From d364d5e7ca76b6be7e04f979ebc43a02feed05f1 Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Wed, 7 Oct 2026 00:01:33 -0700 Subject: [PATCH] refactor: expose public hidden_params accessors (#44668) * refactor: expose public hidden_params accessors Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix: keep hidden params writes on duck-typed responses Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * refactor: move hidden params helpers to core utils Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix: preserve hidden params in duck-typed cost responses Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test: cover hidden params accessors across response types Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix: preserve hidden params for dynamic responses Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix: expose logs through backend component Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * Revert "fix: expose logs through backend component" This reverts commit 621e5ff25e7235ff04eeb765b673bab1b9180dd1. * test: cover dict hidden params helper path Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix: preserve hidden params storage on item writes Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test: cover hidden params accessors for OpenAI types Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test: cover hidden params helper and response paths Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix: complete hidden params accessor migration Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(litellm): tighten hidden parameter validation Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(litellm): read HiddenParams model storage through get_hidden_params Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(litellm): transfer hidden params storage instead of the mapping view Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(litellm): expose only set HiddenParams keys through the mapping view Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * perf(litellm): read HiddenParams view keys without dumping the model Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(litellm): assert HiddenParams extras through model_extra Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(router): copy streaming hidden params into a plain dict Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --------- Co-authored-by: mateo Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- .../proxy/hooks/managed_files.py | 32 ++- litellm/caching/caching_handler.py | 35 ++- .../transformation.py | 24 +- litellm/cost_calculator.py | 9 +- litellm/decisions/main.py | 2 +- litellm/images/main.py | 2 +- litellm/litellm_core_utils/core_helpers.py | 10 +- litellm/litellm_core_utils/hidden_params.py | 114 ++++++++ litellm/litellm_core_utils/litellm_logging.py | 30 +-- .../convert_dict_to_response.py | 16 +- .../streaming_chunk_builder_utils.py | 9 +- .../litellm_core_utils/streaming_handler.py | 73 +++-- litellm/llms/anthropic/chat/transformation.py | 4 +- .../adapters/streaming_iterator.py | 3 +- .../pass_through/adapters/transformation.py | 10 +- .../pass_through/messages/response_cache.py | 2 +- .../messages/streaming_iterator.py | 8 + .../audio_transcription/transformation.py | 2 +- litellm/llms/azure/azure.py | 2 +- litellm/llms/azure/search/transformation.py | 2 +- litellm/llms/azure_ai/agents/handler.py | 9 +- .../azure_model_router/transformation.py | 21 +- .../azure_ai/embed/cohere_transformation.py | 2 +- .../llms/azure_ai/rerank/transformation.py | 5 +- litellm/llms/base_llm/ocr/transformation.py | 8 + .../llms/base_llm/sandbox/transformation.py | 17 ++ .../llms/base_llm/search/transformation.py | 9 + ...n_nova_canvas_image_edit_transformation.py | 15 +- .../image_edit/stability_transformation.py | 13 +- litellm/llms/bytez/chat/transformation.py | 2 +- .../llms/chatgpt/responses/transformation.py | 14 +- .../audio_transcription/transformation.py | 2 +- .../llms/deepinfra/rerank/transformation.py | 4 +- litellm/llms/e2b/sandbox/transformation.py | 21 +- .../audio_transcription/transformation.py | 2 +- .../flux_pro_v11_ultra_transformation.py | 14 +- .../ideogram_v3_transformation.py | 10 +- .../imagen4_transformation.py | 10 +- .../stable_diffusion_transformation.py | 14 +- .../llms/fireworks_ai/chat/transformation.py | 2 +- .../gemini/interactions/transformation.py | 6 +- .../huggingface/embedding/transformation.py | 2 +- .../llms/manus/responses/transformation.py | 8 +- .../audio_transcription/transformation.py | 2 +- litellm/llms/mistral/chat/transformation.py | 2 +- litellm/llms/oci/chat/transformation.py | 2 +- litellm/llms/openai/completion/handler.py | 2 +- .../llms/openai/completion/transformation.py | 7 +- .../llms/openai/containers/transformation.py | 14 +- .../llms/openai/responses/transformation.py | 16 +- .../llms/openai_like/chat/transformation.py | 2 +- .../llms/openrouter/chat/transformation.py | 17 +- .../openrouter/image_edit/transformation.py | 18 +- .../image_generation/transformation.py | 33 ++- .../opensandbox/sandbox/transformation.py | 56 ++-- .../audio_transcription/transformation.py | 2 +- litellm/llms/predibase/chat/transformation.py | 2 +- .../audio_transcription/transformation.py | 2 +- litellm/llms/snowflake/chat/transformation.py | 4 +- .../snowflake/embedding/transformation.py | 2 +- .../soniox/audio_transcription/handler.py | 4 +- .../audio_transcription/transformation.py | 2 +- .../stability/image_edit/transformations.py | 13 +- .../audio_transcription/transformation.py | 2 +- .../vertex_and_google_ai_studio_gemini.py | 20 +- .../volcengine/responses/transformation.py | 12 +- litellm/main.py | 42 +-- litellm/proxy/batches_endpoints/endpoints.py | 60 +++-- .../proxy/fine_tuning_endpoints/endpoints.py | 7 +- litellm/proxy/hooks/dynamic_rate_limiter.py | 30 ++- .../storage_backend_service.py | 6 +- .../cohere_passthrough_logging_handler.py | 5 +- .../gemini_passthrough_logging_handler.py | 4 +- .../openai_passthrough_logging_handler.py | 11 +- .../vertex_passthrough_logging_handler.py | 4 +- .../responses/file_search/emulated_handler.py | 2 +- .../streaming_iterator.py | 9 +- .../transformation.py | 16 +- litellm/responses/main.py | 4 +- .../responses/mcp/chat_completions_handler.py | 10 +- litellm/responses/streaming_iterator.py | 4 +- litellm/responses/utils.py | 2 +- litellm/router.py | 57 ++-- .../complexity_router/complexity_router.py | 7 +- .../add_retry_fallback_headers.py | 13 +- litellm/types/agents.py | 40 +++ litellm/types/containers/main.py | 16 ++ litellm/types/decisions.py | 4 + litellm/types/google_genai/main.py | 16 ++ litellm/types/interactions/generated.py | 32 +++ litellm/types/llms/openai.py | 49 ++++ litellm/types/rerank.py | 8 + litellm/types/responses/main.py | 9 + litellm/types/utils.py | 67 +++++ litellm/types/videos/main.py | 16 ++ ...test_vertex_gemini_image_batch_cost_sdk.py | 3 +- tests/unit/caching/test_caching_handler.py | 45 +++- tests/unit/decisions/test_main.py | 9 + .../test_health_check_helpers.py | 30 ++- .../litellm_core_utils/test_hidden_params.py | 252 ++++++++++++++++++ .../test_litellm_logging.py | 33 +++ .../fine_tuning_endpoints/test_endpoints.py | 17 +- .../proxy/hooks/test_dynamic_rate_limiter.py | 31 +++ ...test_cohere_passthrough_logging_handler.py | 1 + ...test_openai_passthrough_logging_handler.py | 4 +- .../test_litellm_completion_responses.py | 39 +++ .../router_strategy/test_complexity_router.py | 1 + .../test_add_retry_fallback_headers.py | 71 ++++- ...test_router_streaming_fallback_metadata.py | 90 ++++++- .../unit/types/llms/test_types_llms_openai.py | 76 +++++- tests/unit/types/test_types_utils.py | 101 ++++++- 111 files changed, 1779 insertions(+), 409 deletions(-) create mode 100644 litellm/litellm_core_utils/hidden_params.py create mode 100644 tests/unit/litellm_core_utils/test_hidden_params.py diff --git a/enterprise/litellm_enterprise/proxy/hooks/managed_files.py b/enterprise/litellm_enterprise/proxy/hooks/managed_files.py index aae49402146..c07bb216602 100644 --- a/enterprise/litellm_enterprise/proxy/hooks/managed_files.py +++ b/enterprise/litellm_enterprise/proxy/hooks/managed_files.py @@ -33,6 +33,7 @@ from litellm.caching.caching import DualCache from litellm.constants import MAX_FILE_LIST_LIMIT from litellm.files.types import FileRetrieveCallOptions, FileRetrieveProvider from litellm.integrations.custom_logger import CustomLogger +from litellm.litellm_core_utils.hidden_params import HIDDEN_PARAMS_ATTR from litellm.litellm_core_utils.prompt_templates.common_utils import ( extract_file_metadata, ) @@ -1315,7 +1316,10 @@ class PROXY_LiteLLMManagedFiles(CustomLogger, BaseFileEndpoints): model_mappings: Dict[str, str] = {} for file_object in responses: - model_file_id_mapping = file_object._hidden_params.get("model_file_id_mapping") + file_hidden_params = cast( # cast-ok: preserve mapping operations on dynamic file metadata + dict[str, object], getattr(file_object, HIDDEN_PARAMS_ATTR) + ) + model_file_id_mapping = file_hidden_params.get("model_file_id_mapping") if model_file_id_mapping and isinstance(model_file_id_mapping, dict): model_mappings.update(model_file_id_mapping) @@ -1365,7 +1369,8 @@ class PROXY_LiteLLMManagedFiles(CustomLogger, BaseFileEndpoints): _, file_type = extract_file_metadata(create_file_request["file"]) output_file_id = file_objects[0].id - model_id = file_objects[0]._hidden_params.get("model_id") + file_hidden_params: Final = cast(dict[str, object], getattr(file_objects[0], HIDDEN_PARAMS_ATTR)) + model_id = file_hidden_params.get("model_id") unified_file_id = SpecialEnums.LITELLM_MANAGED_FILE_COMPLETE_STR.value.format( file_type, @@ -1431,11 +1436,12 @@ class PROXY_LiteLLMManagedFiles(CustomLogger, BaseFileEndpoints): if decoded_batch_id and is_litellm_executed_batch(decoded_batch_id): return response ## Check if unified_file_id is in the response - unified_file_id = response._hidden_params.get("unified_file_id") # managed file id - unified_batch_id = response._hidden_params.get("unified_batch_id") # managed batch id - is_batch_create: Final = response._hidden_params.get(BATCH_CREATE_HIDDEN_PARAM) is True - model_id = cast(Optional[str], response._hidden_params.get("model_id")) - model_name = cast(Optional[str], response._hidden_params.get("model_name")) + response_hidden_params: Final = cast(dict[str, object], getattr(response, HIDDEN_PARAMS_ATTR)) + unified_file_id = response_hidden_params.get("unified_file_id") + unified_batch_id = response_hidden_params.get("unified_batch_id") + is_batch_create: Final = response_hidden_params.get(BATCH_CREATE_HIDDEN_PARAM) is True + model_id = cast(Optional[str], response_hidden_params.get("model_id")) + model_name = cast(Optional[str], response_hidden_params.get("model_name")) resolved_model_name = resolve_managed_output_file_model_name( unified_input_file_id=unified_file_id if isinstance(unified_file_id, str) else response.input_file_id, @@ -1547,12 +1553,12 @@ class PROXY_LiteLLMManagedFiles(CustomLogger, BaseFileEndpoints): elif isinstance(response, LiteLLMFineTuningJob): ## Check if unified_file_id is in the response - unified_file_id = response._hidden_params.get("unified_file_id") # managed file id - unified_finetuning_job_id = response._hidden_params.get( - "unified_finetuning_job_id" - ) # managed finetuning job id - model_id = cast(Optional[str], response._hidden_params.get("model_id")) - model_name = cast(Optional[str], response._hidden_params.get("model_name")) + finetuning_response_hidden_params: Final = cast( # cast-ok: preserve dynamic mapping behavior + dict[str, object], getattr(response, HIDDEN_PARAMS_ATTR) + ) + unified_file_id = finetuning_response_hidden_params.get("unified_file_id") + unified_finetuning_job_id = finetuning_response_hidden_params.get("unified_finetuning_job_id") + model_id = cast(Optional[str], finetuning_response_hidden_params.get("model_id")) original_response_id = response.id if (unified_file_id or unified_finetuning_job_id) and model_id: response.id = self.get_unified_generic_response_id(model_id=model_id, generic_response_id=response.id) diff --git a/litellm/caching/caching_handler.py b/litellm/caching/caching_handler.py index 4cb84eafa9e..43d7bdb61b1 100644 --- a/litellm/caching/caching_handler.py +++ b/litellm/caching/caching_handler.py @@ -29,6 +29,7 @@ from litellm._logging import print_verbose, verbose_logger from litellm.caching import InMemoryCache from litellm.caching.caching import S3Cache, response_cache_phase from litellm.constants import CACHE_WRITE_SHUTDOWN_FLUSH_TIMEOUT_SECONDS +from litellm.litellm_core_utils.hidden_params import get_hidden_params from litellm.litellm_core_utils.llm_response_utils.response_metadata import ( update_response_metadata, ) @@ -212,6 +213,15 @@ def _request_cache_key(request_kwargs: Mapping[str, Any]) -> str | None: return request_kwargs.get("cache_key", None) +def _set_cached_hidden_param(response: object, key: str, value: object) -> None: + if isinstance(response, TextCompletionResponse): + setattr(response.hidden_params, key, value) + return + hidden_params: Final = get_hidden_params(response) + if hidden_params is not None: + hidden_params[key] = value + + class _CachedEmbeddingRecord(LiteLLMBaseModel): model_config = ConfigDict(frozen=True) @@ -371,8 +381,7 @@ class LLMCachingHandler: or self.request_kwargs.get("cache_key") or litellm.cache.get_cache_key(**self.request_kwargs) ) - if hasattr(cached_result, "_hidden_params"): - cached_result._hidden_params["cache_key"] = cache_key + _set_cached_hidden_param(cached_result, "cache_key", cache_key) return CachingHandlerResponse(cached_result=cached_result) elif ( call_type == CallTypes.aembedding.value @@ -493,8 +502,7 @@ class LLMCachingHandler: or self.request_kwargs.get("cache_key") or litellm.cache.get_cache_key(**self.request_kwargs) ) - if hasattr(cached_result, "_hidden_params"): - cached_result._hidden_params["cache_key"] = cache_key + _set_cached_hidden_param(cached_result, "cache_key", cache_key) return CachingHandlerResponse(cached_result=cached_result) return CachingHandlerResponse(cached_result=cached_result) @@ -574,7 +582,7 @@ class LLMCachingHandler: model=model_name, data=[None] * len(kwargs_input_as_list), ) - final_embedding_cached_response._hidden_params["cache_hit"] = True + final_embedding_cached_response.hidden_params["cache_hit"] = True prompt_tokens = 0 aggregated_details: dict | None = None @@ -749,6 +757,7 @@ class LLMCachingHandler: if cached.usage is not None and embedding_response.usage is not None else cached.usage ) + cached_hidden_params: Final = cached.hidden_params merged: Final = EmbeddingResponse( model=cached.model, data=[ @@ -759,7 +768,7 @@ class LLMCachingHandler: ], usage=merged_usage, hidden_params={ - **cached._hidden_params, + **cached_hidden_params, "cache_hit": True, }, _response_headers=cached._response_headers, @@ -1029,12 +1038,7 @@ class LLMCachingHandler: ) response_obj: Final = ResponsesAPIResponse(**cached_result) - if ( - hasattr(response_obj, "_hidden_params") - and response_obj._hidden_params is not None - and isinstance(response_obj._hidden_params, dict) - ): - response_obj._hidden_params["cache_hit"] = True + _set_cached_hidden_param(response_obj, "cache_hit", True) if _stream_replay_requested(kwargs): cached_result = CachedResponsesAPIStreamingIterator( @@ -1046,12 +1050,7 @@ class LLMCachingHandler: else: cached_result = response_obj - if ( - hasattr(cached_result, "_hidden_params") - and cached_result._hidden_params is not None - and isinstance(cached_result._hidden_params, dict) - ): - cached_result._hidden_params["cache_hit"] = True + _set_cached_hidden_param(cached_result, "cache_hit", True) ######################################################### # Add final timing metrics to the cached result diff --git a/litellm/completion_extras/litellm_responses_transformation/transformation.py b/litellm/completion_extras/litellm_responses_transformation/transformation.py index 790588de29a..eeed2ce784f 100644 --- a/litellm/completion_extras/litellm_responses_transformation/transformation.py +++ b/litellm/completion_extras/litellm_responses_transformation/transformation.py @@ -24,6 +24,7 @@ from pydantic import BaseModel import litellm from litellm import ModelResponse from litellm._logging import verbose_logger +from litellm.litellm_core_utils.hidden_params import get_hidden_params, get_or_create_hidden_params from litellm.litellm_core_utils.prompt_templates.common_utils import ( responses_reasoning_items_from_thinking_blocks, with_prompt_cache_breakpoint, @@ -993,20 +994,25 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge): # Preserve hidden params from the ResponsesAPIResponse, especially the headers # which contain important provider information like x-request-id - raw_response_hidden_params: Final = getattr(raw_response, "_hidden_params", {}) + raw_response_hidden_params: Final = get_hidden_params(raw_response) or {} if raw_response_hidden_params: - if not hasattr(model_response, "_hidden_params") or model_response._hidden_params is None: - model_response._hidden_params = {} + model_response_hidden_params: Final = get_or_create_hidden_params(model_response) # Merge the raw_response hidden params with model_response hidden params # Preserve existing keys in model_response but add/override with raw_response params for key, value in raw_response_hidden_params.items(): - if key == "additional_headers" and key in model_response._hidden_params: - # Merge additional_headers to preserve both sets - existing_additional_headers = model_response._hidden_params.get("additional_headers", {}) - merged_headers = {**value, **existing_additional_headers} - model_response._hidden_params[key] = merged_headers + if key == "additional_headers" and key in model_response_hidden_params: + existing_additional_headers = model_response_hidden_params.get("additional_headers", {}) + merged_headers = { + **cast( # cast-ok: preserve mapping operations on dynamic response metadata + "dict[str, object]", value + ), + **cast( # cast-ok: preserve mapping operations on dynamic response metadata + "dict[str, object]", existing_additional_headers + ), + } + model_response_hidden_params[key] = merged_headers else: - model_response._hidden_params[key] = value + model_response_hidden_params[key] = value return model_response diff --git a/litellm/cost_calculator.py b/litellm/cost_calculator.py index 10d61087ed9..b637676836c 100644 --- a/litellm/cost_calculator.py +++ b/litellm/cost_calculator.py @@ -18,6 +18,7 @@ from litellm.constants import ( DEFAULT_MAX_LRU_CACHE_SIZE, DEFAULT_REPLICATE_GPU_PRICE_PER_SECOND, ) +from litellm.litellm_core_utils.hidden_params import HIDDEN_PARAMS_ATTR from litellm.litellm_core_utils.llm_cost_calc.tool_call_cost_tracking import ( StandardBuiltInToolCostTracking, ) @@ -2028,8 +2029,12 @@ def response_cost_calculator( response_cost = 0.0 else: if isinstance(response_object, BaseModel): - if hasattr(response_object, "_hidden_params"): - provider_response_cost: Final = get_response_cost_from_hidden_params(response_object._hidden_params) + if hasattr(response_object, HIDDEN_PARAMS_ATTR): + hidden_params: Final = cast( # cast-ok: cost metadata supports dict and Pydantic storage + dict[str, object] | BaseModel, + getattr(response_object, HIDDEN_PARAMS_ATTR), + ) + provider_response_cost: Final = get_response_cost_from_hidden_params(hidden_params) if provider_response_cost is not None: return provider_response_cost diff --git a/litellm/decisions/main.py b/litellm/decisions/main.py index 4864fcaa0e4..44a685e3235 100644 --- a/litellm/decisions/main.py +++ b/litellm/decisions/main.py @@ -206,7 +206,7 @@ def _parse_response( response.raise_for_status() payload: Final[object] = _DECISIONS_PAYLOAD_ADAPTER.validate_json(response.content) result: Final = _DECISIONS_RESPONSE_ADAPTER.validate_python(prepared.config.unwrap_response(payload)) - result._hidden_params.update( + result.hidden_params.update( { "model": f"{prepared.provider}/{prepared.upstream_model}", "custom_llm_provider": prepared.provider, diff --git a/litellm/images/main.py b/litellm/images/main.py index a5d905a58e9..af01e051211 100644 --- a/litellm/images/main.py +++ b/litellm/images/main.py @@ -230,7 +230,7 @@ def image_generation( else: model = "dall-e-2" custom_llm_provider = "openai" # default to dall-e-2 on openai - model_response._hidden_params["model"] = model + model_response.hidden_params["model"] = model openai_params: Final = [ "user", "request_timeout", diff --git a/litellm/litellm_core_utils/core_helpers.py b/litellm/litellm_core_utils/core_helpers.py index de0e096a764..e857461785c 100644 --- a/litellm/litellm_core_utils/core_helpers.py +++ b/litellm/litellm_core_utils/core_helpers.py @@ -773,12 +773,18 @@ _NO_HEADERS: Final[Mapping[str, object]] = MappingProxyType({}) class _CarriesHiddenParams(Protocol): _hidden_params: dict[str, object] # mutable-ok: the responses billed here keep hidden params in a plain dict + @property + def hidden_params(self) -> dict[str, object]: ... # mutable-ok: API requires mutation + + @hidden_params.setter + def hidden_params(self, hidden_params: dict[str, object]) -> None: ... # mutable-ok: API requires mutation + def set_response_cost_in_hidden_params(response: _CarriesHiddenParams, cost: float | None) -> None: """Record a provider-reported cost where the cost calculator looks before the price map.""" if cost is None: return - hidden_params: Final = response._hidden_params # pyright: ignore[reportPrivateUsage] # no public accessor + hidden_params: Final = response.hidden_params additional_headers: Final[object] = hidden_params.get("additional_headers") merged: Final[dict[str, object]] = { # mutable-ok: assigned into the plain-dict hidden params **(additional_headers if isinstance(additional_headers, Mapping) else _NO_HEADERS), @@ -794,7 +800,7 @@ _PROVIDER_HEADERS_ADAPTER: Final = TypeAdapter(Mapping[str, str]) def set_provider_response_headers_in_hidden_params( response: _CarriesHiddenParams, headers: httpx.Headers | Mapping[str, str] ) -> None: - hidden_params: Final = response._hidden_params # pyright: ignore[reportPrivateUsage] # no public accessor + hidden_params: Final = response.hidden_params existing_additional_headers: Final[object] = hidden_params.get("additional_headers") raw_headers: Final[dict[str, str]] = dict(headers) # mutable-ok: stored as the plain-dict hidden param additional_headers: Final[dict[str, object]] = { # mutable-ok: assigned into the plain-dict hidden params diff --git a/litellm/litellm_core_utils/hidden_params.py b/litellm/litellm_core_utils/hidden_params.py new file mode 100644 index 00000000000..714b258413e --- /dev/null +++ b/litellm/litellm_core_utils/hidden_params.py @@ -0,0 +1,114 @@ +from collections.abc import Iterator, MutableMapping +from typing import Final, Protocol, cast + +from litellm.types.llms.base import HiddenParams + +_HIDDEN_PARAMS_ATTR: Final = "_hidden_params" +HIDDEN_PARAMS_ATTR: Final = _HIDDEN_PARAMS_ATTR + + +class _SupportsItemAssignment(Protocol): + def __setitem__(self, key: str, value: object) -> None: ... + + +class HiddenParamsModelView(MutableMapping[str, object]): + def __init__(self, hidden_params: HiddenParams) -> None: + self._hidden_params = hidden_params + + def __getitem__(self, key: str) -> object: + fields: Final = self._keys() + if key not in fields: + raise KeyError(key) + try: + return cast(object, getattr(self._hidden_params, key)) + except AttributeError as error: + raise KeyError(key) from error + + def __setitem__(self, key: str, value: object) -> None: + setattr(self._hidden_params, key, value) + + def __delitem__(self, key: str) -> None: + fields: Final = self._keys() + if key not in fields: + raise KeyError(key) + try: + delattr(self._hidden_params, key) + except AttributeError as error: + raise KeyError(key) from error + self._hidden_params.model_fields_set.discard(key) + + def __iter__(self) -> Iterator[str]: + return iter(self._keys()) + + def __len__(self) -> int: + return len(self._keys()) + + def _keys(self) -> frozenset[str]: + return frozenset(self._hidden_params.model_fields_set) | frozenset(self._hidden_params.model_extra or {}) + + +def _get_hidden_params_storage(obj: object) -> object | None: + return obj.get(_HIDDEN_PARAMS_ATTR) if isinstance(obj, dict) else getattr(obj, _HIDDEN_PARAMS_ATTR, None) + + +def get_hidden_params_storage(obj: object) -> dict[str, object] | HiddenParams | None: + hidden_params: Final[object | None] = _get_hidden_params_storage(obj) + if isinstance(hidden_params, dict): + return cast( # cast-ok: runtime dict validation preserves dynamically typed legacy storage + dict[str, object], hidden_params + ) + if isinstance(hidden_params, HiddenParams): + return hidden_params + return None + + +def _as_hidden_params_mapping(hidden_params: object | None) -> MutableMapping[str, object] | None: + if isinstance(hidden_params, dict): + return cast( # cast-ok: runtime dict validation preserves dynamically typed legacy storage + dict[str, object], hidden_params + ) + if isinstance(hidden_params, HiddenParams): + return HiddenParamsModelView(hidden_params) + return None + + +def get_hidden_params(obj: object) -> MutableMapping[str, object] | None: + hidden_params: Final[object | None] = _get_hidden_params_storage(obj) + return _as_hidden_params_mapping(hidden_params) + + +def set_hidden_params(obj: object, hidden_params: dict[str, object] | HiddenParams) -> None: + if isinstance(obj, dict): + obj[_HIDDEN_PARAMS_ATTR] = hidden_params + else: + setattr(obj, _HIDDEN_PARAMS_ATTR, hidden_params) + + +def set_hidden_param(obj: object, key: str, value: object) -> None: + hidden_params: Final[object | None] = _get_hidden_params_storage(obj) + if hidden_params is None: + set_hidden_params(obj, {key: value}) + return + if isinstance(hidden_params, dict): + cast( # cast-ok: runtime dict validation preserves dynamically typed legacy storage + dict[str, object], hidden_params + )[key] = value + return + if hasattr(hidden_params, "__setitem__"): + cast( # cast-ok: hasattr validates the legacy storage assignment protocol + _SupportsItemAssignment, hidden_params + )[key] = value + return + raise TypeError(f"unsupported hidden params storage: {type(hidden_params).__name__}") + + +def get_or_create_hidden_params(obj: object) -> MutableMapping[str, object]: + hidden_params: Final[object | None] = _get_hidden_params_storage(obj) + if hidden_params is None: + created_hidden_params: Final[dict[str, object]] = {} + set_hidden_params(obj, created_hidden_params) + return created_hidden_params + hidden_params_mapping: Final[MutableMapping[str, object] | None] = _as_hidden_params_mapping(hidden_params) + if hidden_params_mapping is not None: + return hidden_params_mapping + raise TypeError(f"unsupported hidden params storage: {type(hidden_params).__name__}") diff --git a/litellm/litellm_core_utils/litellm_logging.py b/litellm/litellm_core_utils/litellm_logging.py index 9f19872360c..e15447c8584 100644 --- a/litellm/litellm_core_utils/litellm_logging.py +++ b/litellm/litellm_core_utils/litellm_logging.py @@ -79,6 +79,7 @@ from litellm.litellm_core_utils.core_helpers import ( ) from litellm.litellm_core_utils.error_normalization import normalize_error from litellm.litellm_core_utils.get_litellm_params import get_litellm_params +from litellm.litellm_core_utils.hidden_params import HIDDEN_PARAMS_ATTR, set_hidden_param from litellm.litellm_core_utils.internal_call_metadata import ( MODEL_ACCESS_GROUP_METADATA_KEY, is_unbilled_non_inference_call, @@ -1865,7 +1866,7 @@ class Logging(LiteLLMLoggingBaseClass): else response_result ) - result_hidden_params: Final = getattr(priced_result, "_hidden_params", None) or MappingProxyType({}) + result_hidden_params: Final = getattr(priced_result, HIDDEN_PARAMS_ATTR, None) or MappingProxyType({}) if isinstance(priced_result, (BaseModel, HttpxBinaryResponseContent)) and hasattr( priced_result, "_hidden_params" ): @@ -2067,7 +2068,7 @@ class Logging(LiteLLMLoggingBaseClass): def _custom_pricing_for(self, result: object) -> bool: litellm_params: Final = getattr(self, "litellm_params", None) - result_hidden_params: Final = getattr(result, "_hidden_params", None) or MappingProxyType({}) + result_hidden_params: Final = getattr(result, HIDDEN_PARAMS_ATTR, None) or MappingProxyType({}) additional_headers: Final = ( result_hidden_params.get("additional_headers") if isinstance(result_hidden_params, dict) @@ -2416,7 +2417,7 @@ class Logging(LiteLLMLoggingBaseClass): """ if logging_result is None: return - hidden_params: Final = getattr(logging_result, "_hidden_params", None) + hidden_params: Final = getattr(logging_result, HIDDEN_PARAMS_ATTR, None) if not hidden_params: return if self.model_call_details.get("litellm_params") is None: @@ -2440,7 +2441,7 @@ class Logging(LiteLLMLoggingBaseClass): ): """Resolve hidden params, compute response cost, and emit the standard logging payload.""" self._surface_response_headers_from_result(logging_result) - hidden_params: Final = getattr(logging_result, "_hidden_params", {}) + hidden_params: Final = getattr(logging_result, HIDDEN_PARAMS_ATTR, {}) if hidden_params: if self.model_call_details.get("litellm_params") is not None: self.model_call_details["litellm_params"].setdefault("metadata", {}) @@ -2715,7 +2716,7 @@ class Logging(LiteLLMLoggingBaseClass): Left in place they overwrite the create's real deployment with the poll's empty one in the payload every logging integration reads. """ - settled_hidden_params: Final = getattr(result, "_hidden_params", None) + settled_hidden_params: Final = getattr(result, HIDDEN_PARAMS_ATTR, None) if isinstance(settled_hidden_params, dict): for poll_scoped_key in ("response_cost", "model_id", "litellm_model_name"): settled_hidden_params.pop(poll_scoped_key, None) @@ -3289,13 +3290,12 @@ class Logging(LiteLLMLoggingBaseClass): batch_successful_requests: Final = kwargs.get("batch_successful_requests", None) batch_failed_requests: Final = kwargs.get("batch_failed_requests", None) has_explicit_batch_data: Final = all(x is not None for x in (batch_cost, batch_usage, batch_models)) - should_compute_batch_data: Final = not has_explicit_batch_data and batch_cost_is_final(result) if has_explicit_batch_data: - result._hidden_params["response_cost"] = batch_cost - result._hidden_params["batch_models"] = batch_models - result._hidden_params["batch_successful_requests"] = batch_successful_requests # pyright: ignore[reportPrivateUsage] # rebind-ok: same result._hidden_params pattern as response_cost/batch_models above - result._hidden_params["batch_failed_requests"] = batch_failed_requests # pyright: ignore[reportPrivateUsage] # rebind-ok: same pattern as above + set_hidden_param(result, "response_cost", batch_cost) + set_hidden_param(result, "batch_models", batch_models) + set_hidden_param(result, "batch_successful_requests", batch_successful_requests) + set_hidden_param(result, "batch_failed_requests", batch_failed_requests) result.usage = batch_usage batch_prompt_cost: Final = kwargs.get("batch_prompt_cost", None) batch_completion_cost: Final = kwargs.get("batch_completion_cost", None) @@ -3320,10 +3320,10 @@ class Logging(LiteLLMLoggingBaseClass): model_info=self.get_router_deployment_model_info(), ) - result._hidden_params["response_cost"] = batch_result.cost - result._hidden_params["batch_models"] = batch_result.models - result._hidden_params["batch_successful_requests"] = batch_result.successful_requests # pyright: ignore[reportPrivateUsage] # rebind-ok: same pattern as above - result._hidden_params["batch_failed_requests"] = batch_result.failed_requests # pyright: ignore[reportPrivateUsage] # rebind-ok: same pattern as above + set_hidden_param(result, "response_cost", batch_result.cost) + set_hidden_param(result, "batch_models", batch_result.models) + set_hidden_param(result, "batch_successful_requests", batch_result.successful_requests) + set_hidden_param(result, "batch_failed_requests", batch_result.failed_requests) result.usage = batch_result.usage self.set_cost_breakdown( input_cost=batch_result.prompt_cost, @@ -6516,7 +6516,7 @@ def _extract_response_obj_and_hidden_params( ) -> tuple[dict, dict | None]: """Extract response_obj and hidden_params from init_response_obj.""" hidden_params: dict | None = ( - getattr(init_response_obj, "_hidden_params", None) + getattr(init_response_obj, HIDDEN_PARAMS_ATTR, None) if isinstance(init_response_obj, BaseModel | HttpxBinaryResponseContent) else None ) diff --git a/litellm/litellm_core_utils/llm_response_utils/convert_dict_to_response.py b/litellm/litellm_core_utils/llm_response_utils/convert_dict_to_response.py index 547329b7f51..cd050715e38 100644 --- a/litellm/litellm_core_utils/llm_response_utils/convert_dict_to_response.py +++ b/litellm/litellm_core_utils/llm_response_utils/convert_dict_to_response.py @@ -9,6 +9,7 @@ from typing import Final, Literal, cast import litellm from litellm._logging import verbose_logger from litellm.constants import RESPONSE_FORMAT_TOOL_NAME +from litellm.litellm_core_utils.hidden_params import get_hidden_params from litellm.litellm_core_utils.prompt_templates.common_utils import ( extract_reasoning_content, ) @@ -516,7 +517,8 @@ class LiteLLMResponseObjectHandler: text_completion_response["choices"] = choices_list text_completion_response["usage"] = response.get("usage", None) - text_completion_response._hidden_params = HiddenParams(**response._hidden_params) + response_hidden_params: Final = get_hidden_params(response) or {} + text_completion_response.hidden_params = HiddenParams.model_validate(response_hidden_params) return text_completion_response @staticmethod @@ -747,9 +749,9 @@ def convert_to_model_response_object( model_response_object._response_ms = (end_time - start_time).total_seconds() * 1000 if hidden_params is not None: - if model_response_object._hidden_params is None: - model_response_object._hidden_params = {} - model_response_object._hidden_params.update(hidden_params) + if model_response_object.hidden_params is None: + model_response_object.hidden_params = {} + model_response_object.hidden_params.update(hidden_params) if _response_headers is not None: model_response_object._response_headers = _response_headers @@ -787,7 +789,7 @@ def convert_to_model_response_object( ).total_seconds() * 1000 # return response latency in ms like openai if hidden_params is not None: - model_response_object._hidden_params = hidden_params + model_response_object.hidden_params = hidden_params if _response_headers is not None: model_response_object._response_headers = _response_headers @@ -833,13 +835,13 @@ def convert_to_model_response_object( setattr(model_response_object, "usage", tr_usage_object) if hidden_params is not None: - model_response_object._hidden_params = hidden_params + model_response_object.hidden_params = hidden_params # Store internally-calculated duration in _hidden_params for cost # tracking without exposing it in the response body. Must be set # after hidden_params assignment to avoid being overwritten. if "_audio_transcription_duration" in response_object: - model_response_object._hidden_params["audio_transcription_duration"] = response_object[ + model_response_object.hidden_params["audio_transcription_duration"] = response_object[ "_audio_transcription_duration" ] diff --git a/litellm/litellm_core_utils/streaming_chunk_builder_utils.py b/litellm/litellm_core_utils/streaming_chunk_builder_utils.py index be9a17a5dd2..0b2a8722354 100644 --- a/litellm/litellm_core_utils/streaming_chunk_builder_utils.py +++ b/litellm/litellm_core_utils/streaming_chunk_builder_utils.py @@ -8,6 +8,7 @@ from typing import TYPE_CHECKING, Any, Final, TypeAlias, TypedDict, Union, cast from typing_extensions import ReadOnly, Required from litellm._logging import verbose_logger +from litellm.litellm_core_utils.hidden_params import HIDDEN_PARAMS_ATTR, set_hidden_params from litellm.types.llms.openai import ( ChatCompletionAssistantContentValue, ChatCompletionAudioDelta, @@ -265,7 +266,11 @@ class ChunkProcessor: return model_response # set hidden params from chunk to model_response if model_response is not None and hasattr(model_response, "_hidden_params"): - model_response._hidden_params = chunk.get("_hidden_params", {}) + chunk_hidden_params: Final = chunk.get("_hidden_params", {}) + if isinstance(chunk_hidden_params, dict): + set_hidden_params(model_response, chunk_hidden_params) + else: + setattr(model_response, HIDDEN_PARAMS_ATTR, chunk_hidden_params) return model_response @staticmethod @@ -841,7 +846,7 @@ class ChunkProcessor: elif (isinstance(chunk, ModelResponse) or isinstance(chunk, ModelResponseStream)) and hasattr( chunk, "_hidden_params" ): - usage_chunk = chunk._hidden_params.get("usage", None) + usage_chunk = chunk.hidden_params.get("usage", None) if isinstance(usage_chunk, dict): return Usage(**usage_chunk) diff --git a/litellm/litellm_core_utils/streaming_handler.py b/litellm/litellm_core_utils/streaming_handler.py index 28402624297..335e33e1f1a 100644 --- a/litellm/litellm_core_utils/streaming_handler.py +++ b/litellm/litellm_core_utils/streaming_handler.py @@ -20,6 +20,7 @@ import litellm from litellm import verbose_logger from litellm._uuid import uuid from litellm.litellm_core_utils.asyncify import asyncify +from litellm.litellm_core_utils.hidden_params import HIDDEN_PARAMS_ATTR, get_hidden_params from litellm.litellm_core_utils.model_response_utils import ( is_model_response_stream_empty, ) @@ -187,7 +188,7 @@ def _provider_hidden_params( chunk: object, provider_response_model: str | None, ) -> Mapping[str, object] | None: - hidden: Final[object] = getattr(chunk, "_hidden_params", None) + hidden: Final = get_hidden_params(chunk) parsed: Final = _parsed_provider_hidden_params(hidden) provider_specific_fields: Final[object | None] = ( dict(parsed.provider_specific_fields) if parsed is not None and parsed.provider_specific_fields else None @@ -206,6 +207,14 @@ def _provider_hidden_params( class CustomStreamWrapper: + @property + def hidden_params(self) -> dict[str, object]: # mutable-ok: API requires mutation + return self._hidden_params + + @hidden_params.setter + def hidden_params(self, hidden_params: dict[str, object]) -> None: # mutable-ok: API requires mutation + self._hidden_params = hidden_params + def __init__( self, completion_stream, @@ -258,7 +267,7 @@ class CustomStreamWrapper: optional_params=self.logging_obj.model_call_details.get("litellm_params", {}), ) - self._hidden_params = { + self._hidden_params: dict[str, object] = { "model_id": (_model_info.get("id", None)), "api_base": _api_base, } # returned as x-litellm-model-id response header in proxy @@ -280,7 +289,7 @@ class CustomStreamWrapper: self._repeated_messages_count = 1 self.is_function_call = self.check_is_function_call(logging_obj=logging_obj) self.created: int | None = None - self._last_returned_hidden_params: dict | None = None + self._last_returned_hidden_params: dict[str, object] | None = None _cached_logging_provider: Final = self.logging_obj.model_call_details.get("custom_llm_provider", None) self._cached_logging_llm_provider: str | None = _cached_logging_provider @@ -624,8 +633,12 @@ class CustomStreamWrapper: pass if str_line.choices[0].finish_reason: is_finished = True # check if str_line._hidden_params["is_finished"] is True - if hasattr(str_line, "_hidden_params") and str_line._hidden_params.get("is_finished") is not None: - is_finished = str_line._hidden_params.get("is_finished") + str_line_hidden_params: Final = get_hidden_params(str_line) + is_finished_from_hidden_params: Final = ( + str_line_hidden_params.get("is_finished") if str_line_hidden_params is not None else None + ) + if isinstance(is_finished_from_hidden_params, bool): + is_finished = is_finished_from_hidden_params finish_reason = str_line.choices[0].finish_reason # checking for logprobs @@ -758,14 +771,14 @@ class CustomStreamWrapper: # must win over both caller-supplied hidden_params and the computed # custom_llm_provider/created_at values, so it comes last. if hidden_params is not None: - model_response._hidden_params = { + model_response.hidden_params = { **hidden_params, "custom_llm_provider": _logging_obj_llm_provider, "created_at": time.time(), **self._base_hidden_params, } else: - model_response._hidden_params = { + model_response.hidden_params = { "custom_llm_provider": _logging_obj_llm_provider, "created_at": time.time(), **self._base_hidden_params, @@ -796,7 +809,7 @@ class CustomStreamWrapper: self.response_id = id if id and isinstance(id, str) and id.strip(): - model_response._hidden_params["received_model_id"] = id + model_response.hidden_params["received_model_id"] = id if self.response_id is not None and isinstance(self.response_id, str): model_response.id = self.response_id @@ -1762,9 +1775,9 @@ class CustomStreamWrapper: _usage: Final[Usage | None] = getattr(response, "usage", None) _cost: Final = CustomStreamWrapper._resolve_provider_reported_cost(getattr(_usage, "cost", None)) if _cost is not None: - if "additional_headers" not in response._hidden_params: - response._hidden_params["additional_headers"] = {} - response._hidden_params["additional_headers"]["llm_provider-x-litellm-response-cost"] = _cost + if "additional_headers" not in response.hidden_params: + response.hidden_params["additional_headers"] = {} + response.hidden_params["additional_headers"]["llm_provider-x-litellm-response-cost"] = _cost def __next__(self) -> "ModelResponseStream": cache_hit = False @@ -1823,14 +1836,22 @@ class CustomStreamWrapper: if getattr(response, "usage", None) is not None: usage_to_preserve = response.usage if usage_to_preserve: - response._hidden_params["usage"] = usage_to_preserve + response_hidden_params_for_usage = cast( # cast-ok: preserve dynamic mapping behavior + dict[str, object], getattr(response, HIDDEN_PARAMS_ATTR) + ) + response_hidden_params_for_usage["usage"] = usage_to_preserve obj_dict = response.model_dump() if "usage" in obj_dict: del obj_dict["usage"] - response = self.model_response_creator(chunk=obj_dict, hidden_params=response._hidden_params) + response_hidden_params_for_model = cast( # cast-ok: preserve dynamic mapping behavior + Mapping[str, object], getattr(response, HIDDEN_PARAMS_ATTR) + ) + response = self.model_response_creator( + chunk=obj_dict, hidden_params=response_hidden_params_for_model + ) ## check if empty is_empty = is_model_response_stream_empty(model_response=cast(ModelResponseStream, response)) @@ -1839,8 +1860,11 @@ class CustomStreamWrapper: # add usage as hidden param if self.sent_last_chunk is True and self.stream_options is None: usage = calculate_total_usage(chunks=self.chunks) - response._hidden_params["usage"] = usage - self._last_returned_hidden_params = response._hidden_params + response_hidden_params_for_final_chunk = cast( # cast-ok: preserve dynamic mapping behavior + dict[str, object], getattr(response, HIDDEN_PARAMS_ATTR) + ) + response_hidden_params_for_final_chunk["usage"] = usage + self._last_returned_hidden_params = response_hidden_params_for_final_chunk # Add MCP metadata to final chunk if present response = self._add_mcp_metadata_to_final_chunk(response) # RETURN RESULT @@ -1937,7 +1961,7 @@ class CustomStreamWrapper: self.chunks.append(processed_chunk) if self.stream_options is None: # add usage as hidden param usage = calculate_total_usage(chunks=self.chunks) - processed_chunk._hidden_params["usage"] = usage + processed_chunk.hidden_params["usage"] = usage ## LOGGING executor.submit( self.run_success_logging_and_cache_storage, @@ -2035,7 +2059,10 @@ class CustomStreamWrapper: if "usage" in obj_dict: del obj_dict["usage"] processed_chunk = self.model_response_creator( - chunk=obj_dict, hidden_params=processed_chunk._hidden_params + chunk=obj_dict, + hidden_params=cast( # cast-ok: preserve mapping operations on dynamic storage + Mapping[str, object], getattr(processed_chunk, HIDDEN_PARAMS_ATTR) + ), ) is_empty = is_model_response_stream_empty( model_response=cast(ModelResponseStream, processed_chunk) @@ -2049,8 +2076,11 @@ class CustomStreamWrapper: # add usage as hidden param if self.sent_last_chunk is True and self.stream_options is None: usage = calculate_total_usage(chunks=self.chunks) - processed_chunk._hidden_params["usage"] = usage - self._last_returned_hidden_params = processed_chunk._hidden_params + processed_chunk_hidden_params = cast( # cast-ok: preserve dynamic mapping behavior + dict[str, object], getattr(processed_chunk, HIDDEN_PARAMS_ATTR) + ) + processed_chunk_hidden_params["usage"] = usage + self._last_returned_hidden_params = processed_chunk_hidden_params # Call post-call streaming deployment hook for final chunk if self.sent_last_chunk is True: @@ -2205,7 +2235,10 @@ class CustomStreamWrapper: self.chunks.append(processed_chunk) if self.stream_options is None: usage: Final = calculate_total_usage(chunks=self.chunks) - processed_chunk._hidden_params["usage"] = usage # pyright: ignore[reportPrivateUsage] # sync parity + processed_chunk_hidden_params: Final = cast( # cast-ok: preserve dynamic mapping behavior + dict[str, object], getattr(processed_chunk, HIDDEN_PARAMS_ATTR) + ) + processed_chunk_hidden_params["usage"] = usage # see sync __next__'s sibling branch: deliberately do NOT restore # here - this chunk is still this call's own data, and restoring # before returning it would corrupt the caller's own log diff --git a/litellm/llms/anthropic/chat/transformation.py b/litellm/llms/anthropic/chat/transformation.py index 5aa14b0442a..96e181b7dcf 100644 --- a/litellm/llms/anthropic/chat/transformation.py +++ b/litellm/llms/anthropic/chat/transformation.py @@ -2668,7 +2668,7 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig): _message = json_mode_message model_response.choices[0].message = _message - model_response._hidden_params["original_response"] = completion_response["content"] + model_response.hidden_params["original_response"] = completion_response["content"] model_response.choices[0].finish_reason = cast( OpenAIChatCompletionFinishReason, map_finish_reason(completion_response["stop_reason"]), @@ -2686,7 +2686,7 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig): model_response.model = completion_response["model"] _hidden_params["provider_specific_fields"] = provider_specific_fields - model_response._hidden_params = _hidden_params + model_response.hidden_params = _hidden_params return model_response def get_prefix_prompt(self, messages: list[AllMessageValues]) -> str | None: diff --git a/litellm/llms/anthropic/pass_through/adapters/streaming_iterator.py b/litellm/llms/anthropic/pass_through/adapters/streaming_iterator.py index 4ef6c305cf5..6432469366c 100644 --- a/litellm/llms/anthropic/pass_through/adapters/streaming_iterator.py +++ b/litellm/llms/anthropic/pass_through/adapters/streaming_iterator.py @@ -20,6 +20,7 @@ from typing_extensions import assert_never from litellm._logging import verbose_logger from litellm._uuid import uuid from litellm.exceptions import MidStreamFallbackError +from litellm.litellm_core_utils.hidden_params import set_hidden_params from litellm.llms.base_llm.chat.transformation import BaseLLMException from litellm.types.llms.anthropic import ( AppliedEdit, @@ -165,7 +166,7 @@ class _CombinedChunkSplitter: chunk.usage = None hidden_params: Final = getattr(chunk, "_hidden_params", None) if isinstance(hidden_params, dict) and "usage" in hidden_params: - chunk._hidden_params = {key: value for key, value in hidden_params.items() if key != "usage"} + set_hidden_params(chunk, {key: value for key, value in hidden_params.items() if key != "usage"}) @staticmethod def _split_by_payload_kind(chunk: "ModelResponseStream") -> "tuple[ModelResponseStream, ...]": diff --git a/litellm/llms/anthropic/pass_through/adapters/transformation.py b/litellm/llms/anthropic/pass_through/adapters/transformation.py index 88b99eb733e..8e0dc83e563 100644 --- a/litellm/llms/anthropic/pass_through/adapters/transformation.py +++ b/litellm/llms/anthropic/pass_through/adapters/transformation.py @@ -8,6 +8,7 @@ from typing import TYPE_CHECKING, Any, Final, Literal, TypeAlias, TypeVar, cast from pydantic import JsonValue, TypeAdapter import litellm +from litellm.litellm_core_utils.hidden_params import get_hidden_params from litellm.llms.anthropic.pass_through.utils import ( is_reasoning_auto_summary_enabled, prompt_cache_key_from_user_id, @@ -1739,10 +1740,13 @@ class LiteLLMAnthropicMessagesAdapter: ) if getattr(response, "usage", None) is not None: litellm_usage_chunk: Usage | None = response.usage - elif hasattr(response, "_hidden_params") and "usage" in response._hidden_params: - litellm_usage_chunk = response._hidden_params["usage"] else: - litellm_usage_chunk = None + response_hidden_params: Final = get_hidden_params(response) + litellm_usage_chunk = ( + Usage.model_validate(response_hidden_params["usage"]) + if response_hidden_params is not None and "usage" in response_hidden_params + else None + ) if litellm_usage_chunk is not None: usage_delta = self._translate_openai_usage_to_anthropic_usage_delta(litellm_usage_chunk) else: diff --git a/litellm/llms/anthropic/pass_through/messages/response_cache.py b/litellm/llms/anthropic/pass_through/messages/response_cache.py index 0d8b98962cf..f35c4c80f2b 100644 --- a/litellm/llms/anthropic/pass_through/messages/response_cache.py +++ b/litellm/llms/anthropic/pass_through/messages/response_cache.py @@ -45,7 +45,7 @@ class AnthropicMessagesStreamCacheWriter: self.collected_chunks: list[bytes] = [] # mutable-ok: rebuilding a tuple per SSE chunk is quadratic self.persisted = False self._hidden_params: dict[str, object] = dict( # mutable-ok: callers stamp cache_key in here - stream._hidden_params if isinstance(stream, AnthropicMessagesStreamingResponse) else _EMPTY_MAPPING + stream.hidden_params if isinstance(stream, AnthropicMessagesStreamingResponse) else _EMPTY_MAPPING ) @property diff --git a/litellm/llms/anthropic/pass_through/messages/streaming_iterator.py b/litellm/llms/anthropic/pass_through/messages/streaming_iterator.py index 13e1a114bdd..150b1a730ef 100644 --- a/litellm/llms/anthropic/pass_through/messages/streaming_iterator.py +++ b/litellm/llms/anthropic/pass_through/messages/streaming_iterator.py @@ -365,6 +365,14 @@ class AnthropicMessagesStreamingResponse: self.completion_stream = completion_stream self._hidden_params = hidden_params + @property + def hidden_params(self) -> AnthropicMessagesStreamHiddenParams: + return self._hidden_params + + @hidden_params.setter + def hidden_params(self, hidden_params: AnthropicMessagesStreamHiddenParams) -> None: + self._hidden_params = hidden_params + @property def has_buffered_provider_output(self) -> bool: return getattr(self.completion_stream, "has_buffered_provider_output", False) is True diff --git a/litellm/llms/azure/audio_transcription/transformation.py b/litellm/llms/azure/audio_transcription/transformation.py index 17597da3895..858f122f65a 100644 --- a/litellm/llms/azure/audio_transcription/transformation.py +++ b/litellm/llms/azure/audio_transcription/transformation.py @@ -142,7 +142,7 @@ class AzureSpeechAudioTranscriptionConfig(BaseAudioTranscriptionConfig): text: Final = self._extract_text(payload) response: Final = TranscriptionResponse(text=text) - response._hidden_params = response_json + response.hidden_params = response_json return response def get_error_class(self, error_message: str, status_code: int, headers: dict | httpx.Headers) -> BaseLLMException: diff --git a/litellm/llms/azure/azure.py b/litellm/llms/azure/azure.py index 35a41304e1b..64441fb0a43 100644 --- a/litellm/llms/azure/azure.py +++ b/litellm/llms/azure/azure.py @@ -1254,7 +1254,7 @@ class AzureChatCompletion(BaseAzureLLM, BaseLLM): and litellm_params is not None and litellm_params.get("base_model", None) is not None ): - model_response._hidden_params["model"] = litellm_params.get("base_model", None) + model_response.hidden_params["model"] = litellm_params.get("base_model", None) # Azure image generation API doesn't support extra_body parameter extra_body: Final = optional_params.pop("extra_body", {}) diff --git a/litellm/llms/azure/search/transformation.py b/litellm/llms/azure/search/transformation.py index 185aedc4ad6..4264dae018a 100644 --- a/litellm/llms/azure/search/transformation.py +++ b/litellm/llms/azure/search/transformation.py @@ -422,7 +422,7 @@ class BingGroundingSearchConfig(BaseSearchConfig): ) if get_secret_str(CONNECTION_ID_ENV): return response - response._hidden_params["additional_headers"] = {_RESPONSE_COST_HEADER: 0.0} + response.hidden_params["additional_headers"] = {_RESPONSE_COST_HEADER: 0.0} return response def get_error_class( diff --git a/litellm/llms/azure_ai/agents/handler.py b/litellm/llms/azure_ai/agents/handler.py index f7382190fca..3f0eeef9c7a 100644 --- a/litellm/llms/azure_ai/agents/handler.py +++ b/litellm/llms/azure_ai/agents/handler.py @@ -29,6 +29,7 @@ import httpx from typing_extensions import ReadOnly from litellm._logging import verbose_logger +from litellm.litellm_core_utils.hidden_params import set_hidden_param from litellm.litellm_core_utils.url_utils import encode_url_path_segment from litellm.llms.azure_ai.agents.transformation import ( AzureAIAgentsConfig, @@ -231,9 +232,7 @@ class AzureAIAgentsHandler: model_response.model = model # Store thread_id for conversation continuity - if not hasattr(model_response, "_hidden_params") or model_response._hidden_params is None: - model_response._hidden_params = {} - model_response._hidden_params["thread_id"] = thread_id + set_hidden_param(model_response, "thread_id", thread_id) # Estimate token usage try: @@ -660,7 +659,7 @@ class AzureAIAgentsHandler: ], ) if thread_id: - final_chunk._hidden_params = {"thread_id": thread_id} + final_chunk.hidden_params = {"thread_id": thread_id} yield final_chunk return @@ -706,7 +705,7 @@ class AzureAIAgentsHandler: ], ) if thread_id: - chunk._hidden_params = {"thread_id": thread_id} + chunk.hidden_params = {"thread_id": thread_id} yield chunk diff --git a/litellm/llms/azure_ai/azure_model_router/transformation.py b/litellm/llms/azure_ai/azure_model_router/transformation.py index 226cd9c13f9..9fb22abbe5e 100644 --- a/litellm/llms/azure_ai/azure_model_router/transformation.py +++ b/litellm/llms/azure_ai/azure_model_router/transformation.py @@ -72,13 +72,14 @@ class AzureModelRouterConfig(AzureAIStudioConfig): Also stamps that model onto ``_hidden_params`` so downstream consumers (spend logs, response restamping) can read it instead of guessing the route from the model string. """ + from litellm.litellm_core_utils.hidden_params import ( + get_hidden_params, + set_hidden_params, + ) from litellm.llms.azure_ai.common_utils import ( AZURE_MODEL_ROUTER_SELECTED_MODEL_KEY, AzureFoundryModelInfo, ) - from litellm.router_utils.add_retry_fallback_headers import ( - get_hidden_params_dict, - ) # Get base model for the parent call (strips routing prefixes for API compatibility) base_model: Final[str] = AzureFoundryModelInfo.get_base_model(model) @@ -100,12 +101,14 @@ class AzureModelRouterConfig(AzureAIStudioConfig): ) selected_model: Final = transformed_response.model if selected_model: - # Rebuilt rather than mutated in place: ModelResponseBase declares _hidden_params as a - # class-level dict, so an in-place write can bleed into unrelated responses. - transformed_response._hidden_params = { # pyright: ignore[reportPrivateUsage] # ModelResponse exposes no public hidden-params setter - **get_hidden_params_dict(transformed_response), - AZURE_MODEL_ROUTER_SELECTED_MODEL_KEY: selected_model, - } + transformed_hidden_params: Final = get_hidden_params(transformed_response) or {} + set_hidden_params( + transformed_response, + { + **transformed_hidden_params, + AZURE_MODEL_ROUTER_SELECTED_MODEL_KEY: selected_model, + }, + ) return transformed_response def calculate_additional_costs(self, model: str, prompt_tokens: int, completion_tokens: int) -> dict | None: diff --git a/litellm/llms/azure_ai/embed/cohere_transformation.py b/litellm/llms/azure_ai/embed/cohere_transformation.py index e60d419255c..453e48d6819 100644 --- a/litellm/llms/azure_ai/embed/cohere_transformation.py +++ b/litellm/llms/azure_ai/embed/cohere_transformation.py @@ -68,7 +68,7 @@ class AzureAICohereConfig: return image_embeddings_request, v1_embeddings_request, image_embedding_idx def _transform_response(self, response: EmbeddingResponse) -> EmbeddingResponse: - additional_headers: Final[dict | None] = response._hidden_params.get("additional_headers") + additional_headers: Final[dict | None] = response.hidden_params.get("additional_headers") if additional_headers: # CALCULATE USAGE input_tokens: Final[str | None] = additional_headers.get("llm_provider-num_tokens") diff --git a/litellm/llms/azure_ai/rerank/transformation.py b/litellm/llms/azure_ai/rerank/transformation.py index 24317f8057e..4c26ba58dd8 100644 --- a/litellm/llms/azure_ai/rerank/transformation.py +++ b/litellm/llms/azure_ai/rerank/transformation.py @@ -105,8 +105,9 @@ class AzureAIRerankConfig(CohereRerankConfig): optional_params=optional_params, litellm_params=litellm_params, ) - base_model: Final = self._get_base_model(rerank_response._hidden_params.get("llm_provider-azureml-model-group")) - rerank_response._hidden_params["model"] = base_model + azure_model_group: Final = rerank_response.hidden_params.get("llm_provider-azureml-model-group") + base_model: Final = self._get_base_model(azure_model_group if isinstance(azure_model_group, str) else None) + rerank_response.hidden_params["model"] = base_model return rerank_response def _get_base_model(self, azure_model_group: str | None) -> str | None: diff --git a/litellm/llms/base_llm/ocr/transformation.py b/litellm/llms/base_llm/ocr/transformation.py index f0c85dd379e..97d402e7a1b 100644 --- a/litellm/llms/base_llm/ocr/transformation.py +++ b/litellm/llms/base_llm/ocr/transformation.py @@ -91,6 +91,14 @@ class OCRResponse(LiteLLMPydanticObjectBase): _hidden_params: dict = PrivateAttr(default_factory=dict) + @property + def hidden_params(self) -> dict[str, builtins.object]: # mutable-ok: API requires mutation + return self._hidden_params + + @hidden_params.setter + def hidden_params(self, hidden_params: dict[str, builtins.object]) -> None: # mutable-ok: API requires mutation + self._hidden_params = hidden_params + def set_provider_native_response(self, native_response: Mapping[str, builtins.object]) -> None: """Keep the provider's own response payload alongside the normalized one.""" self._hidden_params[PROVIDER_NATIVE_RESPONSE_KEY] = native_response diff --git a/litellm/llms/base_llm/sandbox/transformation.py b/litellm/llms/base_llm/sandbox/transformation.py index d19c0744b08..877edc15ec7 100644 --- a/litellm/llms/base_llm/sandbox/transformation.py +++ b/litellm/llms/base_llm/sandbox/transformation.py @@ -6,6 +6,7 @@ returns whatever the sandbox produced. The lifecycle is create container -> run code -> delete container; `code_interpreter_tool` combines all three. """ +import builtins from typing import Any, Final import httpx @@ -27,6 +28,14 @@ class ContainerHandle(LiteLLMPydanticObjectBase): _hidden_params: dict = PrivateAttr(default_factory=dict) + @property + def hidden_params(self) -> dict[str, builtins.object]: # mutable-ok: API requires mutation + return self._hidden_params + + @hidden_params.setter + def hidden_params(self, hidden_params: dict[str, builtins.object]) -> None: # mutable-ok: API requires mutation + self._hidden_params = hidden_params + class CodeExecutionResult(LiteLLMPydanticObjectBase): """Passthrough of the sandbox's own execution output.""" @@ -42,6 +51,14 @@ class CodeExecutionResult(LiteLLMPydanticObjectBase): _hidden_params: dict = PrivateAttr(default_factory=dict) + @property + def hidden_params(self) -> dict[str, builtins.object]: # mutable-ok: API requires mutation + return self._hidden_params + + @hidden_params.setter + def hidden_params(self, hidden_params: dict[str, builtins.object]) -> None: # mutable-ok: API requires mutation + self._hidden_params = hidden_params + class BaseSandboxConfig: """Provider-agnostic sandbox operations.""" diff --git a/litellm/llms/base_llm/search/transformation.py b/litellm/llms/base_llm/search/transformation.py index 4edbf260d99..8ff1beac782 100644 --- a/litellm/llms/base_llm/search/transformation.py +++ b/litellm/llms/base_llm/search/transformation.py @@ -2,6 +2,7 @@ Base Search transformation configuration. """ +import builtins from typing import TYPE_CHECKING, Any, Final, Literal from urllib.parse import urlsplit @@ -77,6 +78,14 @@ class SearchResponse(LiteLLMPydanticObjectBase): # Define private attributes using PrivateAttr _hidden_params: dict = PrivateAttr(default_factory=dict) + @property + def hidden_params(self) -> dict[str, builtins.object]: # mutable-ok: API requires mutation + return self._hidden_params + + @hidden_params.setter + def hidden_params(self, hidden_params: dict[str, builtins.object]) -> None: # mutable-ok: API requires mutation + self._hidden_params = hidden_params + class BaseSearchConfig: """ diff --git a/litellm/llms/bedrock/image_edit/amazon_nova_canvas_image_edit_transformation.py b/litellm/llms/bedrock/image_edit/amazon_nova_canvas_image_edit_transformation.py index 463511ff333..f94f4a3657c 100644 --- a/litellm/llms/bedrock/image_edit/amazon_nova_canvas_image_edit_transformation.py +++ b/litellm/llms/bedrock/image_edit/amazon_nova_canvas_image_edit_transformation.py @@ -14,7 +14,7 @@ from __future__ import annotations import base64 import os -from typing import TYPE_CHECKING, Any, Final +from typing import TYPE_CHECKING, Any, Final, cast import httpx @@ -449,17 +449,18 @@ class BedrockAmazonNovaCanvasImageEditConfig(BaseImageEditConfig): ) if not hasattr(model_response, "_hidden_params"): - model_response._hidden_params = {} - if "additional_headers" not in model_response._hidden_params: - model_response._hidden_params["additional_headers"] = {} + model_response.hidden_params = {} + additional_headers: Final = cast( # cast-ok: provider headers are stored as a mutable mapping + dict[str, object], model_response.hidden_params.setdefault("additional_headers", {}) + ) try: model_info: Final = get_model_info(model, custom_llm_provider="bedrock") cost_per_image: Final = model_info.get("output_cost_per_image", 0) if cost_per_image is not None and model_response.data: - model_response._hidden_params["additional_headers"]["llm_provider-x-litellm-response-cost"] = float( - cost_per_image - ) * len(model_response.data) + additional_headers["llm_provider-x-litellm-response-cost"] = float(cost_per_image) * len( + model_response.data + ) except Exception: pass diff --git a/litellm/llms/bedrock/image_edit/stability_transformation.py b/litellm/llms/bedrock/image_edit/stability_transformation.py index f66dcf130b8..dc8932b76d0 100644 --- a/litellm/llms/bedrock/image_edit/stability_transformation.py +++ b/litellm/llms/bedrock/image_edit/stability_transformation.py @@ -22,7 +22,7 @@ API Reference: https://docs.aws.amazon.com/bedrock/latest/userguide/model-parame """ import base64 -from typing import TYPE_CHECKING, Any, Final +from typing import TYPE_CHECKING, Any, Final, cast import httpx from httpx._types import RequestFiles @@ -335,17 +335,16 @@ class BedrockStabilityImageEditConfig(BaseImageEditConfig): ) if not hasattr(model_response, "_hidden_params"): - model_response._hidden_params = {} - if "additional_headers" not in model_response._hidden_params: - model_response._hidden_params["additional_headers"] = {} + model_response.hidden_params = {} + additional_headers: Final = cast( # cast-ok: provider headers are stored as a mutable mapping + dict[str, object], model_response.hidden_params.setdefault("additional_headers", {}) + ) # Set cost based on model model_info: Final = get_model_info(model, custom_llm_provider="bedrock") cost_per_image: Final = model_info.get("output_cost_per_image", 0) if cost_per_image is not None: - model_response._hidden_params["additional_headers"]["llm_provider-x-litellm-response-cost"] = float( - cost_per_image - ) + additional_headers["llm_provider-x-litellm-response-cost"] = float(cost_per_image) return model_response diff --git a/litellm/llms/bytez/chat/transformation.py b/litellm/llms/bytez/chat/transformation.py index 12338d83017..6a5f97e46b4 100644 --- a/litellm/llms/bytez/chat/transformation.py +++ b/litellm/llms/bytez/chat/transformation.py @@ -231,7 +231,7 @@ class BytezChatConfig(BaseConfig): model_response.usage = usage - model_response._hidden_params["additional_headers"] = raw_response.headers + model_response.hidden_params["additional_headers"] = raw_response.headers message.provider_specific_fields = { "ratelimit-limit": raw_response.headers.get("ratelimit-limit"), "ratelimit-remaining": raw_response.headers.get("ratelimit-remaining"), diff --git a/litellm/llms/chatgpt/responses/transformation.py b/litellm/llms/chatgpt/responses/transformation.py index ae998238cd0..7163cc5d3e0 100644 --- a/litellm/llms/chatgpt/responses/transformation.py +++ b/litellm/llms/chatgpt/responses/transformation.py @@ -1,10 +1,11 @@ from collections.abc import Mapping -from typing import TYPE_CHECKING, Any, Final +from typing import TYPE_CHECKING, Any, Final, cast import httpx from litellm.exceptions import AuthenticationError from litellm.litellm_core_utils.core_helpers import process_response_headers +from litellm.litellm_core_utils.hidden_params import HIDDEN_PARAMS_ATTR, set_hidden_params from litellm.litellm_core_utils.llm_response_utils.convert_dict_to_response import ( safe_convert_created_field, ) @@ -237,10 +238,13 @@ class ChatGPTResponsesAPIConfig(OpenAIResponsesAPIConfig): ) -> None: raw_headers: Final = dict(raw_response.headers) processed_headers: Final = process_response_headers(raw_headers) - if not hasattr(completed_response, "_hidden_params"): - setattr(completed_response, "_hidden_params", {}) - completed_response._hidden_params["additional_headers"] = processed_headers - completed_response._hidden_params["headers"] = raw_headers + if not hasattr(completed_response, HIDDEN_PARAMS_ATTR): + set_hidden_params(completed_response, {}) + hidden_params: Final = cast( # cast-ok: preserve dynamic mapping behavior + dict[str, object], getattr(completed_response, HIDDEN_PARAMS_ATTR) + ) + hidden_params["additional_headers"] = processed_headers + hidden_params["headers"] = raw_headers def get_complete_url( self, diff --git a/litellm/llms/deepgram/audio_transcription/transformation.py b/litellm/llms/deepgram/audio_transcription/transformation.py index c5c9fe12423..a3812dac8ef 100644 --- a/litellm/llms/deepgram/audio_transcription/transformation.py +++ b/litellm/llms/deepgram/audio_transcription/transformation.py @@ -123,7 +123,7 @@ class DeepgramAudioTranscriptionConfig(BaseAudioTranscriptionConfig): ] # Store full response in hidden params - response._hidden_params = response_json + response.hidden_params = response_json return response diff --git a/litellm/llms/deepinfra/rerank/transformation.py b/litellm/llms/deepinfra/rerank/transformation.py index a3d0482af0a..5196c44d504 100644 --- a/litellm/llms/deepinfra/rerank/transformation.py +++ b/litellm/llms/deepinfra/rerank/transformation.py @@ -213,7 +213,7 @@ class DeepinfraRerankConfig(BaseRerankConfig): rerank_response = RerankResponse(id=request_id or str(uuid.uuid4()), results=results, meta=meta) # Store additional information in hidden params - rerank_response._hidden_params = { + rerank_response.hidden_params = { "status": status, "runtime_ms": runtime_ms, "cost": cost, @@ -237,7 +237,7 @@ class DeepinfraRerankConfig(BaseRerankConfig): litellm_params=litellm_params, ) - rerank_response._hidden_params["model"] = model + rerank_response.hidden_params["model"] = model return rerank_response def get_supported_cohere_rerank_params(self, model: str) -> list: diff --git a/litellm/llms/e2b/sandbox/transformation.py b/litellm/llms/e2b/sandbox/transformation.py index d2702b116dc..dbf3996248a 100644 --- a/litellm/llms/e2b/sandbox/transformation.py +++ b/litellm/llms/e2b/sandbox/transformation.py @@ -9,11 +9,12 @@ Talks to e2b's REST API directly over httpx (no e2b SDK dependency): import json from collections.abc import Mapping -from typing import Final +from typing import Final, cast import httpx from pydantic import ConfigDict, TypeAdapter +from litellm.litellm_core_utils.hidden_params import HIDDEN_PARAMS_ATTR from litellm.llms.base_llm.sandbox.transformation import ( SANDBOX_MAX_OUTPUT_BYTES, BaseSandboxConfig, @@ -90,7 +91,7 @@ class E2BSandboxConfig(BaseSandboxConfig): "domain": data.get("domain") or E2B_DEFAULT_DOMAIN, } ) - handle._hidden_params = { + handle.hidden_params = { "envd_access_token": data.get("envdAccessToken"), "traffic_access_token": data.get("trafficAccessToken"), "api_key": key, @@ -109,8 +110,11 @@ class E2BSandboxConfig(BaseSandboxConfig): **kwargs, ) -> CodeExecutionResult: handle: Final = self._as_handle(container) + hidden_params: Final = cast( # cast-ok: preserve mapping operations on the validated handle + dict[str, object], getattr(handle, HIDDEN_PARAMS_ATTR) + ) - token: Final = handle._hidden_params.get("envd_access_token") + token: Final = hidden_params.get("envd_access_token") if not token: raise ValueError( "Cannot run code from a sandbox id alone. e2b secure sandboxes " @@ -119,7 +123,7 @@ class E2BSandboxConfig(BaseSandboxConfig): ) headers: Final = {"Content-Type": "application/json", "X-Access-Token": token} - traffic_token: Final = handle._hidden_params.get("traffic_access_token") + traffic_token: Final = hidden_params.get("traffic_access_token") if traffic_token: headers["E2B-Traffic-Access-Token"] = traffic_token @@ -143,8 +147,11 @@ class E2BSandboxConfig(BaseSandboxConfig): **kwargs, ) -> bool: handle: Final = self._as_handle(container) - key: Final = api_key or handle._hidden_params.get("api_key") or self.validate_environment() - base: Final = api_base or handle._hidden_params.get("api_base") or E2B_API_BASE + hidden_params: Final = cast( # cast-ok: preserve mapping operations on the validated handle + dict[str, object], getattr(handle, HIDDEN_PARAMS_ATTR) + ) + key: Final = api_key or hidden_params.get("api_key") or self.validate_environment() + base: Final = api_base or hidden_params.get("api_base") or E2B_API_BASE try: response: Final = await self._http(client).delete( url=f"{base}/sandboxes/{handle.id}", @@ -161,7 +168,7 @@ class E2BSandboxConfig(BaseSandboxConfig): if isinstance(container, ContainerHandle): return container handle: Final = ContainerHandle(id=str(container), provider="e2b", domain=E2B_DEFAULT_DOMAIN) - handle._hidden_params = {} + handle.hidden_params = {} return handle @staticmethod diff --git a/litellm/llms/elevenlabs/audio_transcription/transformation.py b/litellm/llms/elevenlabs/audio_transcription/transformation.py index 0dbf61917e2..04d2717d64f 100644 --- a/litellm/llms/elevenlabs/audio_transcription/transformation.py +++ b/litellm/llms/elevenlabs/audio_transcription/transformation.py @@ -147,7 +147,7 @@ class ElevenLabsAudioTranscriptionConfig(BaseAudioTranscriptionConfig): ) # Store full response in hidden params - response._hidden_params = response_json + response.hidden_params = response_json return response diff --git a/litellm/llms/fal_ai/image_generation/flux_pro_v11_ultra_transformation.py b/litellm/llms/fal_ai/image_generation/flux_pro_v11_ultra_transformation.py index d4d2941db43..0fdd89aec91 100644 --- a/litellm/llms/fal_ai/image_generation/flux_pro_v11_ultra_transformation.py +++ b/litellm/llms/fal_ai/image_generation/flux_pro_v11_ultra_transformation.py @@ -1,9 +1,10 @@ from collections.abc import Mapping -from typing import TYPE_CHECKING, Any, Final +from typing import TYPE_CHECKING, Any, Final, cast import httpx from pydantic import ConfigDict, TypeAdapter +from litellm.litellm_core_utils.hidden_params import HIDDEN_PARAMS_ATTR from litellm.types.llms.openai import OpenAIImageGenerationOptionalParams from litellm.types.utils import ImageResponse @@ -237,12 +238,15 @@ class FalAIFluxProV11UltraConfig(FalAIBaseConfig): model_response.data.extend(fal_images_to_image_objects(images)) # Add additional metadata from Flux Pro response - if hasattr(model_response, "_hidden_params"): + if hasattr(model_response, HIDDEN_PARAMS_ATTR): + hidden_params: Final = cast( # cast-ok: preserve mapping operations on dynamic response metadata + dict[str, object], getattr(model_response, HIDDEN_PARAMS_ATTR) + ) if "seed" in response_object: - model_response._hidden_params["seed"] = response_object["seed"] + hidden_params["seed"] = response_object["seed"] if "timings" in response_object: - model_response._hidden_params["timings"] = response_object["timings"] + hidden_params["timings"] = response_object["timings"] if "has_nsfw_concepts" in response_object: - model_response._hidden_params["has_nsfw_concepts"] = response_object["has_nsfw_concepts"] + hidden_params["has_nsfw_concepts"] = response_object["has_nsfw_concepts"] return model_response diff --git a/litellm/llms/fal_ai/image_generation/ideogram_v3_transformation.py b/litellm/llms/fal_ai/image_generation/ideogram_v3_transformation.py index 2ca423fafff..ff56f3473d5 100644 --- a/litellm/llms/fal_ai/image_generation/ideogram_v3_transformation.py +++ b/litellm/llms/fal_ai/image_generation/ideogram_v3_transformation.py @@ -1,9 +1,10 @@ from collections.abc import Mapping -from typing import TYPE_CHECKING, Any, Final +from typing import TYPE_CHECKING, Any, Final, cast import httpx from pydantic import ConfigDict, TypeAdapter +from litellm.litellm_core_utils.hidden_params import HIDDEN_PARAMS_ATTR from litellm.types.llms.openai import OpenAIImageGenerationOptionalParams from litellm.types.utils import ImageObject, ImageResponse @@ -189,7 +190,10 @@ class FalAIIdeogramV3Config(FalAIBaseConfig): ) ) - if hasattr(model_response, "_hidden_params") and "seed" in response_object: - model_response._hidden_params["seed"] = response_object["seed"] + if hasattr(model_response, HIDDEN_PARAMS_ATTR) and "seed" in response_object: + hidden_params: Final = cast( # cast-ok: preserve mapping operations on dynamic response metadata + dict[str, object], getattr(model_response, HIDDEN_PARAMS_ATTR) + ) + hidden_params["seed"] = response_object["seed"] return model_response diff --git a/litellm/llms/fal_ai/image_generation/imagen4_transformation.py b/litellm/llms/fal_ai/image_generation/imagen4_transformation.py index 3624b76a4a3..c5a5f0112ca 100644 --- a/litellm/llms/fal_ai/image_generation/imagen4_transformation.py +++ b/litellm/llms/fal_ai/image_generation/imagen4_transformation.py @@ -1,7 +1,8 @@ -from typing import TYPE_CHECKING, Any, Final +from typing import TYPE_CHECKING, Any, Final, cast import httpx +from litellm.litellm_core_utils.hidden_params import HIDDEN_PARAMS_ATTR from litellm.types.llms.openai import OpenAIImageGenerationOptionalParams from litellm.types.utils import ImageObject, ImageResponse @@ -234,8 +235,11 @@ class FalAIImagen4Config(FalAIBaseConfig): ) # Add seed metadata from Imagen4 response - if hasattr(model_response, "_hidden_params"): + if hasattr(model_response, HIDDEN_PARAMS_ATTR): + hidden_params: Final = cast( # cast-ok: preserve mapping operations on dynamic response metadata + dict[str, object], getattr(model_response, HIDDEN_PARAMS_ATTR) + ) if "seed" in response_data: - model_response._hidden_params["seed"] = response_data["seed"] + hidden_params["seed"] = response_data["seed"] return model_response diff --git a/litellm/llms/fal_ai/image_generation/stable_diffusion_transformation.py b/litellm/llms/fal_ai/image_generation/stable_diffusion_transformation.py index 659011ae537..30d51cc8084 100644 --- a/litellm/llms/fal_ai/image_generation/stable_diffusion_transformation.py +++ b/litellm/llms/fal_ai/image_generation/stable_diffusion_transformation.py @@ -1,9 +1,10 @@ from collections.abc import Mapping -from typing import TYPE_CHECKING, Any, Final +from typing import TYPE_CHECKING, Any, Final, cast import httpx from pydantic import ConfigDict, TypeAdapter +from litellm.litellm_core_utils.hidden_params import HIDDEN_PARAMS_ATTR from litellm.types.llms.openai import OpenAIImageGenerationOptionalParams from litellm.types.utils import ImageObject, ImageResponse @@ -268,12 +269,15 @@ class FalAIStableDiffusionConfig(FalAIBaseConfig): ) # Add additional metadata from Stable Diffusion response - if hasattr(model_response, "_hidden_params"): + if hasattr(model_response, HIDDEN_PARAMS_ATTR): + hidden_params: Final = cast( # cast-ok: preserve mapping operations on dynamic response metadata + dict[str, object], getattr(model_response, HIDDEN_PARAMS_ATTR) + ) if "seed" in response_object: - model_response._hidden_params["seed"] = response_object["seed"] + hidden_params["seed"] = response_object["seed"] if "timings" in response_object: - model_response._hidden_params["timings"] = response_object["timings"] + hidden_params["timings"] = response_object["timings"] if "has_nsfw_concepts" in response_object: - model_response._hidden_params["has_nsfw_concepts"] = response_object["has_nsfw_concepts"] + hidden_params["has_nsfw_concepts"] = response_object["has_nsfw_concepts"] return model_response diff --git a/litellm/llms/fireworks_ai/chat/transformation.py b/litellm/llms/fireworks_ai/chat/transformation.py index e8f17fe0ba9..0a8a010f2b9 100644 --- a/litellm/llms/fireworks_ai/chat/transformation.py +++ b/litellm/llms/fireworks_ai/chat/transformation.py @@ -756,7 +756,7 @@ class FireworksAIConfig(FireworksAIMixin, OpenAIGPTConfig): tool_calls=optional_params.get("tools", None), ) - response._hidden_params = { + response.hidden_params = { "additional_headers": additional_headers, **_extract_fireworks_hidden_params(completion_response), } diff --git a/litellm/llms/gemini/interactions/transformation.py b/litellm/llms/gemini/interactions/transformation.py index 0e898147d90..f0e6bb12d39 100644 --- a/litellm/llms/gemini/interactions/transformation.py +++ b/litellm/llms/gemini/interactions/transformation.py @@ -274,8 +274,8 @@ class GoogleAIStudioInteractionsConfig(BaseInteractionsAPIConfig): verbose_logger.debug("Google AI Interactions response: %s", raw_json) response: Final = InteractionsAPIResponse(**raw_json) - response._hidden_params["headers"] = dict(raw_response.headers) - response._hidden_params["additional_headers"] = process_response_headers(dict(raw_response.headers)) + response.hidden_params["headers"] = dict(raw_response.headers) + response.hidden_params["additional_headers"] = process_response_headers(dict(raw_response.headers)) return response @@ -328,7 +328,7 @@ class GoogleAIStudioInteractionsConfig(BaseInteractionsAPIConfig): headers=dict(raw_response.headers), ) response: Final = InteractionsAPIResponse(**raw_json) - response._hidden_params["headers"] = dict(raw_response.headers) + response.hidden_params["headers"] = dict(raw_response.headers) return response def transform_delete_interaction_request( diff --git a/litellm/llms/huggingface/embedding/transformation.py b/litellm/llms/huggingface/embedding/transformation.py index 3fdd4abda73..334fc9a5d9a 100644 --- a/litellm/llms/huggingface/embedding/transformation.py +++ b/litellm/llms/huggingface/embedding/transformation.py @@ -465,7 +465,7 @@ class HuggingFaceEmbeddingConfig(BaseConfig): total_tokens=prompt_tokens + completion_tokens, ) setattr(model_response, "usage", usage) - model_response._hidden_params["original_response"] = completion_response + model_response.hidden_params["original_response"] = completion_response return model_response def transform_response( diff --git a/litellm/llms/manus/responses/transformation.py b/litellm/llms/manus/responses/transformation.py index 775ae3e3ad8..41ccee3589d 100644 --- a/litellm/llms/manus/responses/transformation.py +++ b/litellm/llms/manus/responses/transformation.py @@ -225,8 +225,8 @@ class ManusResponsesAPIConfig(OpenAIResponsesAPIConfig): response = ResponsesAPIResponse.model_construct(**raw_response_json) # Store processed headers in additional_headers so they get returned to the client - response._hidden_params["additional_headers"] = processed_headers - response._hidden_params["headers"] = raw_response_headers + response.hidden_params["additional_headers"] = processed_headers + response.hidden_params["headers"] = raw_response_headers return response def supports_native_websocket(self) -> bool: @@ -315,6 +315,6 @@ class ManusResponsesAPIConfig(OpenAIResponsesAPIConfig): response = ResponsesAPIResponse.model_construct(**raw_response_json) # Store processed headers in additional_headers so they get returned to the client - response._hidden_params["additional_headers"] = processed_headers - response._hidden_params["headers"] = raw_response_headers + response.hidden_params["additional_headers"] = processed_headers + response.hidden_params["headers"] = raw_response_headers return response diff --git a/litellm/llms/mistral/audio_transcription/transformation.py b/litellm/llms/mistral/audio_transcription/transformation.py index 9b1d343b94c..4fce7a95350 100644 --- a/litellm/llms/mistral/audio_transcription/transformation.py +++ b/litellm/llms/mistral/audio_transcription/transformation.py @@ -147,5 +147,5 @@ class MistralAudioTranscriptionConfig(BaseAudioTranscriptionConfig): if "language" in response_json: response["language"] = response_json["language"] - response._hidden_params = response_json + response.hidden_params = response_json return response diff --git a/litellm/llms/mistral/chat/transformation.py b/litellm/llms/mistral/chat/transformation.py index b9d70244b0a..ede9467f7f8 100644 --- a/litellm/llms/mistral/chat/transformation.py +++ b/litellm/llms/mistral/chat/transformation.py @@ -229,7 +229,7 @@ class MistralConfig(OpenAIGPTConfig): @overload def _transform_messages( self, messages: list[AllMessageValues], model: str, is_async: Literal[True] - ) -> Coroutine[object, object, list[AllMessageValues]]: + ) -> Coroutine[object, object, list[AllMessageValues]]: ... @overload diff --git a/litellm/llms/oci/chat/transformation.py b/litellm/llms/oci/chat/transformation.py index e3c4b8d93f2..7ec0e1c9cf6 100644 --- a/litellm/llms/oci/chat/transformation.py +++ b/litellm/llms/oci/chat/transformation.py @@ -626,7 +626,7 @@ class OCIChatConfig(BaseConfig): else: model_response = handle_generic_response(response_json, model, model_response, raw_response) - model_response._hidden_params["additional_headers"] = raw_response.headers + model_response.hidden_params["additional_headers"] = raw_response.headers return model_response @track_llm_api_timing() diff --git a/litellm/llms/openai/completion/handler.py b/litellm/llms/openai/completion/handler.py index f9267873ea2..abacb0b815a 100644 --- a/litellm/llms/openai/completion/handler.py +++ b/litellm/llms/openai/completion/handler.py @@ -204,7 +204,7 @@ class OpenAITextCompletion(BaseLLM): ) ## RESPONSE OBJECT response_obj: Final = TextCompletionResponse(**response_json) - response_obj._hidden_params.original_response = json.dumps(response_json) + response_obj.hidden_params.original_response = json.dumps(response_json) return response_obj except Exception as e: status_code: Final = getattr(e, "status_code", 500) diff --git a/litellm/llms/openai/completion/transformation.py b/litellm/llms/openai/completion/transformation.py index 19ba79720a5..a690a34f326 100644 --- a/litellm/llms/openai/completion/transformation.py +++ b/litellm/llms/openai/completion/transformation.py @@ -5,6 +5,7 @@ Support for gpt model family from collections.abc import Mapping from typing import Final +from litellm.litellm_core_utils.hidden_params import set_hidden_param from litellm.llms.base_llm.completion.transformation import BaseTextCompletionConfig from litellm.types.llms.openai import AllMessageValues, OpenAITextCompletionUserMessage from litellm.types.utils import Choices, Message, ModelResponse, TextCompletionResponse @@ -112,8 +113,10 @@ class OpenAITextCompletionConfig(BaseTextCompletionConfig, OpenAIGPTConfig): if "model" in response_object: model_response_object.model = response_object["model"] - model_response_object._hidden_params["original_response"] = ( - response_object # track original response, if users make a litellm.text_completion() request, we can return the original response + set_hidden_param( + model_response_object, + "original_response", + response_object, ) return model_response_object except Exception as e: diff --git a/litellm/llms/openai/containers/transformation.py b/litellm/llms/openai/containers/transformation.py index 1a5211d5ff5..bff5e0cea10 100644 --- a/litellm/llms/openai/containers/transformation.py +++ b/litellm/llms/openai/containers/transformation.py @@ -1,10 +1,11 @@ from collections.abc import Mapping -from typing import TYPE_CHECKING, Any, Final, Literal +from typing import TYPE_CHECKING, Any, Final, Literal, cast import httpx from typing_extensions import ReadOnly, TypedDict import litellm +from litellm.litellm_core_utils.hidden_params import get_or_create_hidden_params from litellm.litellm_core_utils.llm_cost_calc.tool_call_cost_tracking import ( StandardBuiltInToolCostTracking, ) @@ -165,11 +166,12 @@ class OpenAIContainerConfig(BaseContainerConfig): provider="openai", ) - if not hasattr(container_obj, "_hidden_params") or container_obj._hidden_params is None: - container_obj._hidden_params = {} - if "additional_headers" not in container_obj._hidden_params: - container_obj._hidden_params["additional_headers"] = {} - container_obj._hidden_params["additional_headers"]["llm_provider-x-litellm-response-cost"] = container_cost + container_hidden_params: Final = get_or_create_hidden_params(container_obj) + container_hidden_params.setdefault("additional_headers", {}) + container_additional_headers: Final = cast( # cast-ok: preserve mapping operations on response metadata + "dict[str, object]", container_hidden_params["additional_headers"] + ) + container_additional_headers["llm_provider-x-litellm-response-cost"] = container_cost return container_obj diff --git a/litellm/llms/openai/responses/transformation.py b/litellm/llms/openai/responses/transformation.py index e7463fb66f6..ec4f0c04475 100644 --- a/litellm/llms/openai/responses/transformation.py +++ b/litellm/llms/openai/responses/transformation.py @@ -593,8 +593,8 @@ class OpenAIResponsesAPIConfig(BaseResponsesAPIConfig): response = ResponsesAPIResponse.model_construct(**raw_response_json) # Store processed headers in additional_headers so they get returned to the client - response._hidden_params["additional_headers"] = processed_headers - response._hidden_params["headers"] = raw_response_headers + response.hidden_params["additional_headers"] = processed_headers + response.hidden_params["headers"] = raw_response_headers return response def validate_environment(self, headers: dict, model: str, litellm_params: GenericLiteLLMParams | None) -> dict: @@ -841,8 +841,8 @@ class OpenAIResponsesAPIConfig(BaseResponsesAPIConfig): raw_response_headers: Final = dict(raw_response.headers) processed_headers: Final = process_response_headers(raw_response_headers) response: Final = ResponsesAPIResponse.model_validate(raw_response_json) - response._hidden_params["additional_headers"] = processed_headers - response._hidden_params["headers"] = raw_response_headers + response.hidden_params["additional_headers"] = processed_headers + response.hidden_params["headers"] = raw_response_headers return response @@ -923,8 +923,8 @@ class OpenAIResponsesAPIConfig(BaseResponsesAPIConfig): processed_headers: Final = process_response_headers(raw_response_headers) response: Final = ResponsesAPIResponse.model_validate(raw_response_json) - response._hidden_params["additional_headers"] = processed_headers - response._hidden_params["headers"] = raw_response_headers + response.hidden_params["additional_headers"] = processed_headers + response.hidden_params["headers"] = raw_response_headers return response @@ -993,7 +993,7 @@ class OpenAIResponsesAPIConfig(BaseResponsesAPIConfig): ) response = ResponsesAPIResponse.model_construct(**raw_response_json) - response._hidden_params["additional_headers"] = processed_headers - response._hidden_params["headers"] = raw_response_headers + response.hidden_params["additional_headers"] = processed_headers + response.hidden_params["headers"] = raw_response_headers return response diff --git a/litellm/llms/openai_like/chat/transformation.py b/litellm/llms/openai_like/chat/transformation.py index e5d6cbb7e5e..b0770f1a7e1 100644 --- a/litellm/llms/openai_like/chat/transformation.py +++ b/litellm/llms/openai_like/chat/transformation.py @@ -117,7 +117,7 @@ class OpenAILikeChatConfig(OpenAIGPTConfig): returned_response.model = custom_llm_provider + "/" + (returned_response.model or "") if base_model is not None: - returned_response._hidden_params["model"] = base_model + returned_response.hidden_params["model"] = base_model return returned_response def transform_response( diff --git a/litellm/llms/openrouter/chat/transformation.py b/litellm/llms/openrouter/chat/transformation.py index 08d43c169f4..e97962a1c48 100644 --- a/litellm/llms/openrouter/chat/transformation.py +++ b/litellm/llms/openrouter/chat/transformation.py @@ -13,6 +13,7 @@ from typing import TYPE_CHECKING, Any, Final, cast import httpx import litellm +from litellm.litellm_core_utils.hidden_params import HIDDEN_PARAMS_ATTR, set_hidden_params from litellm.llms.base_llm.base_model_iterator import BaseModelResponseIterator from litellm.llms.base_llm.chat.transformation import BaseLLMException from litellm.types.llms.openai import AllMessageValues, ChatCompletionToolParam @@ -216,13 +217,17 @@ class OpenrouterConfig(OpenAIGPTConfig): response_cost: Final = response_json["usage"].get("cost") if response_cost is not None: # Store cost in hidden params for the cost calculator to use - if not hasattr(model_response, "_hidden_params"): - model_response._hidden_params = {} - if "additional_headers" not in model_response._hidden_params: - model_response._hidden_params["additional_headers"] = {} - model_response._hidden_params["additional_headers"]["llm_provider-x-litellm-response-cost"] = float( - response_cost + if not hasattr(model_response, HIDDEN_PARAMS_ATTR): + set_hidden_params(model_response, {}) + hidden_params: Final = cast( # cast-ok: preserve mapping operations on dynamic response metadata + dict[str, object], getattr(model_response, HIDDEN_PARAMS_ATTR) ) + if "additional_headers" not in hidden_params: + hidden_params["additional_headers"] = {} + additional_headers: Final = cast( # cast-ok: preserve mapping operations on response metadata + dict[str, object], hidden_params["additional_headers"] + ) + additional_headers["llm_provider-x-litellm-response-cost"] = float(response_cost) except Exception: # If we can't extract cost, continue without it - don't fail the response pass diff --git a/litellm/llms/openrouter/image_edit/transformation.py b/litellm/llms/openrouter/image_edit/transformation.py index 3d46277a69e..96b16e3e510 100644 --- a/litellm/llms/openrouter/image_edit/transformation.py +++ b/litellm/llms/openrouter/image_edit/transformation.py @@ -334,20 +334,20 @@ class OpenRouterImageEditConfig(BaseImageEditConfig): cost: Final = usage_data.get("cost") if cost is not None: if not hasattr(model_response, "_hidden_params"): - model_response._hidden_params = {} - if "additional_headers" not in model_response._hidden_params: - model_response._hidden_params["additional_headers"] = {} - model_response._hidden_params["additional_headers"]["llm_provider-x-litellm-response-cost"] = float( - cost + model_response.hidden_params = {} + additional_headers: Final = cast( # cast-ok: provider headers are stored as a mutable mapping + dict[str, object], model_response.hidden_params.setdefault("additional_headers", {}) ) + additional_headers["llm_provider-x-litellm-response-cost"] = float(cost) cost_details: Final = usage_data.get("cost_details", {}) if cost_details: - if "response_cost_details" not in model_response._hidden_params: - model_response._hidden_params["response_cost_details"] = {} - model_response._hidden_params["response_cost_details"].update(cost_details) + response_cost_details: Final = cast( # cast-ok: provider cost details are stored as a mutable mapping + dict[str, object], model_response.hidden_params.setdefault("response_cost_details", {}) + ) + response_cost_details.update(cost_details) - model_response._hidden_params["model"] = response_json.get("model", model) + model_response.hidden_params["model"] = response_json.get("model", model) def _read_image_bytes(self, image: FileTypes) -> bytes: """Read raw bytes from various image input types.""" diff --git a/litellm/llms/openrouter/image_generation/transformation.py b/litellm/llms/openrouter/image_generation/transformation.py index cfca853e280..355df81b87a 100644 --- a/litellm/llms/openrouter/image_generation/transformation.py +++ b/litellm/llms/openrouter/image_generation/transformation.py @@ -28,12 +28,13 @@ Response format: """ from collections.abc import Iterable, Mapping -from typing import TYPE_CHECKING, Any, Final +from typing import TYPE_CHECKING, Any, Final, cast import httpx from pydantic import ConfigDict, TypeAdapter import litellm +from litellm.litellm_core_utils.hidden_params import HIDDEN_PARAMS_ATTR, set_hidden_params from litellm.llms.base_llm.chat.transformation import BaseLLMException from litellm.llms.base_llm.image_generation.transformation import ( BaseImageGenerationConfig, @@ -226,21 +227,31 @@ class OpenRouterImageGenerationConfig(BaseImageGenerationConfig): cost: Final = usage_data.get("cost") if cost is not None: - if not hasattr(model_response, "_hidden_params"): - model_response._hidden_params = {} - if "additional_headers" not in model_response._hidden_params: - model_response._hidden_params["additional_headers"] = {} - model_response._hidden_params["additional_headers"]["llm_provider-x-litellm-response-cost"] = float( - cost + if not hasattr(model_response, HIDDEN_PARAMS_ATTR): + set_hidden_params(model_response, {}) + hidden_params: Final = cast( # cast-ok: preserve mapping operations on dynamic response metadata + dict[str, object], getattr(model_response, HIDDEN_PARAMS_ATTR) ) + if "additional_headers" not in hidden_params: + hidden_params["additional_headers"] = {} + additional_headers: Final = cast( # cast-ok: preserve mapping operations on response metadata + dict[str, object], hidden_params["additional_headers"] + ) + additional_headers["llm_provider-x-litellm-response-cost"] = float(cost) cost_details: Final = usage_data.get("cost_details", {}) if cost_details: - if "response_cost_details" not in model_response._hidden_params: - model_response._hidden_params["response_cost_details"] = {} - model_response._hidden_params["response_cost_details"].update(cost_details) + cost_details_hidden_params: Final = cast( # cast-ok: preserve dynamic mapping behavior + dict[str, object], getattr(model_response, HIDDEN_PARAMS_ATTR) + ) + if "response_cost_details" not in cost_details_hidden_params: + cost_details_hidden_params["response_cost_details"] = {} + response_cost_details: Final = cast( # cast-ok: preserve mapping operations on response metadata + dict[str, object], cost_details_hidden_params["response_cost_details"] + ) + response_cost_details.update(cost_details) - model_response._hidden_params["model"] = response_json.get("model", model) + model_response.hidden_params["model"] = response_json.get("model", model) def get_complete_url( self, diff --git a/litellm/llms/opensandbox/sandbox/transformation.py b/litellm/llms/opensandbox/sandbox/transformation.py index 5126db9bbc9..d2c558241f0 100644 --- a/litellm/llms/opensandbox/sandbox/transformation.py +++ b/litellm/llms/opensandbox/sandbox/transformation.py @@ -1,7 +1,7 @@ import asyncio import json import time -from typing import Final +from typing import Final, cast import httpx @@ -18,6 +18,7 @@ from litellm.constants import ( OPEN_SANDBOX_POLL_INTERVAL, OPEN_SANDBOX_READY_TIMEOUT, ) +from litellm.litellm_core_utils.hidden_params import HIDDEN_PARAMS_ATTR, set_hidden_params from litellm.llms.base_llm.sandbox.transformation import ( SANDBOX_MAX_OUTPUT_BYTES, BaseSandboxConfig, @@ -115,7 +116,7 @@ class OpenSandboxSandboxConfig(BaseSandboxConfig): ) handle: Final = ContainerHandle(id=sandbox_id, provider="opensandbox", domain=base) - handle._hidden_params = { + handle.hidden_params = { "api_base": base, "api_key": key, "execd_endpoint": endpoint, @@ -147,9 +148,12 @@ class OpenSandboxSandboxConfig(BaseSandboxConfig): poll_interval=(float(poll_interval) if poll_interval is not None else DEFAULT_POLL_INTERVAL), client=client, ) - endpoint: Final = str(handle._hidden_params["execd_endpoint"]) - endpoint_headers: Final = self._as_str_dict(handle._hidden_params.get("execd_headers")) - base: Final = str(handle._hidden_params.get("api_base") or handle.domain or self._api_base(api_base)) + hidden_params: Final = cast( # cast-ok: preserve mapping operations on the validated handle + dict[str, object], getattr(handle, HIDDEN_PARAMS_ATTR) + ) + endpoint: Final = str(hidden_params["execd_endpoint"]) + endpoint_headers: Final = self._as_str_dict(hidden_params.get("execd_headers")) + base: Final = str(hidden_params.get("api_base") or handle.domain or self._api_base(api_base)) lines: Final = await self._post_code( url=f"{self._endpoint_base_url(endpoint, base)}/code", headers={ @@ -176,7 +180,10 @@ class OpenSandboxSandboxConfig(BaseSandboxConfig): **kwargs, ) -> bool: handle: Final = self._as_handle(container, api_base=api_base) - base: Final = str(handle._hidden_params.get("api_base") or self._api_base(api_base)) + hidden_params: Final = cast( # cast-ok: preserve mapping operations on the validated handle + dict[str, object], getattr(handle, HIDDEN_PARAMS_ATTR) + ) + base: Final = str(hidden_params.get("api_base") or self._api_base(api_base)) key: Final = self._api_key(api_key=api_key, handle=handle) try: response: Final = await self._http(client).delete( @@ -201,12 +208,15 @@ class OpenSandboxSandboxConfig(BaseSandboxConfig): client: AsyncHTTPHandler | None, ) -> ContainerHandle: handle: Final = self._as_handle(container, api_base=api_base) - if handle._hidden_params.get("execd_endpoint"): + hidden_params: Final = cast( # cast-ok: preserve mapping operations on the validated handle + dict[str, object], getattr(handle, HIDDEN_PARAMS_ATTR) + ) + if hidden_params.get("execd_endpoint"): return handle - base: Final = str(handle._hidden_params.get("api_base") or self._api_base(api_base)) + base: Final = str(hidden_params.get("api_base") or self._api_base(api_base)) key: Final = self._api_key(api_key=api_key, handle=handle) - resolved_use_server_proxy: Final = bool(handle._hidden_params.get("use_server_proxy", use_server_proxy)) + resolved_use_server_proxy: Final = bool(hidden_params.get("use_server_proxy", use_server_proxy)) endpoint, endpoint_headers = await self._wait_for_execd_endpoint( sandbox_id=handle.id, api_base=base, @@ -217,14 +227,17 @@ class OpenSandboxSandboxConfig(BaseSandboxConfig): poll_interval=poll_interval, ) handle.domain = base - handle._hidden_params = { - **handle._hidden_params, - "api_base": base, - "api_key": key, - "execd_endpoint": endpoint, - "execd_headers": endpoint_headers, - "use_server_proxy": resolved_use_server_proxy, - } + set_hidden_params( + handle, + { + **hidden_params, + "api_base": base, + "api_key": key, + "execd_endpoint": endpoint, + "execd_headers": endpoint_headers, + "use_server_proxy": resolved_use_server_proxy, + }, + ) return handle async def _wait_until_running( @@ -329,8 +342,11 @@ class OpenSandboxSandboxConfig(BaseSandboxConfig): def _api_key(self, *, api_key: str | None, handle: ContainerHandle) -> str: if api_key is not None: return api_key - if "api_key" in handle._hidden_params: - return str(handle._hidden_params["api_key"]) + hidden_params: Final = cast( # cast-ok: preserve mapping operations on the validated handle + dict[str, object], getattr(handle, HIDDEN_PARAMS_ATTR) + ) + if "api_key" in hidden_params: + return str(hidden_params["api_key"]) return self.validate_environment() @staticmethod @@ -421,7 +437,7 @@ class OpenSandboxSandboxConfig(BaseSandboxConfig): provider="opensandbox", domain=OpenSandboxSandboxConfig._api_base(api_base), ) - handle._hidden_params = {} + handle.hidden_params = {} return handle @staticmethod diff --git a/litellm/llms/ovhcloud/audio_transcription/transformation.py b/litellm/llms/ovhcloud/audio_transcription/transformation.py index 086c1afbcd4..3ae5f6fa69c 100644 --- a/litellm/llms/ovhcloud/audio_transcription/transformation.py +++ b/litellm/llms/ovhcloud/audio_transcription/transformation.py @@ -163,5 +163,5 @@ class OVHCloudAudioTranscriptionConfig(BaseAudioTranscriptionConfig): if duration is not None: response_json["duration"] = duration - response._hidden_params = response_json + response.hidden_params = response_json return response diff --git a/litellm/llms/predibase/chat/transformation.py b/litellm/llms/predibase/chat/transformation.py index 69924396a1e..bb9b808df0a 100644 --- a/litellm/llms/predibase/chat/transformation.py +++ b/litellm/llms/predibase/chat/transformation.py @@ -260,7 +260,7 @@ class PredibaseConfig(BaseConfig): if k.startswith("x-"): response_headers[f"llm_provider-{k}"] = v - model_response._hidden_params["additional_headers"] = response_headers + model_response.hidden_params["additional_headers"] = response_headers return model_response diff --git a/litellm/llms/scaleway/audio_transcription/transformation.py b/litellm/llms/scaleway/audio_transcription/transformation.py index 525a43ad86d..7ebf4475355 100644 --- a/litellm/llms/scaleway/audio_transcription/transformation.py +++ b/litellm/llms/scaleway/audio_transcription/transformation.py @@ -147,5 +147,5 @@ class ScalewayAudioTranscriptionConfig(BaseAudioTranscriptionConfig): if "language" in response_json: response["language"] = response_json["language"] - response._hidden_params = response_json + response.hidden_params = response_json return response diff --git a/litellm/llms/snowflake/chat/transformation.py b/litellm/llms/snowflake/chat/transformation.py index cc51e1162e3..a6c060732c5 100644 --- a/litellm/llms/snowflake/chat/transformation.py +++ b/litellm/llms/snowflake/chat/transformation.py @@ -558,7 +558,7 @@ class SnowflakeConfig(SnowflakeBaseConfig, OpenAIGPTConfig): returned_response.model = "snowflake/" + (returned_response.model or "") if model is not None: - returned_response._hidden_params["model"] = model + returned_response.hidden_params["model"] = model return returned_response @@ -623,7 +623,7 @@ class SnowflakeConfig(SnowflakeBaseConfig, OpenAIGPTConfig): model_response.id = response_json.get("id", "") if model is not None: - model_response._hidden_params["model"] = model + model_response.hidden_params["model"] = model return model_response diff --git a/litellm/llms/snowflake/embedding/transformation.py b/litellm/llms/snowflake/embedding/transformation.py index 6aa66de1db5..ec482337e52 100644 --- a/litellm/llms/snowflake/embedding/transformation.py +++ b/litellm/llms/snowflake/embedding/transformation.py @@ -62,7 +62,7 @@ class SnowflakeEmbeddingConfig(SnowflakeBaseConfig, BaseEmbeddingConfig): returned_response.model = "snowflake/" + (returned_response.model or "") if model is not None: - returned_response._hidden_params["model"] = model + returned_response.hidden_params["model"] = model return returned_response def get_error_class(self, error_message: str, status_code: int, headers: dict | httpx.Headers) -> BaseLLMException: diff --git a/litellm/llms/soniox/audio_transcription/handler.py b/litellm/llms/soniox/audio_transcription/handler.py index a335caa65c2..a9723125f17 100644 --- a/litellm/llms/soniox/audio_transcription/handler.py +++ b/litellm/llms/soniox/audio_transcription/handler.py @@ -458,7 +458,7 @@ class SonioxAudioTranscriptionHandler: self._safe_log_post_call(logging_obj, audio_file, api_key, body, payload) audio_duration_ms: Final = transcription_meta.get("audio_duration_ms") - response._hidden_params.update( + response.hidden_params.update( { "model": model, "custom_llm_provider": "soniox", @@ -692,7 +692,7 @@ class SonioxAudioTranscriptionHandler: self._safe_log_post_call(logging_obj, audio_file, api_key, body, payload) audio_duration_ms: Final = transcription_meta.get("audio_duration_ms") - response._hidden_params.update( + response.hidden_params.update( { "model": model, "custom_llm_provider": "soniox", diff --git a/litellm/llms/soniox/audio_transcription/transformation.py b/litellm/llms/soniox/audio_transcription/transformation.py index 8507ae73305..48510015f44 100644 --- a/litellm/llms/soniox/audio_transcription/transformation.py +++ b/litellm/llms/soniox/audio_transcription/transformation.py @@ -260,7 +260,7 @@ class SonioxAudioTranscriptionConfig(BaseAudioTranscriptionConfig): # Stash the raw Soniox payload so power-users can read tokens, segments, # speaker/language data, etc. - response._hidden_params.update( + response.hidden_params.update( { "soniox_raw": { "transcription": transcription_meta, diff --git a/litellm/llms/stability/image_edit/transformations.py b/litellm/llms/stability/image_edit/transformations.py index 94711d21b50..5d5fad6128e 100644 --- a/litellm/llms/stability/image_edit/transformations.py +++ b/litellm/llms/stability/image_edit/transformations.py @@ -6,7 +6,7 @@ Handles transformation between OpenAI-compatible format and Stability AI API for API Reference: https://platform.stability.ai/docs/api-reference """ -from typing import TYPE_CHECKING, Any, Final +from typing import TYPE_CHECKING, Any, Final, cast import httpx from httpx._types import RequestFiles @@ -300,16 +300,15 @@ class StabilityImageEditConfig(BaseImageEditConfig): ) if not hasattr(model_response, "_hidden_params"): - model_response._hidden_params = {} - if "additional_headers" not in model_response._hidden_params: - model_response._hidden_params["additional_headers"] = {} + model_response.hidden_params = {} + additional_headers: Final = cast( # cast-ok: provider headers are stored as a mutable mapping + dict[str, object], model_response.hidden_params.setdefault("additional_headers", {}) + ) # Override: fetch model-cost from model_cost map based on the provided model name model_info: Final = get_model_info(model, custom_llm_provider="stability") cost_per_image: Final = model_info.get("output_cost_per_image", 0) if cost_per_image is not None: - model_response._hidden_params["additional_headers"]["llm_provider-x-litellm-response-cost"] = float( - cost_per_image - ) + additional_headers["llm_provider-x-litellm-response-cost"] = float(cost_per_image) return model_response def use_multipart_form_data(self) -> bool: diff --git a/litellm/llms/vertex_ai/audio_transcription/transformation.py b/litellm/llms/vertex_ai/audio_transcription/transformation.py index b1284e15def..5603be81095 100644 --- a/litellm/llms/vertex_ai/audio_transcription/transformation.py +++ b/litellm/llms/vertex_ai/audio_transcription/transformation.py @@ -187,7 +187,7 @@ class VertexAIAudioTranscriptionConfig(BaseAudioTranscriptionConfig, VertexBase) billed_duration = _parse_duration_seconds(parsed.metadata.totalBilledDuration if parsed.metadata else None) if billed_duration is not None: response["duration"] = billed_duration - response._hidden_params = response_json + response.hidden_params = response_json return response diff --git a/litellm/llms/vertex_ai/gemini/vertex_and_google_ai_studio_gemini.py b/litellm/llms/vertex_ai/gemini/vertex_and_google_ai_studio_gemini.py index b8a70c63a4c..1cf1d9e1b49 100644 --- a/litellm/llms/vertex_ai/gemini/vertex_and_google_ai_studio_gemini.py +++ b/litellm/llms/vertex_ai/gemini/vertex_and_google_ai_studio_gemini.py @@ -2100,10 +2100,10 @@ class VertexGeminiConfig(VertexAIBaseConfig, BaseConfig): ) -> None: setattr(model_response, "vertex_ai_grounding_metadata", grounding_metadata) if grounding_metadata: - model_response._hidden_params["vertex_ai_grounding_metadata"] = grounding_metadata + model_response.hidden_params["vertex_ai_grounding_metadata"] = grounding_metadata setattr(model_response, "vertex_ai_url_context_metadata", url_context_metadata) if url_context_metadata: - model_response._hidden_params["vertex_ai_url_context_metadata"] = url_context_metadata + model_response.hidden_params["vertex_ai_url_context_metadata"] = url_context_metadata setattr(model_response, "vertex_ai_safety_ratings", safety_ratings) setattr(model_response, "vertex_ai_safety_results", safety_ratings) if safety_ratings: @@ -2130,7 +2130,7 @@ class VertexGeminiConfig(VertexAIBaseConfig, BaseConfig): merged.append(value) if merged: setattr(response, field_name, merged) - response._hidden_params[field_name] = merged + response.hidden_params[field_name] = merged @staticmethod def _convert_grounding_metadata_to_annotations( @@ -2472,27 +2472,27 @@ class VertexGeminiConfig(VertexAIBaseConfig, BaseConfig): ## ADD METADATA TO RESPONSE ## setattr(model_response, "vertex_ai_grounding_metadata", grounding_metadata) - model_response._hidden_params["vertex_ai_grounding_metadata"] = grounding_metadata + model_response.hidden_params["vertex_ai_grounding_metadata"] = grounding_metadata setattr(model_response, "vertex_ai_url_context_metadata", url_context_metadata) - model_response._hidden_params["vertex_ai_url_context_metadata"] = url_context_metadata + model_response.hidden_params["vertex_ai_url_context_metadata"] = url_context_metadata setattr(model_response, "vertex_ai_safety_results", safety_ratings) - model_response._hidden_params["vertex_ai_safety_results"] = ( + model_response.hidden_params["vertex_ai_safety_results"] = ( safety_ratings # older approach - maintaining to prevent regressions ) ## ADD CITATION METADATA ## setattr(model_response, "vertex_ai_citation_metadata", citation_metadata) - model_response._hidden_params["vertex_ai_citation_metadata"] = ( + model_response.hidden_params["vertex_ai_citation_metadata"] = ( citation_metadata # older approach - maintaining to prevent regressions ) ## ADD TRAFFIC TYPE ## traffic_type: Final = completion_response.get("usageMetadata", {}).get("trafficType") if traffic_type: - model_response._hidden_params.setdefault("provider_specific_fields", {})["traffic_type"] = traffic_type + model_response.hidden_params.setdefault("provider_specific_fields", {})["traffic_type"] = traffic_type ## ADD SERVICE TIER ## if getattr(raw_response, "headers", None): @@ -3223,7 +3223,7 @@ class ModelResponseIterator: traffic_type: Final = processed_chunk.get("usageMetadata", {}).get("trafficType") if traffic_type: - model_response._hidden_params.setdefault("provider_specific_fields", {})["traffic_type"] = traffic_type + model_response.hidden_params.setdefault("provider_specific_fields", {})["traffic_type"] = traffic_type service_tier: Final = self.response_headers.get("x-gemini-service-tier") if service_tier: @@ -3280,7 +3280,7 @@ class ModelResponseIterator: setattr(model_response, "usage", usage) - model_response._hidden_params["is_finished"] = False + model_response.hidden_params["is_finished"] = False return model_response except json.JSONDecodeError: diff --git a/litellm/llms/volcengine/responses/transformation.py b/litellm/llms/volcengine/responses/transformation.py index 89a54e67c0d..6e0aa88b6b6 100644 --- a/litellm/llms/volcengine/responses/transformation.py +++ b/litellm/llms/volcengine/responses/transformation.py @@ -259,8 +259,8 @@ class VolcEngineResponsesAPIConfig(OpenAIResponsesAPIConfig): construct_response: Final[Callable[..., ResponsesAPIResponse]] = ResponsesAPIResponse.model_construct response = construct_response(**raw_response_json) - response._hidden_params["additional_headers"] = processed_headers - response._hidden_params["headers"] = raw_response_headers + response.hidden_params["additional_headers"] = processed_headers + response.hidden_params["headers"] = raw_response_headers return response ######################################################### @@ -325,8 +325,8 @@ class VolcEngineResponsesAPIConfig(OpenAIResponsesAPIConfig): processed_headers: Final = process_response_headers(raw_response_headers) response: Final = ResponsesAPIResponse.model_validate(raw_response_json) - response._hidden_params["additional_headers"] = processed_headers - response._hidden_params["headers"] = raw_response_headers + response.hidden_params["additional_headers"] = processed_headers + response.hidden_params["headers"] = raw_response_headers return response ######################################################### @@ -398,8 +398,8 @@ class VolcEngineResponsesAPIConfig(OpenAIResponsesAPIConfig): processed_headers: Final = process_response_headers(raw_response_headers) response: Final = ResponsesAPIResponse.model_validate(raw_response_json) - response._hidden_params["additional_headers"] = processed_headers - response._hidden_params["headers"] = raw_response_headers + response.hidden_params["additional_headers"] = processed_headers + response.hidden_params["headers"] = raw_response_headers return response def should_fake_stream( diff --git a/litellm/main.py b/litellm/main.py index 4f95ce33191..3736051459d 100644 --- a/litellm/main.py +++ b/litellm/main.py @@ -96,6 +96,7 @@ from litellm.litellm_core_utils.health_check_utils import ( _filter_model_params, create_health_check_response, ) +from litellm.litellm_core_utils.hidden_params import HIDDEN_PARAMS_ATTR, get_hidden_params from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj from litellm.litellm_core_utils.mock_functions import ( mock_embedding, @@ -1036,11 +1037,11 @@ def mock_completion( ) if custom_llm_provider is not None: - model_response._hidden_params["custom_llm_provider"] = custom_llm_provider + model_response.hidden_params["custom_llm_provider"] = custom_llm_provider else: try: _, inferred_provider, _, _ = litellm.utils.get_llm_provider(model=model) - model_response._hidden_params["custom_llm_provider"] = inferred_provider + model_response.hidden_params["custom_llm_provider"] = inferred_provider except Exception: # dont let setting a hidden param block a mock_respose pass @@ -5516,9 +5517,12 @@ def completion( ) ) - if model_response is not None and hasattr(model_response, "_hidden_params"): - model_response._hidden_params["custom_llm_provider"] = custom_llm_provider - model_response._hidden_params["region_name"] = kwargs.get( + if model_response is not None and hasattr(model_response, HIDDEN_PARAMS_ATTR): + model_response_hidden_params: Final = cast( # cast-ok: preserve dynamic mapping behavior + dict[str, object], getattr(model_response, HIDDEN_PARAMS_ATTR) + ) + model_response_hidden_params["custom_llm_provider"] = custom_llm_provider + model_response_hidden_params["region_name"] = kwargs.get( "aws_region_name", None ) # support region-based pricing for bedrock @@ -6247,7 +6251,9 @@ async def aembedding(*args, **kwargs) -> EmbeddingResponse: elif asyncio.iscoroutine(init_response): response = await init_response if response is not None and isinstance(response, EmbeddingResponse) and hasattr(response, "_hidden_params"): - response._hidden_params["custom_llm_provider"] = custom_llm_provider + response_hidden_params: Final = get_hidden_params(response) + if response_hidden_params is not None: + response_hidden_params["custom_llm_provider"] = custom_llm_provider if response is None: raise ValueError("Unable to get Embedding Response. Please pass a valid llm_provider.") @@ -7364,7 +7370,9 @@ def embedding( else: raise LiteLLMUnknownProvider(model=model, custom_llm_provider=custom_llm_provider) if response is not None and hasattr(response, "_hidden_params") and isinstance(response, EmbeddingResponse): - response._hidden_params["custom_llm_provider"] = custom_llm_provider + response_hidden_params: Final = get_hidden_params(response) + if response_hidden_params is not None: + response_hidden_params["custom_llm_provider"] = custom_llm_provider if response is None: raise LiteLLMUnknownProvider(model=model, custom_llm_provider=custom_llm_provider) @@ -7929,7 +7937,7 @@ async def atranscription(*args, **kwargs) -> TranscriptionResponse: if existing_duration is None: calculated_duration: Final = calculate_request_duration(file) if calculated_duration is not None: - response._hidden_params["audio_transcription_duration"] = calculated_duration + response.hidden_params["audio_transcription_duration"] = calculated_duration return response except Exception as e: @@ -8202,7 +8210,10 @@ def transcription( if existing_duration is None: calculated_duration: Final = calculate_request_duration(file) if calculated_duration is not None: - response._hidden_params["audio_transcription_duration"] = calculated_duration + response_hidden_params: Final = cast( # cast-ok: preserve dynamic mapping behavior + dict[str, object], getattr(response, HIDDEN_PARAMS_ATTR) + ) + response_hidden_params["audio_transcription_duration"] = calculated_duration if response is None: raise ValueError("Unmapped provider passed in. Unable to get the response.") @@ -8816,8 +8827,9 @@ async def ahealth_check( if mode in mode_handlers: _response: Final = await mode_handlers[mode]() - # Only process headers for chat mode - _response_headers: Final[dict] = getattr(_response, "_hidden_params", {}).get("headers", {}) or {} + _response_headers: Final = cast( # cast-ok: provider headers are stored as a string-keyed mapping + Mapping[str, object], (get_hidden_params(_response) or {}).get("headers", {}) or {} + ) return create_health_check_response(_response_headers) else: raise Exception(f"Mode {mode} not supported. See modes here: https://docs.litellm.ai/docs/proxy/health") @@ -9076,9 +9088,9 @@ def stream_chunk_builder( if isinstance(chunk, dict): hidden = chunk.get("_hidden_params") else: - hidden = getattr(chunk, "_hidden_params", None) + hidden = getattr(chunk, HIDDEN_PARAMS_ATTR, None) if isinstance(hidden, dict) and "provider_specific_fields" in hidden: - response._hidden_params.setdefault("provider_specific_fields", {}).update( + response.hidden_params.setdefault("provider_specific_fields", {}).update( hidden["provider_specific_fields"] ) break @@ -9255,9 +9267,9 @@ def stream_chunk_builder( if isinstance(chunk, dict): hidden = chunk.get("_hidden_params") else: - hidden = getattr(chunk, "_hidden_params", None) + hidden = getattr(chunk, HIDDEN_PARAMS_ATTR, None) if isinstance(hidden, dict) and "provider_specific_fields" in hidden: - response._hidden_params.setdefault("provider_specific_fields", {}).update( + response.hidden_params.setdefault("provider_specific_fields", {}).update( hidden["provider_specific_fields"] ) break diff --git a/litellm/proxy/batches_endpoints/endpoints.py b/litellm/proxy/batches_endpoints/endpoints.py index 17565a420d9..2972b5043da 100644 --- a/litellm/proxy/batches_endpoints/endpoints.py +++ b/litellm/proxy/batches_endpoints/endpoints.py @@ -17,6 +17,7 @@ from pydantic import TypeAdapter import litellm from litellm._logging import verbose_proxy_logger from litellm.batches.main import CancelBatchRequest, RetrieveBatchRequest +from litellm.litellm_core_utils.hidden_params import get_hidden_params, set_hidden_param from litellm.proxy._types import * from litellm.proxy.auth.user_api_key_auth import user_api_key_auth from litellm.proxy.batches_endpoints.common_utils import validate_batch_list_limit @@ -79,6 +80,11 @@ router: Final = APIRouter() _METADATA_ADAPTER: Final[TypeAdapter[Mapping[str, object]]] = TypeAdapter(Mapping[str, object]) +def _hidden_param_string(hidden_params: Mapping[str, object], key: str) -> str: + value: Final = hidden_params.get(key) + return value if isinstance(value, str) else "" + + def _request_tags(data: Mapping[str, object]) -> tuple[str, ...] | None: metadata: Final = data.get("litellm_metadata") if metadata is None: @@ -211,7 +217,7 @@ async def _create_provider_batch_for_managed_file( } response: Final = await llm_router.acreate_batch(**request) response.input_file_id = input_file_id - response._hidden_params["unified_file_id"] = unified_file_id + set_hidden_param(response, "unified_file_id", unified_file_id) return response @@ -484,7 +490,7 @@ async def create_batch( **_create_batch_data, ) - response._hidden_params[BATCH_CREATE_HIDDEN_PARAM] = True + set_hidden_param(response, BATCH_CREATE_HIDDEN_PARAM, True) ### CALL HOOKS ### - modify outgoing data response = await proxy_logging_obj.post_call_success_hook( @@ -497,10 +503,10 @@ async def create_batch( ) ### RESPONSE HEADERS ### - hidden_params: Final = getattr(response, "_hidden_params", {}) or {} - model_id: Final = hidden_params.get("model_id", None) or "" - cache_key: Final = hidden_params.get("cache_key", None) or "" - api_base: Final = hidden_params.get("api_base", None) or "" + hidden_params: Final = get_hidden_params(response) or {} + model_id: Final = _hidden_param_string(hidden_params, "model_id") + cache_key: Final = _hidden_param_string(hidden_params, "cache_key") + api_base: Final = _hidden_param_string(hidden_params, "api_base") fastapi_response.headers.update( ProxyBaseLLMRequestProcessing.get_custom_headers( @@ -655,10 +661,10 @@ async def retrieve_batch( ) ) - hidden_params = getattr(response, "_hidden_params", {}) or {} - model_id = hidden_params.get("model_id", None) or "" - cache_key = hidden_params.get("cache_key", None) or "" - api_base = hidden_params.get("api_base", None) or "" + hidden_params = get_hidden_params(response) or {} + model_id = _hidden_param_string(hidden_params, "model_id") + cache_key = _hidden_param_string(hidden_params, "cache_key") + api_base = _hidden_param_string(hidden_params, "api_base") fastapi_response.headers.update( ProxyBaseLLMRequestProcessing.get_custom_headers( @@ -736,11 +742,11 @@ async def retrieve_batch( ) response = await llm_router.aretrieve_batch(**data) - response._hidden_params["unified_batch_id"] = unified_batch_id + set_hidden_param(response, "unified_batch_id", unified_batch_id) if unified_batch_id: model_id_from_batch: Final = get_model_id_from_unified_batch_id(unified_batch_id) if model_id_from_batch: - response._hidden_params["model_id"] = model_id_from_batch + set_hidden_param(response, "model_id", model_id_from_batch) # SCENARIO 3: Fallback to custom_llm_provider (uses env variables) else: @@ -802,10 +808,10 @@ async def retrieve_batch( ) ### RESPONSE HEADERS ### - hidden_params = getattr(response, "_hidden_params", {}) or {} - model_id = hidden_params.get("model_id", None) or "" - cache_key = hidden_params.get("cache_key", None) or "" - api_base = hidden_params.get("api_base", None) or "" + hidden_params = get_hidden_params(response) or {} + model_id = _hidden_param_string(hidden_params, "model_id") + cache_key = _hidden_param_string(hidden_params, "cache_key") + api_base = _hidden_param_string(hidden_params, "api_base") fastapi_response.headers.update( ProxyBaseLLMRequestProcessing.get_custom_headers( @@ -986,10 +992,10 @@ async def list_batches( response = _response ### RESPONSE HEADERS ### - hidden_params: Final = getattr(response, "_hidden_params", {}) or {} - model_id: Final = hidden_params.get("model_id", None) or "" - cache_key: Final = hidden_params.get("cache_key", None) or "" - api_base: Final = hidden_params.get("api_base", None) or "" + hidden_params: Final = get_hidden_params(response) or {} + model_id: Final = _hidden_param_string(hidden_params, "model_id") + cache_key: Final = _hidden_param_string(hidden_params, "cache_key") + api_base: Final = _hidden_param_string(hidden_params, "api_base") fastapi_response.headers.update( ProxyBaseLLMRequestProcessing.get_custom_headers( @@ -1168,10 +1174,10 @@ async def cancel_batch( data["model"] = model_id_from_batch data["batch_id"] = get_batch_id_from_unified_batch_id(unified_batch_id) response = await llm_router.acancel_batch(**data) - response._hidden_params["unified_batch_id"] = unified_batch_id + set_hidden_param(response, "unified_batch_id", unified_batch_id) - if not response._hidden_params.get("model_id") and data.get("model"): - response._hidden_params["model_id"] = data["model"] + if not (get_hidden_params(response) or {}).get("model_id") and data.get("model"): + set_hidden_param(response, "model_id", data["model"]) # SCENARIO 3: Fallback to custom_llm_provider (uses env variables) else: @@ -1229,10 +1235,10 @@ async def cancel_batch( ) ### RESPONSE HEADERS ### - hidden_params: Final = getattr(response, "_hidden_params", {}) or {} - model_id: Final = hidden_params.get("model_id", None) or "" - cache_key: Final = hidden_params.get("cache_key", None) or "" - api_base: Final = hidden_params.get("api_base", None) or "" + hidden_params: Final = get_hidden_params(response) or {} + model_id: Final = _hidden_param_string(hidden_params, "model_id") + cache_key: Final = _hidden_param_string(hidden_params, "cache_key") + api_base: Final = _hidden_param_string(hidden_params, "api_base") fastapi_response.headers.update( ProxyBaseLLMRequestProcessing.get_custom_headers( diff --git a/litellm/proxy/fine_tuning_endpoints/endpoints.py b/litellm/proxy/fine_tuning_endpoints/endpoints.py index e1fe7c21754..886b8da1454 100644 --- a/litellm/proxy/fine_tuning_endpoints/endpoints.py +++ b/litellm/proxy/fine_tuning_endpoints/endpoints.py @@ -12,6 +12,7 @@ from fastapi import APIRouter, Depends, HTTPException, Query, Request, Response import litellm from litellm._logging import verbose_proxy_logger +from litellm.litellm_core_utils.hidden_params import set_hidden_param from litellm.proxy._types import * from litellm.proxy.auth.user_api_key_auth import user_api_key_auth from litellm.proxy.common_request_processing import ProxyBaseLLMRequestProcessing @@ -161,7 +162,7 @@ async def create_fine_tuning_job( response = cast(LiteLLMFineTuningJob, await llm_router.acreate_fine_tuning_job(**data)) response.training_file = unified_file_id - response._hidden_params["unified_file_id"] = unified_file_id + set_hidden_param(response, "unified_file_id", unified_file_id) ## ELSE, Route based on custom_llm_provider elif fine_tuning_request.custom_llm_provider: # get configs for custom_llm_provider @@ -304,7 +305,7 @@ async def retrieve_fine_tuning_job( **data, ), ) - response._hidden_params["unified_finetuning_job_id"] = unified_finetuning_job_id + set_hidden_param(response, "unified_finetuning_job_id", unified_finetuning_job_id) elif custom_llm_provider: # get configs for custom_llm_provider llm_provider_config: Final = get_fine_tuning_provider_config(custom_llm_provider=custom_llm_provider) @@ -577,7 +578,7 @@ async def cancel_fine_tuning_job( **data, ), ) - response._hidden_params["unified_finetuning_job_id"] = unified_finetuning_job_id + set_hidden_param(response, "unified_finetuning_job_id", unified_finetuning_job_id) else: # get configs for custom_llm_provider llm_provider_config: Final = get_fine_tuning_provider_config(custom_llm_provider=custom_llm_provider) diff --git a/litellm/proxy/hooks/dynamic_rate_limiter.py b/litellm/proxy/hooks/dynamic_rate_limiter.py index 8c41eb8d2d3..5c99a1bacc1 100644 --- a/litellm/proxy/hooks/dynamic_rate_limiter.py +++ b/litellm/proxy/hooks/dynamic_rate_limiter.py @@ -15,6 +15,7 @@ from litellm._logging import verbose_proxy_logger from litellm.caching.caching import DualCache from litellm.exceptions import RateLimitType from litellm.integrations.custom_logger import CustomLogger +from litellm.litellm_core_utils.hidden_params import set_hidden_param from litellm.proxy._types import UserAPIKeyAuth from litellm.proxy.common_utils.proxy_rate_limit_error import ProxyRateLimitError from litellm.proxy.hooks.rate_limiter_utils import ( @@ -243,10 +244,11 @@ class _PROXY_DynamicRateLimitHandler(CustomLogger): async def async_post_call_success_hook(self, data: dict, user_api_key_dict: UserAPIKeyAuth, response): try: if isinstance(response, ModelResponse): - model_info: Final = self.llm_router.get_model_info(id=response._hidden_params["model_id"]) - assert model_info is not None, "Model info for model with id={} is None".format( - response._hidden_params["model_id"] - ) + model_id: Final = response.hidden_params["model_id"] + if not isinstance(model_id, str): + return response + model_info: Final = self.llm_router.get_model_info(id=model_id) + assert model_info is not None, f"Model info for model with id={model_id} is None" key_priority: Final[str | None] = user_api_key_dict.metadata.get("priority", None) ( available_tpm, @@ -255,14 +257,18 @@ class _PROXY_DynamicRateLimitHandler(CustomLogger): model_rpm, active_projects, ) = await self.check_available_usage(model=model_info["model_name"], priority=key_priority) - response._hidden_params["additional_headers"] = { # Add additional response headers - easier debugging - "x-litellm-model_group": model_info["model_name"], - "x-ratelimit-remaining-litellm-project-tokens": available_tpm, - "x-ratelimit-remaining-litellm-project-requests": available_rpm, - "x-ratelimit-remaining-model-tokens": model_tpm, - "x-ratelimit-remaining-model-requests": model_rpm, - "x-ratelimit-current-active-projects": active_projects, - } + set_hidden_param( + response, + "additional_headers", + { + "x-litellm-model_group": model_info["model_name"], + "x-ratelimit-remaining-litellm-project-tokens": available_tpm, + "x-ratelimit-remaining-litellm-project-requests": available_rpm, + "x-ratelimit-remaining-model-tokens": model_tpm, + "x-ratelimit-remaining-model-requests": model_rpm, + "x-ratelimit-current-active-projects": active_projects, + }, + ) return response return await super().async_post_call_success_hook( diff --git a/litellm/proxy/openai_files_endpoints/storage_backend_service.py b/litellm/proxy/openai_files_endpoints/storage_backend_service.py index 53f48d93aa2..afc1e1c96fa 100644 --- a/litellm/proxy/openai_files_endpoints/storage_backend_service.py +++ b/litellm/proxy/openai_files_endpoints/storage_backend_service.py @@ -12,6 +12,7 @@ from typing import Any, Final, cast from litellm._logging import verbose_proxy_logger from litellm._uuid import uuid as uuid_module +from litellm.litellm_core_utils.hidden_params import get_or_create_hidden_params from litellm.llms.base_llm.files.storage_backend import BaseFileStorageBackend from litellm.llms.base_llm.files.storage_backend_factory import get_storage_backend from litellm.llms.base_llm.files.transformation import BaseFileEndpoints @@ -170,9 +171,8 @@ class StorageBackendFileService: ) # Store storage metadata in hidden params - if not hasattr(file_object, "_hidden_params") or file_object._hidden_params is None: - file_object._hidden_params = {} - file_object._hidden_params.update( + file_object_hidden_params: Final = get_or_create_hidden_params(file_object) + file_object_hidden_params.update( { "storage_backend": target_storage, "storage_url": storage_url, diff --git a/litellm/proxy/pass_through_endpoints/llm_provider_handlers/cohere_passthrough_logging_handler.py b/litellm/proxy/pass_through_endpoints/llm_provider_handlers/cohere_passthrough_logging_handler.py index 597c2e742b3..8bba698ec71 100644 --- a/litellm/proxy/pass_through_endpoints/llm_provider_handlers/cohere_passthrough_logging_handler.py +++ b/litellm/proxy/pass_through_endpoints/llm_provider_handlers/cohere_passthrough_logging_handler.py @@ -5,6 +5,7 @@ import httpx import litellm from litellm import stream_chunk_builder +from litellm.litellm_core_utils.hidden_params import set_hidden_param from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj from litellm.litellm_core_utils.litellm_logging import ( get_standard_logging_object_payload, @@ -114,9 +115,7 @@ class CoherePassthroughLoggingHandler(BasePassthroughLoggingHandler): ) # Set the calculated cost in _hidden_params to prevent recalculation - if not hasattr(litellm_model_response, "_hidden_params"): - litellm_model_response._hidden_params = {} - litellm_model_response._hidden_params["response_cost"] = response_cost + set_hidden_param(litellm_model_response, "response_cost", response_cost) kwargs["response_cost"] = response_cost kwargs["model"] = model diff --git a/litellm/proxy/pass_through_endpoints/llm_provider_handlers/gemini_passthrough_logging_handler.py b/litellm/proxy/pass_through_endpoints/llm_provider_handlers/gemini_passthrough_logging_handler.py index 9efa371ed7f..49679fe838b 100644 --- a/litellm/proxy/pass_through_endpoints/llm_provider_handlers/gemini_passthrough_logging_handler.py +++ b/litellm/proxy/pass_through_endpoints/llm_provider_handlers/gemini_passthrough_logging_handler.py @@ -81,8 +81,8 @@ class GeminiPassthroughLoggingHandler: # Set response_cost in _hidden_params to prevent recalculation if not hasattr(litellm_video_response, "_hidden_params"): - litellm_video_response._hidden_params = {} - litellm_video_response._hidden_params["response_cost"] = response_cost + litellm_video_response.hidden_params = {} + litellm_video_response.hidden_params["response_cost"] = response_cost kwargs["response_cost"] = response_cost kwargs["model"] = model diff --git a/litellm/proxy/pass_through_endpoints/llm_provider_handlers/openai_passthrough_logging_handler.py b/litellm/proxy/pass_through_endpoints/llm_provider_handlers/openai_passthrough_logging_handler.py index 6aef278963d..89c21d092ff 100644 --- a/litellm/proxy/pass_through_endpoints/llm_provider_handlers/openai_passthrough_logging_handler.py +++ b/litellm/proxy/pass_through_endpoints/llm_provider_handlers/openai_passthrough_logging_handler.py @@ -13,6 +13,7 @@ import httpx import litellm from litellm._logging import verbose_proxy_logger +from litellm.litellm_core_utils.hidden_params import set_hidden_param from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj from litellm.litellm_core_utils.litellm_logging import ( get_standard_logging_object_payload, @@ -414,7 +415,7 @@ class OpenAIPassthroughLoggingHandler(BasePassthroughLoggingHandler): model=model, custom_llm_provider=custom_llm_provider, ) - litellm_model_response._hidden_params["response_cost"] = response_cost + set_hidden_param(litellm_model_response, "response_cost", response_cost) elif is_image_generation: # Handle image generation cost calculation response_cost = OpenAIPassthroughLoggingHandler._calculate_image_generation_cost( @@ -433,9 +434,7 @@ class OpenAIPassthroughLoggingHandler(BasePassthroughLoggingHandler): model=model, ) # Set the calculated cost in _hidden_params to prevent recalculation - if not hasattr(litellm_model_response, "_hidden_params"): - litellm_model_response._hidden_params = {} - litellm_model_response._hidden_params["response_cost"] = response_cost + set_hidden_param(litellm_model_response, "response_cost", response_cost) elif is_image_editing: # Handle image editing cost calculation response_cost = OpenAIPassthroughLoggingHandler._calculate_image_editing_cost( @@ -454,9 +453,7 @@ class OpenAIPassthroughLoggingHandler(BasePassthroughLoggingHandler): model=model, ) # Set the calculated cost in _hidden_params to prevent recalculation - if not hasattr(litellm_model_response, "_hidden_params"): - litellm_model_response._hidden_params = {} - litellm_model_response._hidden_params["response_cost"] = response_cost + set_hidden_param(litellm_model_response, "response_cost", response_cost) elif is_responses: # Responses-API cost tracking — see # `_build_responses_api_response_and_cost` for why this needs diff --git a/litellm/proxy/pass_through_endpoints/llm_provider_handlers/vertex_passthrough_logging_handler.py b/litellm/proxy/pass_through_endpoints/llm_provider_handlers/vertex_passthrough_logging_handler.py index 9f0ddb09953..e749c2b992e 100644 --- a/litellm/proxy/pass_through_endpoints/llm_provider_handlers/vertex_passthrough_logging_handler.py +++ b/litellm/proxy/pass_through_endpoints/llm_provider_handlers/vertex_passthrough_logging_handler.py @@ -175,8 +175,8 @@ class VertexPassthroughLoggingHandler: # Set response_cost in _hidden_params to prevent recalculation if not hasattr(litellm_video_response, "_hidden_params"): - litellm_video_response._hidden_params = {} - litellm_video_response._hidden_params["response_cost"] = response_cost + litellm_video_response.hidden_params = {} + litellm_video_response.hidden_params["response_cost"] = response_cost kwargs["response_cost"] = response_cost kwargs["model"] = model diff --git a/litellm/responses/file_search/emulated_handler.py b/litellm/responses/file_search/emulated_handler.py index 887d1a9ff93..e2bb2ae40ea 100644 --- a/litellm/responses/file_search/emulated_handler.py +++ b/litellm/responses/file_search/emulated_handler.py @@ -380,7 +380,7 @@ def _synthesize_responses_api_response( if first_cost is not None: current_cost: Final = hidden.get("response_cost") if isinstance(hidden, dict) else 0 hidden["response_cost"] = (current_cost or 0) + first_cost - synthesized._hidden_params = hidden + synthesized.hidden_params = hidden return synthesized diff --git a/litellm/responses/litellm_completion_transformation/streaming_iterator.py b/litellm/responses/litellm_completion_transformation/streaming_iterator.py index 341324b0c74..eccca70fde1 100644 --- a/litellm/responses/litellm_completion_transformation/streaming_iterator.py +++ b/litellm/responses/litellm_completion_transformation/streaming_iterator.py @@ -4,6 +4,7 @@ from collections.abc import Sequence from typing import Any, Final, cast import litellm +from litellm.litellm_core_utils.hidden_params import get_or_create_hidden_params from litellm.main import stream_chunk_builder from litellm.responses.litellm_completion_transformation.custom_tools import ( build_tool_call_item_kwargs, @@ -649,11 +650,11 @@ class LiteLLMCompletionStreamingIterator(ResponsesAPIStreamingIterator): ), ) if response is not None and self._accumulated_provider_specific_fields: - if not hasattr(response, "_hidden_params") or response._hidden_params is None: - response._hidden_params = {} - response._hidden_params.setdefault("provider_specific_fields", {}).update( - self._accumulated_provider_specific_fields + response_hidden_params: Final = get_or_create_hidden_params(response) + provider_specific_fields: Final = cast( # cast-ok: provider fields are stored as a mutable mapping + dict[str, object], response_hidden_params.setdefault("provider_specific_fields", {}) ) + provider_specific_fields.update(self._accumulated_provider_specific_fields) return response @staticmethod diff --git a/litellm/responses/litellm_completion_transformation/transformation.py b/litellm/responses/litellm_completion_transformation/transformation.py index cb3c04b6452..b1dc6b3a98a 100644 --- a/litellm/responses/litellm_completion_transformation/transformation.py +++ b/litellm/responses/litellm_completion_transformation/transformation.py @@ -39,6 +39,11 @@ from litellm.constants import REDACTED_BY_LITELLM, REDACTED_TOOL_CALL_ARGUMENTS_ from litellm.litellm_core_utils.get_supported_openai_params import ( get_supported_openai_params, ) +from litellm.litellm_core_utils.hidden_params import ( + get_hidden_params, + get_hidden_params_storage, + set_hidden_params, +) from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj from litellm.responses.litellm_completion_transformation.session_handler import ( ResponsesSessionHandler, @@ -2490,10 +2495,17 @@ class LiteLLMCompletionResponsesConfig: user=echoed.get("user"), store=echoed.get("store"), ) - responses_api_response._hidden_params = getattr(chat_completion_response, "_hidden_params", {}) + chat_completion_hidden_params: Final = get_hidden_params_storage(chat_completion_response) + set_hidden_params( + responses_api_response, + chat_completion_hidden_params if chat_completion_hidden_params is not None else {}, + ) # Surface provider-specific fields (generic passthrough from any provider) - provider_fields: Final = responses_api_response._hidden_params.get("provider_specific_fields") + response_hidden_params: Final = get_hidden_params(responses_api_response) + provider_fields: Final = ( + response_hidden_params.get("provider_specific_fields") if response_hidden_params is not None else None + ) if provider_fields: setattr(responses_api_response, "provider_specific_fields", provider_fields) diff --git a/litellm/responses/main.py b/litellm/responses/main.py index f50bdfa42ef..c4632f09723 100644 --- a/litellm/responses/main.py +++ b/litellm/responses/main.py @@ -748,7 +748,7 @@ async def aresponses( ) # Stamp custom_llm_provider so callbacks can identify the provider # (mirrors litellm/main.py:1371 for chat completions) - response._hidden_params["custom_llm_provider"] = custom_llm_provider + response.hidden_params["custom_llm_provider"] = custom_llm_provider if response is None: raise ValueError(f"Got an unexpected None response from the Responses API: {response}") @@ -1453,7 +1453,7 @@ def responses( ) # Stamp custom_llm_provider so callbacks can identify the provider # (mirrors litellm/main.py:1371 for chat completions) - response._hidden_params["custom_llm_provider"] = custom_llm_provider + response.hidden_params["custom_llm_provider"] = custom_llm_provider return response except Exception as e: diff --git a/litellm/responses/mcp/chat_completions_handler.py b/litellm/responses/mcp/chat_completions_handler.py index 3095a319cf7..5a05dfe0245 100644 --- a/litellm/responses/mcp/chat_completions_handler.py +++ b/litellm/responses/mcp/chat_completions_handler.py @@ -6,6 +6,7 @@ from typing import TYPE_CHECKING, Final, cast from typing_extensions import TypedDict, Unpack +from litellm.litellm_core_utils.hidden_params import HIDDEN_PARAMS_ATTR, set_hidden_params from litellm.responses.mcp.litellm_proxy_mcp_handler import ( LiteLLM_Proxy_MCP_Handler, ) @@ -42,8 +43,8 @@ def _add_mcp_metadata_to_response( # For streaming, store MCP metadata in _hidden_params # CustomStreamWrapper._add_mcp_metadata_to_final_chunk() will automatically # add it to the final chunk's delta.provider_specific_fields - if not hasattr(response, "_hidden_params"): - response._hidden_params = {} + if not hasattr(response, HIDDEN_PARAMS_ATTR): + set_hidden_params(response, {}) mcp_metadata: Final = {} if openai_tools: @@ -54,7 +55,10 @@ def _add_mcp_metadata_to_response( mcp_metadata["mcp_call_results"] = tool_results if mcp_metadata: - response._hidden_params["mcp_metadata"] = mcp_metadata + hidden_params: Final = cast( # cast-ok: preserve mapping operations on dynamic response metadata + dict[str, object], getattr(response, HIDDEN_PARAMS_ATTR) + ) + hidden_params["mcp_metadata"] = mcp_metadata return if not isinstance(response, ModelResponse): diff --git a/litellm/responses/streaming_iterator.py b/litellm/responses/streaming_iterator.py index fd88a4a5de4..32a4f4284be 100644 --- a/litellm/responses/streaming_iterator.py +++ b/litellm/responses/streaming_iterator.py @@ -591,7 +591,7 @@ class BaseResponsesAPIStreamingIterator: target: Final[object] = getattr(logging_response, "response", None) if not isinstance(target, ResponsesAPIResponse): return - existing: Final[Mapping[str, object]] = target._hidden_params + existing: Final[Mapping[str, object]] = target.hidden_params source_hidden: Final[object] = getattr( getattr(self.completed_response, "response", None), "_hidden_params", None ) @@ -602,7 +602,7 @@ class BaseResponsesAPIStreamingIterator: raw_headers: Final[Mapping[str, object]] = raw if isinstance(raw, Mapping) else EMPTY_MAPPING # rebuild by value and let existing keys win: sharing the source dicts would alias what the proxy # splats into the client's HTTP headers, and copying non-header keys would carry response_cost - target._hidden_params = { + target.hidden_params = { "additional_headers": {**headers}, "headers": {**raw_headers}, **existing, diff --git a/litellm/responses/utils.py b/litellm/responses/utils.py index 3e308b73d12..f4705e7132e 100644 --- a/litellm/responses/utils.py +++ b/litellm/responses/utils.py @@ -325,7 +325,7 @@ class ResponsesAPIRequestUtils: responses_api_response: dict[str, object], custom_llm_provider: str | None, litellm_metadata: dict[str, object] | None = None, - ) -> dict[str, object]: + ) -> dict[str, object]: ... # fmt: on diff --git a/litellm/router.py b/litellm/router.py index 634fe2f3734..9ac6eae9e3e 100644 --- a/litellm/router.py +++ b/litellm/router.py @@ -91,6 +91,13 @@ from litellm.litellm_core_utils.get_llm_provider_logic import ( declared_authenticating_provider, is_registered_custom_provider, ) +from litellm.litellm_core_utils.hidden_params import ( + HIDDEN_PARAMS_ATTR as _HIDDEN_PARAMS_ATTR, +) +from litellm.litellm_core_utils.hidden_params import ( + get_hidden_params, + set_hidden_params, +) from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLogging from litellm.litellm_core_utils.llm_cost_calc.utils import SERVICE_TIER_COST_KEY_SUFFIXES from litellm.litellm_core_utils.ptu_pricing import ( @@ -149,7 +156,6 @@ from litellm.router_strategy.tag_based_routing import ( ) from litellm.router_utils.access_windows import access_windows_config_error, filter_reserved_deployments from litellm.router_utils.add_retry_fallback_headers import ( - HiddenParamsHost, add_fallback_headers_to_response, add_retry_headers_to_response, apply_quality_router_decision_headers, @@ -2989,7 +2995,7 @@ class Router: fallback_item: object, prepared_fallback_hidden_params: tuple[dict[str, object], dict[str, object]], ) -> None: - if fallback_item is None or not hasattr(fallback_item, "_hidden_params"): + if fallback_item is None or not hasattr(fallback_item, _HIDDEN_PARAMS_ATTR): return fallback_hidden_params, fallback_headers = prepared_fallback_hidden_params @@ -2998,11 +3004,14 @@ class Router: if not isinstance(item_headers, dict): item_headers = {} - cast(HiddenParamsHost, fallback_item)._hidden_params = { - **item_hidden_params, - **fallback_hidden_params, - "additional_headers": {**item_headers, **fallback_headers}, - } + set_hidden_params( + fallback_item, + { + **item_hidden_params, + **fallback_hidden_params, + "additional_headers": {**item_headers, **fallback_headers}, + }, + ) async def _acompletion_streaming_iterator( self, @@ -3038,8 +3047,9 @@ class Router: if isinstance(inner_chunks, list): self.chunks = inner_chunks # Preserve hidden params (including litellm_overhead_time_ms) from original response - if hasattr(model_response, "_hidden_params"): - self._hidden_params = model_response._hidden_params.copy() + model_response_hidden_params: Final = get_hidden_params(model_response) + if model_response_hidden_params is not None: + self._hidden_params = dict(model_response_hidden_params) def __aiter__(self): return self @@ -3656,8 +3666,9 @@ class Router: _response_headers=getattr(model_response, "_response_headers", None), ) self._sync_generator = sync_generator - if hasattr(model_response, "_hidden_params"): - self._hidden_params = model_response._hidden_params.copy() + model_response_hidden_params: Final = get_hidden_params(model_response) + if model_response_hidden_params is not None: + self._hidden_params = dict(model_response_hidden_params) def __iter__(self): return self @@ -4484,7 +4495,8 @@ class Router: if result is not None: # Return the first successful result - result._hidden_params["fastest_response_batch_completion"] = True + if (result_hidden_params := get_hidden_params(result)) is not None: + result_hidden_params["fastest_response_batch_completion"] = True return result # If we exit the loop without returning, all tasks failed @@ -4553,8 +4565,12 @@ class Router: if make_request: try: _response: Final = await self.acompletion(model=model, messages=messages, stream=stream, **kwargs) - _response._hidden_params.setdefault("additional_headers", {}) - _response._hidden_params["additional_headers"].update({"x-litellm-request-prioritization-used": True}) + response_hidden_params: Final = get_hidden_params(_response) + if response_hidden_params is not None: + additional_headers: Final = cast( # cast-ok: router headers are stored as a mutable mapping + dict[str, object], response_hidden_params.setdefault("additional_headers", {}) + ) + additional_headers.update({"x-litellm-request-prioritization-used": True}) return _response except Exception as e: setattr(e, "priority", priority) @@ -4613,11 +4629,12 @@ class Router: if make_request: try: _response: Final = await original_function(*args, **kwargs) - if isinstance(_response._hidden_params, dict): - _response._hidden_params.setdefault("additional_headers", {}) - _response._hidden_params["additional_headers"].update( - {"x-litellm-request-prioritization-used": True} + response_hidden_params: Final = get_hidden_params(_response) + if response_hidden_params is not None: + additional_headers: Final = cast( # cast-ok: router headers are stored as a mutable mapping + dict[str, object], response_hidden_params.setdefault("additional_headers", {}) ) + additional_headers.update({"x-litellm-request-prioritization-used": True}) return _response except Exception as e: setattr(e, "priority", priority) @@ -6453,7 +6470,9 @@ class Router: healthy_deployments=healthy_deployments, responses=responses ) returned_response: Final = cast(OpenAIFileObject, responses[0]) - returned_response._hidden_params["model_file_id_mapping"] = model_file_id_mapping + returned_response_hidden_params: Final = get_hidden_params(returned_response) + if returned_response_hidden_params is not None: + returned_response_hidden_params["model_file_id_mapping"] = model_file_id_mapping return returned_response except Exception as e: verbose_router_logger.exception( diff --git a/litellm/router_strategy/complexity_router/complexity_router.py b/litellm/router_strategy/complexity_router/complexity_router.py index 4177771987c..acc9770ab41 100644 --- a/litellm/router_strategy/complexity_router/complexity_router.py +++ b/litellm/router_strategy/complexity_router/complexity_router.py @@ -49,6 +49,7 @@ from litellm.litellm_core_utils.core_helpers import ( get_parent_otel_span_from_kwargs, is_codex_user_agent, ) +from litellm.litellm_core_utils.hidden_params import get_hidden_params from litellm.litellm_core_utils.internal_call_metadata import forwarded_internal_call_metadata from litellm.litellm_core_utils.prompt_templates.common_utils import ( as_openai_image_part, @@ -413,9 +414,9 @@ def _parent_session_kwargs(request_kwargs: Mapping[str, object] | None) -> Mappi return {k: kwargs[k] for k in ("litellm_session_id", "litellm_trace_id") if kwargs.get(k) is not None} -def _response_cost_or_none(response: ModelResponse | ResponsesAPIResponse) -> float | None: - hidden_params: Final = response._hidden_params - if not isinstance(hidden_params, dict): +def _response_cost_or_none(response: object) -> float | None: + hidden_params: Final = get_hidden_params(response) + if hidden_params is None: return None cost: Final = hidden_params.get("response_cost") if isinstance(cost, bool) or not isinstance(cost, (int, float)): diff --git a/litellm/router_utils/add_retry_fallback_headers.py b/litellm/router_utils/add_retry_fallback_headers.py index ee5d6c1194b..12b26954e4c 100644 --- a/litellm/router_utils/add_retry_fallback_headers.py +++ b/litellm/router_utils/add_retry_fallback_headers.py @@ -6,6 +6,13 @@ from typing import Final, Protocol, TypedDict, cast from pydantic import BaseModel, TypeAdapter, ValidationError +from litellm.litellm_core_utils.hidden_params import ( + HIDDEN_PARAMS_ATTR as _HIDDEN_PARAMS_ATTR, +) +from litellm.litellm_core_utils.hidden_params import ( + set_hidden_params, +) + class FallbackErrorInfo(TypedDict): message: str @@ -227,10 +234,8 @@ def get_hidden_params_dict( def _write_hidden_params(response: object, hidden_params: dict[str, object]) -> None: - if isinstance(response, dict): - response["_hidden_params"] = hidden_params - elif hasattr(response, "_hidden_params"): - setattr(response, "_hidden_params", hidden_params) + if isinstance(response, dict) or hasattr(response, _HIDDEN_PARAMS_ATTR): + set_hidden_params(response, hidden_params) def _ensure_additional_headers_dict( diff --git a/litellm/types/agents.py b/litellm/types/agents.py index b3aa86c249d..97dd67d3ff8 100644 --- a/litellm/types/agents.py +++ b/litellm/types/agents.py @@ -360,6 +360,14 @@ class AgentCreateResponse(LiteLLMPydanticObjectBase): _hidden_params: dict[str, object] = PrivateAttr(default_factory=dict) + @property + def hidden_params(self) -> dict[str, object]: # mutable-ok: API requires mutation + return self._hidden_params + + @hidden_params.setter + def hidden_params(self, hidden_params: dict[str, object]) -> None: # mutable-ok: API requires mutation + self._hidden_params = hidden_params + class AgentDeleteResult(LiteLLMPydanticObjectBase): """Result of a provider-side agent deletion (e.g. Gemini DELETE /v1beta/agents/{name}). @@ -374,6 +382,14 @@ class AgentDeleteResult(LiteLLMPydanticObjectBase): _hidden_params: dict[str, object] = PrivateAttr(default_factory=dict) + @property + def hidden_params(self) -> dict[str, object]: # mutable-ok: API requires mutation + return self._hidden_params + + @hidden_params.setter + def hidden_params(self, hidden_params: dict[str, object]) -> None: # mutable-ok: API requires mutation + self._hidden_params = hidden_params + class AgentListResponse(LiteLLMPydanticObjectBase): """Response from listing agents on the provider side (e.g. Gemini GET /v1beta/agents). @@ -388,6 +404,14 @@ class AgentListResponse(LiteLLMPydanticObjectBase): _hidden_params: dict[str, object] = PrivateAttr(default_factory=dict) + @property + def hidden_params(self) -> dict[str, object]: # mutable-ok: API requires mutation + return self._hidden_params + + @hidden_params.setter + def hidden_params(self, hidden_params: dict[str, object]) -> None: # mutable-ok: API requires mutation + self._hidden_params = hidden_params + class AgentVersionsResponse(LiteLLMPydanticObjectBase): """Response from listing versions of an agent (e.g. Gemini GET /v1beta/agents/{name}/versions). @@ -402,6 +426,14 @@ class AgentVersionsResponse(LiteLLMPydanticObjectBase): _hidden_params: dict[str, object] = PrivateAttr(default_factory=dict) + @property + def hidden_params(self) -> dict[str, object]: # mutable-ok: API requires mutation + return self._hidden_params + + @hidden_params.setter + def hidden_params(self, hidden_params: dict[str, object]) -> None: # mutable-ok: API requires mutation + self._hidden_params = hidden_params + class AgentMakePublicResponse(LiteLLMBaseModel): message: str @@ -468,6 +500,14 @@ class LiteLLMSendMessageResponse(LiteLLMPydanticObjectBase): # LiteLLM private attributes for logging/cost tracking _hidden_params: dict[str, object] = PrivateAttr(default_factory=dict) + @property + def hidden_params(self) -> dict[str, object]: # mutable-ok: API requires mutation + return self._hidden_params + + @hidden_params.setter + def hidden_params(self, hidden_params: dict[str, object]) -> None: # mutable-ok: API requires mutation + self._hidden_params = hidden_params + @classmethod def from_a2a_response( cls, diff --git a/litellm/types/containers/main.py b/litellm/types/containers/main.py index afa07462926..24a2418a5d0 100644 --- a/litellm/types/containers/main.py +++ b/litellm/types/containers/main.py @@ -27,6 +27,14 @@ class ContainerObject(LiteLLMBaseModel): name: str | None = None _hidden_params: dict[str, Any] = PrivateAttr(default={}) + @property + def hidden_params(self) -> dict[str, builtins.object]: # mutable-ok: API requires mutation + return self._hidden_params + + @hidden_params.setter + def hidden_params(self, hidden_params: dict[str, builtins.object]) -> None: # mutable-ok: API requires mutation + self._hidden_params = hidden_params + def __contains__(self, key: str) -> bool: # Define custom behavior for the 'in' operator return hasattr(self, key) @@ -144,6 +152,14 @@ class ContainerFileObject(LiteLLMBaseModel): source: str _hidden_params: dict[str, builtins.object] = PrivateAttr(default={}) + @property + def hidden_params(self) -> dict[str, builtins.object]: # mutable-ok: API requires mutation + return self._hidden_params + + @hidden_params.setter + def hidden_params(self, hidden_params: dict[str, builtins.object]) -> None: # mutable-ok: API requires mutation + self._hidden_params = hidden_params + def __contains__(self, key: str) -> bool: return hasattr(self, key) diff --git a/litellm/types/decisions.py b/litellm/types/decisions.py index 80db26f201d..2745e00b6c6 100644 --- a/litellm/types/decisions.py +++ b/litellm/types/decisions.py @@ -121,3 +121,7 @@ class DecisionsResponse(LiteLLMPydanticObjectBase): model_config = ConfigDict(extra="allow", frozen=True) _hidden_params: dict[str, object] = PrivateAttr(default_factory=dict) + + @property + def hidden_params(self) -> dict[str, object]: # mutable-ok: API requires mutation + return self._hidden_params diff --git a/litellm/types/google_genai/main.py b/litellm/types/google_genai/main.py index 13ba9423a53..0241c9fa809 100644 --- a/litellm/types/google_genai/main.py +++ b/litellm/types/google_genai/main.py @@ -23,6 +23,14 @@ if TYPE_CHECKING: class GenerateContentResponse(GoogleGenAIGenerateContentResponse, BaseLiteLLMOpenAIResponseObject): _hidden_params: dict = {} + @property + def hidden_params(self) -> dict[str, object]: # mutable-ok: API requires mutation + return self._hidden_params + + @hidden_params.setter + def hidden_params(self, hidden_params: dict[str, object]) -> None: # mutable-ok: API requires mutation + self._hidden_params = hidden_params + else: # Fallback types when google.genai is not available ContentListUnion = Any @@ -50,6 +58,14 @@ else: super().__init__(**kwargs) class GenerateContentResponse(BaseLiteLLMOpenAIResponseObject): + @property + def hidden_params(self) -> dict[str, object]: # mutable-ok: API requires mutation + return self._hidden_params + + @hidden_params.setter + def hidden_params(self, hidden_params: dict[str, object]) -> None: # mutable-ok: API requires mutation + self._hidden_params = hidden_params + def __init__(self, **kwargs) -> None: super().__init__(**kwargs) self._hidden_params = kwargs.get("_hidden_params", {}) diff --git a/litellm/types/interactions/generated.py b/litellm/types/interactions/generated.py index fada4d5728a..c4f7a2d8a7b 100644 --- a/litellm/types/interactions/generated.py +++ b/litellm/types/interactions/generated.py @@ -1083,6 +1083,14 @@ class InteractionsAPIResponse(BaseLiteLLMOpenAIResponseObject): _hidden_params: dict = PrivateAttr(default_factory=dict) + @property + def hidden_params(self) -> dict[str, object]: # mutable-ok: API requires mutation + return self._hidden_params + + @hidden_params.setter + def hidden_params(self, hidden_params: dict[str, object]) -> None: # mutable-ok: API requires mutation + self._hidden_params = hidden_params + class InteractionsAPIStreamingResponse(BaseLiteLLMOpenAIResponseObject): """ @@ -1121,6 +1129,14 @@ class InteractionsAPIStreamingResponse(BaseLiteLLMOpenAIResponseObject): _hidden_params: dict = PrivateAttr(default_factory=dict) + @property + def hidden_params(self) -> dict[str, object]: # mutable-ok: API requires mutation + return self._hidden_params + + @hidden_params.setter + def hidden_params(self, hidden_params: dict[str, object]) -> None: # mutable-ok: API requires mutation + self._hidden_params = hidden_params + class DeleteInteractionResult(BaseLiteLLMOpenAIResponseObject): """Result of deleting an interaction.""" @@ -1130,6 +1146,14 @@ class DeleteInteractionResult(BaseLiteLLMOpenAIResponseObject): _hidden_params: dict = PrivateAttr(default_factory=dict) + @property + def hidden_params(self) -> dict[str, object]: # mutable-ok: API requires mutation + return self._hidden_params + + @hidden_params.setter + def hidden_params(self, hidden_params: dict[str, object]) -> None: # mutable-ok: API requires mutation + self._hidden_params = hidden_params + class CancelInteractionResult(BaseLiteLLMOpenAIResponseObject): """Result of cancelling an interaction.""" @@ -1139,6 +1163,14 @@ class CancelInteractionResult(BaseLiteLLMOpenAIResponseObject): _hidden_params: dict = PrivateAttr(default_factory=dict) + @property + def hidden_params(self) -> dict[str, object]: # mutable-ok: API requires mutation + return self._hidden_params + + @hidden_params.setter + def hidden_params(self, hidden_params: dict[str, object]) -> None: # mutable-ok: API requires mutation + self._hidden_params = hidden_params + # Backwards compatibility aliases InteractionTool = Tool diff --git a/litellm/types/llms/openai.py b/litellm/types/llms/openai.py index f241c4a7a27..78196e90003 100644 --- a/litellm/types/llms/openai.py +++ b/litellm/types/llms/openai.py @@ -1,3 +1,4 @@ +import builtins from collections.abc import Iterable, Mapping from enum import Enum from os import PathLike @@ -120,6 +121,14 @@ class BinaryResponseSummary(TypedDict): class HttpxBinaryResponseContent(_HttpxBinaryResponseContent): _hidden_params: dict + @property + def hidden_params(self) -> dict[str, builtins.object]: # mutable-ok: API requires mutation + return self._hidden_params + + @hidden_params.setter + def hidden_params(self, hidden_params: dict[str, builtins.object]) -> None: # mutable-ok: API requires mutation + self._hidden_params = hidden_params + def __init__(self, response: httpx.Response) -> None: super().__init__(response) self._hidden_params = {} @@ -406,6 +415,14 @@ class OpenAIFileObject(LiteLLMBaseModel): _hidden_params: dict = PrivateAttr(default={"response_cost": 0.0}) # no cost for writing a file + @property + def hidden_params(self) -> dict[str, builtins.object]: # mutable-ok: API requires mutation + return self._hidden_params + + @hidden_params.setter + def hidden_params(self, hidden_params: dict[str, builtins.object]) -> None: # mutable-ok: API requires mutation + self._hidden_params = hidden_params + @model_serializer(mode="wrap") def _omit_absent_batch_guardrail( # noqa: ANN202 # annotating it replaces the model's serialization schema self, handler: SerializerFunctionWrapHandler @@ -1422,6 +1439,14 @@ class ResponsesAPIResponse(BaseLiteLLMOpenAIResponseObject): # Define private attributes using PrivateAttr _hidden_params: dict = PrivateAttr(default_factory=dict) + @property + def hidden_params(self) -> dict[str, builtins.object]: # mutable-ok: API requires mutation + return self._hidden_params + + @hidden_params.setter + def hidden_params(self, hidden_params: dict[str, builtins.object]) -> None: # mutable-ok: API requires mutation + self._hidden_params = hidden_params + @field_validator("reasoning", mode="before") @classmethod def validate_reasoning_to_dict(cls, value: Any) -> dict[str, Any] | None: @@ -1597,6 +1622,14 @@ class ResponseCompletedEvent(BaseLiteLLMOpenAIResponseObject): response: ResponsesAPIResponse _hidden_params: dict = PrivateAttr(default_factory=dict) + @property + def hidden_params(self) -> dict[str, object]: # mutable-ok: API requires mutation + return self._hidden_params + + @hidden_params.setter + def hidden_params(self, hidden_params: dict[str, object]) -> None: # mutable-ok: API requires mutation + self._hidden_params = hidden_params + class ResponseFailedEvent(BaseLiteLLMOpenAIResponseObject): type: Literal[ResponsesAPIStreamEvents.RESPONSE_FAILED] @@ -2397,6 +2430,14 @@ class OpenAIModerationResponse(BaseLiteLLMOpenAIResponseObject): # Define private attributes using PrivateAttr _hidden_params: dict = PrivateAttr(default_factory=dict) + @property + def hidden_params(self) -> dict[str, object]: # mutable-ok: API requires mutation + return self._hidden_params + + @hidden_params.setter + def hidden_params(self, hidden_params: dict[str, object]) -> None: # mutable-ok: API requires mutation + self._hidden_params = hidden_params + class OpenAIChatCompletionLogprobs(TypedDict, total=False): content: list[OpenAIChatCompletionLogprobsContent] @@ -2559,6 +2600,14 @@ class OpenAIVideoObject(LiteLLMBaseModel): _hidden_params: dict[str, _JsonValue] = PrivateAttr(default={}) + @property + def hidden_params(self) -> dict[str, _JsonValue]: # mutable-ok: API requires mutation + return self._hidden_params + + @hidden_params.setter + def hidden_params(self, hidden_params: dict[str, _JsonValue]) -> None: # mutable-ok: API requires mutation + self._hidden_params = hidden_params + def __contains__(self, key) -> bool: return hasattr(self, key) diff --git a/litellm/types/rerank.py b/litellm/types/rerank.py index 81cb1b7e315..6d1ab85012b 100644 --- a/litellm/types/rerank.py +++ b/litellm/types/rerank.py @@ -87,6 +87,14 @@ class RerankResponse(LiteLLMBaseModel): # Define private attributes using PrivateAttr _hidden_params: dict = PrivateAttr(default_factory=dict) + @property + def hidden_params(self) -> dict[str, object]: # mutable-ok: API requires mutation + return self._hidden_params + + @hidden_params.setter + def hidden_params(self, hidden_params: dict[str, object]) -> None: # mutable-ok: API requires mutation + self._hidden_params = hidden_params + def __getitem__(self, key): return self.__dict__[key] diff --git a/litellm/types/responses/main.py b/litellm/types/responses/main.py index 2381c7ff3a1..59d34d6e815 100644 --- a/litellm/types/responses/main.py +++ b/litellm/types/responses/main.py @@ -1,3 +1,4 @@ +import builtins from collections.abc import Mapping, Sequence from typing import Final, Literal, Optional, Union @@ -164,6 +165,14 @@ class DeleteResponseResult(BaseLiteLLMOpenAIResponseObject): # Define private attributes using PrivateAttr _hidden_params: dict = PrivateAttr(default_factory=dict) + @property + def hidden_params(self) -> dict[str, builtins.object]: # mutable-ok: API requires mutation + return self._hidden_params + + @hidden_params.setter + def hidden_params(self, hidden_params: dict[str, builtins.object]) -> None: # mutable-ok: API requires mutation + self._hidden_params = hidden_params + class DecodedResponseId(TypedDict, total=False): """Structure representing a decoded response ID""" diff --git a/litellm/types/utils.py b/litellm/types/utils.py index 3ec5e91af14..f8f8388b6f7 100644 --- a/litellm/types/utils.py +++ b/litellm/types/utils.py @@ -2092,6 +2092,14 @@ class ModelResponseBase(OpenAIObject): _hidden_params: dict = {} + @property + def hidden_params(self) -> dict[str, object]: # mutable-ok: API requires mutation + return self._hidden_params + + @hidden_params.setter + def hidden_params(self, hidden_params: dict[str, object]) -> None: # mutable-ok: API requires mutation + self._hidden_params = hidden_params + _response_headers: dict | None = None def set_provider_response_headers(self, headers: httpx.Headers) -> None: @@ -2314,6 +2322,15 @@ class EmbeddingResponse(OpenAIObject): """Usage statistics for the embedding request.""" _hidden_params: dict = {} + + @property + def hidden_params(self) -> dict[str, object]: # mutable-ok: API requires mutation + return self._hidden_params + + @hidden_params.setter + def hidden_params(self, hidden_params: dict[str, object]) -> None: # mutable-ok: API requires mutation + self._hidden_params = hidden_params + _response_headers: dict | None = None _response_ms: float | None = None @@ -2454,6 +2471,14 @@ class TextCompletionResponse(OpenAIObject): _response_ms: int | None = None _hidden_params: HiddenParams + @property + def hidden_params(self) -> HiddenParams: + return self._hidden_params + + @hidden_params.setter + def hidden_params(self, hidden_params: HiddenParams) -> None: + self._hidden_params = hidden_params + def __init__( self, id=None, @@ -2618,6 +2643,14 @@ from openai.types.images_response import ImagesResponse as OpenAIImageResponse class ImageResponse(OpenAIImageResponse, BaseLiteLLMOpenAIResponseObject): _hidden_params: dict = {} + @property + def hidden_params(self) -> dict[str, object]: # mutable-ok: API requires mutation + return self._hidden_params + + @hidden_params.setter + def hidden_params(self, hidden_params: dict[str, object]) -> None: # mutable-ok: API requires mutation + self._hidden_params = hidden_params + usage: ImageUsage | None = None """ Users might use litellm with older python versions, we don't want this to break for them. @@ -2753,6 +2786,14 @@ class TranscriptionResponse(OpenAIObject): _hidden_params: dict = {} _response_headers: dict | None = None + @property + def hidden_params(self) -> dict[str, object]: # mutable-ok: API requires mutation + return self._hidden_params + + @hidden_params.setter + def hidden_params(self, hidden_params: dict[str, object]) -> None: # mutable-ok: API requires mutation + self._hidden_params = hidden_params + def __init__(self, text=None) -> None: super().__init__(text=text) @@ -4343,6 +4384,15 @@ class SelectTokenizerResponse(TypedDict): class LiteLLMFineTuningJob(FineTuningJob): _hidden_params: dict = {} + + @property + def hidden_params(self) -> dict[str, object]: # mutable-ok: API requires mutation + return self._hidden_params + + @hidden_params.setter + def hidden_params(self, hidden_params: dict[str, object]) -> None: # mutable-ok: API requires mutation + self._hidden_params = hidden_params + seed: int | None = None def __init__(self, **kwargs) -> None: @@ -4356,6 +4406,15 @@ class LiteLLMFineTuningJob(FineTuningJob): class LiteLLMBatch(Batch): _hidden_params: dict = {} + + @property + def hidden_params(self) -> dict[str, object]: # mutable-ok: API requires mutation + return self._hidden_params + + @hidden_params.setter + def hidden_params(self, hidden_params: dict[str, object]) -> None: # mutable-ok: API requires mutation + self._hidden_params = hidden_params + usage: Usage | None = None def __contains__(self, key) -> bool: @@ -4388,6 +4447,14 @@ class LiteLLMRealtimeStreamLoggingObject(LiteLLMPydanticObjectBase): service_tier: str | None = None _hidden_params: dict = {} + @property + def hidden_params(self) -> dict[str, object]: # mutable-ok: API requires mutation + return self._hidden_params + + @hidden_params.setter + def hidden_params(self, hidden_params: dict[str, object]) -> None: # mutable-ok: API requires mutation + self._hidden_params = hidden_params + @field_serializer("results") def _serialize_results(self, results: OpenAIRealtimeStreamList) -> list[dict[str, Any]]: return [dict(event) for event in results] diff --git a/litellm/types/videos/main.py b/litellm/types/videos/main.py index 7c22fe18745..da0fe9e7b67 100644 --- a/litellm/types/videos/main.py +++ b/litellm/types/videos/main.py @@ -26,6 +26,14 @@ class VideoObject(LiteLLMBaseModel): usage: dict[str, Any] | None = None _hidden_params: dict[str, builtins.object] = PrivateAttr(default={}) + @property + def hidden_params(self) -> dict[str, builtins.object]: # mutable-ok: API requires mutation + return self._hidden_params + + @hidden_params.setter + def hidden_params(self, hidden_params: dict[str, builtins.object]) -> None: # mutable-ok: API requires mutation + self._hidden_params = hidden_params + def __contains__(self, key) -> bool: # Define custom behavior for the 'in' operator return hasattr(self, key) @@ -115,6 +123,14 @@ class CharacterObject(LiteLLMBaseModel): name: str _hidden_params: dict[str, builtins.object] = PrivateAttr(default={}) + @property + def hidden_params(self) -> dict[str, builtins.object]: # mutable-ok: API requires mutation + return self._hidden_params + + @hidden_params.setter + def hidden_params(self, hidden_params: dict[str, builtins.object]) -> None: # mutable-ok: API requires mutation + self._hidden_params = hidden_params + def __contains__(self, key) -> bool: return hasattr(self, key) diff --git a/tests/integration/sdk/test_vertex_gemini_image_batch_cost_sdk.py b/tests/integration/sdk/test_vertex_gemini_image_batch_cost_sdk.py index 555c4fd7bbe..bb6f0c6f0f7 100644 --- a/tests/integration/sdk/test_vertex_gemini_image_batch_cost_sdk.py +++ b/tests/integration/sdk/test_vertex_gemini_image_batch_cost_sdk.py @@ -194,6 +194,7 @@ _SDK_SCRIPT: Final = textwrap.dedent( import litellm from litellm.integrations.custom_logger import CustomLogger + from litellm.litellm_core_utils.hidden_params import get_hidden_params from litellm.litellm_core_utils.logging_worker import GLOBAL_LOGGING_WORKER from litellm.types.utils import ModelInfo @@ -208,7 +209,7 @@ _SDK_SCRIPT: Final = textwrap.dedent( end_time: object, ) -> None: standard: Final = kwargs.get("standard_logging_object") - hidden: Final = getattr(response_obj, "_hidden_params", None) + hidden: Final = get_hidden_params(response_obj) if not isinstance(standard, dict): raise RuntimeError("success callback omitted standard_logging_object") if standard.get("call_type") != "aretrieve_batch": diff --git a/tests/unit/caching/test_caching_handler.py b/tests/unit/caching/test_caching_handler.py index 29398a45c95..22eea9935f6 100644 --- a/tests/unit/caching/test_caching_handler.py +++ b/tests/unit/caching/test_caching_handler.py @@ -2146,6 +2146,49 @@ async def test_cache_hit_records_the_looked_up_key_as_the_preset_cache_key(monke assert hit.cached_result._hidden_params["cache_key"] == handler.preset_cache_key +@pytest.mark.asyncio +async def test_text_completion_cache_hit_records_cache_key_in_hidden_params(monkeypatch): + import litellm + from litellm.caching.caching import Cache + from litellm.types.utils import CallTypes + + async def atext_completion(**kwargs): + return None + + monkeypatch.setattr(litellm, "cache", Cache(type="local")) + kwargs = {"model": "gpt-5.4", "prompt": "hello", "caching": True} + await litellm.cache.async_add_cache( + litellm.TextCompletionResponse( + id="cached-text-response", + choices=[litellm.utils.TextChoices(text="cached")], + model="gpt-5.4", + ), + **kwargs, + ) + handler = LLMCachingHandler( + original_function=atext_completion, + request_kwargs=kwargs, + start_time=datetime.now(), + ) + logging_obj = _build_logging_obj(CallTypes.atext_completion.value, stream=False) + logging_obj.async_success_handler = AsyncMock() + + hit = await handler._async_get_cache( + model="gpt-5.4", + original_function=atext_completion, + logging_obj=logging_obj, + start_time=datetime.now(), + call_type=CallTypes.atext_completion.value, + kwargs=kwargs, + args=(), + ) + + assert hit.cached_result is not None + assert isinstance(hit.cached_result, litellm.TextCompletionResponse) + assert handler.preset_cache_key is not None + assert hit.cached_result.hidden_params.get("cache_key") == handler.preset_cache_key + + @pytest.mark.asyncio async def test_converted_stream_cache_hit_replayed_as_plain_object_logs_at_hit_time(monkeypatch): import litellm @@ -2580,7 +2623,7 @@ async def test_acompletion_after_a_response_without_choices_calls_the_provider_a second = await litellm.acompletion(model="gpt-4o", messages=messages, mock_response="hi", caching=True) assert second.choices[0].message.content == "hi" - assert second._hidden_params.get("cache_hit") is not True + assert second.hidden_params.get("cache_hit") is not True def _stream_chunk(choices: list[StreamingChoices]) -> ModelResponseStream: diff --git a/tests/unit/decisions/test_main.py b/tests/unit/decisions/test_main.py index 106710d328f..684248b3561 100644 --- a/tests/unit/decisions/test_main.py +++ b/tests/unit/decisions/test_main.py @@ -267,6 +267,15 @@ def test_decisions_cost_uses_litellm_token_pricing() -> None: assert cost == pytest.approx(expected_cost) +def test_decisions_response_hidden_params_getter_preserves_mutable_identity() -> None: + response: Final = DecisionsResponse(model="decider", answers={}, usage=None) + + assert response.hidden_params is response._hidden_params + + response.hidden_params["mutation"] = "visible" + assert response._hidden_params["mutation"] == "visible" + + @pytest.mark.asyncio async def test_decisions_cost_is_in_standard_logging_object(respx_mock: respx.MockRouter) -> None: respx_mock.post("https://api.perplexity.ai/v1/decisions").respond(json=_RESPONSE) diff --git a/tests/unit/litellm_core_utils/test_health_check_helpers.py b/tests/unit/litellm_core_utils/test_health_check_helpers.py index f677589e1d6..51cf0a057b7 100644 --- a/tests/unit/litellm_core_utils/test_health_check_helpers.py +++ b/tests/unit/litellm_core_utils/test_health_check_helpers.py @@ -21,7 +21,13 @@ from litellm.litellm_core_utils.health_check_helpers import ( ) from litellm.main import ahealth_check from litellm.proxy._types import UserAPIKeyAuth -from litellm.types.utils import LIST_BATCHES_SUPPORTED_PROVIDERS +from litellm.types.llms.base import HiddenParams +from litellm.types.utils import ( + LIST_BATCHES_SUPPORTED_PROVIDERS, + TextChoices, + TextCompletionResponse, + Usage, +) def _png_chunks(png: bytes, offset: int = 8) -> tuple[tuple[bytes, bytes], ...]: @@ -146,6 +152,28 @@ async def test_ahealth_check_supports_image_edit_mode(): assert "Mode image_edit not supported" not in str(result) +@pytest.mark.asyncio +async def test_ahealth_check_completion_includes_headers_from_hidden_params_model() -> None: + response: Final = TextCompletionResponse( + id="cmpl-test", + object="text_completion", + created=1, + model="gpt-3.5-turbo-instruct", + choices=[TextChoices(text="hello", index=0, logprobs=None, finish_reason="stop")], + usage=Usage(prompt_tokens=1, completion_tokens=1, total_tokens=2), + ) + response.hidden_params = HiddenParams(headers={"x-ratelimit-remaining-requests": "5"}) + + with patch("litellm.atext_completion", new_callable=AsyncMock, return_value=response) as mock_atext_completion: + result: Final = await ahealth_check( + {"model": "gpt-3.5-turbo-instruct", "api_key": "sk-test"}, + mode="completion", + ) + + mock_atext_completion.assert_awaited_once() + assert result == {"x-ratelimit-remaining-requests": "5"} + + def test_update_model_params_with_health_check_tracking_information(): """Test update_model_params_with_health_check_tracking_information adds required tracking info.""" initial_model_params = {"model": "gpt-3.5-turbo", "api_key": "test_key"} diff --git a/tests/unit/litellm_core_utils/test_hidden_params.py b/tests/unit/litellm_core_utils/test_hidden_params.py new file mode 100644 index 00000000000..9d1e1e8dceb --- /dev/null +++ b/tests/unit/litellm_core_utils/test_hidden_params.py @@ -0,0 +1,252 @@ +from typing import Final + +import pytest + +from litellm.litellm_core_utils.hidden_params import ( + HIDDEN_PARAMS_ATTR, + get_hidden_params, + get_hidden_params_storage, + get_or_create_hidden_params, + set_hidden_param, + set_hidden_params, +) +from litellm.llms.openai.completion.transformation import OpenAITextCompletionConfig +from litellm.types.decisions import DecisionsResponse +from litellm.types.llms.base import HiddenParams +from litellm.types.utils import ModelResponse, TextChoices, TextCompletionResponse, Usage + + +def test_get_and_set_hidden_params_on_plain_object() -> None: + class PlainResponse: + def __init__(self) -> None: + self._hidden_params = {"existing": True} + + response: Final = PlainResponse() + stored: Final = get_hidden_params(response) + assert stored is response._hidden_params + + replacement: Final = {"replacement": True} + set_hidden_params(response, replacement) + + assert response._hidden_params is replacement + assert get_hidden_params(response) is replacement + + +def test_get_and_set_hidden_params_on_dict() -> None: + response: Final = {"_hidden_params": {"existing": True}} + stored: Final = get_hidden_params(response) + + assert stored is response["_hidden_params"] + + replacement: Final = {"replacement": True} + set_hidden_params(response, replacement) + + assert response["_hidden_params"] is replacement + assert get_hidden_params(response) is replacement + + +def test_get_hidden_params_storage_preserves_supported_storage_identity() -> None: + class PlainResponse: + pass + + dict_storage: Final[dict[str, object]] = {"model_id": "dict-model"} + model_storage: Final = HiddenParams(model_id="model-storage") + dict_response: Final = PlainResponse() + model_response: Final = PlainResponse() + setattr(dict_response, HIDDEN_PARAMS_ATTR, dict_storage) + setattr(model_response, HIDDEN_PARAMS_ATTR, model_storage) + + assert get_hidden_params_storage(dict_response) is dict_storage + assert get_hidden_params_storage(model_response) is model_storage + + +def test_get_or_create_hidden_params_sets_empty_dict_on_plain_object() -> None: + class PlainResponse: + pass + + response: Final = PlainResponse() + hidden_params: Final = get_or_create_hidden_params(response) + + assert response._hidden_params is hidden_params + assert hidden_params == {} + + +def test_set_hidden_param_writes_existing_hidden_params_in_place() -> None: + class PlainResponse: + pass + + storage: Final = HiddenParams(response_cost=0.25) + response: Final = PlainResponse() + setattr(response, HIDDEN_PARAMS_ATTR, storage) + + set_hidden_param(response, "k", "value") + + assert getattr(response, HIDDEN_PARAMS_ATTR) is storage + assert storage["k"] == "value" + + +def test_set_hidden_param_creates_dict_storage_when_missing_or_none() -> None: + class PlainResponse: + pass + + missing_response: Final = PlainResponse() + none_response: Final = PlainResponse() + setattr(none_response, HIDDEN_PARAMS_ATTR, None) + + set_hidden_param(missing_response, "k", "missing") + set_hidden_param(none_response, "k", "none") + + assert getattr(missing_response, HIDDEN_PARAMS_ATTR) == {"k": "missing"} + assert getattr(none_response, HIDDEN_PARAMS_ATTR) == {"k": "none"} + + +def test_set_hidden_param_writes_existing_dict_storage_in_place() -> None: + class PlainResponse: + pass + + storage: Final[dict[str, object]] = {"existing": True} + response: Final = PlainResponse() + setattr(response, HIDDEN_PARAMS_ATTR, storage) + + set_hidden_param(response, "k", "value") + + assert getattr(response, HIDDEN_PARAMS_ATTR) is storage + assert storage == {"existing": True, "k": "value"} + + +def test_set_hidden_param_rejects_unsupported_storage_without_replacing_it() -> None: + class PlainResponse: + pass + + storage: Final = object() + response: Final = PlainResponse() + setattr(response, HIDDEN_PARAMS_ATTR, storage) + + with pytest.raises(TypeError, match="unsupported hidden params storage: object"): + set_hidden_param(response, "k", "value") + + assert getattr(response, HIDDEN_PARAMS_ATTR) is storage + + +def test_get_or_create_hidden_params_wraps_hidden_params_storage() -> None: + class PlainResponse: + pass + + storage: Final = HiddenParams( + response_cost=0.25, + headers={"x-ratelimit-remaining-requests": "5"}, + ) + response: Final = PlainResponse() + setattr(response, HIDDEN_PARAMS_ATTR, storage) + + hidden_params: Final = get_or_create_hidden_params(response) + + assert getattr(response, HIDDEN_PARAMS_ATTR) is storage + assert hidden_params["response_cost"] == 0.25 + assert hidden_params["headers"] == {"x-ratelimit-remaining-requests": "5"} + assert "headers" in hidden_params + assert "headers" in tuple(hidden_params) + + nested: Final = hidden_params.setdefault("nested", {}) + assert hidden_params.setdefault("nested", {}) is nested + assert storage.nested is nested + + del hidden_params["headers"] + assert "headers" not in hidden_params + assert "headers" not in tuple(hidden_params) + + +def test_hidden_params_model_view_excludes_unset_fields() -> None: + class PlainResponse: + pass + + storage: Final = HiddenParams() + response: Final = PlainResponse() + setattr(response, HIDDEN_PARAMS_ATTR, storage) + view: Final = get_or_create_hidden_params(response) + + assert not view + assert list(view) == [] + assert "response_cost" not in view + assert "_response_ms" not in view + assert view.get("model_id", "d") == "d" + assert "_response_ms" not in storage.model_fields_set + + with pytest.raises(KeyError, match="response_cost"): + del view["response_cost"] + + view["model_id"] = "m" + view["extra"] = 1 + + assert "model_id" in view + assert "extra" in view + assert set(view) == {"model_id", "extra"} + assert "model_id" in storage.model_fields_set + assert "extra" in (storage.model_extra or {}) + assert storage.model_id == "m" + assert storage.model_extra == {"extra": 1} + + fallback_response: Final = PlainResponse() + setattr(fallback_response, HIDDEN_PARAMS_ATTR, HiddenParams()) + fallback_view: Final = get_or_create_hidden_params(fallback_response) + merged_after_fallback: Final = {**fallback_view, **{"model_id": "m1"}} + merged_before_fallback: Final = {**{"model_id": "m1"}, **fallback_view} + + assert merged_after_fallback == {"model_id": "m1"} + assert merged_before_fallback == {"model_id": "m1"} + + +def test_hidden_params_model_dump_includes_response_ms_when_excluding_unset() -> None: + storage: Final = HiddenParams(response_cost=1.0) + + assert "_response_ms" in storage.model_dump(exclude_unset=True) + + +def test_openai_text_completion_conversion_preserves_hidden_params_storage() -> None: + source: Final = TextCompletionResponse( + id="cmpl-test", + object="text_completion", + created=1, + model="gpt-3.5-turbo-instruct", + choices=[TextChoices(text="hello", index=0, logprobs=None, finish_reason="stop")], + usage=Usage(prompt_tokens=1, completion_tokens=1, total_tokens=2), + ) + response: Final = ModelResponse() + storage: Final = HiddenParams(response_cost=0.25) + setattr(response, HIDDEN_PARAMS_ATTR, storage) + + converted: Final = OpenAITextCompletionConfig().convert_to_chat_model_response_object( + response_object=source, + model_response_object=response, + ) + + assert converted is response + assert getattr(response, HIDDEN_PARAMS_ATTR) is storage + assert storage["original_response"] is source + + +def test_get_hidden_params_preserves_model_response_identity() -> None: + response: Final = ModelResponse() + + assert get_hidden_params(response) is response.hidden_params + + +def test_get_hidden_params_returns_none_for_non_dict_storage() -> None: + class PlainResponse: + def __init__(self) -> None: + self._hidden_params = object() + + response: Final = PlainResponse() + + assert get_hidden_params(response) is None + assert get_hidden_params_storage(response) is None + + +def test_set_hidden_params_replaces_frozen_decisions_response_private_attr() -> None: + response: Final = DecisionsResponse(model="decider", answers={}, usage=None) + replacement: Final = {"replacement": True} + + set_hidden_params(response, replacement) + + assert response.hidden_params is replacement + assert response._hidden_params is replacement diff --git a/tests/unit/litellm_core_utils/test_litellm_logging.py b/tests/unit/litellm_core_utils/test_litellm_logging.py index 54d058f2e0b..02eacc8494d 100644 --- a/tests/unit/litellm_core_utils/test_litellm_logging.py +++ b/tests/unit/litellm_core_utils/test_litellm_logging.py @@ -21,6 +21,7 @@ import pytest_asyncio from mcp.types import AudioContent, CallToolResult, ImageContent, TextContent from openai import AsyncOpenAI from openai._legacy_response import HttpxBinaryResponseContent +from pydantic import BaseModel import litellm from litellm._internal_context import in_post_response_phase @@ -547,6 +548,38 @@ def test_response_cost_calculator_uses_router_model_id_from_litellm_metadata(): litellm.model_cost.pop(custom_model_id, None) +def test_logging_success_path_reads_custom_pydantic_hidden_params() -> None: + class CustomLLMResponse(BaseModel): + _hidden_params = {"response_cost": 0.25, "custom_field": "preserved"} + + response: Final = CustomLLMResponse() + logging_obj: Final = _make_dict_logging_obj() + metadata: Final[dict[str, object]] = {"request_tag": "preserved"} + logging_obj.model_call_details["litellm_params"] = {"metadata": metadata} + + with ( + patch.object( + logging_obj, + "_build_standard_logging_payload", + return_value={"response_cost": 0.25}, + ), + patch("litellm.litellm_core_utils.litellm_logging.emit_standard_logging_payload"), + patch.object(logging_obj, "_is_recognized_call_type_for_logging", return_value=True), + patch.object(logging_obj, "_transform_usage_objects", side_effect=lambda result: result), + ): + logging_obj.success_handler( + result=response, + start_time=time.time(), + end_time=time.time(), + ) + + assert logging_obj.model_call_details["response_cost"] == 0.25 + assert metadata == { + "request_tag": "preserved", + "hidden_params": {"response_cost": 0.25, "custom_field": "preserved"}, + } + + class TestZeroCostDiagnostic: DEPLOYMENT_ID: Final = "lit7898-query-only-priced-deployment" MODEL_GROUP: Final = "query-only-priced-chat" diff --git a/tests/unit/proxy/fine_tuning_endpoints/test_endpoints.py b/tests/unit/proxy/fine_tuning_endpoints/test_endpoints.py index d1b110f288a..99a79801698 100644 --- a/tests/unit/proxy/fine_tuning_endpoints/test_endpoints.py +++ b/tests/unit/proxy/fine_tuning_endpoints/test_endpoints.py @@ -49,6 +49,11 @@ def _unified_job_id() -> str: return base64.urlsafe_b64encode(unified.encode()).decode().rstrip("=") +def _decode_unified_id(encoded_id: str) -> str: + padding: Final = "=" * (-len(encoded_id) % 4) + return base64.urlsafe_b64decode(f"{encoded_id}{padding}").decode() + + def _job() -> LiteLLMFineTuningJob: job = LiteLLMFineTuningJob( id=RAW_JOB_ID, @@ -211,11 +216,13 @@ async def test_create__raw_validation_file_rejected_when_managed_files_required( @pytest.mark.asyncio async def test_create__unified_training_file_allowed_when_managed_files_required(seams): seams.logging.get_proxy_hook.return_value = ManagedResourceAccessCheckerStub() + unified_file_id: Final = _unified_file_id() with patch.object(litellm, "require_managed_files", True): - await _create(_unified_file_id()) + response: Final = await _create(unified_file_id) assert seams.router.acreate_fine_tuning_job.call_count == 1 + assert response.hidden_params["unified_file_id"] == _decode_unified_id(unified_file_id) @pytest.mark.asyncio @@ -251,11 +258,13 @@ async def test_retrieve__raw_job_id_rejected_when_managed_files_required(seams): @pytest.mark.asyncio async def test_retrieve__unified_job_id_allowed_when_managed_files_required(seams): seams.logging.get_proxy_hook.return_value = ManagedResourceAccessCheckerStub() + unified_job_id: Final = _unified_job_id() with patch.object(litellm, "require_managed_files", True): - await _retrieve(_unified_job_id()) + response: Final = await _retrieve(unified_job_id) assert seams.router.aretrieve_fine_tuning_job.call_count == 1 + assert response.hidden_params["unified_finetuning_job_id"] == _decode_unified_id(unified_job_id) @pytest.mark.asyncio @@ -271,11 +280,13 @@ async def test_cancel__raw_job_id_rejected_when_managed_files_required(seams): @pytest.mark.asyncio async def test_cancel__unified_job_id_allowed_when_managed_files_required(seams): seams.logging.get_proxy_hook.return_value = ManagedResourceAccessCheckerStub() + unified_job_id: Final = _unified_job_id() with patch.object(litellm, "require_managed_files", True): - await _cancel(_unified_job_id()) + response: Final = await _cancel(unified_job_id) assert seams.router.acancel_fine_tuning_job.call_count == 1 + assert response.hidden_params["unified_finetuning_job_id"] == _decode_unified_id(unified_job_id) FINE_TUNING_API_BASE: Final = "https://fine-tuning.test/v1" diff --git a/tests/unit/proxy/hooks/test_dynamic_rate_limiter.py b/tests/unit/proxy/hooks/test_dynamic_rate_limiter.py index a1fc3e5dd89..611924ed658 100644 --- a/tests/unit/proxy/hooks/test_dynamic_rate_limiter.py +++ b/tests/unit/proxy/hooks/test_dynamic_rate_limiter.py @@ -1,4 +1,5 @@ from datetime import datetime, timezone +from typing import Final import asyncio, importlib, litellm, os, pytest @@ -8,6 +9,7 @@ from litellm.proxy.hooks.dynamic_rate_limiter import( DynamicRateLimiterCache, _PROXY_DynamicRateLimitHandler, ) +from litellm.types.utils import HiddenParams, ModelResponse from litellm import DualCache as DualCache_dynamic_rate, Router from litellm._uuid import uuid from litellm.litellm_core_utils.logging_worker import GLOBAL_LOGGING_WORKER @@ -52,6 +54,35 @@ async def test_handler_threads_time_fn_to_internal_cache(): assert await handler.internal_usage_cache.async_get_cache(model="my-fake-model") == 2 +@pytest.mark.asyncio +async def test_success_hook_updates_existing_hidden_params_storage() -> None: + model_id: Final = "rate-limit-deployment" + router: Final = Router( + model_list=[ + { + "model_name": "my-fake-model", + "litellm_params": {"model": "gpt-3.5-turbo", "api_key": "test-key", "tpm": 100, "rpm": 10}, + "model_info": {"id": model_id}, + } + ] + ) + handler: Final = _PROXY_DynamicRateLimitHandler(internal_usage_cache=DualCache()) + handler.update_variables(llm_router=router) + response: Final = ModelResponse() + hidden_params: Final = HiddenParams(model_id=model_id) + response._hidden_params = hidden_params + + result: Final = await handler.async_post_call_success_hook( + data={}, + user_api_key_dict=UserAPIKeyAuth(metadata={}), + response=response, + ) + + assert result is response + assert response._hidden_params is hidden_params + assert response.hidden_params["additional_headers"]["x-litellm-model_group"] == "my-fake-model" + + @pytest.fixture() def _vcr_outcome_gate(request, vcr): install_live_call_probe(request, vcr) diff --git a/tests/unit/proxy/pass_through_endpoints/llm_provider_handlers/test_cohere_passthrough_logging_handler.py b/tests/unit/proxy/pass_through_endpoints/llm_provider_handlers/test_cohere_passthrough_logging_handler.py index 814c1a14f3d..32e87d93fdb 100644 --- a/tests/unit/proxy/pass_through_endpoints/llm_provider_handlers/test_cohere_passthrough_logging_handler.py +++ b/tests/unit/proxy/pass_through_endpoints/llm_provider_handlers/test_cohere_passthrough_logging_handler.py @@ -122,6 +122,7 @@ class TestCoherePassthroughLoggingHandler: assert "kwargs" in result assert result["kwargs"]["model"] == "embed-english-v3.0" assert result["kwargs"]["custom_llm_provider"] == "cohere" + assert result["result"].hidden_params["response_cost"] == 3.6e-07 # Verify cost calculation was called with correct parameters mock_completion_cost.assert_called_once() diff --git a/tests/unit/proxy/pass_through_endpoints/llm_provider_handlers/test_openai_passthrough_logging_handler.py b/tests/unit/proxy/pass_through_endpoints/llm_provider_handlers/test_openai_passthrough_logging_handler.py index eeac2c9f02d..1115b55e027 100644 --- a/tests/unit/proxy/pass_through_endpoints/llm_provider_handlers/test_openai_passthrough_logging_handler.py +++ b/tests/unit/proxy/pass_through_endpoints/llm_provider_handlers/test_openai_passthrough_logging_handler.py @@ -1249,7 +1249,7 @@ class TestOpenAIPassthroughIntegration: assert result["kwargs"]["response_cost"] == 2.8e-07 assert result["kwargs"]["model"] == "text-embedding-3-small" assert result["kwargs"]["custom_llm_provider"] == "openai" - assert result["result"]._hidden_params["response_cost"] == 2.8e-07 + assert result["result"].hidden_params["response_cost"] == 2.8e-07 mock_completion_cost.assert_called_once() assert mock_completion_cost.call_args.kwargs["call_type"] == "aembedding" assert mock_logging_obj.model_call_details["response_cost"] == 2.8e-07 @@ -1742,6 +1742,7 @@ class TestOpenAIPassthroughIntegration: assert result["kwargs"]["response_cost"] == 0.040 assert result["kwargs"]["model"] == "dall-e-3" assert result["kwargs"]["custom_llm_provider"] == "openai" + assert result["result"].hidden_params["response_cost"] == 0.040 # Verify cost calculation was called mock_image_cost_calculator.assert_called_once() @@ -1803,6 +1804,7 @@ class TestOpenAIPassthroughIntegration: assert result["kwargs"]["response_cost"] == 0.020 assert result["kwargs"]["model"] == "dall-e-2" assert result["kwargs"]["custom_llm_provider"] == "openai" + assert result["result"].hidden_params["response_cost"] == 0.020 # Verify cost calculation was called mock_image_cost_calculator.assert_called_once() diff --git a/tests/unit/responses/litellm_completion_transformation/test_litellm_completion_responses.py b/tests/unit/responses/litellm_completion_transformation/test_litellm_completion_responses.py index c17f5443025..2090ff3c9fa 100644 --- a/tests/unit/responses/litellm_completion_transformation/test_litellm_completion_responses.py +++ b/tests/unit/responses/litellm_completion_transformation/test_litellm_completion_responses.py @@ -12,11 +12,13 @@ from openai.types.responses.response_function_web_search import ( ) import litellm +from litellm.litellm_core_utils.hidden_params import HIDDEN_PARAMS_ATTR from litellm.litellm_core_utils.prompt_templates.factory import anthropic_messages_pt from litellm.responses.litellm_completion_transformation.transformation import ( TOOL_CALLS_CACHE, LiteLLMCompletionResponsesConfig, ) +from litellm.types.llms.base import HiddenParams from litellm.types.responses.main import build_web_search_call from litellm.types.utils import ( ChatCompletionMessageToolCall, @@ -766,6 +768,43 @@ class TestLiteLLMCompletionResponsesConfig: "custom_llm_provider": "openai", } + @pytest.mark.parametrize( + "storage", + [ + HiddenParams(model_id="m1", provider_specific_fields={"k": "v"}), + {"model_id": "m1", "provider_specific_fields": {"k": "v"}}, + ], + ) + def test_transform_chat_completion_response_preserves_hidden_params_storage_identity( + self, storage: dict[str, object] | HiddenParams + ) -> None: + chat_completion_response = ModelResponse( + id="test-response-id", + created=1234567890, + model="test-model", + object="chat.completion", + choices=[ + Choices( + finish_reason="stop", + index=0, + message=Message(content="Test response", role="assistant"), + ) + ], + ) + setattr(chat_completion_response, HIDDEN_PARAMS_ATTR, storage) + + responses_api_response = ( + LiteLLMCompletionResponsesConfig.transform_chat_completion_response_to_responses_api_response( + request_input="Test", + responses_api_request={}, + chat_completion_response=chat_completion_response, + ) + ) + + assert getattr(responses_api_response, HIDDEN_PARAMS_ATTR) is storage + assert responses_api_response.hidden_params["model_id"] == "m1" + assert responses_api_response.provider_specific_fields == {"k": "v"} + def test_transform_chat_completion_response_handles_missing_hidden_params(self): """Test that missing _hidden_params defaults to empty dict""" # Setup - no _hidden_params set diff --git a/tests/unit/router_strategy/test_complexity_router.py b/tests/unit/router_strategy/test_complexity_router.py index d943e57fa51..92457f2ce8e 100644 --- a/tests/unit/router_strategy/test_complexity_router.py +++ b/tests/unit/router_strategy/test_complexity_router.py @@ -2748,6 +2748,7 @@ def _llm_response(content: str, response_cost: float | None = None): response.choices = [MagicMock()] response.choices[0].message.content = content response._hidden_params = {} if response_cost is None else {"response_cost": response_cost} + response.hidden_params = response._hidden_params return response diff --git a/tests/unit/router_utils/test_add_retry_fallback_headers.py b/tests/unit/router_utils/test_add_retry_fallback_headers.py index eb7490d76f8..1f4ac3be6ac 100644 --- a/tests/unit/router_utils/test_add_retry_fallback_headers.py +++ b/tests/unit/router_utils/test_add_retry_fallback_headers.py @@ -1,6 +1,6 @@ import json from collections.abc import Mapping -from typing import Literal +from typing import Final, Literal import pytest from pydantic import BaseModel @@ -9,10 +9,12 @@ from litellm.router_utils.add_retry_fallback_headers import ( add_fallback_headers_to_response, add_retry_headers_to_response, complexity_router_decision_headers, + ensure_response_additional_headers, get_fallback_errors_from_headers, get_hidden_params_dict, replace_complexity_router_headers, ) +from litellm.types.decisions import DecisionsResponse class StreamingWrapper: @@ -151,6 +153,32 @@ def test_add_fallback_headers_to_streaming_wrapper(): } +def test_add_fallback_headers_updates_plain_duck_backing_storage() -> None: + class PlainDuckResponse: + def __init__(self) -> None: + self._hidden_params: dict[str, object] = { + "model_id": "deployment-1", + "custom_metadata": {"keep": True}, + "additional_headers": {"x-existing": "keep"}, + } + + response: Final = PlainDuckResponse() + original_hidden_params: Final = response._hidden_params + + result: Final = add_fallback_headers_to_response(response=response, attempted_fallbacks=2) + + assert result is response + assert response._hidden_params is original_hidden_params + assert response._hidden_params == { + "model_id": "deployment-1", + "custom_metadata": {"keep": True}, + "additional_headers": { + "x-existing": "keep", + "x-litellm-attempted-fallbacks": 2, + }, + } + + def test_add_fallback_headers_serializes_fallback_errors(): response = StreamingWrapper() fallback_errors = [ @@ -230,6 +258,47 @@ def test_add_fallback_headers_when_no_existing_additional_headers(): assert response._hidden_params["additional_headers"]["x-litellm-attempted-fallbacks"] == 2 +def test_add_fallback_headers_to_frozen_decisions_response() -> None: + response: Final = DecisionsResponse(model="decider", answers={}, usage=None) + + result: Final = add_fallback_headers_to_response(response=response, attempted_fallbacks=1) + + assert result is response + assert response.hidden_params["additional_headers"] == {"x-litellm-attempted-fallbacks": 1} + + +def test_ensure_response_additional_headers_updates_frozen_decisions_response() -> None: + response: Final = DecisionsResponse(model="decider", answers={}, usage=None) + + additional_headers: Final = ensure_response_additional_headers(response) + + assert additional_headers == {} + assert response.hidden_params["additional_headers"] is additional_headers + + +def test_ensure_response_additional_headers_preserves_plain_duck_storage() -> None: + class PlainDuckResponse: + def __init__(self) -> None: + self._hidden_params: dict[str, object] = { + "model_id": "deployment-1", + "custom_metadata": {"keep": True}, + "additional_headers": {"x-existing": "keep"}, + } + + response: Final = PlainDuckResponse() + original_hidden_params: Final = response._hidden_params + additional_headers: Final = ensure_response_additional_headers(response) + additional_headers["x-added"] = "value" + + assert response._hidden_params is original_hidden_params + assert additional_headers is original_hidden_params["additional_headers"] + assert response._hidden_params == { + "model_id": "deployment-1", + "custom_metadata": {"keep": True}, + "additional_headers": {"x-existing": "keep", "x-added": "value"}, + } + + def test_add_fallback_headers_returns_none_when_response_is_none(): result = add_fallback_headers_to_response(response=None, attempted_fallbacks=1) assert result is None diff --git a/tests/unit/test_router_streaming_fallback_metadata.py b/tests/unit/test_router_streaming_fallback_metadata.py index 6ed70dc7cfe..7e0cb30f1e3 100644 --- a/tests/unit/test_router_streaming_fallback_metadata.py +++ b/tests/unit/test_router_streaming_fallback_metadata.py @@ -1,4 +1,5 @@ import json +from typing import Final from unittest.mock import MagicMock import pytest @@ -7,6 +8,7 @@ import litellm from litellm.proxy.proxy_server import _should_include_fallback_errors from litellm.router import Router from litellm.router_utils.add_retry_fallback_headers import get_hidden_params_dict +from litellm.types.llms.base import HiddenParams def test_apply_fallback_hidden_params_copies_from_fallback_response(): @@ -54,6 +56,64 @@ def test_apply_fallback_hidden_params_copies_from_fallback_response(): } +def test_apply_fallback_hidden_params_updates_plain_duck_chunk(): + class PlainChunk: + def __init__(self) -> None: + self._hidden_params = { + "additional_headers": {"x-existing-chunk-header": "keep"}, + "model_id": "chunk-model-id", + } + + chunk: Final = PlainChunk() + fallback_response = { + "_hidden_params": { + "additional_headers": { + "x-litellm-attempted-fallbacks": 1, + }, + "api_base": "https://fallback.example", + } + } + + Router._apply_fallback_hidden_params_to_item( + fallback_item=chunk, + prepared_fallback_hidden_params=Router._prepare_fallback_hidden_params(fallback_response), + ) + + assert chunk._hidden_params["api_base"] == "https://fallback.example" + assert chunk._hidden_params["model_id"] == "chunk-model-id" + assert chunk._hidden_params["additional_headers"] == { + "x-existing-chunk-header": "keep", + "x-litellm-attempted-fallbacks": 1, + } + + +def test_apply_fallback_hidden_params_normalizes_hidden_params_on_plain_duck_chunk(): + class PlainChunk: + def __init__(self) -> None: + self._hidden_params = HiddenParams( + api_base="https://original.example", + model_id="original-model-id", + additional_headers={"x-existing-chunk-header": "keep"}, + ) + + chunk: Final = PlainChunk() + Router._apply_fallback_hidden_params_to_item( + fallback_item=chunk, + prepared_fallback_hidden_params=( + {"api_base": "https://fallback.example", "model_id": "fallback-model-id"}, + {"x-litellm-attempted-fallbacks": 1}, + ), + ) + + assert isinstance(chunk._hidden_params, dict) + assert chunk._hidden_params["api_base"] == "https://fallback.example" + assert chunk._hidden_params["model_id"] == "fallback-model-id" + assert chunk._hidden_params["additional_headers"] == { + "x-existing-chunk-header": "keep", + "x-litellm-attempted-fallbacks": 1, + } + + def _two_group_fallback_router() -> Router: return litellm.Router( model_list=[ @@ -74,6 +134,20 @@ def _additional_headers(response: object) -> dict: return get_hidden_params_dict(response).get("additional_headers", {}) +@pytest.mark.asyncio +async def test_fastest_response_marks_the_winning_response_hidden_params() -> None: + router: Final = _two_group_fallback_router() + + response: Final = await router.abatch_completion_fastest_response( + model="primary-model, fallback-model", + messages=[{"role": "user", "content": "Hello"}], + mock_testing_fallbacks=True, + mock_response="fastest response", + ) + + assert response.hidden_params["fastest_response_batch_completion"] is True + + @pytest.mark.asyncio async def test_include_fallback_errors_propagates_through_router(): router = _two_group_fallback_router() @@ -125,10 +199,12 @@ def test_apply_fallback_hidden_params_to_item_none_item(): def test_apply_fallback_hidden_params_to_item_no_existing_additional_headers(): - class FakeChunk: - _hidden_params = {"model_id": "test-id"} - - chunk = FakeChunk() + chunk: Final = litellm.ModelResponseStream( + id="test", + model="openai/internal-fallback", + choices=[], + ) + chunk.hidden_params["model_id"] = "test-id" Router._apply_fallback_hidden_params_to_item( chunk, ( @@ -137,9 +213,9 @@ def test_apply_fallback_hidden_params_to_item_no_existing_additional_headers(): ), ) - assert chunk._hidden_params["api_base"] == "http://fallback.example" - assert chunk._hidden_params["model_id"] == "test-id" - assert chunk._hidden_params["additional_headers"] == { + assert chunk.hidden_params["api_base"] == "http://fallback.example" + assert chunk.hidden_params["model_id"] == "test-id" + assert chunk.hidden_params["additional_headers"] == { "x-litellm-attempted-fallbacks": 1 } diff --git a/tests/unit/types/llms/test_types_llms_openai.py b/tests/unit/types/llms/test_types_llms_openai.py index 64ec09838e8..e59643f6509 100644 --- a/tests/unit/types/llms/test_types_llms_openai.py +++ b/tests/unit/types/llms/test_types_llms_openai.py @@ -1,5 +1,5 @@ import asyncio -from typing import Optional +from typing import Final, Optional from unittest.mock import AsyncMock, patch import pytest @@ -7,7 +7,13 @@ import pytest import json import litellm -from litellm.types.llms.openai import HttpxBinaryResponseContent +from litellm.types.llms.openai import ( + HttpxBinaryResponseContent, + OpenAIModerationResponse, + OpenAIVideoObject, + ResponseCompletedEvent, + ResponsesAPIResponse, +) @pytest.mark.parametrize("stream", (False, True)) @@ -559,6 +565,14 @@ class TestOpenAIFileObjectBatchGuardrailSerialization: page = FileListPage(object="list", data=[self._file_object()], has_more=False) assert "litellm_batch_guardrail" not in page.model_dump(mode="json")["data"][0] + def test_file_object_hidden_params_public_accessor_preserves_identity(self): + file_object: Final = self._file_object() + assert file_object.hidden_params is file_object._hidden_params + + replacement: Final[dict[str, object]] = {"public_key": "visible"} + file_object.hidden_params = replacement + assert file_object._hidden_params is replacement + def _binary_content(payload: bytes) -> HttpxBinaryResponseContent: import httpx @@ -570,9 +584,14 @@ def test_httpx_binary_response_content_hidden_params_are_per_instance(): first = _binary_content(b"first") second = _binary_content(b"second") - first._hidden_params["response_cost"] = 0.5 + first.hidden_params["response_cost"] = 0.5 - assert second._hidden_params == {} + assert second.hidden_params == {} + + replacement: Final[dict[str, object]] = {"public_key": "visible"} + first.hidden_params = replacement + assert first._hidden_params is replacement + assert first.hidden_params is replacement def test_set_response_cost_none_leaves_hidden_params_empty(): @@ -589,3 +608,52 @@ def test_set_response_cost_none_leaves_hidden_params_empty(): binary_response.set_response_cost(None) assert "response_cost" not in binary_response._hidden_params + + +def test_responses_api_response_hidden_params_public_accessor_is_instance_scoped() -> None: + first: Final = ResponsesAPIResponse(id="resp_first", created_at=1, output=[]) + second: Final = ResponsesAPIResponse(id="resp_second", created_at=2, output=[]) + + assert first.hidden_params is first._hidden_params + + first.hidden_params["public_key"] = "visible" + assert first._hidden_params["public_key"] == "visible" + assert first.hidden_params is not second.hidden_params + assert "public_key" not in second.hidden_params + + replacement: Final = {"replacement_key": "replacement_value"} + first.hidden_params = replacement + assert first._hidden_params is replacement + assert first.hidden_params is replacement + + +def test_response_completed_event_hidden_params_public_accessor_preserves_identity() -> None: + response: Final = ResponsesAPIResponse(id="resp_event", created_at=1, output=[]) + event: Final = ResponseCompletedEvent(type="response.completed", response=response) + + assert event.hidden_params is event._hidden_params + + replacement: Final[dict[str, object]] = {"public_key": "visible"} + event.hidden_params = replacement + assert event._hidden_params is replacement + + +def test_moderation_response_hidden_params_public_accessor_preserves_identity() -> None: + response: Final = OpenAIModerationResponse(id="modr_test", model="moderation", results=[]) + + assert response.hidden_params is response._hidden_params + + replacement: Final[dict[str, object]] = {"public_key": "visible"} + response.hidden_params = replacement + assert response._hidden_params is replacement + + +def test_openai_video_object_hidden_params_public_accessor_preserves_identity() -> None: + video: Final = OpenAIVideoObject(id="video_test", object="video", status="completed", created_at=1) + + assert video.hidden_params is video._hidden_params + + replacement: Final = {**video.hidden_params, "public_key": "visible"} + video.hidden_params = replacement + assert video._hidden_params is replacement + assert video.hidden_params["public_key"] == "visible" diff --git a/tests/unit/types/test_types_utils.py b/tests/unit/types/test_types_utils.py index 0a8c9414a0d..ebec1b1b924 100644 --- a/tests/unit/types/test_types_utils.py +++ b/tests/unit/types/test_types_utils.py @@ -1,18 +1,46 @@ import json -from typing import Final +from typing import Final, Protocol, cast import pytest - +from litellm.litellm_core_utils.hidden_params import HIDDEN_PARAMS_ATTR +from litellm.types.agents import ( + AgentCreateResponse, + AgentDeleteResult, + AgentListResponse, + AgentVersionsResponse, + LiteLLMSendMessageResponse, +) +from litellm.types.containers.main import ContainerFileObject, ContainerObject +from litellm.types.google_genai.main import GenerateContentResponse +from litellm.types.interactions.generated import ( + CancelInteractionResult, + DeleteInteractionResult, + InteractionsAPIResponse, + InteractionsAPIStreamingResponse, +) +from litellm.types.responses.main import DeleteResponseResult from litellm.types.utils import ( + EmbeddingResponse, HiddenParams, ImageObject, ImageResponse, + ModelResponse, + ModelResponseStream, + TextCompletionResponse, all_litellm_params, text_tokens_without_nested_reasoning, ) +class _DictHiddenParamsAccessor(Protocol): + @property + def hidden_params(self) -> dict[str, object]: ... + + @hidden_params.setter + def hidden_params(self, hidden_params: dict[str, object]) -> None: ... + + def test_rust_is_a_known_litellm_param(): assert "rust" in all_litellm_params @@ -24,6 +52,75 @@ def test_hidden_params_response_ms(): assert hidden_params_dict.get("_response_ms") == 100 +@pytest.mark.parametrize("response_type", (ModelResponse, ModelResponseStream, EmbeddingResponse)) +def test_hidden_params_public_accessor_preserves_identity_and_instance_isolation( + response_type: type[ModelResponse] | type[ModelResponseStream] | type[EmbeddingResponse], +) -> None: + response: Final = response_type() + other_response: Final = response_type() + + assert response.hidden_params is response._hidden_params + + response.hidden_params["public_key"] = "visible" + assert response._hidden_params["public_key"] == "visible" + assert response.hidden_params is not other_response.hidden_params + assert "public_key" not in other_response.hidden_params + + replacement: Final = {"replacement_key": "replacement_value"} + response.hidden_params = replacement + assert response._hidden_params is replacement + assert response.hidden_params is replacement + + +def test_text_completion_response_hidden_params_setter_preserves_identity() -> None: + response: Final = TextCompletionResponse(id="response-id", choices=[], created=1, model="model") + replacement: Final = HiddenParams(model_id="replacement") + + response.hidden_params = replacement + + assert response._hidden_params is replacement + assert response.hidden_params["model_id"] == "replacement" + + +def test_image_response_hidden_params_setter_preserves_identity() -> None: + response: Final = ImageResponse(created=1, data=[]) + replacement: Final = {"response_cost": 0.25} + + response.hidden_params = replacement + + assert response._hidden_params is replacement + assert response.hidden_params["response_cost"] == 0.25 + + +@pytest.mark.parametrize( + "response", + ( + AgentCreateResponse.model_construct(), + AgentDeleteResult.model_construct(), + AgentListResponse.model_construct(), + AgentVersionsResponse.model_construct(), + LiteLLMSendMessageResponse.model_construct(), + ContainerObject.model_construct(), + ContainerFileObject.model_construct(), + GenerateContentResponse(), + InteractionsAPIResponse.model_construct(), + InteractionsAPIStreamingResponse.model_construct(), + DeleteInteractionResult.model_construct(), + CancelInteractionResult.model_construct(), + DeleteResponseResult.model_construct(), + ), +) +def test_hidden_params_public_accessor_updates_its_backing_storage( + response: _DictHiddenParamsAccessor, +) -> None: + current: Final = response.hidden_params + assert cast(object, getattr(response, HIDDEN_PARAMS_ATTR)) is current + + replacement: Final = {"replacement": "visible"} + response.hidden_params = replacement + assert cast(object, getattr(response, HIDDEN_PARAMS_ATTR)) is replacement + + def test_chat_completion_delta_tool_call(): from litellm.types.utils import ChatCompletionDeltaToolCall, Function