mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-08 03:08:45 +00:00
refactor: expose public hidden_params accessors (#44668)
* refactor: expose public hidden_params accessors
Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
* fix: keep hidden params writes on duck-typed responses
Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
* refactor: move hidden params helpers to core utils
Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
* fix: preserve hidden params in duck-typed cost responses
Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
* test: cover hidden params accessors across response types
Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
* fix: preserve hidden params for dynamic responses
Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
* fix: expose logs through backend component
Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
* Revert "fix: expose logs through backend component"
This reverts commit 621e5ff25e.
* test: cover dict hidden params helper path
Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
* fix: preserve hidden params storage on item writes
Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
* test: cover hidden params accessors for OpenAI types
Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
* test: cover hidden params helper and response paths
Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
* fix: complete hidden params accessor migration
Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
* fix(litellm): tighten hidden parameter validation
Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
* fix(litellm): read HiddenParams model storage through get_hidden_params
Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
* fix(litellm): transfer hidden params storage instead of the mapping view
Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
* fix(litellm): expose only set HiddenParams keys through the mapping view
Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
* perf(litellm): read HiddenParams view keys without dumping the model
Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
* test(litellm): assert HiddenParams extras through model_extra
Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
* fix(router): copy streaming hidden params into a plain dict
Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
---------
Co-authored-by: mateo <mateo@berri.ai>
Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
parent
bce020f8d8
commit
d364d5e7ca
111 changed files with 1779 additions and 409 deletions
|
|
@ -33,6 +33,7 @@ from litellm.caching.caching import DualCache
|
|||
from litellm.constants import MAX_FILE_LIST_LIMIT
|
||||
from litellm.files.types import FileRetrieveCallOptions, FileRetrieveProvider
|
||||
from litellm.integrations.custom_logger import CustomLogger
|
||||
from litellm.litellm_core_utils.hidden_params import HIDDEN_PARAMS_ATTR
|
||||
from litellm.litellm_core_utils.prompt_templates.common_utils import (
|
||||
extract_file_metadata,
|
||||
)
|
||||
|
|
@ -1315,7 +1316,10 @@ class PROXY_LiteLLMManagedFiles(CustomLogger, BaseFileEndpoints):
|
|||
model_mappings: Dict[str, str] = {}
|
||||
|
||||
for file_object in responses:
|
||||
model_file_id_mapping = file_object._hidden_params.get("model_file_id_mapping")
|
||||
file_hidden_params = cast( # cast-ok: preserve mapping operations on dynamic file metadata
|
||||
dict[str, object], getattr(file_object, HIDDEN_PARAMS_ATTR)
|
||||
)
|
||||
model_file_id_mapping = file_hidden_params.get("model_file_id_mapping")
|
||||
if model_file_id_mapping and isinstance(model_file_id_mapping, dict):
|
||||
model_mappings.update(model_file_id_mapping)
|
||||
|
||||
|
|
@ -1365,7 +1369,8 @@ class PROXY_LiteLLMManagedFiles(CustomLogger, BaseFileEndpoints):
|
|||
_, file_type = extract_file_metadata(create_file_request["file"])
|
||||
|
||||
output_file_id = file_objects[0].id
|
||||
model_id = file_objects[0]._hidden_params.get("model_id")
|
||||
file_hidden_params: Final = cast(dict[str, object], getattr(file_objects[0], HIDDEN_PARAMS_ATTR))
|
||||
model_id = file_hidden_params.get("model_id")
|
||||
|
||||
unified_file_id = SpecialEnums.LITELLM_MANAGED_FILE_COMPLETE_STR.value.format(
|
||||
file_type,
|
||||
|
|
@ -1431,11 +1436,12 @@ class PROXY_LiteLLMManagedFiles(CustomLogger, BaseFileEndpoints):
|
|||
if decoded_batch_id and is_litellm_executed_batch(decoded_batch_id):
|
||||
return response
|
||||
## Check if unified_file_id is in the response
|
||||
unified_file_id = response._hidden_params.get("unified_file_id") # managed file id
|
||||
unified_batch_id = response._hidden_params.get("unified_batch_id") # managed batch id
|
||||
is_batch_create: Final = response._hidden_params.get(BATCH_CREATE_HIDDEN_PARAM) is True
|
||||
model_id = cast(Optional[str], response._hidden_params.get("model_id"))
|
||||
model_name = cast(Optional[str], response._hidden_params.get("model_name"))
|
||||
response_hidden_params: Final = cast(dict[str, object], getattr(response, HIDDEN_PARAMS_ATTR))
|
||||
unified_file_id = response_hidden_params.get("unified_file_id")
|
||||
unified_batch_id = response_hidden_params.get("unified_batch_id")
|
||||
is_batch_create: Final = response_hidden_params.get(BATCH_CREATE_HIDDEN_PARAM) is True
|
||||
model_id = cast(Optional[str], response_hidden_params.get("model_id"))
|
||||
model_name = cast(Optional[str], response_hidden_params.get("model_name"))
|
||||
|
||||
resolved_model_name = resolve_managed_output_file_model_name(
|
||||
unified_input_file_id=unified_file_id if isinstance(unified_file_id, str) else response.input_file_id,
|
||||
|
|
@ -1547,12 +1553,12 @@ class PROXY_LiteLLMManagedFiles(CustomLogger, BaseFileEndpoints):
|
|||
|
||||
elif isinstance(response, LiteLLMFineTuningJob):
|
||||
## Check if unified_file_id is in the response
|
||||
unified_file_id = response._hidden_params.get("unified_file_id") # managed file id
|
||||
unified_finetuning_job_id = response._hidden_params.get(
|
||||
"unified_finetuning_job_id"
|
||||
) # managed finetuning job id
|
||||
model_id = cast(Optional[str], response._hidden_params.get("model_id"))
|
||||
model_name = cast(Optional[str], response._hidden_params.get("model_name"))
|
||||
finetuning_response_hidden_params: Final = cast( # cast-ok: preserve dynamic mapping behavior
|
||||
dict[str, object], getattr(response, HIDDEN_PARAMS_ATTR)
|
||||
)
|
||||
unified_file_id = finetuning_response_hidden_params.get("unified_file_id")
|
||||
unified_finetuning_job_id = finetuning_response_hidden_params.get("unified_finetuning_job_id")
|
||||
model_id = cast(Optional[str], finetuning_response_hidden_params.get("model_id"))
|
||||
original_response_id = response.id
|
||||
if (unified_file_id or unified_finetuning_job_id) and model_id:
|
||||
response.id = self.get_unified_generic_response_id(model_id=model_id, generic_response_id=response.id)
|
||||
|
|
|
|||
|
|
@ -29,6 +29,7 @@ from litellm._logging import print_verbose, verbose_logger
|
|||
from litellm.caching import InMemoryCache
|
||||
from litellm.caching.caching import S3Cache, response_cache_phase
|
||||
from litellm.constants import CACHE_WRITE_SHUTDOWN_FLUSH_TIMEOUT_SECONDS
|
||||
from litellm.litellm_core_utils.hidden_params import get_hidden_params
|
||||
from litellm.litellm_core_utils.llm_response_utils.response_metadata import (
|
||||
update_response_metadata,
|
||||
)
|
||||
|
|
@ -212,6 +213,15 @@ def _request_cache_key(request_kwargs: Mapping[str, Any]) -> str | None:
|
|||
return request_kwargs.get("cache_key", None)
|
||||
|
||||
|
||||
def _set_cached_hidden_param(response: object, key: str, value: object) -> None:
|
||||
if isinstance(response, TextCompletionResponse):
|
||||
setattr(response.hidden_params, key, value)
|
||||
return
|
||||
hidden_params: Final = get_hidden_params(response)
|
||||
if hidden_params is not None:
|
||||
hidden_params[key] = value
|
||||
|
||||
|
||||
class _CachedEmbeddingRecord(LiteLLMBaseModel):
|
||||
model_config = ConfigDict(frozen=True)
|
||||
|
||||
|
|
@ -371,8 +381,7 @@ class LLMCachingHandler:
|
|||
or self.request_kwargs.get("cache_key")
|
||||
or litellm.cache.get_cache_key(**self.request_kwargs)
|
||||
)
|
||||
if hasattr(cached_result, "_hidden_params"):
|
||||
cached_result._hidden_params["cache_key"] = cache_key
|
||||
_set_cached_hidden_param(cached_result, "cache_key", cache_key)
|
||||
return CachingHandlerResponse(cached_result=cached_result)
|
||||
elif (
|
||||
call_type == CallTypes.aembedding.value
|
||||
|
|
@ -493,8 +502,7 @@ class LLMCachingHandler:
|
|||
or self.request_kwargs.get("cache_key")
|
||||
or litellm.cache.get_cache_key(**self.request_kwargs)
|
||||
)
|
||||
if hasattr(cached_result, "_hidden_params"):
|
||||
cached_result._hidden_params["cache_key"] = cache_key
|
||||
_set_cached_hidden_param(cached_result, "cache_key", cache_key)
|
||||
return CachingHandlerResponse(cached_result=cached_result)
|
||||
return CachingHandlerResponse(cached_result=cached_result)
|
||||
|
||||
|
|
@ -574,7 +582,7 @@ class LLMCachingHandler:
|
|||
model=model_name,
|
||||
data=[None] * len(kwargs_input_as_list),
|
||||
)
|
||||
final_embedding_cached_response._hidden_params["cache_hit"] = True
|
||||
final_embedding_cached_response.hidden_params["cache_hit"] = True
|
||||
|
||||
prompt_tokens = 0
|
||||
aggregated_details: dict | None = None
|
||||
|
|
@ -749,6 +757,7 @@ class LLMCachingHandler:
|
|||
if cached.usage is not None and embedding_response.usage is not None
|
||||
else cached.usage
|
||||
)
|
||||
cached_hidden_params: Final = cached.hidden_params
|
||||
merged: Final = EmbeddingResponse(
|
||||
model=cached.model,
|
||||
data=[
|
||||
|
|
@ -759,7 +768,7 @@ class LLMCachingHandler:
|
|||
],
|
||||
usage=merged_usage,
|
||||
hidden_params={
|
||||
**cached._hidden_params,
|
||||
**cached_hidden_params,
|
||||
"cache_hit": True,
|
||||
},
|
||||
_response_headers=cached._response_headers,
|
||||
|
|
@ -1029,12 +1038,7 @@ class LLMCachingHandler:
|
|||
)
|
||||
|
||||
response_obj: Final = ResponsesAPIResponse(**cached_result)
|
||||
if (
|
||||
hasattr(response_obj, "_hidden_params")
|
||||
and response_obj._hidden_params is not None
|
||||
and isinstance(response_obj._hidden_params, dict)
|
||||
):
|
||||
response_obj._hidden_params["cache_hit"] = True
|
||||
_set_cached_hidden_param(response_obj, "cache_hit", True)
|
||||
|
||||
if _stream_replay_requested(kwargs):
|
||||
cached_result = CachedResponsesAPIStreamingIterator(
|
||||
|
|
@ -1046,12 +1050,7 @@ class LLMCachingHandler:
|
|||
else:
|
||||
cached_result = response_obj
|
||||
|
||||
if (
|
||||
hasattr(cached_result, "_hidden_params")
|
||||
and cached_result._hidden_params is not None
|
||||
and isinstance(cached_result._hidden_params, dict)
|
||||
):
|
||||
cached_result._hidden_params["cache_hit"] = True
|
||||
_set_cached_hidden_param(cached_result, "cache_hit", True)
|
||||
|
||||
#########################################################
|
||||
# Add final timing metrics to the cached result
|
||||
|
|
|
|||
|
|
@ -24,6 +24,7 @@ from pydantic import BaseModel
|
|||
import litellm
|
||||
from litellm import ModelResponse
|
||||
from litellm._logging import verbose_logger
|
||||
from litellm.litellm_core_utils.hidden_params import get_hidden_params, get_or_create_hidden_params
|
||||
from litellm.litellm_core_utils.prompt_templates.common_utils import (
|
||||
responses_reasoning_items_from_thinking_blocks,
|
||||
with_prompt_cache_breakpoint,
|
||||
|
|
@ -993,20 +994,25 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge):
|
|||
|
||||
# Preserve hidden params from the ResponsesAPIResponse, especially the headers
|
||||
# which contain important provider information like x-request-id
|
||||
raw_response_hidden_params: Final = getattr(raw_response, "_hidden_params", {})
|
||||
raw_response_hidden_params: Final = get_hidden_params(raw_response) or {}
|
||||
if raw_response_hidden_params:
|
||||
if not hasattr(model_response, "_hidden_params") or model_response._hidden_params is None:
|
||||
model_response._hidden_params = {}
|
||||
model_response_hidden_params: Final = get_or_create_hidden_params(model_response)
|
||||
# Merge the raw_response hidden params with model_response hidden params
|
||||
# Preserve existing keys in model_response but add/override with raw_response params
|
||||
for key, value in raw_response_hidden_params.items():
|
||||
if key == "additional_headers" and key in model_response._hidden_params:
|
||||
# Merge additional_headers to preserve both sets
|
||||
existing_additional_headers = model_response._hidden_params.get("additional_headers", {})
|
||||
merged_headers = {**value, **existing_additional_headers}
|
||||
model_response._hidden_params[key] = merged_headers
|
||||
if key == "additional_headers" and key in model_response_hidden_params:
|
||||
existing_additional_headers = model_response_hidden_params.get("additional_headers", {})
|
||||
merged_headers = {
|
||||
**cast( # cast-ok: preserve mapping operations on dynamic response metadata
|
||||
"dict[str, object]", value
|
||||
),
|
||||
**cast( # cast-ok: preserve mapping operations on dynamic response metadata
|
||||
"dict[str, object]", existing_additional_headers
|
||||
),
|
||||
}
|
||||
model_response_hidden_params[key] = merged_headers
|
||||
else:
|
||||
model_response._hidden_params[key] = value
|
||||
model_response_hidden_params[key] = value
|
||||
|
||||
return model_response
|
||||
|
||||
|
|
|
|||
|
|
@ -18,6 +18,7 @@ from litellm.constants import (
|
|||
DEFAULT_MAX_LRU_CACHE_SIZE,
|
||||
DEFAULT_REPLICATE_GPU_PRICE_PER_SECOND,
|
||||
)
|
||||
from litellm.litellm_core_utils.hidden_params import HIDDEN_PARAMS_ATTR
|
||||
from litellm.litellm_core_utils.llm_cost_calc.tool_call_cost_tracking import (
|
||||
StandardBuiltInToolCostTracking,
|
||||
)
|
||||
|
|
@ -2028,8 +2029,12 @@ def response_cost_calculator(
|
|||
response_cost = 0.0
|
||||
else:
|
||||
if isinstance(response_object, BaseModel):
|
||||
if hasattr(response_object, "_hidden_params"):
|
||||
provider_response_cost: Final = get_response_cost_from_hidden_params(response_object._hidden_params)
|
||||
if hasattr(response_object, HIDDEN_PARAMS_ATTR):
|
||||
hidden_params: Final = cast( # cast-ok: cost metadata supports dict and Pydantic storage
|
||||
dict[str, object] | BaseModel,
|
||||
getattr(response_object, HIDDEN_PARAMS_ATTR),
|
||||
)
|
||||
provider_response_cost: Final = get_response_cost_from_hidden_params(hidden_params)
|
||||
if provider_response_cost is not None:
|
||||
return provider_response_cost
|
||||
|
||||
|
|
|
|||
|
|
@ -206,7 +206,7 @@ def _parse_response(
|
|||
response.raise_for_status()
|
||||
payload: Final[object] = _DECISIONS_PAYLOAD_ADAPTER.validate_json(response.content)
|
||||
result: Final = _DECISIONS_RESPONSE_ADAPTER.validate_python(prepared.config.unwrap_response(payload))
|
||||
result._hidden_params.update(
|
||||
result.hidden_params.update(
|
||||
{
|
||||
"model": f"{prepared.provider}/{prepared.upstream_model}",
|
||||
"custom_llm_provider": prepared.provider,
|
||||
|
|
|
|||
|
|
@ -230,7 +230,7 @@ def image_generation(
|
|||
else:
|
||||
model = "dall-e-2"
|
||||
custom_llm_provider = "openai" # default to dall-e-2 on openai
|
||||
model_response._hidden_params["model"] = model
|
||||
model_response.hidden_params["model"] = model
|
||||
openai_params: Final = [
|
||||
"user",
|
||||
"request_timeout",
|
||||
|
|
|
|||
|
|
@ -773,12 +773,18 @@ _NO_HEADERS: Final[Mapping[str, object]] = MappingProxyType({})
|
|||
class _CarriesHiddenParams(Protocol):
|
||||
_hidden_params: dict[str, object] # mutable-ok: the responses billed here keep hidden params in a plain dict
|
||||
|
||||
@property
|
||||
def hidden_params(self) -> dict[str, object]: ... # mutable-ok: API requires mutation
|
||||
|
||||
@hidden_params.setter
|
||||
def hidden_params(self, hidden_params: dict[str, object]) -> None: ... # mutable-ok: API requires mutation
|
||||
|
||||
|
||||
def set_response_cost_in_hidden_params(response: _CarriesHiddenParams, cost: float | None) -> None:
|
||||
"""Record a provider-reported cost where the cost calculator looks before the price map."""
|
||||
if cost is None:
|
||||
return
|
||||
hidden_params: Final = response._hidden_params # pyright: ignore[reportPrivateUsage] # no public accessor
|
||||
hidden_params: Final = response.hidden_params
|
||||
additional_headers: Final[object] = hidden_params.get("additional_headers")
|
||||
merged: Final[dict[str, object]] = { # mutable-ok: assigned into the plain-dict hidden params
|
||||
**(additional_headers if isinstance(additional_headers, Mapping) else _NO_HEADERS),
|
||||
|
|
@ -794,7 +800,7 @@ _PROVIDER_HEADERS_ADAPTER: Final = TypeAdapter(Mapping[str, str])
|
|||
def set_provider_response_headers_in_hidden_params(
|
||||
response: _CarriesHiddenParams, headers: httpx.Headers | Mapping[str, str]
|
||||
) -> None:
|
||||
hidden_params: Final = response._hidden_params # pyright: ignore[reportPrivateUsage] # no public accessor
|
||||
hidden_params: Final = response.hidden_params
|
||||
existing_additional_headers: Final[object] = hidden_params.get("additional_headers")
|
||||
raw_headers: Final[dict[str, str]] = dict(headers) # mutable-ok: stored as the plain-dict hidden param
|
||||
additional_headers: Final[dict[str, object]] = { # mutable-ok: assigned into the plain-dict hidden params
|
||||
|
|
|
|||
114
litellm/litellm_core_utils/hidden_params.py
Normal file
114
litellm/litellm_core_utils/hidden_params.py
Normal file
|
|
@ -0,0 +1,114 @@
|
|||
from collections.abc import Iterator, MutableMapping
|
||||
from typing import Final, Protocol, cast
|
||||
|
||||
from litellm.types.llms.base import HiddenParams
|
||||
|
||||
_HIDDEN_PARAMS_ATTR: Final = "_hidden_params"
|
||||
HIDDEN_PARAMS_ATTR: Final = _HIDDEN_PARAMS_ATTR
|
||||
|
||||
|
||||
class _SupportsItemAssignment(Protocol):
|
||||
def __setitem__(self, key: str, value: object) -> None: ...
|
||||
|
||||
|
||||
class HiddenParamsModelView(MutableMapping[str, object]):
|
||||
def __init__(self, hidden_params: HiddenParams) -> None:
|
||||
self._hidden_params = hidden_params
|
||||
|
||||
def __getitem__(self, key: str) -> object:
|
||||
fields: Final = self._keys()
|
||||
if key not in fields:
|
||||
raise KeyError(key)
|
||||
try:
|
||||
return cast(object, getattr(self._hidden_params, key))
|
||||
except AttributeError as error:
|
||||
raise KeyError(key) from error
|
||||
|
||||
def __setitem__(self, key: str, value: object) -> None:
|
||||
setattr(self._hidden_params, key, value)
|
||||
|
||||
def __delitem__(self, key: str) -> None:
|
||||
fields: Final = self._keys()
|
||||
if key not in fields:
|
||||
raise KeyError(key)
|
||||
try:
|
||||
delattr(self._hidden_params, key)
|
||||
except AttributeError as error:
|
||||
raise KeyError(key) from error
|
||||
self._hidden_params.model_fields_set.discard(key)
|
||||
|
||||
def __iter__(self) -> Iterator[str]:
|
||||
return iter(self._keys())
|
||||
|
||||
def __len__(self) -> int:
|
||||
return len(self._keys())
|
||||
|
||||
def _keys(self) -> frozenset[str]:
|
||||
return frozenset(self._hidden_params.model_fields_set) | frozenset(self._hidden_params.model_extra or {})
|
||||
|
||||
|
||||
def _get_hidden_params_storage(obj: object) -> object | None:
|
||||
return obj.get(_HIDDEN_PARAMS_ATTR) if isinstance(obj, dict) else getattr(obj, _HIDDEN_PARAMS_ATTR, None)
|
||||
|
||||
|
||||
def get_hidden_params_storage(obj: object) -> dict[str, object] | HiddenParams | None:
|
||||
hidden_params: Final[object | None] = _get_hidden_params_storage(obj)
|
||||
if isinstance(hidden_params, dict):
|
||||
return cast( # cast-ok: runtime dict validation preserves dynamically typed legacy storage
|
||||
dict[str, object], hidden_params
|
||||
)
|
||||
if isinstance(hidden_params, HiddenParams):
|
||||
return hidden_params
|
||||
return None
|
||||
|
||||
|
||||
def _as_hidden_params_mapping(hidden_params: object | None) -> MutableMapping[str, object] | None:
|
||||
if isinstance(hidden_params, dict):
|
||||
return cast( # cast-ok: runtime dict validation preserves dynamically typed legacy storage
|
||||
dict[str, object], hidden_params
|
||||
)
|
||||
if isinstance(hidden_params, HiddenParams):
|
||||
return HiddenParamsModelView(hidden_params)
|
||||
return None
|
||||
|
||||
|
||||
def get_hidden_params(obj: object) -> MutableMapping[str, object] | None:
|
||||
hidden_params: Final[object | None] = _get_hidden_params_storage(obj)
|
||||
return _as_hidden_params_mapping(hidden_params)
|
||||
|
||||
|
||||
def set_hidden_params(obj: object, hidden_params: dict[str, object] | HiddenParams) -> None:
|
||||
if isinstance(obj, dict):
|
||||
obj[_HIDDEN_PARAMS_ATTR] = hidden_params
|
||||
else:
|
||||
setattr(obj, _HIDDEN_PARAMS_ATTR, hidden_params)
|
||||
|
||||
|
||||
def set_hidden_param(obj: object, key: str, value: object) -> None:
|
||||
hidden_params: Final[object | None] = _get_hidden_params_storage(obj)
|
||||
if hidden_params is None:
|
||||
set_hidden_params(obj, {key: value})
|
||||
return
|
||||
if isinstance(hidden_params, dict):
|
||||
cast( # cast-ok: runtime dict validation preserves dynamically typed legacy storage
|
||||
dict[str, object], hidden_params
|
||||
)[key] = value
|
||||
return
|
||||
if hasattr(hidden_params, "__setitem__"):
|
||||
cast( # cast-ok: hasattr validates the legacy storage assignment protocol
|
||||
_SupportsItemAssignment, hidden_params
|
||||
)[key] = value
|
||||
return
|
||||
raise TypeError(f"unsupported hidden params storage: {type(hidden_params).__name__}")
|
||||
|
||||
|
||||
def get_or_create_hidden_params(obj: object) -> MutableMapping[str, object]:
|
||||
hidden_params: Final[object | None] = _get_hidden_params_storage(obj)
|
||||
if hidden_params is None:
|
||||
created_hidden_params: Final[dict[str, object]] = {}
|
||||
set_hidden_params(obj, created_hidden_params)
|
||||
return created_hidden_params
|
||||
hidden_params_mapping: Final[MutableMapping[str, object] | None] = _as_hidden_params_mapping(hidden_params)
|
||||
if hidden_params_mapping is not None:
|
||||
return hidden_params_mapping
|
||||
raise TypeError(f"unsupported hidden params storage: {type(hidden_params).__name__}")
|
||||
|
|
@ -79,6 +79,7 @@ from litellm.litellm_core_utils.core_helpers import (
|
|||
)
|
||||
from litellm.litellm_core_utils.error_normalization import normalize_error
|
||||
from litellm.litellm_core_utils.get_litellm_params import get_litellm_params
|
||||
from litellm.litellm_core_utils.hidden_params import HIDDEN_PARAMS_ATTR, set_hidden_param
|
||||
from litellm.litellm_core_utils.internal_call_metadata import (
|
||||
MODEL_ACCESS_GROUP_METADATA_KEY,
|
||||
is_unbilled_non_inference_call,
|
||||
|
|
@ -1865,7 +1866,7 @@ class Logging(LiteLLMLoggingBaseClass):
|
|||
else response_result
|
||||
)
|
||||
|
||||
result_hidden_params: Final = getattr(priced_result, "_hidden_params", None) or MappingProxyType({})
|
||||
result_hidden_params: Final = getattr(priced_result, HIDDEN_PARAMS_ATTR, None) or MappingProxyType({})
|
||||
if isinstance(priced_result, (BaseModel, HttpxBinaryResponseContent)) and hasattr(
|
||||
priced_result, "_hidden_params"
|
||||
):
|
||||
|
|
@ -2067,7 +2068,7 @@ class Logging(LiteLLMLoggingBaseClass):
|
|||
|
||||
def _custom_pricing_for(self, result: object) -> bool:
|
||||
litellm_params: Final = getattr(self, "litellm_params", None)
|
||||
result_hidden_params: Final = getattr(result, "_hidden_params", None) or MappingProxyType({})
|
||||
result_hidden_params: Final = getattr(result, HIDDEN_PARAMS_ATTR, None) or MappingProxyType({})
|
||||
additional_headers: Final = (
|
||||
result_hidden_params.get("additional_headers")
|
||||
if isinstance(result_hidden_params, dict)
|
||||
|
|
@ -2416,7 +2417,7 @@ class Logging(LiteLLMLoggingBaseClass):
|
|||
"""
|
||||
if logging_result is None:
|
||||
return
|
||||
hidden_params: Final = getattr(logging_result, "_hidden_params", None)
|
||||
hidden_params: Final = getattr(logging_result, HIDDEN_PARAMS_ATTR, None)
|
||||
if not hidden_params:
|
||||
return
|
||||
if self.model_call_details.get("litellm_params") is None:
|
||||
|
|
@ -2440,7 +2441,7 @@ class Logging(LiteLLMLoggingBaseClass):
|
|||
):
|
||||
"""Resolve hidden params, compute response cost, and emit the standard logging payload."""
|
||||
self._surface_response_headers_from_result(logging_result)
|
||||
hidden_params: Final = getattr(logging_result, "_hidden_params", {})
|
||||
hidden_params: Final = getattr(logging_result, HIDDEN_PARAMS_ATTR, {})
|
||||
if hidden_params:
|
||||
if self.model_call_details.get("litellm_params") is not None:
|
||||
self.model_call_details["litellm_params"].setdefault("metadata", {})
|
||||
|
|
@ -2715,7 +2716,7 @@ class Logging(LiteLLMLoggingBaseClass):
|
|||
Left in place they overwrite the create's real deployment with the
|
||||
poll's empty one in the payload every logging integration reads.
|
||||
"""
|
||||
settled_hidden_params: Final = getattr(result, "_hidden_params", None)
|
||||
settled_hidden_params: Final = getattr(result, HIDDEN_PARAMS_ATTR, None)
|
||||
if isinstance(settled_hidden_params, dict):
|
||||
for poll_scoped_key in ("response_cost", "model_id", "litellm_model_name"):
|
||||
settled_hidden_params.pop(poll_scoped_key, None)
|
||||
|
|
@ -3289,13 +3290,12 @@ class Logging(LiteLLMLoggingBaseClass):
|
|||
batch_successful_requests: Final = kwargs.get("batch_successful_requests", None)
|
||||
batch_failed_requests: Final = kwargs.get("batch_failed_requests", None)
|
||||
has_explicit_batch_data: Final = all(x is not None for x in (batch_cost, batch_usage, batch_models))
|
||||
|
||||
should_compute_batch_data: Final = not has_explicit_batch_data and batch_cost_is_final(result)
|
||||
if has_explicit_batch_data:
|
||||
result._hidden_params["response_cost"] = batch_cost
|
||||
result._hidden_params["batch_models"] = batch_models
|
||||
result._hidden_params["batch_successful_requests"] = batch_successful_requests # pyright: ignore[reportPrivateUsage] # rebind-ok: same result._hidden_params pattern as response_cost/batch_models above
|
||||
result._hidden_params["batch_failed_requests"] = batch_failed_requests # pyright: ignore[reportPrivateUsage] # rebind-ok: same pattern as above
|
||||
set_hidden_param(result, "response_cost", batch_cost)
|
||||
set_hidden_param(result, "batch_models", batch_models)
|
||||
set_hidden_param(result, "batch_successful_requests", batch_successful_requests)
|
||||
set_hidden_param(result, "batch_failed_requests", batch_failed_requests)
|
||||
result.usage = batch_usage
|
||||
batch_prompt_cost: Final = kwargs.get("batch_prompt_cost", None)
|
||||
batch_completion_cost: Final = kwargs.get("batch_completion_cost", None)
|
||||
|
|
@ -3320,10 +3320,10 @@ class Logging(LiteLLMLoggingBaseClass):
|
|||
model_info=self.get_router_deployment_model_info(),
|
||||
)
|
||||
|
||||
result._hidden_params["response_cost"] = batch_result.cost
|
||||
result._hidden_params["batch_models"] = batch_result.models
|
||||
result._hidden_params["batch_successful_requests"] = batch_result.successful_requests # pyright: ignore[reportPrivateUsage] # rebind-ok: same pattern as above
|
||||
result._hidden_params["batch_failed_requests"] = batch_result.failed_requests # pyright: ignore[reportPrivateUsage] # rebind-ok: same pattern as above
|
||||
set_hidden_param(result, "response_cost", batch_result.cost)
|
||||
set_hidden_param(result, "batch_models", batch_result.models)
|
||||
set_hidden_param(result, "batch_successful_requests", batch_result.successful_requests)
|
||||
set_hidden_param(result, "batch_failed_requests", batch_result.failed_requests)
|
||||
result.usage = batch_result.usage
|
||||
self.set_cost_breakdown(
|
||||
input_cost=batch_result.prompt_cost,
|
||||
|
|
@ -6516,7 +6516,7 @@ def _extract_response_obj_and_hidden_params(
|
|||
) -> tuple[dict, dict | None]:
|
||||
"""Extract response_obj and hidden_params from init_response_obj."""
|
||||
hidden_params: dict | None = (
|
||||
getattr(init_response_obj, "_hidden_params", None)
|
||||
getattr(init_response_obj, HIDDEN_PARAMS_ATTR, None)
|
||||
if isinstance(init_response_obj, BaseModel | HttpxBinaryResponseContent)
|
||||
else None
|
||||
)
|
||||
|
|
|
|||
|
|
@ -9,6 +9,7 @@ from typing import Final, Literal, cast
|
|||
import litellm
|
||||
from litellm._logging import verbose_logger
|
||||
from litellm.constants import RESPONSE_FORMAT_TOOL_NAME
|
||||
from litellm.litellm_core_utils.hidden_params import get_hidden_params
|
||||
from litellm.litellm_core_utils.prompt_templates.common_utils import (
|
||||
extract_reasoning_content,
|
||||
)
|
||||
|
|
@ -516,7 +517,8 @@ class LiteLLMResponseObjectHandler:
|
|||
|
||||
text_completion_response["choices"] = choices_list
|
||||
text_completion_response["usage"] = response.get("usage", None)
|
||||
text_completion_response._hidden_params = HiddenParams(**response._hidden_params)
|
||||
response_hidden_params: Final = get_hidden_params(response) or {}
|
||||
text_completion_response.hidden_params = HiddenParams.model_validate(response_hidden_params)
|
||||
return text_completion_response
|
||||
|
||||
@staticmethod
|
||||
|
|
@ -747,9 +749,9 @@ def convert_to_model_response_object(
|
|||
model_response_object._response_ms = (end_time - start_time).total_seconds() * 1000
|
||||
|
||||
if hidden_params is not None:
|
||||
if model_response_object._hidden_params is None:
|
||||
model_response_object._hidden_params = {}
|
||||
model_response_object._hidden_params.update(hidden_params)
|
||||
if model_response_object.hidden_params is None:
|
||||
model_response_object.hidden_params = {}
|
||||
model_response_object.hidden_params.update(hidden_params)
|
||||
|
||||
if _response_headers is not None:
|
||||
model_response_object._response_headers = _response_headers
|
||||
|
|
@ -787,7 +789,7 @@ def convert_to_model_response_object(
|
|||
).total_seconds() * 1000 # return response latency in ms like openai
|
||||
|
||||
if hidden_params is not None:
|
||||
model_response_object._hidden_params = hidden_params
|
||||
model_response_object.hidden_params = hidden_params
|
||||
|
||||
if _response_headers is not None:
|
||||
model_response_object._response_headers = _response_headers
|
||||
|
|
@ -833,13 +835,13 @@ def convert_to_model_response_object(
|
|||
setattr(model_response_object, "usage", tr_usage_object)
|
||||
|
||||
if hidden_params is not None:
|
||||
model_response_object._hidden_params = hidden_params
|
||||
model_response_object.hidden_params = hidden_params
|
||||
|
||||
# Store internally-calculated duration in _hidden_params for cost
|
||||
# tracking without exposing it in the response body. Must be set
|
||||
# after hidden_params assignment to avoid being overwritten.
|
||||
if "_audio_transcription_duration" in response_object:
|
||||
model_response_object._hidden_params["audio_transcription_duration"] = response_object[
|
||||
model_response_object.hidden_params["audio_transcription_duration"] = response_object[
|
||||
"_audio_transcription_duration"
|
||||
]
|
||||
|
||||
|
|
|
|||
|
|
@ -8,6 +8,7 @@ from typing import TYPE_CHECKING, Any, Final, TypeAlias, TypedDict, Union, cast
|
|||
from typing_extensions import ReadOnly, Required
|
||||
|
||||
from litellm._logging import verbose_logger
|
||||
from litellm.litellm_core_utils.hidden_params import HIDDEN_PARAMS_ATTR, set_hidden_params
|
||||
from litellm.types.llms.openai import (
|
||||
ChatCompletionAssistantContentValue,
|
||||
ChatCompletionAudioDelta,
|
||||
|
|
@ -265,7 +266,11 @@ class ChunkProcessor:
|
|||
return model_response
|
||||
# set hidden params from chunk to model_response
|
||||
if model_response is not None and hasattr(model_response, "_hidden_params"):
|
||||
model_response._hidden_params = chunk.get("_hidden_params", {})
|
||||
chunk_hidden_params: Final = chunk.get("_hidden_params", {})
|
||||
if isinstance(chunk_hidden_params, dict):
|
||||
set_hidden_params(model_response, chunk_hidden_params)
|
||||
else:
|
||||
setattr(model_response, HIDDEN_PARAMS_ATTR, chunk_hidden_params)
|
||||
return model_response
|
||||
|
||||
@staticmethod
|
||||
|
|
@ -841,7 +846,7 @@ class ChunkProcessor:
|
|||
elif (isinstance(chunk, ModelResponse) or isinstance(chunk, ModelResponseStream)) and hasattr(
|
||||
chunk, "_hidden_params"
|
||||
):
|
||||
usage_chunk = chunk._hidden_params.get("usage", None)
|
||||
usage_chunk = chunk.hidden_params.get("usage", None)
|
||||
|
||||
if isinstance(usage_chunk, dict):
|
||||
return Usage(**usage_chunk)
|
||||
|
|
|
|||
|
|
@ -20,6 +20,7 @@ import litellm
|
|||
from litellm import verbose_logger
|
||||
from litellm._uuid import uuid
|
||||
from litellm.litellm_core_utils.asyncify import asyncify
|
||||
from litellm.litellm_core_utils.hidden_params import HIDDEN_PARAMS_ATTR, get_hidden_params
|
||||
from litellm.litellm_core_utils.model_response_utils import (
|
||||
is_model_response_stream_empty,
|
||||
)
|
||||
|
|
@ -187,7 +188,7 @@ def _provider_hidden_params(
|
|||
chunk: object,
|
||||
provider_response_model: str | None,
|
||||
) -> Mapping[str, object] | None:
|
||||
hidden: Final[object] = getattr(chunk, "_hidden_params", None)
|
||||
hidden: Final = get_hidden_params(chunk)
|
||||
parsed: Final = _parsed_provider_hidden_params(hidden)
|
||||
provider_specific_fields: Final[object | None] = (
|
||||
dict(parsed.provider_specific_fields) if parsed is not None and parsed.provider_specific_fields else None
|
||||
|
|
@ -206,6 +207,14 @@ def _provider_hidden_params(
|
|||
|
||||
|
||||
class CustomStreamWrapper:
|
||||
@property
|
||||
def hidden_params(self) -> dict[str, object]: # mutable-ok: API requires mutation
|
||||
return self._hidden_params
|
||||
|
||||
@hidden_params.setter
|
||||
def hidden_params(self, hidden_params: dict[str, object]) -> None: # mutable-ok: API requires mutation
|
||||
self._hidden_params = hidden_params
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
completion_stream,
|
||||
|
|
@ -258,7 +267,7 @@ class CustomStreamWrapper:
|
|||
optional_params=self.logging_obj.model_call_details.get("litellm_params", {}),
|
||||
)
|
||||
|
||||
self._hidden_params = {
|
||||
self._hidden_params: dict[str, object] = {
|
||||
"model_id": (_model_info.get("id", None)),
|
||||
"api_base": _api_base,
|
||||
} # returned as x-litellm-model-id response header in proxy
|
||||
|
|
@ -280,7 +289,7 @@ class CustomStreamWrapper:
|
|||
self._repeated_messages_count = 1
|
||||
self.is_function_call = self.check_is_function_call(logging_obj=logging_obj)
|
||||
self.created: int | None = None
|
||||
self._last_returned_hidden_params: dict | None = None
|
||||
self._last_returned_hidden_params: dict[str, object] | None = None
|
||||
|
||||
_cached_logging_provider: Final = self.logging_obj.model_call_details.get("custom_llm_provider", None)
|
||||
self._cached_logging_llm_provider: str | None = _cached_logging_provider
|
||||
|
|
@ -624,8 +633,12 @@ class CustomStreamWrapper:
|
|||
pass
|
||||
if str_line.choices[0].finish_reason:
|
||||
is_finished = True # check if str_line._hidden_params["is_finished"] is True
|
||||
if hasattr(str_line, "_hidden_params") and str_line._hidden_params.get("is_finished") is not None:
|
||||
is_finished = str_line._hidden_params.get("is_finished")
|
||||
str_line_hidden_params: Final = get_hidden_params(str_line)
|
||||
is_finished_from_hidden_params: Final = (
|
||||
str_line_hidden_params.get("is_finished") if str_line_hidden_params is not None else None
|
||||
)
|
||||
if isinstance(is_finished_from_hidden_params, bool):
|
||||
is_finished = is_finished_from_hidden_params
|
||||
finish_reason = str_line.choices[0].finish_reason
|
||||
|
||||
# checking for logprobs
|
||||
|
|
@ -758,14 +771,14 @@ class CustomStreamWrapper:
|
|||
# must win over both caller-supplied hidden_params and the computed
|
||||
# custom_llm_provider/created_at values, so it comes last.
|
||||
if hidden_params is not None:
|
||||
model_response._hidden_params = {
|
||||
model_response.hidden_params = {
|
||||
**hidden_params,
|
||||
"custom_llm_provider": _logging_obj_llm_provider,
|
||||
"created_at": time.time(),
|
||||
**self._base_hidden_params,
|
||||
}
|
||||
else:
|
||||
model_response._hidden_params = {
|
||||
model_response.hidden_params = {
|
||||
"custom_llm_provider": _logging_obj_llm_provider,
|
||||
"created_at": time.time(),
|
||||
**self._base_hidden_params,
|
||||
|
|
@ -796,7 +809,7 @@ class CustomStreamWrapper:
|
|||
self.response_id = id
|
||||
|
||||
if id and isinstance(id, str) and id.strip():
|
||||
model_response._hidden_params["received_model_id"] = id
|
||||
model_response.hidden_params["received_model_id"] = id
|
||||
|
||||
if self.response_id is not None and isinstance(self.response_id, str):
|
||||
model_response.id = self.response_id
|
||||
|
|
@ -1762,9 +1775,9 @@ class CustomStreamWrapper:
|
|||
_usage: Final[Usage | None] = getattr(response, "usage", None)
|
||||
_cost: Final = CustomStreamWrapper._resolve_provider_reported_cost(getattr(_usage, "cost", None))
|
||||
if _cost is not None:
|
||||
if "additional_headers" not in response._hidden_params:
|
||||
response._hidden_params["additional_headers"] = {}
|
||||
response._hidden_params["additional_headers"]["llm_provider-x-litellm-response-cost"] = _cost
|
||||
if "additional_headers" not in response.hidden_params:
|
||||
response.hidden_params["additional_headers"] = {}
|
||||
response.hidden_params["additional_headers"]["llm_provider-x-litellm-response-cost"] = _cost
|
||||
|
||||
def __next__(self) -> "ModelResponseStream":
|
||||
cache_hit = False
|
||||
|
|
@ -1823,14 +1836,22 @@ class CustomStreamWrapper:
|
|||
if getattr(response, "usage", None) is not None:
|
||||
usage_to_preserve = response.usage
|
||||
if usage_to_preserve:
|
||||
response._hidden_params["usage"] = usage_to_preserve
|
||||
response_hidden_params_for_usage = cast( # cast-ok: preserve dynamic mapping behavior
|
||||
dict[str, object], getattr(response, HIDDEN_PARAMS_ATTR)
|
||||
)
|
||||
response_hidden_params_for_usage["usage"] = usage_to_preserve
|
||||
|
||||
obj_dict = response.model_dump()
|
||||
|
||||
if "usage" in obj_dict:
|
||||
del obj_dict["usage"]
|
||||
|
||||
response = self.model_response_creator(chunk=obj_dict, hidden_params=response._hidden_params)
|
||||
response_hidden_params_for_model = cast( # cast-ok: preserve dynamic mapping behavior
|
||||
Mapping[str, object], getattr(response, HIDDEN_PARAMS_ATTR)
|
||||
)
|
||||
response = self.model_response_creator(
|
||||
chunk=obj_dict, hidden_params=response_hidden_params_for_model
|
||||
)
|
||||
## check if empty
|
||||
is_empty = is_model_response_stream_empty(model_response=cast(ModelResponseStream, response))
|
||||
|
||||
|
|
@ -1839,8 +1860,11 @@ class CustomStreamWrapper:
|
|||
# add usage as hidden param
|
||||
if self.sent_last_chunk is True and self.stream_options is None:
|
||||
usage = calculate_total_usage(chunks=self.chunks)
|
||||
response._hidden_params["usage"] = usage
|
||||
self._last_returned_hidden_params = response._hidden_params
|
||||
response_hidden_params_for_final_chunk = cast( # cast-ok: preserve dynamic mapping behavior
|
||||
dict[str, object], getattr(response, HIDDEN_PARAMS_ATTR)
|
||||
)
|
||||
response_hidden_params_for_final_chunk["usage"] = usage
|
||||
self._last_returned_hidden_params = response_hidden_params_for_final_chunk
|
||||
# Add MCP metadata to final chunk if present
|
||||
response = self._add_mcp_metadata_to_final_chunk(response)
|
||||
# RETURN RESULT
|
||||
|
|
@ -1937,7 +1961,7 @@ class CustomStreamWrapper:
|
|||
self.chunks.append(processed_chunk)
|
||||
if self.stream_options is None: # add usage as hidden param
|
||||
usage = calculate_total_usage(chunks=self.chunks)
|
||||
processed_chunk._hidden_params["usage"] = usage
|
||||
processed_chunk.hidden_params["usage"] = usage
|
||||
## LOGGING
|
||||
executor.submit(
|
||||
self.run_success_logging_and_cache_storage,
|
||||
|
|
@ -2035,7 +2059,10 @@ class CustomStreamWrapper:
|
|||
if "usage" in obj_dict:
|
||||
del obj_dict["usage"]
|
||||
processed_chunk = self.model_response_creator(
|
||||
chunk=obj_dict, hidden_params=processed_chunk._hidden_params
|
||||
chunk=obj_dict,
|
||||
hidden_params=cast( # cast-ok: preserve mapping operations on dynamic storage
|
||||
Mapping[str, object], getattr(processed_chunk, HIDDEN_PARAMS_ATTR)
|
||||
),
|
||||
)
|
||||
is_empty = is_model_response_stream_empty(
|
||||
model_response=cast(ModelResponseStream, processed_chunk)
|
||||
|
|
@ -2049,8 +2076,11 @@ class CustomStreamWrapper:
|
|||
# add usage as hidden param
|
||||
if self.sent_last_chunk is True and self.stream_options is None:
|
||||
usage = calculate_total_usage(chunks=self.chunks)
|
||||
processed_chunk._hidden_params["usage"] = usage
|
||||
self._last_returned_hidden_params = processed_chunk._hidden_params
|
||||
processed_chunk_hidden_params = cast( # cast-ok: preserve dynamic mapping behavior
|
||||
dict[str, object], getattr(processed_chunk, HIDDEN_PARAMS_ATTR)
|
||||
)
|
||||
processed_chunk_hidden_params["usage"] = usage
|
||||
self._last_returned_hidden_params = processed_chunk_hidden_params
|
||||
|
||||
# Call post-call streaming deployment hook for final chunk
|
||||
if self.sent_last_chunk is True:
|
||||
|
|
@ -2205,7 +2235,10 @@ class CustomStreamWrapper:
|
|||
self.chunks.append(processed_chunk)
|
||||
if self.stream_options is None:
|
||||
usage: Final = calculate_total_usage(chunks=self.chunks)
|
||||
processed_chunk._hidden_params["usage"] = usage # pyright: ignore[reportPrivateUsage] # sync parity
|
||||
processed_chunk_hidden_params: Final = cast( # cast-ok: preserve dynamic mapping behavior
|
||||
dict[str, object], getattr(processed_chunk, HIDDEN_PARAMS_ATTR)
|
||||
)
|
||||
processed_chunk_hidden_params["usage"] = usage
|
||||
# see sync __next__'s sibling branch: deliberately do NOT restore
|
||||
# here - this chunk is still this call's own data, and restoring
|
||||
# before returning it would corrupt the caller's own log
|
||||
|
|
|
|||
|
|
@ -2668,7 +2668,7 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig):
|
|||
_message = json_mode_message
|
||||
|
||||
model_response.choices[0].message = _message
|
||||
model_response._hidden_params["original_response"] = completion_response["content"]
|
||||
model_response.hidden_params["original_response"] = completion_response["content"]
|
||||
model_response.choices[0].finish_reason = cast(
|
||||
OpenAIChatCompletionFinishReason,
|
||||
map_finish_reason(completion_response["stop_reason"]),
|
||||
|
|
@ -2686,7 +2686,7 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig):
|
|||
model_response.model = completion_response["model"]
|
||||
|
||||
_hidden_params["provider_specific_fields"] = provider_specific_fields
|
||||
model_response._hidden_params = _hidden_params
|
||||
model_response.hidden_params = _hidden_params
|
||||
return model_response
|
||||
|
||||
def get_prefix_prompt(self, messages: list[AllMessageValues]) -> str | None:
|
||||
|
|
|
|||
|
|
@ -20,6 +20,7 @@ from typing_extensions import assert_never
|
|||
from litellm._logging import verbose_logger
|
||||
from litellm._uuid import uuid
|
||||
from litellm.exceptions import MidStreamFallbackError
|
||||
from litellm.litellm_core_utils.hidden_params import set_hidden_params
|
||||
from litellm.llms.base_llm.chat.transformation import BaseLLMException
|
||||
from litellm.types.llms.anthropic import (
|
||||
AppliedEdit,
|
||||
|
|
@ -165,7 +166,7 @@ class _CombinedChunkSplitter:
|
|||
chunk.usage = None
|
||||
hidden_params: Final = getattr(chunk, "_hidden_params", None)
|
||||
if isinstance(hidden_params, dict) and "usage" in hidden_params:
|
||||
chunk._hidden_params = {key: value for key, value in hidden_params.items() if key != "usage"}
|
||||
set_hidden_params(chunk, {key: value for key, value in hidden_params.items() if key != "usage"})
|
||||
|
||||
@staticmethod
|
||||
def _split_by_payload_kind(chunk: "ModelResponseStream") -> "tuple[ModelResponseStream, ...]":
|
||||
|
|
|
|||
|
|
@ -8,6 +8,7 @@ from typing import TYPE_CHECKING, Any, Final, Literal, TypeAlias, TypeVar, cast
|
|||
from pydantic import JsonValue, TypeAdapter
|
||||
|
||||
import litellm
|
||||
from litellm.litellm_core_utils.hidden_params import get_hidden_params
|
||||
from litellm.llms.anthropic.pass_through.utils import (
|
||||
is_reasoning_auto_summary_enabled,
|
||||
prompt_cache_key_from_user_id,
|
||||
|
|
@ -1739,10 +1740,13 @@ class LiteLLMAnthropicMessagesAdapter:
|
|||
)
|
||||
if getattr(response, "usage", None) is not None:
|
||||
litellm_usage_chunk: Usage | None = response.usage
|
||||
elif hasattr(response, "_hidden_params") and "usage" in response._hidden_params:
|
||||
litellm_usage_chunk = response._hidden_params["usage"]
|
||||
else:
|
||||
litellm_usage_chunk = None
|
||||
response_hidden_params: Final = get_hidden_params(response)
|
||||
litellm_usage_chunk = (
|
||||
Usage.model_validate(response_hidden_params["usage"])
|
||||
if response_hidden_params is not None and "usage" in response_hidden_params
|
||||
else None
|
||||
)
|
||||
if litellm_usage_chunk is not None:
|
||||
usage_delta = self._translate_openai_usage_to_anthropic_usage_delta(litellm_usage_chunk)
|
||||
else:
|
||||
|
|
|
|||
|
|
@ -45,7 +45,7 @@ class AnthropicMessagesStreamCacheWriter:
|
|||
self.collected_chunks: list[bytes] = [] # mutable-ok: rebuilding a tuple per SSE chunk is quadratic
|
||||
self.persisted = False
|
||||
self._hidden_params: dict[str, object] = dict( # mutable-ok: callers stamp cache_key in here
|
||||
stream._hidden_params if isinstance(stream, AnthropicMessagesStreamingResponse) else _EMPTY_MAPPING
|
||||
stream.hidden_params if isinstance(stream, AnthropicMessagesStreamingResponse) else _EMPTY_MAPPING
|
||||
)
|
||||
|
||||
@property
|
||||
|
|
|
|||
|
|
@ -365,6 +365,14 @@ class AnthropicMessagesStreamingResponse:
|
|||
self.completion_stream = completion_stream
|
||||
self._hidden_params = hidden_params
|
||||
|
||||
@property
|
||||
def hidden_params(self) -> AnthropicMessagesStreamHiddenParams:
|
||||
return self._hidden_params
|
||||
|
||||
@hidden_params.setter
|
||||
def hidden_params(self, hidden_params: AnthropicMessagesStreamHiddenParams) -> None:
|
||||
self._hidden_params = hidden_params
|
||||
|
||||
@property
|
||||
def has_buffered_provider_output(self) -> bool:
|
||||
return getattr(self.completion_stream, "has_buffered_provider_output", False) is True
|
||||
|
|
|
|||
|
|
@ -142,7 +142,7 @@ class AzureSpeechAudioTranscriptionConfig(BaseAudioTranscriptionConfig):
|
|||
|
||||
text: Final = self._extract_text(payload)
|
||||
response: Final = TranscriptionResponse(text=text)
|
||||
response._hidden_params = response_json
|
||||
response.hidden_params = response_json
|
||||
return response
|
||||
|
||||
def get_error_class(self, error_message: str, status_code: int, headers: dict | httpx.Headers) -> BaseLLMException:
|
||||
|
|
|
|||
|
|
@ -1254,7 +1254,7 @@ class AzureChatCompletion(BaseAzureLLM, BaseLLM):
|
|||
and litellm_params is not None
|
||||
and litellm_params.get("base_model", None) is not None
|
||||
):
|
||||
model_response._hidden_params["model"] = litellm_params.get("base_model", None)
|
||||
model_response.hidden_params["model"] = litellm_params.get("base_model", None)
|
||||
|
||||
# Azure image generation API doesn't support extra_body parameter
|
||||
extra_body: Final = optional_params.pop("extra_body", {})
|
||||
|
|
|
|||
|
|
@ -422,7 +422,7 @@ class BingGroundingSearchConfig(BaseSearchConfig):
|
|||
)
|
||||
if get_secret_str(CONNECTION_ID_ENV):
|
||||
return response
|
||||
response._hidden_params["additional_headers"] = {_RESPONSE_COST_HEADER: 0.0}
|
||||
response.hidden_params["additional_headers"] = {_RESPONSE_COST_HEADER: 0.0}
|
||||
return response
|
||||
|
||||
def get_error_class(
|
||||
|
|
|
|||
|
|
@ -29,6 +29,7 @@ import httpx
|
|||
from typing_extensions import ReadOnly
|
||||
|
||||
from litellm._logging import verbose_logger
|
||||
from litellm.litellm_core_utils.hidden_params import set_hidden_param
|
||||
from litellm.litellm_core_utils.url_utils import encode_url_path_segment
|
||||
from litellm.llms.azure_ai.agents.transformation import (
|
||||
AzureAIAgentsConfig,
|
||||
|
|
@ -231,9 +232,7 @@ class AzureAIAgentsHandler:
|
|||
model_response.model = model
|
||||
|
||||
# Store thread_id for conversation continuity
|
||||
if not hasattr(model_response, "_hidden_params") or model_response._hidden_params is None:
|
||||
model_response._hidden_params = {}
|
||||
model_response._hidden_params["thread_id"] = thread_id
|
||||
set_hidden_param(model_response, "thread_id", thread_id)
|
||||
|
||||
# Estimate token usage
|
||||
try:
|
||||
|
|
@ -660,7 +659,7 @@ class AzureAIAgentsHandler:
|
|||
],
|
||||
)
|
||||
if thread_id:
|
||||
final_chunk._hidden_params = {"thread_id": thread_id}
|
||||
final_chunk.hidden_params = {"thread_id": thread_id}
|
||||
yield final_chunk
|
||||
return
|
||||
|
||||
|
|
@ -706,7 +705,7 @@ class AzureAIAgentsHandler:
|
|||
],
|
||||
)
|
||||
if thread_id:
|
||||
chunk._hidden_params = {"thread_id": thread_id}
|
||||
chunk.hidden_params = {"thread_id": thread_id}
|
||||
yield chunk
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -72,13 +72,14 @@ class AzureModelRouterConfig(AzureAIStudioConfig):
|
|||
Also stamps that model onto ``_hidden_params`` so downstream consumers (spend logs,
|
||||
response restamping) can read it instead of guessing the route from the model string.
|
||||
"""
|
||||
from litellm.litellm_core_utils.hidden_params import (
|
||||
get_hidden_params,
|
||||
set_hidden_params,
|
||||
)
|
||||
from litellm.llms.azure_ai.common_utils import (
|
||||
AZURE_MODEL_ROUTER_SELECTED_MODEL_KEY,
|
||||
AzureFoundryModelInfo,
|
||||
)
|
||||
from litellm.router_utils.add_retry_fallback_headers import (
|
||||
get_hidden_params_dict,
|
||||
)
|
||||
|
||||
# Get base model for the parent call (strips routing prefixes for API compatibility)
|
||||
base_model: Final[str] = AzureFoundryModelInfo.get_base_model(model)
|
||||
|
|
@ -100,12 +101,14 @@ class AzureModelRouterConfig(AzureAIStudioConfig):
|
|||
)
|
||||
selected_model: Final = transformed_response.model
|
||||
if selected_model:
|
||||
# Rebuilt rather than mutated in place: ModelResponseBase declares _hidden_params as a
|
||||
# class-level dict, so an in-place write can bleed into unrelated responses.
|
||||
transformed_response._hidden_params = { # pyright: ignore[reportPrivateUsage] # ModelResponse exposes no public hidden-params setter
|
||||
**get_hidden_params_dict(transformed_response),
|
||||
AZURE_MODEL_ROUTER_SELECTED_MODEL_KEY: selected_model,
|
||||
}
|
||||
transformed_hidden_params: Final = get_hidden_params(transformed_response) or {}
|
||||
set_hidden_params(
|
||||
transformed_response,
|
||||
{
|
||||
**transformed_hidden_params,
|
||||
AZURE_MODEL_ROUTER_SELECTED_MODEL_KEY: selected_model,
|
||||
},
|
||||
)
|
||||
return transformed_response
|
||||
|
||||
def calculate_additional_costs(self, model: str, prompt_tokens: int, completion_tokens: int) -> dict | None:
|
||||
|
|
|
|||
|
|
@ -68,7 +68,7 @@ class AzureAICohereConfig:
|
|||
return image_embeddings_request, v1_embeddings_request, image_embedding_idx
|
||||
|
||||
def _transform_response(self, response: EmbeddingResponse) -> EmbeddingResponse:
|
||||
additional_headers: Final[dict | None] = response._hidden_params.get("additional_headers")
|
||||
additional_headers: Final[dict | None] = response.hidden_params.get("additional_headers")
|
||||
if additional_headers:
|
||||
# CALCULATE USAGE
|
||||
input_tokens: Final[str | None] = additional_headers.get("llm_provider-num_tokens")
|
||||
|
|
|
|||
|
|
@ -105,8 +105,9 @@ class AzureAIRerankConfig(CohereRerankConfig):
|
|||
optional_params=optional_params,
|
||||
litellm_params=litellm_params,
|
||||
)
|
||||
base_model: Final = self._get_base_model(rerank_response._hidden_params.get("llm_provider-azureml-model-group"))
|
||||
rerank_response._hidden_params["model"] = base_model
|
||||
azure_model_group: Final = rerank_response.hidden_params.get("llm_provider-azureml-model-group")
|
||||
base_model: Final = self._get_base_model(azure_model_group if isinstance(azure_model_group, str) else None)
|
||||
rerank_response.hidden_params["model"] = base_model
|
||||
return rerank_response
|
||||
|
||||
def _get_base_model(self, azure_model_group: str | None) -> str | None:
|
||||
|
|
|
|||
|
|
@ -91,6 +91,14 @@ class OCRResponse(LiteLLMPydanticObjectBase):
|
|||
|
||||
_hidden_params: dict = PrivateAttr(default_factory=dict)
|
||||
|
||||
@property
|
||||
def hidden_params(self) -> dict[str, builtins.object]: # mutable-ok: API requires mutation
|
||||
return self._hidden_params
|
||||
|
||||
@hidden_params.setter
|
||||
def hidden_params(self, hidden_params: dict[str, builtins.object]) -> None: # mutable-ok: API requires mutation
|
||||
self._hidden_params = hidden_params
|
||||
|
||||
def set_provider_native_response(self, native_response: Mapping[str, builtins.object]) -> None:
|
||||
"""Keep the provider's own response payload alongside the normalized one."""
|
||||
self._hidden_params[PROVIDER_NATIVE_RESPONSE_KEY] = native_response
|
||||
|
|
|
|||
|
|
@ -6,6 +6,7 @@ returns whatever the sandbox produced. The lifecycle is create container ->
|
|||
run code -> delete container; `code_interpreter_tool` combines all three.
|
||||
"""
|
||||
|
||||
import builtins
|
||||
from typing import Any, Final
|
||||
|
||||
import httpx
|
||||
|
|
@ -27,6 +28,14 @@ class ContainerHandle(LiteLLMPydanticObjectBase):
|
|||
|
||||
_hidden_params: dict = PrivateAttr(default_factory=dict)
|
||||
|
||||
@property
|
||||
def hidden_params(self) -> dict[str, builtins.object]: # mutable-ok: API requires mutation
|
||||
return self._hidden_params
|
||||
|
||||
@hidden_params.setter
|
||||
def hidden_params(self, hidden_params: dict[str, builtins.object]) -> None: # mutable-ok: API requires mutation
|
||||
self._hidden_params = hidden_params
|
||||
|
||||
|
||||
class CodeExecutionResult(LiteLLMPydanticObjectBase):
|
||||
"""Passthrough of the sandbox's own execution output."""
|
||||
|
|
@ -42,6 +51,14 @@ class CodeExecutionResult(LiteLLMPydanticObjectBase):
|
|||
|
||||
_hidden_params: dict = PrivateAttr(default_factory=dict)
|
||||
|
||||
@property
|
||||
def hidden_params(self) -> dict[str, builtins.object]: # mutable-ok: API requires mutation
|
||||
return self._hidden_params
|
||||
|
||||
@hidden_params.setter
|
||||
def hidden_params(self, hidden_params: dict[str, builtins.object]) -> None: # mutable-ok: API requires mutation
|
||||
self._hidden_params = hidden_params
|
||||
|
||||
|
||||
class BaseSandboxConfig:
|
||||
"""Provider-agnostic sandbox operations."""
|
||||
|
|
|
|||
|
|
@ -2,6 +2,7 @@
|
|||
Base Search transformation configuration.
|
||||
"""
|
||||
|
||||
import builtins
|
||||
from typing import TYPE_CHECKING, Any, Final, Literal
|
||||
from urllib.parse import urlsplit
|
||||
|
||||
|
|
@ -77,6 +78,14 @@ class SearchResponse(LiteLLMPydanticObjectBase):
|
|||
# Define private attributes using PrivateAttr
|
||||
_hidden_params: dict = PrivateAttr(default_factory=dict)
|
||||
|
||||
@property
|
||||
def hidden_params(self) -> dict[str, builtins.object]: # mutable-ok: API requires mutation
|
||||
return self._hidden_params
|
||||
|
||||
@hidden_params.setter
|
||||
def hidden_params(self, hidden_params: dict[str, builtins.object]) -> None: # mutable-ok: API requires mutation
|
||||
self._hidden_params = hidden_params
|
||||
|
||||
|
||||
class BaseSearchConfig:
|
||||
"""
|
||||
|
|
|
|||
|
|
@ -14,7 +14,7 @@ from __future__ import annotations
|
|||
|
||||
import base64
|
||||
import os
|
||||
from typing import TYPE_CHECKING, Any, Final
|
||||
from typing import TYPE_CHECKING, Any, Final, cast
|
||||
|
||||
import httpx
|
||||
|
||||
|
|
@ -449,17 +449,18 @@ class BedrockAmazonNovaCanvasImageEditConfig(BaseImageEditConfig):
|
|||
)
|
||||
|
||||
if not hasattr(model_response, "_hidden_params"):
|
||||
model_response._hidden_params = {}
|
||||
if "additional_headers" not in model_response._hidden_params:
|
||||
model_response._hidden_params["additional_headers"] = {}
|
||||
model_response.hidden_params = {}
|
||||
additional_headers: Final = cast( # cast-ok: provider headers are stored as a mutable mapping
|
||||
dict[str, object], model_response.hidden_params.setdefault("additional_headers", {})
|
||||
)
|
||||
|
||||
try:
|
||||
model_info: Final = get_model_info(model, custom_llm_provider="bedrock")
|
||||
cost_per_image: Final = model_info.get("output_cost_per_image", 0)
|
||||
if cost_per_image is not None and model_response.data:
|
||||
model_response._hidden_params["additional_headers"]["llm_provider-x-litellm-response-cost"] = float(
|
||||
cost_per_image
|
||||
) * len(model_response.data)
|
||||
additional_headers["llm_provider-x-litellm-response-cost"] = float(cost_per_image) * len(
|
||||
model_response.data
|
||||
)
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
|
|
|
|||
|
|
@ -22,7 +22,7 @@ API Reference: https://docs.aws.amazon.com/bedrock/latest/userguide/model-parame
|
|||
"""
|
||||
|
||||
import base64
|
||||
from typing import TYPE_CHECKING, Any, Final
|
||||
from typing import TYPE_CHECKING, Any, Final, cast
|
||||
|
||||
import httpx
|
||||
from httpx._types import RequestFiles
|
||||
|
|
@ -335,17 +335,16 @@ class BedrockStabilityImageEditConfig(BaseImageEditConfig):
|
|||
)
|
||||
|
||||
if not hasattr(model_response, "_hidden_params"):
|
||||
model_response._hidden_params = {}
|
||||
if "additional_headers" not in model_response._hidden_params:
|
||||
model_response._hidden_params["additional_headers"] = {}
|
||||
model_response.hidden_params = {}
|
||||
additional_headers: Final = cast( # cast-ok: provider headers are stored as a mutable mapping
|
||||
dict[str, object], model_response.hidden_params.setdefault("additional_headers", {})
|
||||
)
|
||||
|
||||
# Set cost based on model
|
||||
model_info: Final = get_model_info(model, custom_llm_provider="bedrock")
|
||||
cost_per_image: Final = model_info.get("output_cost_per_image", 0)
|
||||
if cost_per_image is not None:
|
||||
model_response._hidden_params["additional_headers"]["llm_provider-x-litellm-response-cost"] = float(
|
||||
cost_per_image
|
||||
)
|
||||
additional_headers["llm_provider-x-litellm-response-cost"] = float(cost_per_image)
|
||||
|
||||
return model_response
|
||||
|
||||
|
|
|
|||
|
|
@ -231,7 +231,7 @@ class BytezChatConfig(BaseConfig):
|
|||
|
||||
model_response.usage = usage
|
||||
|
||||
model_response._hidden_params["additional_headers"] = raw_response.headers
|
||||
model_response.hidden_params["additional_headers"] = raw_response.headers
|
||||
message.provider_specific_fields = {
|
||||
"ratelimit-limit": raw_response.headers.get("ratelimit-limit"),
|
||||
"ratelimit-remaining": raw_response.headers.get("ratelimit-remaining"),
|
||||
|
|
|
|||
|
|
@ -1,10 +1,11 @@
|
|||
from collections.abc import Mapping
|
||||
from typing import TYPE_CHECKING, Any, Final
|
||||
from typing import TYPE_CHECKING, Any, Final, cast
|
||||
|
||||
import httpx
|
||||
|
||||
from litellm.exceptions import AuthenticationError
|
||||
from litellm.litellm_core_utils.core_helpers import process_response_headers
|
||||
from litellm.litellm_core_utils.hidden_params import HIDDEN_PARAMS_ATTR, set_hidden_params
|
||||
from litellm.litellm_core_utils.llm_response_utils.convert_dict_to_response import (
|
||||
safe_convert_created_field,
|
||||
)
|
||||
|
|
@ -237,10 +238,13 @@ class ChatGPTResponsesAPIConfig(OpenAIResponsesAPIConfig):
|
|||
) -> None:
|
||||
raw_headers: Final = dict(raw_response.headers)
|
||||
processed_headers: Final = process_response_headers(raw_headers)
|
||||
if not hasattr(completed_response, "_hidden_params"):
|
||||
setattr(completed_response, "_hidden_params", {})
|
||||
completed_response._hidden_params["additional_headers"] = processed_headers
|
||||
completed_response._hidden_params["headers"] = raw_headers
|
||||
if not hasattr(completed_response, HIDDEN_PARAMS_ATTR):
|
||||
set_hidden_params(completed_response, {})
|
||||
hidden_params: Final = cast( # cast-ok: preserve dynamic mapping behavior
|
||||
dict[str, object], getattr(completed_response, HIDDEN_PARAMS_ATTR)
|
||||
)
|
||||
hidden_params["additional_headers"] = processed_headers
|
||||
hidden_params["headers"] = raw_headers
|
||||
|
||||
def get_complete_url(
|
||||
self,
|
||||
|
|
|
|||
|
|
@ -123,7 +123,7 @@ class DeepgramAudioTranscriptionConfig(BaseAudioTranscriptionConfig):
|
|||
]
|
||||
|
||||
# Store full response in hidden params
|
||||
response._hidden_params = response_json
|
||||
response.hidden_params = response_json
|
||||
|
||||
return response
|
||||
|
||||
|
|
|
|||
|
|
@ -213,7 +213,7 @@ class DeepinfraRerankConfig(BaseRerankConfig):
|
|||
rerank_response = RerankResponse(id=request_id or str(uuid.uuid4()), results=results, meta=meta)
|
||||
|
||||
# Store additional information in hidden params
|
||||
rerank_response._hidden_params = {
|
||||
rerank_response.hidden_params = {
|
||||
"status": status,
|
||||
"runtime_ms": runtime_ms,
|
||||
"cost": cost,
|
||||
|
|
@ -237,7 +237,7 @@ class DeepinfraRerankConfig(BaseRerankConfig):
|
|||
litellm_params=litellm_params,
|
||||
)
|
||||
|
||||
rerank_response._hidden_params["model"] = model
|
||||
rerank_response.hidden_params["model"] = model
|
||||
return rerank_response
|
||||
|
||||
def get_supported_cohere_rerank_params(self, model: str) -> list:
|
||||
|
|
|
|||
|
|
@ -9,11 +9,12 @@ Talks to e2b's REST API directly over httpx (no e2b SDK dependency):
|
|||
|
||||
import json
|
||||
from collections.abc import Mapping
|
||||
from typing import Final
|
||||
from typing import Final, cast
|
||||
|
||||
import httpx
|
||||
from pydantic import ConfigDict, TypeAdapter
|
||||
|
||||
from litellm.litellm_core_utils.hidden_params import HIDDEN_PARAMS_ATTR
|
||||
from litellm.llms.base_llm.sandbox.transformation import (
|
||||
SANDBOX_MAX_OUTPUT_BYTES,
|
||||
BaseSandboxConfig,
|
||||
|
|
@ -90,7 +91,7 @@ class E2BSandboxConfig(BaseSandboxConfig):
|
|||
"domain": data.get("domain") or E2B_DEFAULT_DOMAIN,
|
||||
}
|
||||
)
|
||||
handle._hidden_params = {
|
||||
handle.hidden_params = {
|
||||
"envd_access_token": data.get("envdAccessToken"),
|
||||
"traffic_access_token": data.get("trafficAccessToken"),
|
||||
"api_key": key,
|
||||
|
|
@ -109,8 +110,11 @@ class E2BSandboxConfig(BaseSandboxConfig):
|
|||
**kwargs,
|
||||
) -> CodeExecutionResult:
|
||||
handle: Final = self._as_handle(container)
|
||||
hidden_params: Final = cast( # cast-ok: preserve mapping operations on the validated handle
|
||||
dict[str, object], getattr(handle, HIDDEN_PARAMS_ATTR)
|
||||
)
|
||||
|
||||
token: Final = handle._hidden_params.get("envd_access_token")
|
||||
token: Final = hidden_params.get("envd_access_token")
|
||||
if not token:
|
||||
raise ValueError(
|
||||
"Cannot run code from a sandbox id alone. e2b secure sandboxes "
|
||||
|
|
@ -119,7 +123,7 @@ class E2BSandboxConfig(BaseSandboxConfig):
|
|||
)
|
||||
|
||||
headers: Final = {"Content-Type": "application/json", "X-Access-Token": token}
|
||||
traffic_token: Final = handle._hidden_params.get("traffic_access_token")
|
||||
traffic_token: Final = hidden_params.get("traffic_access_token")
|
||||
if traffic_token:
|
||||
headers["E2B-Traffic-Access-Token"] = traffic_token
|
||||
|
||||
|
|
@ -143,8 +147,11 @@ class E2BSandboxConfig(BaseSandboxConfig):
|
|||
**kwargs,
|
||||
) -> bool:
|
||||
handle: Final = self._as_handle(container)
|
||||
key: Final = api_key or handle._hidden_params.get("api_key") or self.validate_environment()
|
||||
base: Final = api_base or handle._hidden_params.get("api_base") or E2B_API_BASE
|
||||
hidden_params: Final = cast( # cast-ok: preserve mapping operations on the validated handle
|
||||
dict[str, object], getattr(handle, HIDDEN_PARAMS_ATTR)
|
||||
)
|
||||
key: Final = api_key or hidden_params.get("api_key") or self.validate_environment()
|
||||
base: Final = api_base or hidden_params.get("api_base") or E2B_API_BASE
|
||||
try:
|
||||
response: Final = await self._http(client).delete(
|
||||
url=f"{base}/sandboxes/{handle.id}",
|
||||
|
|
@ -161,7 +168,7 @@ class E2BSandboxConfig(BaseSandboxConfig):
|
|||
if isinstance(container, ContainerHandle):
|
||||
return container
|
||||
handle: Final = ContainerHandle(id=str(container), provider="e2b", domain=E2B_DEFAULT_DOMAIN)
|
||||
handle._hidden_params = {}
|
||||
handle.hidden_params = {}
|
||||
return handle
|
||||
|
||||
@staticmethod
|
||||
|
|
|
|||
|
|
@ -147,7 +147,7 @@ class ElevenLabsAudioTranscriptionConfig(BaseAudioTranscriptionConfig):
|
|||
)
|
||||
|
||||
# Store full response in hidden params
|
||||
response._hidden_params = response_json
|
||||
response.hidden_params = response_json
|
||||
|
||||
return response
|
||||
|
||||
|
|
|
|||
|
|
@ -1,9 +1,10 @@
|
|||
from collections.abc import Mapping
|
||||
from typing import TYPE_CHECKING, Any, Final
|
||||
from typing import TYPE_CHECKING, Any, Final, cast
|
||||
|
||||
import httpx
|
||||
from pydantic import ConfigDict, TypeAdapter
|
||||
|
||||
from litellm.litellm_core_utils.hidden_params import HIDDEN_PARAMS_ATTR
|
||||
from litellm.types.llms.openai import OpenAIImageGenerationOptionalParams
|
||||
from litellm.types.utils import ImageResponse
|
||||
|
||||
|
|
@ -237,12 +238,15 @@ class FalAIFluxProV11UltraConfig(FalAIBaseConfig):
|
|||
model_response.data.extend(fal_images_to_image_objects(images))
|
||||
|
||||
# Add additional metadata from Flux Pro response
|
||||
if hasattr(model_response, "_hidden_params"):
|
||||
if hasattr(model_response, HIDDEN_PARAMS_ATTR):
|
||||
hidden_params: Final = cast( # cast-ok: preserve mapping operations on dynamic response metadata
|
||||
dict[str, object], getattr(model_response, HIDDEN_PARAMS_ATTR)
|
||||
)
|
||||
if "seed" in response_object:
|
||||
model_response._hidden_params["seed"] = response_object["seed"]
|
||||
hidden_params["seed"] = response_object["seed"]
|
||||
if "timings" in response_object:
|
||||
model_response._hidden_params["timings"] = response_object["timings"]
|
||||
hidden_params["timings"] = response_object["timings"]
|
||||
if "has_nsfw_concepts" in response_object:
|
||||
model_response._hidden_params["has_nsfw_concepts"] = response_object["has_nsfw_concepts"]
|
||||
hidden_params["has_nsfw_concepts"] = response_object["has_nsfw_concepts"]
|
||||
|
||||
return model_response
|
||||
|
|
|
|||
|
|
@ -1,9 +1,10 @@
|
|||
from collections.abc import Mapping
|
||||
from typing import TYPE_CHECKING, Any, Final
|
||||
from typing import TYPE_CHECKING, Any, Final, cast
|
||||
|
||||
import httpx
|
||||
from pydantic import ConfigDict, TypeAdapter
|
||||
|
||||
from litellm.litellm_core_utils.hidden_params import HIDDEN_PARAMS_ATTR
|
||||
from litellm.types.llms.openai import OpenAIImageGenerationOptionalParams
|
||||
from litellm.types.utils import ImageObject, ImageResponse
|
||||
|
||||
|
|
@ -189,7 +190,10 @@ class FalAIIdeogramV3Config(FalAIBaseConfig):
|
|||
)
|
||||
)
|
||||
|
||||
if hasattr(model_response, "_hidden_params") and "seed" in response_object:
|
||||
model_response._hidden_params["seed"] = response_object["seed"]
|
||||
if hasattr(model_response, HIDDEN_PARAMS_ATTR) and "seed" in response_object:
|
||||
hidden_params: Final = cast( # cast-ok: preserve mapping operations on dynamic response metadata
|
||||
dict[str, object], getattr(model_response, HIDDEN_PARAMS_ATTR)
|
||||
)
|
||||
hidden_params["seed"] = response_object["seed"]
|
||||
|
||||
return model_response
|
||||
|
|
|
|||
|
|
@ -1,7 +1,8 @@
|
|||
from typing import TYPE_CHECKING, Any, Final
|
||||
from typing import TYPE_CHECKING, Any, Final, cast
|
||||
|
||||
import httpx
|
||||
|
||||
from litellm.litellm_core_utils.hidden_params import HIDDEN_PARAMS_ATTR
|
||||
from litellm.types.llms.openai import OpenAIImageGenerationOptionalParams
|
||||
from litellm.types.utils import ImageObject, ImageResponse
|
||||
|
||||
|
|
@ -234,8 +235,11 @@ class FalAIImagen4Config(FalAIBaseConfig):
|
|||
)
|
||||
|
||||
# Add seed metadata from Imagen4 response
|
||||
if hasattr(model_response, "_hidden_params"):
|
||||
if hasattr(model_response, HIDDEN_PARAMS_ATTR):
|
||||
hidden_params: Final = cast( # cast-ok: preserve mapping operations on dynamic response metadata
|
||||
dict[str, object], getattr(model_response, HIDDEN_PARAMS_ATTR)
|
||||
)
|
||||
if "seed" in response_data:
|
||||
model_response._hidden_params["seed"] = response_data["seed"]
|
||||
hidden_params["seed"] = response_data["seed"]
|
||||
|
||||
return model_response
|
||||
|
|
|
|||
|
|
@ -1,9 +1,10 @@
|
|||
from collections.abc import Mapping
|
||||
from typing import TYPE_CHECKING, Any, Final
|
||||
from typing import TYPE_CHECKING, Any, Final, cast
|
||||
|
||||
import httpx
|
||||
from pydantic import ConfigDict, TypeAdapter
|
||||
|
||||
from litellm.litellm_core_utils.hidden_params import HIDDEN_PARAMS_ATTR
|
||||
from litellm.types.llms.openai import OpenAIImageGenerationOptionalParams
|
||||
from litellm.types.utils import ImageObject, ImageResponse
|
||||
|
||||
|
|
@ -268,12 +269,15 @@ class FalAIStableDiffusionConfig(FalAIBaseConfig):
|
|||
)
|
||||
|
||||
# Add additional metadata from Stable Diffusion response
|
||||
if hasattr(model_response, "_hidden_params"):
|
||||
if hasattr(model_response, HIDDEN_PARAMS_ATTR):
|
||||
hidden_params: Final = cast( # cast-ok: preserve mapping operations on dynamic response metadata
|
||||
dict[str, object], getattr(model_response, HIDDEN_PARAMS_ATTR)
|
||||
)
|
||||
if "seed" in response_object:
|
||||
model_response._hidden_params["seed"] = response_object["seed"]
|
||||
hidden_params["seed"] = response_object["seed"]
|
||||
if "timings" in response_object:
|
||||
model_response._hidden_params["timings"] = response_object["timings"]
|
||||
hidden_params["timings"] = response_object["timings"]
|
||||
if "has_nsfw_concepts" in response_object:
|
||||
model_response._hidden_params["has_nsfw_concepts"] = response_object["has_nsfw_concepts"]
|
||||
hidden_params["has_nsfw_concepts"] = response_object["has_nsfw_concepts"]
|
||||
|
||||
return model_response
|
||||
|
|
|
|||
|
|
@ -756,7 +756,7 @@ class FireworksAIConfig(FireworksAIMixin, OpenAIGPTConfig):
|
|||
tool_calls=optional_params.get("tools", None),
|
||||
)
|
||||
|
||||
response._hidden_params = {
|
||||
response.hidden_params = {
|
||||
"additional_headers": additional_headers,
|
||||
**_extract_fireworks_hidden_params(completion_response),
|
||||
}
|
||||
|
|
|
|||
|
|
@ -274,8 +274,8 @@ class GoogleAIStudioInteractionsConfig(BaseInteractionsAPIConfig):
|
|||
verbose_logger.debug("Google AI Interactions response: %s", raw_json)
|
||||
|
||||
response: Final = InteractionsAPIResponse(**raw_json)
|
||||
response._hidden_params["headers"] = dict(raw_response.headers)
|
||||
response._hidden_params["additional_headers"] = process_response_headers(dict(raw_response.headers))
|
||||
response.hidden_params["headers"] = dict(raw_response.headers)
|
||||
response.hidden_params["additional_headers"] = process_response_headers(dict(raw_response.headers))
|
||||
|
||||
return response
|
||||
|
||||
|
|
@ -328,7 +328,7 @@ class GoogleAIStudioInteractionsConfig(BaseInteractionsAPIConfig):
|
|||
headers=dict(raw_response.headers),
|
||||
)
|
||||
response: Final = InteractionsAPIResponse(**raw_json)
|
||||
response._hidden_params["headers"] = dict(raw_response.headers)
|
||||
response.hidden_params["headers"] = dict(raw_response.headers)
|
||||
return response
|
||||
|
||||
def transform_delete_interaction_request(
|
||||
|
|
|
|||
|
|
@ -465,7 +465,7 @@ class HuggingFaceEmbeddingConfig(BaseConfig):
|
|||
total_tokens=prompt_tokens + completion_tokens,
|
||||
)
|
||||
setattr(model_response, "usage", usage)
|
||||
model_response._hidden_params["original_response"] = completion_response
|
||||
model_response.hidden_params["original_response"] = completion_response
|
||||
return model_response
|
||||
|
||||
def transform_response(
|
||||
|
|
|
|||
|
|
@ -225,8 +225,8 @@ class ManusResponsesAPIConfig(OpenAIResponsesAPIConfig):
|
|||
response = ResponsesAPIResponse.model_construct(**raw_response_json)
|
||||
|
||||
# Store processed headers in additional_headers so they get returned to the client
|
||||
response._hidden_params["additional_headers"] = processed_headers
|
||||
response._hidden_params["headers"] = raw_response_headers
|
||||
response.hidden_params["additional_headers"] = processed_headers
|
||||
response.hidden_params["headers"] = raw_response_headers
|
||||
return response
|
||||
|
||||
def supports_native_websocket(self) -> bool:
|
||||
|
|
@ -315,6 +315,6 @@ class ManusResponsesAPIConfig(OpenAIResponsesAPIConfig):
|
|||
response = ResponsesAPIResponse.model_construct(**raw_response_json)
|
||||
|
||||
# Store processed headers in additional_headers so they get returned to the client
|
||||
response._hidden_params["additional_headers"] = processed_headers
|
||||
response._hidden_params["headers"] = raw_response_headers
|
||||
response.hidden_params["additional_headers"] = processed_headers
|
||||
response.hidden_params["headers"] = raw_response_headers
|
||||
return response
|
||||
|
|
|
|||
|
|
@ -147,5 +147,5 @@ class MistralAudioTranscriptionConfig(BaseAudioTranscriptionConfig):
|
|||
if "language" in response_json:
|
||||
response["language"] = response_json["language"]
|
||||
|
||||
response._hidden_params = response_json
|
||||
response.hidden_params = response_json
|
||||
return response
|
||||
|
|
|
|||
|
|
@ -229,7 +229,7 @@ class MistralConfig(OpenAIGPTConfig):
|
|||
@overload
|
||||
def _transform_messages(
|
||||
self, messages: list[AllMessageValues], model: str, is_async: Literal[True]
|
||||
) -> Coroutine[object, object, list[AllMessageValues]]:
|
||||
) -> Coroutine[object, object, list[AllMessageValues]]:
|
||||
...
|
||||
|
||||
@overload
|
||||
|
|
|
|||
|
|
@ -626,7 +626,7 @@ class OCIChatConfig(BaseConfig):
|
|||
else:
|
||||
model_response = handle_generic_response(response_json, model, model_response, raw_response)
|
||||
|
||||
model_response._hidden_params["additional_headers"] = raw_response.headers
|
||||
model_response.hidden_params["additional_headers"] = raw_response.headers
|
||||
return model_response
|
||||
|
||||
@track_llm_api_timing()
|
||||
|
|
|
|||
|
|
@ -204,7 +204,7 @@ class OpenAITextCompletion(BaseLLM):
|
|||
)
|
||||
## RESPONSE OBJECT
|
||||
response_obj: Final = TextCompletionResponse(**response_json)
|
||||
response_obj._hidden_params.original_response = json.dumps(response_json)
|
||||
response_obj.hidden_params.original_response = json.dumps(response_json)
|
||||
return response_obj
|
||||
except Exception as e:
|
||||
status_code: Final = getattr(e, "status_code", 500)
|
||||
|
|
|
|||
|
|
@ -5,6 +5,7 @@ Support for gpt model family
|
|||
from collections.abc import Mapping
|
||||
from typing import Final
|
||||
|
||||
from litellm.litellm_core_utils.hidden_params import set_hidden_param
|
||||
from litellm.llms.base_llm.completion.transformation import BaseTextCompletionConfig
|
||||
from litellm.types.llms.openai import AllMessageValues, OpenAITextCompletionUserMessage
|
||||
from litellm.types.utils import Choices, Message, ModelResponse, TextCompletionResponse
|
||||
|
|
@ -112,8 +113,10 @@ class OpenAITextCompletionConfig(BaseTextCompletionConfig, OpenAIGPTConfig):
|
|||
if "model" in response_object:
|
||||
model_response_object.model = response_object["model"]
|
||||
|
||||
model_response_object._hidden_params["original_response"] = (
|
||||
response_object # track original response, if users make a litellm.text_completion() request, we can return the original response
|
||||
set_hidden_param(
|
||||
model_response_object,
|
||||
"original_response",
|
||||
response_object,
|
||||
)
|
||||
return model_response_object
|
||||
except Exception as e:
|
||||
|
|
|
|||
|
|
@ -1,10 +1,11 @@
|
|||
from collections.abc import Mapping
|
||||
from typing import TYPE_CHECKING, Any, Final, Literal
|
||||
from typing import TYPE_CHECKING, Any, Final, Literal, cast
|
||||
|
||||
import httpx
|
||||
from typing_extensions import ReadOnly, TypedDict
|
||||
|
||||
import litellm
|
||||
from litellm.litellm_core_utils.hidden_params import get_or_create_hidden_params
|
||||
from litellm.litellm_core_utils.llm_cost_calc.tool_call_cost_tracking import (
|
||||
StandardBuiltInToolCostTracking,
|
||||
)
|
||||
|
|
@ -165,11 +166,12 @@ class OpenAIContainerConfig(BaseContainerConfig):
|
|||
provider="openai",
|
||||
)
|
||||
|
||||
if not hasattr(container_obj, "_hidden_params") or container_obj._hidden_params is None:
|
||||
container_obj._hidden_params = {}
|
||||
if "additional_headers" not in container_obj._hidden_params:
|
||||
container_obj._hidden_params["additional_headers"] = {}
|
||||
container_obj._hidden_params["additional_headers"]["llm_provider-x-litellm-response-cost"] = container_cost
|
||||
container_hidden_params: Final = get_or_create_hidden_params(container_obj)
|
||||
container_hidden_params.setdefault("additional_headers", {})
|
||||
container_additional_headers: Final = cast( # cast-ok: preserve mapping operations on response metadata
|
||||
"dict[str, object]", container_hidden_params["additional_headers"]
|
||||
)
|
||||
container_additional_headers["llm_provider-x-litellm-response-cost"] = container_cost
|
||||
|
||||
return container_obj
|
||||
|
||||
|
|
|
|||
|
|
@ -593,8 +593,8 @@ class OpenAIResponsesAPIConfig(BaseResponsesAPIConfig):
|
|||
response = ResponsesAPIResponse.model_construct(**raw_response_json)
|
||||
|
||||
# Store processed headers in additional_headers so they get returned to the client
|
||||
response._hidden_params["additional_headers"] = processed_headers
|
||||
response._hidden_params["headers"] = raw_response_headers
|
||||
response.hidden_params["additional_headers"] = processed_headers
|
||||
response.hidden_params["headers"] = raw_response_headers
|
||||
return response
|
||||
|
||||
def validate_environment(self, headers: dict, model: str, litellm_params: GenericLiteLLMParams | None) -> dict:
|
||||
|
|
@ -841,8 +841,8 @@ class OpenAIResponsesAPIConfig(BaseResponsesAPIConfig):
|
|||
raw_response_headers: Final = dict(raw_response.headers)
|
||||
processed_headers: Final = process_response_headers(raw_response_headers)
|
||||
response: Final = ResponsesAPIResponse.model_validate(raw_response_json)
|
||||
response._hidden_params["additional_headers"] = processed_headers
|
||||
response._hidden_params["headers"] = raw_response_headers
|
||||
response.hidden_params["additional_headers"] = processed_headers
|
||||
response.hidden_params["headers"] = raw_response_headers
|
||||
|
||||
return response
|
||||
|
||||
|
|
@ -923,8 +923,8 @@ class OpenAIResponsesAPIConfig(BaseResponsesAPIConfig):
|
|||
processed_headers: Final = process_response_headers(raw_response_headers)
|
||||
|
||||
response: Final = ResponsesAPIResponse.model_validate(raw_response_json)
|
||||
response._hidden_params["additional_headers"] = processed_headers
|
||||
response._hidden_params["headers"] = raw_response_headers
|
||||
response.hidden_params["additional_headers"] = processed_headers
|
||||
response.hidden_params["headers"] = raw_response_headers
|
||||
|
||||
return response
|
||||
|
||||
|
|
@ -993,7 +993,7 @@ class OpenAIResponsesAPIConfig(BaseResponsesAPIConfig):
|
|||
)
|
||||
response = ResponsesAPIResponse.model_construct(**raw_response_json)
|
||||
|
||||
response._hidden_params["additional_headers"] = processed_headers
|
||||
response._hidden_params["headers"] = raw_response_headers
|
||||
response.hidden_params["additional_headers"] = processed_headers
|
||||
response.hidden_params["headers"] = raw_response_headers
|
||||
|
||||
return response
|
||||
|
|
|
|||
|
|
@ -117,7 +117,7 @@ class OpenAILikeChatConfig(OpenAIGPTConfig):
|
|||
returned_response.model = custom_llm_provider + "/" + (returned_response.model or "")
|
||||
|
||||
if base_model is not None:
|
||||
returned_response._hidden_params["model"] = base_model
|
||||
returned_response.hidden_params["model"] = base_model
|
||||
return returned_response
|
||||
|
||||
def transform_response(
|
||||
|
|
|
|||
|
|
@ -13,6 +13,7 @@ from typing import TYPE_CHECKING, Any, Final, cast
|
|||
import httpx
|
||||
|
||||
import litellm
|
||||
from litellm.litellm_core_utils.hidden_params import HIDDEN_PARAMS_ATTR, set_hidden_params
|
||||
from litellm.llms.base_llm.base_model_iterator import BaseModelResponseIterator
|
||||
from litellm.llms.base_llm.chat.transformation import BaseLLMException
|
||||
from litellm.types.llms.openai import AllMessageValues, ChatCompletionToolParam
|
||||
|
|
@ -216,13 +217,17 @@ class OpenrouterConfig(OpenAIGPTConfig):
|
|||
response_cost: Final = response_json["usage"].get("cost")
|
||||
if response_cost is not None:
|
||||
# Store cost in hidden params for the cost calculator to use
|
||||
if not hasattr(model_response, "_hidden_params"):
|
||||
model_response._hidden_params = {}
|
||||
if "additional_headers" not in model_response._hidden_params:
|
||||
model_response._hidden_params["additional_headers"] = {}
|
||||
model_response._hidden_params["additional_headers"]["llm_provider-x-litellm-response-cost"] = float(
|
||||
response_cost
|
||||
if not hasattr(model_response, HIDDEN_PARAMS_ATTR):
|
||||
set_hidden_params(model_response, {})
|
||||
hidden_params: Final = cast( # cast-ok: preserve mapping operations on dynamic response metadata
|
||||
dict[str, object], getattr(model_response, HIDDEN_PARAMS_ATTR)
|
||||
)
|
||||
if "additional_headers" not in hidden_params:
|
||||
hidden_params["additional_headers"] = {}
|
||||
additional_headers: Final = cast( # cast-ok: preserve mapping operations on response metadata
|
||||
dict[str, object], hidden_params["additional_headers"]
|
||||
)
|
||||
additional_headers["llm_provider-x-litellm-response-cost"] = float(response_cost)
|
||||
except Exception:
|
||||
# If we can't extract cost, continue without it - don't fail the response
|
||||
pass
|
||||
|
|
|
|||
|
|
@ -334,20 +334,20 @@ class OpenRouterImageEditConfig(BaseImageEditConfig):
|
|||
cost: Final = usage_data.get("cost")
|
||||
if cost is not None:
|
||||
if not hasattr(model_response, "_hidden_params"):
|
||||
model_response._hidden_params = {}
|
||||
if "additional_headers" not in model_response._hidden_params:
|
||||
model_response._hidden_params["additional_headers"] = {}
|
||||
model_response._hidden_params["additional_headers"]["llm_provider-x-litellm-response-cost"] = float(
|
||||
cost
|
||||
model_response.hidden_params = {}
|
||||
additional_headers: Final = cast( # cast-ok: provider headers are stored as a mutable mapping
|
||||
dict[str, object], model_response.hidden_params.setdefault("additional_headers", {})
|
||||
)
|
||||
additional_headers["llm_provider-x-litellm-response-cost"] = float(cost)
|
||||
|
||||
cost_details: Final = usage_data.get("cost_details", {})
|
||||
if cost_details:
|
||||
if "response_cost_details" not in model_response._hidden_params:
|
||||
model_response._hidden_params["response_cost_details"] = {}
|
||||
model_response._hidden_params["response_cost_details"].update(cost_details)
|
||||
response_cost_details: Final = cast( # cast-ok: provider cost details are stored as a mutable mapping
|
||||
dict[str, object], model_response.hidden_params.setdefault("response_cost_details", {})
|
||||
)
|
||||
response_cost_details.update(cost_details)
|
||||
|
||||
model_response._hidden_params["model"] = response_json.get("model", model)
|
||||
model_response.hidden_params["model"] = response_json.get("model", model)
|
||||
|
||||
def _read_image_bytes(self, image: FileTypes) -> bytes:
|
||||
"""Read raw bytes from various image input types."""
|
||||
|
|
|
|||
|
|
@ -28,12 +28,13 @@ Response format:
|
|||
"""
|
||||
|
||||
from collections.abc import Iterable, Mapping
|
||||
from typing import TYPE_CHECKING, Any, Final
|
||||
from typing import TYPE_CHECKING, Any, Final, cast
|
||||
|
||||
import httpx
|
||||
from pydantic import ConfigDict, TypeAdapter
|
||||
|
||||
import litellm
|
||||
from litellm.litellm_core_utils.hidden_params import HIDDEN_PARAMS_ATTR, set_hidden_params
|
||||
from litellm.llms.base_llm.chat.transformation import BaseLLMException
|
||||
from litellm.llms.base_llm.image_generation.transformation import (
|
||||
BaseImageGenerationConfig,
|
||||
|
|
@ -226,21 +227,31 @@ class OpenRouterImageGenerationConfig(BaseImageGenerationConfig):
|
|||
|
||||
cost: Final = usage_data.get("cost")
|
||||
if cost is not None:
|
||||
if not hasattr(model_response, "_hidden_params"):
|
||||
model_response._hidden_params = {}
|
||||
if "additional_headers" not in model_response._hidden_params:
|
||||
model_response._hidden_params["additional_headers"] = {}
|
||||
model_response._hidden_params["additional_headers"]["llm_provider-x-litellm-response-cost"] = float(
|
||||
cost
|
||||
if not hasattr(model_response, HIDDEN_PARAMS_ATTR):
|
||||
set_hidden_params(model_response, {})
|
||||
hidden_params: Final = cast( # cast-ok: preserve mapping operations on dynamic response metadata
|
||||
dict[str, object], getattr(model_response, HIDDEN_PARAMS_ATTR)
|
||||
)
|
||||
if "additional_headers" not in hidden_params:
|
||||
hidden_params["additional_headers"] = {}
|
||||
additional_headers: Final = cast( # cast-ok: preserve mapping operations on response metadata
|
||||
dict[str, object], hidden_params["additional_headers"]
|
||||
)
|
||||
additional_headers["llm_provider-x-litellm-response-cost"] = float(cost)
|
||||
|
||||
cost_details: Final = usage_data.get("cost_details", {})
|
||||
if cost_details:
|
||||
if "response_cost_details" not in model_response._hidden_params:
|
||||
model_response._hidden_params["response_cost_details"] = {}
|
||||
model_response._hidden_params["response_cost_details"].update(cost_details)
|
||||
cost_details_hidden_params: Final = cast( # cast-ok: preserve dynamic mapping behavior
|
||||
dict[str, object], getattr(model_response, HIDDEN_PARAMS_ATTR)
|
||||
)
|
||||
if "response_cost_details" not in cost_details_hidden_params:
|
||||
cost_details_hidden_params["response_cost_details"] = {}
|
||||
response_cost_details: Final = cast( # cast-ok: preserve mapping operations on response metadata
|
||||
dict[str, object], cost_details_hidden_params["response_cost_details"]
|
||||
)
|
||||
response_cost_details.update(cost_details)
|
||||
|
||||
model_response._hidden_params["model"] = response_json.get("model", model)
|
||||
model_response.hidden_params["model"] = response_json.get("model", model)
|
||||
|
||||
def get_complete_url(
|
||||
self,
|
||||
|
|
|
|||
|
|
@ -1,7 +1,7 @@
|
|||
import asyncio
|
||||
import json
|
||||
import time
|
||||
from typing import Final
|
||||
from typing import Final, cast
|
||||
|
||||
import httpx
|
||||
|
||||
|
|
@ -18,6 +18,7 @@ from litellm.constants import (
|
|||
OPEN_SANDBOX_POLL_INTERVAL,
|
||||
OPEN_SANDBOX_READY_TIMEOUT,
|
||||
)
|
||||
from litellm.litellm_core_utils.hidden_params import HIDDEN_PARAMS_ATTR, set_hidden_params
|
||||
from litellm.llms.base_llm.sandbox.transformation import (
|
||||
SANDBOX_MAX_OUTPUT_BYTES,
|
||||
BaseSandboxConfig,
|
||||
|
|
@ -115,7 +116,7 @@ class OpenSandboxSandboxConfig(BaseSandboxConfig):
|
|||
)
|
||||
|
||||
handle: Final = ContainerHandle(id=sandbox_id, provider="opensandbox", domain=base)
|
||||
handle._hidden_params = {
|
||||
handle.hidden_params = {
|
||||
"api_base": base,
|
||||
"api_key": key,
|
||||
"execd_endpoint": endpoint,
|
||||
|
|
@ -147,9 +148,12 @@ class OpenSandboxSandboxConfig(BaseSandboxConfig):
|
|||
poll_interval=(float(poll_interval) if poll_interval is not None else DEFAULT_POLL_INTERVAL),
|
||||
client=client,
|
||||
)
|
||||
endpoint: Final = str(handle._hidden_params["execd_endpoint"])
|
||||
endpoint_headers: Final = self._as_str_dict(handle._hidden_params.get("execd_headers"))
|
||||
base: Final = str(handle._hidden_params.get("api_base") or handle.domain or self._api_base(api_base))
|
||||
hidden_params: Final = cast( # cast-ok: preserve mapping operations on the validated handle
|
||||
dict[str, object], getattr(handle, HIDDEN_PARAMS_ATTR)
|
||||
)
|
||||
endpoint: Final = str(hidden_params["execd_endpoint"])
|
||||
endpoint_headers: Final = self._as_str_dict(hidden_params.get("execd_headers"))
|
||||
base: Final = str(hidden_params.get("api_base") or handle.domain or self._api_base(api_base))
|
||||
lines: Final = await self._post_code(
|
||||
url=f"{self._endpoint_base_url(endpoint, base)}/code",
|
||||
headers={
|
||||
|
|
@ -176,7 +180,10 @@ class OpenSandboxSandboxConfig(BaseSandboxConfig):
|
|||
**kwargs,
|
||||
) -> bool:
|
||||
handle: Final = self._as_handle(container, api_base=api_base)
|
||||
base: Final = str(handle._hidden_params.get("api_base") or self._api_base(api_base))
|
||||
hidden_params: Final = cast( # cast-ok: preserve mapping operations on the validated handle
|
||||
dict[str, object], getattr(handle, HIDDEN_PARAMS_ATTR)
|
||||
)
|
||||
base: Final = str(hidden_params.get("api_base") or self._api_base(api_base))
|
||||
key: Final = self._api_key(api_key=api_key, handle=handle)
|
||||
try:
|
||||
response: Final = await self._http(client).delete(
|
||||
|
|
@ -201,12 +208,15 @@ class OpenSandboxSandboxConfig(BaseSandboxConfig):
|
|||
client: AsyncHTTPHandler | None,
|
||||
) -> ContainerHandle:
|
||||
handle: Final = self._as_handle(container, api_base=api_base)
|
||||
if handle._hidden_params.get("execd_endpoint"):
|
||||
hidden_params: Final = cast( # cast-ok: preserve mapping operations on the validated handle
|
||||
dict[str, object], getattr(handle, HIDDEN_PARAMS_ATTR)
|
||||
)
|
||||
if hidden_params.get("execd_endpoint"):
|
||||
return handle
|
||||
|
||||
base: Final = str(handle._hidden_params.get("api_base") or self._api_base(api_base))
|
||||
base: Final = str(hidden_params.get("api_base") or self._api_base(api_base))
|
||||
key: Final = self._api_key(api_key=api_key, handle=handle)
|
||||
resolved_use_server_proxy: Final = bool(handle._hidden_params.get("use_server_proxy", use_server_proxy))
|
||||
resolved_use_server_proxy: Final = bool(hidden_params.get("use_server_proxy", use_server_proxy))
|
||||
endpoint, endpoint_headers = await self._wait_for_execd_endpoint(
|
||||
sandbox_id=handle.id,
|
||||
api_base=base,
|
||||
|
|
@ -217,14 +227,17 @@ class OpenSandboxSandboxConfig(BaseSandboxConfig):
|
|||
poll_interval=poll_interval,
|
||||
)
|
||||
handle.domain = base
|
||||
handle._hidden_params = {
|
||||
**handle._hidden_params,
|
||||
"api_base": base,
|
||||
"api_key": key,
|
||||
"execd_endpoint": endpoint,
|
||||
"execd_headers": endpoint_headers,
|
||||
"use_server_proxy": resolved_use_server_proxy,
|
||||
}
|
||||
set_hidden_params(
|
||||
handle,
|
||||
{
|
||||
**hidden_params,
|
||||
"api_base": base,
|
||||
"api_key": key,
|
||||
"execd_endpoint": endpoint,
|
||||
"execd_headers": endpoint_headers,
|
||||
"use_server_proxy": resolved_use_server_proxy,
|
||||
},
|
||||
)
|
||||
return handle
|
||||
|
||||
async def _wait_until_running(
|
||||
|
|
@ -329,8 +342,11 @@ class OpenSandboxSandboxConfig(BaseSandboxConfig):
|
|||
def _api_key(self, *, api_key: str | None, handle: ContainerHandle) -> str:
|
||||
if api_key is not None:
|
||||
return api_key
|
||||
if "api_key" in handle._hidden_params:
|
||||
return str(handle._hidden_params["api_key"])
|
||||
hidden_params: Final = cast( # cast-ok: preserve mapping operations on the validated handle
|
||||
dict[str, object], getattr(handle, HIDDEN_PARAMS_ATTR)
|
||||
)
|
||||
if "api_key" in hidden_params:
|
||||
return str(hidden_params["api_key"])
|
||||
return self.validate_environment()
|
||||
|
||||
@staticmethod
|
||||
|
|
@ -421,7 +437,7 @@ class OpenSandboxSandboxConfig(BaseSandboxConfig):
|
|||
provider="opensandbox",
|
||||
domain=OpenSandboxSandboxConfig._api_base(api_base),
|
||||
)
|
||||
handle._hidden_params = {}
|
||||
handle.hidden_params = {}
|
||||
return handle
|
||||
|
||||
@staticmethod
|
||||
|
|
|
|||
|
|
@ -163,5 +163,5 @@ class OVHCloudAudioTranscriptionConfig(BaseAudioTranscriptionConfig):
|
|||
if duration is not None:
|
||||
response_json["duration"] = duration
|
||||
|
||||
response._hidden_params = response_json
|
||||
response.hidden_params = response_json
|
||||
return response
|
||||
|
|
|
|||
|
|
@ -260,7 +260,7 @@ class PredibaseConfig(BaseConfig):
|
|||
if k.startswith("x-"):
|
||||
response_headers[f"llm_provider-{k}"] = v
|
||||
|
||||
model_response._hidden_params["additional_headers"] = response_headers
|
||||
model_response.hidden_params["additional_headers"] = response_headers
|
||||
|
||||
return model_response
|
||||
|
||||
|
|
|
|||
|
|
@ -147,5 +147,5 @@ class ScalewayAudioTranscriptionConfig(BaseAudioTranscriptionConfig):
|
|||
if "language" in response_json:
|
||||
response["language"] = response_json["language"]
|
||||
|
||||
response._hidden_params = response_json
|
||||
response.hidden_params = response_json
|
||||
return response
|
||||
|
|
|
|||
|
|
@ -558,7 +558,7 @@ class SnowflakeConfig(SnowflakeBaseConfig, OpenAIGPTConfig):
|
|||
returned_response.model = "snowflake/" + (returned_response.model or "")
|
||||
|
||||
if model is not None:
|
||||
returned_response._hidden_params["model"] = model
|
||||
returned_response.hidden_params["model"] = model
|
||||
|
||||
return returned_response
|
||||
|
||||
|
|
@ -623,7 +623,7 @@ class SnowflakeConfig(SnowflakeBaseConfig, OpenAIGPTConfig):
|
|||
model_response.id = response_json.get("id", "")
|
||||
|
||||
if model is not None:
|
||||
model_response._hidden_params["model"] = model
|
||||
model_response.hidden_params["model"] = model
|
||||
|
||||
return model_response
|
||||
|
||||
|
|
|
|||
|
|
@ -62,7 +62,7 @@ class SnowflakeEmbeddingConfig(SnowflakeBaseConfig, BaseEmbeddingConfig):
|
|||
returned_response.model = "snowflake/" + (returned_response.model or "")
|
||||
|
||||
if model is not None:
|
||||
returned_response._hidden_params["model"] = model
|
||||
returned_response.hidden_params["model"] = model
|
||||
return returned_response
|
||||
|
||||
def get_error_class(self, error_message: str, status_code: int, headers: dict | httpx.Headers) -> BaseLLMException:
|
||||
|
|
|
|||
|
|
@ -458,7 +458,7 @@ class SonioxAudioTranscriptionHandler:
|
|||
self._safe_log_post_call(logging_obj, audio_file, api_key, body, payload)
|
||||
|
||||
audio_duration_ms: Final = transcription_meta.get("audio_duration_ms")
|
||||
response._hidden_params.update(
|
||||
response.hidden_params.update(
|
||||
{
|
||||
"model": model,
|
||||
"custom_llm_provider": "soniox",
|
||||
|
|
@ -692,7 +692,7 @@ class SonioxAudioTranscriptionHandler:
|
|||
self._safe_log_post_call(logging_obj, audio_file, api_key, body, payload)
|
||||
|
||||
audio_duration_ms: Final = transcription_meta.get("audio_duration_ms")
|
||||
response._hidden_params.update(
|
||||
response.hidden_params.update(
|
||||
{
|
||||
"model": model,
|
||||
"custom_llm_provider": "soniox",
|
||||
|
|
|
|||
|
|
@ -260,7 +260,7 @@ class SonioxAudioTranscriptionConfig(BaseAudioTranscriptionConfig):
|
|||
|
||||
# Stash the raw Soniox payload so power-users can read tokens, segments,
|
||||
# speaker/language data, etc.
|
||||
response._hidden_params.update(
|
||||
response.hidden_params.update(
|
||||
{
|
||||
"soniox_raw": {
|
||||
"transcription": transcription_meta,
|
||||
|
|
|
|||
|
|
@ -6,7 +6,7 @@ Handles transformation between OpenAI-compatible format and Stability AI API for
|
|||
API Reference: https://platform.stability.ai/docs/api-reference
|
||||
"""
|
||||
|
||||
from typing import TYPE_CHECKING, Any, Final
|
||||
from typing import TYPE_CHECKING, Any, Final, cast
|
||||
|
||||
import httpx
|
||||
from httpx._types import RequestFiles
|
||||
|
|
@ -300,16 +300,15 @@ class StabilityImageEditConfig(BaseImageEditConfig):
|
|||
)
|
||||
|
||||
if not hasattr(model_response, "_hidden_params"):
|
||||
model_response._hidden_params = {}
|
||||
if "additional_headers" not in model_response._hidden_params:
|
||||
model_response._hidden_params["additional_headers"] = {}
|
||||
model_response.hidden_params = {}
|
||||
additional_headers: Final = cast( # cast-ok: provider headers are stored as a mutable mapping
|
||||
dict[str, object], model_response.hidden_params.setdefault("additional_headers", {})
|
||||
)
|
||||
# Override: fetch model-cost from model_cost map based on the provided model name
|
||||
model_info: Final = get_model_info(model, custom_llm_provider="stability")
|
||||
cost_per_image: Final = model_info.get("output_cost_per_image", 0)
|
||||
if cost_per_image is not None:
|
||||
model_response._hidden_params["additional_headers"]["llm_provider-x-litellm-response-cost"] = float(
|
||||
cost_per_image
|
||||
)
|
||||
additional_headers["llm_provider-x-litellm-response-cost"] = float(cost_per_image)
|
||||
return model_response
|
||||
|
||||
def use_multipart_form_data(self) -> bool:
|
||||
|
|
|
|||
|
|
@ -187,7 +187,7 @@ class VertexAIAudioTranscriptionConfig(BaseAudioTranscriptionConfig, VertexBase)
|
|||
billed_duration = _parse_duration_seconds(parsed.metadata.totalBilledDuration if parsed.metadata else None)
|
||||
if billed_duration is not None:
|
||||
response["duration"] = billed_duration
|
||||
response._hidden_params = response_json
|
||||
response.hidden_params = response_json
|
||||
return response
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -2100,10 +2100,10 @@ class VertexGeminiConfig(VertexAIBaseConfig, BaseConfig):
|
|||
) -> None:
|
||||
setattr(model_response, "vertex_ai_grounding_metadata", grounding_metadata)
|
||||
if grounding_metadata:
|
||||
model_response._hidden_params["vertex_ai_grounding_metadata"] = grounding_metadata
|
||||
model_response.hidden_params["vertex_ai_grounding_metadata"] = grounding_metadata
|
||||
setattr(model_response, "vertex_ai_url_context_metadata", url_context_metadata)
|
||||
if url_context_metadata:
|
||||
model_response._hidden_params["vertex_ai_url_context_metadata"] = url_context_metadata
|
||||
model_response.hidden_params["vertex_ai_url_context_metadata"] = url_context_metadata
|
||||
setattr(model_response, "vertex_ai_safety_ratings", safety_ratings)
|
||||
setattr(model_response, "vertex_ai_safety_results", safety_ratings)
|
||||
if safety_ratings:
|
||||
|
|
@ -2130,7 +2130,7 @@ class VertexGeminiConfig(VertexAIBaseConfig, BaseConfig):
|
|||
merged.append(value)
|
||||
if merged:
|
||||
setattr(response, field_name, merged)
|
||||
response._hidden_params[field_name] = merged
|
||||
response.hidden_params[field_name] = merged
|
||||
|
||||
@staticmethod
|
||||
def _convert_grounding_metadata_to_annotations(
|
||||
|
|
@ -2472,27 +2472,27 @@ class VertexGeminiConfig(VertexAIBaseConfig, BaseConfig):
|
|||
## ADD METADATA TO RESPONSE ##
|
||||
|
||||
setattr(model_response, "vertex_ai_grounding_metadata", grounding_metadata)
|
||||
model_response._hidden_params["vertex_ai_grounding_metadata"] = grounding_metadata
|
||||
model_response.hidden_params["vertex_ai_grounding_metadata"] = grounding_metadata
|
||||
|
||||
setattr(model_response, "vertex_ai_url_context_metadata", url_context_metadata)
|
||||
|
||||
model_response._hidden_params["vertex_ai_url_context_metadata"] = url_context_metadata
|
||||
model_response.hidden_params["vertex_ai_url_context_metadata"] = url_context_metadata
|
||||
|
||||
setattr(model_response, "vertex_ai_safety_results", safety_ratings)
|
||||
model_response._hidden_params["vertex_ai_safety_results"] = (
|
||||
model_response.hidden_params["vertex_ai_safety_results"] = (
|
||||
safety_ratings # older approach - maintaining to prevent regressions
|
||||
)
|
||||
|
||||
## ADD CITATION METADATA ##
|
||||
setattr(model_response, "vertex_ai_citation_metadata", citation_metadata)
|
||||
model_response._hidden_params["vertex_ai_citation_metadata"] = (
|
||||
model_response.hidden_params["vertex_ai_citation_metadata"] = (
|
||||
citation_metadata # older approach - maintaining to prevent regressions
|
||||
)
|
||||
|
||||
## ADD TRAFFIC TYPE ##
|
||||
traffic_type: Final = completion_response.get("usageMetadata", {}).get("trafficType")
|
||||
if traffic_type:
|
||||
model_response._hidden_params.setdefault("provider_specific_fields", {})["traffic_type"] = traffic_type
|
||||
model_response.hidden_params.setdefault("provider_specific_fields", {})["traffic_type"] = traffic_type
|
||||
|
||||
## ADD SERVICE TIER ##
|
||||
if getattr(raw_response, "headers", None):
|
||||
|
|
@ -3223,7 +3223,7 @@ class ModelResponseIterator:
|
|||
|
||||
traffic_type: Final = processed_chunk.get("usageMetadata", {}).get("trafficType")
|
||||
if traffic_type:
|
||||
model_response._hidden_params.setdefault("provider_specific_fields", {})["traffic_type"] = traffic_type
|
||||
model_response.hidden_params.setdefault("provider_specific_fields", {})["traffic_type"] = traffic_type
|
||||
|
||||
service_tier: Final = self.response_headers.get("x-gemini-service-tier")
|
||||
if service_tier:
|
||||
|
|
@ -3280,7 +3280,7 @@ class ModelResponseIterator:
|
|||
|
||||
setattr(model_response, "usage", usage)
|
||||
|
||||
model_response._hidden_params["is_finished"] = False
|
||||
model_response.hidden_params["is_finished"] = False
|
||||
return model_response
|
||||
|
||||
except json.JSONDecodeError:
|
||||
|
|
|
|||
|
|
@ -259,8 +259,8 @@ class VolcEngineResponsesAPIConfig(OpenAIResponsesAPIConfig):
|
|||
construct_response: Final[Callable[..., ResponsesAPIResponse]] = ResponsesAPIResponse.model_construct
|
||||
response = construct_response(**raw_response_json)
|
||||
|
||||
response._hidden_params["additional_headers"] = processed_headers
|
||||
response._hidden_params["headers"] = raw_response_headers
|
||||
response.hidden_params["additional_headers"] = processed_headers
|
||||
response.hidden_params["headers"] = raw_response_headers
|
||||
return response
|
||||
|
||||
#########################################################
|
||||
|
|
@ -325,8 +325,8 @@ class VolcEngineResponsesAPIConfig(OpenAIResponsesAPIConfig):
|
|||
processed_headers: Final = process_response_headers(raw_response_headers)
|
||||
|
||||
response: Final = ResponsesAPIResponse.model_validate(raw_response_json)
|
||||
response._hidden_params["additional_headers"] = processed_headers
|
||||
response._hidden_params["headers"] = raw_response_headers
|
||||
response.hidden_params["additional_headers"] = processed_headers
|
||||
response.hidden_params["headers"] = raw_response_headers
|
||||
return response
|
||||
|
||||
#########################################################
|
||||
|
|
@ -398,8 +398,8 @@ class VolcEngineResponsesAPIConfig(OpenAIResponsesAPIConfig):
|
|||
processed_headers: Final = process_response_headers(raw_response_headers)
|
||||
|
||||
response: Final = ResponsesAPIResponse.model_validate(raw_response_json)
|
||||
response._hidden_params["additional_headers"] = processed_headers
|
||||
response._hidden_params["headers"] = raw_response_headers
|
||||
response.hidden_params["additional_headers"] = processed_headers
|
||||
response.hidden_params["headers"] = raw_response_headers
|
||||
return response
|
||||
|
||||
def should_fake_stream(
|
||||
|
|
|
|||
|
|
@ -96,6 +96,7 @@ from litellm.litellm_core_utils.health_check_utils import (
|
|||
_filter_model_params,
|
||||
create_health_check_response,
|
||||
)
|
||||
from litellm.litellm_core_utils.hidden_params import HIDDEN_PARAMS_ATTR, get_hidden_params
|
||||
from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj
|
||||
from litellm.litellm_core_utils.mock_functions import (
|
||||
mock_embedding,
|
||||
|
|
@ -1036,11 +1037,11 @@ def mock_completion(
|
|||
)
|
||||
|
||||
if custom_llm_provider is not None:
|
||||
model_response._hidden_params["custom_llm_provider"] = custom_llm_provider
|
||||
model_response.hidden_params["custom_llm_provider"] = custom_llm_provider
|
||||
else:
|
||||
try:
|
||||
_, inferred_provider, _, _ = litellm.utils.get_llm_provider(model=model)
|
||||
model_response._hidden_params["custom_llm_provider"] = inferred_provider
|
||||
model_response.hidden_params["custom_llm_provider"] = inferred_provider
|
||||
except Exception:
|
||||
# dont let setting a hidden param block a mock_respose
|
||||
pass
|
||||
|
|
@ -5516,9 +5517,12 @@ def completion(
|
|||
)
|
||||
)
|
||||
|
||||
if model_response is not None and hasattr(model_response, "_hidden_params"):
|
||||
model_response._hidden_params["custom_llm_provider"] = custom_llm_provider
|
||||
model_response._hidden_params["region_name"] = kwargs.get(
|
||||
if model_response is not None and hasattr(model_response, HIDDEN_PARAMS_ATTR):
|
||||
model_response_hidden_params: Final = cast( # cast-ok: preserve dynamic mapping behavior
|
||||
dict[str, object], getattr(model_response, HIDDEN_PARAMS_ATTR)
|
||||
)
|
||||
model_response_hidden_params["custom_llm_provider"] = custom_llm_provider
|
||||
model_response_hidden_params["region_name"] = kwargs.get(
|
||||
"aws_region_name", None
|
||||
) # support region-based pricing for bedrock
|
||||
|
||||
|
|
@ -6247,7 +6251,9 @@ async def aembedding(*args, **kwargs) -> EmbeddingResponse:
|
|||
elif asyncio.iscoroutine(init_response):
|
||||
response = await init_response
|
||||
if response is not None and isinstance(response, EmbeddingResponse) and hasattr(response, "_hidden_params"):
|
||||
response._hidden_params["custom_llm_provider"] = custom_llm_provider
|
||||
response_hidden_params: Final = get_hidden_params(response)
|
||||
if response_hidden_params is not None:
|
||||
response_hidden_params["custom_llm_provider"] = custom_llm_provider
|
||||
|
||||
if response is None:
|
||||
raise ValueError("Unable to get Embedding Response. Please pass a valid llm_provider.")
|
||||
|
|
@ -7364,7 +7370,9 @@ def embedding(
|
|||
else:
|
||||
raise LiteLLMUnknownProvider(model=model, custom_llm_provider=custom_llm_provider)
|
||||
if response is not None and hasattr(response, "_hidden_params") and isinstance(response, EmbeddingResponse):
|
||||
response._hidden_params["custom_llm_provider"] = custom_llm_provider
|
||||
response_hidden_params: Final = get_hidden_params(response)
|
||||
if response_hidden_params is not None:
|
||||
response_hidden_params["custom_llm_provider"] = custom_llm_provider
|
||||
|
||||
if response is None:
|
||||
raise LiteLLMUnknownProvider(model=model, custom_llm_provider=custom_llm_provider)
|
||||
|
|
@ -7929,7 +7937,7 @@ async def atranscription(*args, **kwargs) -> TranscriptionResponse:
|
|||
if existing_duration is None:
|
||||
calculated_duration: Final = calculate_request_duration(file)
|
||||
if calculated_duration is not None:
|
||||
response._hidden_params["audio_transcription_duration"] = calculated_duration
|
||||
response.hidden_params["audio_transcription_duration"] = calculated_duration
|
||||
|
||||
return response
|
||||
except Exception as e:
|
||||
|
|
@ -8202,7 +8210,10 @@ def transcription(
|
|||
if existing_duration is None:
|
||||
calculated_duration: Final = calculate_request_duration(file)
|
||||
if calculated_duration is not None:
|
||||
response._hidden_params["audio_transcription_duration"] = calculated_duration
|
||||
response_hidden_params: Final = cast( # cast-ok: preserve dynamic mapping behavior
|
||||
dict[str, object], getattr(response, HIDDEN_PARAMS_ATTR)
|
||||
)
|
||||
response_hidden_params["audio_transcription_duration"] = calculated_duration
|
||||
|
||||
if response is None:
|
||||
raise ValueError("Unmapped provider passed in. Unable to get the response.")
|
||||
|
|
@ -8816,8 +8827,9 @@ async def ahealth_check(
|
|||
|
||||
if mode in mode_handlers:
|
||||
_response: Final = await mode_handlers[mode]()
|
||||
# Only process headers for chat mode
|
||||
_response_headers: Final[dict] = getattr(_response, "_hidden_params", {}).get("headers", {}) or {}
|
||||
_response_headers: Final = cast( # cast-ok: provider headers are stored as a string-keyed mapping
|
||||
Mapping[str, object], (get_hidden_params(_response) or {}).get("headers", {}) or {}
|
||||
)
|
||||
return create_health_check_response(_response_headers)
|
||||
else:
|
||||
raise Exception(f"Mode {mode} not supported. See modes here: https://docs.litellm.ai/docs/proxy/health")
|
||||
|
|
@ -9076,9 +9088,9 @@ def stream_chunk_builder(
|
|||
if isinstance(chunk, dict):
|
||||
hidden = chunk.get("_hidden_params")
|
||||
else:
|
||||
hidden = getattr(chunk, "_hidden_params", None)
|
||||
hidden = getattr(chunk, HIDDEN_PARAMS_ATTR, None)
|
||||
if isinstance(hidden, dict) and "provider_specific_fields" in hidden:
|
||||
response._hidden_params.setdefault("provider_specific_fields", {}).update(
|
||||
response.hidden_params.setdefault("provider_specific_fields", {}).update(
|
||||
hidden["provider_specific_fields"]
|
||||
)
|
||||
break
|
||||
|
|
@ -9255,9 +9267,9 @@ def stream_chunk_builder(
|
|||
if isinstance(chunk, dict):
|
||||
hidden = chunk.get("_hidden_params")
|
||||
else:
|
||||
hidden = getattr(chunk, "_hidden_params", None)
|
||||
hidden = getattr(chunk, HIDDEN_PARAMS_ATTR, None)
|
||||
if isinstance(hidden, dict) and "provider_specific_fields" in hidden:
|
||||
response._hidden_params.setdefault("provider_specific_fields", {}).update(
|
||||
response.hidden_params.setdefault("provider_specific_fields", {}).update(
|
||||
hidden["provider_specific_fields"]
|
||||
)
|
||||
break
|
||||
|
|
|
|||
|
|
@ -17,6 +17,7 @@ from pydantic import TypeAdapter
|
|||
import litellm
|
||||
from litellm._logging import verbose_proxy_logger
|
||||
from litellm.batches.main import CancelBatchRequest, RetrieveBatchRequest
|
||||
from litellm.litellm_core_utils.hidden_params import get_hidden_params, set_hidden_param
|
||||
from litellm.proxy._types import *
|
||||
from litellm.proxy.auth.user_api_key_auth import user_api_key_auth
|
||||
from litellm.proxy.batches_endpoints.common_utils import validate_batch_list_limit
|
||||
|
|
@ -79,6 +80,11 @@ router: Final = APIRouter()
|
|||
_METADATA_ADAPTER: Final[TypeAdapter[Mapping[str, object]]] = TypeAdapter(Mapping[str, object])
|
||||
|
||||
|
||||
def _hidden_param_string(hidden_params: Mapping[str, object], key: str) -> str:
|
||||
value: Final = hidden_params.get(key)
|
||||
return value if isinstance(value, str) else ""
|
||||
|
||||
|
||||
def _request_tags(data: Mapping[str, object]) -> tuple[str, ...] | None:
|
||||
metadata: Final = data.get("litellm_metadata")
|
||||
if metadata is None:
|
||||
|
|
@ -211,7 +217,7 @@ async def _create_provider_batch_for_managed_file(
|
|||
}
|
||||
response: Final = await llm_router.acreate_batch(**request)
|
||||
response.input_file_id = input_file_id
|
||||
response._hidden_params["unified_file_id"] = unified_file_id
|
||||
set_hidden_param(response, "unified_file_id", unified_file_id)
|
||||
return response
|
||||
|
||||
|
||||
|
|
@ -484,7 +490,7 @@ async def create_batch(
|
|||
**_create_batch_data,
|
||||
)
|
||||
|
||||
response._hidden_params[BATCH_CREATE_HIDDEN_PARAM] = True
|
||||
set_hidden_param(response, BATCH_CREATE_HIDDEN_PARAM, True)
|
||||
|
||||
### CALL HOOKS ### - modify outgoing data
|
||||
response = await proxy_logging_obj.post_call_success_hook(
|
||||
|
|
@ -497,10 +503,10 @@ async def create_batch(
|
|||
)
|
||||
|
||||
### RESPONSE HEADERS ###
|
||||
hidden_params: Final = getattr(response, "_hidden_params", {}) or {}
|
||||
model_id: Final = hidden_params.get("model_id", None) or ""
|
||||
cache_key: Final = hidden_params.get("cache_key", None) or ""
|
||||
api_base: Final = hidden_params.get("api_base", None) or ""
|
||||
hidden_params: Final = get_hidden_params(response) or {}
|
||||
model_id: Final = _hidden_param_string(hidden_params, "model_id")
|
||||
cache_key: Final = _hidden_param_string(hidden_params, "cache_key")
|
||||
api_base: Final = _hidden_param_string(hidden_params, "api_base")
|
||||
|
||||
fastapi_response.headers.update(
|
||||
ProxyBaseLLMRequestProcessing.get_custom_headers(
|
||||
|
|
@ -655,10 +661,10 @@ async def retrieve_batch(
|
|||
)
|
||||
)
|
||||
|
||||
hidden_params = getattr(response, "_hidden_params", {}) or {}
|
||||
model_id = hidden_params.get("model_id", None) or ""
|
||||
cache_key = hidden_params.get("cache_key", None) or ""
|
||||
api_base = hidden_params.get("api_base", None) or ""
|
||||
hidden_params = get_hidden_params(response) or {}
|
||||
model_id = _hidden_param_string(hidden_params, "model_id")
|
||||
cache_key = _hidden_param_string(hidden_params, "cache_key")
|
||||
api_base = _hidden_param_string(hidden_params, "api_base")
|
||||
|
||||
fastapi_response.headers.update(
|
||||
ProxyBaseLLMRequestProcessing.get_custom_headers(
|
||||
|
|
@ -736,11 +742,11 @@ async def retrieve_batch(
|
|||
)
|
||||
|
||||
response = await llm_router.aretrieve_batch(**data)
|
||||
response._hidden_params["unified_batch_id"] = unified_batch_id
|
||||
set_hidden_param(response, "unified_batch_id", unified_batch_id)
|
||||
if unified_batch_id:
|
||||
model_id_from_batch: Final = get_model_id_from_unified_batch_id(unified_batch_id)
|
||||
if model_id_from_batch:
|
||||
response._hidden_params["model_id"] = model_id_from_batch
|
||||
set_hidden_param(response, "model_id", model_id_from_batch)
|
||||
|
||||
# SCENARIO 3: Fallback to custom_llm_provider (uses env variables)
|
||||
else:
|
||||
|
|
@ -802,10 +808,10 @@ async def retrieve_batch(
|
|||
)
|
||||
|
||||
### RESPONSE HEADERS ###
|
||||
hidden_params = getattr(response, "_hidden_params", {}) or {}
|
||||
model_id = hidden_params.get("model_id", None) or ""
|
||||
cache_key = hidden_params.get("cache_key", None) or ""
|
||||
api_base = hidden_params.get("api_base", None) or ""
|
||||
hidden_params = get_hidden_params(response) or {}
|
||||
model_id = _hidden_param_string(hidden_params, "model_id")
|
||||
cache_key = _hidden_param_string(hidden_params, "cache_key")
|
||||
api_base = _hidden_param_string(hidden_params, "api_base")
|
||||
|
||||
fastapi_response.headers.update(
|
||||
ProxyBaseLLMRequestProcessing.get_custom_headers(
|
||||
|
|
@ -986,10 +992,10 @@ async def list_batches(
|
|||
response = _response
|
||||
|
||||
### RESPONSE HEADERS ###
|
||||
hidden_params: Final = getattr(response, "_hidden_params", {}) or {}
|
||||
model_id: Final = hidden_params.get("model_id", None) or ""
|
||||
cache_key: Final = hidden_params.get("cache_key", None) or ""
|
||||
api_base: Final = hidden_params.get("api_base", None) or ""
|
||||
hidden_params: Final = get_hidden_params(response) or {}
|
||||
model_id: Final = _hidden_param_string(hidden_params, "model_id")
|
||||
cache_key: Final = _hidden_param_string(hidden_params, "cache_key")
|
||||
api_base: Final = _hidden_param_string(hidden_params, "api_base")
|
||||
|
||||
fastapi_response.headers.update(
|
||||
ProxyBaseLLMRequestProcessing.get_custom_headers(
|
||||
|
|
@ -1168,10 +1174,10 @@ async def cancel_batch(
|
|||
data["model"] = model_id_from_batch
|
||||
data["batch_id"] = get_batch_id_from_unified_batch_id(unified_batch_id)
|
||||
response = await llm_router.acancel_batch(**data)
|
||||
response._hidden_params["unified_batch_id"] = unified_batch_id
|
||||
set_hidden_param(response, "unified_batch_id", unified_batch_id)
|
||||
|
||||
if not response._hidden_params.get("model_id") and data.get("model"):
|
||||
response._hidden_params["model_id"] = data["model"]
|
||||
if not (get_hidden_params(response) or {}).get("model_id") and data.get("model"):
|
||||
set_hidden_param(response, "model_id", data["model"])
|
||||
|
||||
# SCENARIO 3: Fallback to custom_llm_provider (uses env variables)
|
||||
else:
|
||||
|
|
@ -1229,10 +1235,10 @@ async def cancel_batch(
|
|||
)
|
||||
|
||||
### RESPONSE HEADERS ###
|
||||
hidden_params: Final = getattr(response, "_hidden_params", {}) or {}
|
||||
model_id: Final = hidden_params.get("model_id", None) or ""
|
||||
cache_key: Final = hidden_params.get("cache_key", None) or ""
|
||||
api_base: Final = hidden_params.get("api_base", None) or ""
|
||||
hidden_params: Final = get_hidden_params(response) or {}
|
||||
model_id: Final = _hidden_param_string(hidden_params, "model_id")
|
||||
cache_key: Final = _hidden_param_string(hidden_params, "cache_key")
|
||||
api_base: Final = _hidden_param_string(hidden_params, "api_base")
|
||||
|
||||
fastapi_response.headers.update(
|
||||
ProxyBaseLLMRequestProcessing.get_custom_headers(
|
||||
|
|
|
|||
|
|
@ -12,6 +12,7 @@ from fastapi import APIRouter, Depends, HTTPException, Query, Request, Response
|
|||
|
||||
import litellm
|
||||
from litellm._logging import verbose_proxy_logger
|
||||
from litellm.litellm_core_utils.hidden_params import set_hidden_param
|
||||
from litellm.proxy._types import *
|
||||
from litellm.proxy.auth.user_api_key_auth import user_api_key_auth
|
||||
from litellm.proxy.common_request_processing import ProxyBaseLLMRequestProcessing
|
||||
|
|
@ -161,7 +162,7 @@ async def create_fine_tuning_job(
|
|||
|
||||
response = cast(LiteLLMFineTuningJob, await llm_router.acreate_fine_tuning_job(**data))
|
||||
response.training_file = unified_file_id
|
||||
response._hidden_params["unified_file_id"] = unified_file_id
|
||||
set_hidden_param(response, "unified_file_id", unified_file_id)
|
||||
## ELSE, Route based on custom_llm_provider
|
||||
elif fine_tuning_request.custom_llm_provider:
|
||||
# get configs for custom_llm_provider
|
||||
|
|
@ -304,7 +305,7 @@ async def retrieve_fine_tuning_job(
|
|||
**data,
|
||||
),
|
||||
)
|
||||
response._hidden_params["unified_finetuning_job_id"] = unified_finetuning_job_id
|
||||
set_hidden_param(response, "unified_finetuning_job_id", unified_finetuning_job_id)
|
||||
elif custom_llm_provider:
|
||||
# get configs for custom_llm_provider
|
||||
llm_provider_config: Final = get_fine_tuning_provider_config(custom_llm_provider=custom_llm_provider)
|
||||
|
|
@ -577,7 +578,7 @@ async def cancel_fine_tuning_job(
|
|||
**data,
|
||||
),
|
||||
)
|
||||
response._hidden_params["unified_finetuning_job_id"] = unified_finetuning_job_id
|
||||
set_hidden_param(response, "unified_finetuning_job_id", unified_finetuning_job_id)
|
||||
else:
|
||||
# get configs for custom_llm_provider
|
||||
llm_provider_config: Final = get_fine_tuning_provider_config(custom_llm_provider=custom_llm_provider)
|
||||
|
|
|
|||
|
|
@ -15,6 +15,7 @@ from litellm._logging import verbose_proxy_logger
|
|||
from litellm.caching.caching import DualCache
|
||||
from litellm.exceptions import RateLimitType
|
||||
from litellm.integrations.custom_logger import CustomLogger
|
||||
from litellm.litellm_core_utils.hidden_params import set_hidden_param
|
||||
from litellm.proxy._types import UserAPIKeyAuth
|
||||
from litellm.proxy.common_utils.proxy_rate_limit_error import ProxyRateLimitError
|
||||
from litellm.proxy.hooks.rate_limiter_utils import (
|
||||
|
|
@ -243,10 +244,11 @@ class _PROXY_DynamicRateLimitHandler(CustomLogger):
|
|||
async def async_post_call_success_hook(self, data: dict, user_api_key_dict: UserAPIKeyAuth, response):
|
||||
try:
|
||||
if isinstance(response, ModelResponse):
|
||||
model_info: Final = self.llm_router.get_model_info(id=response._hidden_params["model_id"])
|
||||
assert model_info is not None, "Model info for model with id={} is None".format(
|
||||
response._hidden_params["model_id"]
|
||||
)
|
||||
model_id: Final = response.hidden_params["model_id"]
|
||||
if not isinstance(model_id, str):
|
||||
return response
|
||||
model_info: Final = self.llm_router.get_model_info(id=model_id)
|
||||
assert model_info is not None, f"Model info for model with id={model_id} is None"
|
||||
key_priority: Final[str | None] = user_api_key_dict.metadata.get("priority", None)
|
||||
(
|
||||
available_tpm,
|
||||
|
|
@ -255,14 +257,18 @@ class _PROXY_DynamicRateLimitHandler(CustomLogger):
|
|||
model_rpm,
|
||||
active_projects,
|
||||
) = await self.check_available_usage(model=model_info["model_name"], priority=key_priority)
|
||||
response._hidden_params["additional_headers"] = { # Add additional response headers - easier debugging
|
||||
"x-litellm-model_group": model_info["model_name"],
|
||||
"x-ratelimit-remaining-litellm-project-tokens": available_tpm,
|
||||
"x-ratelimit-remaining-litellm-project-requests": available_rpm,
|
||||
"x-ratelimit-remaining-model-tokens": model_tpm,
|
||||
"x-ratelimit-remaining-model-requests": model_rpm,
|
||||
"x-ratelimit-current-active-projects": active_projects,
|
||||
}
|
||||
set_hidden_param(
|
||||
response,
|
||||
"additional_headers",
|
||||
{
|
||||
"x-litellm-model_group": model_info["model_name"],
|
||||
"x-ratelimit-remaining-litellm-project-tokens": available_tpm,
|
||||
"x-ratelimit-remaining-litellm-project-requests": available_rpm,
|
||||
"x-ratelimit-remaining-model-tokens": model_tpm,
|
||||
"x-ratelimit-remaining-model-requests": model_rpm,
|
||||
"x-ratelimit-current-active-projects": active_projects,
|
||||
},
|
||||
)
|
||||
|
||||
return response
|
||||
return await super().async_post_call_success_hook(
|
||||
|
|
|
|||
|
|
@ -12,6 +12,7 @@ from typing import Any, Final, cast
|
|||
|
||||
from litellm._logging import verbose_proxy_logger
|
||||
from litellm._uuid import uuid as uuid_module
|
||||
from litellm.litellm_core_utils.hidden_params import get_or_create_hidden_params
|
||||
from litellm.llms.base_llm.files.storage_backend import BaseFileStorageBackend
|
||||
from litellm.llms.base_llm.files.storage_backend_factory import get_storage_backend
|
||||
from litellm.llms.base_llm.files.transformation import BaseFileEndpoints
|
||||
|
|
@ -170,9 +171,8 @@ class StorageBackendFileService:
|
|||
)
|
||||
|
||||
# Store storage metadata in hidden params
|
||||
if not hasattr(file_object, "_hidden_params") or file_object._hidden_params is None:
|
||||
file_object._hidden_params = {}
|
||||
file_object._hidden_params.update(
|
||||
file_object_hidden_params: Final = get_or_create_hidden_params(file_object)
|
||||
file_object_hidden_params.update(
|
||||
{
|
||||
"storage_backend": target_storage,
|
||||
"storage_url": storage_url,
|
||||
|
|
|
|||
|
|
@ -5,6 +5,7 @@ import httpx
|
|||
|
||||
import litellm
|
||||
from litellm import stream_chunk_builder
|
||||
from litellm.litellm_core_utils.hidden_params import set_hidden_param
|
||||
from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj
|
||||
from litellm.litellm_core_utils.litellm_logging import (
|
||||
get_standard_logging_object_payload,
|
||||
|
|
@ -114,9 +115,7 @@ class CoherePassthroughLoggingHandler(BasePassthroughLoggingHandler):
|
|||
)
|
||||
|
||||
# Set the calculated cost in _hidden_params to prevent recalculation
|
||||
if not hasattr(litellm_model_response, "_hidden_params"):
|
||||
litellm_model_response._hidden_params = {}
|
||||
litellm_model_response._hidden_params["response_cost"] = response_cost
|
||||
set_hidden_param(litellm_model_response, "response_cost", response_cost)
|
||||
|
||||
kwargs["response_cost"] = response_cost
|
||||
kwargs["model"] = model
|
||||
|
|
|
|||
|
|
@ -81,8 +81,8 @@ class GeminiPassthroughLoggingHandler:
|
|||
|
||||
# Set response_cost in _hidden_params to prevent recalculation
|
||||
if not hasattr(litellm_video_response, "_hidden_params"):
|
||||
litellm_video_response._hidden_params = {}
|
||||
litellm_video_response._hidden_params["response_cost"] = response_cost
|
||||
litellm_video_response.hidden_params = {}
|
||||
litellm_video_response.hidden_params["response_cost"] = response_cost
|
||||
|
||||
kwargs["response_cost"] = response_cost
|
||||
kwargs["model"] = model
|
||||
|
|
|
|||
|
|
@ -13,6 +13,7 @@ import httpx
|
|||
|
||||
import litellm
|
||||
from litellm._logging import verbose_proxy_logger
|
||||
from litellm.litellm_core_utils.hidden_params import set_hidden_param
|
||||
from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj
|
||||
from litellm.litellm_core_utils.litellm_logging import (
|
||||
get_standard_logging_object_payload,
|
||||
|
|
@ -414,7 +415,7 @@ class OpenAIPassthroughLoggingHandler(BasePassthroughLoggingHandler):
|
|||
model=model,
|
||||
custom_llm_provider=custom_llm_provider,
|
||||
)
|
||||
litellm_model_response._hidden_params["response_cost"] = response_cost
|
||||
set_hidden_param(litellm_model_response, "response_cost", response_cost)
|
||||
elif is_image_generation:
|
||||
# Handle image generation cost calculation
|
||||
response_cost = OpenAIPassthroughLoggingHandler._calculate_image_generation_cost(
|
||||
|
|
@ -433,9 +434,7 @@ class OpenAIPassthroughLoggingHandler(BasePassthroughLoggingHandler):
|
|||
model=model,
|
||||
)
|
||||
# Set the calculated cost in _hidden_params to prevent recalculation
|
||||
if not hasattr(litellm_model_response, "_hidden_params"):
|
||||
litellm_model_response._hidden_params = {}
|
||||
litellm_model_response._hidden_params["response_cost"] = response_cost
|
||||
set_hidden_param(litellm_model_response, "response_cost", response_cost)
|
||||
elif is_image_editing:
|
||||
# Handle image editing cost calculation
|
||||
response_cost = OpenAIPassthroughLoggingHandler._calculate_image_editing_cost(
|
||||
|
|
@ -454,9 +453,7 @@ class OpenAIPassthroughLoggingHandler(BasePassthroughLoggingHandler):
|
|||
model=model,
|
||||
)
|
||||
# Set the calculated cost in _hidden_params to prevent recalculation
|
||||
if not hasattr(litellm_model_response, "_hidden_params"):
|
||||
litellm_model_response._hidden_params = {}
|
||||
litellm_model_response._hidden_params["response_cost"] = response_cost
|
||||
set_hidden_param(litellm_model_response, "response_cost", response_cost)
|
||||
elif is_responses:
|
||||
# Responses-API cost tracking — see
|
||||
# `_build_responses_api_response_and_cost` for why this needs
|
||||
|
|
|
|||
|
|
@ -175,8 +175,8 @@ class VertexPassthroughLoggingHandler:
|
|||
|
||||
# Set response_cost in _hidden_params to prevent recalculation
|
||||
if not hasattr(litellm_video_response, "_hidden_params"):
|
||||
litellm_video_response._hidden_params = {}
|
||||
litellm_video_response._hidden_params["response_cost"] = response_cost
|
||||
litellm_video_response.hidden_params = {}
|
||||
litellm_video_response.hidden_params["response_cost"] = response_cost
|
||||
|
||||
kwargs["response_cost"] = response_cost
|
||||
kwargs["model"] = model
|
||||
|
|
|
|||
|
|
@ -380,7 +380,7 @@ def _synthesize_responses_api_response(
|
|||
if first_cost is not None:
|
||||
current_cost: Final = hidden.get("response_cost") if isinstance(hidden, dict) else 0
|
||||
hidden["response_cost"] = (current_cost or 0) + first_cost
|
||||
synthesized._hidden_params = hidden
|
||||
synthesized.hidden_params = hidden
|
||||
return synthesized
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -4,6 +4,7 @@ from collections.abc import Sequence
|
|||
from typing import Any, Final, cast
|
||||
|
||||
import litellm
|
||||
from litellm.litellm_core_utils.hidden_params import get_or_create_hidden_params
|
||||
from litellm.main import stream_chunk_builder
|
||||
from litellm.responses.litellm_completion_transformation.custom_tools import (
|
||||
build_tool_call_item_kwargs,
|
||||
|
|
@ -649,11 +650,11 @@ class LiteLLMCompletionStreamingIterator(ResponsesAPIStreamingIterator):
|
|||
),
|
||||
)
|
||||
if response is not None and self._accumulated_provider_specific_fields:
|
||||
if not hasattr(response, "_hidden_params") or response._hidden_params is None:
|
||||
response._hidden_params = {}
|
||||
response._hidden_params.setdefault("provider_specific_fields", {}).update(
|
||||
self._accumulated_provider_specific_fields
|
||||
response_hidden_params: Final = get_or_create_hidden_params(response)
|
||||
provider_specific_fields: Final = cast( # cast-ok: provider fields are stored as a mutable mapping
|
||||
dict[str, object], response_hidden_params.setdefault("provider_specific_fields", {})
|
||||
)
|
||||
provider_specific_fields.update(self._accumulated_provider_specific_fields)
|
||||
return response
|
||||
|
||||
@staticmethod
|
||||
|
|
|
|||
|
|
@ -39,6 +39,11 @@ from litellm.constants import REDACTED_BY_LITELLM, REDACTED_TOOL_CALL_ARGUMENTS_
|
|||
from litellm.litellm_core_utils.get_supported_openai_params import (
|
||||
get_supported_openai_params,
|
||||
)
|
||||
from litellm.litellm_core_utils.hidden_params import (
|
||||
get_hidden_params,
|
||||
get_hidden_params_storage,
|
||||
set_hidden_params,
|
||||
)
|
||||
from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj
|
||||
from litellm.responses.litellm_completion_transformation.session_handler import (
|
||||
ResponsesSessionHandler,
|
||||
|
|
@ -2490,10 +2495,17 @@ class LiteLLMCompletionResponsesConfig:
|
|||
user=echoed.get("user"),
|
||||
store=echoed.get("store"),
|
||||
)
|
||||
responses_api_response._hidden_params = getattr(chat_completion_response, "_hidden_params", {})
|
||||
chat_completion_hidden_params: Final = get_hidden_params_storage(chat_completion_response)
|
||||
set_hidden_params(
|
||||
responses_api_response,
|
||||
chat_completion_hidden_params if chat_completion_hidden_params is not None else {},
|
||||
)
|
||||
|
||||
# Surface provider-specific fields (generic passthrough from any provider)
|
||||
provider_fields: Final = responses_api_response._hidden_params.get("provider_specific_fields")
|
||||
response_hidden_params: Final = get_hidden_params(responses_api_response)
|
||||
provider_fields: Final = (
|
||||
response_hidden_params.get("provider_specific_fields") if response_hidden_params is not None else None
|
||||
)
|
||||
if provider_fields:
|
||||
setattr(responses_api_response, "provider_specific_fields", provider_fields)
|
||||
|
||||
|
|
|
|||
|
|
@ -748,7 +748,7 @@ async def aresponses(
|
|||
)
|
||||
# Stamp custom_llm_provider so callbacks can identify the provider
|
||||
# (mirrors litellm/main.py:1371 for chat completions)
|
||||
response._hidden_params["custom_llm_provider"] = custom_llm_provider
|
||||
response.hidden_params["custom_llm_provider"] = custom_llm_provider
|
||||
|
||||
if response is None:
|
||||
raise ValueError(f"Got an unexpected None response from the Responses API: {response}")
|
||||
|
|
@ -1453,7 +1453,7 @@ def responses(
|
|||
)
|
||||
# Stamp custom_llm_provider so callbacks can identify the provider
|
||||
# (mirrors litellm/main.py:1371 for chat completions)
|
||||
response._hidden_params["custom_llm_provider"] = custom_llm_provider
|
||||
response.hidden_params["custom_llm_provider"] = custom_llm_provider
|
||||
|
||||
return response
|
||||
except Exception as e:
|
||||
|
|
|
|||
|
|
@ -6,6 +6,7 @@ from typing import TYPE_CHECKING, Final, cast
|
|||
|
||||
from typing_extensions import TypedDict, Unpack
|
||||
|
||||
from litellm.litellm_core_utils.hidden_params import HIDDEN_PARAMS_ATTR, set_hidden_params
|
||||
from litellm.responses.mcp.litellm_proxy_mcp_handler import (
|
||||
LiteLLM_Proxy_MCP_Handler,
|
||||
)
|
||||
|
|
@ -42,8 +43,8 @@ def _add_mcp_metadata_to_response(
|
|||
# For streaming, store MCP metadata in _hidden_params
|
||||
# CustomStreamWrapper._add_mcp_metadata_to_final_chunk() will automatically
|
||||
# add it to the final chunk's delta.provider_specific_fields
|
||||
if not hasattr(response, "_hidden_params"):
|
||||
response._hidden_params = {}
|
||||
if not hasattr(response, HIDDEN_PARAMS_ATTR):
|
||||
set_hidden_params(response, {})
|
||||
|
||||
mcp_metadata: Final = {}
|
||||
if openai_tools:
|
||||
|
|
@ -54,7 +55,10 @@ def _add_mcp_metadata_to_response(
|
|||
mcp_metadata["mcp_call_results"] = tool_results
|
||||
|
||||
if mcp_metadata:
|
||||
response._hidden_params["mcp_metadata"] = mcp_metadata
|
||||
hidden_params: Final = cast( # cast-ok: preserve mapping operations on dynamic response metadata
|
||||
dict[str, object], getattr(response, HIDDEN_PARAMS_ATTR)
|
||||
)
|
||||
hidden_params["mcp_metadata"] = mcp_metadata
|
||||
return
|
||||
|
||||
if not isinstance(response, ModelResponse):
|
||||
|
|
|
|||
|
|
@ -591,7 +591,7 @@ class BaseResponsesAPIStreamingIterator:
|
|||
target: Final[object] = getattr(logging_response, "response", None)
|
||||
if not isinstance(target, ResponsesAPIResponse):
|
||||
return
|
||||
existing: Final[Mapping[str, object]] = target._hidden_params
|
||||
existing: Final[Mapping[str, object]] = target.hidden_params
|
||||
source_hidden: Final[object] = getattr(
|
||||
getattr(self.completed_response, "response", None), "_hidden_params", None
|
||||
)
|
||||
|
|
@ -602,7 +602,7 @@ class BaseResponsesAPIStreamingIterator:
|
|||
raw_headers: Final[Mapping[str, object]] = raw if isinstance(raw, Mapping) else EMPTY_MAPPING
|
||||
# rebuild by value and let existing keys win: sharing the source dicts would alias what the proxy
|
||||
# splats into the client's HTTP headers, and copying non-header keys would carry response_cost
|
||||
target._hidden_params = {
|
||||
target.hidden_params = {
|
||||
"additional_headers": {**headers},
|
||||
"headers": {**raw_headers},
|
||||
**existing,
|
||||
|
|
|
|||
|
|
@ -325,7 +325,7 @@ class ResponsesAPIRequestUtils:
|
|||
responses_api_response: dict[str, object],
|
||||
custom_llm_provider: str | None,
|
||||
litellm_metadata: dict[str, object] | None = None,
|
||||
) -> dict[str, object]:
|
||||
) -> dict[str, object]:
|
||||
...
|
||||
|
||||
# fmt: on
|
||||
|
|
|
|||
|
|
@ -91,6 +91,13 @@ from litellm.litellm_core_utils.get_llm_provider_logic import (
|
|||
declared_authenticating_provider,
|
||||
is_registered_custom_provider,
|
||||
)
|
||||
from litellm.litellm_core_utils.hidden_params import (
|
||||
HIDDEN_PARAMS_ATTR as _HIDDEN_PARAMS_ATTR,
|
||||
)
|
||||
from litellm.litellm_core_utils.hidden_params import (
|
||||
get_hidden_params,
|
||||
set_hidden_params,
|
||||
)
|
||||
from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLogging
|
||||
from litellm.litellm_core_utils.llm_cost_calc.utils import SERVICE_TIER_COST_KEY_SUFFIXES
|
||||
from litellm.litellm_core_utils.ptu_pricing import (
|
||||
|
|
@ -149,7 +156,6 @@ from litellm.router_strategy.tag_based_routing import (
|
|||
)
|
||||
from litellm.router_utils.access_windows import access_windows_config_error, filter_reserved_deployments
|
||||
from litellm.router_utils.add_retry_fallback_headers import (
|
||||
HiddenParamsHost,
|
||||
add_fallback_headers_to_response,
|
||||
add_retry_headers_to_response,
|
||||
apply_quality_router_decision_headers,
|
||||
|
|
@ -2989,7 +2995,7 @@ class Router:
|
|||
fallback_item: object,
|
||||
prepared_fallback_hidden_params: tuple[dict[str, object], dict[str, object]],
|
||||
) -> None:
|
||||
if fallback_item is None or not hasattr(fallback_item, "_hidden_params"):
|
||||
if fallback_item is None or not hasattr(fallback_item, _HIDDEN_PARAMS_ATTR):
|
||||
return
|
||||
|
||||
fallback_hidden_params, fallback_headers = prepared_fallback_hidden_params
|
||||
|
|
@ -2998,11 +3004,14 @@ class Router:
|
|||
if not isinstance(item_headers, dict):
|
||||
item_headers = {}
|
||||
|
||||
cast(HiddenParamsHost, fallback_item)._hidden_params = {
|
||||
**item_hidden_params,
|
||||
**fallback_hidden_params,
|
||||
"additional_headers": {**item_headers, **fallback_headers},
|
||||
}
|
||||
set_hidden_params(
|
||||
fallback_item,
|
||||
{
|
||||
**item_hidden_params,
|
||||
**fallback_hidden_params,
|
||||
"additional_headers": {**item_headers, **fallback_headers},
|
||||
},
|
||||
)
|
||||
|
||||
async def _acompletion_streaming_iterator(
|
||||
self,
|
||||
|
|
@ -3038,8 +3047,9 @@ class Router:
|
|||
if isinstance(inner_chunks, list):
|
||||
self.chunks = inner_chunks
|
||||
# Preserve hidden params (including litellm_overhead_time_ms) from original response
|
||||
if hasattr(model_response, "_hidden_params"):
|
||||
self._hidden_params = model_response._hidden_params.copy()
|
||||
model_response_hidden_params: Final = get_hidden_params(model_response)
|
||||
if model_response_hidden_params is not None:
|
||||
self._hidden_params = dict(model_response_hidden_params)
|
||||
|
||||
def __aiter__(self):
|
||||
return self
|
||||
|
|
@ -3656,8 +3666,9 @@ class Router:
|
|||
_response_headers=getattr(model_response, "_response_headers", None),
|
||||
)
|
||||
self._sync_generator = sync_generator
|
||||
if hasattr(model_response, "_hidden_params"):
|
||||
self._hidden_params = model_response._hidden_params.copy()
|
||||
model_response_hidden_params: Final = get_hidden_params(model_response)
|
||||
if model_response_hidden_params is not None:
|
||||
self._hidden_params = dict(model_response_hidden_params)
|
||||
|
||||
def __iter__(self):
|
||||
return self
|
||||
|
|
@ -4484,7 +4495,8 @@ class Router:
|
|||
|
||||
if result is not None:
|
||||
# Return the first successful result
|
||||
result._hidden_params["fastest_response_batch_completion"] = True
|
||||
if (result_hidden_params := get_hidden_params(result)) is not None:
|
||||
result_hidden_params["fastest_response_batch_completion"] = True
|
||||
return result
|
||||
|
||||
# If we exit the loop without returning, all tasks failed
|
||||
|
|
@ -4553,8 +4565,12 @@ class Router:
|
|||
if make_request:
|
||||
try:
|
||||
_response: Final = await self.acompletion(model=model, messages=messages, stream=stream, **kwargs)
|
||||
_response._hidden_params.setdefault("additional_headers", {})
|
||||
_response._hidden_params["additional_headers"].update({"x-litellm-request-prioritization-used": True})
|
||||
response_hidden_params: Final = get_hidden_params(_response)
|
||||
if response_hidden_params is not None:
|
||||
additional_headers: Final = cast( # cast-ok: router headers are stored as a mutable mapping
|
||||
dict[str, object], response_hidden_params.setdefault("additional_headers", {})
|
||||
)
|
||||
additional_headers.update({"x-litellm-request-prioritization-used": True})
|
||||
return _response
|
||||
except Exception as e:
|
||||
setattr(e, "priority", priority)
|
||||
|
|
@ -4613,11 +4629,12 @@ class Router:
|
|||
if make_request:
|
||||
try:
|
||||
_response: Final = await original_function(*args, **kwargs)
|
||||
if isinstance(_response._hidden_params, dict):
|
||||
_response._hidden_params.setdefault("additional_headers", {})
|
||||
_response._hidden_params["additional_headers"].update(
|
||||
{"x-litellm-request-prioritization-used": True}
|
||||
response_hidden_params: Final = get_hidden_params(_response)
|
||||
if response_hidden_params is not None:
|
||||
additional_headers: Final = cast( # cast-ok: router headers are stored as a mutable mapping
|
||||
dict[str, object], response_hidden_params.setdefault("additional_headers", {})
|
||||
)
|
||||
additional_headers.update({"x-litellm-request-prioritization-used": True})
|
||||
return _response
|
||||
except Exception as e:
|
||||
setattr(e, "priority", priority)
|
||||
|
|
@ -6453,7 +6470,9 @@ class Router:
|
|||
healthy_deployments=healthy_deployments, responses=responses
|
||||
)
|
||||
returned_response: Final = cast(OpenAIFileObject, responses[0])
|
||||
returned_response._hidden_params["model_file_id_mapping"] = model_file_id_mapping
|
||||
returned_response_hidden_params: Final = get_hidden_params(returned_response)
|
||||
if returned_response_hidden_params is not None:
|
||||
returned_response_hidden_params["model_file_id_mapping"] = model_file_id_mapping
|
||||
return returned_response
|
||||
except Exception as e:
|
||||
verbose_router_logger.exception(
|
||||
|
|
|
|||
|
|
@ -49,6 +49,7 @@ from litellm.litellm_core_utils.core_helpers import (
|
|||
get_parent_otel_span_from_kwargs,
|
||||
is_codex_user_agent,
|
||||
)
|
||||
from litellm.litellm_core_utils.hidden_params import get_hidden_params
|
||||
from litellm.litellm_core_utils.internal_call_metadata import forwarded_internal_call_metadata
|
||||
from litellm.litellm_core_utils.prompt_templates.common_utils import (
|
||||
as_openai_image_part,
|
||||
|
|
@ -413,9 +414,9 @@ def _parent_session_kwargs(request_kwargs: Mapping[str, object] | None) -> Mappi
|
|||
return {k: kwargs[k] for k in ("litellm_session_id", "litellm_trace_id") if kwargs.get(k) is not None}
|
||||
|
||||
|
||||
def _response_cost_or_none(response: ModelResponse | ResponsesAPIResponse) -> float | None:
|
||||
hidden_params: Final = response._hidden_params
|
||||
if not isinstance(hidden_params, dict):
|
||||
def _response_cost_or_none(response: object) -> float | None:
|
||||
hidden_params: Final = get_hidden_params(response)
|
||||
if hidden_params is None:
|
||||
return None
|
||||
cost: Final = hidden_params.get("response_cost")
|
||||
if isinstance(cost, bool) or not isinstance(cost, (int, float)):
|
||||
|
|
|
|||
|
|
@ -6,6 +6,13 @@ from typing import Final, Protocol, TypedDict, cast
|
|||
|
||||
from pydantic import BaseModel, TypeAdapter, ValidationError
|
||||
|
||||
from litellm.litellm_core_utils.hidden_params import (
|
||||
HIDDEN_PARAMS_ATTR as _HIDDEN_PARAMS_ATTR,
|
||||
)
|
||||
from litellm.litellm_core_utils.hidden_params import (
|
||||
set_hidden_params,
|
||||
)
|
||||
|
||||
|
||||
class FallbackErrorInfo(TypedDict):
|
||||
message: str
|
||||
|
|
@ -227,10 +234,8 @@ def get_hidden_params_dict(
|
|||
|
||||
|
||||
def _write_hidden_params(response: object, hidden_params: dict[str, object]) -> None:
|
||||
if isinstance(response, dict):
|
||||
response["_hidden_params"] = hidden_params
|
||||
elif hasattr(response, "_hidden_params"):
|
||||
setattr(response, "_hidden_params", hidden_params)
|
||||
if isinstance(response, dict) or hasattr(response, _HIDDEN_PARAMS_ATTR):
|
||||
set_hidden_params(response, hidden_params)
|
||||
|
||||
|
||||
def _ensure_additional_headers_dict(
|
||||
|
|
|
|||
|
|
@ -360,6 +360,14 @@ class AgentCreateResponse(LiteLLMPydanticObjectBase):
|
|||
|
||||
_hidden_params: dict[str, object] = PrivateAttr(default_factory=dict)
|
||||
|
||||
@property
|
||||
def hidden_params(self) -> dict[str, object]: # mutable-ok: API requires mutation
|
||||
return self._hidden_params
|
||||
|
||||
@hidden_params.setter
|
||||
def hidden_params(self, hidden_params: dict[str, object]) -> None: # mutable-ok: API requires mutation
|
||||
self._hidden_params = hidden_params
|
||||
|
||||
|
||||
class AgentDeleteResult(LiteLLMPydanticObjectBase):
|
||||
"""Result of a provider-side agent deletion (e.g. Gemini DELETE /v1beta/agents/{name}).
|
||||
|
|
@ -374,6 +382,14 @@ class AgentDeleteResult(LiteLLMPydanticObjectBase):
|
|||
|
||||
_hidden_params: dict[str, object] = PrivateAttr(default_factory=dict)
|
||||
|
||||
@property
|
||||
def hidden_params(self) -> dict[str, object]: # mutable-ok: API requires mutation
|
||||
return self._hidden_params
|
||||
|
||||
@hidden_params.setter
|
||||
def hidden_params(self, hidden_params: dict[str, object]) -> None: # mutable-ok: API requires mutation
|
||||
self._hidden_params = hidden_params
|
||||
|
||||
|
||||
class AgentListResponse(LiteLLMPydanticObjectBase):
|
||||
"""Response from listing agents on the provider side (e.g. Gemini GET /v1beta/agents).
|
||||
|
|
@ -388,6 +404,14 @@ class AgentListResponse(LiteLLMPydanticObjectBase):
|
|||
|
||||
_hidden_params: dict[str, object] = PrivateAttr(default_factory=dict)
|
||||
|
||||
@property
|
||||
def hidden_params(self) -> dict[str, object]: # mutable-ok: API requires mutation
|
||||
return self._hidden_params
|
||||
|
||||
@hidden_params.setter
|
||||
def hidden_params(self, hidden_params: dict[str, object]) -> None: # mutable-ok: API requires mutation
|
||||
self._hidden_params = hidden_params
|
||||
|
||||
|
||||
class AgentVersionsResponse(LiteLLMPydanticObjectBase):
|
||||
"""Response from listing versions of an agent (e.g. Gemini GET /v1beta/agents/{name}/versions).
|
||||
|
|
@ -402,6 +426,14 @@ class AgentVersionsResponse(LiteLLMPydanticObjectBase):
|
|||
|
||||
_hidden_params: dict[str, object] = PrivateAttr(default_factory=dict)
|
||||
|
||||
@property
|
||||
def hidden_params(self) -> dict[str, object]: # mutable-ok: API requires mutation
|
||||
return self._hidden_params
|
||||
|
||||
@hidden_params.setter
|
||||
def hidden_params(self, hidden_params: dict[str, object]) -> None: # mutable-ok: API requires mutation
|
||||
self._hidden_params = hidden_params
|
||||
|
||||
|
||||
class AgentMakePublicResponse(LiteLLMBaseModel):
|
||||
message: str
|
||||
|
|
@ -468,6 +500,14 @@ class LiteLLMSendMessageResponse(LiteLLMPydanticObjectBase):
|
|||
# LiteLLM private attributes for logging/cost tracking
|
||||
_hidden_params: dict[str, object] = PrivateAttr(default_factory=dict)
|
||||
|
||||
@property
|
||||
def hidden_params(self) -> dict[str, object]: # mutable-ok: API requires mutation
|
||||
return self._hidden_params
|
||||
|
||||
@hidden_params.setter
|
||||
def hidden_params(self, hidden_params: dict[str, object]) -> None: # mutable-ok: API requires mutation
|
||||
self._hidden_params = hidden_params
|
||||
|
||||
@classmethod
|
||||
def from_a2a_response(
|
||||
cls,
|
||||
|
|
|
|||
|
|
@ -27,6 +27,14 @@ class ContainerObject(LiteLLMBaseModel):
|
|||
name: str | None = None
|
||||
_hidden_params: dict[str, Any] = PrivateAttr(default={})
|
||||
|
||||
@property
|
||||
def hidden_params(self) -> dict[str, builtins.object]: # mutable-ok: API requires mutation
|
||||
return self._hidden_params
|
||||
|
||||
@hidden_params.setter
|
||||
def hidden_params(self, hidden_params: dict[str, builtins.object]) -> None: # mutable-ok: API requires mutation
|
||||
self._hidden_params = hidden_params
|
||||
|
||||
def __contains__(self, key: str) -> bool:
|
||||
# Define custom behavior for the 'in' operator
|
||||
return hasattr(self, key)
|
||||
|
|
@ -144,6 +152,14 @@ class ContainerFileObject(LiteLLMBaseModel):
|
|||
source: str
|
||||
_hidden_params: dict[str, builtins.object] = PrivateAttr(default={})
|
||||
|
||||
@property
|
||||
def hidden_params(self) -> dict[str, builtins.object]: # mutable-ok: API requires mutation
|
||||
return self._hidden_params
|
||||
|
||||
@hidden_params.setter
|
||||
def hidden_params(self, hidden_params: dict[str, builtins.object]) -> None: # mutable-ok: API requires mutation
|
||||
self._hidden_params = hidden_params
|
||||
|
||||
def __contains__(self, key: str) -> bool:
|
||||
return hasattr(self, key)
|
||||
|
||||
|
|
|
|||
|
|
@ -121,3 +121,7 @@ class DecisionsResponse(LiteLLMPydanticObjectBase):
|
|||
model_config = ConfigDict(extra="allow", frozen=True)
|
||||
|
||||
_hidden_params: dict[str, object] = PrivateAttr(default_factory=dict)
|
||||
|
||||
@property
|
||||
def hidden_params(self) -> dict[str, object]: # mutable-ok: API requires mutation
|
||||
return self._hidden_params
|
||||
|
|
|
|||
|
|
@ -23,6 +23,14 @@ if TYPE_CHECKING:
|
|||
class GenerateContentResponse(GoogleGenAIGenerateContentResponse, BaseLiteLLMOpenAIResponseObject):
|
||||
_hidden_params: dict = {}
|
||||
|
||||
@property
|
||||
def hidden_params(self) -> dict[str, object]: # mutable-ok: API requires mutation
|
||||
return self._hidden_params
|
||||
|
||||
@hidden_params.setter
|
||||
def hidden_params(self, hidden_params: dict[str, object]) -> None: # mutable-ok: API requires mutation
|
||||
self._hidden_params = hidden_params
|
||||
|
||||
else:
|
||||
# Fallback types when google.genai is not available
|
||||
ContentListUnion = Any
|
||||
|
|
@ -50,6 +58,14 @@ else:
|
|||
super().__init__(**kwargs)
|
||||
|
||||
class GenerateContentResponse(BaseLiteLLMOpenAIResponseObject):
|
||||
@property
|
||||
def hidden_params(self) -> dict[str, object]: # mutable-ok: API requires mutation
|
||||
return self._hidden_params
|
||||
|
||||
@hidden_params.setter
|
||||
def hidden_params(self, hidden_params: dict[str, object]) -> None: # mutable-ok: API requires mutation
|
||||
self._hidden_params = hidden_params
|
||||
|
||||
def __init__(self, **kwargs) -> None:
|
||||
super().__init__(**kwargs)
|
||||
self._hidden_params = kwargs.get("_hidden_params", {})
|
||||
|
|
|
|||
|
|
@ -1083,6 +1083,14 @@ class InteractionsAPIResponse(BaseLiteLLMOpenAIResponseObject):
|
|||
|
||||
_hidden_params: dict = PrivateAttr(default_factory=dict)
|
||||
|
||||
@property
|
||||
def hidden_params(self) -> dict[str, object]: # mutable-ok: API requires mutation
|
||||
return self._hidden_params
|
||||
|
||||
@hidden_params.setter
|
||||
def hidden_params(self, hidden_params: dict[str, object]) -> None: # mutable-ok: API requires mutation
|
||||
self._hidden_params = hidden_params
|
||||
|
||||
|
||||
class InteractionsAPIStreamingResponse(BaseLiteLLMOpenAIResponseObject):
|
||||
"""
|
||||
|
|
@ -1121,6 +1129,14 @@ class InteractionsAPIStreamingResponse(BaseLiteLLMOpenAIResponseObject):
|
|||
|
||||
_hidden_params: dict = PrivateAttr(default_factory=dict)
|
||||
|
||||
@property
|
||||
def hidden_params(self) -> dict[str, object]: # mutable-ok: API requires mutation
|
||||
return self._hidden_params
|
||||
|
||||
@hidden_params.setter
|
||||
def hidden_params(self, hidden_params: dict[str, object]) -> None: # mutable-ok: API requires mutation
|
||||
self._hidden_params = hidden_params
|
||||
|
||||
|
||||
class DeleteInteractionResult(BaseLiteLLMOpenAIResponseObject):
|
||||
"""Result of deleting an interaction."""
|
||||
|
|
@ -1130,6 +1146,14 @@ class DeleteInteractionResult(BaseLiteLLMOpenAIResponseObject):
|
|||
|
||||
_hidden_params: dict = PrivateAttr(default_factory=dict)
|
||||
|
||||
@property
|
||||
def hidden_params(self) -> dict[str, object]: # mutable-ok: API requires mutation
|
||||
return self._hidden_params
|
||||
|
||||
@hidden_params.setter
|
||||
def hidden_params(self, hidden_params: dict[str, object]) -> None: # mutable-ok: API requires mutation
|
||||
self._hidden_params = hidden_params
|
||||
|
||||
|
||||
class CancelInteractionResult(BaseLiteLLMOpenAIResponseObject):
|
||||
"""Result of cancelling an interaction."""
|
||||
|
|
@ -1139,6 +1163,14 @@ class CancelInteractionResult(BaseLiteLLMOpenAIResponseObject):
|
|||
|
||||
_hidden_params: dict = PrivateAttr(default_factory=dict)
|
||||
|
||||
@property
|
||||
def hidden_params(self) -> dict[str, object]: # mutable-ok: API requires mutation
|
||||
return self._hidden_params
|
||||
|
||||
@hidden_params.setter
|
||||
def hidden_params(self, hidden_params: dict[str, object]) -> None: # mutable-ok: API requires mutation
|
||||
self._hidden_params = hidden_params
|
||||
|
||||
|
||||
# Backwards compatibility aliases
|
||||
InteractionTool = Tool
|
||||
|
|
|
|||
|
|
@ -1,3 +1,4 @@
|
|||
import builtins
|
||||
from collections.abc import Iterable, Mapping
|
||||
from enum import Enum
|
||||
from os import PathLike
|
||||
|
|
@ -120,6 +121,14 @@ class BinaryResponseSummary(TypedDict):
|
|||
class HttpxBinaryResponseContent(_HttpxBinaryResponseContent):
|
||||
_hidden_params: dict
|
||||
|
||||
@property
|
||||
def hidden_params(self) -> dict[str, builtins.object]: # mutable-ok: API requires mutation
|
||||
return self._hidden_params
|
||||
|
||||
@hidden_params.setter
|
||||
def hidden_params(self, hidden_params: dict[str, builtins.object]) -> None: # mutable-ok: API requires mutation
|
||||
self._hidden_params = hidden_params
|
||||
|
||||
def __init__(self, response: httpx.Response) -> None:
|
||||
super().__init__(response)
|
||||
self._hidden_params = {}
|
||||
|
|
@ -406,6 +415,14 @@ class OpenAIFileObject(LiteLLMBaseModel):
|
|||
|
||||
_hidden_params: dict = PrivateAttr(default={"response_cost": 0.0}) # no cost for writing a file
|
||||
|
||||
@property
|
||||
def hidden_params(self) -> dict[str, builtins.object]: # mutable-ok: API requires mutation
|
||||
return self._hidden_params
|
||||
|
||||
@hidden_params.setter
|
||||
def hidden_params(self, hidden_params: dict[str, builtins.object]) -> None: # mutable-ok: API requires mutation
|
||||
self._hidden_params = hidden_params
|
||||
|
||||
@model_serializer(mode="wrap")
|
||||
def _omit_absent_batch_guardrail( # noqa: ANN202 # annotating it replaces the model's serialization schema
|
||||
self, handler: SerializerFunctionWrapHandler
|
||||
|
|
@ -1422,6 +1439,14 @@ class ResponsesAPIResponse(BaseLiteLLMOpenAIResponseObject):
|
|||
# Define private attributes using PrivateAttr
|
||||
_hidden_params: dict = PrivateAttr(default_factory=dict)
|
||||
|
||||
@property
|
||||
def hidden_params(self) -> dict[str, builtins.object]: # mutable-ok: API requires mutation
|
||||
return self._hidden_params
|
||||
|
||||
@hidden_params.setter
|
||||
def hidden_params(self, hidden_params: dict[str, builtins.object]) -> None: # mutable-ok: API requires mutation
|
||||
self._hidden_params = hidden_params
|
||||
|
||||
@field_validator("reasoning", mode="before")
|
||||
@classmethod
|
||||
def validate_reasoning_to_dict(cls, value: Any) -> dict[str, Any] | None:
|
||||
|
|
@ -1597,6 +1622,14 @@ class ResponseCompletedEvent(BaseLiteLLMOpenAIResponseObject):
|
|||
response: ResponsesAPIResponse
|
||||
_hidden_params: dict = PrivateAttr(default_factory=dict)
|
||||
|
||||
@property
|
||||
def hidden_params(self) -> dict[str, object]: # mutable-ok: API requires mutation
|
||||
return self._hidden_params
|
||||
|
||||
@hidden_params.setter
|
||||
def hidden_params(self, hidden_params: dict[str, object]) -> None: # mutable-ok: API requires mutation
|
||||
self._hidden_params = hidden_params
|
||||
|
||||
|
||||
class ResponseFailedEvent(BaseLiteLLMOpenAIResponseObject):
|
||||
type: Literal[ResponsesAPIStreamEvents.RESPONSE_FAILED]
|
||||
|
|
@ -2397,6 +2430,14 @@ class OpenAIModerationResponse(BaseLiteLLMOpenAIResponseObject):
|
|||
# Define private attributes using PrivateAttr
|
||||
_hidden_params: dict = PrivateAttr(default_factory=dict)
|
||||
|
||||
@property
|
||||
def hidden_params(self) -> dict[str, object]: # mutable-ok: API requires mutation
|
||||
return self._hidden_params
|
||||
|
||||
@hidden_params.setter
|
||||
def hidden_params(self, hidden_params: dict[str, object]) -> None: # mutable-ok: API requires mutation
|
||||
self._hidden_params = hidden_params
|
||||
|
||||
|
||||
class OpenAIChatCompletionLogprobs(TypedDict, total=False):
|
||||
content: list[OpenAIChatCompletionLogprobsContent]
|
||||
|
|
@ -2559,6 +2600,14 @@ class OpenAIVideoObject(LiteLLMBaseModel):
|
|||
|
||||
_hidden_params: dict[str, _JsonValue] = PrivateAttr(default={})
|
||||
|
||||
@property
|
||||
def hidden_params(self) -> dict[str, _JsonValue]: # mutable-ok: API requires mutation
|
||||
return self._hidden_params
|
||||
|
||||
@hidden_params.setter
|
||||
def hidden_params(self, hidden_params: dict[str, _JsonValue]) -> None: # mutable-ok: API requires mutation
|
||||
self._hidden_params = hidden_params
|
||||
|
||||
def __contains__(self, key) -> bool:
|
||||
return hasattr(self, key)
|
||||
|
||||
|
|
|
|||
|
|
@ -87,6 +87,14 @@ class RerankResponse(LiteLLMBaseModel):
|
|||
# Define private attributes using PrivateAttr
|
||||
_hidden_params: dict = PrivateAttr(default_factory=dict)
|
||||
|
||||
@property
|
||||
def hidden_params(self) -> dict[str, object]: # mutable-ok: API requires mutation
|
||||
return self._hidden_params
|
||||
|
||||
@hidden_params.setter
|
||||
def hidden_params(self, hidden_params: dict[str, object]) -> None: # mutable-ok: API requires mutation
|
||||
self._hidden_params = hidden_params
|
||||
|
||||
def __getitem__(self, key):
|
||||
return self.__dict__[key]
|
||||
|
||||
|
|
|
|||
|
|
@ -1,3 +1,4 @@
|
|||
import builtins
|
||||
from collections.abc import Mapping, Sequence
|
||||
from typing import Final, Literal, Optional, Union
|
||||
|
||||
|
|
@ -164,6 +165,14 @@ class DeleteResponseResult(BaseLiteLLMOpenAIResponseObject):
|
|||
# Define private attributes using PrivateAttr
|
||||
_hidden_params: dict = PrivateAttr(default_factory=dict)
|
||||
|
||||
@property
|
||||
def hidden_params(self) -> dict[str, builtins.object]: # mutable-ok: API requires mutation
|
||||
return self._hidden_params
|
||||
|
||||
@hidden_params.setter
|
||||
def hidden_params(self, hidden_params: dict[str, builtins.object]) -> None: # mutable-ok: API requires mutation
|
||||
self._hidden_params = hidden_params
|
||||
|
||||
|
||||
class DecodedResponseId(TypedDict, total=False):
|
||||
"""Structure representing a decoded response ID"""
|
||||
|
|
|
|||
|
|
@ -2092,6 +2092,14 @@ class ModelResponseBase(OpenAIObject):
|
|||
|
||||
_hidden_params: dict = {}
|
||||
|
||||
@property
|
||||
def hidden_params(self) -> dict[str, object]: # mutable-ok: API requires mutation
|
||||
return self._hidden_params
|
||||
|
||||
@hidden_params.setter
|
||||
def hidden_params(self, hidden_params: dict[str, object]) -> None: # mutable-ok: API requires mutation
|
||||
self._hidden_params = hidden_params
|
||||
|
||||
_response_headers: dict | None = None
|
||||
|
||||
def set_provider_response_headers(self, headers: httpx.Headers) -> None:
|
||||
|
|
@ -2314,6 +2322,15 @@ class EmbeddingResponse(OpenAIObject):
|
|||
"""Usage statistics for the embedding request."""
|
||||
|
||||
_hidden_params: dict = {}
|
||||
|
||||
@property
|
||||
def hidden_params(self) -> dict[str, object]: # mutable-ok: API requires mutation
|
||||
return self._hidden_params
|
||||
|
||||
@hidden_params.setter
|
||||
def hidden_params(self, hidden_params: dict[str, object]) -> None: # mutable-ok: API requires mutation
|
||||
self._hidden_params = hidden_params
|
||||
|
||||
_response_headers: dict | None = None
|
||||
_response_ms: float | None = None
|
||||
|
||||
|
|
@ -2454,6 +2471,14 @@ class TextCompletionResponse(OpenAIObject):
|
|||
_response_ms: int | None = None
|
||||
_hidden_params: HiddenParams
|
||||
|
||||
@property
|
||||
def hidden_params(self) -> HiddenParams:
|
||||
return self._hidden_params
|
||||
|
||||
@hidden_params.setter
|
||||
def hidden_params(self, hidden_params: HiddenParams) -> None:
|
||||
self._hidden_params = hidden_params
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
id=None,
|
||||
|
|
@ -2618,6 +2643,14 @@ from openai.types.images_response import ImagesResponse as OpenAIImageResponse
|
|||
class ImageResponse(OpenAIImageResponse, BaseLiteLLMOpenAIResponseObject):
|
||||
_hidden_params: dict = {}
|
||||
|
||||
@property
|
||||
def hidden_params(self) -> dict[str, object]: # mutable-ok: API requires mutation
|
||||
return self._hidden_params
|
||||
|
||||
@hidden_params.setter
|
||||
def hidden_params(self, hidden_params: dict[str, object]) -> None: # mutable-ok: API requires mutation
|
||||
self._hidden_params = hidden_params
|
||||
|
||||
usage: ImageUsage | None = None
|
||||
"""
|
||||
Users might use litellm with older python versions, we don't want this to break for them.
|
||||
|
|
@ -2753,6 +2786,14 @@ class TranscriptionResponse(OpenAIObject):
|
|||
_hidden_params: dict = {}
|
||||
_response_headers: dict | None = None
|
||||
|
||||
@property
|
||||
def hidden_params(self) -> dict[str, object]: # mutable-ok: API requires mutation
|
||||
return self._hidden_params
|
||||
|
||||
@hidden_params.setter
|
||||
def hidden_params(self, hidden_params: dict[str, object]) -> None: # mutable-ok: API requires mutation
|
||||
self._hidden_params = hidden_params
|
||||
|
||||
def __init__(self, text=None) -> None:
|
||||
super().__init__(text=text)
|
||||
|
||||
|
|
@ -4343,6 +4384,15 @@ class SelectTokenizerResponse(TypedDict):
|
|||
|
||||
class LiteLLMFineTuningJob(FineTuningJob):
|
||||
_hidden_params: dict = {}
|
||||
|
||||
@property
|
||||
def hidden_params(self) -> dict[str, object]: # mutable-ok: API requires mutation
|
||||
return self._hidden_params
|
||||
|
||||
@hidden_params.setter
|
||||
def hidden_params(self, hidden_params: dict[str, object]) -> None: # mutable-ok: API requires mutation
|
||||
self._hidden_params = hidden_params
|
||||
|
||||
seed: int | None = None
|
||||
|
||||
def __init__(self, **kwargs) -> None:
|
||||
|
|
@ -4356,6 +4406,15 @@ class LiteLLMFineTuningJob(FineTuningJob):
|
|||
|
||||
class LiteLLMBatch(Batch):
|
||||
_hidden_params: dict = {}
|
||||
|
||||
@property
|
||||
def hidden_params(self) -> dict[str, object]: # mutable-ok: API requires mutation
|
||||
return self._hidden_params
|
||||
|
||||
@hidden_params.setter
|
||||
def hidden_params(self, hidden_params: dict[str, object]) -> None: # mutable-ok: API requires mutation
|
||||
self._hidden_params = hidden_params
|
||||
|
||||
usage: Usage | None = None
|
||||
|
||||
def __contains__(self, key) -> bool:
|
||||
|
|
@ -4388,6 +4447,14 @@ class LiteLLMRealtimeStreamLoggingObject(LiteLLMPydanticObjectBase):
|
|||
service_tier: str | None = None
|
||||
_hidden_params: dict = {}
|
||||
|
||||
@property
|
||||
def hidden_params(self) -> dict[str, object]: # mutable-ok: API requires mutation
|
||||
return self._hidden_params
|
||||
|
||||
@hidden_params.setter
|
||||
def hidden_params(self, hidden_params: dict[str, object]) -> None: # mutable-ok: API requires mutation
|
||||
self._hidden_params = hidden_params
|
||||
|
||||
@field_serializer("results")
|
||||
def _serialize_results(self, results: OpenAIRealtimeStreamList) -> list[dict[str, Any]]:
|
||||
return [dict(event) for event in results]
|
||||
|
|
|
|||
|
|
@ -26,6 +26,14 @@ class VideoObject(LiteLLMBaseModel):
|
|||
usage: dict[str, Any] | None = None
|
||||
_hidden_params: dict[str, builtins.object] = PrivateAttr(default={})
|
||||
|
||||
@property
|
||||
def hidden_params(self) -> dict[str, builtins.object]: # mutable-ok: API requires mutation
|
||||
return self._hidden_params
|
||||
|
||||
@hidden_params.setter
|
||||
def hidden_params(self, hidden_params: dict[str, builtins.object]) -> None: # mutable-ok: API requires mutation
|
||||
self._hidden_params = hidden_params
|
||||
|
||||
def __contains__(self, key) -> bool:
|
||||
# Define custom behavior for the 'in' operator
|
||||
return hasattr(self, key)
|
||||
|
|
@ -115,6 +123,14 @@ class CharacterObject(LiteLLMBaseModel):
|
|||
name: str
|
||||
_hidden_params: dict[str, builtins.object] = PrivateAttr(default={})
|
||||
|
||||
@property
|
||||
def hidden_params(self) -> dict[str, builtins.object]: # mutable-ok: API requires mutation
|
||||
return self._hidden_params
|
||||
|
||||
@hidden_params.setter
|
||||
def hidden_params(self, hidden_params: dict[str, builtins.object]) -> None: # mutable-ok: API requires mutation
|
||||
self._hidden_params = hidden_params
|
||||
|
||||
def __contains__(self, key) -> bool:
|
||||
return hasattr(self, key)
|
||||
|
||||
|
|
|
|||
|
|
@ -194,6 +194,7 @@ _SDK_SCRIPT: Final = textwrap.dedent(
|
|||
|
||||
import litellm
|
||||
from litellm.integrations.custom_logger import CustomLogger
|
||||
from litellm.litellm_core_utils.hidden_params import get_hidden_params
|
||||
from litellm.litellm_core_utils.logging_worker import GLOBAL_LOGGING_WORKER
|
||||
from litellm.types.utils import ModelInfo
|
||||
|
||||
|
|
@ -208,7 +209,7 @@ _SDK_SCRIPT: Final = textwrap.dedent(
|
|||
end_time: object,
|
||||
) -> None:
|
||||
standard: Final = kwargs.get("standard_logging_object")
|
||||
hidden: Final = getattr(response_obj, "_hidden_params", None)
|
||||
hidden: Final = get_hidden_params(response_obj)
|
||||
if not isinstance(standard, dict):
|
||||
raise RuntimeError("success callback omitted standard_logging_object")
|
||||
if standard.get("call_type") != "aretrieve_batch":
|
||||
|
|
|
|||
|
|
@ -2146,6 +2146,49 @@ async def test_cache_hit_records_the_looked_up_key_as_the_preset_cache_key(monke
|
|||
assert hit.cached_result._hidden_params["cache_key"] == handler.preset_cache_key
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_text_completion_cache_hit_records_cache_key_in_hidden_params(monkeypatch):
|
||||
import litellm
|
||||
from litellm.caching.caching import Cache
|
||||
from litellm.types.utils import CallTypes
|
||||
|
||||
async def atext_completion(**kwargs):
|
||||
return None
|
||||
|
||||
monkeypatch.setattr(litellm, "cache", Cache(type="local"))
|
||||
kwargs = {"model": "gpt-5.4", "prompt": "hello", "caching": True}
|
||||
await litellm.cache.async_add_cache(
|
||||
litellm.TextCompletionResponse(
|
||||
id="cached-text-response",
|
||||
choices=[litellm.utils.TextChoices(text="cached")],
|
||||
model="gpt-5.4",
|
||||
),
|
||||
**kwargs,
|
||||
)
|
||||
handler = LLMCachingHandler(
|
||||
original_function=atext_completion,
|
||||
request_kwargs=kwargs,
|
||||
start_time=datetime.now(),
|
||||
)
|
||||
logging_obj = _build_logging_obj(CallTypes.atext_completion.value, stream=False)
|
||||
logging_obj.async_success_handler = AsyncMock()
|
||||
|
||||
hit = await handler._async_get_cache(
|
||||
model="gpt-5.4",
|
||||
original_function=atext_completion,
|
||||
logging_obj=logging_obj,
|
||||
start_time=datetime.now(),
|
||||
call_type=CallTypes.atext_completion.value,
|
||||
kwargs=kwargs,
|
||||
args=(),
|
||||
)
|
||||
|
||||
assert hit.cached_result is not None
|
||||
assert isinstance(hit.cached_result, litellm.TextCompletionResponse)
|
||||
assert handler.preset_cache_key is not None
|
||||
assert hit.cached_result.hidden_params.get("cache_key") == handler.preset_cache_key
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_converted_stream_cache_hit_replayed_as_plain_object_logs_at_hit_time(monkeypatch):
|
||||
import litellm
|
||||
|
|
@ -2580,7 +2623,7 @@ async def test_acompletion_after_a_response_without_choices_calls_the_provider_a
|
|||
|
||||
second = await litellm.acompletion(model="gpt-4o", messages=messages, mock_response="hi", caching=True)
|
||||
assert second.choices[0].message.content == "hi"
|
||||
assert second._hidden_params.get("cache_hit") is not True
|
||||
assert second.hidden_params.get("cache_hit") is not True
|
||||
|
||||
|
||||
def _stream_chunk(choices: list[StreamingChoices]) -> ModelResponseStream:
|
||||
|
|
|
|||
|
|
@ -267,6 +267,15 @@ def test_decisions_cost_uses_litellm_token_pricing() -> None:
|
|||
assert cost == pytest.approx(expected_cost)
|
||||
|
||||
|
||||
def test_decisions_response_hidden_params_getter_preserves_mutable_identity() -> None:
|
||||
response: Final = DecisionsResponse(model="decider", answers={}, usage=None)
|
||||
|
||||
assert response.hidden_params is response._hidden_params
|
||||
|
||||
response.hidden_params["mutation"] = "visible"
|
||||
assert response._hidden_params["mutation"] == "visible"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_decisions_cost_is_in_standard_logging_object(respx_mock: respx.MockRouter) -> None:
|
||||
respx_mock.post("https://api.perplexity.ai/v1/decisions").respond(json=_RESPONSE)
|
||||
|
|
|
|||
|
|
@ -21,7 +21,13 @@ from litellm.litellm_core_utils.health_check_helpers import (
|
|||
)
|
||||
from litellm.main import ahealth_check
|
||||
from litellm.proxy._types import UserAPIKeyAuth
|
||||
from litellm.types.utils import LIST_BATCHES_SUPPORTED_PROVIDERS
|
||||
from litellm.types.llms.base import HiddenParams
|
||||
from litellm.types.utils import (
|
||||
LIST_BATCHES_SUPPORTED_PROVIDERS,
|
||||
TextChoices,
|
||||
TextCompletionResponse,
|
||||
Usage,
|
||||
)
|
||||
|
||||
|
||||
def _png_chunks(png: bytes, offset: int = 8) -> tuple[tuple[bytes, bytes], ...]:
|
||||
|
|
@ -146,6 +152,28 @@ async def test_ahealth_check_supports_image_edit_mode():
|
|||
assert "Mode image_edit not supported" not in str(result)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_ahealth_check_completion_includes_headers_from_hidden_params_model() -> None:
|
||||
response: Final = TextCompletionResponse(
|
||||
id="cmpl-test",
|
||||
object="text_completion",
|
||||
created=1,
|
||||
model="gpt-3.5-turbo-instruct",
|
||||
choices=[TextChoices(text="hello", index=0, logprobs=None, finish_reason="stop")],
|
||||
usage=Usage(prompt_tokens=1, completion_tokens=1, total_tokens=2),
|
||||
)
|
||||
response.hidden_params = HiddenParams(headers={"x-ratelimit-remaining-requests": "5"})
|
||||
|
||||
with patch("litellm.atext_completion", new_callable=AsyncMock, return_value=response) as mock_atext_completion:
|
||||
result: Final = await ahealth_check(
|
||||
{"model": "gpt-3.5-turbo-instruct", "api_key": "sk-test"},
|
||||
mode="completion",
|
||||
)
|
||||
|
||||
mock_atext_completion.assert_awaited_once()
|
||||
assert result == {"x-ratelimit-remaining-requests": "5"}
|
||||
|
||||
|
||||
def test_update_model_params_with_health_check_tracking_information():
|
||||
"""Test update_model_params_with_health_check_tracking_information adds required tracking info."""
|
||||
initial_model_params = {"model": "gpt-3.5-turbo", "api_key": "test_key"}
|
||||
|
|
|
|||
252
tests/unit/litellm_core_utils/test_hidden_params.py
Normal file
252
tests/unit/litellm_core_utils/test_hidden_params.py
Normal file
|
|
@ -0,0 +1,252 @@
|
|||
from typing import Final
|
||||
|
||||
import pytest
|
||||
|
||||
from litellm.litellm_core_utils.hidden_params import (
|
||||
HIDDEN_PARAMS_ATTR,
|
||||
get_hidden_params,
|
||||
get_hidden_params_storage,
|
||||
get_or_create_hidden_params,
|
||||
set_hidden_param,
|
||||
set_hidden_params,
|
||||
)
|
||||
from litellm.llms.openai.completion.transformation import OpenAITextCompletionConfig
|
||||
from litellm.types.decisions import DecisionsResponse
|
||||
from litellm.types.llms.base import HiddenParams
|
||||
from litellm.types.utils import ModelResponse, TextChoices, TextCompletionResponse, Usage
|
||||
|
||||
|
||||
def test_get_and_set_hidden_params_on_plain_object() -> None:
|
||||
class PlainResponse:
|
||||
def __init__(self) -> None:
|
||||
self._hidden_params = {"existing": True}
|
||||
|
||||
response: Final = PlainResponse()
|
||||
stored: Final = get_hidden_params(response)
|
||||
assert stored is response._hidden_params
|
||||
|
||||
replacement: Final = {"replacement": True}
|
||||
set_hidden_params(response, replacement)
|
||||
|
||||
assert response._hidden_params is replacement
|
||||
assert get_hidden_params(response) is replacement
|
||||
|
||||
|
||||
def test_get_and_set_hidden_params_on_dict() -> None:
|
||||
response: Final = {"_hidden_params": {"existing": True}}
|
||||
stored: Final = get_hidden_params(response)
|
||||
|
||||
assert stored is response["_hidden_params"]
|
||||
|
||||
replacement: Final = {"replacement": True}
|
||||
set_hidden_params(response, replacement)
|
||||
|
||||
assert response["_hidden_params"] is replacement
|
||||
assert get_hidden_params(response) is replacement
|
||||
|
||||
|
||||
def test_get_hidden_params_storage_preserves_supported_storage_identity() -> None:
|
||||
class PlainResponse:
|
||||
pass
|
||||
|
||||
dict_storage: Final[dict[str, object]] = {"model_id": "dict-model"}
|
||||
model_storage: Final = HiddenParams(model_id="model-storage")
|
||||
dict_response: Final = PlainResponse()
|
||||
model_response: Final = PlainResponse()
|
||||
setattr(dict_response, HIDDEN_PARAMS_ATTR, dict_storage)
|
||||
setattr(model_response, HIDDEN_PARAMS_ATTR, model_storage)
|
||||
|
||||
assert get_hidden_params_storage(dict_response) is dict_storage
|
||||
assert get_hidden_params_storage(model_response) is model_storage
|
||||
|
||||
|
||||
def test_get_or_create_hidden_params_sets_empty_dict_on_plain_object() -> None:
|
||||
class PlainResponse:
|
||||
pass
|
||||
|
||||
response: Final = PlainResponse()
|
||||
hidden_params: Final = get_or_create_hidden_params(response)
|
||||
|
||||
assert response._hidden_params is hidden_params
|
||||
assert hidden_params == {}
|
||||
|
||||
|
||||
def test_set_hidden_param_writes_existing_hidden_params_in_place() -> None:
|
||||
class PlainResponse:
|
||||
pass
|
||||
|
||||
storage: Final = HiddenParams(response_cost=0.25)
|
||||
response: Final = PlainResponse()
|
||||
setattr(response, HIDDEN_PARAMS_ATTR, storage)
|
||||
|
||||
set_hidden_param(response, "k", "value")
|
||||
|
||||
assert getattr(response, HIDDEN_PARAMS_ATTR) is storage
|
||||
assert storage["k"] == "value"
|
||||
|
||||
|
||||
def test_set_hidden_param_creates_dict_storage_when_missing_or_none() -> None:
|
||||
class PlainResponse:
|
||||
pass
|
||||
|
||||
missing_response: Final = PlainResponse()
|
||||
none_response: Final = PlainResponse()
|
||||
setattr(none_response, HIDDEN_PARAMS_ATTR, None)
|
||||
|
||||
set_hidden_param(missing_response, "k", "missing")
|
||||
set_hidden_param(none_response, "k", "none")
|
||||
|
||||
assert getattr(missing_response, HIDDEN_PARAMS_ATTR) == {"k": "missing"}
|
||||
assert getattr(none_response, HIDDEN_PARAMS_ATTR) == {"k": "none"}
|
||||
|
||||
|
||||
def test_set_hidden_param_writes_existing_dict_storage_in_place() -> None:
|
||||
class PlainResponse:
|
||||
pass
|
||||
|
||||
storage: Final[dict[str, object]] = {"existing": True}
|
||||
response: Final = PlainResponse()
|
||||
setattr(response, HIDDEN_PARAMS_ATTR, storage)
|
||||
|
||||
set_hidden_param(response, "k", "value")
|
||||
|
||||
assert getattr(response, HIDDEN_PARAMS_ATTR) is storage
|
||||
assert storage == {"existing": True, "k": "value"}
|
||||
|
||||
|
||||
def test_set_hidden_param_rejects_unsupported_storage_without_replacing_it() -> None:
|
||||
class PlainResponse:
|
||||
pass
|
||||
|
||||
storage: Final = object()
|
||||
response: Final = PlainResponse()
|
||||
setattr(response, HIDDEN_PARAMS_ATTR, storage)
|
||||
|
||||
with pytest.raises(TypeError, match="unsupported hidden params storage: object"):
|
||||
set_hidden_param(response, "k", "value")
|
||||
|
||||
assert getattr(response, HIDDEN_PARAMS_ATTR) is storage
|
||||
|
||||
|
||||
def test_get_or_create_hidden_params_wraps_hidden_params_storage() -> None:
|
||||
class PlainResponse:
|
||||
pass
|
||||
|
||||
storage: Final = HiddenParams(
|
||||
response_cost=0.25,
|
||||
headers={"x-ratelimit-remaining-requests": "5"},
|
||||
)
|
||||
response: Final = PlainResponse()
|
||||
setattr(response, HIDDEN_PARAMS_ATTR, storage)
|
||||
|
||||
hidden_params: Final = get_or_create_hidden_params(response)
|
||||
|
||||
assert getattr(response, HIDDEN_PARAMS_ATTR) is storage
|
||||
assert hidden_params["response_cost"] == 0.25
|
||||
assert hidden_params["headers"] == {"x-ratelimit-remaining-requests": "5"}
|
||||
assert "headers" in hidden_params
|
||||
assert "headers" in tuple(hidden_params)
|
||||
|
||||
nested: Final = hidden_params.setdefault("nested", {})
|
||||
assert hidden_params.setdefault("nested", {}) is nested
|
||||
assert storage.nested is nested
|
||||
|
||||
del hidden_params["headers"]
|
||||
assert "headers" not in hidden_params
|
||||
assert "headers" not in tuple(hidden_params)
|
||||
|
||||
|
||||
def test_hidden_params_model_view_excludes_unset_fields() -> None:
|
||||
class PlainResponse:
|
||||
pass
|
||||
|
||||
storage: Final = HiddenParams()
|
||||
response: Final = PlainResponse()
|
||||
setattr(response, HIDDEN_PARAMS_ATTR, storage)
|
||||
view: Final = get_or_create_hidden_params(response)
|
||||
|
||||
assert not view
|
||||
assert list(view) == []
|
||||
assert "response_cost" not in view
|
||||
assert "_response_ms" not in view
|
||||
assert view.get("model_id", "d") == "d"
|
||||
assert "_response_ms" not in storage.model_fields_set
|
||||
|
||||
with pytest.raises(KeyError, match="response_cost"):
|
||||
del view["response_cost"]
|
||||
|
||||
view["model_id"] = "m"
|
||||
view["extra"] = 1
|
||||
|
||||
assert "model_id" in view
|
||||
assert "extra" in view
|
||||
assert set(view) == {"model_id", "extra"}
|
||||
assert "model_id" in storage.model_fields_set
|
||||
assert "extra" in (storage.model_extra or {})
|
||||
assert storage.model_id == "m"
|
||||
assert storage.model_extra == {"extra": 1}
|
||||
|
||||
fallback_response: Final = PlainResponse()
|
||||
setattr(fallback_response, HIDDEN_PARAMS_ATTR, HiddenParams())
|
||||
fallback_view: Final = get_or_create_hidden_params(fallback_response)
|
||||
merged_after_fallback: Final = {**fallback_view, **{"model_id": "m1"}}
|
||||
merged_before_fallback: Final = {**{"model_id": "m1"}, **fallback_view}
|
||||
|
||||
assert merged_after_fallback == {"model_id": "m1"}
|
||||
assert merged_before_fallback == {"model_id": "m1"}
|
||||
|
||||
|
||||
def test_hidden_params_model_dump_includes_response_ms_when_excluding_unset() -> None:
|
||||
storage: Final = HiddenParams(response_cost=1.0)
|
||||
|
||||
assert "_response_ms" in storage.model_dump(exclude_unset=True)
|
||||
|
||||
|
||||
def test_openai_text_completion_conversion_preserves_hidden_params_storage() -> None:
|
||||
source: Final = TextCompletionResponse(
|
||||
id="cmpl-test",
|
||||
object="text_completion",
|
||||
created=1,
|
||||
model="gpt-3.5-turbo-instruct",
|
||||
choices=[TextChoices(text="hello", index=0, logprobs=None, finish_reason="stop")],
|
||||
usage=Usage(prompt_tokens=1, completion_tokens=1, total_tokens=2),
|
||||
)
|
||||
response: Final = ModelResponse()
|
||||
storage: Final = HiddenParams(response_cost=0.25)
|
||||
setattr(response, HIDDEN_PARAMS_ATTR, storage)
|
||||
|
||||
converted: Final = OpenAITextCompletionConfig().convert_to_chat_model_response_object(
|
||||
response_object=source,
|
||||
model_response_object=response,
|
||||
)
|
||||
|
||||
assert converted is response
|
||||
assert getattr(response, HIDDEN_PARAMS_ATTR) is storage
|
||||
assert storage["original_response"] is source
|
||||
|
||||
|
||||
def test_get_hidden_params_preserves_model_response_identity() -> None:
|
||||
response: Final = ModelResponse()
|
||||
|
||||
assert get_hidden_params(response) is response.hidden_params
|
||||
|
||||
|
||||
def test_get_hidden_params_returns_none_for_non_dict_storage() -> None:
|
||||
class PlainResponse:
|
||||
def __init__(self) -> None:
|
||||
self._hidden_params = object()
|
||||
|
||||
response: Final = PlainResponse()
|
||||
|
||||
assert get_hidden_params(response) is None
|
||||
assert get_hidden_params_storage(response) is None
|
||||
|
||||
|
||||
def test_set_hidden_params_replaces_frozen_decisions_response_private_attr() -> None:
|
||||
response: Final = DecisionsResponse(model="decider", answers={}, usage=None)
|
||||
replacement: Final = {"replacement": True}
|
||||
|
||||
set_hidden_params(response, replacement)
|
||||
|
||||
assert response.hidden_params is replacement
|
||||
assert response._hidden_params is replacement
|
||||
Some files were not shown because too many files have changed in this diff Show more
Loading…
Add table
Reference in a new issue