From d01efcb3084a0a3bd7302294d1072ab8ab247ecd Mon Sep 17 00:00:00 2001 From: AlexsanderHamir Date: Sat, 15 Nov 2025 15:58:41 -0800 Subject: [PATCH 001/121] speech set up --- no_cache_hits.py | 48 +++++++++++++++++++++++++++++++++++++++++++++ speech.mp3 | Bin 0 -> 104 bytes speech_config.yaml | 9 +++++++++ 3 files changed, 57 insertions(+) create mode 100644 no_cache_hits.py create mode 100644 speech.mp3 create mode 100644 speech_config.yaml diff --git a/no_cache_hits.py b/no_cache_hits.py new file mode 100644 index 00000000000..1b3bf895f77 --- /dev/null +++ b/no_cache_hits.py @@ -0,0 +1,48 @@ +from locust import HttpUser, between, task + + +class MyUser(HttpUser): + """ + Minimal Locust user for repeatedly hitting `/v1/audio/speech`. + The goal is to measure server-side performance, so we avoid any extra work + (file writes, random generation, manual timing, custom event hooks, etc.) + that could inflate client-side latency. + """ + + wait_time = between(0.5, 1) + host = "http://0.0.0.0:8090" + + def on_start(self): + self.api_key = "sk-1234" + self.model_name = "fake-openai-speech" + self.headers = { + "Authorization": f"Bearer {self.api_key}", + "Content-Type": "application/json", + } + self.prompt_counter = 0 + + @task + def audio_speech_request(self): + self.prompt_counter += 1 + # Ensure prompts differ slightly so the backend can't reuse cached audio. + prompt = ( + "Generate a short spoken status update mentioning counter " + f"{self.prompt_counter}." + ) + + response = self.client.post( + "v1/audio/speech", + json={ + "model": self.model_name, + "input": prompt, + "voice": "alloy", + "format": "mp3", + }, + headers=self.headers, + name="audio_speech", + ) + + if response.status_code != 200: + # log the errors in error.txt + with open("error.txt", "a") as error_log: + error_log.write(response.text + "\n") \ No newline at end of file diff --git a/speech.mp3 b/speech.mp3 new file mode 100644 index 0000000000000000000000000000000000000000..f4f854d9bd215e2493d48b4bc4d39804bd79c038 GIT binary patch literal 104 NcmezWdjbPJ000JT0*e3u literal 0 HcmV?d00001 diff --git a/speech_config.yaml b/speech_config.yaml new file mode 100644 index 00000000000..ad9920a2793 --- /dev/null +++ b/speech_config.yaml @@ -0,0 +1,9 @@ +model_list: + - model_name: fake-openai-speech + litellm_params: + model: openai/gpt-4o-mini-tts + api_base: http://0.0.0.0:8090/ + api_key: sk-1234 + model_info: + mode: audio_speech + \ No newline at end of file From 44f2013495c6987bef3918ca045f942777b21c34 Mon Sep 17 00:00:00 2001 From: AlexsanderHamir Date: Sat, 15 Nov 2025 17:02:15 -0800 Subject: [PATCH 002/121] fix: change chunk_size for aiter_bytes 1KB is too small for audio and is lowering the RPS when testing with medium to large files --- litellm/proxy/proxy_server.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/litellm/proxy/proxy_server.py b/litellm/proxy/proxy_server.py index a6e73199f0e..36178652a7d 100644 --- a/litellm/proxy/proxy_server.py +++ b/litellm/proxy/proxy_server.py @@ -5305,7 +5305,7 @@ async def audio_speech( # Printing each chunk size async def generate(_response: HttpxBinaryResponseContent): - _generator = await _response.aiter_bytes(chunk_size=1024) + _generator = await _response.aiter_bytes(chunk_size=4096) async for chunk in _generator: yield chunk From 348d28d871a8c8d00ec1263d6d61caff4843cd7b Mon Sep 17 00:00:00 2001 From: AlexsanderHamir Date: Sat, 15 Nov 2025 17:15:26 -0800 Subject: [PATCH 003/121] fix: remove function definition from every request --- litellm/proxy/proxy_server.py | 19 ++++++++++++------- 1 file changed, 12 insertions(+), 7 deletions(-) diff --git a/litellm/proxy/proxy_server.py b/litellm/proxy/proxy_server.py index 36178652a7d..c65c7f77f0d 100644 --- a/litellm/proxy/proxy_server.py +++ b/litellm/proxy/proxy_server.py @@ -14,6 +14,7 @@ from datetime import datetime, timedelta from typing import ( TYPE_CHECKING, Any, + AsyncGenerator, List, Literal, Optional, @@ -5231,6 +5232,14 @@ async def moderations( ) +async def _audio_speech_chunk_generator( + _response: HttpxBinaryResponseContent, +) -> AsyncGenerator[bytes, None]: + _generator = await _response.aiter_bytes(chunk_size=4096) + async for chunk in _generator: + yield chunk + + @router.post( "/v1/audio/speech", dependencies=[Depends(user_api_key_auth)], @@ -5303,12 +5312,6 @@ async def audio_speech( response_cost = hidden_params.get("response_cost", None) or "" litellm_call_id = hidden_params.get("litellm_call_id", None) or "" - # Printing each chunk size - async def generate(_response: HttpxBinaryResponseContent): - _generator = await _response.aiter_bytes(chunk_size=4096) - async for chunk in _generator: - yield chunk - custom_headers = ProxyBaseLLMRequestProcessing.get_custom_headers( user_api_key_dict=user_api_key_dict, model_id=model_id, @@ -5337,7 +5340,9 @@ async def audio_speech( media_type = "audio/wav" # Gemini TTS returns WAV format after conversion return StreamingResponse( - generate(response), media_type=media_type, headers=custom_headers # type: ignore + _audio_speech_chunk_generator(response), # type: ignore[arg-type] + media_type=media_type, + headers=custom_headers, # type: ignore ) except Exception as e: From 8ea0e31678863d2d700bf857bedac4d25338e008 Mon Sep 17 00:00:00 2001 From: AlexsanderHamir Date: Sat, 15 Nov 2025 17:30:58 -0800 Subject: [PATCH 004/121] Optimize streaming response accumulation Refactor async_data_generator to build streamed text via list accumulation and ''.join() instead of repeated string concatenation. This improves performance for long responses without changing streaming behavior. --- litellm/proxy/proxy_server.py | 7 ++++--- 1 file changed, 4 insertions(+), 3 deletions(-) diff --git a/litellm/proxy/proxy_server.py b/litellm/proxy/proxy_server.py index c65c7f77f0d..b67de4a87a7 100644 --- a/litellm/proxy/proxy_server.py +++ b/litellm/proxy/proxy_server.py @@ -4017,7 +4017,8 @@ async def async_data_generator( ): verbose_proxy_logger.debug("inside generator") try: - str_so_far = "" + # Use a list to accumulate response segments to avoid O(n^2) string concatenation + str_so_far_parts: list[str] = [] error_message: Optional[str] = None async for chunk in proxy_logging_obj.async_post_call_streaming_iterator_hook( user_api_key_dict=user_api_key_dict, @@ -4033,12 +4034,12 @@ async def async_data_generator( user_api_key_dict=user_api_key_dict, response=chunk, data=request_data, - str_so_far=str_so_far, + str_so_far="".join(str_so_far_parts), ) if isinstance(chunk, (ModelResponse, ModelResponseStream)): response_str = litellm.get_response_string(response_obj=chunk) - str_so_far += response_str + str_so_far_parts.append(response_str) if isinstance(chunk, BaseModel): chunk = chunk.model_dump_json(exclude_none=True, exclude_unset=True) From 98e2b64040f6e5f882648b433a23799582d710a1 Mon Sep 17 00:00:00 2001 From: AlexsanderHamir Date: Sat, 15 Nov 2025 17:37:01 -0800 Subject: [PATCH 005/121] Optimize response string construction Use list accumulation and join in get_response_string to avoid O(n^2) string concatenation and add a brief comment explaining the performance rationale. --- litellm/utils.py | 20 ++++++++++++-------- 1 file changed, 12 insertions(+), 8 deletions(-) diff --git a/litellm/utils.py b/litellm/utils.py index 783d462a7af..201e8145254 100644 --- a/litellm/utils.py +++ b/litellm/utils.py @@ -4437,17 +4437,20 @@ def get_response_string(response_obj: Union[ModelResponse, ModelResponseStream]) responses_api_response = getattr(response_obj, "response", None) if responses_api_response and hasattr(responses_api_response, "output"): output_list = responses_api_response.output - response_str = "" + # Use list accumulation to avoid O(n^2) string concatenation: + # repeatedly doing `response_str += part` copies the full string each time + # because Python strings are immutable, so total work grows with n^2. + response_output_parts: List[str] = [] for output_item in output_list: # Handle output items with content array if hasattr(output_item, "content"): for content_part in output_item.content: if hasattr(content_part, "text"): - response_str += content_part.text + response_output_parts.append(content_part.text) # Handle output items with direct text field elif hasattr(output_item, "text"): - response_str += output_item.text - return response_str + response_output_parts.append(output_item.text) + return "".join(response_output_parts) # Handle Responses API text delta events if hasattr(response_obj, "type") and hasattr(response_obj, "delta"): @@ -4461,16 +4464,17 @@ def get_response_string(response_obj: Union[ModelResponse, ModelResponseStream]) response_obj.choices ) - response_str = "" + # Use list accumulation to avoid O(n^2) string concatenation across choices + response_parts: List[str] = [] for choice in _choices: if isinstance(choice, Choices): if choice.message.content is not None: - response_str += choice.message.content + response_parts.append(str(choice.message.content)) elif isinstance(choice, StreamingChoices): if choice.delta.content is not None: - response_str += choice.delta.content + response_parts.append(str(choice.delta.content)) - return response_str + return "".join(response_parts) def get_api_key(llm_provider: str, dynamic_api_key: Optional[str]): From 4614e528dc4b3385582810f3c565d62668373d3c Mon Sep 17 00:00:00 2001 From: AlexsanderHamir Date: Mon, 17 Nov 2025 10:11:55 -0800 Subject: [PATCH 006/121] fix: remove deadcode The optimizations related to `select_data_generator` had no effect because its output which is the generator wasn't being used. --- litellm/proxy/proxy_server.py | 5 ----- 1 file changed, 5 deletions(-) diff --git a/litellm/proxy/proxy_server.py b/litellm/proxy/proxy_server.py index b67de4a87a7..d6a98a1ca5b 100644 --- a/litellm/proxy/proxy_server.py +++ b/litellm/proxy/proxy_server.py @@ -5327,11 +5327,6 @@ async def audio_speech( hidden_params=hidden_params, ) - select_data_generator( - response=response, - user_api_key_dict=user_api_key_dict, - request_data=data, - ) # Determine media type based on model type media_type = "audio/mpeg" # Default for OpenAI TTS request_model = data.get("model", "") From c8c12298590885bc845088d4982c7059996f04a9 Mon Sep 17 00:00:00 2001 From: AlexsanderHamir Date: Mon, 17 Nov 2025 10:24:34 -0800 Subject: [PATCH 007/121] fix: call_type mistake & remove repetitive .lower() calls --- litellm/proxy/proxy_server.py | 12 +++++++----- 1 file changed, 7 insertions(+), 5 deletions(-) diff --git a/litellm/proxy/proxy_server.py b/litellm/proxy/proxy_server.py index d6a98a1ca5b..cb43ef0cecf 100644 --- a/litellm/proxy/proxy_server.py +++ b/litellm/proxy/proxy_server.py @@ -5286,7 +5286,7 @@ async def audio_speech( ### CALL HOOKS ### - modify incoming data / reject request before calling the model data = await proxy_logging_obj.pre_call_hook( - user_api_key_dict=user_api_key_dict, data=data, call_type="image_generation" + user_api_key_dict=user_api_key_dict, data=data, call_type="aspeech" ) ## ROUTE TO CORRECT ENDPOINT ## @@ -5330,10 +5330,12 @@ async def audio_speech( # Determine media type based on model type media_type = "audio/mpeg" # Default for OpenAI TTS request_model = data.get("model", "") - if "gemini" in request_model.lower() and ( - "tts" in request_model.lower() or "preview-tts" in request_model.lower() - ): - media_type = "audio/wav" # Gemini TTS returns WAV format after conversion + if request_model: + request_model_lower = request_model.lower() + if "gemini" in request_model_lower and ( + "tts" in request_model_lower or "preview-tts" in request_model_lower + ): + media_type = "audio/wav" # Gemini TTS returns WAV format after conversion return StreamingResponse( _audio_speech_chunk_generator(response), # type: ignore[arg-type] From 697cb0906011cb84f2cc8c34b4ff7a2d7402dc13 Mon Sep 17 00:00:00 2001 From: AlexsanderHamir Date: Mon, 17 Nov 2025 10:51:18 -0800 Subject: [PATCH 008/121] fix: shared_sessions not being used --- litellm/llms/openai/openai.py | 5 +++++ litellm/main.py | 2 ++ 2 files changed, 7 insertions(+) diff --git a/litellm/llms/openai/openai.py b/litellm/llms/openai/openai.py index 2949e35e5e7..3282b7665c0 100644 --- a/litellm/llms/openai/openai.py +++ b/litellm/llms/openai/openai.py @@ -1414,6 +1414,7 @@ class OpenAIChatCompletion(BaseLLM, BaseOpenAILLM): timeout: Union[float, httpx.Timeout], aspeech: Optional[bool] = None, client=None, + shared_session: Optional["ClientSession"] = None, ) -> HttpxBinaryResponseContent: if aspeech is not None and aspeech is True: return self.async_audio_speech( @@ -1428,6 +1429,7 @@ class OpenAIChatCompletion(BaseLLM, BaseOpenAILLM): max_retries=max_retries, timeout=timeout, client=client, + shared_session=shared_session, ) # type: ignore openai_client = self._get_openai_client( @@ -1437,6 +1439,7 @@ class OpenAIChatCompletion(BaseLLM, BaseOpenAILLM): timeout=timeout, max_retries=max_retries, client=client, + shared_session=shared_session, ) response = cast(OpenAI, openai_client).audio.speech.create( @@ -1460,6 +1463,7 @@ class OpenAIChatCompletion(BaseLLM, BaseOpenAILLM): max_retries: int, timeout: Union[float, httpx.Timeout], client=None, + shared_session: Optional["ClientSession"] = None, ) -> HttpxBinaryResponseContent: openai_client = cast( AsyncOpenAI, @@ -1470,6 +1474,7 @@ class OpenAIChatCompletion(BaseLLM, BaseOpenAILLM): timeout=timeout, max_retries=max_retries, client=client, + shared_session=shared_session, ), ) diff --git a/litellm/main.py b/litellm/main.py index 14d0b04b7b5..412d7f1c38e 100644 --- a/litellm/main.py +++ b/litellm/main.py @@ -5747,6 +5747,7 @@ def speech( # noqa: PLR0915 proxy_server_request = kwargs.get("proxy_server_request", None) extra_headers = kwargs.get("extra_headers", None) model_info = kwargs.get("model_info", None) + shared_session = kwargs.get("shared_session", None) model, custom_llm_provider, dynamic_api_key, api_base = get_llm_provider( model=model, custom_llm_provider=custom_llm_provider, api_base=api_base ) # type: ignore @@ -5856,6 +5857,7 @@ def speech( # noqa: PLR0915 timeout=timeout, client=client, # pass AsyncOpenAI, OpenAI client aspeech=aspeech, + shared_session=shared_session, ) elif custom_llm_provider == "azure": # Check if this is Azure Speech Service (Cognitive Services TTS) From f1895265e6b5643ef0e5be97b6f4cd65fcc2e78f Mon Sep 17 00:00:00 2001 From: AlexsanderHamir Date: Mon, 17 Nov 2025 12:55:58 -0800 Subject: [PATCH 009/121] fix: increase chunk_size to 8 KB for optimal latency --- litellm/proxy/proxy_server.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/litellm/proxy/proxy_server.py b/litellm/proxy/proxy_server.py index cb43ef0cecf..16988205530 100644 --- a/litellm/proxy/proxy_server.py +++ b/litellm/proxy/proxy_server.py @@ -5236,7 +5236,7 @@ async def moderations( async def _audio_speech_chunk_generator( _response: HttpxBinaryResponseContent, ) -> AsyncGenerator[bytes, None]: - _generator = await _response.aiter_bytes(chunk_size=4096) + _generator = await _response.aiter_bytes(chunk_size=8192) async for chunk in _generator: yield chunk From 7241b4e9b505ea433563a88916cece2d062cd063 Mon Sep 17 00:00:00 2001 From: AlexsanderHamir Date: Mon, 17 Nov 2025 12:59:48 -0800 Subject: [PATCH 010/121] add: comment above optimization For anybody that would change this value for whatever reason, the comment makes the tradeoff clear. --- litellm/proxy/proxy_server.py | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/litellm/proxy/proxy_server.py b/litellm/proxy/proxy_server.py index 16988205530..da592c90720 100644 --- a/litellm/proxy/proxy_server.py +++ b/litellm/proxy/proxy_server.py @@ -5236,6 +5236,10 @@ async def moderations( async def _audio_speech_chunk_generator( _response: HttpxBinaryResponseContent, ) -> AsyncGenerator[bytes, None]: + # chunk_size has a big impact on latency, it can't be too small or too large + # too small: latency is high + # too large: latency is low, but memory usage is high + # 8192 is a good compromise _generator = await _response.aiter_bytes(chunk_size=8192) async for chunk in _generator: yield chunk From b4e25a68a4690e93110fb4ca2c4aa56f1c08a25a Mon Sep 17 00:00:00 2001 From: AlexsanderHamir Date: Tue, 18 Nov 2025 09:40:26 -0800 Subject: [PATCH 011/121] fix: remove test files --- no_cache_hits.py | 48 --------------------------------------------- speech.mp3 | Bin 104 -> 0 bytes speech_config.yaml | 9 --------- 3 files changed, 57 deletions(-) delete mode 100644 no_cache_hits.py delete mode 100644 speech.mp3 delete mode 100644 speech_config.yaml diff --git a/no_cache_hits.py b/no_cache_hits.py deleted file mode 100644 index 1b3bf895f77..00000000000 --- a/no_cache_hits.py +++ /dev/null @@ -1,48 +0,0 @@ -from locust import HttpUser, between, task - - -class MyUser(HttpUser): - """ - Minimal Locust user for repeatedly hitting `/v1/audio/speech`. - The goal is to measure server-side performance, so we avoid any extra work - (file writes, random generation, manual timing, custom event hooks, etc.) - that could inflate client-side latency. - """ - - wait_time = between(0.5, 1) - host = "http://0.0.0.0:8090" - - def on_start(self): - self.api_key = "sk-1234" - self.model_name = "fake-openai-speech" - self.headers = { - "Authorization": f"Bearer {self.api_key}", - "Content-Type": "application/json", - } - self.prompt_counter = 0 - - @task - def audio_speech_request(self): - self.prompt_counter += 1 - # Ensure prompts differ slightly so the backend can't reuse cached audio. - prompt = ( - "Generate a short spoken status update mentioning counter " - f"{self.prompt_counter}." - ) - - response = self.client.post( - "v1/audio/speech", - json={ - "model": self.model_name, - "input": prompt, - "voice": "alloy", - "format": "mp3", - }, - headers=self.headers, - name="audio_speech", - ) - - if response.status_code != 200: - # log the errors in error.txt - with open("error.txt", "a") as error_log: - error_log.write(response.text + "\n") \ No newline at end of file diff --git a/speech.mp3 b/speech.mp3 deleted file mode 100644 index f4f854d9bd215e2493d48b4bc4d39804bd79c038..0000000000000000000000000000000000000000 GIT binary patch literal 0 HcmV?d00001 literal 104 NcmezWdjbPJ000JT0*e3u diff --git a/speech_config.yaml b/speech_config.yaml deleted file mode 100644 index ad9920a2793..00000000000 --- a/speech_config.yaml +++ /dev/null @@ -1,9 +0,0 @@ -model_list: - - model_name: fake-openai-speech - litellm_params: - model: openai/gpt-4o-mini-tts - api_base: http://0.0.0.0:8090/ - api_key: sk-1234 - model_info: - mode: audio_speech - \ No newline at end of file From 1c67b7e1daf28a9d9afc24d647adfb35c71d322b Mon Sep 17 00:00:00 2001 From: AlexsanderHamir Date: Thu, 20 Nov 2025 17:48:40 -0800 Subject: [PATCH 012/121] fix: place hardcoded value on constants.py --- litellm/constants.py | 1 + litellm/proxy/proxy_server.py | 3 ++- 2 files changed, 3 insertions(+), 1 deletion(-) diff --git a/litellm/constants.py b/litellm/constants.py index 3f763cad926..fc26e1cf817 100644 --- a/litellm/constants.py +++ b/litellm/constants.py @@ -246,6 +246,7 @@ TOGETHER_AI_EMBEDDING_350_M = int(os.getenv("TOGETHER_AI_EMBEDDING_350_M", 350)) QDRANT_SCALAR_QUANTILE = float(os.getenv("QDRANT_SCALAR_QUANTILE", 0.99)) QDRANT_VECTOR_SIZE = int(os.getenv("QDRANT_VECTOR_SIZE", 1536)) CACHED_STREAMING_CHUNK_DELAY = float(os.getenv("CACHED_STREAMING_CHUNK_DELAY", 0.02)) +AUDIO_SPEECH_CHUNK_SIZE = 8192 # chunk_size for audio speech streaming. Balance between latency and memory usage MAX_SIZE_PER_ITEM_IN_MEMORY_CACHE_IN_KB = int( os.getenv("MAX_SIZE_PER_ITEM_IN_MEMORY_CACHE_IN_KB", 512) ) diff --git a/litellm/proxy/proxy_server.py b/litellm/proxy/proxy_server.py index da592c90720..47c30e9ce84 100644 --- a/litellm/proxy/proxy_server.py +++ b/litellm/proxy/proxy_server.py @@ -32,6 +32,7 @@ from litellm.constants import ( AIOHTTP_CONNECTOR_LIMIT, AIOHTTP_KEEPALIVE_TIMEOUT, AIOHTTP_TTL_DNS_CACHE, + AUDIO_SPEECH_CHUNK_SIZE, BASE_MCP_ROUTE, DEFAULT_MAX_RECURSE_DEPTH, DEFAULT_SHARED_HEALTH_CHECK_LOCK_TTL, @@ -5240,7 +5241,7 @@ async def _audio_speech_chunk_generator( # too small: latency is high # too large: latency is low, but memory usage is high # 8192 is a good compromise - _generator = await _response.aiter_bytes(chunk_size=8192) + _generator = await _response.aiter_bytes(chunk_size=AUDIO_SPEECH_CHUNK_SIZE) async for chunk in _generator: yield chunk From dba9946b989a6d76f563878219f3246009e1a1a0 Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Wed, 26 Nov 2025 21:04:15 +0530 Subject: [PATCH 013/121] Update new feats as reviewed --- litellm/llms/anthropic/chat/transformation.py | 17 +++- litellm/llms/anthropic/common_utils.py | 84 ++++++++++++++++++- .../bedrock/chat/converse_transformation.py | 16 +++- .../anthropic_claude3_transformation.py | 52 +++++++++++- .../anthropic_claude3_transformation.py | 42 +++++++++- .../anthropic/transformation.py | 20 +++++ 6 files changed, 213 insertions(+), 18 deletions(-) diff --git a/litellm/llms/anthropic/chat/transformation.py b/litellm/llms/anthropic/chat/transformation.py index ac1c9b1e000..4221dfacf34 100644 --- a/litellm/llms/anthropic/chat/transformation.py +++ b/litellm/llms/anthropic/chat/transformation.py @@ -119,6 +119,10 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig): def get_config(cls): return super().get_config() + def _is_claude_opus_4_5(self, model: str) -> bool: + """Check if the model is Claude Opus 4.5.""" + return "opus-4-5" in model.lower() or "opus_4_5" in model.lower() + def get_supported_openai_params(self, model: str): params = [ "stream", @@ -626,7 +630,7 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig): return hosted_web_search_tool - def map_openai_params( + def map_openai_params( # noqa: PLR0915 self, non_default_params: dict, optional_params: dict, @@ -712,9 +716,14 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig): if param == "thinking": optional_params["thinking"] = value elif param == "reasoning_effort" and isinstance(value, str): - optional_params["thinking"] = AnthropicConfig._map_reasoning_effort( - value - ) + # For Claude Opus 4.5, map reasoning_effort to output_config + if self._is_claude_opus_4_5(model): + optional_params["output_config"] = {"effort": value} + else: + # For other models, map to thinking parameter + optional_params["thinking"] = AnthropicConfig._map_reasoning_effort( + value + ) elif param == "web_search_options" and isinstance(value, dict): hosted_web_search_tool = self.map_web_search_tool( cast(OpenAIWebSearchOptions, value) diff --git a/litellm/llms/anthropic/common_utils.py b/litellm/llms/anthropic/common_utils.py index 9f5688f9e01..6b339f169cd 100644 --- a/litellm/llms/anthropic/common_utils.py +++ b/litellm/llms/anthropic/common_utils.py @@ -151,15 +151,22 @@ class AnthropicModelInfo(BaseLLMModelInfo): return False - def is_effort_used(self, optional_params: Optional[dict]) -> bool: + def is_effort_used(self, optional_params: Optional[dict], model: Optional[str] = None) -> bool: """ - Check if effort parameter is being used via output_config. + Check if effort parameter is being used. - Returns True if output_config with effort field is present. + Returns True if effort-related parameters are present. """ if not optional_params: return False + # Check if reasoning_effort is provided for Claude Opus 4.5 + if model and ("opus-4-5" in model.lower() or "opus_4_5" in model.lower()): + reasoning_effort = optional_params.get("reasoning_effort") + if reasoning_effort and isinstance(reasoning_effort, str): + return True + + # Check if output_config is directly provided output_config = optional_params.get("output_config") if output_config and isinstance(output_config, dict): effort = output_config.get("effort") @@ -193,6 +200,75 @@ class AnthropicModelInfo(BaseLLMModelInfo): computer_tool_version, "computer-use-2024-10-22" # Default fallback ) + def get_anthropic_beta_list( + self, + model: str, + custom_llm_provider: str, + tools: Optional[List] = None, + optional_params: Optional[dict] = None, + computer_tool_used: Optional[str] = None, + prompt_caching_set: bool = False, + file_id_used: bool = False, + mcp_server_used: bool = False, + ) -> List[str]: + """ + Get list of beta headers based on provider and features used. + + This method provides provider-specific beta header values for different Anthropic features. + Different providers (Anthropic API, Bedrock, VertexAI, Microsoft Foundry) may require + different beta header values for the same feature. + + Returns: + List of beta header strings + """ + from litellm.types.llms.anthropic import ( + ANTHROPIC_EFFORT_BETA_HEADER, + ANTHROPIC_TOOL_SEARCH_BETA_HEADER, + ) + + betas = [] + + # Detect features + tool_search_used = self.is_tool_search_used(tools) + programmatic_tool_calling_used = self.is_programmatic_tool_calling_used(tools) + input_examples_used = self.is_input_examples_used(tools) + effort_used = self.is_effort_used(optional_params, model) + + # Add beta headers based on provider + if custom_llm_provider in ["vertex_ai", "vertex_ai_beta"]: + if tool_search_used: + betas.append("tool-search-tool-2025-10-19") + # VertexAI doesn't support programmatic tool calling or input_examples yet + elif custom_llm_provider == "bedrock": + # Bedrock: tool-search only for Opus 4.5, advanced-tool-use for programmatic/input_examples + if tool_search_used and ("opus-4" in model.lower() or "opus_4" in model.lower()): + betas.append("tool-search-tool-2025-10-19") + if programmatic_tool_calling_used or input_examples_used: + betas.append(ANTHROPIC_TOOL_SEARCH_BETA_HEADER) # advanced-tool-use-2025-11-20 + else: # anthropic, azure (Microsoft Foundry), and others + # Direct API and Microsoft Foundry use advanced-tool-use for all + if tool_search_used or programmatic_tool_calling_used or input_examples_used: + betas.append(ANTHROPIC_TOOL_SEARCH_BETA_HEADER) # advanced-tool-use-2025-11-20 + + if effort_used: + betas.append(ANTHROPIC_EFFORT_BETA_HEADER) # effort-2025-11-24 + + if computer_tool_used: + beta_header = self.get_computer_tool_beta_header(computer_tool_used) + betas.append(beta_header) + + if prompt_caching_set: + betas.append("prompt-caching-2024-07-31") + + if file_id_used: + betas.append("files-api-2025-04-14") + betas.append("code-execution-2025-05-22") + + if mcp_server_used: + betas.append("mcp-client-2025-04-04") + + return list(set(betas)) + def get_anthropic_headers( self, api_key: str, @@ -278,7 +354,7 @@ class AnthropicModelInfo(BaseLLMModelInfo): tool_search_used = self.is_tool_search_used(tools=tools) programmatic_tool_calling_used = self.is_programmatic_tool_calling_used(tools=tools) input_examples_used = self.is_input_examples_used(tools=tools) - effort_used = self.is_effort_used(optional_params=optional_params) + effort_used = self.is_effort_used(optional_params=optional_params, model=model) user_anthropic_beta_headers = self._get_user_anthropic_beta_headers( anthropic_beta_header=headers.get("anthropic-beta") ) diff --git a/litellm/llms/bedrock/chat/converse_transformation.py b/litellm/llms/bedrock/chat/converse_transformation.py index d76a3c31b51..759311c38fe 100644 --- a/litellm/llms/bedrock/chat/converse_transformation.py +++ b/litellm/llms/bedrock/chat/converse_transformation.py @@ -815,11 +815,21 @@ class AmazonConverseConfig(BaseConfig): user_betas = get_anthropic_beta_from_headers(headers) anthropic_beta_list.extend(user_betas) + # Filter out tool search tools - Bedrock Converse API doesn't support them + filtered_tools = [] + if original_tools: + for tool in original_tools: + tool_type = tool.get("type", "") + if tool_type in ("tool_search_tool_regex_20251119", "tool_search_tool_bm25_20251119"): + # Tool search not supported in Converse API - skip it + continue + filtered_tools.append(tool) + # Only separate tools if computer use tools are actually present - if original_tools and self.is_computer_use_tool_used(original_tools, model): + if filtered_tools and self.is_computer_use_tool_used(filtered_tools, model): # Separate computer use tools from regular function tools computer_use_tools, regular_tools = self._separate_computer_use_tools( - original_tools, model + filtered_tools, model ) # Process regular function tools using existing logic @@ -835,7 +845,7 @@ class AmazonConverseConfig(BaseConfig): additional_request_params["tools"] = transformed_computer_tools else: # No computer use tools, process all tools as regular tools - bedrock_tools = _bedrock_tools_pt(original_tools) + bedrock_tools = _bedrock_tools_pt(filtered_tools) # Set anthropic_beta in additional_request_params if we have any beta features if anthropic_beta_list: diff --git a/litellm/llms/bedrock/chat/invoke_transformations/anthropic_claude3_transformation.py b/litellm/llms/bedrock/chat/invoke_transformations/anthropic_claude3_transformation.py index 02b8fd57115..d618451f73e 100644 --- a/litellm/llms/bedrock/chat/invoke_transformations/anthropic_claude3_transformation.py +++ b/litellm/llms/bedrock/chat/invoke_transformations/anthropic_claude3_transformation.py @@ -76,6 +76,7 @@ class AmazonAnthropicClaudeConfig(AmazonInvokeConfig, AnthropicConfig): for k, v in optional_params.items() if k not in self.aws_authentication_params } + filtered_params = self._normalize_bedrock_tool_search_tools(filtered_params) _anthropic_request = AnthropicConfig.transform_request( self, @@ -91,13 +92,58 @@ class AmazonAnthropicClaudeConfig(AmazonInvokeConfig, AnthropicConfig): if "anthropic_version" not in _anthropic_request: _anthropic_request["anthropic_version"] = self.anthropic_version - # Handle anthropic_beta from user headers - anthropic_beta_list = get_anthropic_beta_from_headers(headers) + anthropic_beta_list = [] + + user_betas = get_anthropic_beta_from_headers(headers) + if user_betas: + anthropic_beta_list.extend(user_betas) + + # Auto-detect and add beta headers using the new method + tools = optional_params.get("tools") + auto_betas = self.get_anthropic_beta_list( + model=model, + custom_llm_provider=self.custom_llm_provider or "bedrock", + tools=tools, + optional_params=optional_params, + computer_tool_used=self.is_computer_tool_used(tools), + prompt_caching_set=self.is_cache_control_set(messages), + file_id_used=self.is_file_id_used(messages), + mcp_server_used=self.is_mcp_server_used(optional_params.get("mcp_servers")), + ) + anthropic_beta_list.extend(auto_betas) + if anthropic_beta_list: - _anthropic_request["anthropic_beta"] = anthropic_beta_list + _anthropic_request["anthropic_beta"] = list(set(anthropic_beta_list)) return _anthropic_request + def _normalize_bedrock_tool_search_tools(self, optional_params: dict) -> dict: + """ + Convert tool search entries to the format supported by the Bedrock Invoke API. + """ + tools = optional_params.get("tools") + if not tools or not isinstance(tools, list): + return optional_params + + normalized_tools = [] + for tool in tools: + tool_type = tool.get("type") + if tool_type == "tool_search_tool_bm25_20251119": + # Bedrock Invoke does not support the BM25 variant, so skip it. + continue + if tool_type == "tool_search_tool_regex_20251119": + normalized_tool = tool.copy() + normalized_tool["type"] = "tool_search_tool_regex" + normalized_tool["name"] = normalized_tool.get( + "name", "tool_search_tool_regex" + ) + normalized_tools.append(normalized_tool) + continue + normalized_tools.append(tool) + + optional_params["tools"] = normalized_tools + return optional_params + def transform_response( self, model: str, diff --git a/litellm/llms/bedrock/messages/invoke_transformations/anthropic_claude3_transformation.py b/litellm/llms/bedrock/messages/invoke_transformations/anthropic_claude3_transformation.py index be782d35766..6f5165241ca 100644 --- a/litellm/llms/bedrock/messages/invoke_transformations/anthropic_claude3_transformation.py +++ b/litellm/llms/bedrock/messages/invoke_transformations/anthropic_claude3_transformation.py @@ -1,7 +1,18 @@ -from typing import TYPE_CHECKING, Any, AsyncIterator, Dict, List, Optional, Tuple, Union +from typing import ( + TYPE_CHECKING, + Any, + AsyncIterator, + Dict, + List, + Optional, + Tuple, + Union, + cast, +) import httpx +from litellm.llms.anthropic.common_utils import AnthropicModelInfo from litellm.llms.anthropic.experimental_pass_through.messages.transformation import ( AnthropicMessagesConfig, ) @@ -13,6 +24,7 @@ from litellm.llms.bedrock.chat.invoke_transformations.base_invoke_transformation AmazonInvokeConfig, ) from litellm.llms.bedrock.common_utils import get_anthropic_beta_from_headers +from litellm.types.llms.openai import AllMessageValues from litellm.types.router import GenericLiteLLMParams from litellm.types.utils import GenericStreamingChunk from litellm.types.utils import GenericStreamingChunk as GChunk @@ -129,10 +141,32 @@ class AmazonAnthropicClaudeMessagesConfig( if "model" in anthropic_messages_request: anthropic_messages_request.pop("model", None) - # 4. Handle anthropic_beta from user headers - anthropic_beta_list = get_anthropic_beta_from_headers(headers) + # 4. AUTO-INJECT beta headers based on features used + anthropic_beta_list = [] + + # Get user-provided beta headers first + user_betas = get_anthropic_beta_from_headers(headers) + if user_betas: + anthropic_beta_list.extend(user_betas) + + anthropic_model_info = AnthropicModelInfo() + tools = anthropic_messages_optional_request_params.get("tools") + messages_typed = cast(List[AllMessageValues], messages) + auto_betas = anthropic_model_info.get_anthropic_beta_list( + model=model, + custom_llm_provider="bedrock", + tools=tools, + optional_params=anthropic_messages_optional_request_params, + computer_tool_used=anthropic_model_info.is_computer_tool_used(tools), + prompt_caching_set=anthropic_model_info.is_cache_control_set(messages_typed), + file_id_used=anthropic_model_info.is_file_id_used(messages_typed), + mcp_server_used=anthropic_model_info.is_mcp_server_used(anthropic_messages_optional_request_params.get("mcp_servers")), + ) + anthropic_beta_list.extend(auto_betas) + + # Remove duplicates and set in request body if any beta headers exist if anthropic_beta_list: - anthropic_messages_request["anthropic_beta"] = anthropic_beta_list + anthropic_messages_request["anthropic_beta"] = list(set(anthropic_beta_list)) return anthropic_messages_request diff --git a/litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/transformation.py b/litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/transformation.py index 7ba788e335c..69651ca4358 100644 --- a/litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/transformation.py +++ b/litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/transformation.py @@ -68,6 +68,26 @@ class VertexAIAnthropicConfig(AnthropicConfig): ) data.pop("model", None) # vertex anthropic doesn't accept 'model' parameter + + tools = optional_params.get("tools") + anthropic_beta_list = [] + + auto_betas = self.get_anthropic_beta_list( + model=model, + custom_llm_provider=self.custom_llm_provider or "vertex_ai", + tools=tools, + optional_params=optional_params, + computer_tool_used=self.is_computer_tool_used(tools), + prompt_caching_set=self.is_cache_control_set(messages), + file_id_used=self.is_file_id_used(messages), + mcp_server_used=self.is_mcp_server_used(optional_params.get("mcp_servers")), + ) + anthropic_beta_list.extend(auto_betas) + + # Note: VertexAI uses tool-search-tool-2025-10-19 for tool search (different from direct API) + if anthropic_beta_list: + data["anthropic_beta"] = list(set(anthropic_beta_list)) + return data def transform_response( From c7ef668d783870b485fd3b0c9211c3e844944d42 Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Wed, 26 Nov 2025 21:18:47 +0530 Subject: [PATCH 014/121] Update documentation for azure 4 feats --- .../index.md | 301 +----------------- docs/my-website/docs/providers/anthropic.md | 4 +- .../docs/providers/anthropic_effort.md | 27 +- .../anthropic_programmatic_tool_calling.md | 17 +- .../anthropic_tool_input_examples.md | 19 +- .../docs/providers/anthropic_tool_search.md | 17 +- 6 files changed, 67 insertions(+), 318 deletions(-) diff --git a/docs/my-website/blog/anthropic_opus_4_5_and_advanced_features/index.md b/docs/my-website/blog/anthropic_opus_4_5_and_advanced_features/index.md index b545e936186..6df4823b188 100644 --- a/docs/my-website/blog/anthropic_opus_4_5_and_advanced_features/index.md +++ b/docs/my-website/blog/anthropic_opus_4_5_and_advanced_features/index.md @@ -897,14 +897,13 @@ curl --location 'http://0.0.0.0:4000/chat/completions' \ ## Effort Parameter: Control Token Usage {#effort-parameter} -Controls aspects like how much effort the model puts into its response, via `output_config={"effort": ..}`. +Control how much effort Claude puts into its response using the `reasoning_effort` parameter. This allows you to trade off between response thoroughness and token efficiency. :::info - -Soon, we will map OpenAI's `reasoning_effort` parameter to this. +LiteLLM automatically maps `reasoning_effort` to Anthropic's `output_config` format and adds the required `effort-2025-11-24` beta header for Claude Opus 4.5. ::: -Potential Values for `effort` parameter: `"high"`, `"medium"`, `"low"`. +Potential values for `reasoning_effort` parameter: `"high"`, `"medium"`, `"low"`. ### Usage Example @@ -920,7 +919,7 @@ message = "Analyze the trade-offs between microservices and monolithic architect response_high = litellm.completion( model="anthropic/claude-opus-4-5-20251101", messages=[{"role": "user", "content": message}], - output_config={"effort": "high"} + reasoning_effort="high" ) print("High effort response:") @@ -931,7 +930,7 @@ print(f"Tokens used: {response_high.usage.completion_tokens}\n") response_medium = litellm.completion( model="anthropic/claude-opus-4-5-20251101", messages=[{"role": "user", "content": message}], - output_config={"effort": "medium"} + reasoning_effort="medium" ) print("Medium effort response:") @@ -942,7 +941,7 @@ print(f"Tokens used: {response_medium.usage.completion_tokens}\n") response_low = litellm.completion( model="anthropic/claude-opus-4-5-20251101", messages=[{"role": "user", "content": message}], - output_config={"effort": "low"} + reasoning_effort="low" ) print("Low effort response:") @@ -987,295 +986,9 @@ curl --location 'http://0.0.0.0:4000/chat/completions' \ "role": "user", "content": "Analyze the trade-offs between microservices and monolithic architectures" }], - "output_config": { - "effort": "high" - } + "reasoning_effort": "high" } ' ``` - - -## Cost Tracking: Monitor Tool Search Usage {#cost-tracking} - -### Understanding Tool Search Costs - -Tool search operations are tracked separately in the usage object, allowing you to monitor and optimize costs. - -It is available in the `usage` object, under `server_tool_use.tool_search_requests`. - -Anthropic charges $0.0001 per tool search request. - -### Tracking Example - - - - -```python -import litellm - -tools = [ - { - "type": "tool_search_tool_regex_20251119", - "name": "tool_search_tool_regex" - }, - # ... 100 deferred tools -] - -response = litellm.completion( - model="anthropic/claude-sonnet-4-5-20250929", - messages=[{ - "role": "user", - "content": "Find and use the weather tool for San Francisco" - }], - tools=tools -) - -# Standard token usage -print("Token Usage:") -print(f" Input tokens: {response.usage.prompt_tokens}") -print(f" Output tokens: {response.usage.completion_tokens}") -print(f" Total tokens: {response.usage.total_tokens}") - -# Tool search specific usage -if hasattr(response.usage, 'server_tool_use') and response.usage.server_tool_use: - print(f"\nTool Search Usage:") - print(f" Search requests: {response.usage.server_tool_use.tool_search_requests}") - - # Calculate cost (example pricing) - input_cost = response.usage.prompt_tokens * 0.000003 # $3 per 1M tokens - output_cost = response.usage.completion_tokens * 0.000015 # $15 per 1M tokens - search_cost = response.usage.server_tool_use.tool_search_requests * 0.0001 # Example - - total_cost = input_cost + output_cost + search_cost - - print(f"\nCost Breakdown:") - print(f" Input tokens: ${input_cost:.6f}") - print(f" Output tokens: ${output_cost:.6f}") - print(f" Tool searches: ${search_cost:.6f}") - print(f" Total: ${total_cost:.6f}") -``` - - - - -1. Setup config.yaml - -```yaml -model_list: - - model_name: claude-4 - litellm_params: - model: anthropic/claude-opus-4-5-20251101 - api_key: os.environ/ANTHROPIC_API_KEY -``` - -2. Start the proxy - -```bash -litellm --config /path/to/config.yaml -``` - -3. Test it! - -```bash -curl --location 'http://0.0.0.0:4000/chat/completions' \ ---header 'Content-Type: application/json' \ ---header 'Authorization: Bearer $LITELLM_KEY' \ ---data ' { - "model": "claude-4", - "messages": [{ - "role": "user", - "content": "Find and use the weather tool for San Francisco" - }], - "tools": [ - { - "type": "tool_search_tool_regex_20251119", - "name": "tool_search_tool_regex" - }, - # ... 100 deferred tools - ] - } -' -``` - -Expected Response: - -```json -{ - ..., - "usage": { - ..., - "server_tool_use": { - "tool_search_requests": 1 - } - } -} -``` - - - - -### Cost Optimization Tips - -1. **Keep frequently used tools non-deferred** (3-5 tools) -2. **Use tool search for large catalogs** (10+ tools) -3. **Monitor search requests** to identify optimization opportunities -4. **Combine with effort parameter** for maximum efficiency - - ---- - -## Combining Features {#combining-features} - -### The Power of Integration - -These features work together seamlessly. Here's a real-world example combining all of them: - - - - -```python -import litellm -import json - -# Large tool catalog with search, programmatic calling, and examples -tools = [ - # Enable tool search - { - "type": "tool_search_tool_regex_20251119", - "name": "tool_search_tool_regex" - }, - # Enable programmatic calling - { - "type": "code_execution_20250825", - "name": "code_execution" - }, - # Database tool with all features - { - "type": "function", - "function": { - "name": "query_database", - "description": "Execute SQL queries against the analytics database. Returns JSON array of results.", - "parameters": { - "type": "object", - "properties": { - "sql": { - "type": "string", - "description": "SQL SELECT statement" - }, - "limit": { - "type": "integer", - "description": "Maximum rows to return" - } - }, - "required": ["sql"] - } - }, - "defer_loading": True, # Tool search - "allowed_callers": ["code_execution_20250825"], # Programmatic calling - "input_examples": [ # Input examples - { - "sql": "SELECT region, SUM(revenue) as total FROM sales GROUP BY region", - "limit": 100 - } - ] - }, - # ... 50 more tools with defer_loading -] - -# Make request with effort control -response = litellm.completion( - model="anthropic/claude-opus-4-5-20251101", - messages=[{ - "role": "user", - "content": "Analyze sales by region for the last quarter and identify top performers" - }], - tools=tools, - output_config={"effort": "medium"} # Balanced efficiency -) - -# Track comprehensive usage -print("Complete Usage Metrics:") -print(f" Input tokens: {response.usage.prompt_tokens}") -print(f" Output tokens: {response.usage.completion_tokens}") -print(f" Total tokens: {response.usage.total_tokens}") - -if hasattr(response.usage, 'server_tool_use') and response.usage.server_tool_use: - print(f" Tool searches: {response.usage.server_tool_use.tool_search_requests}") - -print(f"\nResponse: {response.choices[0].message.content}") -``` - - - - -1. Setup config.yaml - -```yaml -model_list: - - model_name: claude-4 - litellm_params: - model: anthropic/claude-opus-4-5-20251101 - api_key: os.environ/ANTHROPIC_API_KEY -``` - -2. Start the proxy - -```bash -litellm --config /path/to/config.yaml -``` - -3. Test it! - -```bash -curl --location 'http://0.0.0.0:4000/chat/completions' \ ---header 'Content-Type: application/json' \ ---header 'Authorization: Bearer $LITELLM_KEY' \ ---data ' { - "model": "claude-4", - "messages": [{ - "role": "user", - "content": "Analyze sales by region for the last quarter and identify top performers" - }], - "tools": [ - { - "type": "tool_search_tool_regex_20251119", - "name": "tool_search_tool_regex" - }, - # ... 100 deferred tools - ], - "output_config": { - "effort": "medium" - } - } -' -``` - -Expected Response: - -```json -{ - ..., - "usage": { - ..., - "server_tool_use": { - "tool_search_requests": 1 - } - } -} -``` - - - - -### Real-World Benefits - -This combination enables: - -1. **Massive scale** - Handle 1000+ tools efficiently -2. **Low latency** - Programmatic calling reduces round trips -3. **High accuracy** - Input examples ensure correct tool usage -4. **Cost control** - Effort parameter optimizes token spend -5. **Full visibility** - Track all usage metrics - diff --git a/docs/my-website/docs/providers/anthropic.md b/docs/my-website/docs/providers/anthropic.md index 24365f0cc47..d84c1c23048 100644 --- a/docs/my-website/docs/providers/anthropic.md +++ b/docs/my-website/docs/providers/anthropic.md @@ -41,7 +41,8 @@ Check this in code, [here](../completion/input.md#translated-openai-params) "extra_headers", "parallel_tool_calls", "response_format", -"user" +"user", +"reasoning_effort", ``` :::info @@ -49,6 +50,7 @@ Check this in code, [here](../completion/input.md#translated-openai-params) **Notes:** - Anthropic API fails requests when `max_tokens` are not passed. Due to this litellm passes `max_tokens=4096` when no `max_tokens` are passed. - `response_format` is fully supported for Claude Sonnet 4.5 and Opus 4.1 models (see [Structured Outputs](#structured-outputs) section) +- `reasoning_effort` is automatically mapped to `output_config={"effort": ...}` for Claude Opus 4.5 models (see [Effort Parameter](./anthropic_effort.md)) ::: diff --git a/docs/my-website/docs/providers/anthropic_effort.md b/docs/my-website/docs/providers/anthropic_effort.md index 0015162a95b..e4bfd50e6c2 100644 --- a/docs/my-website/docs/providers/anthropic_effort.md +++ b/docs/my-website/docs/providers/anthropic_effort.md @@ -9,7 +9,10 @@ Control how many tokens Claude uses when responding with the `effort` parameter, The `effort` parameter allows you to control how eager Claude is about spending tokens when responding to requests. This gives you the ability to trade off between response thoroughness and token efficiency, all with a single model. -**Note**: The effort parameter is currently in beta and only supported by Claude Opus 4.5. You must include the beta header `effort-2025-11-24` when using this feature (LiteLLM automatically adds this header when `output_config` with `effort` is detected). +**Note**: The effort parameter is currently in beta and only supported by Claude Opus 4.5. LiteLLM automatically adds the `effort-2025-11-24` beta header when: +- `reasoning_effort` parameter is provided (for Claude Opus 4.5 only) + +For Claude Opus 4.5, `reasoning_effort="medium"`—both are automatically mapped to the correct format. ## How Effort Works @@ -52,9 +55,7 @@ response = litellm.completion( "role": "user", "content": "Analyze the trade-offs between microservices and monolithic architectures" }], - output_config={ - "effort": "medium" - } + reasoning_effort="medium" # Automatically mapped to output_config for Opus 4.5 ) print(response.choices[0].message.content) @@ -217,11 +218,14 @@ response = litellm.completion( The effort parameter is supported across all Anthropic-compatible providers: -- **Standard Anthropic**: ✅ Supported (Claude Opus 4.5) -- **Azure Anthropic**: ✅ Supported (Claude Opus 4.5) -- **Vertex AI Anthropic**: ✅ Supported (Claude Opus 4.5) +- **Standard Anthropic API**: ✅ Supported (Claude Opus 4.5) +- **Azure Anthropic / Microsoft Foundry**: ✅ Supported (Claude Opus 4.5) +- **Amazon Bedrock**: ✅ Supported (Claude Opus 4.5) +- **Google Cloud Vertex AI**: ✅ Supported (Claude Opus 4.5) -LiteLLM automatically handles the beta header injection for all providers. +LiteLLM automatically handles: +- Beta header injection (`effort-2025-11-24`) for all providers +- Parameter mapping: `reasoning_effort` → `output_config={"effort": ...}` for Claude Opus 4.5 ## Usage and Pricing @@ -242,9 +246,12 @@ print(f"Total tokens: {response.usage.total_tokens}") ### Beta header not being added -LiteLLM automatically adds the `effort-2025-11-24` beta header when `output_config` with `effort` is detected. If you're not seeing the header: +LiteLLM automatically adds the `effort-2025-11-24` beta header when: +- `reasoning_effort` parameter is provided (for Claude Opus 4.5 only) -1. Ensure you're using `output_config` with an `effort` field +If you're not seeing the header: + +1. Ensure you're using `reasoning_effort` parameter 2. Verify the model is Claude Opus 4.5 3. Check that LiteLLM version supports this feature diff --git a/docs/my-website/docs/providers/anthropic_programmatic_tool_calling.md b/docs/my-website/docs/providers/anthropic_programmatic_tool_calling.md index 6d3e15785e5..574dd7b0935 100644 --- a/docs/my-website/docs/providers/anthropic_programmatic_tool_calling.md +++ b/docs/my-website/docs/providers/anthropic_programmatic_tool_calling.md @@ -3,7 +3,11 @@ Programmatic tool calling allows Claude to write code that calls your tools programmatically within a code execution container, rather than requiring round trips through the model for each tool invocation. This reduces latency for multi-tool workflows and decreases token consumption by allowing Claude to filter or process data before it reaches the model's context window. :::info -Programmatic tool calling is currently in public beta. LiteLLM automatically adds the required `advanced-tool-use-2025-11-20` beta header when it detects tools with the `allowed_callers` field. +Programmatic tool calling is currently in public beta. LiteLLM automatically detects tools with the `allowed_callers` field and adds the appropriate beta header based on your provider: + +- **Anthropic API & Microsoft Foundry**: `advanced-tool-use-2025-11-20` +- **Amazon Bedrock**: `advanced-tool-use-2025-11-20` +- **Google Cloud Vertex AI**: Not supported This feature requires the code execution tool to be enabled. ::: @@ -380,13 +384,14 @@ For example, calling 10 tools directly uses ~10x the tokens of calling them prog ## Provider Support -LiteLLM supports programmatic tool calling across all Anthropic-compatible providers: +LiteLLM supports programmatic tool calling across the following Anthropic-compatible providers: -- **Standard Anthropic API** (`anthropic/claude-sonnet-4-5-20250929`) -- **Azure Anthropic** (`azure/claude-sonnet-4-5-20250929`) -- **Vertex AI Anthropic** (`vertex_ai/claude-sonnet-4-5-20250929`) +- **Standard Anthropic API** (`anthropic/claude-sonnet-4-5-20250929`) ✅ +- **Azure Anthropic / Microsoft Foundry** (`azure/claude-sonnet-4-5-20250929`) ✅ +- **Amazon Bedrock** (`bedrock/invoke/anthropic.claude-sonnet-4-5-20250929-v1:0`) ✅ +- **Google Cloud Vertex AI** (`vertex_ai/claude-sonnet-4-5-20250929`) ❌ Not supported -The beta header is automatically added when LiteLLM detects tools with `allowed_callers` field. +The beta header (`advanced-tool-use-2025-11-20`) is automatically added when LiteLLM detects tools with the `allowed_callers` field. ## Limitations diff --git a/docs/my-website/docs/providers/anthropic_tool_input_examples.md b/docs/my-website/docs/providers/anthropic_tool_input_examples.md index d0b7cc1762c..39f4d8555f4 100644 --- a/docs/my-website/docs/providers/anthropic_tool_input_examples.md +++ b/docs/my-website/docs/providers/anthropic_tool_input_examples.md @@ -3,7 +3,13 @@ Provide concrete examples of valid tool inputs to help Claude understand how to use your tools more effectively. This is particularly useful for complex tools with nested objects, optional parameters, or format-sensitive inputs. :::info -Tool input examples is a beta feature. LiteLLM automatically adds the required `advanced-tool-use-2025-11-20` beta header when it detects tools with the `input_examples` field. +Tool input examples is a beta feature. LiteLLM automatically detects tools with the `input_examples` field and adds the appropriate beta header based on your provider: + +- **Anthropic API & Microsoft Foundry**: `advanced-tool-use-2025-11-20` +- **Amazon Bedrock**: `advanced-tool-use-2025-11-20` (Claude Opus 4.5 only) +- **Google Cloud Vertex AI**: Not supported + +You don't need to manually specify beta headers—LiteLLM handles this automatically. ::: ## When to Use Input Examples @@ -378,13 +384,14 @@ Input examples work seamlessly with other Anthropic tool features: ## Provider Support -LiteLLM supports input examples across all Anthropic-compatible providers: +LiteLLM supports input examples across the following Anthropic-compatible providers: -- **Standard Anthropic API** (`anthropic/claude-sonnet-4-5-20250929`) -- **Azure Anthropic** (`azure/claude-sonnet-4-5-20250929`) -- **Vertex AI Anthropic** (`vertex_ai/claude-sonnet-4-5-20250929`) +- **Standard Anthropic API** (`anthropic/claude-sonnet-4-5-20250929`) ✅ +- **Azure Anthropic / Microsoft Foundry** (`azure/claude-sonnet-4-5-20250929`) ✅ +- **Amazon Bedrock** (`bedrock/invoke/anthropic.claude-opus-4-5-20251101-v1:0`) ✅ (Opus 4.5 only) +- **Google Cloud Vertex AI** (`vertex_ai/claude-sonnet-4-5-20250929`) ❌ Not supported -The beta header is automatically added when LiteLLM detects tools with `input_examples` field. +The beta header (`advanced-tool-use-2025-11-20`) is automatically added when LiteLLM detects tools with the `input_examples` field. ## Troubleshooting diff --git a/docs/my-website/docs/providers/anthropic_tool_search.md b/docs/my-website/docs/providers/anthropic_tool_search.md index 7b9e7cfaa72..28ce5688eeb 100644 --- a/docs/my-website/docs/providers/anthropic_tool_search.md +++ b/docs/my-website/docs/providers/anthropic_tool_search.md @@ -290,7 +290,13 @@ response = client.chat.completions.create( ### Beta Header -LiteLLM automatically adds the `advanced-tool-use-2025-11-20` beta header when tool search tools are detected. You don't need to manually specify it. +LiteLLM automatically detects tool search tools and adds the appropriate beta header based on your provider: + +- **Anthropic API & Microsoft Foundry**: `advanced-tool-use-2025-11-20` +- **Google Cloud Vertex AI**: `tool-search-tool-2025-10-19` +- **Amazon Bedrock** (Invoke API, Opus 4.5 only): `tool-search-tool-2025-10-19` + +You don't need to manually specify beta headers—LiteLLM handles this automatically. ### Deferred Loading @@ -387,9 +393,18 @@ If Claude references a tool that isn't in your deferred tools list, you'll get a - Not compatible with tool use examples - Requires Claude Opus 4.5 or Sonnet 4.5 - On Bedrock, only available via invoke API (not converse API) +- On Bedrock, only supported for Claude Opus 4.5 (not Sonnet 4.5) +- BM25 variant (`tool_search_tool_bm25_20251119`) is not supported on Bedrock - Maximum 10,000 tools in catalog - Returns 3-5 most relevant tools per search +### Bedrock-Specific Notes + +When using Bedrock's Invoke API: +- The regex variant (`tool_search_tool_regex_20251119`) is automatically normalized to `tool_search_tool_regex` +- The BM25 variant (`tool_search_tool_bm25_20251119`) is automatically filtered out as it's not supported +- Tool search is only available for Claude Opus 4.5 models + ## Additional Resources - [Anthropic Tool Search Documentation](https://docs.anthropic.com/en/docs/build-with-claude/tool-use/tool-search) From 210560e1e76f5d2f1ad4c94dcb3e41303c710ae2 Mon Sep 17 00:00:00 2001 From: AlexsanderHamir Date: Wed, 26 Nov 2025 15:48:43 -0800 Subject: [PATCH 015/121] Add paginated /spend/logs/v2 endpoint - Add /spend/logs/v2 endpoint that shares implementation with /spend/logs/ui - Provides paginated access to spend logs with comprehensive filtering - Replaces non-paginated /spend/logs endpoint to prevent performance issues - Both v2 and ui endpoints share the same function for consistency --- .../spend_management_endpoints.py | 27 ++++++++++++------- 1 file changed, 18 insertions(+), 9 deletions(-) diff --git a/litellm/proxy/spend_tracking/spend_management_endpoints.py b/litellm/proxy/spend_tracking/spend_management_endpoints.py index 5dfedcc0b87..2c9dc3fc09a 100644 --- a/litellm/proxy/spend_tracking/spend_management_endpoints.py +++ b/litellm/proxy/spend_tracking/spend_management_endpoints.py @@ -1,5 +1,6 @@ #### SPEND MANAGEMENT ##### import collections +import json import os from datetime import datetime, timedelta, timezone from functools import lru_cache @@ -1609,6 +1610,14 @@ async def calculate_spend(request: SpendCalculateRequest): ) +@router.get( + "/spend/logs/v2", + tags=["Budget & Spend Tracking"], + dependencies=[Depends(user_api_key_auth)], + responses={ + 200: {"model": Dict[str, Any]}, + }, +) @router.get( "/spend/logs/ui", tags=["Budget & Spend Tracking"], @@ -1672,16 +1681,16 @@ async def ui_view_spend_logs( # noqa: PLR0915 ), ): """ - View spend logs for UI with pagination support + View spend logs with pagination support. + Available at both `/spend/logs/v2` (public API) and `/spend/logs/ui` (internal UI). - Returns: - { - "data": List[LiteLLM_SpendLogs], # Paginated spend logs - "total": int, # Total number of records - "page": int, # Current page number - "page_size": int, # Number of items per page - "total_pages": int # Total number of pages - } + Returns paginated response with data, total, page, page_size, and total_pages. + + Example: + ``` + curl -X GET "http://0.0.0.0:8000/spend/logs/v2?start_date=2025-11-25%2000:00:00&end_date=2025-11-26%2023:59:59&page=1&page_size=50" \ +-H "Authorization: Bearer sk-1234" + ``` """ from litellm.proxy.proxy_server import prisma_client From 1ad9e014abfea81174bb2bcf4112af1f6d3e658f Mon Sep 17 00:00:00 2001 From: AlexsanderHamir Date: Wed, 26 Nov 2025 16:11:53 -0800 Subject: [PATCH 016/121] Add endpoint-based date parsing for /spend/logs/v2 - Add flexible date parsing for v2 endpoint (supports both YYYY-MM-DD and YYYY-MM-DD HH:MM:SS) - Keep strict timestamp format for /spend/logs/ui endpoint for backward compatibility - Parse dates based on which endpoint was called using request path --- .../spend_management_endpoints.py | 28 +++++++++++++------ 1 file changed, 20 insertions(+), 8 deletions(-) diff --git a/litellm/proxy/spend_tracking/spend_management_endpoints.py b/litellm/proxy/spend_tracking/spend_management_endpoints.py index 2c9dc3fc09a..608db6af9a6 100644 --- a/litellm/proxy/spend_tracking/spend_management_endpoints.py +++ b/litellm/proxy/spend_tracking/spend_management_endpoints.py @@ -7,7 +7,7 @@ from functools import lru_cache from typing import TYPE_CHECKING, Any, Dict, List, Literal, Optional import fastapi -from fastapi import APIRouter, Depends, HTTPException, status +from fastapi import APIRouter, Depends, HTTPException, Request, status import litellm from litellm._logging import verbose_proxy_logger @@ -1628,6 +1628,7 @@ async def calculate_spend(request: SpendCalculateRequest): }, ) async def ui_view_spend_logs( # noqa: PLR0915 + request: Request, api_key: Optional[str] = fastapi.Query( default=None, description="Get spend logs based on api key", @@ -1711,13 +1712,24 @@ async def ui_view_spend_logs( # noqa: PLR0915 ) try: - # Convert the date strings to datetime objects - start_date_obj = datetime.strptime(start_date, "%Y-%m-%d %H:%M:%S").replace( - tzinfo=timezone.utc - ) - end_date_obj = datetime.strptime(end_date, "%Y-%m-%d %H:%M:%S").replace( - tzinfo=timezone.utc - ) + is_v2 = "/spend/logs/v2" in request.url.path + formats = ["%Y-%m-%d %H:%M:%S", "%Y-%m-%d"] if is_v2 else ["%Y-%m-%d %H:%M:%S"] + + def parse_date(date_str: str) -> datetime: + date_str = date_str.strip() + for fmt in formats: + try: + return datetime.strptime(date_str, fmt).replace(tzinfo=timezone.utc) + except ValueError: + continue + expected = "'YYYY-MM-DD' or 'YYYY-MM-DD HH:MM:SS'" if is_v2 else "'YYYY-MM-DD HH:MM:SS'" + raise HTTPException( + status_code=status.HTTP_400_BAD_REQUEST, + detail=f"Invalid date format: {date_str}. Expected: {expected}", + ) + + start_date_obj = parse_date(start_date) + end_date_obj = parse_date(end_date) # Convert to ISO format strings for Prisma start_date_iso = start_date_obj.isoformat() # Already in UTC, no need to add Z From df190d25b87ced96cce2d2940e0f2b770256856c Mon Sep 17 00:00:00 2001 From: AlexsanderHamir Date: Wed, 26 Nov 2025 16:16:13 -0800 Subject: [PATCH 017/121] Add deprecation notice to /spend/logs endpoint - Mark /spend/logs as deprecated in docstring - Direct users to use /spend/logs/v2 for paginated access - Warns about performance issues with non-paginated endpoint --- litellm/proxy/spend_tracking/spend_management_endpoints.py | 3 +++ 1 file changed, 3 insertions(+) diff --git a/litellm/proxy/spend_tracking/spend_management_endpoints.py b/litellm/proxy/spend_tracking/spend_management_endpoints.py index 608db6af9a6..2d3fc023a39 100644 --- a/litellm/proxy/spend_tracking/spend_management_endpoints.py +++ b/litellm/proxy/spend_tracking/spend_management_endpoints.py @@ -1917,6 +1917,9 @@ async def view_spend_logs( # noqa: PLR0915 user_api_key_dict: UserAPIKeyAuth = Depends(user_api_key_auth), ): """ + [DEPRECATED] This endpoint is not paginated and can cause performance issues. + Please use `/spend/logs/v2` instead for paginated access to spend logs. + View all spend logs, if request_id is provided, only logs for that request_id will be returned When start_date and end_date are provided: From 247160277e11b28c59db05feef76c1adc105c886 Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Thu, 27 Nov 2025 15:46:04 +0530 Subject: [PATCH 018/121] Added support for twelvelabs pegasus --- litellm/__init__.py | 3 + litellm/constants.py | 1 + ...mazon_twelvelabs_pegasus_transformation.py | 133 ++++++++++++++++++ litellm/llms/bedrock/common_utils.py | 2 + .../test_twelvelabs_pegasus_transformation.py | 85 +++++++++++ 5 files changed, 224 insertions(+) create mode 100644 litellm/llms/bedrock/chat/invoke_transformations/amazon_twelvelabs_pegasus_transformation.py create mode 100644 tests/test_litellm/llms/bedrock/chat/invoke_transformations/test_twelvelabs_pegasus_transformation.py diff --git a/litellm/__init__.py b/litellm/__init__.py index aebf1404196..e6af8a21ff5 100644 --- a/litellm/__init__.py +++ b/litellm/__init__.py @@ -1222,6 +1222,9 @@ from .llms.bedrock.chat.invoke_transformations.amazon_mistral_transformation imp from .llms.bedrock.chat.invoke_transformations.amazon_titan_transformation import ( AmazonTitanConfig, ) +from .llms.bedrock.chat.invoke_transformations.amazon_twelvelabs_pegasus_transformation import ( + AmazonTwelveLabsPegasusConfig, +) from .llms.bedrock.chat.invoke_transformations.base_invoke_transformation import ( AmazonInvokeConfig, ) diff --git a/litellm/constants.py b/litellm/constants.py index cf3d4c6e742..00e17788466 100644 --- a/litellm/constants.py +++ b/litellm/constants.py @@ -851,6 +851,7 @@ BEDROCK_INVOKE_PROVIDERS_LITERAL = Literal[ "nova", "deepseek_r1", "qwen3", + "twelvelabs", ] BEDROCK_EMBEDDING_PROVIDERS_LITERAL = Literal[ diff --git a/litellm/llms/bedrock/chat/invoke_transformations/amazon_twelvelabs_pegasus_transformation.py b/litellm/llms/bedrock/chat/invoke_transformations/amazon_twelvelabs_pegasus_transformation.py new file mode 100644 index 00000000000..7b72968ea3d --- /dev/null +++ b/litellm/llms/bedrock/chat/invoke_transformations/amazon_twelvelabs_pegasus_transformation.py @@ -0,0 +1,133 @@ +""" +Transforms OpenAI-style requests into TwelveLabs Pegasus 1.2 requests for Bedrock. + +Reference: +https://docs.twelvelabs.io/docs/models/pegasus +""" + +from typing import Any, Dict, List, Optional + +from litellm.llms.base_llm.base_utils import type_to_response_format_param +from litellm.llms.base_llm.chat.transformation import BaseConfig +from litellm.llms.bedrock.chat.invoke_transformations.base_invoke_transformation import ( + AmazonInvokeConfig, +) +from litellm.types.llms.openai import AllMessageValues +from litellm.utils import get_base64_str + + +class AmazonTwelveLabsPegasusConfig(AmazonInvokeConfig, BaseConfig): + """ + Handles transforming OpenAI-style requests into Bedrock InvokeModel requests for + `twelvelabs.pegasus-1-2-v1:0`. + + Pegasus 1.2 requires an `inputPrompt` and a `mediaSource` that either references + an S3 object or a base64-encoded clip. Optional OpenAI params (temperature, + response_format, max_tokens) are translated to the TwelveLabs schema. + """ + + def get_supported_openai_params(self, model: str) -> List[str]: + return [ + "max_tokens", + "max_completion_tokens", + "temperature", + "response_format", + ] + + def map_openai_params( + self, + non_default_params: dict, + optional_params: dict, + model: str, + drop_params: bool, + ) -> dict: + for param, value in non_default_params.items(): + if param in {"max_tokens", "max_completion_tokens"}: + optional_params["maxOutputTokens"] = value + if param == "temperature": + optional_params["temperature"] = value + if param == "response_format": + optional_params["responseFormat"] = self._normalize_response_format( + value + ) + return optional_params + + def _normalize_response_format(self, value: Any) -> Any: + if isinstance(value, dict): + return value + return type_to_response_format_param(response_format=value) or value + + def transform_request( + self, + model: str, + messages: List[AllMessageValues], + optional_params: dict, + litellm_params: dict, + headers: dict, + ) -> dict: + input_prompt = self._convert_messages_to_prompt(messages=messages) + request_data: Dict[str, Any] = {"inputPrompt": input_prompt} + + media_source = self._build_media_source(optional_params) + if media_source is not None: + request_data["mediaSource"] = media_source + + for key in ("temperature", "maxOutputTokens", "responseFormat"): + if key in optional_params: + request_data[key] = optional_params.get(key) + return request_data + + def _build_media_source(self, optional_params: dict) -> Optional[dict]: + direct_source = optional_params.get("mediaSource") or optional_params.get( + "media_source" + ) + if isinstance(direct_source, dict): + return direct_source + + base64_input = optional_params.get("video_base64") or optional_params.get( + "base64_string" + ) + if base64_input: + return {"base64String": get_base64_str(base64_input)} + + s3_uri = ( + optional_params.get("video_s3_uri") + or optional_params.get("s3_uri") + or optional_params.get("media_source_s3_uri") + ) + if s3_uri: + s3_location = {"uri": s3_uri} + bucket_owner = ( + optional_params.get("video_s3_bucket_owner") + or optional_params.get("s3_bucket_owner") + or optional_params.get("media_source_bucket_owner") + ) + if bucket_owner: + s3_location["bucketOwner"] = bucket_owner + return {"s3Location": s3_location} + return None + + def _convert_messages_to_prompt(self, messages: List[AllMessageValues]) -> str: + prompt_parts: List[str] = [] + for message in messages: + role = message.get("role", "user") + content = message.get("content", "") + if isinstance(content, list): + text_fragments = [] + for item in content: + if isinstance(item, dict): + item_type = item.get("type") + if item_type == "text": + text_fragments.append(item.get("text", "")) + elif item_type == "image_url": + text_fragments.append("") + elif item_type == "video_url": + text_fragments.append("