From 73ef4bef36c5c74ad5c0b0df8d4d48e68522de8e Mon Sep 17 00:00:00 2001
From: mateo-berri <277851410+mateo-berri@users.noreply.github.com>
Date: Sun, 21 Jun 2026 03:07:30 +0000
Subject: [PATCH] refactor(bedrock): remove dead BedrockLLM invoke path and its
tests
The Feb 2025 refactor routed Bedrock Invoke through AmazonInvokeConfig and
base_llm_http_handler, leaving BedrockLLM.completion and its transitive
helpers (process_response, convert_messages_to_prompt, async_completion,
async_streaming, _async_anthropic_messages_completion) unreachable, along
with AmazonAnthropicClaudeConfig.async_transform_request. Delete them and
the unit tests that exercised only that dead path. The class is kept as a
thin shell because get_bedrock_invoke_provider is still referenced through
it, and the live make_call/make_sync_call dispatch and stream decoders stay
put.
---
litellm/llms/bedrock/chat/invoke_handler.py | 1038 +----------------
.../anthropic_claude3_transformation.py | 28 -
.../test_bedrock_completion.py | 123 --
.../llms/bedrock/chat/test_invoke_handler.py | 34 -
4 files changed, 6 insertions(+), 1217 deletions(-)
diff --git a/litellm/llms/bedrock/chat/invoke_handler.py b/litellm/llms/bedrock/chat/invoke_handler.py
index 753e41b5343..5772c1edf9a 100644
--- a/litellm/llms/bedrock/chat/invoke_handler.py
+++ b/litellm/llms/bedrock/chat/invoke_handler.py
@@ -1,19 +1,16 @@
"""
-TODO: DELETE FILE. Bedrock LLM is no longer used. Goto `litellm/llms/bedrock/chat/invoke_transformations/base_invoke_transformation.py`
+Bedrock Invoke streaming dispatch (make_call/make_sync_call) and the event
+stream decoders live here. The request/response transforms moved to
+`litellm/llms/bedrock/chat/invoke_transformations/base_invoke_transformation.py`.
"""
-import copy
-import time
import types
-from functools import partial
from typing import (
AsyncIterator,
- Callable,
Iterator,
Optional,
Tuple,
cast,
- get_args,
)
import httpx # type: ignore
@@ -25,16 +22,6 @@ from litellm.caching.caching import InMemoryCache
from litellm.constants import RESPONSE_FORMAT_TOOL_NAME
from litellm.litellm_core_utils.core_helpers import map_finish_reason
from litellm.litellm_core_utils.litellm_logging import Logging
-from litellm.litellm_core_utils.logging_utils import track_llm_api_timing
-from litellm.litellm_core_utils.prompt_templates.factory import (
- cohere_message_pt,
- construct_tool_use_system_prompt,
- contains_tag,
- custom_prompt,
- extract_between_tags,
- parse_xml_params,
- prompt_factory,
-)
from litellm.llms.anthropic.chat.handler import (
ModelResponseIterator as AnthropicModelResponseIterator,
)
@@ -65,12 +52,10 @@ from litellm.types.utils import (
StreamingChoices,
Usage,
)
-from litellm.utils import CustomStreamWrapper, get_secret
from ..base_aws_llm import BaseAWSLLM
from ..common_utils import (
BedrockError,
- ModelResponseIterator,
get_bedrock_response_stream_shape,
get_bedrock_tool_name,
)
@@ -79,9 +64,6 @@ bedrock_tool_name_mappings: InMemoryCache = InMemoryCache(
max_size_in_memory=50, default_ttl=600
)
from litellm.llms.bedrock.chat.converse_transformation import AmazonConverseConfig
-from litellm.llms.bedrock.chat.invoke_transformations.amazon_openai_transformation import (
- AmazonBedrockOpenAIConfig,
-)
converse_config = AmazonConverseConfig()
@@ -382,1021 +364,13 @@ def make_sync_call(
class BedrockLLM(BaseAWSLLM):
- """
- Example call
-
- ```
- curl --location --request POST 'https://bedrock-runtime.{aws_region_name}.amazonaws.com/model/{bedrock_model_name}/invoke' \
- --header 'Content-Type: application/json' \
- --header 'Accept: application/json' \
- --user "$AWS_ACCESS_KEY_ID":"$AWS_SECRET_ACCESS_KEY" \
- --aws-sigv4 "aws:amz:us-east-1:bedrock" \
- --data-raw '{
- "prompt": "Hi",
- "temperature": 0,
- "p": 0.9,
- "max_tokens": 4096
- }'
- ```
- """
+ """Legacy shell retained only for ``get_bedrock_invoke_provider``, inherited
+ from BaseAWSLLM and still referenced by name. Invoke requests now run through
+ AmazonInvokeConfig and base_llm_http_handler."""
def __init__(self) -> None:
super().__init__()
- @staticmethod
- def is_claude_messages_api_model(model: str) -> bool:
- """
- Check if the model uses the Claude Messages API (Claude 3+).
-
- Handles:
- - Regional prefixes: eu.anthropic.claude-*, us.anthropic.claude-*
- - Claude 3 models: claude-3-haiku, claude-3-sonnet, claude-3-opus, claude-3-5-*, claude-3-7-*
- - Claude 4 models: claude-opus-4, claude-sonnet-4, claude-haiku-4
- """
- # Normalize model string to lowercase for matching
- model_lower = model.lower()
-
- # Claude 3+ indicators (all use Messages API)
- messages_api_indicators = [
- "claude-3", # Claude 3.x models
- "claude-opus-4", # Claude Opus 4
- "claude-sonnet-4", # Claude Sonnet 4
- "claude-haiku-4", # Claude Haiku 4
- ]
-
- return any(indicator in model_lower for indicator in messages_api_indicators)
-
- def convert_messages_to_prompt(
- self, model, messages, provider, custom_prompt_dict
- ) -> Tuple[str, Optional[list]]:
- # handle anthropic prompts and amazon titan prompts
- prompt = ""
- chat_history: Optional[list] = None
- ## CUSTOM PROMPT
- if model in custom_prompt_dict:
- # check if the model has a registered custom prompt
- model_prompt_details = custom_prompt_dict[model]
- prompt = custom_prompt(
- role_dict=model_prompt_details["roles"],
- initial_prompt_value=model_prompt_details.get(
- "initial_prompt_value", ""
- ),
- final_prompt_value=model_prompt_details.get("final_prompt_value", ""),
- messages=messages,
- )
- return prompt, None
- ## ELSE
- if provider == "anthropic" or provider == "amazon":
- prompt = prompt_factory(
- model=model, messages=messages, custom_llm_provider="bedrock"
- )
- elif provider == "mistral":
- prompt = prompt_factory(
- model=model, messages=messages, custom_llm_provider="bedrock"
- )
- elif provider == "meta" or provider == "llama":
- prompt = prompt_factory(
- model=model, messages=messages, custom_llm_provider="bedrock"
- )
- elif provider == "openai":
- # OpenAI uses messages directly, no prompt conversion needed
- # Return empty prompt as it won't be used
- prompt = ""
- elif provider == "cohere":
- prompt, chat_history = cohere_message_pt(messages=messages)
- else:
- prompt = ""
- for message in messages:
- if "role" in message:
- if message["role"] == "user":
- prompt += f"{message['content']}"
- else:
- prompt += f"{message['content']}"
- else:
- prompt += f"{message['content']}"
- return prompt, chat_history # type: ignore
-
- def process_response(
- self,
- model: str,
- response: httpx.Response,
- model_response: ModelResponse,
- stream: Optional[bool],
- logging_obj: Logging,
- optional_params: dict,
- api_key: str,
- data: Union[dict, str],
- messages: List,
- print_verbose,
- encoding,
- ) -> Union[ModelResponse, CustomStreamWrapper]:
- provider = self.get_bedrock_invoke_provider(model)
- ## LOGGING
- logging_obj.post_call(
- input=messages,
- api_key=api_key,
- original_response=response.text,
- additional_args={"complete_input_dict": data},
- )
- print_verbose(f"raw model_response: {response.text}")
-
- ## RESPONSE OBJECT
- try:
- completion_response = response.json()
- except Exception:
- raise BedrockError(message=response.text, status_code=422)
-
- outputText: Optional[str] = None
- try:
- if provider == "cohere":
- if "text" in completion_response:
- outputText = completion_response["text"] # type: ignore
- elif "generations" in completion_response:
- outputText = completion_response["generations"][0]["text"]
- model_response.choices[0].finish_reason = map_finish_reason(
- completion_response["generations"][0]["finish_reason"]
- )
- elif provider == "anthropic":
- if self.is_claude_messages_api_model(model):
- json_schemas: dict = {}
- _is_function_call = False
- ## Handle Tool Calling
- if "tools" in optional_params:
- _is_function_call = True
- for tool in optional_params["tools"]:
- json_schemas[tool["function"]["name"]] = tool[
- "function"
- ].get("parameters", None)
- outputText = completion_response.get("content")[0].get("text", None)
- if outputText is not None and contains_tag(
- "invoke", outputText
- ): # OUTPUT PARSE FUNCTION CALL
- function_name = extract_between_tags("tool_name", outputText)[0]
- function_arguments_str = extract_between_tags(
- "invoke", outputText
- )[0].strip()
- function_arguments_str = (
- f"{function_arguments_str}"
- )
- function_arguments = parse_xml_params(
- function_arguments_str,
- json_schema=json_schemas.get(
- function_name, None
- ), # check if we have a json schema for this function name)
- )
- _message = litellm.Message(
- tool_calls=[
- {
- "id": f"call_{uuid.uuid4()}",
- "type": "function",
- "function": {
- "name": function_name,
- "arguments": json.dumps(function_arguments),
- },
- }
- ],
- content=None,
- )
- model_response.choices[0].message = _message # type: ignore
- model_response._hidden_params["original_response"] = (
- outputText # allow user to access raw anthropic tool calling response
- )
- if (
- _is_function_call is True
- and stream is not None
- and stream is True
- ):
- print_verbose(
- "INSIDE BEDROCK STREAMING TOOL CALLING CONDITION BLOCK"
- )
- # return an iterator
- streaming_model_response = ModelResponseStream()
- streaming_model_response.choices[0].finish_reason = getattr(
- model_response.choices[0], "finish_reason", "stop"
- )
- # streaming_model_response.choices = [litellm.utils.StreamingChoices()]
- streaming_choice = litellm.utils.StreamingChoices()
- streaming_choice.index = model_response.choices[0].index
- _tool_calls = []
- print_verbose(
- f"type of model_response.choices[0]: {type(model_response.choices[0])}"
- )
- print_verbose(
- f"type of streaming_choice: {type(streaming_choice)}"
- )
- if isinstance(model_response.choices[0], litellm.Choices):
- if getattr(
- model_response.choices[0].message, "tool_calls", None
- ) is not None and isinstance(
- model_response.choices[0].message.tool_calls, list
- ):
- for tool_call in model_response.choices[
- 0
- ].message.tool_calls:
- _tool_call = {**tool_call.dict(), "index": 0}
- _tool_calls.append(_tool_call)
- delta_obj = Delta(
- content=getattr(
- model_response.choices[0].message, "content", None
- ),
- role=model_response.choices[0].message.role,
- tool_calls=_tool_calls,
- )
- streaming_choice.delta = delta_obj
- streaming_model_response.choices = [streaming_choice]
- completion_stream = ModelResponseIterator(
- model_response=streaming_model_response
- )
- print_verbose(
- "Returns anthropic CustomStreamWrapper with 'cached_response' streaming object"
- )
- return litellm.CustomStreamWrapper(
- completion_stream=completion_stream,
- model=model,
- custom_llm_provider="cached_response",
- logging_obj=logging_obj,
- )
-
- model_response.choices[0].finish_reason = map_finish_reason(
- completion_response.get("stop_reason", "")
- )
- _usage = litellm.Usage(
- prompt_tokens=completion_response["usage"]["input_tokens"],
- completion_tokens=completion_response["usage"]["output_tokens"],
- total_tokens=completion_response["usage"]["input_tokens"]
- + completion_response["usage"]["output_tokens"],
- )
- setattr(model_response, "usage", _usage)
- else:
- outputText = completion_response["completion"]
-
- model_response.choices[0].finish_reason = completion_response[
- "stop_reason"
- ]
- elif provider == "ai21":
- outputText = (
- completion_response.get("completions")[0].get("data").get("text")
- )
- elif provider == "meta" or provider == "llama":
- outputText = completion_response["generation"]
- elif provider == "openai":
- # OpenAI imported models use OpenAI Chat Completions format
- if (
- "choices" in completion_response
- and len(completion_response["choices"]) > 0
- ):
- choice = completion_response["choices"][0]
- if "message" in choice:
- outputText = choice["message"].get("content")
- elif "text" in choice: # fallback for completion format
- outputText = choice["text"]
-
- # Set finish reason
- if "finish_reason" in choice:
- model_response.choices[0].finish_reason = map_finish_reason(
- choice["finish_reason"]
- )
-
- # Set usage if available
- if "usage" in completion_response:
- usage = completion_response["usage"]
- _usage = litellm.Usage(
- prompt_tokens=usage.get("prompt_tokens", 0),
- completion_tokens=usage.get("completion_tokens", 0),
- total_tokens=usage.get("total_tokens", 0),
- )
- setattr(model_response, "usage", _usage)
- elif provider == "mistral":
- outputText = completion_response["outputs"][0]["text"]
- model_response.choices[0].finish_reason = completion_response[
- "outputs"
- ][0]["stop_reason"]
- else: # amazon titan
- outputText = completion_response.get("results")[0].get("outputText")
- except Exception as e:
- raise BedrockError(
- message="Error processing={}, Received error={}".format(
- response.text, str(e)
- ),
- status_code=422,
- )
-
- try:
- if (
- outputText is not None
- and len(outputText) > 0
- and hasattr(model_response.choices[0], "message")
- and getattr(model_response.choices[0].message, "tool_calls", None) # type: ignore
- is None
- ):
- model_response.choices[0].message.content = outputText # type: ignore
- elif (
- hasattr(model_response.choices[0], "message")
- and getattr(model_response.choices[0].message, "tool_calls", None) # type: ignore
- is not None
- ):
- pass
- else:
- raise Exception()
- except Exception as e:
- raise BedrockError(
- message="Error parsing received text={}.\nError-{}".format(
- outputText, str(e)
- ),
- status_code=response.status_code,
- )
-
- if stream and provider == "ai21":
- streaming_model_response = ModelResponseStream()
- streaming_model_response.choices[0].finish_reason = model_response.choices[ # type: ignore
- 0
- ].finish_reason
- # streaming_model_response.choices = [litellm.utils.StreamingChoices()]
- streaming_choice = litellm.utils.StreamingChoices()
- streaming_choice.index = model_response.choices[0].index
- delta_obj = litellm.utils.Delta(
- content=getattr(model_response.choices[0].message, "content", None), # type: ignore
- role=model_response.choices[0].message.role, # type: ignore
- )
- streaming_choice.delta = delta_obj
- streaming_model_response.choices = [streaming_choice]
- mri = ModelResponseIterator(model_response=streaming_model_response)
- return CustomStreamWrapper(
- completion_stream=mri,
- model=model,
- custom_llm_provider="cached_response",
- logging_obj=logging_obj,
- )
-
- ## CALCULATING USAGE - bedrock returns usage in the headers
- # Skip if usage was already set (e.g., from JSON response for OpenAI provider)
- if (
- not hasattr(model_response, "usage")
- or getattr(model_response, "usage", None) is None
- ):
- bedrock_input_tokens = response.headers.get(
- "x-amzn-bedrock-input-token-count", None
- )
- bedrock_output_tokens = response.headers.get(
- "x-amzn-bedrock-output-token-count", None
- )
-
- prompt_tokens = int(
- bedrock_input_tokens or litellm.token_counter(messages=messages)
- )
-
- completion_tokens = int(
- bedrock_output_tokens
- or litellm.token_counter(
- text=model_response.choices[0].message.content, # type: ignore
- count_response_tokens=True,
- )
- )
-
- model_response.created = int(time.time())
- model_response.model = model
- usage = Usage(
- prompt_tokens=prompt_tokens,
- completion_tokens=completion_tokens,
- total_tokens=prompt_tokens + completion_tokens,
- )
- setattr(model_response, "usage", usage)
- else:
- # Ensure created and model are set even if usage was already set
- model_response.created = int(time.time())
- model_response.model = model
-
- return model_response
-
- def completion(
- self,
- model: str,
- messages: list,
- api_base: Optional[str],
- custom_prompt_dict: dict,
- model_response: ModelResponse,
- print_verbose: Callable,
- encoding,
- logging_obj: Logging,
- optional_params: dict,
- acompletion: bool,
- timeout: Optional[Union[float, httpx.Timeout]],
- litellm_params=None,
- logger_fn=None,
- extra_headers: Optional[dict] = None,
- client: Optional[Union[AsyncHTTPHandler, HTTPHandler]] = None,
- ) -> Union[ModelResponse, CustomStreamWrapper]:
- try:
- from botocore.credentials import Credentials
- except ImportError:
- raise ImportError("Missing boto3 to call bedrock. Run 'pip install boto3'.")
-
- ## SETUP ##
- stream = optional_params.pop("stream", None)
- stream_chunk_size = optional_params.pop("stream_chunk_size", None)
-
- provider = self.get_bedrock_invoke_provider(model)
- modelId = self.get_bedrock_model_id(
- model=model,
- provider=provider,
- optional_params=optional_params,
- )
-
- ## CREDENTIALS ##
- # pop aws_secret_access_key, aws_access_key_id, aws_session_token, aws_region_name from kwargs, since completion calls fail with them
- aws_secret_access_key = optional_params.pop("aws_secret_access_key", None)
- aws_access_key_id = optional_params.pop("aws_access_key_id", None)
- aws_session_token = optional_params.pop("aws_session_token", None)
- aws_region_name = optional_params.pop("aws_region_name", None)
- aws_role_name = optional_params.pop("aws_role_name", None)
- aws_session_name = optional_params.pop("aws_session_name", None)
- aws_profile_name = optional_params.pop("aws_profile_name", None)
- aws_bedrock_runtime_endpoint = optional_params.pop(
- "aws_bedrock_runtime_endpoint", None
- ) # https://bedrock-runtime.{region_name}.amazonaws.com
- aws_web_identity_token = optional_params.pop("aws_web_identity_token", None)
- aws_sts_endpoint = optional_params.pop("aws_sts_endpoint", None)
- ssl_verify = optional_params.pop("ssl_verify", None)
-
- ### SET REGION NAME ###
- if aws_region_name is None:
- # check env #
- litellm_aws_region_name = get_secret("AWS_REGION_NAME", None)
-
- if litellm_aws_region_name is not None and isinstance(
- litellm_aws_region_name, str
- ):
- aws_region_name = litellm_aws_region_name
-
- standard_aws_region_name = get_secret("AWS_REGION", None)
- if standard_aws_region_name is not None and isinstance(
- standard_aws_region_name, str
- ):
- aws_region_name = standard_aws_region_name
-
- if aws_region_name is None:
- aws_region_name = "us-west-2"
-
- credentials: Credentials = self.get_credentials(
- aws_access_key_id=aws_access_key_id,
- aws_secret_access_key=aws_secret_access_key,
- aws_session_token=aws_session_token,
- aws_region_name=aws_region_name,
- aws_session_name=aws_session_name,
- aws_profile_name=aws_profile_name,
- aws_role_name=aws_role_name,
- aws_web_identity_token=aws_web_identity_token,
- aws_sts_endpoint=aws_sts_endpoint,
- ssl_verify=ssl_verify,
- )
-
- ### SET RUNTIME ENDPOINT ###
- endpoint_url, proxy_endpoint_url = self.get_runtime_endpoint(
- api_base=api_base,
- aws_bedrock_runtime_endpoint=aws_bedrock_runtime_endpoint,
- aws_region_name=aws_region_name,
- )
-
- if (stream is not None and stream is True) and provider != "ai21":
- endpoint_url = f"{endpoint_url}/model/{modelId}/invoke-with-response-stream"
- proxy_endpoint_url = (
- f"{proxy_endpoint_url}/model/{modelId}/invoke-with-response-stream"
- )
- else:
- endpoint_url = f"{endpoint_url}/model/{modelId}/invoke"
- proxy_endpoint_url = f"{proxy_endpoint_url}/model/{modelId}/invoke"
-
- if (
- acompletion
- and provider == "anthropic"
- and self.is_claude_messages_api_model(model)
- ):
- if isinstance(client, HTTPHandler):
- client = None
- return self._async_anthropic_messages_completion(
- model=model,
- messages=messages,
- endpoint_url=endpoint_url,
- proxy_endpoint_url=proxy_endpoint_url,
- credentials=credentials,
- aws_region_name=aws_region_name,
- model_response=model_response,
- print_verbose=print_verbose,
- encoding=encoding,
- logging_obj=logging_obj,
- optional_params=optional_params,
- stream=stream,
- litellm_params=litellm_params,
- logger_fn=logger_fn,
- extra_headers=extra_headers,
- timeout=timeout,
- client=client,
- stream_chunk_size=stream_chunk_size,
- ) # type: ignore[return-value]
-
- prompt, chat_history = self.convert_messages_to_prompt(
- model, messages, provider, custom_prompt_dict
- )
- inference_params = copy.deepcopy(optional_params)
- json_schemas: dict = {}
- if provider == "cohere":
- if model.startswith("cohere.command-r"):
- ## LOAD CONFIG
- config = litellm.AmazonCohereChatConfig().get_config()
- for k, v in config.items():
- if (
- k not in inference_params
- ): # completion(top_k=3) > anthropic_config(top_k=3) <- allows for dynamic variables to be passed in
- inference_params[k] = v
- _data = {"message": prompt, **inference_params}
- if chat_history is not None:
- _data["chat_history"] = chat_history
- data = json.dumps(_data)
- else:
- ## LOAD CONFIG
- config = litellm.AmazonCohereConfig.get_config()
- for k, v in config.items():
- if (
- k not in inference_params
- ): # completion(top_k=3) > anthropic_config(top_k=3) <- allows for dynamic variables to be passed in
- inference_params[k] = v
- if stream is True:
- inference_params["stream"] = (
- True # cohere requires stream = True in inference params
- )
- data = json.dumps({"prompt": prompt, **inference_params})
- elif provider == "anthropic":
- if self.is_claude_messages_api_model(model):
- # Separate system prompt from rest of message
- system_prompt_idx: list[int] = []
- system_messages: list[str] = []
- for idx, message in enumerate(messages):
- if message["role"] == "system":
- system_messages.append(message["content"])
- system_prompt_idx.append(idx)
- if len(system_prompt_idx) > 0:
- inference_params["system"] = "\n".join(system_messages)
- messages = [
- i for j, i in enumerate(messages) if j not in system_prompt_idx
- ]
- # Format rest of message according to anthropic guidelines
- messages = prompt_factory(
- model=model, messages=messages, custom_llm_provider="anthropic_xml"
- ) # type: ignore
- ## LOAD CONFIG
- config = litellm.AmazonAnthropicClaudeConfig.get_config()
- for k, v in config.items():
- if (
- k not in inference_params
- ): # completion(top_k=3) > anthropic_config(top_k=3) <- allows for dynamic variables to be passed in
- inference_params[k] = v
- ## Handle Tool Calling
- if "tools" in inference_params:
- _is_function_call = True
- for tool in inference_params["tools"]:
- json_schemas[tool["function"]["name"]] = tool["function"].get(
- "parameters", None
- )
- tool_calling_system_prompt = construct_tool_use_system_prompt(
- tools=inference_params["tools"]
- )
- inference_params["system"] = (
- inference_params.get("system", "\n")
- + tool_calling_system_prompt
- ) # add the anthropic tool calling prompt to the system prompt
- inference_params.pop("tools")
- data = json.dumps({"messages": messages, **inference_params})
- else:
- ## LOAD CONFIG
- config = litellm.AmazonAnthropicConfig.get_config()
- for k, v in config.items():
- if (
- k not in inference_params
- ): # completion(top_k=3) > anthropic_config(top_k=3) <- allows for dynamic variables to be passed in
- inference_params[k] = v
- data = json.dumps({"prompt": prompt, **inference_params})
- elif provider == "ai21":
- ## LOAD CONFIG
- config = litellm.AmazonAI21Config.get_config()
- for k, v in config.items():
- if (
- k not in inference_params
- ): # completion(top_k=3) > anthropic_config(top_k=3) <- allows for dynamic variables to be passed in
- inference_params[k] = v
-
- data = json.dumps({"prompt": prompt, **inference_params})
- elif provider == "mistral":
- ## LOAD CONFIG
- config = litellm.AmazonMistralConfig.get_config()
- for k, v in config.items():
- if (
- k not in inference_params
- ): # completion(top_k=3) > amazon_config(top_k=3) <- allows for dynamic variables to be passed in
- inference_params[k] = v
-
- data = json.dumps({"prompt": prompt, **inference_params})
- elif provider == "amazon": # amazon titan
- ## LOAD CONFIG
- config = litellm.AmazonTitanConfig.get_config()
- for k, v in config.items():
- if (
- k not in inference_params
- ): # completion(top_k=3) > amazon_config(top_k=3) <- allows for dynamic variables to be passed in
- inference_params[k] = v
-
- data = json.dumps(
- {
- "inputText": prompt,
- "textGenerationConfig": inference_params,
- }
- )
- elif provider == "meta" or provider == "llama":
- ## LOAD CONFIG
- config = litellm.AmazonLlamaConfig.get_config()
- for k, v in config.items():
- if (
- k not in inference_params
- ): # completion(top_k=3) > anthropic_config(top_k=3) <- allows for dynamic variables to be passed in
- inference_params[k] = v
- data = json.dumps({"prompt": prompt, **inference_params})
- elif provider == "openai":
- ## OpenAI imported models use OpenAI Chat Completions format (messages-based)
- # Use AmazonBedrockOpenAIConfig for proper OpenAI transformation
- openai_config = AmazonBedrockOpenAIConfig()
- supported_params = openai_config.get_supported_openai_params(model=model)
-
- # Filter to only supported OpenAI params
- filtered_params = {
- k: v for k, v in inference_params.items() if k in supported_params
- }
-
- # OpenAI uses messages format, not prompt
- data = json.dumps({"messages": messages, **filtered_params})
- else:
- ## LOGGING
- logging_obj.pre_call(
- input=messages,
- api_key="",
- additional_args={
- "complete_input_dict": inference_params,
- },
- )
- raise BedrockError(
- status_code=404,
- message="Bedrock Invoke HTTPX: Unknown provider={}, model={}. Try calling via converse route - `bedrock/converse/`.".format(
- provider, model
- ),
- )
-
- ## COMPLETION CALL
-
- headers = {"Content-Type": "application/json"}
- if extra_headers is not None:
- headers = {"Content-Type": "application/json", **extra_headers}
- prepped = self.get_request_headers(
- credentials=credentials,
- aws_region_name=aws_region_name,
- extra_headers=extra_headers,
- endpoint_url=endpoint_url,
- data=data,
- headers=headers,
- )
-
- ## LOGGING
- logging_obj.pre_call(
- input=messages,
- api_key="",
- additional_args={
- "complete_input_dict": data,
- "api_base": proxy_endpoint_url,
- "headers": prepped.headers,
- },
- )
-
- ### ROUTING (ASYNC, STREAMING, SYNC)
- if acompletion:
- if isinstance(client, HTTPHandler):
- client = None
- if stream is True and provider != "ai21":
- return self.async_streaming(
- model=model,
- messages=messages,
- data=data,
- api_base=proxy_endpoint_url,
- model_response=model_response,
- print_verbose=print_verbose,
- encoding=encoding,
- logging_obj=logging_obj,
- optional_params=optional_params,
- stream=True,
- litellm_params=litellm_params,
- logger_fn=logger_fn,
- headers=prepped.headers,
- timeout=timeout,
- client=client,
- stream_chunk_size=stream_chunk_size,
- ) # type: ignore
- ### ASYNC COMPLETION
- return self.async_completion(
- model=model,
- messages=messages,
- data=data,
- api_base=proxy_endpoint_url,
- model_response=model_response,
- print_verbose=print_verbose,
- encoding=encoding,
- logging_obj=logging_obj,
- optional_params=optional_params,
- stream=stream, # type: ignore
- litellm_params=litellm_params,
- logger_fn=logger_fn,
- headers=prepped.headers,
- timeout=timeout,
- client=client,
- ) # type: ignore
-
- if client is None or isinstance(client, AsyncHTTPHandler):
- _params = {}
- if timeout is not None:
- if isinstance(timeout, float) or isinstance(timeout, int):
- timeout = httpx.Timeout(timeout)
- _params["timeout"] = timeout
- self.client = _get_httpx_client(_params) # type: ignore
- else:
- self.client = client
- if (stream is not None and stream is True) and provider != "ai21":
- response = self.client.post(
- url=proxy_endpoint_url,
- headers=prepped.headers, # type: ignore
- data=data,
- stream=stream,
- logging_obj=logging_obj,
- )
-
- if response.status_code != 200:
- raise BedrockError(
- status_code=response.status_code, message=str(response.read())
- )
-
- decoder = AWSEventStreamDecoder(model=model)
-
- completion_stream = decoder.iter_bytes(
- response.iter_bytes(chunk_size=stream_chunk_size)
- )
- streaming_response = CustomStreamWrapper(
- completion_stream=completion_stream,
- model=model,
- custom_llm_provider="bedrock",
- logging_obj=logging_obj,
- )
-
- ## LOGGING
- logging_obj.post_call(
- input=messages,
- api_key="",
- original_response=streaming_response,
- additional_args={"complete_input_dict": data},
- )
- return streaming_response
-
- try:
- response = self.client.post(
- url=proxy_endpoint_url,
- headers=dict(prepped.headers),
- data=data,
- logging_obj=logging_obj,
- )
- response.raise_for_status()
- except httpx.HTTPStatusError as err:
- error_code = err.response.status_code
- raise BedrockError(status_code=error_code, message=err.response.text)
- except httpx.TimeoutException:
- raise BedrockError(status_code=408, message="Timeout error occurred.")
-
- return self.process_response(
- model=model,
- response=response,
- model_response=model_response,
- stream=stream,
- logging_obj=logging_obj,
- optional_params=optional_params,
- api_key="",
- data=data,
- messages=messages,
- print_verbose=print_verbose,
- encoding=encoding,
- )
-
- async def _async_anthropic_messages_completion(
- self,
- model: str,
- messages: list,
- endpoint_url: str,
- proxy_endpoint_url: str,
- credentials,
- aws_region_name: str,
- model_response: ModelResponse,
- print_verbose: Callable,
- encoding,
- logging_obj: Logging,
- optional_params: dict,
- stream,
- litellm_params=None,
- logger_fn=None,
- extra_headers: Optional[dict] = None,
- timeout: Optional[Union[float, httpx.Timeout]] = None,
- client: Optional[AsyncHTTPHandler] = None,
- stream_chunk_size: Optional[int] = None,
- ) -> Union[ModelResponse, CustomStreamWrapper]:
- transformed_request = (
- await litellm.AmazonAnthropicClaudeConfig().async_transform_request(
- model=model,
- messages=messages,
- optional_params=optional_params,
- litellm_params=litellm_params or {},
- headers=extra_headers or {},
- )
- )
- data = json.dumps(transformed_request)
-
- headers = {"Content-Type": "application/json"}
- if extra_headers is not None:
- headers = {"Content-Type": "application/json", **extra_headers}
- prepped = self.get_request_headers(
- credentials=credentials,
- aws_region_name=aws_region_name,
- extra_headers=extra_headers,
- endpoint_url=endpoint_url,
- data=data,
- headers=headers,
- )
-
- logging_obj.pre_call(
- input=messages,
- api_key="",
- additional_args={
- "complete_input_dict": data,
- "api_base": proxy_endpoint_url,
- "headers": prepped.headers,
- },
- )
-
- if stream is True:
- return await self.async_streaming(
- model=model,
- messages=messages,
- data=data,
- api_base=proxy_endpoint_url,
- model_response=model_response,
- print_verbose=print_verbose,
- encoding=encoding,
- logging_obj=logging_obj,
- optional_params=optional_params,
- stream=True,
- litellm_params=litellm_params,
- logger_fn=logger_fn,
- headers=prepped.headers,
- timeout=timeout,
- client=client,
- stream_chunk_size=stream_chunk_size,
- )
- return await self.async_completion(
- model=model,
- messages=messages,
- data=data,
- api_base=proxy_endpoint_url,
- model_response=model_response,
- print_verbose=print_verbose,
- encoding=encoding,
- logging_obj=logging_obj,
- optional_params=optional_params,
- stream=stream, # type: ignore
- litellm_params=litellm_params,
- logger_fn=logger_fn,
- headers=prepped.headers,
- timeout=timeout,
- client=client,
- )
-
- async def async_completion(
- self,
- model: str,
- messages: list,
- api_base: str,
- model_response: ModelResponse,
- print_verbose: Callable,
- data: str,
- timeout: Optional[Union[float, httpx.Timeout]],
- encoding,
- logging_obj: Logging,
- stream,
- optional_params: dict,
- litellm_params=None,
- logger_fn=None,
- headers={},
- client: Optional[AsyncHTTPHandler] = None,
- ) -> Union[ModelResponse, CustomStreamWrapper]:
- if client is None:
- _params = {}
- if timeout is not None:
- if isinstance(timeout, float) or isinstance(timeout, int):
- timeout = httpx.Timeout(timeout)
- _params["timeout"] = timeout
- client = get_async_httpx_client(params=_params, llm_provider=litellm.LlmProviders.BEDROCK) # type: ignore
- else:
- client = client # type: ignore
-
- try:
- response = await client.post(
- api_base,
- headers=headers,
- data=data,
- timeout=timeout,
- logging_obj=logging_obj,
- )
- response.raise_for_status()
- except httpx.HTTPStatusError as err:
- error_code = err.response.status_code
- raise BedrockError(status_code=error_code, message=err.response.text)
- except httpx.TimeoutException:
- raise BedrockError(status_code=408, message="Timeout error occurred.")
-
- return self.process_response(
- model=model,
- response=response,
- model_response=model_response,
- stream=stream if isinstance(stream, bool) else False,
- logging_obj=logging_obj,
- api_key="",
- data=data,
- messages=messages,
- print_verbose=print_verbose,
- optional_params=optional_params,
- encoding=encoding,
- )
-
- @track_llm_api_timing() # for streaming, we need to instrument the function calling the wrapper
- async def async_streaming(
- self,
- model: str,
- messages: list,
- api_base: str,
- model_response: ModelResponse,
- print_verbose: Callable,
- data: str,
- timeout: Optional[Union[float, httpx.Timeout]],
- encoding,
- logging_obj: Logging,
- stream,
- optional_params: dict,
- litellm_params=None,
- logger_fn=None,
- headers={},
- client: Optional[AsyncHTTPHandler] = None,
- stream_chunk_size: Optional[int] = None,
- ) -> CustomStreamWrapper:
- # The call is not made here; instead, we prepare the necessary objects for the stream.
-
- streaming_response = CustomStreamWrapper(
- completion_stream=None,
- make_call=partial(
- make_call,
- client=client,
- api_base=api_base,
- headers=headers,
- data=data, # type: ignore
- model=model,
- messages=messages,
- logging_obj=logging_obj,
- fake_stream=True if "ai21" in api_base else False,
- stream_chunk_size=stream_chunk_size,
- ),
- model=model,
- custom_llm_provider="bedrock",
- logging_obj=logging_obj,
- )
- return streaming_response
-
- @staticmethod
- def _get_provider_from_model_path(
- model_path: str,
- ) -> Optional[litellm.BEDROCK_INVOKE_PROVIDERS_LITERAL]:
- """
- Helper function to get the provider from a model path with format: provider/model-name
-
- Args:
- model_path (str): The model path (e.g., 'llama/arn:aws:bedrock:us-east-1:086734376398:imported-model/r4c4kewx2s0n' or 'anthropic/model-name')
-
- Returns:
- Optional[str]: The provider name, or None if no valid provider found
- """
- parts = model_path.split("/")
- if len(parts) >= 1:
- provider = parts[0]
- if provider in get_args(litellm.BEDROCK_INVOKE_PROVIDERS_LITERAL):
- return cast(litellm.BEDROCK_INVOKE_PROVIDERS_LITERAL, provider)
- return None
-
class AWSEventStreamDecoder:
def __init__(self, model: str, json_mode: Optional[bool] = False) -> None:
diff --git a/litellm/llms/bedrock/chat/invoke_transformations/anthropic_claude3_transformation.py b/litellm/llms/bedrock/chat/invoke_transformations/anthropic_claude3_transformation.py
index e84989e4c1a..c7b99ddd215 100644
--- a/litellm/llms/bedrock/chat/invoke_transformations/anthropic_claude3_transformation.py
+++ b/litellm/llms/bedrock/chat/invoke_transformations/anthropic_claude3_transformation.py
@@ -154,34 +154,6 @@ class AmazonAnthropicClaudeConfig(AmazonInvokeConfig, AnthropicConfig):
return _anthropic_request
- async def async_transform_request(
- self,
- model: str,
- messages: List[AllMessageValues],
- optional_params: dict,
- litellm_params: dict,
- headers: dict,
- ) -> dict:
- _anthropic_request = self._build_bedrock_anthropic_request_base(
- model=model,
- messages=messages,
- optional_params=optional_params,
- litellm_params=litellm_params,
- headers=headers,
- )
-
- await self._async_convert_document_url_sources_to_base64(_anthropic_request)
- beta_list = self._compute_bedrock_invoke_beta_headers(
- model=model,
- messages=messages,
- optional_params=optional_params,
- headers=headers,
- )
- if beta_list:
- _anthropic_request["anthropic_beta"] = beta_list
-
- return _anthropic_request
-
def _build_bedrock_anthropic_request_base(
self,
model: str,
diff --git a/tests/llm_translation/test_bedrock_completion.py b/tests/llm_translation/test_bedrock_completion.py
index fa22ff6b392..655e5284462 100644
--- a/tests/llm_translation/test_bedrock_completion.py
+++ b/tests/llm_translation/test_bedrock_completion.py
@@ -3559,89 +3559,6 @@ def test_bedrock_openai_model_id_extraction():
print(f"✓ Model ID extracted and encoded: {model_id}")
-def test_bedrock_openai_convert_messages_to_prompt():
- """
- Test that convert_messages_to_prompt returns empty string for OpenAI models.
- """
- from litellm.llms.bedrock.chat.invoke_handler import BedrockLLM
-
- bedrock_llm = BedrockLLM()
- messages = [
- {"role": "system", "content": "You are helpful"},
- {"role": "user", "content": "Hello"},
- ]
-
- prompt, chat_history = bedrock_llm.convert_messages_to_prompt(
- model="test-model", messages=messages, provider="openai", custom_prompt_dict={}
- )
-
- # OpenAI models use messages directly, no prompt conversion
- assert prompt == ""
- assert chat_history is None
- print("✓ convert_messages_to_prompt returns empty for OpenAI")
-
-
-def test_bedrock_openai_response_parsing():
- """
- Test that OpenAI responses are correctly parsed.
- """
- from litellm.llms.bedrock.chat.invoke_handler import BedrockLLM
- from litellm import ModelResponse
- from unittest.mock import Mock
- import json
-
- bedrock_llm = BedrockLLM()
-
- # Mock OpenAI-style response
- openai_response = {
- "choices": [
- {
- "message": {
- "content": "The capital of France is Paris.",
- "role": "assistant",
- },
- "finish_reason": "stop",
- "index": 0,
- }
- ],
- "usage": {"prompt_tokens": 10, "completion_tokens": 8, "total_tokens": 18},
- }
-
- mock_response = Mock()
- mock_response.json.return_value = openai_response
- mock_response.text = json.dumps(openai_response)
- mock_response.status_code = 200
- mock_response.headers = {}
-
- model_response = ModelResponse()
- mock_logging = Mock()
-
- result = bedrock_llm.process_response(
- model="openai/arn:aws:bedrock:us-east-1:123:imported-model/test",
- response=mock_response,
- model_response=model_response,
- stream=False,
- logging_obj=mock_logging,
- optional_params={},
- api_key="",
- data={},
- messages=[{"role": "user", "content": "What is the capital of France?"}],
- print_verbose=lambda x: None,
- encoding=None,
- )
-
- # Verify response content
- assert result.choices[0].message.content == "The capital of France is Paris."
- assert result.choices[0].finish_reason == "stop"
-
- # Verify usage
- assert result.usage.prompt_tokens == 10
- assert result.usage.completion_tokens == 8
- assert result.usage.total_tokens == 18
-
- print("✓ OpenAI response parsing works correctly")
-
-
def test_bedrock_openai_request_transformation():
"""
Test that the request is correctly transformed for OpenAI models.
@@ -3831,46 +3748,6 @@ def test_bedrock_openai_multiple_message_types():
print("✓ Multiple message types handled correctly")
-def test_bedrock_openai_error_handling():
- """
- Test that errors from OpenAI models are properly handled.
- """
- from litellm.llms.bedrock.chat.invoke_handler import BedrockLLM
- from litellm import ModelResponse
- from litellm.llms.bedrock.common_utils import BedrockError
- from unittest.mock import Mock
- import json
-
- bedrock_llm = BedrockLLM()
-
- # Mock error response
- mock_response = Mock()
- mock_response.json.side_effect = Exception("Invalid JSON")
- mock_response.text = "Invalid response"
- mock_response.status_code = 422
-
- model_response = ModelResponse()
- mock_logging = Mock()
-
- with pytest.raises(BedrockError) as exc_info:
- bedrock_llm.process_response(
- model="openai/arn:aws:bedrock:us-east-1:123:imported-model/test",
- response=mock_response,
- model_response=model_response,
- stream=False,
- logging_obj=mock_logging,
- optional_params={},
- api_key="",
- data={},
- messages=[],
- print_verbose=lambda x: None,
- encoding=None,
- )
-
- assert exc_info.value.status_code == 422
- print("✓ Error handling works correctly")
-
-
# ============================================================================
# Nova Grounding (web_search_options) Unit Tests (Mocked)
# ============================================================================
diff --git a/tests/test_litellm/llms/bedrock/chat/test_invoke_handler.py b/tests/test_litellm/llms/bedrock/chat/test_invoke_handler.py
index 34ce75abe9a..9c9d6fc44a3 100644
--- a/tests/test_litellm/llms/bedrock/chat/test_invoke_handler.py
+++ b/tests/test_litellm/llms/bedrock/chat/test_invoke_handler.py
@@ -8,14 +8,11 @@ sys.path.insert(
0, os.path.abspath("../../../../..")
) # Adds the parent directory to the system path
-import litellm
from litellm.llms.bedrock.chat.invoke_handler import (
AWSEventStreamDecoder,
- BedrockLLM,
make_call,
make_sync_call,
)
-from litellm.llms.custom_httpx.http_handler import HTTPHandler
def test_transform_thinking_blocks_with_redacted_content():
@@ -336,34 +333,3 @@ def test_make_sync_call_guards_against_leaked_control_param():
)
client.post.assert_not_called()
-
-
-def test_legacy_bedrock_llm_streaming_does_not_rechunk_by_default():
- mock_response = MagicMock()
- mock_response.status_code = 200
- mock_response.iter_bytes = MagicMock(return_value=iter([]))
- client = HTTPHandler()
- client.post = MagicMock(return_value=mock_response)
-
- BedrockLLM().completion(
- model="cohere.command-text-v14",
- messages=[{"role": "user", "content": "hi"}],
- api_base=None,
- custom_prompt_dict={},
- model_response=litellm.ModelResponse(),
- print_verbose=lambda *args, **kwargs: None,
- encoding=litellm.encoding,
- logging_obj=MagicMock(),
- optional_params={
- "stream": True,
- "aws_access_key_id": "fake",
- "aws_secret_access_key": "fake",
- "aws_region_name": "us-east-1",
- },
- acompletion=False,
- timeout=None,
- litellm_params={},
- client=client,
- )
-
- mock_response.iter_bytes.assert_called_once_with(chunk_size=None)