mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-07 02:59:05 +00:00
fix(bedrock): route overlong GPT version digits to Converse and send native chat completions to the runtime endpoint
A model id with more than 4300 version digits raised ValueError in the route check; the digits are now bounded so such ids fall back to Converse. The native chat completions URL now follows Converse's precedence: aws_bedrock_runtime_endpoint (or AWS_BEDROCK_RUNTIME_ENDPOINT) wins over api_base, so a deployment that sets both keeps sending to the same host
This commit is contained in:
parent
708f5a5729
commit
dae29c138b
3 changed files with 37 additions and 3 deletions
|
|
@ -277,12 +277,12 @@ class AmazonBedrockRuntimeChatCompletionsConfig(OpenAILikeChatConfig):
|
|||
aws_region_name: Final = self._aws_signer._get_aws_region_name( # pyright: ignore[reportPrivateUsage] # BaseAWSLLM has no public region resolver
|
||||
optional_params=self._params_with_region_from_path(optional_params, model), model=model
|
||||
)
|
||||
endpoint_url, _ = self._aws_signer.get_runtime_endpoint(
|
||||
_, proxy_endpoint_url = self._aws_signer.get_runtime_endpoint(
|
||||
api_base=api_base,
|
||||
aws_bedrock_runtime_endpoint=optional_params.get("aws_bedrock_runtime_endpoint"),
|
||||
aws_region_name=aws_region_name,
|
||||
)
|
||||
base: Final = endpoint_url.rstrip("/")
|
||||
base: Final = proxy_endpoint_url.rstrip("/")
|
||||
if base.endswith("/openai/v1/chat/completions"):
|
||||
return base
|
||||
if base.endswith("/openai/v1"):
|
||||
|
|
|
|||
|
|
@ -39,7 +39,7 @@ if TYPE_CHECKING:
|
|||
|
||||
_ERROR_REQUEST_URL: Final = "https://docs.litellm.ai/docs"
|
||||
_OPENAI_FAMILY_MODEL_RE: Final = re.compile(r"(^|[./])openai\.")
|
||||
_OPENAI_GPT_VERSION_RE: Final = re.compile(r"(^|[./])openai\.gpt-(\d+)(?:\.(\d+))?")
|
||||
_OPENAI_GPT_VERSION_RE: Final = re.compile(r"(^|[./])openai\.gpt-(\d{1,3})(?!\d)(?:\.(\d{1,3})(?!\d))?")
|
||||
_BEDROCK_RUNTIME_CHAT_COMPLETIONS_DEFAULT_SINCE: Final = (5, 6)
|
||||
_BEDROCK_RUNTIME_CHAT_COMPLETIONS_ENDPOINT: Final = "/v1/chat/completions"
|
||||
BedrockRoute = Literal[
|
||||
|
|
|
|||
|
|
@ -149,6 +149,40 @@ def test_complete_url_appends_to_openai_v1_base():
|
|||
assert url == "https://bedrock-runtime.us-west-2.amazonaws.com/openai/v1/chat/completions"
|
||||
|
||||
|
||||
def test_complete_url_sends_to_the_runtime_endpoint_over_api_base_like_converse(monkeypatch):
|
||||
monkeypatch.delenv("AWS_BEDROCK_RUNTIME_ENDPOINT", raising=False)
|
||||
cfg = AmazonBedrockRuntimeChatCompletionsConfig()
|
||||
url = cfg.get_complete_url(
|
||||
api_base="https://signing-host.example.com",
|
||||
api_key=None,
|
||||
model="us.openai.gpt-5.6-sol",
|
||||
optional_params={"aws_region_name": "us-east-1", "aws_bedrock_runtime_endpoint": "https://egress.example.com/"},
|
||||
litellm_params={},
|
||||
)
|
||||
assert url == "https://egress.example.com/openai/v1/chat/completions"
|
||||
|
||||
|
||||
def test_complete_url_sends_to_the_env_runtime_endpoint_over_api_base_like_converse(monkeypatch):
|
||||
monkeypatch.setenv("AWS_BEDROCK_RUNTIME_ENDPOINT", "https://env-egress.example.com")
|
||||
cfg = AmazonBedrockRuntimeChatCompletionsConfig()
|
||||
url = cfg.get_complete_url(
|
||||
api_base="https://signing-host.example.com",
|
||||
api_key=None,
|
||||
model="us.openai.gpt-5.6-sol",
|
||||
optional_params={"aws_region_name": "us-east-1"},
|
||||
litellm_params={},
|
||||
)
|
||||
assert url == "https://env-egress.example.com/openai/v1/chat/completions"
|
||||
|
||||
|
||||
@pytest.mark.parametrize("digits", [4, 4301, 30000])
|
||||
@pytest.mark.parametrize("template", ["openai.gpt-{run}", "us.openai.gpt-5.{run}", "openai.gpt-{run}.{run}-sol"])
|
||||
def test_overlong_gpt_version_digits_route_to_converse_without_raising(local_cost_map, template, digits):
|
||||
model = template.format(run="9" * digits)
|
||||
assert bedrock_runtime_chat_completions_is_default(model) is False
|
||||
assert bedrock_route_for_request(model, {}, None) == "converse"
|
||||
|
||||
|
||||
def test_project_id_is_not_sent_as_openai_project_header():
|
||||
cfg = AmazonBedrockRuntimeChatCompletionsConfig()
|
||||
headers = cfg.validate_environment(
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue