mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-03 02:22:24 +00:00
Merge 71cd2c30d1 into dab2deb5ed
This commit is contained in:
commit
01178d9479
3 changed files with 73 additions and 13 deletions
|
|
@ -52,6 +52,7 @@ DEFAULT_S3_MAX_CONCURRENT_UPLOADS: Final = int(os.getenv("DEFAULT_S3_MAX_CONCURR
|
|||
DEFAULT_S3_MAX_ADAPTIVE_CONCURRENCY: Final = get_env_int("DEFAULT_S3_MAX_ADAPTIVE_CONCURRENCY", 200)
|
||||
# https://docs.aws.amazon.com/AmazonS3/latest/userguide/object-keys.html
|
||||
MAX_S3_OBJECT_KEY_BYTES: Final = 1024
|
||||
MAX_S3_OBJECT_KEY_FILENAME_BYTES: Final = 255
|
||||
S3_BOUNDED_OBJECT_KEY_HEAD_BYTES: Final = 64
|
||||
S3_PREFIX_DIGEST_CHARS: Final = 16
|
||||
# s3 allows 2048 bytes of combined metadata headers, which Content-Disposition counts against
|
||||
|
|
|
|||
|
|
@ -14,6 +14,7 @@ from litellm._logging import print_verbose, verbose_logger
|
|||
from litellm.constants import (
|
||||
MAX_S3_OBJECT_DOWNLOAD_FILENAME_BYTES,
|
||||
MAX_S3_OBJECT_KEY_BYTES,
|
||||
MAX_S3_OBJECT_KEY_FILENAME_BYTES,
|
||||
S3_BOUNDED_OBJECT_KEY_HEAD_BYTES,
|
||||
S3_LOG_PROMPTS_ONLY_ENV_VAR,
|
||||
S3_PREFIX_DIGEST_CHARS,
|
||||
|
|
@ -375,21 +376,28 @@ def get_s3_object_key(
|
|||
sanitized_s3_file_name: Final = s3_file_name.replace("/", "_").replace(":", "_")
|
||||
configured_prefix: Final = (s3_path.rstrip("/") + "/" if s3_path else "") + prefix
|
||||
date_segment: Final = start_time.strftime("%Y-%m-%d") + "/"
|
||||
extension: Final = ".json"
|
||||
file_name_budget: Final = MAX_S3_OBJECT_KEY_FILENAME_BYTES - len(extension.encode("utf-8"))
|
||||
# we need the s3 key to include the time, so we log cache hits too
|
||||
s3_object_key: Final = configured_prefix + date_segment + sanitized_s3_file_name + ".json"
|
||||
if len(s3_object_key.encode("utf-8")) <= MAX_S3_OBJECT_KEY_BYTES:
|
||||
s3_object_key: Final = configured_prefix + date_segment + sanitized_s3_file_name + extension
|
||||
if (
|
||||
len(s3_object_key.encode("utf-8")) <= MAX_S3_OBJECT_KEY_BYTES
|
||||
and len(sanitized_s3_file_name.encode("utf-8")) <= file_name_budget
|
||||
):
|
||||
return s3_object_key
|
||||
|
||||
# shorten the response id first and only trim the configured prefix if that is what does not
|
||||
# fit, so prefix scoped IAM policies and lifecycle rules keep matching
|
||||
budget: Final = MAX_S3_OBJECT_KEY_BYTES - len(date_segment.encode("utf-8")) - len(b".json")
|
||||
budget: Final = MAX_S3_OBJECT_KEY_BYTES - len(date_segment.encode("utf-8")) - len(extension.encode("utf-8"))
|
||||
prefix_bytes: Final = len(configured_prefix.encode("utf-8"))
|
||||
if prefix_bytes + S3_MIN_BOUNDED_FILE_NAME_BYTES <= budget:
|
||||
bounded_file_name: Final = _bounded_s3_file_name(s3_file_name, sanitized_s3_file_name, budget - prefix_bytes)
|
||||
return configured_prefix + date_segment + bounded_file_name + ".json"
|
||||
bounded_file_name: Final = _bounded_s3_file_name(
|
||||
s3_file_name, sanitized_s3_file_name, min(file_name_budget, budget - prefix_bytes)
|
||||
)
|
||||
return configured_prefix + date_segment + bounded_file_name + extension
|
||||
|
||||
shortest_file_name: Final = _bounded_s3_file_name(
|
||||
s3_file_name, sanitized_s3_file_name, S3_MIN_BOUNDED_FILE_NAME_BYTES
|
||||
)
|
||||
bounded_prefix: Final = _bounded_s3_prefix(configured_prefix, budget - len(shortest_file_name.encode("utf-8")))
|
||||
return bounded_prefix + date_segment + shortest_file_name + ".json"
|
||||
return bounded_prefix + date_segment + shortest_file_name + extension
|
||||
|
|
|
|||
|
|
@ -1387,25 +1387,76 @@ def test_create_s3_batch_logging_element_flat_key_for_arn_response_id():
|
|||
|
||||
|
||||
# --------------------------------------------------------------
|
||||
# object keys bounded to S3's 1024 UTF-8 byte limit
|
||||
# object keys bounded to S3 and filesystem-compatible limits
|
||||
# --------------------------------------------------------------
|
||||
def _oversized_response_id() -> str:
|
||||
return "resp_" + "A" * 1100
|
||||
|
||||
|
||||
def test_s3_object_key_at_the_byte_limit_is_left_alone():
|
||||
"""A key that still fits is left byte-identical."""
|
||||
from litellm.constants import MAX_S3_OBJECT_KEY_BYTES
|
||||
def test_s3_object_key_at_the_segment_byte_limit_is_left_alone():
|
||||
"""A final key segment that still fits is left byte-identical."""
|
||||
from litellm.integrations.s3 import get_s3_object_key
|
||||
|
||||
start_time = datetime(2026, 8, 24, 6, 18, 41, 948021)
|
||||
fixed_len = len("input/2026-08-24/.json")
|
||||
file_name = "x" * (MAX_S3_OBJECT_KEY_BYTES - fixed_len)
|
||||
file_name = "x" * 250
|
||||
|
||||
key = get_s3_object_key(s3_path="input", prefix="", start_time=start_time, s3_file_name=file_name)
|
||||
|
||||
assert key == f"input/2026-08-24/{file_name}.json"
|
||||
assert len(key.encode("utf-8")) == MAX_S3_OBJECT_KEY_BYTES
|
||||
assert len(key.rsplit("/", 1)[1].encode("utf-8")) == 255
|
||||
|
||||
|
||||
def test_s3_object_key_just_over_the_file_name_limit_is_bounded():
|
||||
"""A 256-byte final key segment is shortened below the local filesystem limit."""
|
||||
from litellm.integrations.s3 import get_s3_object_key
|
||||
|
||||
key = get_s3_object_key(
|
||||
s3_path="input",
|
||||
prefix="",
|
||||
start_time=datetime(2026, 8, 24, 6, 18, 41, 948021),
|
||||
s3_file_name="x" * 251,
|
||||
)
|
||||
|
||||
assert len(key.rsplit("/", 1)[1].encode("utf-8")) <= 255
|
||||
|
||||
|
||||
def test_s3_object_key_bounds_the_file_segment_when_the_full_key_fits():
|
||||
"""A locally stored object name is bounded even when its full S3 key is below 1024 bytes."""
|
||||
import hashlib
|
||||
|
||||
from litellm.constants import MAX_S3_OBJECT_KEY_BYTES
|
||||
from litellm.integrations.s3 import get_s3_object_key
|
||||
|
||||
file_name = "time-01-14-19-015434_resp_" + "A" * 300
|
||||
|
||||
key = get_s3_object_key(
|
||||
s3_path="requests",
|
||||
prefix="",
|
||||
start_time=datetime(2026, 9, 12, 1, 14, 19, 15434),
|
||||
s3_file_name=file_name,
|
||||
)
|
||||
|
||||
file_segment = key.rsplit("/", 1)[1]
|
||||
assert len(key.encode("utf-8")) < MAX_S3_OBJECT_KEY_BYTES
|
||||
assert len(file_segment.encode("utf-8")) <= 255
|
||||
assert file_segment.startswith("time-01-14-19-015434_resp_")
|
||||
assert file_segment.endswith(f"_{hashlib.sha256(file_name.encode('utf-8')).hexdigest()}.json")
|
||||
|
||||
|
||||
def test_s3_object_key_file_segment_limit_counts_utf8_bytes():
|
||||
"""The final key segment is bounded by encoded bytes without splitting a character."""
|
||||
from litellm.integrations.s3 import get_s3_object_key
|
||||
|
||||
key = get_s3_object_key(
|
||||
s3_path="requests",
|
||||
prefix="",
|
||||
start_time=datetime(2026, 9, 12, 1, 14, 19, 15434),
|
||||
s3_file_name="time-01-14-19-015434_resp_" + "日" * 100,
|
||||
)
|
||||
|
||||
file_segment = key.rsplit("/", 1)[1]
|
||||
assert len(file_segment.encode("utf-8")) <= 255
|
||||
assert "\ufffd" not in file_segment
|
||||
|
||||
|
||||
def test_s3_object_key_is_bounded_for_oversized_response_id():
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue