azure_sentinel_truncate Move truncation size to env for more user control

This commit is contained in:
alihacks 2026-04-25 19:38:51 -04:00
parent 374217115d
commit 7c74005b7a
2 changed files with 37 additions and 35 deletions

View file

@ -121,13 +121,18 @@ class AzureSentinelLogger(CustomBatchLogger):
asyncio.create_task(self.periodic_flush())
self.log_queue: List[StandardLoggingPayload] = []
# When True, string fields (messages, response) are truncated to the
# Azure Log Analytics column limit (256 KB / 262,144 chars). Azure
# silently truncates at this limit anyway; doing it ourselves lets us
# keep the tail (most recent content) and record metadata.
# Controlled by AZURE_SENTINEL_TRUNCATE_CONTENT env var (default: false).
truncate_env = os.getenv("AZURE_SENTINEL_TRUNCATE_CONTENT", "false")
self.truncate_content = truncate_env.lower() in ("true", "1", "yes")
# When set to a positive integer, string fields (messages, response)
# are truncated to that many characters. Azure Log Analytics silently
# truncates string columns when over the limits doing it here
# ourselves lets us keep the tail (most recent content) and record
# metadata. Set AZURE_SENTINEL_TRUNCATE_BYTES to the desired char
# limit (e.g. "262144"). When unset or "0", truncation is disabled.
truncate_bytes_env = os.getenv("AZURE_SENTINEL_TRUNCATE_BYTES", "0")
try:
self.truncate_max_chars = int(truncate_bytes_env)
except ValueError:
self.truncate_max_chars = 0
self.truncate_content = self.truncate_max_chars > 0
async def _get_oauth_token(self) -> str:
"""
@ -253,18 +258,13 @@ class AzureSentinelLogger(CustomBatchLogger):
# We target a conservative threshold to stay safely under the limit.
MAX_BATCH_SIZE_BYTES = 950_000 # ~950KB uncompressed target per batch
# Azure Log Analytics silently truncates string column values at 256 KB.
# We enforce this limit ourselves so we can keep the tail (most recent
# content) and record truncation metadata.
MAX_COLUMN_CHARS = 262_144 # 256 KB Azure Log Analytics column limit
def _enforce_column_limits(
self, payload: StandardLoggingPayload
) -> StandardLoggingPayload:
"""
Truncate messages/response string fields to the Azure Log Analytics
column limit (262,144 chars). Keeps the *tail* of each field so that
the most recent conversation turns and response text are preserved.
Truncate messages/response string fields to ``self.truncate_max_chars``
characters. Keeps the *tail* of each field so that the most recent
conversation turns and response text are preserved.
Only called when ``self.truncate_content`` is True.
@ -272,7 +272,7 @@ class AzureSentinelLogger(CustomBatchLogger):
limit, otherwise returns a deep copy with truncated fields and
truncation metadata added.
"""
limit = self.MAX_COLUMN_CHARS
limit = self.truncate_max_chars
messages = payload.get("messages")
response = payload.get("response")
msg_str = str(messages) if messages is not None else ""
@ -328,8 +328,8 @@ class AzureSentinelLogger(CustomBatchLogger):
Splits payloads into gzip-compressed batches that stay under
MAX_BATCH_SIZE_BYTES (uncompressed) per batch.
When truncate_content is enabled, enforces the Azure Log Analytics
column limit (256 KB) on messages/response fields before batching.
When truncate_content is enabled, enforces the configured character
limit on messages/response fields before batching.
Returns a list of gzip-compressed byte strings, each representing
a JSON array of log entries.

View file

@ -139,8 +139,7 @@ async def test_azure_sentinel_batch_splitting():
@pytest.mark.asyncio
async def test_azure_sentinel_split_into_batches_single_oversized_entry():
"""Test that a single entry larger than MAX_BATCH_SIZE_BYTES is sent alone"""
with patch.dict(os.environ, {"AZURE_SENTINEL_TRUNCATE_CONTENT": "false"}):
logger = _make_logger()
logger = _make_logger()
# One very large payload that exceeds the batch size on its own.
# Use varied content so JSON serialization keeps it large.
@ -177,11 +176,11 @@ async def test_azure_sentinel_split_into_batches_single_oversized_entry():
def test_column_limit_truncates_large_fields():
"""Test that fields exceeding 262,144 chars are truncated (keeping the tail)"""
with patch.dict(os.environ, {"AZURE_SENTINEL_TRUNCATE_CONTENT": "true"}):
"""Test that fields exceeding the configured limit are truncated (keeping the tail)"""
with patch.dict(os.environ, {"AZURE_SENTINEL_TRUNCATE_BYTES": "262144"}):
logger = _make_logger()
# Content larger than 256 KB column limit
# Content larger than the configured limit
big_messages = "A" * 300_000
big_response = "B" * 300_000
@ -199,12 +198,12 @@ def test_column_limit_truncates_large_fields():
# Messages field should be truncated, keeping tail
msg_str = str(result["messages"])
assert msg_str.startswith("[truncated by litellm]...")
assert len(msg_str) <= logger.MAX_COLUMN_CHARS
assert len(msg_str) <= logger.truncate_max_chars
# Response field should be truncated, keeping tail
resp_str = str(result["response"])
assert resp_str.startswith("[truncated by litellm]...")
assert len(resp_str) <= logger.MAX_COLUMN_CHARS
assert len(resp_str) <= logger.truncate_max_chars
# Truncation metadata present
metadata = result.get("metadata", {})
@ -217,15 +216,15 @@ def test_column_limit_truncates_large_fields():
assert "messages" in trunc_info["truncated_fields"]
assert "response" in trunc_info["truncated_fields"]
assert trunc_info["original_messages_chars"] == len(str(payload["messages"]))
assert trunc_info["max_column_chars"] == 262_144
assert trunc_info["max_column_chars"] == 262144
# Original payload not mutated
assert len(str(payload["messages"])) > 262_144
assert len(str(payload["messages"])) > 262144
def test_column_limit_preserves_small_payloads():
"""Test that payloads under the column limit are returned unchanged"""
with patch.dict(os.environ, {"AZURE_SENTINEL_TRUNCATE_CONTENT": "true"}):
"""Test that payloads under the configured limit are returned unchanged"""
with patch.dict(os.environ, {"AZURE_SENTINEL_TRUNCATE_BYTES": "262144"}):
logger = _make_logger()
payload = _make_payload(
@ -242,13 +241,15 @@ def test_column_limit_preserves_small_payloads():
assert "litellm_content_truncated" not in metadata
def test_truncate_disabled_via_env_var():
"""Test that truncation is skipped when AZURE_SENTINEL_TRUNCATE_CONTENT=false"""
with patch.dict(os.environ, {"AZURE_SENTINEL_TRUNCATE_CONTENT": "false"}):
def test_truncate_disabled_when_env_unset():
"""Test that truncation is skipped when AZURE_SENTINEL_TRUNCATE_BYTES is not set"""
with patch.dict(os.environ, {}, clear=False):
# Ensure the env var is absent
os.environ.pop("AZURE_SENTINEL_TRUNCATE_BYTES", None)
logger = _make_logger()
assert logger.truncate_content is False
# Create payload with content exceeding 256 KB column limit
# Create payload with content exceeding 256 KB
huge_content = "z" * 300_000
payloads = [
_make_payload(
@ -270,11 +271,12 @@ def test_truncate_disabled_via_env_var():
def test_truncate_enabled_in_split_batches():
"""Test that _split_into_batches truncates large fields when enabled"""
with patch.dict(os.environ, {"AZURE_SENTINEL_TRUNCATE_CONTENT": "true"}):
with patch.dict(os.environ, {"AZURE_SENTINEL_TRUNCATE_BYTES": "262144"}):
logger = _make_logger()
assert logger.truncate_content is True
assert logger.truncate_max_chars == 262144
# Content exceeding 256 KB column limit
# Content exceeding the configured limit
huge_content = "X" * 400_000
payloads = [
_make_payload(