mirror of
https://github.com/BerriAI/litellm.git
synced 2026-09-17 23:51:30 +00:00
fix(otel): size the message ceiling so every vocabulary fits beside it
Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
parent
492251c7bc
commit
fcaf2d7d98
2 changed files with 14 additions and 10 deletions
|
|
@ -32,15 +32,18 @@ core telemetry no matter how many vocabularies are configured.
|
|||
"""
|
||||
|
||||
|
||||
MAX_MESSAGE_ATTRS_PER_SPAN: Final = DEFAULT_SPAN_ATTRIBUTE_LIMIT // 4
|
||||
MAX_MESSAGE_ATTRS_PER_SPAN: Final = DEFAULT_SPAN_ATTRIBUTE_LIMIT // 8
|
||||
"""Span-wide ceiling on attributes spent spelling out chat messages per index.
|
||||
|
||||
A conversation is the other unbounded family: two attributes per message, for
|
||||
the prompt and the response alike, on the same span. Past a few dozen turns the
|
||||
family alone exceeds the span attribute limit and evicts the core telemetry
|
||||
written before it. The ceiling covers both directions together, since a budget
|
||||
handed to each direction separately doubles. The complete conversation still
|
||||
rides the JSON blob attributes; only the per-index convenience keys are capped.
|
||||
handed to each direction separately doubles. An eighth is the largest share
|
||||
that still fits beside the tool ceiling and the core of every vocabulary at
|
||||
once, request parameters, cost breakdown and identity included. The complete
|
||||
conversation still rides the JSON blob attributes; only the per-index
|
||||
convenience keys are capped.
|
||||
"""
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -461,10 +461,11 @@ def _conversation_payload(turns, choices=1, **overrides):
|
|||
)
|
||||
|
||||
|
||||
def _conversation_span(mapper_names, payload):
|
||||
def _conversation_span(mapper_names, payload, legacy_compat=False):
|
||||
"""The exported LLM-call span for ``payload`` with content capture on."""
|
||||
cfg = OpenTelemetryV2Config(
|
||||
exporter="in_memory",
|
||||
legacy_compat=legacy_compat,
|
||||
mapper_names=list(mapper_names),
|
||||
capture_message_content="span_only",
|
||||
)
|
||||
|
|
@ -541,13 +542,13 @@ def test_message_cap_is_shared_across_input_and_output():
|
|||
) == (MAX_MESSAGE_ATTRS_PER_SPAN // 2)
|
||||
|
||||
|
||||
def test_fully_populated_arize_span_stays_within_the_attribute_limit():
|
||||
def test_fully_populated_span_with_every_vocabulary_stays_within_the_attribute_limit():
|
||||
"""Every capped family maxed at once still leaves the whole core intact.
|
||||
|
||||
The Arize / Phoenix composition (``genai`` + ``openinference`` + ``legacy``)
|
||||
with every request parameter, every cost component, a hundred-plus tools, a
|
||||
two-hundred-turn prompt and twenty choices is the worst case the two
|
||||
span-wide ceilings have to absorb together.
|
||||
Every vocabulary in the registry plus ``legacy``, every request parameter,
|
||||
every cost component, a hundred-plus tools, a two-hundred-turn prompt and
|
||||
twenty choices is the worst case the two span-wide ceilings have to absorb
|
||||
together.
|
||||
"""
|
||||
payload = _conversation_payload(
|
||||
200,
|
||||
|
|
@ -579,7 +580,7 @@ def test_fully_populated_arize_span_stays_within_the_attribute_limit():
|
|||
)
|
||||
},
|
||||
)
|
||||
span = _conversation_span(["genai", "openinference"], payload)
|
||||
span = _conversation_span(["genai", "openinference", "langfuse", "weave", "langtrace"], payload, legacy_compat=True)
|
||||
a = span.attributes
|
||||
|
||||
assert span.dropped_attributes == 0
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue