fix(otel): repair Arize OTel v2 regressions from #43698 (linear fit, metadata slot, repr tool args) (#44488)

* fix(otel): keep Arize OTel v2 regressions in check — O(n) fit, metadata slot, repr tool args

Three regressions from #43698's OpenInference tool-call/metadata emission:

1. Metadata evicted indexed message attributes: the new `metadata` key
   competed for the 128-attribute span budget, and the fit sheds whole
   message groups BEFORE `span.set_attribute`, so the SDK's dropped
   counter stayed 0 — an invisible eviction (live A/B: input-message
   attributes 86 -> 84). Two-part fix: the fit pins `metadata` behind
   every message group (it sheds only once all indexed messages are
   gone), and the span budget no longer charges pre-set attributes the
   mappers overwrite in place — a boundary-opened LLM span already
   carries keys like `gen_ai.request.model`, so the old accounting
   reserved slots the fit could never spend. Live: 86 input-message
   attributes with the metadata attribute riding alongside.

2. Quadratic shed on long prompts: `_message_shed_groups` rescanned the
   full group map once per message (measured on a real acompletion:
   0.032/0.128/0.478/1.910s at 1000/2000/4000/8000 messages vs
   0.007/0.010/0.022/0.029s at base). Index the tool-call groups once by
   (family, message index): the fit is linear again (0.008/0.007/0.014/
   0.031s, same rig).

3. Malformed Python tool arguments lost the whole span: provider
   adapters and `model_construct` responses hand over raw objects, and
   `json.dumps` raises on tuple-keyed dicts (TypeError) and cycles
   (ValueError) before the span is exported. Serialize with a repr
   fallback; both cases now export with a readable arguments attribute.

The attribute budget change affects every boundary-opened LLM-call span
(strictly more attributes retained, never fewer); the mapper changes
only touch the OpenInference vocabulary.

* refactor(otel): build the tool-call group index in one shot

Review follow-up: the dict.setdefault/append seeding in
_tool_call_groups_by_message violated the no-mutation coding convention
(AGENTS.md: build values in one shot with comprehensions or generators
wrapped in tuple()/MappingProxyType()). Rebuild it as a sorted groupby
comprehension; randomized parity harness confirms the shed order is
byte-identical to the seeded version (400 trials).

Also pin the overflow corner Greptile asked about: a pre-set
indexed-message key the fit sheds keeps its earlier value in place, so
the span total can never exceed the SDK limit (new emitter test).

* fix(otel): key the groupby with an explicit tuple to keep basedpyright at budget

The slice-keyed groupby (group[:2]) widened the key to tuple[str | int],
adding one reportGeneralTypeIssues over the codebase ceiling. Key by the
explicit (family, message index) pair instead; shed order unchanged
(300-trial randomized parity harness).

* fix(otel): read pre-set span keys through a helper typed for both runtime shapes

The SDK annotates ReadableSpan.attributes as a Mapping, but an ended span
hands back a tuple of pairs, so the inline isinstance branch narrowed to
Never and pushed reportGeneralTypeIssues one over the codebase ceiling.
Extract _carried_keys with the runtime union declared on the parameter;
behavior unchanged.
This commit is contained in:
yucheng-berri 2026-10-04 03:09:38 -07:00 • committed by Yucheng He
parent ae5f4ce2e4
commit b21e8ea009
5 changed files with 349 additions and 23 deletions

View file

@ -91,13 +91,32 @@ def span_attribute_limit(span: Span) -> int | None:
return span._limits.max_span_attributes # pyright: ignore[reportPrivateUsage] # SDK has no public getter
def attribute_budget(span: Span, reserved: int) -> int | None:
"""How many mapped attributes fit on ``span`` next to what it already carries and ``reserved`` more."""
def _carried_keys(
attributes: Mapping[str, AttrValue] | tuple[tuple[str, AttrValue], ...],
) -> frozenset[str]:
"""The keys a span already carries: a live span exposes a ``Mapping``, an ended one a tuple of pairs."""
if isinstance(attributes, Mapping):
return frozenset(attributes)
return frozenset(key for key, _value in attributes)
def attribute_budget(span: Span, reserved: int, overwrites: frozenset[str] = frozenset()) -> int | None:
"""How many mapped attributes fit on ``span`` next to what it already carries and ``reserved`` more.
``overwrites`` are the mapped keys already present on ``span``: setting one
replaces the value in place and consumes no slot against the limit, so only
the genuinely new pre-existing keys reduce the budget. Counting the
overwritten ones too reserves slots the fit can never spend and sheds
indexed message attributes for nothing.
"""
limit: Final = span_attribute_limit(span)
if limit is None:
return None
on_span: Final = len(span.attributes or ()) if isinstance(span, ReadableSpan) else 0
return limit - on_span - reserved
if not isinstance(span, ReadableSpan):
return limit - reserved
existing_keys: Final = _carried_keys(span.attributes or ())
fresh: Final = len(existing_keys - overwrites) if overwrites else len(existing_keys)
return limit - fresh - reserved
def stamp_error(
@ -284,7 +303,7 @@ class SpanEmitter:
)
stamped_later: Final = error_attributes(error) if error else _NO_ATTRIBUTES
reserved: Final = len(stamped_later.keys() - mapped.keys())
for key, value in fit_indexed_messages(mapped, attribute_budget(span, reserved)).items():
for key, value in fit_indexed_messages(mapped, attribute_budget(span, reserved, frozenset(mapped))).items():
span.set_attribute(key, value)
if error:
stamped: Final = stamp_error(span, error)

View file

@ -99,17 +99,33 @@ def _message_key_groups(attrs: Mapping[str, AttrValue]) -> Mapping[tuple[str, in
)
def _message_shed_groups(
groups: Mapping[tuple[str, int, int], tuple[str, ...]], family: str, message_idx: int
) -> Iterator[tuple[str, int, int]]:
tool_call_groups: Final = tuple(
sorted(
(group for group in groups if group[:2] == (family, message_idx) and group[2] != _MESSAGE_BASE),
key=lambda group: group[2],
reverse=True,
)
def _tool_call_groups_by_message(
groups: Mapping[tuple[str, int, int], tuple[str, ...]],
) -> Mapping[tuple[str, int], tuple[tuple[str, int, int], ...]]:
"""Tool-call groups indexed by ``(family, message index)``, each tuple highest tool index first.
Indexing once keeps the shed order linear in the group count: rescanning the
full group map per message made attribute fitting quadratic on long prompts.
"""
ordered: Final = sorted(
(group for group in groups if group[2] != _MESSAGE_BASE),
key=lambda group: (group[0], group[1], group[2]),
)
yield from tool_call_groups
return MappingProxyType(
{
message: tuple(reversed(tuple(message_tool_groups)))
for message, message_tool_groups in groupby(ordered, key=lambda group: (group[0], group[1]))
}
)
def _message_shed_groups(
groups: Mapping[tuple[str, int, int], tuple[str, ...]],
tool_call_groups: Mapping[tuple[str, int], tuple[tuple[str, int, int], ...]],
family: str,
message_idx: int,
) -> Iterator[tuple[str, int, int]]:
yield from tool_call_groups.get((family, message_idx), ())
base_group: Final = (family, message_idx, _MESSAGE_BASE)
if base_group in groups:
yield base_group
@ -117,6 +133,7 @@ def _message_shed_groups(
def _shed_order(groups: Mapping[tuple[str, int, int], tuple[str, ...]]) -> tuple[tuple[str, int, int], ...]:
"""Middle inputs, extra choices, pinned inputs, then the first choice, with tool calls before message keys."""
tool_call_groups: Final = _tool_call_groups_by_message(groups)
inputs: Final = sorted(frozenset(idx for family, idx, _ in groups if family == _INPUT_MESSAGES))
outputs: Final = sorted(frozenset(idx for family, idx, _ in groups if family == _OUTPUT_MESSAGES))
pinned_inputs: Final = tuple(dict.fromkeys((*inputs[:1], *inputs[-1:])))
@ -127,24 +144,35 @@ def _shed_order(groups: Mapping[tuple[str, int, int], tuple[str, ...]]) -> tuple
*((_OUTPUT_MESSAGES, idx) for idx in outputs[:1]),
)
return tuple(
chain.from_iterable(_message_shed_groups(groups, family, message_idx) for family, message_idx in message_order)
chain.from_iterable(
_message_shed_groups(groups, tool_call_groups, family, message_idx) for family, message_idx in message_order
)
)
_METADATA_KEY: Final = "metadata"
def fit_indexed_messages(attrs: Mapping[str, AttrValue], budget: int | None) -> Mapping[str, AttrValue]:
"""``attrs`` with indexed message attributes shed, least valuable first, until at most ``budget`` keys remain.
``None`` means the span has no attribute count limit. Every message still rides the ``input.value`` and
``output.value`` blobs, so shedding a per-index pair loses no content.
``output.value`` blobs, so shedding a per-index pair loses no content. The ``metadata`` blob sheds only
after every indexed message attribute: message attributes are the indexed, queryable view (the blobs
carry no per-index keys), so a count-squeezed span keeps them and the single metadata key absorbs only
the residual shortfall. Shedding happens here, before ``span.set_attribute``, so the fit is exact and
the SDK's dropped-attributes counter never silently masks the choice.
"""
if budget is None or len(attrs) <= budget:
return attrs
groups: Final = _message_key_groups(attrs)
order: Final = _shed_order(groups)
running: Final = tuple(accumulate(len(groups[group]) for group in order))
sheddable: Final = (*(groups[group] for group in _shed_order(groups)),)
flex: Final = (_METADATA_KEY,) if _METADATA_KEY in attrs else ()
candidates: Final = (*sheddable, flex) if flex else sheddable
running: Final = tuple(accumulate(len(candidate) for candidate in candidates))
excess: Final = len(attrs) - budget
shed_count: Final = next((n + 1 for n, total in enumerate(running) if total >= excess), len(order))
shed: Final = frozenset(chain.from_iterable(groups[group] for group in order[:shed_count]))
shed_count: Final = next((n + 1 for n, total in enumerate(running) if total >= excess), len(candidates))
shed: Final = frozenset(chain.from_iterable(candidates[:shed_count]))
return MappingProxyType({key: value for key, value in attrs.items() if key not in shed})

View file

@ -159,7 +159,7 @@ def _message_tool_call(value: object) -> MessageToolCall | None:
arguments: Final = (
raw_arguments
if isinstance(raw_arguments, str)
else json.dumps(raw_arguments, default=str)
else _stringify_tool_arguments(raw_arguments)
if raw_arguments is not None
else None
)
@ -171,3 +171,17 @@ def _message_tool_call(value: object) -> MessageToolCall | None:
name=name if isinstance(name, str) else None,
arguments=arguments,
)
def _stringify_tool_arguments(value: object) -> str:
"""Serialize non-string tool-call arguments, falling back to ``repr``.
Arguments normally arrive as already-JSON strings, but provider adapters and
``model_construct`` responses hand over raw Python objects. ``json.dumps``
raises on those (tuple-keyed dicts, cycles) and the escaping exception would
lose the whole span, so keep a readable ``repr`` instead.
"""
try:
return json.dumps(value, default=str)
except (TypeError, ValueError):
return repr(value)

View file

@ -20,7 +20,7 @@ from litellm.integrations.otel import ( # noqa: E402
)
from litellm.integrations.otel.plumbing import context as ctx_mod # noqa: E402
from litellm.integrations.otel.plumbing import providers # noqa: E402
from litellm.integrations.otel.emitter import SpanEmitter, span_attribute_limit # noqa: E402
from litellm.integrations.otel.emitter import SpanEmitter, attribute_budget, span_attribute_limit # noqa: E402
from litellm.integrations.otel.emitter import stamp_error # noqa: E402
from litellm.integrations.otel.mappers.utils import MAX_TOOL_DEFINITION_ATTRS_PER_SPAN # noqa: E402
from litellm.integrations.otel.model.payloads import ( # noqa: E402
@ -558,6 +558,113 @@ def test_prompt_turns_are_shed_before_response_choices():
)
def test_metadata_blob_keeps_the_indexed_messages_it_competes_with():
"""The promoted ``metadata`` blob rides alongside the indexed messages on a boundary-opened span.
The live regression shape: the LLM-call span is opened at the request
boundary and already carries a few stamped attributes, most of which the
mappers re-emit under the same keys. Charging those overwritten keys
against the attribute budget reserved slots the fit could never spend, so
the new ``metadata`` key displaced a whole middle message group — an
eviction invisible to the SDK's dropped-attributes counter because the fit
sheds before ``span.set_attribute``. With the budget counting only keys
the fit does not overwrite, the metadata key rides and the span keeps the
indexed-message count of the metadata-free span (at shedding granularity).
"""
cfg = OpenTelemetryV2Config(
exporter="in_memory",
legacy_compat=False,
mapper_names=["genai", "openinference"],
capture_message_content="span_only",
)
provider, exporter = providers.in_memory_provider(cfg)
engine = SpanEmitter(providers.get_tracer(provider, "litellm-test"), cfg)
def boundary_span(payload):
span = engine.start_span(SpanRole.LLM_CALL, "chat gpt-4o")
# what the boundary opener stamps before the typed payload exists
span.set_attribute(GenAI.REQUEST_MODEL, "gpt-4o")
span.set_attribute(LiteLLM.PROVIDER_MODEL, "gpt-4o-2024")
span.set_attribute("litellm.metadata.user_api_key_alias", "edge-key")
engine.finish_span(
SpanRole.LLM_CALL,
span,
LLMCallSpanData.from_standard_logging_payload(
payload, capture_content=True, metadata_keys=("user_api_key_alias",)
),
)
(finished,) = exporter.get_finished_spans()
exporter.clear()
return finished
with_metadata = boundary_span(_conversation_payload(47, metadata={"user_api_key_alias": "edge-key"}))
without_metadata = boundary_span(_conversation_payload(47))
_assert_core_intact(with_metadata)
a = with_metadata.attributes
assert json.loads(a["metadata"]) == {"user_api_key_alias": "edge-key"}
assert "metadata" not in without_metadata.attributes
kept = _indexed_messages(a, "llm.input_messages")
baseline = _indexed_messages(without_metadata.attributes, "llm.input_messages")
assert kept == baseline or kept == baseline[:-1]
assert kept[0] == 0 and kept[-1] == 46
assert len(a) <= SpanLimits().max_span_attributes
def test_preset_message_key_the_fit_sheds_still_never_overflows_the_span():
"""A pre-set indexed-message key the fit later sheds cannot push the span over its limit.
The budget treats every mapped key already on the span as an overwrite
(free). If the fit then sheds that key, the value stamped earlier simply
stays in its slot, so the span holds one entry for it either way and the
total never exceeds the limit — the SDK's dropped-attributes counter stays
at zero.
"""
cfg = OpenTelemetryV2Config(
exporter="in_memory",
legacy_compat=False,
mapper_names=["genai", "openinference"],
capture_message_content="span_only",
)
provider, exporter = providers.in_memory_provider(cfg)
engine = SpanEmitter(providers.get_tracer(provider, "litellm-test"), cfg)
span = engine.start_span(SpanRole.LLM_CALL, "chat gpt-4o")
span.set_attribute("llm.input_messages.1.message.role", "stale-role")
engine.finish_span(
SpanRole.LLM_CALL,
span,
LLMCallSpanData.from_standard_logging_payload(
_conversation_payload(60), capture_content=True, metadata_keys=("user_api_key_alias",)
),
)
(finished,) = exporter.get_finished_spans()
_assert_core_intact(finished)
assert len(finished.attributes) <= SpanLimits().max_span_attributes
def test_attribute_budget_counts_only_keys_the_fit_does_not_overwrite():
"""Pre-set attributes the mapped set overwrites consume no slot against the span limit.
A boundary-opened LLM-call span already carries a few stamped attributes,
most of which the mappers re-emit under the same keys. Charging those
against the budget reserves slots the fit can never spend and sheds
indexed message attributes for nothing.
"""
cfg = OpenTelemetryV2Config(exporter="in_memory")
provider, _exporter = providers.in_memory_provider(cfg)
span = providers.get_tracer(provider, "litellm-test").start_span("s")
span.set_attribute("gen_ai.request.model", "gpt-4o")
span.set_attribute("litellm.metadata.user_api_key_alias", "edge-key")
try:
assert attribute_budget(span, 0) == SpanLimits().max_span_attributes - 2
assert (
attribute_budget(span, 0, frozenset({"gen_ai.request.model"}))
== SpanLimits().max_span_attributes - 1
)
finally:
span.end()
def test_indexed_messages_respect_a_lower_span_attribute_count_limit(monkeypatch):
"""The budget follows the SDK's configured limit, not a hardcoded default."""
monkeypatch.setenv("OTEL_SPAN_ATTRIBUTE_COUNT_LIMIT", "48")

View file

@ -6,6 +6,7 @@ backends, so one trace lights up every configured destination.
"""
import json
import time
from collections.abc import Mapping
from itertools import chain
from typing import Final
@ -347,6 +348,163 @@ def test_openinference_metadata_contains_only_promoted_metadata():
assert "metadata" not in OpenInferenceMapper().map(_llm_call())
def _long_prompt_messages(count: int) -> tuple[dict[str, str], ...]:
return tuple({"role": "user" if i == 0 else "assistant", "content": f"turn {i}"} for i in range(count))
def _input_message_keys(attrs: Mapping[str, object]) -> frozenset[str]:
return frozenset(key for key in attrs if key.startswith("llm.input_messages."))
def test_openinference_metadata_does_not_displace_indexed_messages_under_budget():
"""The ``metadata`` blob rides in reclaimed slots instead of evicting messages.
``metadata`` competes for the OTel 128-attribute span budget, and shedding
happens before ``span.set_attribute``, so the SDK's dropped-attributes
counter stays at zero — the eviction is invisible. The fit pins the
metadata key behind every message group: a squeezed span keeps the message
attributes it would keep without the metadata blob (at whole-group
granularity, at most one group of headroom difference), and the metadata
key survives alongside them.
"""
messages: Final = _long_prompt_messages(62)
def fitted(promoted: Mapping[str, str], budget: int | None = None):
mapped: Final = OpenInferenceMapper().map(_llm_call(messages_in=messages, promoted_metadata=promoted))
assert len(mapped) > 128, "fixture must squeeze the span attribute budget"
return fit_indexed_messages(mapped, budget if budget is not None else len(mapped) - 5)
without_metadata: Final = fitted({})
with_metadata: Final = fitted({"user_api_key_alias": "edge-key"})
assert "metadata" not in without_metadata
assert json.loads(with_metadata["metadata"]) == {"user_api_key_alias": "edge-key"}
assert _input_message_keys(with_metadata) == _input_message_keys(without_metadata)
# Against the absolute span limit the displacement is bounded by the
# whole-group shedding granularity: at most one message group.
without_at_limit: Final = fitted({}, 128)
with_at_limit: Final = fitted({"user_api_key_alias": "edge-key"}, 128)
assert len(_input_message_keys(with_at_limit)) >= len(_input_message_keys(without_at_limit)) - 2
assert "metadata" in with_at_limit
def test_openinference_metadata_sheds_only_after_every_indexed_message():
"""Metadata sheds last: only once every indexed message attribute is gone."""
mapped: Final = OpenInferenceMapper().map(
_llm_call(messages_in=_long_prompt_messages(6), promoted_metadata={"user_api_key_alias": "edge-key"})
)
message_keys: Final = frozenset(key for key in mapped if ".message." in key)
# Budget too small for the message family alone: everything indexed goes,
# and the metadata blob absorbs the residual shortfall with it.
starved: Final = fit_indexed_messages(mapped, len(mapped) - len(message_keys) - 1)
assert not any(key in starved for key in message_keys)
assert "metadata" not in starved
# One slot more and metadata survives alongside zero indexed messages.
last_standing: Final = fit_indexed_messages(mapped, len(mapped) - len(message_keys))
assert not any(key in last_standing for key in message_keys)
assert "metadata" in last_standing
def test_openinference_raw_tool_arguments_fall_back_to_repr_instead_of_raising():
"""Malformed Python tool arguments must not lose the span.
A plain ``Function()`` constructor JSON-serializes arguments, but provider
adapters and ``model_construct`` responses hand over raw Python objects —
tuple-keyed dicts and cycles that ``json.dumps`` raises on. The mapper
serializes them with a ``repr`` fallback instead of letting the exception
escape before the span is exported.
"""
# rebind-ok: a self-referencing dict cannot be built in one shot — the cycle
# only exists once the finished dict is inserted into itself.
circular: dict[str, object] = {}
circular["self"] = circular
for label, raw_arguments in (("tuple-key", {(1, 2): "v"}), ("circular", circular)):
data: Final = _llm_call(
choices_out=(
{
"finish_reason": "tool_calls",
"message": {
"role": "assistant",
"content": None,
"tool_calls": [
{"id": "tc1", "type": "function", "function": {"name": "f", "arguments": raw_arguments}}
],
},
},
)
)
attrs: Final = OpenInferenceMapper().map(data)
assert (
attrs["llm.output_messages.0.message.tool_calls.0.tool_call.function.arguments"] == repr(raw_arguments)
), label
assert attrs["llm.output_messages.0.message.role"] == "assistant"
def test_openinference_payload_tool_arguments_with_raw_objects_map_without_raising():
"""The standard-logging payload path (provider adapter responses) survives raw argument objects too."""
payload: Final = {
"call_type": "acompletion",
"custom_llm_provider": "openai",
"model": "gpt-4o",
"prompt_tokens": 3,
"completion_tokens": 2,
"total_tokens": 5,
"stream": False,
"model_parameters": {},
"response": {
"id": "resp_bad",
"model": "gpt-4o-2024",
"choices": [
{
"index": 0,
"finish_reason": "tool_calls",
"message": {
"role": "assistant",
"content": "hi",
"tool_calls": [
{"id": "tc1", "type": "function", "function": {"name": "f", "arguments": {(1, 2): "v"}}}
],
},
}
],
},
"metadata": {"user_api_key_alias": "edge-key"},
"status": "success",
"litellm_call_id": "call_raw_args",
"hidden_params": {},
}
data: Final = LLMCallSpanData.from_standard_logging_payload(
payload, capture_content=True, metadata_keys=("user_api_key_alias",)
)
attrs: Final = OpenInferenceMapper().map(data)
arguments: Final = attrs["llm.output_messages.0.message.tool_calls.0.tool_call.function.arguments"]
assert arguments == repr({(1, 2): "v"})
assert json.loads(attrs["metadata"]) == {"user_api_key_alias": "edge-key"}
def test_openinference_attribute_fit_stays_linear_on_long_prompts():
"""Attribute fitting is linear in the message count, not quadratic.
Rescanning the full group map once per message made the fit quadratic in
the prompt length (~1.9s of callback time at 8000 messages); indexing the
groups once keeps it in milliseconds. The bound sits far above the linear
runtime so it cannot flake on slow runners, while the quadratic path
exceeds it several times over.
"""
messages: Final = _long_prompt_messages(4000)
mapped: Final = OpenInferenceMapper().map(
_llm_call(messages_in=messages, promoted_metadata={"user_api_key_alias": "k"})
)
started: Final = time.perf_counter()
fitted: Final = fit_indexed_messages(mapped, 128)
elapsed: Final = time.perf_counter() - started
assert elapsed < 0.25
assert len(fitted) <= 128
assert fitted["llm.input_messages.0.message.role"] == "user"
assert fitted["llm.input_messages.3999.message.role"] == "assistant"
# --------------------------------------------------------------------------- #
# Langfuse
# --------------------------------------------------------------------------- #