mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-11 03:38:38 +00:00
fix(otel): repair Arize OTel v2 regressions from #43698 (linear fit, metadata slot, repr tool args) (#44488)
* fix(otel): keep Arize OTel v2 regressions in check — O(n) fit, metadata slot, repr tool args Three regressions from #43698's OpenInference tool-call/metadata emission: 1. Metadata evicted indexed message attributes: the new `metadata` key competed for the 128-attribute span budget, and the fit sheds whole message groups BEFORE `span.set_attribute`, so the SDK's dropped counter stayed 0 — an invisible eviction (live A/B: input-message attributes 86 -> 84). Two-part fix: the fit pins `metadata` behind every message group (it sheds only once all indexed messages are gone), and the span budget no longer charges pre-set attributes the mappers overwrite in place — a boundary-opened LLM span already carries keys like `gen_ai.request.model`, so the old accounting reserved slots the fit could never spend. Live: 86 input-message attributes with the metadata attribute riding alongside. 2. Quadratic shed on long prompts: `_message_shed_groups` rescanned the full group map once per message (measured on a real acompletion: 0.032/0.128/0.478/1.910s at 1000/2000/4000/8000 messages vs 0.007/0.010/0.022/0.029s at base). Index the tool-call groups once by (family, message index): the fit is linear again (0.008/0.007/0.014/ 0.031s, same rig). 3. Malformed Python tool arguments lost the whole span: provider adapters and `model_construct` responses hand over raw objects, and `json.dumps` raises on tuple-keyed dicts (TypeError) and cycles (ValueError) before the span is exported. Serialize with a repr fallback; both cases now export with a readable arguments attribute. The attribute budget change affects every boundary-opened LLM-call span (strictly more attributes retained, never fewer); the mapper changes only touch the OpenInference vocabulary. * refactor(otel): build the tool-call group index in one shot Review follow-up: the dict.setdefault/append seeding in _tool_call_groups_by_message violated the no-mutation coding convention (AGENTS.md: build values in one shot with comprehensions or generators wrapped in tuple()/MappingProxyType()). Rebuild it as a sorted groupby comprehension; randomized parity harness confirms the shed order is byte-identical to the seeded version (400 trials). Also pin the overflow corner Greptile asked about: a pre-set indexed-message key the fit sheds keeps its earlier value in place, so the span total can never exceed the SDK limit (new emitter test). * fix(otel): key the groupby with an explicit tuple to keep basedpyright at budget The slice-keyed groupby (group[:2]) widened the key to tuple[str | int], adding one reportGeneralTypeIssues over the codebase ceiling. Key by the explicit (family, message index) pair instead; shed order unchanged (300-trial randomized parity harness). * fix(otel): read pre-set span keys through a helper typed for both runtime shapes The SDK annotates ReadableSpan.attributes as a Mapping, but an ended span hands back a tuple of pairs, so the inline isinstance branch narrowed to Never and pushed reportGeneralTypeIssues one over the codebase ceiling. Extract _carried_keys with the runtime union declared on the parameter; behavior unchanged.
This commit is contained in:
parent
ae5f4ce2e4
commit
b21e8ea009
5 changed files with 349 additions and 23 deletions
|
|
@ -91,13 +91,32 @@ def span_attribute_limit(span: Span) -> int | None:
|
|||
return span._limits.max_span_attributes # pyright: ignore[reportPrivateUsage] # SDK has no public getter
|
||||
|
||||
|
||||
def attribute_budget(span: Span, reserved: int) -> int | None:
|
||||
"""How many mapped attributes fit on ``span`` next to what it already carries and ``reserved`` more."""
|
||||
def _carried_keys(
|
||||
attributes: Mapping[str, AttrValue] | tuple[tuple[str, AttrValue], ...],
|
||||
) -> frozenset[str]:
|
||||
"""The keys a span already carries: a live span exposes a ``Mapping``, an ended one a tuple of pairs."""
|
||||
if isinstance(attributes, Mapping):
|
||||
return frozenset(attributes)
|
||||
return frozenset(key for key, _value in attributes)
|
||||
|
||||
|
||||
def attribute_budget(span: Span, reserved: int, overwrites: frozenset[str] = frozenset()) -> int | None:
|
||||
"""How many mapped attributes fit on ``span`` next to what it already carries and ``reserved`` more.
|
||||
|
||||
``overwrites`` are the mapped keys already present on ``span``: setting one
|
||||
replaces the value in place and consumes no slot against the limit, so only
|
||||
the genuinely new pre-existing keys reduce the budget. Counting the
|
||||
overwritten ones too reserves slots the fit can never spend and sheds
|
||||
indexed message attributes for nothing.
|
||||
"""
|
||||
limit: Final = span_attribute_limit(span)
|
||||
if limit is None:
|
||||
return None
|
||||
on_span: Final = len(span.attributes or ()) if isinstance(span, ReadableSpan) else 0
|
||||
return limit - on_span - reserved
|
||||
if not isinstance(span, ReadableSpan):
|
||||
return limit - reserved
|
||||
existing_keys: Final = _carried_keys(span.attributes or ())
|
||||
fresh: Final = len(existing_keys - overwrites) if overwrites else len(existing_keys)
|
||||
return limit - fresh - reserved
|
||||
|
||||
|
||||
def stamp_error(
|
||||
|
|
@ -284,7 +303,7 @@ class SpanEmitter:
|
|||
)
|
||||
stamped_later: Final = error_attributes(error) if error else _NO_ATTRIBUTES
|
||||
reserved: Final = len(stamped_later.keys() - mapped.keys())
|
||||
for key, value in fit_indexed_messages(mapped, attribute_budget(span, reserved)).items():
|
||||
for key, value in fit_indexed_messages(mapped, attribute_budget(span, reserved, frozenset(mapped))).items():
|
||||
span.set_attribute(key, value)
|
||||
if error:
|
||||
stamped: Final = stamp_error(span, error)
|
||||
|
|
|
|||
|
|
@ -99,17 +99,33 @@ def _message_key_groups(attrs: Mapping[str, AttrValue]) -> Mapping[tuple[str, in
|
|||
)
|
||||
|
||||
|
||||
def _message_shed_groups(
|
||||
groups: Mapping[tuple[str, int, int], tuple[str, ...]], family: str, message_idx: int
|
||||
) -> Iterator[tuple[str, int, int]]:
|
||||
tool_call_groups: Final = tuple(
|
||||
sorted(
|
||||
(group for group in groups if group[:2] == (family, message_idx) and group[2] != _MESSAGE_BASE),
|
||||
key=lambda group: group[2],
|
||||
reverse=True,
|
||||
)
|
||||
def _tool_call_groups_by_message(
|
||||
groups: Mapping[tuple[str, int, int], tuple[str, ...]],
|
||||
) -> Mapping[tuple[str, int], tuple[tuple[str, int, int], ...]]:
|
||||
"""Tool-call groups indexed by ``(family, message index)``, each tuple highest tool index first.
|
||||
|
||||
Indexing once keeps the shed order linear in the group count: rescanning the
|
||||
full group map per message made attribute fitting quadratic on long prompts.
|
||||
"""
|
||||
ordered: Final = sorted(
|
||||
(group for group in groups if group[2] != _MESSAGE_BASE),
|
||||
key=lambda group: (group[0], group[1], group[2]),
|
||||
)
|
||||
yield from tool_call_groups
|
||||
return MappingProxyType(
|
||||
{
|
||||
message: tuple(reversed(tuple(message_tool_groups)))
|
||||
for message, message_tool_groups in groupby(ordered, key=lambda group: (group[0], group[1]))
|
||||
}
|
||||
)
|
||||
|
||||
|
||||
def _message_shed_groups(
|
||||
groups: Mapping[tuple[str, int, int], tuple[str, ...]],
|
||||
tool_call_groups: Mapping[tuple[str, int], tuple[tuple[str, int, int], ...]],
|
||||
family: str,
|
||||
message_idx: int,
|
||||
) -> Iterator[tuple[str, int, int]]:
|
||||
yield from tool_call_groups.get((family, message_idx), ())
|
||||
base_group: Final = (family, message_idx, _MESSAGE_BASE)
|
||||
if base_group in groups:
|
||||
yield base_group
|
||||
|
|
@ -117,6 +133,7 @@ def _message_shed_groups(
|
|||
|
||||
def _shed_order(groups: Mapping[tuple[str, int, int], tuple[str, ...]]) -> tuple[tuple[str, int, int], ...]:
|
||||
"""Middle inputs, extra choices, pinned inputs, then the first choice, with tool calls before message keys."""
|
||||
tool_call_groups: Final = _tool_call_groups_by_message(groups)
|
||||
inputs: Final = sorted(frozenset(idx for family, idx, _ in groups if family == _INPUT_MESSAGES))
|
||||
outputs: Final = sorted(frozenset(idx for family, idx, _ in groups if family == _OUTPUT_MESSAGES))
|
||||
pinned_inputs: Final = tuple(dict.fromkeys((*inputs[:1], *inputs[-1:])))
|
||||
|
|
@ -127,24 +144,35 @@ def _shed_order(groups: Mapping[tuple[str, int, int], tuple[str, ...]]) -> tuple
|
|||
*((_OUTPUT_MESSAGES, idx) for idx in outputs[:1]),
|
||||
)
|
||||
return tuple(
|
||||
chain.from_iterable(_message_shed_groups(groups, family, message_idx) for family, message_idx in message_order)
|
||||
chain.from_iterable(
|
||||
_message_shed_groups(groups, tool_call_groups, family, message_idx) for family, message_idx in message_order
|
||||
)
|
||||
)
|
||||
|
||||
|
||||
_METADATA_KEY: Final = "metadata"
|
||||
|
||||
|
||||
def fit_indexed_messages(attrs: Mapping[str, AttrValue], budget: int | None) -> Mapping[str, AttrValue]:
|
||||
"""``attrs`` with indexed message attributes shed, least valuable first, until at most ``budget`` keys remain.
|
||||
|
||||
``None`` means the span has no attribute count limit. Every message still rides the ``input.value`` and
|
||||
``output.value`` blobs, so shedding a per-index pair loses no content.
|
||||
``output.value`` blobs, so shedding a per-index pair loses no content. The ``metadata`` blob sheds only
|
||||
after every indexed message attribute: message attributes are the indexed, queryable view (the blobs
|
||||
carry no per-index keys), so a count-squeezed span keeps them and the single metadata key absorbs only
|
||||
the residual shortfall. Shedding happens here, before ``span.set_attribute``, so the fit is exact and
|
||||
the SDK's dropped-attributes counter never silently masks the choice.
|
||||
"""
|
||||
if budget is None or len(attrs) <= budget:
|
||||
return attrs
|
||||
groups: Final = _message_key_groups(attrs)
|
||||
order: Final = _shed_order(groups)
|
||||
running: Final = tuple(accumulate(len(groups[group]) for group in order))
|
||||
sheddable: Final = (*(groups[group] for group in _shed_order(groups)),)
|
||||
flex: Final = (_METADATA_KEY,) if _METADATA_KEY in attrs else ()
|
||||
candidates: Final = (*sheddable, flex) if flex else sheddable
|
||||
running: Final = tuple(accumulate(len(candidate) for candidate in candidates))
|
||||
excess: Final = len(attrs) - budget
|
||||
shed_count: Final = next((n + 1 for n, total in enumerate(running) if total >= excess), len(order))
|
||||
shed: Final = frozenset(chain.from_iterable(groups[group] for group in order[:shed_count]))
|
||||
shed_count: Final = next((n + 1 for n, total in enumerate(running) if total >= excess), len(candidates))
|
||||
shed: Final = frozenset(chain.from_iterable(candidates[:shed_count]))
|
||||
return MappingProxyType({key: value for key, value in attrs.items() if key not in shed})
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -159,7 +159,7 @@ def _message_tool_call(value: object) -> MessageToolCall | None:
|
|||
arguments: Final = (
|
||||
raw_arguments
|
||||
if isinstance(raw_arguments, str)
|
||||
else json.dumps(raw_arguments, default=str)
|
||||
else _stringify_tool_arguments(raw_arguments)
|
||||
if raw_arguments is not None
|
||||
else None
|
||||
)
|
||||
|
|
@ -171,3 +171,17 @@ def _message_tool_call(value: object) -> MessageToolCall | None:
|
|||
name=name if isinstance(name, str) else None,
|
||||
arguments=arguments,
|
||||
)
|
||||
|
||||
|
||||
def _stringify_tool_arguments(value: object) -> str:
|
||||
"""Serialize non-string tool-call arguments, falling back to ``repr``.
|
||||
|
||||
Arguments normally arrive as already-JSON strings, but provider adapters and
|
||||
``model_construct`` responses hand over raw Python objects. ``json.dumps``
|
||||
raises on those (tuple-keyed dicts, cycles) and the escaping exception would
|
||||
lose the whole span, so keep a readable ``repr`` instead.
|
||||
"""
|
||||
try:
|
||||
return json.dumps(value, default=str)
|
||||
except (TypeError, ValueError):
|
||||
return repr(value)
|
||||
|
|
|
|||
|
|
@ -20,7 +20,7 @@ from litellm.integrations.otel import ( # noqa: E402
|
|||
)
|
||||
from litellm.integrations.otel.plumbing import context as ctx_mod # noqa: E402
|
||||
from litellm.integrations.otel.plumbing import providers # noqa: E402
|
||||
from litellm.integrations.otel.emitter import SpanEmitter, span_attribute_limit # noqa: E402
|
||||
from litellm.integrations.otel.emitter import SpanEmitter, attribute_budget, span_attribute_limit # noqa: E402
|
||||
from litellm.integrations.otel.emitter import stamp_error # noqa: E402
|
||||
from litellm.integrations.otel.mappers.utils import MAX_TOOL_DEFINITION_ATTRS_PER_SPAN # noqa: E402
|
||||
from litellm.integrations.otel.model.payloads import ( # noqa: E402
|
||||
|
|
@ -558,6 +558,113 @@ def test_prompt_turns_are_shed_before_response_choices():
|
|||
)
|
||||
|
||||
|
||||
def test_metadata_blob_keeps_the_indexed_messages_it_competes_with():
|
||||
"""The promoted ``metadata`` blob rides alongside the indexed messages on a boundary-opened span.
|
||||
|
||||
The live regression shape: the LLM-call span is opened at the request
|
||||
boundary and already carries a few stamped attributes, most of which the
|
||||
mappers re-emit under the same keys. Charging those overwritten keys
|
||||
against the attribute budget reserved slots the fit could never spend, so
|
||||
the new ``metadata`` key displaced a whole middle message group — an
|
||||
eviction invisible to the SDK's dropped-attributes counter because the fit
|
||||
sheds before ``span.set_attribute``. With the budget counting only keys
|
||||
the fit does not overwrite, the metadata key rides and the span keeps the
|
||||
indexed-message count of the metadata-free span (at shedding granularity).
|
||||
"""
|
||||
cfg = OpenTelemetryV2Config(
|
||||
exporter="in_memory",
|
||||
legacy_compat=False,
|
||||
mapper_names=["genai", "openinference"],
|
||||
capture_message_content="span_only",
|
||||
)
|
||||
provider, exporter = providers.in_memory_provider(cfg)
|
||||
engine = SpanEmitter(providers.get_tracer(provider, "litellm-test"), cfg)
|
||||
|
||||
def boundary_span(payload):
|
||||
span = engine.start_span(SpanRole.LLM_CALL, "chat gpt-4o")
|
||||
# what the boundary opener stamps before the typed payload exists
|
||||
span.set_attribute(GenAI.REQUEST_MODEL, "gpt-4o")
|
||||
span.set_attribute(LiteLLM.PROVIDER_MODEL, "gpt-4o-2024")
|
||||
span.set_attribute("litellm.metadata.user_api_key_alias", "edge-key")
|
||||
engine.finish_span(
|
||||
SpanRole.LLM_CALL,
|
||||
span,
|
||||
LLMCallSpanData.from_standard_logging_payload(
|
||||
payload, capture_content=True, metadata_keys=("user_api_key_alias",)
|
||||
),
|
||||
)
|
||||
(finished,) = exporter.get_finished_spans()
|
||||
exporter.clear()
|
||||
return finished
|
||||
|
||||
with_metadata = boundary_span(_conversation_payload(47, metadata={"user_api_key_alias": "edge-key"}))
|
||||
without_metadata = boundary_span(_conversation_payload(47))
|
||||
_assert_core_intact(with_metadata)
|
||||
a = with_metadata.attributes
|
||||
|
||||
assert json.loads(a["metadata"]) == {"user_api_key_alias": "edge-key"}
|
||||
assert "metadata" not in without_metadata.attributes
|
||||
kept = _indexed_messages(a, "llm.input_messages")
|
||||
baseline = _indexed_messages(without_metadata.attributes, "llm.input_messages")
|
||||
assert kept == baseline or kept == baseline[:-1]
|
||||
assert kept[0] == 0 and kept[-1] == 46
|
||||
assert len(a) <= SpanLimits().max_span_attributes
|
||||
|
||||
|
||||
def test_preset_message_key_the_fit_sheds_still_never_overflows_the_span():
|
||||
"""A pre-set indexed-message key the fit later sheds cannot push the span over its limit.
|
||||
|
||||
The budget treats every mapped key already on the span as an overwrite
|
||||
(free). If the fit then sheds that key, the value stamped earlier simply
|
||||
stays in its slot, so the span holds one entry for it either way and the
|
||||
total never exceeds the limit — the SDK's dropped-attributes counter stays
|
||||
at zero.
|
||||
"""
|
||||
cfg = OpenTelemetryV2Config(
|
||||
exporter="in_memory",
|
||||
legacy_compat=False,
|
||||
mapper_names=["genai", "openinference"],
|
||||
capture_message_content="span_only",
|
||||
)
|
||||
provider, exporter = providers.in_memory_provider(cfg)
|
||||
engine = SpanEmitter(providers.get_tracer(provider, "litellm-test"), cfg)
|
||||
span = engine.start_span(SpanRole.LLM_CALL, "chat gpt-4o")
|
||||
span.set_attribute("llm.input_messages.1.message.role", "stale-role")
|
||||
engine.finish_span(
|
||||
SpanRole.LLM_CALL,
|
||||
span,
|
||||
LLMCallSpanData.from_standard_logging_payload(
|
||||
_conversation_payload(60), capture_content=True, metadata_keys=("user_api_key_alias",)
|
||||
),
|
||||
)
|
||||
(finished,) = exporter.get_finished_spans()
|
||||
_assert_core_intact(finished)
|
||||
assert len(finished.attributes) <= SpanLimits().max_span_attributes
|
||||
|
||||
|
||||
def test_attribute_budget_counts_only_keys_the_fit_does_not_overwrite():
|
||||
"""Pre-set attributes the mapped set overwrites consume no slot against the span limit.
|
||||
|
||||
A boundary-opened LLM-call span already carries a few stamped attributes,
|
||||
most of which the mappers re-emit under the same keys. Charging those
|
||||
against the budget reserves slots the fit can never spend and sheds
|
||||
indexed message attributes for nothing.
|
||||
"""
|
||||
cfg = OpenTelemetryV2Config(exporter="in_memory")
|
||||
provider, _exporter = providers.in_memory_provider(cfg)
|
||||
span = providers.get_tracer(provider, "litellm-test").start_span("s")
|
||||
span.set_attribute("gen_ai.request.model", "gpt-4o")
|
||||
span.set_attribute("litellm.metadata.user_api_key_alias", "edge-key")
|
||||
try:
|
||||
assert attribute_budget(span, 0) == SpanLimits().max_span_attributes - 2
|
||||
assert (
|
||||
attribute_budget(span, 0, frozenset({"gen_ai.request.model"}))
|
||||
== SpanLimits().max_span_attributes - 1
|
||||
)
|
||||
finally:
|
||||
span.end()
|
||||
|
||||
|
||||
def test_indexed_messages_respect_a_lower_span_attribute_count_limit(monkeypatch):
|
||||
"""The budget follows the SDK's configured limit, not a hardcoded default."""
|
||||
monkeypatch.setenv("OTEL_SPAN_ATTRIBUTE_COUNT_LIMIT", "48")
|
||||
|
|
|
|||
|
|
@ -6,6 +6,7 @@ backends, so one trace lights up every configured destination.
|
|||
"""
|
||||
|
||||
import json
|
||||
import time
|
||||
from collections.abc import Mapping
|
||||
from itertools import chain
|
||||
from typing import Final
|
||||
|
|
@ -347,6 +348,163 @@ def test_openinference_metadata_contains_only_promoted_metadata():
|
|||
assert "metadata" not in OpenInferenceMapper().map(_llm_call())
|
||||
|
||||
|
||||
def _long_prompt_messages(count: int) -> tuple[dict[str, str], ...]:
|
||||
return tuple({"role": "user" if i == 0 else "assistant", "content": f"turn {i}"} for i in range(count))
|
||||
|
||||
|
||||
def _input_message_keys(attrs: Mapping[str, object]) -> frozenset[str]:
|
||||
return frozenset(key for key in attrs if key.startswith("llm.input_messages."))
|
||||
|
||||
|
||||
def test_openinference_metadata_does_not_displace_indexed_messages_under_budget():
|
||||
"""The ``metadata`` blob rides in reclaimed slots instead of evicting messages.
|
||||
|
||||
``metadata`` competes for the OTel 128-attribute span budget, and shedding
|
||||
happens before ``span.set_attribute``, so the SDK's dropped-attributes
|
||||
counter stays at zero — the eviction is invisible. The fit pins the
|
||||
metadata key behind every message group: a squeezed span keeps the message
|
||||
attributes it would keep without the metadata blob (at whole-group
|
||||
granularity, at most one group of headroom difference), and the metadata
|
||||
key survives alongside them.
|
||||
"""
|
||||
messages: Final = _long_prompt_messages(62)
|
||||
|
||||
def fitted(promoted: Mapping[str, str], budget: int | None = None):
|
||||
mapped: Final = OpenInferenceMapper().map(_llm_call(messages_in=messages, promoted_metadata=promoted))
|
||||
assert len(mapped) > 128, "fixture must squeeze the span attribute budget"
|
||||
return fit_indexed_messages(mapped, budget if budget is not None else len(mapped) - 5)
|
||||
|
||||
without_metadata: Final = fitted({})
|
||||
with_metadata: Final = fitted({"user_api_key_alias": "edge-key"})
|
||||
assert "metadata" not in without_metadata
|
||||
assert json.loads(with_metadata["metadata"]) == {"user_api_key_alias": "edge-key"}
|
||||
assert _input_message_keys(with_metadata) == _input_message_keys(without_metadata)
|
||||
|
||||
# Against the absolute span limit the displacement is bounded by the
|
||||
# whole-group shedding granularity: at most one message group.
|
||||
without_at_limit: Final = fitted({}, 128)
|
||||
with_at_limit: Final = fitted({"user_api_key_alias": "edge-key"}, 128)
|
||||
assert len(_input_message_keys(with_at_limit)) >= len(_input_message_keys(without_at_limit)) - 2
|
||||
assert "metadata" in with_at_limit
|
||||
|
||||
|
||||
def test_openinference_metadata_sheds_only_after_every_indexed_message():
|
||||
"""Metadata sheds last: only once every indexed message attribute is gone."""
|
||||
mapped: Final = OpenInferenceMapper().map(
|
||||
_llm_call(messages_in=_long_prompt_messages(6), promoted_metadata={"user_api_key_alias": "edge-key"})
|
||||
)
|
||||
message_keys: Final = frozenset(key for key in mapped if ".message." in key)
|
||||
# Budget too small for the message family alone: everything indexed goes,
|
||||
# and the metadata blob absorbs the residual shortfall with it.
|
||||
starved: Final = fit_indexed_messages(mapped, len(mapped) - len(message_keys) - 1)
|
||||
assert not any(key in starved for key in message_keys)
|
||||
assert "metadata" not in starved
|
||||
# One slot more and metadata survives alongside zero indexed messages.
|
||||
last_standing: Final = fit_indexed_messages(mapped, len(mapped) - len(message_keys))
|
||||
assert not any(key in last_standing for key in message_keys)
|
||||
assert "metadata" in last_standing
|
||||
|
||||
|
||||
def test_openinference_raw_tool_arguments_fall_back_to_repr_instead_of_raising():
|
||||
"""Malformed Python tool arguments must not lose the span.
|
||||
|
||||
A plain ``Function()`` constructor JSON-serializes arguments, but provider
|
||||
adapters and ``model_construct`` responses hand over raw Python objects —
|
||||
tuple-keyed dicts and cycles that ``json.dumps`` raises on. The mapper
|
||||
serializes them with a ``repr`` fallback instead of letting the exception
|
||||
escape before the span is exported.
|
||||
"""
|
||||
# rebind-ok: a self-referencing dict cannot be built in one shot — the cycle
|
||||
# only exists once the finished dict is inserted into itself.
|
||||
circular: dict[str, object] = {}
|
||||
circular["self"] = circular
|
||||
for label, raw_arguments in (("tuple-key", {(1, 2): "v"}), ("circular", circular)):
|
||||
data: Final = _llm_call(
|
||||
choices_out=(
|
||||
{
|
||||
"finish_reason": "tool_calls",
|
||||
"message": {
|
||||
"role": "assistant",
|
||||
"content": None,
|
||||
"tool_calls": [
|
||||
{"id": "tc1", "type": "function", "function": {"name": "f", "arguments": raw_arguments}}
|
||||
],
|
||||
},
|
||||
},
|
||||
)
|
||||
)
|
||||
attrs: Final = OpenInferenceMapper().map(data)
|
||||
assert (
|
||||
attrs["llm.output_messages.0.message.tool_calls.0.tool_call.function.arguments"] == repr(raw_arguments)
|
||||
), label
|
||||
assert attrs["llm.output_messages.0.message.role"] == "assistant"
|
||||
|
||||
|
||||
def test_openinference_payload_tool_arguments_with_raw_objects_map_without_raising():
|
||||
"""The standard-logging payload path (provider adapter responses) survives raw argument objects too."""
|
||||
payload: Final = {
|
||||
"call_type": "acompletion",
|
||||
"custom_llm_provider": "openai",
|
||||
"model": "gpt-4o",
|
||||
"prompt_tokens": 3,
|
||||
"completion_tokens": 2,
|
||||
"total_tokens": 5,
|
||||
"stream": False,
|
||||
"model_parameters": {},
|
||||
"response": {
|
||||
"id": "resp_bad",
|
||||
"model": "gpt-4o-2024",
|
||||
"choices": [
|
||||
{
|
||||
"index": 0,
|
||||
"finish_reason": "tool_calls",
|
||||
"message": {
|
||||
"role": "assistant",
|
||||
"content": "hi",
|
||||
"tool_calls": [
|
||||
{"id": "tc1", "type": "function", "function": {"name": "f", "arguments": {(1, 2): "v"}}}
|
||||
],
|
||||
},
|
||||
}
|
||||
],
|
||||
},
|
||||
"metadata": {"user_api_key_alias": "edge-key"},
|
||||
"status": "success",
|
||||
"litellm_call_id": "call_raw_args",
|
||||
"hidden_params": {},
|
||||
}
|
||||
data: Final = LLMCallSpanData.from_standard_logging_payload(
|
||||
payload, capture_content=True, metadata_keys=("user_api_key_alias",)
|
||||
)
|
||||
attrs: Final = OpenInferenceMapper().map(data)
|
||||
arguments: Final = attrs["llm.output_messages.0.message.tool_calls.0.tool_call.function.arguments"]
|
||||
assert arguments == repr({(1, 2): "v"})
|
||||
assert json.loads(attrs["metadata"]) == {"user_api_key_alias": "edge-key"}
|
||||
|
||||
|
||||
def test_openinference_attribute_fit_stays_linear_on_long_prompts():
|
||||
"""Attribute fitting is linear in the message count, not quadratic.
|
||||
|
||||
Rescanning the full group map once per message made the fit quadratic in
|
||||
the prompt length (~1.9s of callback time at 8000 messages); indexing the
|
||||
groups once keeps it in milliseconds. The bound sits far above the linear
|
||||
runtime so it cannot flake on slow runners, while the quadratic path
|
||||
exceeds it several times over.
|
||||
"""
|
||||
messages: Final = _long_prompt_messages(4000)
|
||||
mapped: Final = OpenInferenceMapper().map(
|
||||
_llm_call(messages_in=messages, promoted_metadata={"user_api_key_alias": "k"})
|
||||
)
|
||||
started: Final = time.perf_counter()
|
||||
fitted: Final = fit_indexed_messages(mapped, 128)
|
||||
elapsed: Final = time.perf_counter() - started
|
||||
|
||||
assert elapsed < 0.25
|
||||
assert len(fitted) <= 128
|
||||
assert fitted["llm.input_messages.0.message.role"] == "user"
|
||||
assert fitted["llm.input_messages.3999.message.role"] == "assistant"
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Langfuse
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue