From b21e8ea009510c7cb8481f387e8878fc11ba2b0d Mon Sep 17 00:00:00 2001 From: yucheng-berri Date: Sun, 4 Oct 2026 03:09:38 -0700 Subject: [PATCH] fix(otel): repair Arize OTel v2 regressions from #43698 (linear fit, metadata slot, repr tool args) (#44488) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * fix(otel): keep Arize OTel v2 regressions in check — O(n) fit, metadata slot, repr tool args Three regressions from #43698's OpenInference tool-call/metadata emission: 1. Metadata evicted indexed message attributes: the new `metadata` key competed for the 128-attribute span budget, and the fit sheds whole message groups BEFORE `span.set_attribute`, so the SDK's dropped counter stayed 0 — an invisible eviction (live A/B: input-message attributes 86 -> 84). Two-part fix: the fit pins `metadata` behind every message group (it sheds only once all indexed messages are gone), and the span budget no longer charges pre-set attributes the mappers overwrite in place — a boundary-opened LLM span already carries keys like `gen_ai.request.model`, so the old accounting reserved slots the fit could never spend. Live: 86 input-message attributes with the metadata attribute riding alongside. 2. Quadratic shed on long prompts: `_message_shed_groups` rescanned the full group map once per message (measured on a real acompletion: 0.032/0.128/0.478/1.910s at 1000/2000/4000/8000 messages vs 0.007/0.010/0.022/0.029s at base). Index the tool-call groups once by (family, message index): the fit is linear again (0.008/0.007/0.014/ 0.031s, same rig). 3. Malformed Python tool arguments lost the whole span: provider adapters and `model_construct` responses hand over raw objects, and `json.dumps` raises on tuple-keyed dicts (TypeError) and cycles (ValueError) before the span is exported. Serialize with a repr fallback; both cases now export with a readable arguments attribute. The attribute budget change affects every boundary-opened LLM-call span (strictly more attributes retained, never fewer); the mapper changes only touch the OpenInference vocabulary. * refactor(otel): build the tool-call group index in one shot Review follow-up: the dict.setdefault/append seeding in _tool_call_groups_by_message violated the no-mutation coding convention (AGENTS.md: build values in one shot with comprehensions or generators wrapped in tuple()/MappingProxyType()). Rebuild it as a sorted groupby comprehension; randomized parity harness confirms the shed order is byte-identical to the seeded version (400 trials). Also pin the overflow corner Greptile asked about: a pre-set indexed-message key the fit sheds keeps its earlier value in place, so the span total can never exceed the SDK limit (new emitter test). * fix(otel): key the groupby with an explicit tuple to keep basedpyright at budget The slice-keyed groupby (group[:2]) widened the key to tuple[str | int], adding one reportGeneralTypeIssues over the codebase ceiling. Key by the explicit (family, message index) pair instead; shed order unchanged (300-trial randomized parity harness). * fix(otel): read pre-set span keys through a helper typed for both runtime shapes The SDK annotates ReadableSpan.attributes as a Mapping, but an ended span hands back a tuple of pairs, so the inline isinstance branch narrowed to Never and pushed reportGeneralTypeIssues one over the codebase ceiling. Extract _carried_keys with the runtime union declared on the parameter; behavior unchanged. --- litellm/integrations/otel/emitter.py | 29 +++- .../otel/mappers/openinference.py | 60 +++++-- litellm/integrations/otel/mappers/utils.py | 16 +- .../integrations/otel/test_otel_v2_emitter.py | 109 +++++++++++- .../otel/test_otel_v2_vendor_mappers.py | 158 ++++++++++++++++++ 5 files changed, 349 insertions(+), 23 deletions(-) diff --git a/litellm/integrations/otel/emitter.py b/litellm/integrations/otel/emitter.py index e9441ee2a9a..a4cb0a84541 100644 --- a/litellm/integrations/otel/emitter.py +++ b/litellm/integrations/otel/emitter.py @@ -91,13 +91,32 @@ def span_attribute_limit(span: Span) -> int | None: return span._limits.max_span_attributes # pyright: ignore[reportPrivateUsage] # SDK has no public getter -def attribute_budget(span: Span, reserved: int) -> int | None: - """How many mapped attributes fit on ``span`` next to what it already carries and ``reserved`` more.""" +def _carried_keys( + attributes: Mapping[str, AttrValue] | tuple[tuple[str, AttrValue], ...], +) -> frozenset[str]: + """The keys a span already carries: a live span exposes a ``Mapping``, an ended one a tuple of pairs.""" + if isinstance(attributes, Mapping): + return frozenset(attributes) + return frozenset(key for key, _value in attributes) + + +def attribute_budget(span: Span, reserved: int, overwrites: frozenset[str] = frozenset()) -> int | None: + """How many mapped attributes fit on ``span`` next to what it already carries and ``reserved`` more. + + ``overwrites`` are the mapped keys already present on ``span``: setting one + replaces the value in place and consumes no slot against the limit, so only + the genuinely new pre-existing keys reduce the budget. Counting the + overwritten ones too reserves slots the fit can never spend and sheds + indexed message attributes for nothing. + """ limit: Final = span_attribute_limit(span) if limit is None: return None - on_span: Final = len(span.attributes or ()) if isinstance(span, ReadableSpan) else 0 - return limit - on_span - reserved + if not isinstance(span, ReadableSpan): + return limit - reserved + existing_keys: Final = _carried_keys(span.attributes or ()) + fresh: Final = len(existing_keys - overwrites) if overwrites else len(existing_keys) + return limit - fresh - reserved def stamp_error( @@ -284,7 +303,7 @@ class SpanEmitter: ) stamped_later: Final = error_attributes(error) if error else _NO_ATTRIBUTES reserved: Final = len(stamped_later.keys() - mapped.keys()) - for key, value in fit_indexed_messages(mapped, attribute_budget(span, reserved)).items(): + for key, value in fit_indexed_messages(mapped, attribute_budget(span, reserved, frozenset(mapped))).items(): span.set_attribute(key, value) if error: stamped: Final = stamp_error(span, error) diff --git a/litellm/integrations/otel/mappers/openinference.py b/litellm/integrations/otel/mappers/openinference.py index 971c9a33664..8d7b4d7cf4a 100644 --- a/litellm/integrations/otel/mappers/openinference.py +++ b/litellm/integrations/otel/mappers/openinference.py @@ -99,17 +99,33 @@ def _message_key_groups(attrs: Mapping[str, AttrValue]) -> Mapping[tuple[str, in ) -def _message_shed_groups( - groups: Mapping[tuple[str, int, int], tuple[str, ...]], family: str, message_idx: int -) -> Iterator[tuple[str, int, int]]: - tool_call_groups: Final = tuple( - sorted( - (group for group in groups if group[:2] == (family, message_idx) and group[2] != _MESSAGE_BASE), - key=lambda group: group[2], - reverse=True, - ) +def _tool_call_groups_by_message( + groups: Mapping[tuple[str, int, int], tuple[str, ...]], +) -> Mapping[tuple[str, int], tuple[tuple[str, int, int], ...]]: + """Tool-call groups indexed by ``(family, message index)``, each tuple highest tool index first. + + Indexing once keeps the shed order linear in the group count: rescanning the + full group map per message made attribute fitting quadratic on long prompts. + """ + ordered: Final = sorted( + (group for group in groups if group[2] != _MESSAGE_BASE), + key=lambda group: (group[0], group[1], group[2]), ) - yield from tool_call_groups + return MappingProxyType( + { + message: tuple(reversed(tuple(message_tool_groups))) + for message, message_tool_groups in groupby(ordered, key=lambda group: (group[0], group[1])) + } + ) + + +def _message_shed_groups( + groups: Mapping[tuple[str, int, int], tuple[str, ...]], + tool_call_groups: Mapping[tuple[str, int], tuple[tuple[str, int, int], ...]], + family: str, + message_idx: int, +) -> Iterator[tuple[str, int, int]]: + yield from tool_call_groups.get((family, message_idx), ()) base_group: Final = (family, message_idx, _MESSAGE_BASE) if base_group in groups: yield base_group @@ -117,6 +133,7 @@ def _message_shed_groups( def _shed_order(groups: Mapping[tuple[str, int, int], tuple[str, ...]]) -> tuple[tuple[str, int, int], ...]: """Middle inputs, extra choices, pinned inputs, then the first choice, with tool calls before message keys.""" + tool_call_groups: Final = _tool_call_groups_by_message(groups) inputs: Final = sorted(frozenset(idx for family, idx, _ in groups if family == _INPUT_MESSAGES)) outputs: Final = sorted(frozenset(idx for family, idx, _ in groups if family == _OUTPUT_MESSAGES)) pinned_inputs: Final = tuple(dict.fromkeys((*inputs[:1], *inputs[-1:]))) @@ -127,24 +144,35 @@ def _shed_order(groups: Mapping[tuple[str, int, int], tuple[str, ...]]) -> tuple *((_OUTPUT_MESSAGES, idx) for idx in outputs[:1]), ) return tuple( - chain.from_iterable(_message_shed_groups(groups, family, message_idx) for family, message_idx in message_order) + chain.from_iterable( + _message_shed_groups(groups, tool_call_groups, family, message_idx) for family, message_idx in message_order + ) ) +_METADATA_KEY: Final = "metadata" + + def fit_indexed_messages(attrs: Mapping[str, AttrValue], budget: int | None) -> Mapping[str, AttrValue]: """``attrs`` with indexed message attributes shed, least valuable first, until at most ``budget`` keys remain. ``None`` means the span has no attribute count limit. Every message still rides the ``input.value`` and - ``output.value`` blobs, so shedding a per-index pair loses no content. + ``output.value`` blobs, so shedding a per-index pair loses no content. The ``metadata`` blob sheds only + after every indexed message attribute: message attributes are the indexed, queryable view (the blobs + carry no per-index keys), so a count-squeezed span keeps them and the single metadata key absorbs only + the residual shortfall. Shedding happens here, before ``span.set_attribute``, so the fit is exact and + the SDK's dropped-attributes counter never silently masks the choice. """ if budget is None or len(attrs) <= budget: return attrs groups: Final = _message_key_groups(attrs) - order: Final = _shed_order(groups) - running: Final = tuple(accumulate(len(groups[group]) for group in order)) + sheddable: Final = (*(groups[group] for group in _shed_order(groups)),) + flex: Final = (_METADATA_KEY,) if _METADATA_KEY in attrs else () + candidates: Final = (*sheddable, flex) if flex else sheddable + running: Final = tuple(accumulate(len(candidate) for candidate in candidates)) excess: Final = len(attrs) - budget - shed_count: Final = next((n + 1 for n, total in enumerate(running) if total >= excess), len(order)) - shed: Final = frozenset(chain.from_iterable(groups[group] for group in order[:shed_count])) + shed_count: Final = next((n + 1 for n, total in enumerate(running) if total >= excess), len(candidates)) + shed: Final = frozenset(chain.from_iterable(candidates[:shed_count])) return MappingProxyType({key: value for key, value in attrs.items() if key not in shed}) diff --git a/litellm/integrations/otel/mappers/utils.py b/litellm/integrations/otel/mappers/utils.py index 39ee56c2175..cfaf9d51a8c 100644 --- a/litellm/integrations/otel/mappers/utils.py +++ b/litellm/integrations/otel/mappers/utils.py @@ -159,7 +159,7 @@ def _message_tool_call(value: object) -> MessageToolCall | None: arguments: Final = ( raw_arguments if isinstance(raw_arguments, str) - else json.dumps(raw_arguments, default=str) + else _stringify_tool_arguments(raw_arguments) if raw_arguments is not None else None ) @@ -171,3 +171,17 @@ def _message_tool_call(value: object) -> MessageToolCall | None: name=name if isinstance(name, str) else None, arguments=arguments, ) + + +def _stringify_tool_arguments(value: object) -> str: + """Serialize non-string tool-call arguments, falling back to ``repr``. + + Arguments normally arrive as already-JSON strings, but provider adapters and + ``model_construct`` responses hand over raw Python objects. ``json.dumps`` + raises on those (tuple-keyed dicts, cycles) and the escaping exception would + lose the whole span, so keep a readable ``repr`` instead. + """ + try: + return json.dumps(value, default=str) + except (TypeError, ValueError): + return repr(value) diff --git a/tests/unit/integrations/otel/test_otel_v2_emitter.py b/tests/unit/integrations/otel/test_otel_v2_emitter.py index 11b2aa5fd67..949eb97bccd 100644 --- a/tests/unit/integrations/otel/test_otel_v2_emitter.py +++ b/tests/unit/integrations/otel/test_otel_v2_emitter.py @@ -20,7 +20,7 @@ from litellm.integrations.otel import ( # noqa: E402 ) from litellm.integrations.otel.plumbing import context as ctx_mod # noqa: E402 from litellm.integrations.otel.plumbing import providers # noqa: E402 -from litellm.integrations.otel.emitter import SpanEmitter, span_attribute_limit # noqa: E402 +from litellm.integrations.otel.emitter import SpanEmitter, attribute_budget, span_attribute_limit # noqa: E402 from litellm.integrations.otel.emitter import stamp_error # noqa: E402 from litellm.integrations.otel.mappers.utils import MAX_TOOL_DEFINITION_ATTRS_PER_SPAN # noqa: E402 from litellm.integrations.otel.model.payloads import ( # noqa: E402 @@ -558,6 +558,113 @@ def test_prompt_turns_are_shed_before_response_choices(): ) +def test_metadata_blob_keeps_the_indexed_messages_it_competes_with(): + """The promoted ``metadata`` blob rides alongside the indexed messages on a boundary-opened span. + + The live regression shape: the LLM-call span is opened at the request + boundary and already carries a few stamped attributes, most of which the + mappers re-emit under the same keys. Charging those overwritten keys + against the attribute budget reserved slots the fit could never spend, so + the new ``metadata`` key displaced a whole middle message group — an + eviction invisible to the SDK's dropped-attributes counter because the fit + sheds before ``span.set_attribute``. With the budget counting only keys + the fit does not overwrite, the metadata key rides and the span keeps the + indexed-message count of the metadata-free span (at shedding granularity). + """ + cfg = OpenTelemetryV2Config( + exporter="in_memory", + legacy_compat=False, + mapper_names=["genai", "openinference"], + capture_message_content="span_only", + ) + provider, exporter = providers.in_memory_provider(cfg) + engine = SpanEmitter(providers.get_tracer(provider, "litellm-test"), cfg) + + def boundary_span(payload): + span = engine.start_span(SpanRole.LLM_CALL, "chat gpt-4o") + # what the boundary opener stamps before the typed payload exists + span.set_attribute(GenAI.REQUEST_MODEL, "gpt-4o") + span.set_attribute(LiteLLM.PROVIDER_MODEL, "gpt-4o-2024") + span.set_attribute("litellm.metadata.user_api_key_alias", "edge-key") + engine.finish_span( + SpanRole.LLM_CALL, + span, + LLMCallSpanData.from_standard_logging_payload( + payload, capture_content=True, metadata_keys=("user_api_key_alias",) + ), + ) + (finished,) = exporter.get_finished_spans() + exporter.clear() + return finished + + with_metadata = boundary_span(_conversation_payload(47, metadata={"user_api_key_alias": "edge-key"})) + without_metadata = boundary_span(_conversation_payload(47)) + _assert_core_intact(with_metadata) + a = with_metadata.attributes + + assert json.loads(a["metadata"]) == {"user_api_key_alias": "edge-key"} + assert "metadata" not in without_metadata.attributes + kept = _indexed_messages(a, "llm.input_messages") + baseline = _indexed_messages(without_metadata.attributes, "llm.input_messages") + assert kept == baseline or kept == baseline[:-1] + assert kept[0] == 0 and kept[-1] == 46 + assert len(a) <= SpanLimits().max_span_attributes + + +def test_preset_message_key_the_fit_sheds_still_never_overflows_the_span(): + """A pre-set indexed-message key the fit later sheds cannot push the span over its limit. + + The budget treats every mapped key already on the span as an overwrite + (free). If the fit then sheds that key, the value stamped earlier simply + stays in its slot, so the span holds one entry for it either way and the + total never exceeds the limit — the SDK's dropped-attributes counter stays + at zero. + """ + cfg = OpenTelemetryV2Config( + exporter="in_memory", + legacy_compat=False, + mapper_names=["genai", "openinference"], + capture_message_content="span_only", + ) + provider, exporter = providers.in_memory_provider(cfg) + engine = SpanEmitter(providers.get_tracer(provider, "litellm-test"), cfg) + span = engine.start_span(SpanRole.LLM_CALL, "chat gpt-4o") + span.set_attribute("llm.input_messages.1.message.role", "stale-role") + engine.finish_span( + SpanRole.LLM_CALL, + span, + LLMCallSpanData.from_standard_logging_payload( + _conversation_payload(60), capture_content=True, metadata_keys=("user_api_key_alias",) + ), + ) + (finished,) = exporter.get_finished_spans() + _assert_core_intact(finished) + assert len(finished.attributes) <= SpanLimits().max_span_attributes + + +def test_attribute_budget_counts_only_keys_the_fit_does_not_overwrite(): + """Pre-set attributes the mapped set overwrites consume no slot against the span limit. + + A boundary-opened LLM-call span already carries a few stamped attributes, + most of which the mappers re-emit under the same keys. Charging those + against the budget reserves slots the fit can never spend and sheds + indexed message attributes for nothing. + """ + cfg = OpenTelemetryV2Config(exporter="in_memory") + provider, _exporter = providers.in_memory_provider(cfg) + span = providers.get_tracer(provider, "litellm-test").start_span("s") + span.set_attribute("gen_ai.request.model", "gpt-4o") + span.set_attribute("litellm.metadata.user_api_key_alias", "edge-key") + try: + assert attribute_budget(span, 0) == SpanLimits().max_span_attributes - 2 + assert ( + attribute_budget(span, 0, frozenset({"gen_ai.request.model"})) + == SpanLimits().max_span_attributes - 1 + ) + finally: + span.end() + + def test_indexed_messages_respect_a_lower_span_attribute_count_limit(monkeypatch): """The budget follows the SDK's configured limit, not a hardcoded default.""" monkeypatch.setenv("OTEL_SPAN_ATTRIBUTE_COUNT_LIMIT", "48") diff --git a/tests/unit/integrations/otel/test_otel_v2_vendor_mappers.py b/tests/unit/integrations/otel/test_otel_v2_vendor_mappers.py index 5169e6b9d5f..efc43b16305 100644 --- a/tests/unit/integrations/otel/test_otel_v2_vendor_mappers.py +++ b/tests/unit/integrations/otel/test_otel_v2_vendor_mappers.py @@ -6,6 +6,7 @@ backends, so one trace lights up every configured destination. """ import json +import time from collections.abc import Mapping from itertools import chain from typing import Final @@ -347,6 +348,163 @@ def test_openinference_metadata_contains_only_promoted_metadata(): assert "metadata" not in OpenInferenceMapper().map(_llm_call()) +def _long_prompt_messages(count: int) -> tuple[dict[str, str], ...]: + return tuple({"role": "user" if i == 0 else "assistant", "content": f"turn {i}"} for i in range(count)) + + +def _input_message_keys(attrs: Mapping[str, object]) -> frozenset[str]: + return frozenset(key for key in attrs if key.startswith("llm.input_messages.")) + + +def test_openinference_metadata_does_not_displace_indexed_messages_under_budget(): + """The ``metadata`` blob rides in reclaimed slots instead of evicting messages. + + ``metadata`` competes for the OTel 128-attribute span budget, and shedding + happens before ``span.set_attribute``, so the SDK's dropped-attributes + counter stays at zero — the eviction is invisible. The fit pins the + metadata key behind every message group: a squeezed span keeps the message + attributes it would keep without the metadata blob (at whole-group + granularity, at most one group of headroom difference), and the metadata + key survives alongside them. + """ + messages: Final = _long_prompt_messages(62) + + def fitted(promoted: Mapping[str, str], budget: int | None = None): + mapped: Final = OpenInferenceMapper().map(_llm_call(messages_in=messages, promoted_metadata=promoted)) + assert len(mapped) > 128, "fixture must squeeze the span attribute budget" + return fit_indexed_messages(mapped, budget if budget is not None else len(mapped) - 5) + + without_metadata: Final = fitted({}) + with_metadata: Final = fitted({"user_api_key_alias": "edge-key"}) + assert "metadata" not in without_metadata + assert json.loads(with_metadata["metadata"]) == {"user_api_key_alias": "edge-key"} + assert _input_message_keys(with_metadata) == _input_message_keys(without_metadata) + + # Against the absolute span limit the displacement is bounded by the + # whole-group shedding granularity: at most one message group. + without_at_limit: Final = fitted({}, 128) + with_at_limit: Final = fitted({"user_api_key_alias": "edge-key"}, 128) + assert len(_input_message_keys(with_at_limit)) >= len(_input_message_keys(without_at_limit)) - 2 + assert "metadata" in with_at_limit + + +def test_openinference_metadata_sheds_only_after_every_indexed_message(): + """Metadata sheds last: only once every indexed message attribute is gone.""" + mapped: Final = OpenInferenceMapper().map( + _llm_call(messages_in=_long_prompt_messages(6), promoted_metadata={"user_api_key_alias": "edge-key"}) + ) + message_keys: Final = frozenset(key for key in mapped if ".message." in key) + # Budget too small for the message family alone: everything indexed goes, + # and the metadata blob absorbs the residual shortfall with it. + starved: Final = fit_indexed_messages(mapped, len(mapped) - len(message_keys) - 1) + assert not any(key in starved for key in message_keys) + assert "metadata" not in starved + # One slot more and metadata survives alongside zero indexed messages. + last_standing: Final = fit_indexed_messages(mapped, len(mapped) - len(message_keys)) + assert not any(key in last_standing for key in message_keys) + assert "metadata" in last_standing + + +def test_openinference_raw_tool_arguments_fall_back_to_repr_instead_of_raising(): + """Malformed Python tool arguments must not lose the span. + + A plain ``Function()`` constructor JSON-serializes arguments, but provider + adapters and ``model_construct`` responses hand over raw Python objects — + tuple-keyed dicts and cycles that ``json.dumps`` raises on. The mapper + serializes them with a ``repr`` fallback instead of letting the exception + escape before the span is exported. + """ + # rebind-ok: a self-referencing dict cannot be built in one shot — the cycle + # only exists once the finished dict is inserted into itself. + circular: dict[str, object] = {} + circular["self"] = circular + for label, raw_arguments in (("tuple-key", {(1, 2): "v"}), ("circular", circular)): + data: Final = _llm_call( + choices_out=( + { + "finish_reason": "tool_calls", + "message": { + "role": "assistant", + "content": None, + "tool_calls": [ + {"id": "tc1", "type": "function", "function": {"name": "f", "arguments": raw_arguments}} + ], + }, + }, + ) + ) + attrs: Final = OpenInferenceMapper().map(data) + assert ( + attrs["llm.output_messages.0.message.tool_calls.0.tool_call.function.arguments"] == repr(raw_arguments) + ), label + assert attrs["llm.output_messages.0.message.role"] == "assistant" + + +def test_openinference_payload_tool_arguments_with_raw_objects_map_without_raising(): + """The standard-logging payload path (provider adapter responses) survives raw argument objects too.""" + payload: Final = { + "call_type": "acompletion", + "custom_llm_provider": "openai", + "model": "gpt-4o", + "prompt_tokens": 3, + "completion_tokens": 2, + "total_tokens": 5, + "stream": False, + "model_parameters": {}, + "response": { + "id": "resp_bad", + "model": "gpt-4o-2024", + "choices": [ + { + "index": 0, + "finish_reason": "tool_calls", + "message": { + "role": "assistant", + "content": "hi", + "tool_calls": [ + {"id": "tc1", "type": "function", "function": {"name": "f", "arguments": {(1, 2): "v"}}} + ], + }, + } + ], + }, + "metadata": {"user_api_key_alias": "edge-key"}, + "status": "success", + "litellm_call_id": "call_raw_args", + "hidden_params": {}, + } + data: Final = LLMCallSpanData.from_standard_logging_payload( + payload, capture_content=True, metadata_keys=("user_api_key_alias",) + ) + attrs: Final = OpenInferenceMapper().map(data) + arguments: Final = attrs["llm.output_messages.0.message.tool_calls.0.tool_call.function.arguments"] + assert arguments == repr({(1, 2): "v"}) + assert json.loads(attrs["metadata"]) == {"user_api_key_alias": "edge-key"} + + +def test_openinference_attribute_fit_stays_linear_on_long_prompts(): + """Attribute fitting is linear in the message count, not quadratic. + + Rescanning the full group map once per message made the fit quadratic in + the prompt length (~1.9s of callback time at 8000 messages); indexing the + groups once keeps it in milliseconds. The bound sits far above the linear + runtime so it cannot flake on slow runners, while the quadratic path + exceeds it several times over. + """ + messages: Final = _long_prompt_messages(4000) + mapped: Final = OpenInferenceMapper().map( + _llm_call(messages_in=messages, promoted_metadata={"user_api_key_alias": "k"}) + ) + started: Final = time.perf_counter() + fitted: Final = fit_indexed_messages(mapped, 128) + elapsed: Final = time.perf_counter() - started + + assert elapsed < 0.25 + assert len(fitted) <= 128 + assert fitted["llm.input_messages.0.message.role"] == "user" + assert fitted["llm.input_messages.3999.message.role"] == "assistant" + + # --------------------------------------------------------------------------- # # Langfuse # --------------------------------------------------------------------------- #