mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-05 02:41:56 +00:00
fix(otel): index the opener and the latest prompt turns, not the oldest
A value length limit clips the input.value blob, so the per-index keys are the only untruncated copy of a message. Indexing the leading prompt messages dropped the live user turn from every span attribute on long conversations. Keep message 0 and the most recent turns under the same span-wide budget, original indices preserved, reply reservation unchanged Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
parent
fcaf2d7d98
commit
a067557dae
2 changed files with 50 additions and 10 deletions
|
|
@ -92,8 +92,13 @@ class OpenInferenceMapper:
|
|||
return {
|
||||
**collect(self._LLM_CALL_ATTRS, data),
|
||||
**collect(self._BLOB_ATTRS, data),
|
||||
**self._messages("llm.input_messages", "input.value", data.messages_in, indexed_in),
|
||||
**self._messages("llm.output_messages", "output.value", outputs, indexed_out),
|
||||
**self._messages(
|
||||
"llm.input_messages",
|
||||
"input.value",
|
||||
data.messages_in,
|
||||
self._prompt_positions(len(data.messages_in), indexed_in),
|
||||
),
|
||||
**self._messages("llm.output_messages", "output.value", outputs, range(indexed_out)),
|
||||
**self._tools(data),
|
||||
}
|
||||
|
||||
|
|
@ -109,18 +114,26 @@ class OpenInferenceMapper:
|
|||
return _MAX_INDEXED_MESSAGES - indexed_out, indexed_out
|
||||
|
||||
@staticmethod
|
||||
def _messages(prefix: str, value_key: str, messages: Sequence[object], indexed: int) -> AttributeMap:
|
||||
"""``{prefix}.{idx}.message.*`` keys for the leading ``indexed`` messages + the ``value_key`` blob of all."""
|
||||
def _prompt_positions(total: int, indexed: int) -> tuple[int, ...]:
|
||||
"""Which prompt messages get per-index attributes: message 0 and the most recent turns.
|
||||
|
||||
A value length limit clips the ``input.value`` blob, so the system prompt and the
|
||||
live turn each keep a short key of their own. The middle of a long prompt does not.
|
||||
"""
|
||||
if total <= indexed:
|
||||
return tuple(range(total))
|
||||
return (0, *range(total - indexed + 1, total))
|
||||
|
||||
@staticmethod
|
||||
def _messages(prefix: str, value_key: str, messages: Sequence[object], positions: Sequence[int]) -> AttributeMap:
|
||||
"""``{prefix}.{idx}.message.*`` keys for the messages at ``positions`` + the ``value_key`` blob of all."""
|
||||
parsed: Final = [(m.get("role") if isinstance(m, dict) else None, message_content(m)) for m in messages]
|
||||
attrs: Final = drop_none(
|
||||
{
|
||||
key: value
|
||||
for idx, (role, content) in enumerate(parsed[:indexed])
|
||||
for idx, (role, content) in ((idx, parsed[idx]) for idx in positions)
|
||||
for key, value in (
|
||||
(
|
||||
f"{prefix}.{idx}.message.role",
|
||||
role if isinstance(role, str) else None,
|
||||
),
|
||||
(f"{prefix}.{idx}.message.role", role if isinstance(role, str) else None),
|
||||
(f"{prefix}.{idx}.message.content", content),
|
||||
)
|
||||
}
|
||||
|
|
|
|||
|
|
@ -505,7 +505,8 @@ def test_long_conversation_does_not_evict_core_attributes(turns):
|
|||
|
||||
assert a["llm.input_messages.0.message.content"] == "turn 0"
|
||||
assert a["llm.output_messages.0.message.content"] == "reply 0"
|
||||
assert f"llm.input_messages.{turns - 1}.message.role" not in a
|
||||
assert a[f"llm.input_messages.{turns - 1}.message.content"] == f"turn {turns - 1}"
|
||||
assert f"llm.input_messages.{turns // 2}.message.role" not in a
|
||||
assert len(json.loads(a["input.value"])) == turns
|
||||
assert len(json.loads(a["output.value"])) == 1
|
||||
assert len(json.loads(a[GenAI.INPUT_MESSAGES])) == turns
|
||||
|
|
@ -520,6 +521,31 @@ def test_short_conversation_keeps_every_message_indexed():
|
|||
assert a[f"llm.output_messages.{idx}.message.content"] == f"reply {idx}"
|
||||
|
||||
|
||||
def test_indexed_prompt_keeps_opener_and_latest_turns_under_a_value_length_limit(monkeypatch):
|
||||
"""The per-index keys are the only untruncated copy once the SDK clips string values.
|
||||
|
||||
Operators bound attribute sizes with ``OTEL_SPAN_ATTRIBUTE_VALUE_LENGTH_LIMIT``,
|
||||
which cuts the ``input.value`` blob short. The system prompt and the live turn
|
||||
then have to survive as their own short keys, whatever the conversation length.
|
||||
"""
|
||||
monkeypatch.setenv("OTEL_SPAN_ATTRIBUTE_VALUE_LENGTH_LIMIT", "256")
|
||||
payload = _conversation_payload(60)
|
||||
payload["messages"][0] = {"role": "system", "content": "be terse"}
|
||||
payload["messages"][-1] = {"role": "user", "content": "LATEST-TURN"}
|
||||
a = _conversation_span(["genai", "openinference"], payload).attributes
|
||||
|
||||
assert len(a["input.value"]) == 256
|
||||
assert a["llm.input_messages.0.message.role"] == "system"
|
||||
assert a["llm.input_messages.0.message.content"] == "be terse"
|
||||
assert a["llm.input_messages.59.message.role"] == "user"
|
||||
assert a["llm.input_messages.59.message.content"] == "LATEST-TURN"
|
||||
assert a["llm.output_messages.0.message.content"] == "reply 0"
|
||||
assert [int(key.split(".")[2]) for key in a if key.endswith("message.content") and key.startswith("llm.input_")] == [
|
||||
0,
|
||||
*range(54, 60),
|
||||
]
|
||||
|
||||
|
||||
def test_message_cap_is_shared_across_input_and_output():
|
||||
"""One span-wide allowance covers both directions, and the response always keeps a share.
|
||||
|
||||
|
|
@ -588,4 +614,5 @@ def test_fully_populated_span_with_every_vocabulary_stays_within_the_attribute_l
|
|||
assert a[f"{LiteLLM.COST_PREFIX}total"] == 0.002
|
||||
assert a[LiteLLM.TOOLS_DECLARED] == 127
|
||||
assert a["llm.input_messages.0.message.content"] == "turn 0"
|
||||
assert a["llm.input_messages.199.message.content"] == "turn 199"
|
||||
assert a["llm.output_messages.0.message.content"] == "reply 0"
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue