fix(langfuse_otel): map the proxy end user to Langfuse user.id, never session.id

Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
yucheng 2026-09-23 07:14:29 +00:00
parent eac73c8278
commit aa1d732210
8 changed files with 385 additions and 290 deletions

View file

@ -410,7 +410,14 @@ def _set_tool_attributes(span: "Span", optional_tools: list | None, metadata_too
)
def set_attributes(span: "Span", kwargs, response_obj, attributes: type[BaseLLMObsOTELAttributes]):
def set_attributes(
span: "Span",
kwargs,
response_obj,
attributes: type[BaseLLMObsOTELAttributes],
*,
emit_session_and_user: bool = True,
):
"""
Populates span with OpenInference-compliant LLM attributes for Arize and Phoenix tracing.
"""
@ -470,8 +477,12 @@ def set_attributes(span: "Span", kwargs, response_obj, attributes: type[BaseLLMO
# Additive emitters. Each is independently guarded so a failure can never
# blank the attributes set by the main try-block above. New attributes are
# written under new keys; existing attributes are not overwritten.
slp: Final = kwargs.get("standard_logging_object")
_safe_emit("session/user attrs", _set_session_and_user_attrs, span, kwargs, slp)
from litellm.integrations.otel.model.utils import as_str_mapping
slp: Final = as_str_mapping(kwargs.get("standard_logging_object"))
if emit_session_and_user:
_safe_emit("session/user attrs", _set_session_and_user_attrs, span, kwargs, slp)
_safe_emit("request context attrs", _set_request_context_attrs, span, slp)
_safe_emit("response cost", _set_response_cost_attr, span, slp)
_safe_emit(
"passthrough normalization",
@ -834,14 +845,12 @@ def _emit_input_message_extras(span: "Span", prefix: str, message: dict) -> None
def _set_session_and_user_attrs(span: "Span", kwargs: dict, standard_logging_payload) -> None:
"""Emit `SESSION_ID` / `USER_ID` / team metadata when source data exists.
"""Emit `SESSION_ID` / `USER_ID` when source data exists.
`SESSION_ID` is emitted only when an explicit end-user identifier exists
(`metadata.user_api_key_end_user_id`). We deliberately do NOT fall back
to `trace_id`, because that would create a distinct "session" for every
single request and distort Arize's Session-grouping analytics. The
`trace_id` is still emitted under its own `litellm.trace_id` key so
spans remain filterable by trace.
single request and distort Arize's Session-grouping analytics.
USER_ID is *only* emitted when no upstream path (model_params.user or
optional_params.user) has already set it, to avoid overwriting an
@ -857,10 +866,6 @@ def _set_session_and_user_attrs(span: "Span", kwargs: dict, standard_logging_pay
if session_id:
safe_set_attribute(span, SpanAttributes.SESSION_ID, str(session_id))
trace_id: Final = standard_logging_payload.get("trace_id")
if trace_id:
safe_set_attribute(span, "litellm.trace_id", str(trace_id))
optional_params: Final = kwargs.get("optional_params") or {}
model_params: Final = standard_logging_payload.get("model_parameters") or {}
has_user_already: Final = bool(
@ -872,6 +877,22 @@ def _set_session_and_user_attrs(span: "Span", kwargs: dict, standard_logging_pay
if user_id:
safe_set_attribute(span, SpanAttributes.USER_ID, str(user_id))
def _set_request_context_attrs(span: "Span", standard_logging_payload: object) -> None:
"""Emit `litellm.trace_id` / team / key context when source data exists."""
from litellm.integrations.otel.model.utils import as_str_mapping
payload: Final = as_str_mapping(standard_logging_payload)
if payload is None:
return
trace_id: Final = payload.get("trace_id")
if trace_id:
safe_set_attribute(span, "litellm.trace_id", str(trace_id))
metadata: Final = as_str_mapping(payload.get("metadata"))
if metadata is None:
return
team_id: Final = metadata.get("user_api_key_team_id")
if team_id:
safe_set_attribute(span, "litellm.team_id", str(team_id))

View file

@ -39,19 +39,20 @@ class LangfuseOtelLogger(OpenTelemetry):
super().__init__(config=config, *args, **kwargs)
@staticmethod
def set_langfuse_otel_attributes(span: Span, kwargs, response_obj):
def set_langfuse_otel_attributes(span: Span, kwargs: dict[str, object], response_obj) -> None:
"""
Sets OpenTelemetry span attributes for Langfuse observability.
Uses the same attribute setting logic as Arize Phoenix for consistency.
"""
_utils.set_attributes(span, kwargs, response_obj, LangfuseLLMObsOTELAttributes)
_utils.set_attributes(span, kwargs, response_obj, LangfuseLLMObsOTELAttributes, emit_session_and_user=False)
span.set_attribute("langfuse.observation.type", "generation")
#########################################################
# Set Langfuse specific attributes
#########################################################
LangfuseOtelLogger._set_langfuse_specific_attributes(span=span, kwargs=kwargs, response_obj=response_obj)
LangfuseOtelLogger._set_trace_user_attribute(span=span, kwargs=kwargs)
@staticmethod
def _extract_langfuse_metadata(kwargs: dict) -> dict:
@ -255,6 +256,28 @@ class LangfuseOtelLogger(OpenTelemetry):
LangfuseOtelLogger._set_observation_output(span=span, response_obj=response_obj)
@staticmethod
def _set_trace_user_attribute(span: Span, kwargs: dict[str, object]) -> None:
from litellm.integrations.arize._utils import safe_set_attribute
from litellm.integrations.otel.model.utils import as_str, as_str_mapping
slp: Final = as_str_mapping(kwargs.get("standard_logging_object"))
slp_metadata: Final = as_str_mapping(slp.get("metadata")) if slp is not None else None
if slp is None or slp_metadata is None:
return
metadata: Final = as_str_mapping(
LangfuseOtelLogger._extract_langfuse_metadata(kwargs) # pyright: ignore[reportUnknownArgumentType,reportUnknownMemberType] # helper returns a loosely typed dict
)
if metadata is None or as_str(metadata.get("trace_user_id")):
return
end_user: Final = (
slp_metadata.get("user_api_key_end_user_id")
or slp.get("end_user")
or as_str(metadata.get("user_api_key_end_user_id"))
)
if end_user:
safe_set_attribute(span, LangfuseSpanAttributes.TRACE_USER_ID.value, str(end_user))
@staticmethod
def _get_langfuse_otel_host() -> str | None:
"""

View file

@ -9,7 +9,7 @@ from litellm.integrations.otel.mappers.langfuse import (
LangfuseMapper,
)
from litellm.integrations.otel.model.request_io import request_input, response_output, stream_output
from litellm.integrations.otel.model.trace_controls import caller_trace_controls
from litellm.integrations.otel.model.trace_controls import langfuse_trace_controls
from litellm.integrations.otel.plumbing.context import request_root_span
if TYPE_CHECKING:
@ -18,13 +18,14 @@ if TYPE_CHECKING:
class LangfuseOpenTelemetryV2(OpenTelemetryV2):
"""Stamps the caller's trace controls (name, user, session, tags) on the request. Langfuse reads them off
the root observation, and the proxy's root span is still recording when the LLM call starts."""
"""Stamps the caller's trace controls (name, user, session, tags) on the request, falling back to the
proxy's end user when the caller names no trace user. Langfuse reads them off the root observation,
and the proxy's root span is still recording when the LLM call starts."""
def log_pre_api_call(self, model: str, messages: object, kwargs: Mapping[str, object]) -> None:
root: Final = request_root_span()
if root is not None and root.is_recording():
root.set_attributes(LangfuseMapper.trace_attributes(caller_trace_controls(kwargs)))
root.set_attributes(LangfuseMapper.trace_attributes(langfuse_trace_controls(kwargs)))
super().log_pre_api_call(model, messages, kwargs)

View file

@ -3,7 +3,7 @@
from __future__ import annotations
from collections.abc import Mapping
from dataclasses import dataclass
from dataclasses import dataclass, replace
from typing import Final
from pydantic import TypeAdapter, ValidationError
@ -33,11 +33,7 @@ def caller_trace_controls(kwargs: Mapping[str, object]) -> TraceControls:
return TraceControls()
proxy_request: Final = as_str_mapping(request.get("proxy_server_request"))
headers: Final = as_str_mapping(proxy_request.get("headers")) if proxy_request is not None else None
bodies: Final = tuple(
metadata
for key in ("metadata", "litellm_metadata")
if (metadata := as_str_mapping(request.get(key))) is not None
)
bodies: Final = _metadata_bodies(request)
def scalar(control: str) -> str | None:
from_header: Final = as_str(headers.get(f"{LANGFUSE_HEADER_PREFIX}{control}")) if headers is not None else None
@ -53,6 +49,28 @@ def caller_trace_controls(kwargs: Mapping[str, object]) -> TraceControls:
)
def langfuse_trace_controls(kwargs: Mapping[str, object]) -> TraceControls:
controls: Final = caller_trace_controls(kwargs)
if controls.user_id:
return controls
request: Final = as_str_mapping(kwargs.get("litellm_params"))
if request is None:
return controls
end_user: Final = next(
(value for body in _metadata_bodies(request) if (value := as_str(body.get("user_api_key_end_user_id")))),
None,
)
return replace(controls, user_id=end_user)
def _metadata_bodies(request: Mapping[str, object]) -> tuple[Mapping[str, object], ...]:
return tuple(
metadata
for key in ("metadata", "litellm_metadata")
if (metadata := as_str_mapping(request.get(key))) is not None
)
def _str_items(value: object) -> tuple[str, ...]:
try:
items: Final = _ITEMS.validate_python(value)

View file

@ -1,9 +1,6 @@
import json
from typing import Optional
# Adds the grandparent directory to sys.path to allow importing project modules
import asyncio
import json
import pytest
@ -70,9 +67,7 @@ def test_arize_set_attributes():
# Simulated LLM response object
response_obj = ModelResponse(
usage={"total_tokens": 100, "completion_tokens": 60, "prompt_tokens": 40},
choices=[
Choices(message={"role": "assistant", "content": "Basic Response Content"})
],
choices=[Choices(message={"role": "assistant", "content": "Basic Response Content"})],
model="gpt-4o",
id="chatcmpl-ID",
)
@ -89,9 +84,7 @@ def test_arize_set_attributes():
assert span.set_attribute.call_count == 26
# Metadata attached to the span
span.set_attribute.assert_any_call(
SpanAttributes.METADATA, json.dumps({"key_1": "value_1", "key_2": None})
)
span.set_attribute.assert_any_call(SpanAttributes.METADATA, json.dumps({"key_1": "value_1", "key_2": None}))
# Basic LLM information
span.set_attribute.assert_any_call(SpanAttributes.LLM_MODEL_NAME, "gpt-4o")
@ -114,16 +107,12 @@ def test_arize_set_attributes():
span.set_attribute.assert_any_call(SpanAttributes.OPENINFERENCE_SPAN_KIND, "LLM")
# And TOOL must never be written for an LLM chat completion call.
span_kind_writes = [
c.args[1]
for c in span.set_attribute.call_args_list
if c.args[0] == SpanAttributes.OPENINFERENCE_SPAN_KIND
c.args[1] for c in span.set_attribute.call_args_list if c.args[0] == SpanAttributes.OPENINFERENCE_SPAN_KIND
]
assert "TOOL" not in span_kind_writes
# Request message content and metadata
span.set_attribute.assert_any_call(
SpanAttributes.INPUT_VALUE, "Basic Request Content"
)
span.set_attribute.assert_any_call(SpanAttributes.INPUT_VALUE, "Basic Request Content")
span.set_attribute.assert_any_call(
f"{SpanAttributes.LLM_INPUT_MESSAGES}.0.{MessageAttributes.MESSAGE_ROLE}",
"user",
@ -134,9 +123,7 @@ def test_arize_set_attributes():
)
# Tool call definitions and function names
span.set_attribute.assert_any_call(
f"{SpanAttributes.LLM_TOOLS}.0.name", "get_weather"
)
span.set_attribute.assert_any_call(f"{SpanAttributes.LLM_TOOLS}.0.name", "get_weather")
span.set_attribute.assert_any_call(
f"{SpanAttributes.LLM_TOOLS}.0.description",
"Fetches weather details.",
@ -146,26 +133,20 @@ def test_arize_set_attributes():
json.dumps(
{
"type": "object",
"properties": {
"location": {"type": "string", "description": "City name"}
},
"properties": {"location": {"type": "string", "description": "City name"}},
"required": ["location"],
}
),
)
# Invocation parameters
span.set_attribute.assert_any_call(
SpanAttributes.LLM_INVOCATION_PARAMETERS, '{"user": "test_user"}'
)
span.set_attribute.assert_any_call(SpanAttributes.LLM_INVOCATION_PARAMETERS, '{"user": "test_user"}')
# User ID
span.set_attribute.assert_any_call(SpanAttributes.USER_ID, "test_user")
# Output message content
span.set_attribute.assert_any_call(
SpanAttributes.OUTPUT_VALUE, "Basic Response Content"
)
span.set_attribute.assert_any_call(SpanAttributes.OUTPUT_VALUE, "Basic Response Content")
span.set_attribute.assert_any_call(
f"{SpanAttributes.LLM_OUTPUT_MESSAGES}.0.{MessageAttributes.MESSAGE_ROLE}",
"assistant",
@ -187,18 +168,20 @@ def test_arize_set_attributes_responses_api():
Verifies that multiple output types are correctly handled.
"""
from unittest.mock import MagicMock
from litellm.types.llms.openai import (
ResponsesAPIResponse,
ResponseAPIUsage,
OutputTokensDetails,
)
from openai.types.responses import (
ResponseReasoningItem,
ResponseOutputMessage,
ResponseOutputText,
ResponseReasoningItem,
)
from openai.types.responses.response_reasoning_item import Summary
from litellm.types.llms.openai import (
OutputTokensDetails,
ResponseAPIUsage,
ResponsesAPIResponse,
)
span = MagicMock() # Mocked tracing span to test attribute setting
# Construct kwargs to simulate a real LLM request scenario
@ -228,9 +211,7 @@ def test_arize_set_attributes_responses_api():
ResponseReasoningItem(
id="reasoning-001",
type="reasoning",
summary=[
Summary(text="First, I need to analyze...", type="summary_text")
],
summary=[Summary(text="First, I need to analyze...", type="summary_text")],
),
ResponseOutputMessage(
id="msg-001",
@ -277,9 +258,7 @@ def test_arize_set_attributes_responses_api():
span.set_attribute.assert_any_call(SpanAttributes.LLM_TOKEN_COUNT_TOTAL, 370)
span.set_attribute.assert_any_call(SpanAttributes.LLM_TOKEN_COUNT_COMPLETION, 250)
span.set_attribute.assert_any_call(SpanAttributes.LLM_TOKEN_COUNT_PROMPT, 120)
span.set_attribute.assert_any_call(
SpanAttributes.LLM_TOKEN_COUNT_COMPLETION_DETAILS_REASONING, 180
)
span.set_attribute.assert_any_call(SpanAttributes.LLM_TOKEN_COUNT_COMPLETION_DETAILS_REASONING, 180)
def test_set_usage_outputs_pydantic_completion_usage():
@ -327,9 +306,7 @@ def test_set_usage_outputs_pydantic_completion_usage():
span.set_attribute.assert_any_call(SpanAttributes.LLM_TOKEN_COUNT_PROMPT, 40)
span.set_attribute.assert_any_call(SpanAttributes.LLM_TOKEN_COUNT_COMPLETION, 60)
# reasoning_tokens for chat completions live in completion_tokens_details
span.set_attribute.assert_any_call(
SpanAttributes.LLM_TOKEN_COUNT_COMPLETION_DETAILS_REASONING, 25
)
span.set_attribute.assert_any_call(SpanAttributes.LLM_TOKEN_COUNT_COMPLETION_DETAILS_REASONING, 25)
def test_set_usage_outputs_pydantic_response_api_usage():
@ -362,9 +339,7 @@ def test_set_usage_outputs_pydantic_response_api_usage():
span.set_attribute.assert_any_call(SpanAttributes.LLM_TOKEN_COUNT_TOTAL, 370)
span.set_attribute.assert_any_call(SpanAttributes.LLM_TOKEN_COUNT_PROMPT, 120)
span.set_attribute.assert_any_call(SpanAttributes.LLM_TOKEN_COUNT_COMPLETION, 250)
span.set_attribute.assert_any_call(
SpanAttributes.LLM_TOKEN_COUNT_COMPLETION_DETAILS_REASONING, 180
)
span.set_attribute.assert_any_call(SpanAttributes.LLM_TOKEN_COUNT_COMPLETION_DETAILS_REASONING, 180)
class TestArizeLogger(CustomLogger):
@ -375,16 +350,12 @@ class TestArizeLogger(CustomLogger):
def __init__(self, *args, **kwargs):
super().__init__(*args, **kwargs)
self.standard_callback_dynamic_params: Optional[
StandardCallbackDynamicParams
] = None
self.standard_callback_dynamic_params: StandardCallbackDynamicParams | None = None
async def async_log_success_event(self, kwargs, response_obj, start_time, end_time):
# Capture dynamic params and print them for verification
print("logged kwargs", json.dumps(kwargs, indent=4, default=str))
self.standard_callback_dynamic_params = kwargs.get(
"standard_callback_dynamic_params"
)
self.standard_callback_dynamic_params = kwargs.get("standard_callback_dynamic_params")
@pytest.mark.asyncio
@ -410,14 +381,8 @@ async def test_arize_dynamic_params():
# Assert dynamic parameters were received in the callback
assert test_arize_logger.standard_callback_dynamic_params is not None
assert (
test_arize_logger.standard_callback_dynamic_params.get("arize_api_key")
== "test_api_key_dynamic"
)
assert (
test_arize_logger.standard_callback_dynamic_params.get("arize_space_key")
== "test_space_key_dynamic"
)
assert test_arize_logger.standard_callback_dynamic_params.get("arize_api_key") == "test_api_key_dynamic"
assert test_arize_logger.standard_callback_dynamic_params.get("arize_space_key") == "test_space_key_dynamic"
def test_construct_dynamic_arize_headers():
@ -428,9 +393,7 @@ def test_construct_dynamic_arize_headers():
from litellm.types.utils import StandardCallbackDynamicParams
# Test with all parameters present
dynamic_params_full = StandardCallbackDynamicParams(
arize_api_key="test_api_key", arize_space_id="test_space_id"
)
dynamic_params_full = StandardCallbackDynamicParams(arize_api_key="test_api_key", arize_space_id="test_space_id")
arize_logger = ArizeLogger()
headers = arize_logger.construct_dynamic_otel_headers(dynamic_params_full)
@ -438,9 +401,7 @@ def test_construct_dynamic_arize_headers():
assert headers == expected_headers
# Test with only space_id
dynamic_params_space_id_only = StandardCallbackDynamicParams(
arize_space_id="test_space_id"
)
dynamic_params_space_id_only = StandardCallbackDynamicParams(arize_space_id="test_space_id")
headers = arize_logger.construct_dynamic_otel_headers(dynamic_params_space_id_only)
expected_headers = {"arize-space-id": "test_space_id"}
@ -456,9 +417,7 @@ def test_construct_dynamic_arize_headers():
dynamic_params_space_key_and_api_key = StandardCallbackDynamicParams(
arize_space_key="test_space_key", arize_api_key="test_api_key"
)
headers = arize_logger.construct_dynamic_otel_headers(
dynamic_params_space_key_and_api_key
)
headers = arize_logger.construct_dynamic_otel_headers(dynamic_params_space_key_and_api_key)
expected_headers = {"arize-space-id": "test_space_key", "api_key": "test_api_key"}
@ -528,9 +487,7 @@ def test_arize_emits_no_cache_tokens_when_absent():
from litellm.integrations.arize._utils import _set_usage_outputs
span = MagicMock()
response_obj = {
"usage": {"total_tokens": 10, "completion_tokens": 4, "prompt_tokens": 6}
}
response_obj = {"usage": {"total_tokens": 10, "completion_tokens": 4, "prompt_tokens": 6}}
_set_usage_outputs(span, response_obj, SpanAttributes)
attrs = _collect_calls(span)
assert SpanAttributes.LLM_TOKEN_COUNT_PROMPT_DETAILS_CACHE_READ not in attrs
@ -542,14 +499,8 @@ def test_passthrough_call_type_resolves_to_llm_span_kind():
from litellm.integrations._types.open_inference import OpenInferenceSpanKindValues
from litellm.integrations.arize._utils import _infer_open_inference_span_kind
assert (
_infer_open_inference_span_kind("allm_passthrough_route")
== OpenInferenceSpanKindValues.LLM.value
)
assert (
_infer_open_inference_span_kind("llm_passthrough_route")
== OpenInferenceSpanKindValues.LLM.value
)
assert _infer_open_inference_span_kind("allm_passthrough_route") == OpenInferenceSpanKindValues.LLM.value
assert _infer_open_inference_span_kind("llm_passthrough_route") == OpenInferenceSpanKindValues.LLM.value
def test_arize_chat_completion_with_tools_stays_llm_span_kind():
@ -605,9 +556,7 @@ def test_arize_chat_completion_with_tools_stays_llm_span_kind():
ArizeLogger.set_arize_attributes(span, kwargs, response_obj)
span_kind_writes = [
c.args[1]
for c in span.set_attribute.call_args_list
if c.args[0] == SpanAttributes.OPENINFERENCE_SPAN_KIND
c.args[1] for c in span.set_attribute.call_args_list if c.args[0] == SpanAttributes.OPENINFERENCE_SPAN_KIND
]
assert span_kind_writes, "span.kind must be written"
assert all(v == "LLM" for v in span_kind_writes)
@ -659,13 +608,8 @@ def test_arize_emits_assistant_tool_calls_on_output_message():
attrs = _collect_calls(span)
base = f"{SpanAttributes.LLM_OUTPUT_MESSAGES}.0.{MessageAttributes.MESSAGE_TOOL_CALLS}.0"
assert attrs[f"{base}.{ToolCallAttributes.TOOL_CALL_ID}"] == "call_abc"
assert (
attrs[f"{base}.{ToolCallAttributes.TOOL_CALL_FUNCTION_NAME}"] == "get_weather"
)
assert (
attrs[f"{base}.{ToolCallAttributes.TOOL_CALL_FUNCTION_ARGUMENTS_JSON}"]
== '{"location": "SF"}'
)
assert attrs[f"{base}.{ToolCallAttributes.TOOL_CALL_FUNCTION_NAME}"] == "get_weather"
assert attrs[f"{base}.{ToolCallAttributes.TOOL_CALL_FUNCTION_ARGUMENTS_JSON}"] == '{"location": "SF"}'
def test_arize_output_value_falls_back_to_tool_calls_summary():
@ -818,9 +762,7 @@ def test_arize_emits_tool_call_id_and_name_on_input_tool_message():
assert attrs[f"{assistant_base}.{ToolCallAttributes.TOOL_CALL_ID}"] == "call_abc"
# Tool message at index 2
tool_prefix = f"{SpanAttributes.LLM_INPUT_MESSAGES}.2"
assert (
attrs[f"{tool_prefix}.{MessageAttributes.MESSAGE_TOOL_CALL_ID}"] == "call_abc"
)
assert attrs[f"{tool_prefix}.{MessageAttributes.MESSAGE_TOOL_CALL_ID}"] == "call_abc"
assert attrs[f"{tool_prefix}.{MessageAttributes.MESSAGE_NAME}"] == "get_weather"
@ -866,10 +808,7 @@ def test_arize_emits_multimodal_input_contents():
assert attrs[f"{base}.0.message_content.type"] == "text"
assert attrs[f"{base}.0.message_content.text"] == "What is in this image?"
assert attrs[f"{base}.1.message_content.type"] == "image"
assert (
attrs[f"{base}.1.message_content.image.image.url"]
== "https://example.com/cat.png"
)
assert attrs[f"{base}.1.message_content.image.image.url"] == "https://example.com/cat.png"
def test_arize_emits_session_and_user_attrs_from_metadata():
@ -974,11 +913,7 @@ def test_arize_does_not_overwrite_user_id_from_optional_params():
id="r2",
)
ArizeLogger.set_arize_attributes(span, kwargs, response_obj)
user_id_writes = [
c.args[1]
for c in span.set_attribute.call_args_list
if c.args[0] == SpanAttributes.USER_ID
]
user_id_writes = [c.args[1] for c in span.set_attribute.call_args_list if c.args[0] == SpanAttributes.USER_ID]
assert "from_metadata" not in user_id_writes
@ -1048,9 +983,7 @@ def test_arize_passthrough_bedrock_anthropic_normalization():
"complete_input_dict": {
"anthropic_version": "bedrock-2023-05-31",
"max_tokens": 64,
"messages": [
{"role": "user", "content": "What is the capital of France?"}
],
"messages": [{"role": "user", "content": "What is the capital of France?"}],
}
},
"standard_logging_object": {
@ -1068,19 +1001,13 @@ def test_arize_passthrough_bedrock_anthropic_normalization():
assert attrs[SpanAttributes.INPUT_VALUE] == "What is the capital of France?"
msg0 = f"{SpanAttributes.LLM_INPUT_MESSAGES}.0"
assert attrs[f"{msg0}.{MessageAttributes.MESSAGE_ROLE}"] == "user"
assert (
attrs[f"{msg0}.{MessageAttributes.MESSAGE_CONTENT}"]
== "What is the capital of France?"
)
assert attrs[f"{msg0}.{MessageAttributes.MESSAGE_CONTENT}"] == "What is the capital of France?"
# Output rendering (Anthropic content[].text)
assert attrs[SpanAttributes.OUTPUT_VALUE] == "The capital of France is Paris."
out0 = f"{SpanAttributes.LLM_OUTPUT_MESSAGES}.0"
assert attrs[f"{out0}.{MessageAttributes.MESSAGE_ROLE}"] == "assistant"
assert (
attrs[f"{out0}.{MessageAttributes.MESSAGE_CONTENT}"]
== "The capital of France is Paris."
)
assert attrs[f"{out0}.{MessageAttributes.MESSAGE_CONTENT}"] == "The capital of France is Paris."
# Token counts (Bedrock input_tokens/output_tokens) — extracted via
# coercion of the non-dict response.
@ -1089,9 +1016,7 @@ def test_arize_passthrough_bedrock_anthropic_normalization():
# Span kind defended even though the call_type is a passthrough variant.
span_kind_writes = [
c.args[1]
for c in span.set_attribute.call_args_list
if c.args[0] == SpanAttributes.OPENINFERENCE_SPAN_KIND
c.args[1] for c in span.set_attribute.call_args_list if c.args[0] == SpanAttributes.OPENINFERENCE_SPAN_KIND
]
assert span_kind_writes # at least one
assert all(v == "LLM" for v in span_kind_writes)
@ -1109,11 +1034,7 @@ def test_arize_passthrough_call_type_does_not_run_on_chat_completion():
span = MagicMock()
_maybe_normalize_passthrough(
span,
{
"additional_args": {
"complete_input_dict": {"messages": [{"role": "user", "content": "x"}]}
}
},
{"additional_args": {"complete_input_dict": {"messages": [{"role": "user", "content": "x"}]}}},
{"choices": [{"message": {"role": "assistant", "content": "y"}}]},
{"choices": [{"message": {"role": "assistant", "content": "y"}}]},
{"call_type": "completion"},
@ -1133,11 +1054,7 @@ def test_arize_passthrough_skipped_when_message_redaction_enabled():
span = MagicMock()
kwargs = {
"additional_args": {
"complete_input_dict": {
"messages": [
{"role": "user", "content": "Patient John Doe, SSN 123-45-6789"}
]
}
"complete_input_dict": {"messages": [{"role": "user", "content": "Patient John Doe, SSN 123-45-6789"}]}
},
# Enables redaction via the dynamic-param path inside
# should_redact_message_logging(), without touching globals.
@ -1211,9 +1128,7 @@ def test_arize_mcp_call_tool_result_does_not_break_attribute_setting():
"optional_params": {},
"litellm_params": {"custom_llm_provider": "mcp"},
}
response_obj = CallToolResult(
content=[TextContent(type="text", text="sunny, 21C")], isError=False
)
response_obj = CallToolResult(content=[TextContent(type="text", text="sunny, 21C")], isError=False)
ArizeLogger.set_arize_attributes(span, kwargs, response_obj)
@ -1295,9 +1210,7 @@ def test_arize_mcp_tool_span_renders_name_input_and_output():
from mcp.types import CallToolResult, TextContent
span = MagicMock()
response_obj = CallToolResult(
content=[TextContent(type="text", text="sunny, 21C")], isError=False
)
response_obj = CallToolResult(content=[TextContent(type="text", text="sunny, 21C")], isError=False)
ArizeLogger.set_arize_attributes(span, _mcp_kwargs(), response_obj)
@ -1336,9 +1249,7 @@ def test_arize_mcp_tool_span_respects_message_redaction():
from mcp.types import CallToolResult, TextContent
span = MagicMock()
response_obj = CallToolResult(
content=[TextContent(type="text", text="SSN 123-45-6789")], isError=False
)
response_obj = CallToolResult(content=[TextContent(type="text", text="SSN 123-45-6789")], isError=False)
ArizeLogger.set_arize_attributes(
span,
@ -1513,3 +1424,34 @@ def test_arize_mcp_emitter_is_inert_without_a_standard_logging_object():
written = {c.args[0]: c.args[1] for c in span.set_attribute.call_args_list}
assert SpanAttributes.TOOL_NAME not in written
def test_arize_session_and_user_attrs_still_emit_from_key_metadata_by_default():
"""The emit_session_and_user split is Langfuse-only: Arize keeps session.id
= end user and user.id = internal key owner when no body user exists."""
from unittest.mock import MagicMock
span = MagicMock()
kwargs = {
"model": "gpt-4o",
"messages": [{"role": "user", "content": "hello"}],
"standard_logging_object": {
"call_type": "acompletion",
"model_parameters": {},
"metadata": {
"user_api_key_end_user_id": "end-1",
"user_api_key_user_id": "internal-1",
"user_api_key_team_id": "team-1",
},
"trace_id": "trace-1",
},
"optional_params": {},
"litellm_params": {"custom_llm_provider": "openai"},
}
ArizeLogger.set_arize_attributes(span, kwargs, {"id": "chatcmpl-1", "choices": [], "usage": {}})
span.set_attribute.assert_any_call(SpanAttributes.SESSION_ID, "end-1")
span.set_attribute.assert_any_call(SpanAttributes.USER_ID, "internal-1")
span.set_attribute.assert_any_call("litellm.trace_id", "trace-1")
span.set_attribute.assert_any_call("litellm.team_id", "team-1")

View file

@ -419,6 +419,31 @@ def test_langfuse_user_and_session_headers_beat_body_metadata_on_both_spans():
assert attrs["session.id"] == "from-header-s"
def test_the_proxy_end_user_fills_user_id_when_the_caller_names_no_trace_user():
logger, exporter = _logger()
root_attrs, generation_attrs = _run_named_request(
logger, exporter, {"metadata": {"user_api_key_end_user_id": "end-1"}, "proxy_server_request": {"headers": {}}}
)
assert root_attrs["user.id"] == "end-1"
def test_a_callers_trace_user_id_still_wins_over_the_proxy_end_user():
logger, exporter = _logger()
root_attrs, generation_attrs = _run_named_request(
logger,
exporter,
{
"metadata": {"trace_user_id": "caller-1", "user_api_key_end_user_id": "end-1"},
"proxy_server_request": {"headers": {}},
},
)
assert root_attrs["user.id"] == "caller-1"
def test_caller_metadata_cannot_override_the_proxy_team_identity():
logger, exporter = _logger()
response: Final = ModelResponse(choices=[Choices(message=Message(role="assistant", content="pong"))])
@ -461,7 +486,9 @@ def test_a_request_without_trace_controls_stamps_none_of_them():
logger, exporter = _logger()
root_attrs, generation_attrs = _run_named_request(
logger, exporter, {"metadata": {"user_api_key_team_id": "t1", "tags": []}, "proxy_server_request": {"headers": {}}}
logger,
exporter,
{"metadata": {"user_api_key_team_id": "t1", "tags": []}, "proxy_server_request": {"headers": {}}},
)
assert set(TRACE_CONTROL_ATTRS).isdisjoint(root_attrs)

View file

@ -45,7 +45,11 @@ from litellm.integrations.otel.model.spans import (
root_roles,
validate_registry,
)
from litellm.integrations.otel.model.trace_controls import TraceControls, caller_trace_controls
from litellm.integrations.otel.model.trace_controls import (
TraceControls,
caller_trace_controls,
langfuse_trace_controls,
)
@pytest.fixture(autouse=True)
@ -1337,6 +1341,43 @@ def test_caller_trace_controls_carry_user_session_and_tags(request_data, expecte
assert LLMCallEvent.from_dict({"litellm_params": request_data}).trace == expected
@pytest.mark.parametrize(
("request_data", "expected"),
[
({"metadata": {"user_api_key_end_user_id": "end-1"}}, TraceControls(user_id="end-1")),
({"litellm_metadata": {"user_api_key_end_user_id": "end-2"}}, TraceControls(user_id="end-2")),
(
{"metadata": {"trace_user_id": "caller-1", "user_api_key_end_user_id": "end-1"}},
TraceControls(user_id="caller-1"),
),
(
{
"proxy_server_request": {"headers": {"langfuse_trace_user_id": "header-1"}},
"metadata": {"user_api_key_end_user_id": "end-1"},
},
TraceControls(user_id="header-1"),
),
(
{"metadata": {"user_api_key_end_user_id": "end-1", "session_id": "s-1"}},
TraceControls(user_id="end-1", session_id="s-1"),
),
({"metadata": {"user_api_key_user_id": "internal-1"}}, TraceControls()),
({}, TraceControls()),
],
ids=[
"body-end-user",
"anthropic-end-user",
"body-caller-wins",
"header-caller-wins",
"session-kept",
"internal-user-ignored",
"empty",
],
)
def test_langfuse_trace_controls_fall_back_to_the_proxy_end_user(request_data, expected):
assert langfuse_trace_controls({"litellm_params": request_data}) == expected
def test_llm_span_data_carries_the_caller_trace_controls():
controls: Final = TraceControls(name="nightly-eval", user_id="u1", session_id="s1", tags=("a", "b"))
data: Final = LLMCallSpanData.from_standard_logging_payload(_sample_payload(), trace=controls)

View file

@ -104,19 +104,17 @@ class TestLangfuseOtelIntegration:
mock_kwargs = {"test": "kwargs"}
mock_response = {"test": "response"}
with patch(
"litellm.integrations.arize._utils.set_attributes"
) as mock_set_attributes:
LangfuseOtelLogger.set_langfuse_otel_attributes(
mock_span, mock_kwargs, mock_response
)
with patch("litellm.integrations.arize._utils.set_attributes") as mock_set_attributes:
LangfuseOtelLogger.set_langfuse_otel_attributes(mock_span, mock_kwargs, mock_response)
mock_set_attributes.assert_called_once_with(
mock_span, mock_kwargs, mock_response, LangfuseLLMObsOTELAttributes
)
mock_span.set_attribute.assert_any_call(
"langfuse.observation.type", "generation"
mock_span,
mock_kwargs,
mock_response,
LangfuseLLMObsOTELAttributes,
emit_session_and_user=False,
)
mock_span.set_attribute.assert_any_call("langfuse.observation.type", "generation")
def test_set_langfuse_environment_attribute(self):
"""Test that Langfuse environment is set correctly when environment variable is present."""
@ -125,17 +123,11 @@ class TestLangfuseOtelIntegration:
test_env = "staging"
with patch.dict(os.environ, {"LANGFUSE_TRACING_ENVIRONMENT": test_env}):
with patch(
"litellm.integrations.arize._utils.safe_set_attribute"
) as mock_safe_set_attribute:
LangfuseOtelLogger._set_langfuse_specific_attributes(
mock_span, mock_kwargs, {}
)
with patch("litellm.integrations.arize._utils.safe_set_attribute") as mock_safe_set_attribute:
LangfuseOtelLogger._set_langfuse_specific_attributes(mock_span, mock_kwargs, {})
# safe_set_attribute(span, key, value) → positional args
mock_safe_set_attribute.assert_called_once_with(
mock_span, "langfuse.environment", test_env
)
mock_safe_set_attribute.assert_called_once_with(mock_span, "langfuse.environment", test_env)
def test_set_langfuse_environment_attribute_prefers_dynamic_param(self):
"""Per-key/team langfuse_environment beats the deployment env var."""
@ -148,18 +140,10 @@ class TestLangfuseOtelIntegration:
self.attributes[key] = value
span = _RecordingSpan()
mock_kwargs = {
"standard_callback_dynamic_params": {
"langfuse_environment": "team-a-env"
}
}
mock_kwargs = {"standard_callback_dynamic_params": {"langfuse_environment": "team-a-env"}}
with patch.dict(
os.environ, {"LANGFUSE_TRACING_ENVIRONMENT": "deployment-wide"}
):
LangfuseOtelLogger._set_langfuse_specific_attributes(
span, mock_kwargs, {}
)
with patch.dict(os.environ, {"LANGFUSE_TRACING_ENVIRONMENT": "deployment-wide"}):
LangfuseOtelLogger._set_langfuse_specific_attributes(span, mock_kwargs, {})
assert span.attributes["langfuse.environment"] == "team-a-env"
@ -223,12 +207,8 @@ class TestLangfuseOtelIntegration:
kwargs = {"litellm_params": {"metadata": metadata}}
# Capture calls to safe_set_attribute
with patch(
"litellm.integrations.arize._utils.safe_set_attribute"
) as mock_safe_set_attribute:
LangfuseOtelLogger._set_langfuse_specific_attributes(
MagicMock(), kwargs, None
)
with patch("litellm.integrations.arize._utils.safe_set_attribute") as mock_safe_set_attribute:
LangfuseOtelLogger._set_langfuse_specific_attributes(MagicMock(), kwargs, None)
# Build expected calls manually for clarity
from litellm.types.integrations.langfuse_otel import LangfuseSpanAttributes
@ -249,9 +229,7 @@ class TestLangfuseOtelIntegration:
LangfuseSpanAttributes.TRACE_METADATA.value: json.dumps({"k": "v"}),
LangfuseSpanAttributes.RELEASE.value: "rel-1",
LangfuseSpanAttributes.EXISTING_TRACE_ID.value: "existing-id",
LangfuseSpanAttributes.UPDATE_TRACE_KEYS.value: json.dumps(
["key1", "key2"]
),
LangfuseSpanAttributes.UPDATE_TRACE_KEYS.value: json.dumps(["key1", "key2"]),
LangfuseSpanAttributes.DEBUG_LANGFUSE.value: True,
}
@ -261,9 +239,7 @@ class TestLangfuseOtelIntegration:
for call in mock_safe_set_attribute.call_args_list
}
assert (
actual == expected
), "Mismatch between expected and actual OTEL attribute mapping."
assert actual == expected, "Mismatch between expected and actual OTEL attribute mapping."
@pytest.mark.parametrize(
"metadata, expected_version",
@ -288,16 +264,10 @@ class TestLangfuseOtelIntegration:
def test_version_emitted_on_langfuse_v4_key(self, metadata, expected_version):
kwargs = {"litellm_params": {"metadata": {"trace_release": "rel-9", **metadata}}}
with patch(
"litellm.integrations.arize._utils.safe_set_attribute"
) as mock_safe_set_attribute:
LangfuseOtelLogger._set_langfuse_specific_attributes(
MagicMock(), kwargs, None
)
with patch("litellm.integrations.arize._utils.safe_set_attribute") as mock_safe_set_attribute:
LangfuseOtelLogger._set_langfuse_specific_attributes(MagicMock(), kwargs, None)
emitted = {
call.args[1]: call.args[2] for call in mock_safe_set_attribute.call_args_list
}
emitted = {call.args[1]: call.args[2] for call in mock_safe_set_attribute.call_args_list}
if expected_version is None:
assert "langfuse.version" not in emitted
@ -335,12 +305,8 @@ class TestLangfuseOtelIntegration:
"messages": [{"role": "user", "content": "What's the weather in Tokyo?"}],
}
with patch(
"litellm.integrations.arize._utils.safe_set_attribute"
) as mock_safe_set_attribute:
LangfuseOtelLogger._set_langfuse_specific_attributes(
MagicMock(), kwargs, response_obj
)
with patch("litellm.integrations.arize._utils.safe_set_attribute") as mock_safe_set_attribute:
LangfuseOtelLogger._set_langfuse_specific_attributes(MagicMock(), kwargs, response_obj)
expect_output = {
LangfuseSpanAttributes.OBSERVATION_INPUT.value: [
@ -353,14 +319,9 @@ class TestLangfuseOtelIntegration:
}
# Flatten the actual calls into {key: value}
actual = {
call.args[1]: json.loads(call.args[2])
for call in mock_safe_set_attribute.call_args_list
}
actual = {call.args[1]: json.loads(call.args[2]) for call in mock_safe_set_attribute.call_args_list}
assert (
actual == expect_output
), "Mismatch in observation input/output OTEL attributes."
assert actual == expect_output, "Mismatch in observation input/output OTEL attributes."
def test_set_langfuse_specific_attributes_with_tool_calls(self):
"""Test that _set_langfuse_specific_attributes correctly sets observation.output with tool calls in Langfuse format."""
@ -384,9 +345,7 @@ class TestLangfuseOtelIntegration:
"content": None,
"tool_calls": [
ChatCompletionMessageToolCall(
function=Function(
arguments='{"location":"Tokyo"}', name="get_weather"
),
function=Function(arguments='{"location":"Tokyo"}', name="get_weather"),
id="call_123",
type="function",
)
@ -396,12 +355,8 @@ class TestLangfuseOtelIntegration:
],
)
with patch(
"litellm.integrations.arize._utils.safe_set_attribute"
) as mock_safe_set_attribute:
LangfuseOtelLogger._set_langfuse_specific_attributes(
MagicMock(), {}, response_obj
)
with patch("litellm.integrations.arize._utils.safe_set_attribute") as mock_safe_set_attribute:
LangfuseOtelLogger._set_langfuse_specific_attributes(MagicMock(), {}, response_obj)
expected = {
LangfuseSpanAttributes.OBSERVATION_OUTPUT.value: [
@ -416,13 +371,8 @@ class TestLangfuseOtelIntegration:
}
# Flatten the actual calls into {key: value}
actual = {
call.args[1]: json.loads(call.args[2])
for call in mock_safe_set_attribute.call_args_list
}
assert (
actual == expected
), "Mismatch in observation output OTEL attribute for tool calls."
actual = {call.args[1]: json.loads(call.args[2]) for call in mock_safe_set_attribute.call_args_list}
assert actual == expected, "Mismatch in observation output OTEL attribute for tool calls."
def test_construct_dynamic_otel_headers_with_langfuse_keys(self):
"""Test that construct_dynamic_otel_headers creates proper auth headers when langfuse keys are provided."""
@ -573,9 +523,7 @@ class TestLangfuseOtelKeyDynamicConfig:
logger = LangfuseOtelLogger()
assert logger.OTEL_EXPORTER == "console"
tracer = logger.get_tracer_to_use_for_request(
{"standard_callback_dynamic_params": self._dynamic_params()}
)
tracer = logger.get_tracer_to_use_for_request({"standard_callback_dynamic_params": self._dynamic_params()})
assert tracer is not logger.tracer
assert len(logger._tracer_provider_cache) == 1
@ -636,9 +584,7 @@ class TestLangfuseOtelKeyDynamicConfig:
with self._clean_env():
logger = LangfuseOtelLogger()
with patch.object(otel_module.verbose_logger, "debug", side_effect=_spy):
logger.get_tracer_to_use_for_request(
{"standard_callback_dynamic_params": self._dynamic_params()}
)
logger.get_tracer_to_use_for_request({"standard_callback_dynamic_params": self._dynamic_params()})
logged = "\n".join(recorded_arguments)
assert "initializing span processor" in logged
@ -698,27 +644,21 @@ class TestLangfuseOtelResponsesAPI:
LangfuseLLMObsOTELAttributes,
)
with patch(
"litellm.integrations.arize._utils.set_attributes"
) as mock_set_attributes:
with patch(
"litellm.integrations.arize._utils.safe_set_attribute"
) as mock_safe_set_attribute:
with patch("litellm.integrations.arize._utils.set_attributes") as mock_set_attributes:
with patch("litellm.integrations.arize._utils.safe_set_attribute") as mock_safe_set_attribute:
logger = LangfuseOtelLogger()
logger.set_langfuse_otel_attributes(mock_span, kwargs, mock_response)
# Verify that set_attributes was called for general attributes
mock_set_attributes.assert_called_once_with(
mock_span, kwargs, mock_response, LangfuseLLMObsOTELAttributes
mock_span, kwargs, mock_response, LangfuseLLMObsOTELAttributes, emit_session_and_user=False
)
# Verify that Langfuse-specific attributes were set
mock_safe_set_attribute.assert_any_call(
mock_span, "langfuse.generation.name", "responses_test_generation"
)
mock_safe_set_attribute.assert_any_call(
mock_span, "langfuse.trace.name", "responses_api_trace"
)
mock_safe_set_attribute.assert_any_call(mock_span, "langfuse.trace.name", "responses_api_trace")
def test_responses_api_metadata_extraction(self):
"""Test that metadata is correctly extracted from ResponsesAPI kwargs."""
@ -768,9 +708,7 @@ class TestLangfuseOtelResponsesAPI:
mock_span = MagicMock()
with patch(
"litellm.integrations.arize._utils.safe_set_attribute"
) as mock_safe_set_attribute:
with patch("litellm.integrations.arize._utils.safe_set_attribute") as mock_safe_set_attribute:
LangfuseOtelLogger._set_langfuse_specific_attributes(mock_span, kwargs, {})
# Verify specific attributes were set
@ -812,11 +750,12 @@ class TestLangfuseOtelResponsesAPI:
def test_responses_api_with_output(self):
"""Test Langfuse OTEL logger with Responses API output (reasoning + message)."""
from openai.types.responses import (
ResponseReasoningItem,
ResponseOutputMessage,
ResponseOutputText,
ResponseReasoningItem,
)
from openai.types.responses.response_reasoning_item import Summary
from litellm.types.integrations.langfuse_otel import LangfuseSpanAttributes
# Create Responses API response with reasoning and message
@ -852,21 +791,15 @@ class TestLangfuseOtelResponsesAPI:
kwargs = {
"call_type": "responses",
"messages": [
{"role": "user", "content": "What's the weather in San Francisco?"}
],
"messages": [{"role": "user", "content": "What's the weather in San Francisco?"}],
"model": "gpt-4o",
"optional_params": {},
}
mock_span = MagicMock()
with patch(
"litellm.integrations.arize._utils.safe_set_attribute"
) as mock_safe_set_attribute:
LangfuseOtelLogger._set_langfuse_specific_attributes(
mock_span, kwargs, response_obj
)
with patch("litellm.integrations.arize._utils.safe_set_attribute") as mock_safe_set_attribute:
LangfuseOtelLogger._set_langfuse_specific_attributes(mock_span, kwargs, response_obj)
# Verify observation output was set
output_calls = [
@ -885,23 +818,18 @@ class TestLangfuseOtelResponsesAPI:
# Verify reasoning summary
assert output_data[0]["role"] == "reasoning_summary"
assert (
output_data[0]["content"]
== "Let me analyze this problem step by step..."
)
assert output_data[0]["content"] == "Let me analyze this problem step by step..."
# Verify message
assert output_data[1]["role"] == "assistant"
assert (
output_data[1]["content"]
== "The weather in San Francisco is sunny, 20°C."
)
assert output_data[1]["content"] == "The weather in San Francisco is sunny, 20°C."
def test_responses_api_with_function_calls(self):
"""Test Langfuse OTEL logger with Responses API function_call output."""
from litellm.types.integrations.langfuse_otel import LangfuseSpanAttributes
from openai.types.responses import ResponseFunctionToolCall
from litellm.types.integrations.langfuse_otel import LangfuseSpanAttributes
# Create Responses API response with function call
response_obj = ResponsesAPIResponse(
id="response-789",
@ -920,21 +848,15 @@ class TestLangfuseOtelResponsesAPI:
kwargs = {
"call_type": "responses",
"messages": [
{"role": "user", "content": "What's the weather in San Francisco?"}
],
"messages": [{"role": "user", "content": "What's the weather in San Francisco?"}],
"model": "gpt-4o",
"optional_params": {},
}
mock_span = MagicMock()
with patch(
"litellm.integrations.arize._utils.safe_set_attribute"
) as mock_safe_set_attribute:
LangfuseOtelLogger._set_langfuse_specific_attributes(
mock_span, kwargs, response_obj
)
with patch("litellm.integrations.arize._utils.safe_set_attribute") as mock_safe_set_attribute:
LangfuseOtelLogger._set_langfuse_specific_attributes(mock_span, kwargs, response_obj)
# Verify observation output was set
output_calls = [
@ -989,9 +911,11 @@ class TestLangfuseOtelResponsesAPI:
mock_span = MagicMock()
with patch( # test-quality-ok: the span attribute sink is the observable boundary; sibling tests in this class stub the same seam
"litellm.integrations.arize._utils.safe_set_attribute"
) as mock_safe_set_attribute:
with (
patch( # test-quality-ok: the span attribute sink is the observable boundary; sibling tests in this class stub the same seam
"litellm.integrations.arize._utils.safe_set_attribute"
) as mock_safe_set_attribute
):
LangfuseOtelLogger._set_langfuse_specific_attributes(mock_span, kwargs, response_obj)
output_calls = [
@ -1006,5 +930,103 @@ class TestLangfuseOtelResponsesAPI:
assert output_data[0]["arguments"] == {}
class TestLangfuseOtelTraceIdentity:
def _recording_span(self):
from opentelemetry.sdk.trace import TracerProvider
return TracerProvider().get_tracer("test").start_span("generation")
def _kwargs(self, slp_metadata=None, litellm_metadata=None, model_parameters=None, slp_extra=None):
return {
"model": "gpt-4o",
"messages": [{"role": "user", "content": "hello"}],
"optional_params": {},
"litellm_params": {"metadata": litellm_metadata or {}, "custom_llm_provider": "openai"},
"standard_logging_object": {
"call_type": "acompletion",
"model_parameters": model_parameters or {},
"metadata": slp_metadata or {},
**(slp_extra or {}),
},
}
def _response_obj(self):
return {
"id": "chatcmpl-1",
"model": "gpt-4o",
"choices": [{"index": 0, "message": {"role": "assistant", "content": "hi"}, "finish_reason": "stop"}],
"usage": {"prompt_tokens": 1, "completion_tokens": 1, "total_tokens": 2},
}
def _identity(self, kwargs):
span = self._recording_span()
LangfuseOtelLogger.set_langfuse_otel_attributes(span, kwargs, self._response_obj())
attributes = dict(span.attributes or {})
return {key: attributes.get(key) for key in ("user.id", "session.id")}, attributes
def test_header_end_user_beats_internal_key_owner_in_user_id(self):
identity, _ = self._identity(
self._kwargs(
slp_metadata={
"user_api_key_end_user_id": "end-1",
"user_api_key_user_id": "internal-1",
}
)
)
assert identity == {"user.id": "end-1", "session.id": None}
def test_header_end_user_lands_in_user_id_for_a_service_key(self):
identity, _ = self._identity(self._kwargs(slp_metadata={"user_api_key_end_user_id": "end-1"}))
assert identity == {"user.id": "end-1", "session.id": None}
def test_body_user_is_never_a_session(self):
identity, _ = self._identity(
self._kwargs(
slp_metadata={"user_api_key_end_user_id": "body-user"},
model_parameters={"user": "body-user"},
)
)
assert identity == {"user.id": "body-user", "session.id": None}
def test_caller_trace_user_id_wins_over_the_end_user(self):
identity, _ = self._identity(
self._kwargs(
slp_metadata={"user_api_key_end_user_id": "end-1"},
litellm_metadata={"trace_user_id": "caller-1"},
)
)
assert identity == {"user.id": "caller-1", "session.id": None}
def test_caller_session_id_stays_the_session_beside_the_end_user(self):
identity, _ = self._identity(
self._kwargs(
slp_metadata={"user_api_key_end_user_id": "end-1"},
litellm_metadata={"session_id": "sess-1"},
)
)
assert identity == {"user.id": "end-1", "session.id": "sess-1"}
def test_internal_user_without_an_end_user_never_lands_in_user_id(self):
identity, _ = self._identity(self._kwargs(slp_metadata={"user_api_key_user_id": "internal-1"}))
assert identity == {"user.id": None, "session.id": None}
def test_request_context_attributes_still_emit(self):
_, attributes = self._identity(
self._kwargs(
slp_metadata={
"user_api_key_end_user_id": "end-1",
"user_api_key_team_id": "team-1",
"user_api_key_team_alias": "team-alias",
"user_api_key_alias": "key-alias",
},
slp_extra={"trace_id": "trace-1"},
)
)
assert attributes["litellm.trace_id"] == "trace-1"
assert attributes["litellm.team_id"] == "team-1"
assert attributes["litellm.team_alias"] == "team-alias"
assert attributes["litellm.key_alias"] == "key-alias"
if __name__ == "__main__":
pytest.main([__file__])