refactor(langfuse_otel): keep arize payload handling unchanged and drop test churn

Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
yucheng 2026-09-23 07:22:15 +00:00
parent aa1d732210
commit 2cd2feb909
5 changed files with 270 additions and 111 deletions

View file

@ -9,6 +9,7 @@ from litellm.integrations.opentelemetry_utils.base_otel_llm_obs_attributes impor
BaseLLMObsOTELAttributes,
safe_set_attribute,
)
from litellm.integrations.otel.model.utils import as_str_mapping
from litellm.litellm_core_utils.redact_messages import (
should_redact_message_logging,
)
@ -477,9 +478,7 @@ def set_attributes(
# Additive emitters. Each is independently guarded so a failure can never
# blank the attributes set by the main try-block above. New attributes are
# written under new keys; existing attributes are not overwritten.
from litellm.integrations.otel.model.utils import as_str_mapping
slp: Final = as_str_mapping(kwargs.get("standard_logging_object"))
slp: Final = kwargs.get("standard_logging_object")
if emit_session_and_user:
_safe_emit("session/user attrs", _set_session_and_user_attrs, span, kwargs, slp)
_safe_emit("request context attrs", _set_request_context_attrs, span, slp)
@ -880,8 +879,6 @@ def _set_session_and_user_attrs(span: "Span", kwargs: dict, standard_logging_pay
def _set_request_context_attrs(span: "Span", standard_logging_payload: object) -> None:
"""Emit `litellm.trace_id` / team / key context when source data exists."""
from litellm.integrations.otel.model.utils import as_str_mapping
payload: Final = as_str_mapping(standard_logging_payload)
if payload is None:
return

View file

@ -6,10 +6,12 @@ from typing import TYPE_CHECKING, Any, Final, Optional
from litellm._logging import verbose_logger
from litellm.integrations.arize import _utils
from litellm.integrations.arize._utils import safe_set_attribute
from litellm.integrations.langfuse.langfuse_otel_attributes import (
LangfuseLLMObsOTELAttributes,
)
from litellm.integrations.opentelemetry import OpenTelemetry, OpenTelemetryConfig
from litellm.integrations.otel.model.utils import as_str, as_str_mapping
from litellm.litellm_core_utils.safe_json_loads import safe_json_loads
from litellm.types.integrations.langfuse_otel import (
LangfuseSpanAttributes,
@ -258,9 +260,6 @@ class LangfuseOtelLogger(OpenTelemetry):
@staticmethod
def _set_trace_user_attribute(span: Span, kwargs: dict[str, object]) -> None:
from litellm.integrations.arize._utils import safe_set_attribute
from litellm.integrations.otel.model.utils import as_str, as_str_mapping
slp: Final = as_str_mapping(kwargs.get("standard_logging_object"))
slp_metadata: Final = as_str_mapping(slp.get("metadata")) if slp is not None else None
if slp is None or slp_metadata is None:

View file

@ -1,6 +1,9 @@
# Adds the grandparent directory to sys.path to allow importing project modules
import asyncio
import json
from typing import Optional
# Adds the grandparent directory to sys.path to allow importing project modules
import asyncio
import pytest
@ -67,7 +70,9 @@ def test_arize_set_attributes():
# Simulated LLM response object
response_obj = ModelResponse(
usage={"total_tokens": 100, "completion_tokens": 60, "prompt_tokens": 40},
choices=[Choices(message={"role": "assistant", "content": "Basic Response Content"})],
choices=[
Choices(message={"role": "assistant", "content": "Basic Response Content"})
],
model="gpt-4o",
id="chatcmpl-ID",
)
@ -84,7 +89,9 @@ def test_arize_set_attributes():
assert span.set_attribute.call_count == 26
# Metadata attached to the span
span.set_attribute.assert_any_call(SpanAttributes.METADATA, json.dumps({"key_1": "value_1", "key_2": None}))
span.set_attribute.assert_any_call(
SpanAttributes.METADATA, json.dumps({"key_1": "value_1", "key_2": None})
)
# Basic LLM information
span.set_attribute.assert_any_call(SpanAttributes.LLM_MODEL_NAME, "gpt-4o")
@ -107,12 +114,16 @@ def test_arize_set_attributes():
span.set_attribute.assert_any_call(SpanAttributes.OPENINFERENCE_SPAN_KIND, "LLM")
# And TOOL must never be written for an LLM chat completion call.
span_kind_writes = [
c.args[1] for c in span.set_attribute.call_args_list if c.args[0] == SpanAttributes.OPENINFERENCE_SPAN_KIND
c.args[1]
for c in span.set_attribute.call_args_list
if c.args[0] == SpanAttributes.OPENINFERENCE_SPAN_KIND
]
assert "TOOL" not in span_kind_writes
# Request message content and metadata
span.set_attribute.assert_any_call(SpanAttributes.INPUT_VALUE, "Basic Request Content")
span.set_attribute.assert_any_call(
SpanAttributes.INPUT_VALUE, "Basic Request Content"
)
span.set_attribute.assert_any_call(
f"{SpanAttributes.LLM_INPUT_MESSAGES}.0.{MessageAttributes.MESSAGE_ROLE}",
"user",
@ -123,7 +134,9 @@ def test_arize_set_attributes():
)
# Tool call definitions and function names
span.set_attribute.assert_any_call(f"{SpanAttributes.LLM_TOOLS}.0.name", "get_weather")
span.set_attribute.assert_any_call(
f"{SpanAttributes.LLM_TOOLS}.0.name", "get_weather"
)
span.set_attribute.assert_any_call(
f"{SpanAttributes.LLM_TOOLS}.0.description",
"Fetches weather details.",
@ -133,20 +146,26 @@ def test_arize_set_attributes():
json.dumps(
{
"type": "object",
"properties": {"location": {"type": "string", "description": "City name"}},
"properties": {
"location": {"type": "string", "description": "City name"}
},
"required": ["location"],
}
),
)
# Invocation parameters
span.set_attribute.assert_any_call(SpanAttributes.LLM_INVOCATION_PARAMETERS, '{"user": "test_user"}')
span.set_attribute.assert_any_call(
SpanAttributes.LLM_INVOCATION_PARAMETERS, '{"user": "test_user"}'
)
# User ID
span.set_attribute.assert_any_call(SpanAttributes.USER_ID, "test_user")
# Output message content
span.set_attribute.assert_any_call(SpanAttributes.OUTPUT_VALUE, "Basic Response Content")
span.set_attribute.assert_any_call(
SpanAttributes.OUTPUT_VALUE, "Basic Response Content"
)
span.set_attribute.assert_any_call(
f"{SpanAttributes.LLM_OUTPUT_MESSAGES}.0.{MessageAttributes.MESSAGE_ROLE}",
"assistant",
@ -168,20 +187,18 @@ def test_arize_set_attributes_responses_api():
Verifies that multiple output types are correctly handled.
"""
from unittest.mock import MagicMock
from litellm.types.llms.openai import (
ResponsesAPIResponse,
ResponseAPIUsage,
OutputTokensDetails,
)
from openai.types.responses import (
ResponseReasoningItem,
ResponseOutputMessage,
ResponseOutputText,
ResponseReasoningItem,
)
from openai.types.responses.response_reasoning_item import Summary
from litellm.types.llms.openai import (
OutputTokensDetails,
ResponseAPIUsage,
ResponsesAPIResponse,
)
span = MagicMock() # Mocked tracing span to test attribute setting
# Construct kwargs to simulate a real LLM request scenario
@ -211,7 +228,9 @@ def test_arize_set_attributes_responses_api():
ResponseReasoningItem(
id="reasoning-001",
type="reasoning",
summary=[Summary(text="First, I need to analyze...", type="summary_text")],
summary=[
Summary(text="First, I need to analyze...", type="summary_text")
],
),
ResponseOutputMessage(
id="msg-001",
@ -258,7 +277,9 @@ def test_arize_set_attributes_responses_api():
span.set_attribute.assert_any_call(SpanAttributes.LLM_TOKEN_COUNT_TOTAL, 370)
span.set_attribute.assert_any_call(SpanAttributes.LLM_TOKEN_COUNT_COMPLETION, 250)
span.set_attribute.assert_any_call(SpanAttributes.LLM_TOKEN_COUNT_PROMPT, 120)
span.set_attribute.assert_any_call(SpanAttributes.LLM_TOKEN_COUNT_COMPLETION_DETAILS_REASONING, 180)
span.set_attribute.assert_any_call(
SpanAttributes.LLM_TOKEN_COUNT_COMPLETION_DETAILS_REASONING, 180
)
def test_set_usage_outputs_pydantic_completion_usage():
@ -306,7 +327,9 @@ def test_set_usage_outputs_pydantic_completion_usage():
span.set_attribute.assert_any_call(SpanAttributes.LLM_TOKEN_COUNT_PROMPT, 40)
span.set_attribute.assert_any_call(SpanAttributes.LLM_TOKEN_COUNT_COMPLETION, 60)
# reasoning_tokens for chat completions live in completion_tokens_details
span.set_attribute.assert_any_call(SpanAttributes.LLM_TOKEN_COUNT_COMPLETION_DETAILS_REASONING, 25)
span.set_attribute.assert_any_call(
SpanAttributes.LLM_TOKEN_COUNT_COMPLETION_DETAILS_REASONING, 25
)
def test_set_usage_outputs_pydantic_response_api_usage():
@ -339,7 +362,9 @@ def test_set_usage_outputs_pydantic_response_api_usage():
span.set_attribute.assert_any_call(SpanAttributes.LLM_TOKEN_COUNT_TOTAL, 370)
span.set_attribute.assert_any_call(SpanAttributes.LLM_TOKEN_COUNT_PROMPT, 120)
span.set_attribute.assert_any_call(SpanAttributes.LLM_TOKEN_COUNT_COMPLETION, 250)
span.set_attribute.assert_any_call(SpanAttributes.LLM_TOKEN_COUNT_COMPLETION_DETAILS_REASONING, 180)
span.set_attribute.assert_any_call(
SpanAttributes.LLM_TOKEN_COUNT_COMPLETION_DETAILS_REASONING, 180
)
class TestArizeLogger(CustomLogger):
@ -350,12 +375,16 @@ class TestArizeLogger(CustomLogger):
def __init__(self, *args, **kwargs):
super().__init__(*args, **kwargs)
self.standard_callback_dynamic_params: StandardCallbackDynamicParams | None = None
self.standard_callback_dynamic_params: Optional[
StandardCallbackDynamicParams
] = None
async def async_log_success_event(self, kwargs, response_obj, start_time, end_time):
# Capture dynamic params and print them for verification
print("logged kwargs", json.dumps(kwargs, indent=4, default=str))
self.standard_callback_dynamic_params = kwargs.get("standard_callback_dynamic_params")
self.standard_callback_dynamic_params = kwargs.get(
"standard_callback_dynamic_params"
)
@pytest.mark.asyncio
@ -381,8 +410,14 @@ async def test_arize_dynamic_params():
# Assert dynamic parameters were received in the callback
assert test_arize_logger.standard_callback_dynamic_params is not None
assert test_arize_logger.standard_callback_dynamic_params.get("arize_api_key") == "test_api_key_dynamic"
assert test_arize_logger.standard_callback_dynamic_params.get("arize_space_key") == "test_space_key_dynamic"
assert (
test_arize_logger.standard_callback_dynamic_params.get("arize_api_key")
== "test_api_key_dynamic"
)
assert (
test_arize_logger.standard_callback_dynamic_params.get("arize_space_key")
== "test_space_key_dynamic"
)
def test_construct_dynamic_arize_headers():
@ -393,7 +428,9 @@ def test_construct_dynamic_arize_headers():
from litellm.types.utils import StandardCallbackDynamicParams
# Test with all parameters present
dynamic_params_full = StandardCallbackDynamicParams(arize_api_key="test_api_key", arize_space_id="test_space_id")
dynamic_params_full = StandardCallbackDynamicParams(
arize_api_key="test_api_key", arize_space_id="test_space_id"
)
arize_logger = ArizeLogger()
headers = arize_logger.construct_dynamic_otel_headers(dynamic_params_full)
@ -401,7 +438,9 @@ def test_construct_dynamic_arize_headers():
assert headers == expected_headers
# Test with only space_id
dynamic_params_space_id_only = StandardCallbackDynamicParams(arize_space_id="test_space_id")
dynamic_params_space_id_only = StandardCallbackDynamicParams(
arize_space_id="test_space_id"
)
headers = arize_logger.construct_dynamic_otel_headers(dynamic_params_space_id_only)
expected_headers = {"arize-space-id": "test_space_id"}
@ -417,7 +456,9 @@ def test_construct_dynamic_arize_headers():
dynamic_params_space_key_and_api_key = StandardCallbackDynamicParams(
arize_space_key="test_space_key", arize_api_key="test_api_key"
)
headers = arize_logger.construct_dynamic_otel_headers(dynamic_params_space_key_and_api_key)
headers = arize_logger.construct_dynamic_otel_headers(
dynamic_params_space_key_and_api_key
)
expected_headers = {"arize-space-id": "test_space_key", "api_key": "test_api_key"}
@ -487,7 +528,9 @@ def test_arize_emits_no_cache_tokens_when_absent():
from litellm.integrations.arize._utils import _set_usage_outputs
span = MagicMock()
response_obj = {"usage": {"total_tokens": 10, "completion_tokens": 4, "prompt_tokens": 6}}
response_obj = {
"usage": {"total_tokens": 10, "completion_tokens": 4, "prompt_tokens": 6}
}
_set_usage_outputs(span, response_obj, SpanAttributes)
attrs = _collect_calls(span)
assert SpanAttributes.LLM_TOKEN_COUNT_PROMPT_DETAILS_CACHE_READ not in attrs
@ -499,8 +542,14 @@ def test_passthrough_call_type_resolves_to_llm_span_kind():
from litellm.integrations._types.open_inference import OpenInferenceSpanKindValues
from litellm.integrations.arize._utils import _infer_open_inference_span_kind
assert _infer_open_inference_span_kind("allm_passthrough_route") == OpenInferenceSpanKindValues.LLM.value
assert _infer_open_inference_span_kind("llm_passthrough_route") == OpenInferenceSpanKindValues.LLM.value
assert (
_infer_open_inference_span_kind("allm_passthrough_route")
== OpenInferenceSpanKindValues.LLM.value
)
assert (
_infer_open_inference_span_kind("llm_passthrough_route")
== OpenInferenceSpanKindValues.LLM.value
)
def test_arize_chat_completion_with_tools_stays_llm_span_kind():
@ -556,7 +605,9 @@ def test_arize_chat_completion_with_tools_stays_llm_span_kind():
ArizeLogger.set_arize_attributes(span, kwargs, response_obj)
span_kind_writes = [
c.args[1] for c in span.set_attribute.call_args_list if c.args[0] == SpanAttributes.OPENINFERENCE_SPAN_KIND
c.args[1]
for c in span.set_attribute.call_args_list
if c.args[0] == SpanAttributes.OPENINFERENCE_SPAN_KIND
]
assert span_kind_writes, "span.kind must be written"
assert all(v == "LLM" for v in span_kind_writes)
@ -608,8 +659,13 @@ def test_arize_emits_assistant_tool_calls_on_output_message():
attrs = _collect_calls(span)
base = f"{SpanAttributes.LLM_OUTPUT_MESSAGES}.0.{MessageAttributes.MESSAGE_TOOL_CALLS}.0"
assert attrs[f"{base}.{ToolCallAttributes.TOOL_CALL_ID}"] == "call_abc"
assert attrs[f"{base}.{ToolCallAttributes.TOOL_CALL_FUNCTION_NAME}"] == "get_weather"
assert attrs[f"{base}.{ToolCallAttributes.TOOL_CALL_FUNCTION_ARGUMENTS_JSON}"] == '{"location": "SF"}'
assert (
attrs[f"{base}.{ToolCallAttributes.TOOL_CALL_FUNCTION_NAME}"] == "get_weather"
)
assert (
attrs[f"{base}.{ToolCallAttributes.TOOL_CALL_FUNCTION_ARGUMENTS_JSON}"]
== '{"location": "SF"}'
)
def test_arize_output_value_falls_back_to_tool_calls_summary():
@ -762,7 +818,9 @@ def test_arize_emits_tool_call_id_and_name_on_input_tool_message():
assert attrs[f"{assistant_base}.{ToolCallAttributes.TOOL_CALL_ID}"] == "call_abc"
# Tool message at index 2
tool_prefix = f"{SpanAttributes.LLM_INPUT_MESSAGES}.2"
assert attrs[f"{tool_prefix}.{MessageAttributes.MESSAGE_TOOL_CALL_ID}"] == "call_abc"
assert (
attrs[f"{tool_prefix}.{MessageAttributes.MESSAGE_TOOL_CALL_ID}"] == "call_abc"
)
assert attrs[f"{tool_prefix}.{MessageAttributes.MESSAGE_NAME}"] == "get_weather"
@ -808,7 +866,10 @@ def test_arize_emits_multimodal_input_contents():
assert attrs[f"{base}.0.message_content.type"] == "text"
assert attrs[f"{base}.0.message_content.text"] == "What is in this image?"
assert attrs[f"{base}.1.message_content.type"] == "image"
assert attrs[f"{base}.1.message_content.image.image.url"] == "https://example.com/cat.png"
assert (
attrs[f"{base}.1.message_content.image.image.url"]
== "https://example.com/cat.png"
)
def test_arize_emits_session_and_user_attrs_from_metadata():
@ -913,7 +974,11 @@ def test_arize_does_not_overwrite_user_id_from_optional_params():
id="r2",
)
ArizeLogger.set_arize_attributes(span, kwargs, response_obj)
user_id_writes = [c.args[1] for c in span.set_attribute.call_args_list if c.args[0] == SpanAttributes.USER_ID]
user_id_writes = [
c.args[1]
for c in span.set_attribute.call_args_list
if c.args[0] == SpanAttributes.USER_ID
]
assert "from_metadata" not in user_id_writes
@ -983,7 +1048,9 @@ def test_arize_passthrough_bedrock_anthropic_normalization():
"complete_input_dict": {
"anthropic_version": "bedrock-2023-05-31",
"max_tokens": 64,
"messages": [{"role": "user", "content": "What is the capital of France?"}],
"messages": [
{"role": "user", "content": "What is the capital of France?"}
],
}
},
"standard_logging_object": {
@ -1001,13 +1068,19 @@ def test_arize_passthrough_bedrock_anthropic_normalization():
assert attrs[SpanAttributes.INPUT_VALUE] == "What is the capital of France?"
msg0 = f"{SpanAttributes.LLM_INPUT_MESSAGES}.0"
assert attrs[f"{msg0}.{MessageAttributes.MESSAGE_ROLE}"] == "user"
assert attrs[f"{msg0}.{MessageAttributes.MESSAGE_CONTENT}"] == "What is the capital of France?"
assert (
attrs[f"{msg0}.{MessageAttributes.MESSAGE_CONTENT}"]
== "What is the capital of France?"
)
# Output rendering (Anthropic content[].text)
assert attrs[SpanAttributes.OUTPUT_VALUE] == "The capital of France is Paris."
out0 = f"{SpanAttributes.LLM_OUTPUT_MESSAGES}.0"
assert attrs[f"{out0}.{MessageAttributes.MESSAGE_ROLE}"] == "assistant"
assert attrs[f"{out0}.{MessageAttributes.MESSAGE_CONTENT}"] == "The capital of France is Paris."
assert (
attrs[f"{out0}.{MessageAttributes.MESSAGE_CONTENT}"]
== "The capital of France is Paris."
)
# Token counts (Bedrock input_tokens/output_tokens) — extracted via
# coercion of the non-dict response.
@ -1016,7 +1089,9 @@ def test_arize_passthrough_bedrock_anthropic_normalization():
# Span kind defended even though the call_type is a passthrough variant.
span_kind_writes = [
c.args[1] for c in span.set_attribute.call_args_list if c.args[0] == SpanAttributes.OPENINFERENCE_SPAN_KIND
c.args[1]
for c in span.set_attribute.call_args_list
if c.args[0] == SpanAttributes.OPENINFERENCE_SPAN_KIND
]
assert span_kind_writes # at least one
assert all(v == "LLM" for v in span_kind_writes)
@ -1034,7 +1109,11 @@ def test_arize_passthrough_call_type_does_not_run_on_chat_completion():
span = MagicMock()
_maybe_normalize_passthrough(
span,
{"additional_args": {"complete_input_dict": {"messages": [{"role": "user", "content": "x"}]}}},
{
"additional_args": {
"complete_input_dict": {"messages": [{"role": "user", "content": "x"}]}
}
},
{"choices": [{"message": {"role": "assistant", "content": "y"}}]},
{"choices": [{"message": {"role": "assistant", "content": "y"}}]},
{"call_type": "completion"},
@ -1054,7 +1133,11 @@ def test_arize_passthrough_skipped_when_message_redaction_enabled():
span = MagicMock()
kwargs = {
"additional_args": {
"complete_input_dict": {"messages": [{"role": "user", "content": "Patient John Doe, SSN 123-45-6789"}]}
"complete_input_dict": {
"messages": [
{"role": "user", "content": "Patient John Doe, SSN 123-45-6789"}
]
}
},
# Enables redaction via the dynamic-param path inside
# should_redact_message_logging(), without touching globals.
@ -1128,7 +1211,9 @@ def test_arize_mcp_call_tool_result_does_not_break_attribute_setting():
"optional_params": {},
"litellm_params": {"custom_llm_provider": "mcp"},
}
response_obj = CallToolResult(content=[TextContent(type="text", text="sunny, 21C")], isError=False)
response_obj = CallToolResult(
content=[TextContent(type="text", text="sunny, 21C")], isError=False
)
ArizeLogger.set_arize_attributes(span, kwargs, response_obj)
@ -1210,7 +1295,9 @@ def test_arize_mcp_tool_span_renders_name_input_and_output():
from mcp.types import CallToolResult, TextContent
span = MagicMock()
response_obj = CallToolResult(content=[TextContent(type="text", text="sunny, 21C")], isError=False)
response_obj = CallToolResult(
content=[TextContent(type="text", text="sunny, 21C")], isError=False
)
ArizeLogger.set_arize_attributes(span, _mcp_kwargs(), response_obj)
@ -1249,7 +1336,9 @@ def test_arize_mcp_tool_span_respects_message_redaction():
from mcp.types import CallToolResult, TextContent
span = MagicMock()
response_obj = CallToolResult(content=[TextContent(type="text", text="SSN 123-45-6789")], isError=False)
response_obj = CallToolResult(
content=[TextContent(type="text", text="SSN 123-45-6789")], isError=False
)
ArizeLogger.set_arize_attributes(
span,

View file

@ -486,9 +486,7 @@ def test_a_request_without_trace_controls_stamps_none_of_them():
logger, exporter = _logger()
root_attrs, generation_attrs = _run_named_request(
logger,
exporter,
{"metadata": {"user_api_key_team_id": "t1", "tags": []}, "proxy_server_request": {"headers": {}}},
logger, exporter, {"metadata": {"user_api_key_team_id": "t1", "tags": []}, "proxy_server_request": {"headers": {}}}
)
assert set(TRACE_CONTROL_ATTRS).isdisjoint(root_attrs)

View file

@ -104,17 +104,19 @@ class TestLangfuseOtelIntegration:
mock_kwargs = {"test": "kwargs"}
mock_response = {"test": "response"}
with patch("litellm.integrations.arize._utils.set_attributes") as mock_set_attributes:
LangfuseOtelLogger.set_langfuse_otel_attributes(mock_span, mock_kwargs, mock_response)
with patch(
"litellm.integrations.arize._utils.set_attributes"
) as mock_set_attributes:
LangfuseOtelLogger.set_langfuse_otel_attributes(
mock_span, mock_kwargs, mock_response
)
mock_set_attributes.assert_called_once_with(
mock_span,
mock_kwargs,
mock_response,
LangfuseLLMObsOTELAttributes,
emit_session_and_user=False,
mock_span, mock_kwargs, mock_response, LangfuseLLMObsOTELAttributes, emit_session_and_user=False
)
mock_span.set_attribute.assert_any_call(
"langfuse.observation.type", "generation"
)
mock_span.set_attribute.assert_any_call("langfuse.observation.type", "generation")
def test_set_langfuse_environment_attribute(self):
"""Test that Langfuse environment is set correctly when environment variable is present."""
@ -123,11 +125,17 @@ class TestLangfuseOtelIntegration:
test_env = "staging"
with patch.dict(os.environ, {"LANGFUSE_TRACING_ENVIRONMENT": test_env}):
with patch("litellm.integrations.arize._utils.safe_set_attribute") as mock_safe_set_attribute:
LangfuseOtelLogger._set_langfuse_specific_attributes(mock_span, mock_kwargs, {})
with patch(
"litellm.integrations.arize._utils.safe_set_attribute"
) as mock_safe_set_attribute:
LangfuseOtelLogger._set_langfuse_specific_attributes(
mock_span, mock_kwargs, {}
)
# safe_set_attribute(span, key, value) → positional args
mock_safe_set_attribute.assert_called_once_with(mock_span, "langfuse.environment", test_env)
mock_safe_set_attribute.assert_called_once_with(
mock_span, "langfuse.environment", test_env
)
def test_set_langfuse_environment_attribute_prefers_dynamic_param(self):
"""Per-key/team langfuse_environment beats the deployment env var."""
@ -140,10 +148,18 @@ class TestLangfuseOtelIntegration:
self.attributes[key] = value
span = _RecordingSpan()
mock_kwargs = {"standard_callback_dynamic_params": {"langfuse_environment": "team-a-env"}}
mock_kwargs = {
"standard_callback_dynamic_params": {
"langfuse_environment": "team-a-env"
}
}
with patch.dict(os.environ, {"LANGFUSE_TRACING_ENVIRONMENT": "deployment-wide"}):
LangfuseOtelLogger._set_langfuse_specific_attributes(span, mock_kwargs, {})
with patch.dict(
os.environ, {"LANGFUSE_TRACING_ENVIRONMENT": "deployment-wide"}
):
LangfuseOtelLogger._set_langfuse_specific_attributes(
span, mock_kwargs, {}
)
assert span.attributes["langfuse.environment"] == "team-a-env"
@ -207,8 +223,12 @@ class TestLangfuseOtelIntegration:
kwargs = {"litellm_params": {"metadata": metadata}}
# Capture calls to safe_set_attribute
with patch("litellm.integrations.arize._utils.safe_set_attribute") as mock_safe_set_attribute:
LangfuseOtelLogger._set_langfuse_specific_attributes(MagicMock(), kwargs, None)
with patch(
"litellm.integrations.arize._utils.safe_set_attribute"
) as mock_safe_set_attribute:
LangfuseOtelLogger._set_langfuse_specific_attributes(
MagicMock(), kwargs, None
)
# Build expected calls manually for clarity
from litellm.types.integrations.langfuse_otel import LangfuseSpanAttributes
@ -229,7 +249,9 @@ class TestLangfuseOtelIntegration:
LangfuseSpanAttributes.TRACE_METADATA.value: json.dumps({"k": "v"}),
LangfuseSpanAttributes.RELEASE.value: "rel-1",
LangfuseSpanAttributes.EXISTING_TRACE_ID.value: "existing-id",
LangfuseSpanAttributes.UPDATE_TRACE_KEYS.value: json.dumps(["key1", "key2"]),
LangfuseSpanAttributes.UPDATE_TRACE_KEYS.value: json.dumps(
["key1", "key2"]
),
LangfuseSpanAttributes.DEBUG_LANGFUSE.value: True,
}
@ -239,7 +261,9 @@ class TestLangfuseOtelIntegration:
for call in mock_safe_set_attribute.call_args_list
}
assert actual == expected, "Mismatch between expected and actual OTEL attribute mapping."
assert (
actual == expected
), "Mismatch between expected and actual OTEL attribute mapping."
@pytest.mark.parametrize(
"metadata, expected_version",
@ -264,10 +288,16 @@ class TestLangfuseOtelIntegration:
def test_version_emitted_on_langfuse_v4_key(self, metadata, expected_version):
kwargs = {"litellm_params": {"metadata": {"trace_release": "rel-9", **metadata}}}
with patch("litellm.integrations.arize._utils.safe_set_attribute") as mock_safe_set_attribute:
LangfuseOtelLogger._set_langfuse_specific_attributes(MagicMock(), kwargs, None)
with patch(
"litellm.integrations.arize._utils.safe_set_attribute"
) as mock_safe_set_attribute:
LangfuseOtelLogger._set_langfuse_specific_attributes(
MagicMock(), kwargs, None
)
emitted = {call.args[1]: call.args[2] for call in mock_safe_set_attribute.call_args_list}
emitted = {
call.args[1]: call.args[2] for call in mock_safe_set_attribute.call_args_list
}
if expected_version is None:
assert "langfuse.version" not in emitted
@ -305,8 +335,12 @@ class TestLangfuseOtelIntegration:
"messages": [{"role": "user", "content": "What's the weather in Tokyo?"}],
}
with patch("litellm.integrations.arize._utils.safe_set_attribute") as mock_safe_set_attribute:
LangfuseOtelLogger._set_langfuse_specific_attributes(MagicMock(), kwargs, response_obj)
with patch(
"litellm.integrations.arize._utils.safe_set_attribute"
) as mock_safe_set_attribute:
LangfuseOtelLogger._set_langfuse_specific_attributes(
MagicMock(), kwargs, response_obj
)
expect_output = {
LangfuseSpanAttributes.OBSERVATION_INPUT.value: [
@ -319,9 +353,14 @@ class TestLangfuseOtelIntegration:
}
# Flatten the actual calls into {key: value}
actual = {call.args[1]: json.loads(call.args[2]) for call in mock_safe_set_attribute.call_args_list}
actual = {
call.args[1]: json.loads(call.args[2])
for call in mock_safe_set_attribute.call_args_list
}
assert actual == expect_output, "Mismatch in observation input/output OTEL attributes."
assert (
actual == expect_output
), "Mismatch in observation input/output OTEL attributes."
def test_set_langfuse_specific_attributes_with_tool_calls(self):
"""Test that _set_langfuse_specific_attributes correctly sets observation.output with tool calls in Langfuse format."""
@ -345,7 +384,9 @@ class TestLangfuseOtelIntegration:
"content": None,
"tool_calls": [
ChatCompletionMessageToolCall(
function=Function(arguments='{"location":"Tokyo"}', name="get_weather"),
function=Function(
arguments='{"location":"Tokyo"}', name="get_weather"
),
id="call_123",
type="function",
)
@ -355,8 +396,12 @@ class TestLangfuseOtelIntegration:
],
)
with patch("litellm.integrations.arize._utils.safe_set_attribute") as mock_safe_set_attribute:
LangfuseOtelLogger._set_langfuse_specific_attributes(MagicMock(), {}, response_obj)
with patch(
"litellm.integrations.arize._utils.safe_set_attribute"
) as mock_safe_set_attribute:
LangfuseOtelLogger._set_langfuse_specific_attributes(
MagicMock(), {}, response_obj
)
expected = {
LangfuseSpanAttributes.OBSERVATION_OUTPUT.value: [
@ -371,8 +416,13 @@ class TestLangfuseOtelIntegration:
}
# Flatten the actual calls into {key: value}
actual = {call.args[1]: json.loads(call.args[2]) for call in mock_safe_set_attribute.call_args_list}
assert actual == expected, "Mismatch in observation output OTEL attribute for tool calls."
actual = {
call.args[1]: json.loads(call.args[2])
for call in mock_safe_set_attribute.call_args_list
}
assert (
actual == expected
), "Mismatch in observation output OTEL attribute for tool calls."
def test_construct_dynamic_otel_headers_with_langfuse_keys(self):
"""Test that construct_dynamic_otel_headers creates proper auth headers when langfuse keys are provided."""
@ -523,7 +573,9 @@ class TestLangfuseOtelKeyDynamicConfig:
logger = LangfuseOtelLogger()
assert logger.OTEL_EXPORTER == "console"
tracer = logger.get_tracer_to_use_for_request({"standard_callback_dynamic_params": self._dynamic_params()})
tracer = logger.get_tracer_to_use_for_request(
{"standard_callback_dynamic_params": self._dynamic_params()}
)
assert tracer is not logger.tracer
assert len(logger._tracer_provider_cache) == 1
@ -584,7 +636,9 @@ class TestLangfuseOtelKeyDynamicConfig:
with self._clean_env():
logger = LangfuseOtelLogger()
with patch.object(otel_module.verbose_logger, "debug", side_effect=_spy):
logger.get_tracer_to_use_for_request({"standard_callback_dynamic_params": self._dynamic_params()})
logger.get_tracer_to_use_for_request(
{"standard_callback_dynamic_params": self._dynamic_params()}
)
logged = "\n".join(recorded_arguments)
assert "initializing span processor" in logged
@ -644,8 +698,12 @@ class TestLangfuseOtelResponsesAPI:
LangfuseLLMObsOTELAttributes,
)
with patch("litellm.integrations.arize._utils.set_attributes") as mock_set_attributes:
with patch("litellm.integrations.arize._utils.safe_set_attribute") as mock_safe_set_attribute:
with patch(
"litellm.integrations.arize._utils.set_attributes"
) as mock_set_attributes:
with patch(
"litellm.integrations.arize._utils.safe_set_attribute"
) as mock_safe_set_attribute:
logger = LangfuseOtelLogger()
logger.set_langfuse_otel_attributes(mock_span, kwargs, mock_response)
@ -658,7 +716,9 @@ class TestLangfuseOtelResponsesAPI:
mock_safe_set_attribute.assert_any_call(
mock_span, "langfuse.generation.name", "responses_test_generation"
)
mock_safe_set_attribute.assert_any_call(mock_span, "langfuse.trace.name", "responses_api_trace")
mock_safe_set_attribute.assert_any_call(
mock_span, "langfuse.trace.name", "responses_api_trace"
)
def test_responses_api_metadata_extraction(self):
"""Test that metadata is correctly extracted from ResponsesAPI kwargs."""
@ -708,7 +768,9 @@ class TestLangfuseOtelResponsesAPI:
mock_span = MagicMock()
with patch("litellm.integrations.arize._utils.safe_set_attribute") as mock_safe_set_attribute:
with patch(
"litellm.integrations.arize._utils.safe_set_attribute"
) as mock_safe_set_attribute:
LangfuseOtelLogger._set_langfuse_specific_attributes(mock_span, kwargs, {})
# Verify specific attributes were set
@ -750,12 +812,11 @@ class TestLangfuseOtelResponsesAPI:
def test_responses_api_with_output(self):
"""Test Langfuse OTEL logger with Responses API output (reasoning + message)."""
from openai.types.responses import (
ResponseReasoningItem,
ResponseOutputMessage,
ResponseOutputText,
ResponseReasoningItem,
)
from openai.types.responses.response_reasoning_item import Summary
from litellm.types.integrations.langfuse_otel import LangfuseSpanAttributes
# Create Responses API response with reasoning and message
@ -791,15 +852,21 @@ class TestLangfuseOtelResponsesAPI:
kwargs = {
"call_type": "responses",
"messages": [{"role": "user", "content": "What's the weather in San Francisco?"}],
"messages": [
{"role": "user", "content": "What's the weather in San Francisco?"}
],
"model": "gpt-4o",
"optional_params": {},
}
mock_span = MagicMock()
with patch("litellm.integrations.arize._utils.safe_set_attribute") as mock_safe_set_attribute:
LangfuseOtelLogger._set_langfuse_specific_attributes(mock_span, kwargs, response_obj)
with patch(
"litellm.integrations.arize._utils.safe_set_attribute"
) as mock_safe_set_attribute:
LangfuseOtelLogger._set_langfuse_specific_attributes(
mock_span, kwargs, response_obj
)
# Verify observation output was set
output_calls = [
@ -818,17 +885,22 @@ class TestLangfuseOtelResponsesAPI:
# Verify reasoning summary
assert output_data[0]["role"] == "reasoning_summary"
assert output_data[0]["content"] == "Let me analyze this problem step by step..."
assert (
output_data[0]["content"]
== "Let me analyze this problem step by step..."
)
# Verify message
assert output_data[1]["role"] == "assistant"
assert output_data[1]["content"] == "The weather in San Francisco is sunny, 20°C."
assert (
output_data[1]["content"]
== "The weather in San Francisco is sunny, 20°C."
)
def test_responses_api_with_function_calls(self):
"""Test Langfuse OTEL logger with Responses API function_call output."""
from openai.types.responses import ResponseFunctionToolCall
from litellm.types.integrations.langfuse_otel import LangfuseSpanAttributes
from openai.types.responses import ResponseFunctionToolCall
# Create Responses API response with function call
response_obj = ResponsesAPIResponse(
@ -848,15 +920,21 @@ class TestLangfuseOtelResponsesAPI:
kwargs = {
"call_type": "responses",
"messages": [{"role": "user", "content": "What's the weather in San Francisco?"}],
"messages": [
{"role": "user", "content": "What's the weather in San Francisco?"}
],
"model": "gpt-4o",
"optional_params": {},
}
mock_span = MagicMock()
with patch("litellm.integrations.arize._utils.safe_set_attribute") as mock_safe_set_attribute:
LangfuseOtelLogger._set_langfuse_specific_attributes(mock_span, kwargs, response_obj)
with patch(
"litellm.integrations.arize._utils.safe_set_attribute"
) as mock_safe_set_attribute:
LangfuseOtelLogger._set_langfuse_specific_attributes(
mock_span, kwargs, response_obj
)
# Verify observation output was set
output_calls = [
@ -911,11 +989,9 @@ class TestLangfuseOtelResponsesAPI:
mock_span = MagicMock()
with (
patch( # test-quality-ok: the span attribute sink is the observable boundary; sibling tests in this class stub the same seam
"litellm.integrations.arize._utils.safe_set_attribute"
) as mock_safe_set_attribute
):
with patch( # test-quality-ok: the span attribute sink is the observable boundary; sibling tests in this class stub the same seam
"litellm.integrations.arize._utils.safe_set_attribute"
) as mock_safe_set_attribute:
LangfuseOtelLogger._set_langfuse_specific_attributes(mock_span, kwargs, response_obj)
output_calls = [