diff --git a/litellm/integrations/arize/_utils.py b/litellm/integrations/arize/_utils.py index ce0b6e2ed48..9594bfc268f 100644 --- a/litellm/integrations/arize/_utils.py +++ b/litellm/integrations/arize/_utils.py @@ -9,6 +9,7 @@ from litellm.integrations.opentelemetry_utils.base_otel_llm_obs_attributes impor BaseLLMObsOTELAttributes, safe_set_attribute, ) +from litellm.integrations.otel.model.utils import as_str_mapping from litellm.litellm_core_utils.redact_messages import ( should_redact_message_logging, ) @@ -477,9 +478,7 @@ def set_attributes( # Additive emitters. Each is independently guarded so a failure can never # blank the attributes set by the main try-block above. New attributes are # written under new keys; existing attributes are not overwritten. - from litellm.integrations.otel.model.utils import as_str_mapping - - slp: Final = as_str_mapping(kwargs.get("standard_logging_object")) + slp: Final = kwargs.get("standard_logging_object") if emit_session_and_user: _safe_emit("session/user attrs", _set_session_and_user_attrs, span, kwargs, slp) _safe_emit("request context attrs", _set_request_context_attrs, span, slp) @@ -880,8 +879,6 @@ def _set_session_and_user_attrs(span: "Span", kwargs: dict, standard_logging_pay def _set_request_context_attrs(span: "Span", standard_logging_payload: object) -> None: """Emit `litellm.trace_id` / team / key context when source data exists.""" - from litellm.integrations.otel.model.utils import as_str_mapping - payload: Final = as_str_mapping(standard_logging_payload) if payload is None: return diff --git a/litellm/integrations/langfuse/langfuse_otel.py b/litellm/integrations/langfuse/langfuse_otel.py index e7181fa37aa..91771995d1d 100644 --- a/litellm/integrations/langfuse/langfuse_otel.py +++ b/litellm/integrations/langfuse/langfuse_otel.py @@ -6,10 +6,12 @@ from typing import TYPE_CHECKING, Any, Final, Optional from litellm._logging import verbose_logger from litellm.integrations.arize import _utils +from litellm.integrations.arize._utils import safe_set_attribute from litellm.integrations.langfuse.langfuse_otel_attributes import ( LangfuseLLMObsOTELAttributes, ) from litellm.integrations.opentelemetry import OpenTelemetry, OpenTelemetryConfig +from litellm.integrations.otel.model.utils import as_str, as_str_mapping from litellm.litellm_core_utils.safe_json_loads import safe_json_loads from litellm.types.integrations.langfuse_otel import ( LangfuseSpanAttributes, @@ -258,9 +260,6 @@ class LangfuseOtelLogger(OpenTelemetry): @staticmethod def _set_trace_user_attribute(span: Span, kwargs: dict[str, object]) -> None: - from litellm.integrations.arize._utils import safe_set_attribute - from litellm.integrations.otel.model.utils import as_str, as_str_mapping - slp: Final = as_str_mapping(kwargs.get("standard_logging_object")) slp_metadata: Final = as_str_mapping(slp.get("metadata")) if slp is not None else None if slp is None or slp_metadata is None: diff --git a/tests/test_litellm/integrations/arize/test_arize_utils.py b/tests/test_litellm/integrations/arize/test_arize_utils.py index 488609eb496..8ce2f6a4202 100644 --- a/tests/test_litellm/integrations/arize/test_arize_utils.py +++ b/tests/test_litellm/integrations/arize/test_arize_utils.py @@ -1,6 +1,9 @@ -# Adds the grandparent directory to sys.path to allow importing project modules -import asyncio import json +from typing import Optional + +# Adds the grandparent directory to sys.path to allow importing project modules + +import asyncio import pytest @@ -67,7 +70,9 @@ def test_arize_set_attributes(): # Simulated LLM response object response_obj = ModelResponse( usage={"total_tokens": 100, "completion_tokens": 60, "prompt_tokens": 40}, - choices=[Choices(message={"role": "assistant", "content": "Basic Response Content"})], + choices=[ + Choices(message={"role": "assistant", "content": "Basic Response Content"}) + ], model="gpt-4o", id="chatcmpl-ID", ) @@ -84,7 +89,9 @@ def test_arize_set_attributes(): assert span.set_attribute.call_count == 26 # Metadata attached to the span - span.set_attribute.assert_any_call(SpanAttributes.METADATA, json.dumps({"key_1": "value_1", "key_2": None})) + span.set_attribute.assert_any_call( + SpanAttributes.METADATA, json.dumps({"key_1": "value_1", "key_2": None}) + ) # Basic LLM information span.set_attribute.assert_any_call(SpanAttributes.LLM_MODEL_NAME, "gpt-4o") @@ -107,12 +114,16 @@ def test_arize_set_attributes(): span.set_attribute.assert_any_call(SpanAttributes.OPENINFERENCE_SPAN_KIND, "LLM") # And TOOL must never be written for an LLM chat completion call. span_kind_writes = [ - c.args[1] for c in span.set_attribute.call_args_list if c.args[0] == SpanAttributes.OPENINFERENCE_SPAN_KIND + c.args[1] + for c in span.set_attribute.call_args_list + if c.args[0] == SpanAttributes.OPENINFERENCE_SPAN_KIND ] assert "TOOL" not in span_kind_writes # Request message content and metadata - span.set_attribute.assert_any_call(SpanAttributes.INPUT_VALUE, "Basic Request Content") + span.set_attribute.assert_any_call( + SpanAttributes.INPUT_VALUE, "Basic Request Content" + ) span.set_attribute.assert_any_call( f"{SpanAttributes.LLM_INPUT_MESSAGES}.0.{MessageAttributes.MESSAGE_ROLE}", "user", @@ -123,7 +134,9 @@ def test_arize_set_attributes(): ) # Tool call definitions and function names - span.set_attribute.assert_any_call(f"{SpanAttributes.LLM_TOOLS}.0.name", "get_weather") + span.set_attribute.assert_any_call( + f"{SpanAttributes.LLM_TOOLS}.0.name", "get_weather" + ) span.set_attribute.assert_any_call( f"{SpanAttributes.LLM_TOOLS}.0.description", "Fetches weather details.", @@ -133,20 +146,26 @@ def test_arize_set_attributes(): json.dumps( { "type": "object", - "properties": {"location": {"type": "string", "description": "City name"}}, + "properties": { + "location": {"type": "string", "description": "City name"} + }, "required": ["location"], } ), ) # Invocation parameters - span.set_attribute.assert_any_call(SpanAttributes.LLM_INVOCATION_PARAMETERS, '{"user": "test_user"}') + span.set_attribute.assert_any_call( + SpanAttributes.LLM_INVOCATION_PARAMETERS, '{"user": "test_user"}' + ) # User ID span.set_attribute.assert_any_call(SpanAttributes.USER_ID, "test_user") # Output message content - span.set_attribute.assert_any_call(SpanAttributes.OUTPUT_VALUE, "Basic Response Content") + span.set_attribute.assert_any_call( + SpanAttributes.OUTPUT_VALUE, "Basic Response Content" + ) span.set_attribute.assert_any_call( f"{SpanAttributes.LLM_OUTPUT_MESSAGES}.0.{MessageAttributes.MESSAGE_ROLE}", "assistant", @@ -168,20 +187,18 @@ def test_arize_set_attributes_responses_api(): Verifies that multiple output types are correctly handled. """ from unittest.mock import MagicMock - + from litellm.types.llms.openai import ( + ResponsesAPIResponse, + ResponseAPIUsage, + OutputTokensDetails, + ) from openai.types.responses import ( + ResponseReasoningItem, ResponseOutputMessage, ResponseOutputText, - ResponseReasoningItem, ) from openai.types.responses.response_reasoning_item import Summary - from litellm.types.llms.openai import ( - OutputTokensDetails, - ResponseAPIUsage, - ResponsesAPIResponse, - ) - span = MagicMock() # Mocked tracing span to test attribute setting # Construct kwargs to simulate a real LLM request scenario @@ -211,7 +228,9 @@ def test_arize_set_attributes_responses_api(): ResponseReasoningItem( id="reasoning-001", type="reasoning", - summary=[Summary(text="First, I need to analyze...", type="summary_text")], + summary=[ + Summary(text="First, I need to analyze...", type="summary_text") + ], ), ResponseOutputMessage( id="msg-001", @@ -258,7 +277,9 @@ def test_arize_set_attributes_responses_api(): span.set_attribute.assert_any_call(SpanAttributes.LLM_TOKEN_COUNT_TOTAL, 370) span.set_attribute.assert_any_call(SpanAttributes.LLM_TOKEN_COUNT_COMPLETION, 250) span.set_attribute.assert_any_call(SpanAttributes.LLM_TOKEN_COUNT_PROMPT, 120) - span.set_attribute.assert_any_call(SpanAttributes.LLM_TOKEN_COUNT_COMPLETION_DETAILS_REASONING, 180) + span.set_attribute.assert_any_call( + SpanAttributes.LLM_TOKEN_COUNT_COMPLETION_DETAILS_REASONING, 180 + ) def test_set_usage_outputs_pydantic_completion_usage(): @@ -306,7 +327,9 @@ def test_set_usage_outputs_pydantic_completion_usage(): span.set_attribute.assert_any_call(SpanAttributes.LLM_TOKEN_COUNT_PROMPT, 40) span.set_attribute.assert_any_call(SpanAttributes.LLM_TOKEN_COUNT_COMPLETION, 60) # reasoning_tokens for chat completions live in completion_tokens_details - span.set_attribute.assert_any_call(SpanAttributes.LLM_TOKEN_COUNT_COMPLETION_DETAILS_REASONING, 25) + span.set_attribute.assert_any_call( + SpanAttributes.LLM_TOKEN_COUNT_COMPLETION_DETAILS_REASONING, 25 + ) def test_set_usage_outputs_pydantic_response_api_usage(): @@ -339,7 +362,9 @@ def test_set_usage_outputs_pydantic_response_api_usage(): span.set_attribute.assert_any_call(SpanAttributes.LLM_TOKEN_COUNT_TOTAL, 370) span.set_attribute.assert_any_call(SpanAttributes.LLM_TOKEN_COUNT_PROMPT, 120) span.set_attribute.assert_any_call(SpanAttributes.LLM_TOKEN_COUNT_COMPLETION, 250) - span.set_attribute.assert_any_call(SpanAttributes.LLM_TOKEN_COUNT_COMPLETION_DETAILS_REASONING, 180) + span.set_attribute.assert_any_call( + SpanAttributes.LLM_TOKEN_COUNT_COMPLETION_DETAILS_REASONING, 180 + ) class TestArizeLogger(CustomLogger): @@ -350,12 +375,16 @@ class TestArizeLogger(CustomLogger): def __init__(self, *args, **kwargs): super().__init__(*args, **kwargs) - self.standard_callback_dynamic_params: StandardCallbackDynamicParams | None = None + self.standard_callback_dynamic_params: Optional[ + StandardCallbackDynamicParams + ] = None async def async_log_success_event(self, kwargs, response_obj, start_time, end_time): # Capture dynamic params and print them for verification print("logged kwargs", json.dumps(kwargs, indent=4, default=str)) - self.standard_callback_dynamic_params = kwargs.get("standard_callback_dynamic_params") + self.standard_callback_dynamic_params = kwargs.get( + "standard_callback_dynamic_params" + ) @pytest.mark.asyncio @@ -381,8 +410,14 @@ async def test_arize_dynamic_params(): # Assert dynamic parameters were received in the callback assert test_arize_logger.standard_callback_dynamic_params is not None - assert test_arize_logger.standard_callback_dynamic_params.get("arize_api_key") == "test_api_key_dynamic" - assert test_arize_logger.standard_callback_dynamic_params.get("arize_space_key") == "test_space_key_dynamic" + assert ( + test_arize_logger.standard_callback_dynamic_params.get("arize_api_key") + == "test_api_key_dynamic" + ) + assert ( + test_arize_logger.standard_callback_dynamic_params.get("arize_space_key") + == "test_space_key_dynamic" + ) def test_construct_dynamic_arize_headers(): @@ -393,7 +428,9 @@ def test_construct_dynamic_arize_headers(): from litellm.types.utils import StandardCallbackDynamicParams # Test with all parameters present - dynamic_params_full = StandardCallbackDynamicParams(arize_api_key="test_api_key", arize_space_id="test_space_id") + dynamic_params_full = StandardCallbackDynamicParams( + arize_api_key="test_api_key", arize_space_id="test_space_id" + ) arize_logger = ArizeLogger() headers = arize_logger.construct_dynamic_otel_headers(dynamic_params_full) @@ -401,7 +438,9 @@ def test_construct_dynamic_arize_headers(): assert headers == expected_headers # Test with only space_id - dynamic_params_space_id_only = StandardCallbackDynamicParams(arize_space_id="test_space_id") + dynamic_params_space_id_only = StandardCallbackDynamicParams( + arize_space_id="test_space_id" + ) headers = arize_logger.construct_dynamic_otel_headers(dynamic_params_space_id_only) expected_headers = {"arize-space-id": "test_space_id"} @@ -417,7 +456,9 @@ def test_construct_dynamic_arize_headers(): dynamic_params_space_key_and_api_key = StandardCallbackDynamicParams( arize_space_key="test_space_key", arize_api_key="test_api_key" ) - headers = arize_logger.construct_dynamic_otel_headers(dynamic_params_space_key_and_api_key) + headers = arize_logger.construct_dynamic_otel_headers( + dynamic_params_space_key_and_api_key + ) expected_headers = {"arize-space-id": "test_space_key", "api_key": "test_api_key"} @@ -487,7 +528,9 @@ def test_arize_emits_no_cache_tokens_when_absent(): from litellm.integrations.arize._utils import _set_usage_outputs span = MagicMock() - response_obj = {"usage": {"total_tokens": 10, "completion_tokens": 4, "prompt_tokens": 6}} + response_obj = { + "usage": {"total_tokens": 10, "completion_tokens": 4, "prompt_tokens": 6} + } _set_usage_outputs(span, response_obj, SpanAttributes) attrs = _collect_calls(span) assert SpanAttributes.LLM_TOKEN_COUNT_PROMPT_DETAILS_CACHE_READ not in attrs @@ -499,8 +542,14 @@ def test_passthrough_call_type_resolves_to_llm_span_kind(): from litellm.integrations._types.open_inference import OpenInferenceSpanKindValues from litellm.integrations.arize._utils import _infer_open_inference_span_kind - assert _infer_open_inference_span_kind("allm_passthrough_route") == OpenInferenceSpanKindValues.LLM.value - assert _infer_open_inference_span_kind("llm_passthrough_route") == OpenInferenceSpanKindValues.LLM.value + assert ( + _infer_open_inference_span_kind("allm_passthrough_route") + == OpenInferenceSpanKindValues.LLM.value + ) + assert ( + _infer_open_inference_span_kind("llm_passthrough_route") + == OpenInferenceSpanKindValues.LLM.value + ) def test_arize_chat_completion_with_tools_stays_llm_span_kind(): @@ -556,7 +605,9 @@ def test_arize_chat_completion_with_tools_stays_llm_span_kind(): ArizeLogger.set_arize_attributes(span, kwargs, response_obj) span_kind_writes = [ - c.args[1] for c in span.set_attribute.call_args_list if c.args[0] == SpanAttributes.OPENINFERENCE_SPAN_KIND + c.args[1] + for c in span.set_attribute.call_args_list + if c.args[0] == SpanAttributes.OPENINFERENCE_SPAN_KIND ] assert span_kind_writes, "span.kind must be written" assert all(v == "LLM" for v in span_kind_writes) @@ -608,8 +659,13 @@ def test_arize_emits_assistant_tool_calls_on_output_message(): attrs = _collect_calls(span) base = f"{SpanAttributes.LLM_OUTPUT_MESSAGES}.0.{MessageAttributes.MESSAGE_TOOL_CALLS}.0" assert attrs[f"{base}.{ToolCallAttributes.TOOL_CALL_ID}"] == "call_abc" - assert attrs[f"{base}.{ToolCallAttributes.TOOL_CALL_FUNCTION_NAME}"] == "get_weather" - assert attrs[f"{base}.{ToolCallAttributes.TOOL_CALL_FUNCTION_ARGUMENTS_JSON}"] == '{"location": "SF"}' + assert ( + attrs[f"{base}.{ToolCallAttributes.TOOL_CALL_FUNCTION_NAME}"] == "get_weather" + ) + assert ( + attrs[f"{base}.{ToolCallAttributes.TOOL_CALL_FUNCTION_ARGUMENTS_JSON}"] + == '{"location": "SF"}' + ) def test_arize_output_value_falls_back_to_tool_calls_summary(): @@ -762,7 +818,9 @@ def test_arize_emits_tool_call_id_and_name_on_input_tool_message(): assert attrs[f"{assistant_base}.{ToolCallAttributes.TOOL_CALL_ID}"] == "call_abc" # Tool message at index 2 tool_prefix = f"{SpanAttributes.LLM_INPUT_MESSAGES}.2" - assert attrs[f"{tool_prefix}.{MessageAttributes.MESSAGE_TOOL_CALL_ID}"] == "call_abc" + assert ( + attrs[f"{tool_prefix}.{MessageAttributes.MESSAGE_TOOL_CALL_ID}"] == "call_abc" + ) assert attrs[f"{tool_prefix}.{MessageAttributes.MESSAGE_NAME}"] == "get_weather" @@ -808,7 +866,10 @@ def test_arize_emits_multimodal_input_contents(): assert attrs[f"{base}.0.message_content.type"] == "text" assert attrs[f"{base}.0.message_content.text"] == "What is in this image?" assert attrs[f"{base}.1.message_content.type"] == "image" - assert attrs[f"{base}.1.message_content.image.image.url"] == "https://example.com/cat.png" + assert ( + attrs[f"{base}.1.message_content.image.image.url"] + == "https://example.com/cat.png" + ) def test_arize_emits_session_and_user_attrs_from_metadata(): @@ -913,7 +974,11 @@ def test_arize_does_not_overwrite_user_id_from_optional_params(): id="r2", ) ArizeLogger.set_arize_attributes(span, kwargs, response_obj) - user_id_writes = [c.args[1] for c in span.set_attribute.call_args_list if c.args[0] == SpanAttributes.USER_ID] + user_id_writes = [ + c.args[1] + for c in span.set_attribute.call_args_list + if c.args[0] == SpanAttributes.USER_ID + ] assert "from_metadata" not in user_id_writes @@ -983,7 +1048,9 @@ def test_arize_passthrough_bedrock_anthropic_normalization(): "complete_input_dict": { "anthropic_version": "bedrock-2023-05-31", "max_tokens": 64, - "messages": [{"role": "user", "content": "What is the capital of France?"}], + "messages": [ + {"role": "user", "content": "What is the capital of France?"} + ], } }, "standard_logging_object": { @@ -1001,13 +1068,19 @@ def test_arize_passthrough_bedrock_anthropic_normalization(): assert attrs[SpanAttributes.INPUT_VALUE] == "What is the capital of France?" msg0 = f"{SpanAttributes.LLM_INPUT_MESSAGES}.0" assert attrs[f"{msg0}.{MessageAttributes.MESSAGE_ROLE}"] == "user" - assert attrs[f"{msg0}.{MessageAttributes.MESSAGE_CONTENT}"] == "What is the capital of France?" + assert ( + attrs[f"{msg0}.{MessageAttributes.MESSAGE_CONTENT}"] + == "What is the capital of France?" + ) # Output rendering (Anthropic content[].text) assert attrs[SpanAttributes.OUTPUT_VALUE] == "The capital of France is Paris." out0 = f"{SpanAttributes.LLM_OUTPUT_MESSAGES}.0" assert attrs[f"{out0}.{MessageAttributes.MESSAGE_ROLE}"] == "assistant" - assert attrs[f"{out0}.{MessageAttributes.MESSAGE_CONTENT}"] == "The capital of France is Paris." + assert ( + attrs[f"{out0}.{MessageAttributes.MESSAGE_CONTENT}"] + == "The capital of France is Paris." + ) # Token counts (Bedrock input_tokens/output_tokens) — extracted via # coercion of the non-dict response. @@ -1016,7 +1089,9 @@ def test_arize_passthrough_bedrock_anthropic_normalization(): # Span kind defended even though the call_type is a passthrough variant. span_kind_writes = [ - c.args[1] for c in span.set_attribute.call_args_list if c.args[0] == SpanAttributes.OPENINFERENCE_SPAN_KIND + c.args[1] + for c in span.set_attribute.call_args_list + if c.args[0] == SpanAttributes.OPENINFERENCE_SPAN_KIND ] assert span_kind_writes # at least one assert all(v == "LLM" for v in span_kind_writes) @@ -1034,7 +1109,11 @@ def test_arize_passthrough_call_type_does_not_run_on_chat_completion(): span = MagicMock() _maybe_normalize_passthrough( span, - {"additional_args": {"complete_input_dict": {"messages": [{"role": "user", "content": "x"}]}}}, + { + "additional_args": { + "complete_input_dict": {"messages": [{"role": "user", "content": "x"}]} + } + }, {"choices": [{"message": {"role": "assistant", "content": "y"}}]}, {"choices": [{"message": {"role": "assistant", "content": "y"}}]}, {"call_type": "completion"}, @@ -1054,7 +1133,11 @@ def test_arize_passthrough_skipped_when_message_redaction_enabled(): span = MagicMock() kwargs = { "additional_args": { - "complete_input_dict": {"messages": [{"role": "user", "content": "Patient John Doe, SSN 123-45-6789"}]} + "complete_input_dict": { + "messages": [ + {"role": "user", "content": "Patient John Doe, SSN 123-45-6789"} + ] + } }, # Enables redaction via the dynamic-param path inside # should_redact_message_logging(), without touching globals. @@ -1128,7 +1211,9 @@ def test_arize_mcp_call_tool_result_does_not_break_attribute_setting(): "optional_params": {}, "litellm_params": {"custom_llm_provider": "mcp"}, } - response_obj = CallToolResult(content=[TextContent(type="text", text="sunny, 21C")], isError=False) + response_obj = CallToolResult( + content=[TextContent(type="text", text="sunny, 21C")], isError=False + ) ArizeLogger.set_arize_attributes(span, kwargs, response_obj) @@ -1210,7 +1295,9 @@ def test_arize_mcp_tool_span_renders_name_input_and_output(): from mcp.types import CallToolResult, TextContent span = MagicMock() - response_obj = CallToolResult(content=[TextContent(type="text", text="sunny, 21C")], isError=False) + response_obj = CallToolResult( + content=[TextContent(type="text", text="sunny, 21C")], isError=False + ) ArizeLogger.set_arize_attributes(span, _mcp_kwargs(), response_obj) @@ -1249,7 +1336,9 @@ def test_arize_mcp_tool_span_respects_message_redaction(): from mcp.types import CallToolResult, TextContent span = MagicMock() - response_obj = CallToolResult(content=[TextContent(type="text", text="SSN 123-45-6789")], isError=False) + response_obj = CallToolResult( + content=[TextContent(type="text", text="SSN 123-45-6789")], isError=False + ) ArizeLogger.set_arize_attributes( span, diff --git a/tests/test_litellm/integrations/otel/test_langfuse_logger.py b/tests/test_litellm/integrations/otel/test_langfuse_logger.py index a0f4e26064c..bf0afec6f54 100644 --- a/tests/test_litellm/integrations/otel/test_langfuse_logger.py +++ b/tests/test_litellm/integrations/otel/test_langfuse_logger.py @@ -486,9 +486,7 @@ def test_a_request_without_trace_controls_stamps_none_of_them(): logger, exporter = _logger() root_attrs, generation_attrs = _run_named_request( - logger, - exporter, - {"metadata": {"user_api_key_team_id": "t1", "tags": []}, "proxy_server_request": {"headers": {}}}, + logger, exporter, {"metadata": {"user_api_key_team_id": "t1", "tags": []}, "proxy_server_request": {"headers": {}}} ) assert set(TRACE_CONTROL_ATTRS).isdisjoint(root_attrs) diff --git a/tests/test_litellm/integrations/test_langfuse_otel.py b/tests/test_litellm/integrations/test_langfuse_otel.py index 2dbc0c209fd..e34670c8488 100644 --- a/tests/test_litellm/integrations/test_langfuse_otel.py +++ b/tests/test_litellm/integrations/test_langfuse_otel.py @@ -104,17 +104,19 @@ class TestLangfuseOtelIntegration: mock_kwargs = {"test": "kwargs"} mock_response = {"test": "response"} - with patch("litellm.integrations.arize._utils.set_attributes") as mock_set_attributes: - LangfuseOtelLogger.set_langfuse_otel_attributes(mock_span, mock_kwargs, mock_response) + with patch( + "litellm.integrations.arize._utils.set_attributes" + ) as mock_set_attributes: + LangfuseOtelLogger.set_langfuse_otel_attributes( + mock_span, mock_kwargs, mock_response + ) mock_set_attributes.assert_called_once_with( - mock_span, - mock_kwargs, - mock_response, - LangfuseLLMObsOTELAttributes, - emit_session_and_user=False, + mock_span, mock_kwargs, mock_response, LangfuseLLMObsOTELAttributes, emit_session_and_user=False + ) + mock_span.set_attribute.assert_any_call( + "langfuse.observation.type", "generation" ) - mock_span.set_attribute.assert_any_call("langfuse.observation.type", "generation") def test_set_langfuse_environment_attribute(self): """Test that Langfuse environment is set correctly when environment variable is present.""" @@ -123,11 +125,17 @@ class TestLangfuseOtelIntegration: test_env = "staging" with patch.dict(os.environ, {"LANGFUSE_TRACING_ENVIRONMENT": test_env}): - with patch("litellm.integrations.arize._utils.safe_set_attribute") as mock_safe_set_attribute: - LangfuseOtelLogger._set_langfuse_specific_attributes(mock_span, mock_kwargs, {}) + with patch( + "litellm.integrations.arize._utils.safe_set_attribute" + ) as mock_safe_set_attribute: + LangfuseOtelLogger._set_langfuse_specific_attributes( + mock_span, mock_kwargs, {} + ) # safe_set_attribute(span, key, value) → positional args - mock_safe_set_attribute.assert_called_once_with(mock_span, "langfuse.environment", test_env) + mock_safe_set_attribute.assert_called_once_with( + mock_span, "langfuse.environment", test_env + ) def test_set_langfuse_environment_attribute_prefers_dynamic_param(self): """Per-key/team langfuse_environment beats the deployment env var.""" @@ -140,10 +148,18 @@ class TestLangfuseOtelIntegration: self.attributes[key] = value span = _RecordingSpan() - mock_kwargs = {"standard_callback_dynamic_params": {"langfuse_environment": "team-a-env"}} + mock_kwargs = { + "standard_callback_dynamic_params": { + "langfuse_environment": "team-a-env" + } + } - with patch.dict(os.environ, {"LANGFUSE_TRACING_ENVIRONMENT": "deployment-wide"}): - LangfuseOtelLogger._set_langfuse_specific_attributes(span, mock_kwargs, {}) + with patch.dict( + os.environ, {"LANGFUSE_TRACING_ENVIRONMENT": "deployment-wide"} + ): + LangfuseOtelLogger._set_langfuse_specific_attributes( + span, mock_kwargs, {} + ) assert span.attributes["langfuse.environment"] == "team-a-env" @@ -207,8 +223,12 @@ class TestLangfuseOtelIntegration: kwargs = {"litellm_params": {"metadata": metadata}} # Capture calls to safe_set_attribute - with patch("litellm.integrations.arize._utils.safe_set_attribute") as mock_safe_set_attribute: - LangfuseOtelLogger._set_langfuse_specific_attributes(MagicMock(), kwargs, None) + with patch( + "litellm.integrations.arize._utils.safe_set_attribute" + ) as mock_safe_set_attribute: + LangfuseOtelLogger._set_langfuse_specific_attributes( + MagicMock(), kwargs, None + ) # Build expected calls manually for clarity from litellm.types.integrations.langfuse_otel import LangfuseSpanAttributes @@ -229,7 +249,9 @@ class TestLangfuseOtelIntegration: LangfuseSpanAttributes.TRACE_METADATA.value: json.dumps({"k": "v"}), LangfuseSpanAttributes.RELEASE.value: "rel-1", LangfuseSpanAttributes.EXISTING_TRACE_ID.value: "existing-id", - LangfuseSpanAttributes.UPDATE_TRACE_KEYS.value: json.dumps(["key1", "key2"]), + LangfuseSpanAttributes.UPDATE_TRACE_KEYS.value: json.dumps( + ["key1", "key2"] + ), LangfuseSpanAttributes.DEBUG_LANGFUSE.value: True, } @@ -239,7 +261,9 @@ class TestLangfuseOtelIntegration: for call in mock_safe_set_attribute.call_args_list } - assert actual == expected, "Mismatch between expected and actual OTEL attribute mapping." + assert ( + actual == expected + ), "Mismatch between expected and actual OTEL attribute mapping." @pytest.mark.parametrize( "metadata, expected_version", @@ -264,10 +288,16 @@ class TestLangfuseOtelIntegration: def test_version_emitted_on_langfuse_v4_key(self, metadata, expected_version): kwargs = {"litellm_params": {"metadata": {"trace_release": "rel-9", **metadata}}} - with patch("litellm.integrations.arize._utils.safe_set_attribute") as mock_safe_set_attribute: - LangfuseOtelLogger._set_langfuse_specific_attributes(MagicMock(), kwargs, None) + with patch( + "litellm.integrations.arize._utils.safe_set_attribute" + ) as mock_safe_set_attribute: + LangfuseOtelLogger._set_langfuse_specific_attributes( + MagicMock(), kwargs, None + ) - emitted = {call.args[1]: call.args[2] for call in mock_safe_set_attribute.call_args_list} + emitted = { + call.args[1]: call.args[2] for call in mock_safe_set_attribute.call_args_list + } if expected_version is None: assert "langfuse.version" not in emitted @@ -305,8 +335,12 @@ class TestLangfuseOtelIntegration: "messages": [{"role": "user", "content": "What's the weather in Tokyo?"}], } - with patch("litellm.integrations.arize._utils.safe_set_attribute") as mock_safe_set_attribute: - LangfuseOtelLogger._set_langfuse_specific_attributes(MagicMock(), kwargs, response_obj) + with patch( + "litellm.integrations.arize._utils.safe_set_attribute" + ) as mock_safe_set_attribute: + LangfuseOtelLogger._set_langfuse_specific_attributes( + MagicMock(), kwargs, response_obj + ) expect_output = { LangfuseSpanAttributes.OBSERVATION_INPUT.value: [ @@ -319,9 +353,14 @@ class TestLangfuseOtelIntegration: } # Flatten the actual calls into {key: value} - actual = {call.args[1]: json.loads(call.args[2]) for call in mock_safe_set_attribute.call_args_list} + actual = { + call.args[1]: json.loads(call.args[2]) + for call in mock_safe_set_attribute.call_args_list + } - assert actual == expect_output, "Mismatch in observation input/output OTEL attributes." + assert ( + actual == expect_output + ), "Mismatch in observation input/output OTEL attributes." def test_set_langfuse_specific_attributes_with_tool_calls(self): """Test that _set_langfuse_specific_attributes correctly sets observation.output with tool calls in Langfuse format.""" @@ -345,7 +384,9 @@ class TestLangfuseOtelIntegration: "content": None, "tool_calls": [ ChatCompletionMessageToolCall( - function=Function(arguments='{"location":"Tokyo"}', name="get_weather"), + function=Function( + arguments='{"location":"Tokyo"}', name="get_weather" + ), id="call_123", type="function", ) @@ -355,8 +396,12 @@ class TestLangfuseOtelIntegration: ], ) - with patch("litellm.integrations.arize._utils.safe_set_attribute") as mock_safe_set_attribute: - LangfuseOtelLogger._set_langfuse_specific_attributes(MagicMock(), {}, response_obj) + with patch( + "litellm.integrations.arize._utils.safe_set_attribute" + ) as mock_safe_set_attribute: + LangfuseOtelLogger._set_langfuse_specific_attributes( + MagicMock(), {}, response_obj + ) expected = { LangfuseSpanAttributes.OBSERVATION_OUTPUT.value: [ @@ -371,8 +416,13 @@ class TestLangfuseOtelIntegration: } # Flatten the actual calls into {key: value} - actual = {call.args[1]: json.loads(call.args[2]) for call in mock_safe_set_attribute.call_args_list} - assert actual == expected, "Mismatch in observation output OTEL attribute for tool calls." + actual = { + call.args[1]: json.loads(call.args[2]) + for call in mock_safe_set_attribute.call_args_list + } + assert ( + actual == expected + ), "Mismatch in observation output OTEL attribute for tool calls." def test_construct_dynamic_otel_headers_with_langfuse_keys(self): """Test that construct_dynamic_otel_headers creates proper auth headers when langfuse keys are provided.""" @@ -523,7 +573,9 @@ class TestLangfuseOtelKeyDynamicConfig: logger = LangfuseOtelLogger() assert logger.OTEL_EXPORTER == "console" - tracer = logger.get_tracer_to_use_for_request({"standard_callback_dynamic_params": self._dynamic_params()}) + tracer = logger.get_tracer_to_use_for_request( + {"standard_callback_dynamic_params": self._dynamic_params()} + ) assert tracer is not logger.tracer assert len(logger._tracer_provider_cache) == 1 @@ -584,7 +636,9 @@ class TestLangfuseOtelKeyDynamicConfig: with self._clean_env(): logger = LangfuseOtelLogger() with patch.object(otel_module.verbose_logger, "debug", side_effect=_spy): - logger.get_tracer_to_use_for_request({"standard_callback_dynamic_params": self._dynamic_params()}) + logger.get_tracer_to_use_for_request( + {"standard_callback_dynamic_params": self._dynamic_params()} + ) logged = "\n".join(recorded_arguments) assert "initializing span processor" in logged @@ -644,8 +698,12 @@ class TestLangfuseOtelResponsesAPI: LangfuseLLMObsOTELAttributes, ) - with patch("litellm.integrations.arize._utils.set_attributes") as mock_set_attributes: - with patch("litellm.integrations.arize._utils.safe_set_attribute") as mock_safe_set_attribute: + with patch( + "litellm.integrations.arize._utils.set_attributes" + ) as mock_set_attributes: + with patch( + "litellm.integrations.arize._utils.safe_set_attribute" + ) as mock_safe_set_attribute: logger = LangfuseOtelLogger() logger.set_langfuse_otel_attributes(mock_span, kwargs, mock_response) @@ -658,7 +716,9 @@ class TestLangfuseOtelResponsesAPI: mock_safe_set_attribute.assert_any_call( mock_span, "langfuse.generation.name", "responses_test_generation" ) - mock_safe_set_attribute.assert_any_call(mock_span, "langfuse.trace.name", "responses_api_trace") + mock_safe_set_attribute.assert_any_call( + mock_span, "langfuse.trace.name", "responses_api_trace" + ) def test_responses_api_metadata_extraction(self): """Test that metadata is correctly extracted from ResponsesAPI kwargs.""" @@ -708,7 +768,9 @@ class TestLangfuseOtelResponsesAPI: mock_span = MagicMock() - with patch("litellm.integrations.arize._utils.safe_set_attribute") as mock_safe_set_attribute: + with patch( + "litellm.integrations.arize._utils.safe_set_attribute" + ) as mock_safe_set_attribute: LangfuseOtelLogger._set_langfuse_specific_attributes(mock_span, kwargs, {}) # Verify specific attributes were set @@ -750,12 +812,11 @@ class TestLangfuseOtelResponsesAPI: def test_responses_api_with_output(self): """Test Langfuse OTEL logger with Responses API output (reasoning + message).""" from openai.types.responses import ( + ResponseReasoningItem, ResponseOutputMessage, ResponseOutputText, - ResponseReasoningItem, ) from openai.types.responses.response_reasoning_item import Summary - from litellm.types.integrations.langfuse_otel import LangfuseSpanAttributes # Create Responses API response with reasoning and message @@ -791,15 +852,21 @@ class TestLangfuseOtelResponsesAPI: kwargs = { "call_type": "responses", - "messages": [{"role": "user", "content": "What's the weather in San Francisco?"}], + "messages": [ + {"role": "user", "content": "What's the weather in San Francisco?"} + ], "model": "gpt-4o", "optional_params": {}, } mock_span = MagicMock() - with patch("litellm.integrations.arize._utils.safe_set_attribute") as mock_safe_set_attribute: - LangfuseOtelLogger._set_langfuse_specific_attributes(mock_span, kwargs, response_obj) + with patch( + "litellm.integrations.arize._utils.safe_set_attribute" + ) as mock_safe_set_attribute: + LangfuseOtelLogger._set_langfuse_specific_attributes( + mock_span, kwargs, response_obj + ) # Verify observation output was set output_calls = [ @@ -818,17 +885,22 @@ class TestLangfuseOtelResponsesAPI: # Verify reasoning summary assert output_data[0]["role"] == "reasoning_summary" - assert output_data[0]["content"] == "Let me analyze this problem step by step..." + assert ( + output_data[0]["content"] + == "Let me analyze this problem step by step..." + ) # Verify message assert output_data[1]["role"] == "assistant" - assert output_data[1]["content"] == "The weather in San Francisco is sunny, 20°C." + assert ( + output_data[1]["content"] + == "The weather in San Francisco is sunny, 20°C." + ) def test_responses_api_with_function_calls(self): """Test Langfuse OTEL logger with Responses API function_call output.""" - from openai.types.responses import ResponseFunctionToolCall - from litellm.types.integrations.langfuse_otel import LangfuseSpanAttributes + from openai.types.responses import ResponseFunctionToolCall # Create Responses API response with function call response_obj = ResponsesAPIResponse( @@ -848,15 +920,21 @@ class TestLangfuseOtelResponsesAPI: kwargs = { "call_type": "responses", - "messages": [{"role": "user", "content": "What's the weather in San Francisco?"}], + "messages": [ + {"role": "user", "content": "What's the weather in San Francisco?"} + ], "model": "gpt-4o", "optional_params": {}, } mock_span = MagicMock() - with patch("litellm.integrations.arize._utils.safe_set_attribute") as mock_safe_set_attribute: - LangfuseOtelLogger._set_langfuse_specific_attributes(mock_span, kwargs, response_obj) + with patch( + "litellm.integrations.arize._utils.safe_set_attribute" + ) as mock_safe_set_attribute: + LangfuseOtelLogger._set_langfuse_specific_attributes( + mock_span, kwargs, response_obj + ) # Verify observation output was set output_calls = [ @@ -911,11 +989,9 @@ class TestLangfuseOtelResponsesAPI: mock_span = MagicMock() - with ( - patch( # test-quality-ok: the span attribute sink is the observable boundary; sibling tests in this class stub the same seam - "litellm.integrations.arize._utils.safe_set_attribute" - ) as mock_safe_set_attribute - ): + with patch( # test-quality-ok: the span attribute sink is the observable boundary; sibling tests in this class stub the same seam + "litellm.integrations.arize._utils.safe_set_attribute" + ) as mock_safe_set_attribute: LangfuseOtelLogger._set_langfuse_specific_attributes(mock_span, kwargs, response_obj) output_calls = [