From 6dee7871193faad27624a4ef4d383656d1ac8aa3 Mon Sep 17 00:00:00 2001 From: Asaf Vertman Date: Thu, 16 Apr 2026 10:03:17 +0300 Subject: [PATCH] fix(langfuse_otel): include responses input in observation payload --- .../integrations/langfuse/langfuse_otel.py | 7 +- .../langfuse/langfuse_otel_attributes.py | 67 +++++++++++++--- .../integrations/test_langfuse_otel.py | 78 ++++++++++++++++++- 3 files changed, 136 insertions(+), 16 deletions(-) diff --git a/litellm/integrations/langfuse/langfuse_otel.py b/litellm/integrations/langfuse/langfuse_otel.py index b96ec72b04e..eff237914f8 100644 --- a/litellm/integrations/langfuse/langfuse_otel.py +++ b/litellm/integrations/langfuse/langfuse_otel.py @@ -8,6 +8,7 @@ from litellm._logging import verbose_logger from litellm.integrations.arize import _utils from litellm.integrations.langfuse.langfuse_otel_attributes import ( LangfuseLLMObsOTELAttributes, + get_langfuse_observation_input_by_type, ) from litellm.integrations.opentelemetry import OpenTelemetry, OpenTelemetryConfig from litellm.types.integrations.langfuse_otel import ( @@ -243,12 +244,12 @@ class LangfuseOtelLogger(OpenTelemetry): metadata = LangfuseOtelLogger._extract_langfuse_metadata(kwargs) LangfuseOtelLogger._set_metadata_attributes(span=span, metadata=metadata) - messages = kwargs.get("messages") - if messages: + input_payload = get_langfuse_observation_input_by_type(kwargs) + if input_payload is not None: safe_set_attribute( span, LangfuseSpanAttributes.OBSERVATION_INPUT.value, - safe_dumps(messages), + safe_dumps(input_payload), ) LangfuseOtelLogger._set_observation_output(span=span, response_obj=response_obj) diff --git a/litellm/integrations/langfuse/langfuse_otel_attributes.py b/litellm/integrations/langfuse/langfuse_otel_attributes.py index fb4a0a6a36c..a8b1de17973 100644 --- a/litellm/integrations/langfuse/langfuse_otel_attributes.py +++ b/litellm/integrations/langfuse/langfuse_otel_attributes.py @@ -14,6 +14,7 @@ from litellm.integrations.opentelemetry_utils.base_otel_llm_obs_attributes impor BaseLLMObsOTELAttributes, safe_set_attribute, ) +from litellm.litellm_core_utils.safe_json_dumps import safe_dumps from litellm.types.llms.openai import HttpxBinaryResponseContent, ResponsesAPIResponse from litellm.types.utils import ( EmbeddingResponse, @@ -82,21 +83,65 @@ def get_output_content_by_type( return "" +def get_langfuse_observation_input_by_type( + kwargs: Dict[str, Any], +) -> Optional[Union[list[Any], dict[str, Any]]]: + """ + Build the Langfuse observation input payload for both chat- and + Responses-style requests. + + Chat requests preserve the existing observation shape of the raw messages + list. Responses requests snapshot the model-visible request fields under a + single `input` payload for prompt debugging. + """ + messages = kwargs.get("messages") + if messages: + return messages + + optional_params = kwargs.get("optional_params", {}) or {} + response_input = kwargs.get("input") + if response_input is not None: + prompt: dict[str, Any] = {"input": response_input} + for key in ( + "instructions", + "functions", + "tools", + "tool_choice", + "reasoning", + "max_output_tokens", + "text", + "parallel_tool_calls", + "truncation", + "temperature", + "top_p", + ): + value = optional_params.get(key) + if value is not None: + prompt[key] = value + return prompt + + prompt: dict[str, Any] = {} + if "messages" in kwargs: + prompt["messages"] = messages + for key in ("functions", "tools"): + value = optional_params.get(key) + if value is not None: + prompt[key] = value + + return prompt or None + + class LangfuseLLMObsOTELAttributes(BaseLLMObsOTELAttributes): @staticmethod @override def set_messages(span: "Span", kwargs: Dict[str, Any]): - prompt = {"messages": kwargs.get("messages")} - optional_params = kwargs.get("optional_params", {}) - functions = optional_params.get("functions") - tools = optional_params.get("tools") - if functions is not None: - prompt["functions"] = functions - if tools is not None: - prompt["tools"] = tools - - input = prompt - safe_set_attribute(span, "langfuse.observation.input", json.dumps(input)) + input_payload = get_langfuse_observation_input_by_type(kwargs) + if input_payload is not None: + safe_set_attribute( + span, + "langfuse.observation.input", + safe_dumps(input_payload), + ) @staticmethod @override diff --git a/tests/test_litellm/integrations/test_langfuse_otel.py b/tests/test_litellm/integrations/test_langfuse_otel.py index 44853d9dce5..661fa1c6155 100644 --- a/tests/test_litellm/integrations/test_langfuse_otel.py +++ b/tests/test_litellm/integrations/test_langfuse_otel.py @@ -441,9 +441,9 @@ class TestLangfuseOtelResponsesAPI: kwargs = { "call_type": "responses", - "messages": [{"role": "user", "content": "Hello"}], + "input": [{"role": "user", "content": "Hello"}], "model": "gpt-4o", - "optional_params": {}, + "optional_params": {"instructions": "Reply briefly."}, "litellm_params": {"metadata": test_metadata}, } @@ -475,6 +475,80 @@ class TestLangfuseOtelResponsesAPI: mock_span, "langfuse.trace.name", "responses_api_trace" ) + def test_responses_api_observation_input_uses_input_items(self): + """Responses API calls should log the actual `input` items in observation.input.""" + response_input = [ + { + "role": "user", + "content": "show me connected applications with high risk", + } + ] + kwargs = { + "call_type": "responses", + "model": "gpt-4.1", + "input": response_input, + "optional_params": { + "instructions": "You are a security analyst.", + "tools": [ + { + "type": "function", + "name": "list_applications", + "description": "List applications", + "parameters": {"type": "object", "properties": {}}, + } + ], + "tool_choice": "auto", + "reasoning": {"effort": "low"}, + "stream": False, + }, + "litellm_params": {"custom_llm_provider": "openai", "metadata": {}}, + "standard_logging_object": { + "metadata": {}, + "call_type": "responses", + "model_parameters": { + "instructions": "You are a security analyst.", + "tools": [{"type": "function", "name": "list_applications"}], + "tool_choice": "auto", + "reasoning": {"effort": "low"}, + "stream": False, + }, + }, + } + response_obj = ResponsesAPIResponse( + id="response-456", + created_at=1234567891, + output=[ + { + "type": "message", + "content": [{"type": "text", "text": "Hello from responses API"}], + } + ], + parallel_tool_calls=False, + tool_choice="auto", + tools=[], + top_p=1.0, + ) + + mock_span = MagicMock() + + LangfuseOtelLogger.set_langfuse_otel_attributes(mock_span, kwargs, response_obj) + + actual_attributes = { + call.args[0]: call.args[1] + for call in mock_span.set_attribute.call_args_list + } + observation_input = json.loads(actual_attributes["langfuse.observation.input"]) + + assert observation_input["input"] == response_input + assert observation_input["instructions"] == "You are a security analyst." + assert observation_input["tool_choice"] == "auto" + assert observation_input["reasoning"] == {"effort": "low"} + assert observation_input["tools"][0]["name"] == "list_applications" + assert ( + "show me connected applications with high risk" + in actual_attributes["langfuse.observation.input"] + ) + def test_responses_api_metadata_extraction(self): """Test that metadata is correctly extracted from ResponsesAPI kwargs.""" # Clean up any existing module mocks