From 951e3284ed5aba4a310d5eaec9f2a39de67d6bc3 Mon Sep 17 00:00:00 2001 From: Harshit Jain Date: Tue, 27 Jan 2026 05:47:15 +0530 Subject: [PATCH 01/10] fix: support multi-project keys and fix trace leakage --- .../integrations/langfuse/langfuse_otel.py | 57 ++- litellm/litellm_core_utils/litellm_logging.py | 16 +- .../integrations/test_langfuse_otel.py | 443 +++++++++++------- 3 files changed, 325 insertions(+), 191 deletions(-) diff --git a/litellm/integrations/langfuse/langfuse_otel.py b/litellm/integrations/langfuse/langfuse_otel.py index 08493a0e8ec..4038745eb5a 100644 --- a/litellm/integrations/langfuse/langfuse_otel.py +++ b/litellm/integrations/langfuse/langfuse_otel.py @@ -37,8 +37,12 @@ LANGFUSE_CLOUD_US_ENDPOINT = "https://us.cloud.langfuse.com/api/public/otel" class LangfuseOtelLogger(OpenTelemetry): - def __init__(self, *args, **kwargs): - super().__init__(*args, **kwargs) + def __init__(self, config=None, *args, **kwargs): + # Prevent LangfuseOtelLogger from modifying global environment variables by constructing config manually + # and passing it to the parent OpenTelemetry class + if config is None: + config = self._create_open_telemetry_config_from_langfuse_env() + super().__init__(config=config, *args, **kwargs) @staticmethod def set_langfuse_otel_attributes(span: Span, kwargs, response_obj): @@ -114,6 +118,10 @@ class LangfuseOtelLogger(OpenTelemetry): for key, enum_attr in mapping.items(): if key in metadata and metadata[key] is not None: value = metadata[key] + if key == "trace_id" and isinstance(value, str): + # trace_id must be 32 hex char no dashes for langfuse : Litellm sends uuid with dashes (might be breaking at some point) + value = value.replace("-", "") + if isinstance(value, (list, dict)): try: value = json.dumps(value) @@ -265,6 +273,45 @@ class LangfuseOtelLogger(OpenTelemetry): """ return os.environ.get("LANGFUSE_OTEL_HOST") or os.environ.get("LANGFUSE_HOST") + def _create_open_telemetry_config_from_langfuse_env(self) -> OpenTelemetryConfig: + """ + Creates OpenTelemetryConfig from Langfuse environment variables. + Does NOT modify global environment variables. + """ + from litellm.integrations.opentelemetry import OpenTelemetryConfig + + public_key = os.environ.get("LANGFUSE_PUBLIC_KEY", None) + secret_key = os.environ.get("LANGFUSE_SECRET_KEY", None) + + if not public_key or not secret_key: + # If no keys, return default from env (likely logging to console or something else) + return OpenTelemetryConfig.from_env() + + # Determine endpoint - default to US cloud + langfuse_host = LangfuseOtelLogger._get_langfuse_otel_host() + + if langfuse_host: + # If LANGFUSE_HOST is provided, construct OTEL endpoint from it + if not langfuse_host.startswith("http"): + langfuse_host = "https://" + langfuse_host + endpoint = f"{langfuse_host.rstrip('/')}/api/public/otel" + verbose_logger.debug(f"Using Langfuse OTEL endpoint from host: {endpoint}") + else: + # Default to US cloud endpoint + endpoint = LANGFUSE_CLOUD_US_ENDPOINT + verbose_logger.debug(f"Using Langfuse US cloud endpoint: {endpoint}") + + auth_header = LangfuseOtelLogger._get_langfuse_authorization_header( + public_key=public_key, secret_key=secret_key + ) + otlp_auth_headers = f"Authorization={auth_header}" + + return OpenTelemetryConfig( + exporter="otlp_http", + endpoint=endpoint, + headers=otlp_auth_headers, + ) + @staticmethod def get_langfuse_otel_config() -> LangfuseOtelConfig: """ @@ -308,9 +355,9 @@ class LangfuseOtelLogger(OpenTelemetry): ) otlp_auth_headers = f"Authorization={auth_header}" - # Set standard OTEL environment variables - os.environ["OTEL_EXPORTER_OTLP_ENDPOINT"] = endpoint - os.environ["OTEL_EXPORTER_OTLP_HEADERS"] = otlp_auth_headers + # Prevent modification of global env vars which causes leakage + # os.environ["OTEL_EXPORTER_OTLP_ENDPOINT"] = endpoint + # os.environ["OTEL_EXPORTER_OTLP_HEADERS"] = otlp_auth_headers return LangfuseOtelConfig( otlp_auth_headers=otlp_auth_headers, protocol="otlp_http" diff --git a/litellm/litellm_core_utils/litellm_logging.py b/litellm/litellm_core_utils/litellm_logging.py index fadeeffa9cc..f0ea9e39819 100644 --- a/litellm/litellm_core_utils/litellm_logging.py +++ b/litellm/litellm_core_utils/litellm_logging.py @@ -3857,18 +3857,6 @@ def _init_custom_logger_compatible_class( # noqa: PLR0915 return langfuse_logger # type: ignore elif logging_integration == "langfuse_otel": from litellm.integrations.langfuse.langfuse_otel import LangfuseOtelLogger - from litellm.integrations.opentelemetry import ( - OpenTelemetry, - OpenTelemetryConfig, - ) - - langfuse_otel_config = LangfuseOtelLogger.get_langfuse_otel_config() - - # The endpoint and headers are now set as environment variables by get_langfuse_otel_config() - otel_config = OpenTelemetryConfig( - exporter=langfuse_otel_config.protocol, - headers=langfuse_otel_config.otlp_auth_headers, - ) for callback in _in_memory_loggers: if ( @@ -3876,8 +3864,10 @@ def _init_custom_logger_compatible_class( # noqa: PLR0915 and callback.callback_name == "langfuse_otel" ): return callback # type: ignore + # Allow LangfuseOtelLogger to initialize its own config safely + # This prevents startup crashes if LANGFUSE keys are not in env (e.g. for dynamic usage) _otel_logger = LangfuseOtelLogger( - config=otel_config, callback_name="langfuse_otel" + config=None, callback_name="langfuse_otel" ) _in_memory_loggers.append(_otel_logger) return _otel_logger # type: ignore diff --git a/tests/test_litellm/integrations/test_langfuse_otel.py b/tests/test_litellm/integrations/test_langfuse_otel.py index f8c662979ad..88602813902 100644 --- a/tests/test_litellm/integrations/test_langfuse_otel.py +++ b/tests/test_litellm/integrations/test_langfuse_otel.py @@ -1,6 +1,5 @@ import json import os -from datetime import datetime from unittest.mock import MagicMock, patch import pytest @@ -11,82 +10,110 @@ from litellm.types.llms.openai import ResponsesAPIResponse class TestLangfuseOtelIntegration: - def test_get_langfuse_otel_config_with_required_env_vars(self): """Test that config is created correctly with required environment variables.""" # Clean environment of any Langfuse-related variables - env_vars_to_clean = ['LANGFUSE_HOST', 'OTEL_EXPORTER_OTLP_ENDPOINT', 'OTEL_EXPORTER_OTLP_HEADERS'] - with patch.dict(os.environ, { - 'LANGFUSE_PUBLIC_KEY': 'test_public_key', - 'LANGFUSE_SECRET_KEY': 'test_secret_key' - }, clear=False): + env_vars_to_clean = [ + "LANGFUSE_HOST", + "OTEL_EXPORTER_OTLP_ENDPOINT", + "OTEL_EXPORTER_OTLP_HEADERS", + ] + with patch.dict( + os.environ, + { + "LANGFUSE_PUBLIC_KEY": "test_public_key", + "LANGFUSE_SECRET_KEY": "test_secret_key", + }, + clear=False, + ): # Remove any existing Langfuse variables for var in env_vars_to_clean: if var in os.environ: del os.environ[var] - + config = LangfuseOtelLogger.get_langfuse_otel_config() - + assert isinstance(config, LangfuseOtelConfig) assert config.protocol == "otlp_http" assert "Authorization=Basic" in config.otlp_auth_headers - # Check that environment variables are set correctly (US default) - assert os.environ.get("OTEL_EXPORTER_OTLP_ENDPOINT") == "https://us.cloud.langfuse.com/api/public/otel" - assert "Authorization=Basic" in os.environ.get("OTEL_EXPORTER_OTLP_HEADERS", "") - + # Note: We no longer set os.environ explicitly to avoid leakage + # assert os.environ.get("OTEL_EXPORTER_OTLP_ENDPOINT") == "https://us.cloud.langfuse.com/api/public/otel" + # assert "Authorization=Basic" in os.environ.get("OTEL_EXPORTER_OTLP_HEADERS", "") + def test_get_langfuse_otel_config_missing_keys(self): """Test that ValueError is raised when required keys are missing.""" with patch.dict(os.environ, {}, clear=True): - with pytest.raises(ValueError, match="LANGFUSE_PUBLIC_KEY and LANGFUSE_SECRET_KEY must be set"): + with pytest.raises( + ValueError, + match="LANGFUSE_PUBLIC_KEY and LANGFUSE_SECRET_KEY must be set", + ): LangfuseOtelLogger.get_langfuse_otel_config() - + def test_get_langfuse_otel_config_with_eu_host(self): """Test config with EU host.""" - with patch.dict(os.environ, { - 'LANGFUSE_PUBLIC_KEY': 'test_public_key', - 'LANGFUSE_SECRET_KEY': 'test_secret_key', - 'LANGFUSE_HOST': 'https://cloud.langfuse.com' - }, clear=False): + with patch.dict( + os.environ, + { + "LANGFUSE_PUBLIC_KEY": "test_public_key", + "LANGFUSE_SECRET_KEY": "test_secret_key", + "LANGFUSE_HOST": "https://cloud.langfuse.com", + }, + clear=False, + ): config = LangfuseOtelLogger.get_langfuse_otel_config() - - assert os.environ.get("OTEL_EXPORTER_OTLP_ENDPOINT") == "https://cloud.langfuse.com/api/public/otel" - + # Endpoint assertion removed as side effect is gone + assert isinstance(config, LangfuseOtelConfig) + def test_get_langfuse_otel_config_with_custom_host(self): """Test config with custom host.""" - with patch.dict(os.environ, { - 'LANGFUSE_PUBLIC_KEY': 'test_public_key', - 'LANGFUSE_SECRET_KEY': 'test_secret_key', - 'LANGFUSE_HOST': 'https://my-langfuse.com' - }, clear=False): + with patch.dict( + os.environ, + { + "LANGFUSE_PUBLIC_KEY": "test_public_key", + "LANGFUSE_SECRET_KEY": "test_secret_key", + "LANGFUSE_HOST": "https://my-langfuse.com", + }, + clear=False, + ): config = LangfuseOtelLogger.get_langfuse_otel_config() - - assert os.environ.get("OTEL_EXPORTER_OTLP_ENDPOINT") == "https://my-langfuse.com/api/public/otel" - + # Endpoint assertion removed as side effect is gone + assert isinstance(config, LangfuseOtelConfig) + def test_get_langfuse_otel_config_with_host_no_protocol(self): """Test config with custom host without protocol.""" - with patch.dict(os.environ, { - 'LANGFUSE_PUBLIC_KEY': 'test_public_key', - 'LANGFUSE_SECRET_KEY': 'test_secret_key', - 'LANGFUSE_HOST': 'my-langfuse.com' - }, clear=False): + with patch.dict( + os.environ, + { + "LANGFUSE_PUBLIC_KEY": "test_public_key", + "LANGFUSE_SECRET_KEY": "test_secret_key", + "LANGFUSE_HOST": "my-langfuse.com", + }, + clear=False, + ): config = LangfuseOtelLogger.get_langfuse_otel_config() - - assert os.environ.get("OTEL_EXPORTER_OTLP_ENDPOINT") == "https://my-langfuse.com/api/public/otel" - + # Endpoint assertion removed as side effect is gone + assert isinstance(config, LangfuseOtelConfig) + def test_set_langfuse_otel_attributes(self): """Test that set_langfuse_otel_attributes calls the Arize utils function.""" from litellm.integrations.langfuse.langfuse_otel_attributes import ( LangfuseLLMObsOTELAttributes, ) - + mock_span = MagicMock() mock_kwargs = {"test": "kwargs"} mock_response = {"test": "response"} - - with patch('litellm.integrations.arize._utils.set_attributes') as mock_set_attributes: - LangfuseOtelLogger.set_langfuse_otel_attributes(mock_span, mock_kwargs, mock_response) - - mock_set_attributes.assert_called_once_with(mock_span, mock_kwargs, mock_response, LangfuseLLMObsOTELAttributes) + + with patch( + "litellm.integrations.arize._utils.set_attributes" + ) as mock_set_attributes: + LangfuseOtelLogger.set_langfuse_otel_attributes( + mock_span, mock_kwargs, mock_response + ) + + mock_set_attributes.assert_called_once_with( + mock_span, mock_kwargs, mock_response, LangfuseLLMObsOTELAttributes + ) def test_set_langfuse_environment_attribute(self): """Test that Langfuse environment is set correctly when environment variable is present.""" @@ -94,15 +121,17 @@ class TestLangfuseOtelIntegration: mock_kwargs = {"test": "kwargs"} test_env = "staging" - with patch.dict(os.environ, {'LANGFUSE_TRACING_ENVIRONMENT': test_env}): - with patch('litellm.integrations.arize._utils.safe_set_attribute') as mock_safe_set_attribute: - LangfuseOtelLogger._set_langfuse_specific_attributes(mock_span, mock_kwargs, {}) - + with patch.dict(os.environ, {"LANGFUSE_TRACING_ENVIRONMENT": test_env}): + with patch( + "litellm.integrations.arize._utils.safe_set_attribute" + ) as mock_safe_set_attribute: + LangfuseOtelLogger._set_langfuse_specific_attributes( + mock_span, mock_kwargs, {} + ) + # safe_set_attribute(span, key, value) → positional args mock_safe_set_attribute.assert_called_once_with( - mock_span, - "langfuse.environment", - test_env + mock_span, "langfuse.environment", test_env ) def test_extract_langfuse_metadata_basic(self): @@ -119,11 +148,13 @@ class TestLangfuseOtelIntegration: # Build a stub module + class on-the-fly stub_module = types.ModuleType("litellm.integrations.langfuse.langfuse") + class StubLFLogger: @staticmethod def add_metadata_from_header(litellm_params, metadata): # Echo back existing metadata plus a marker return {**metadata, "enriched": True} + stub_module.LangFuseLogger = StubLFLogger # type: ignore # Register stub in sys.modules so import inside method succeeds @@ -159,11 +190,16 @@ class TestLangfuseOtelIntegration: kwargs = {"litellm_params": {"metadata": metadata}} # Capture calls to safe_set_attribute - with patch('litellm.integrations.arize._utils.safe_set_attribute') as mock_safe_set_attribute: - LangfuseOtelLogger._set_langfuse_specific_attributes(MagicMock(), kwargs, None) + with patch( + "litellm.integrations.arize._utils.safe_set_attribute" + ) as mock_safe_set_attribute: + LangfuseOtelLogger._set_langfuse_specific_attributes( + MagicMock(), kwargs, None + ) # Build expected calls manually for clarity from litellm.types.integrations.langfuse_otel import LangfuseSpanAttributes + expected = { LangfuseSpanAttributes.GENERATION_NAME.value: "gen-name", LangfuseSpanAttributes.GENERATION_ID.value: "gen-id", @@ -176,12 +212,14 @@ class TestLangfuseOtelIntegration: # Lists / dicts should be JSON strings LangfuseSpanAttributes.TAGS.value: json.dumps(["tagA", "tagB"]), LangfuseSpanAttributes.TRACE_NAME.value: "trace-name", - LangfuseSpanAttributes.TRACE_ID.value: "trace-id", + LangfuseSpanAttributes.TRACE_ID.value: "traceid", # stripped dashes LangfuseSpanAttributes.TRACE_METADATA.value: json.dumps({"k": "v"}), LangfuseSpanAttributes.TRACE_VERSION.value: "t-ver", LangfuseSpanAttributes.TRACE_RELEASE.value: "rel-1", LangfuseSpanAttributes.EXISTING_TRACE_ID.value: "existing-id", - LangfuseSpanAttributes.UPDATE_TRACE_KEYS.value: json.dumps(["key1", "key2"]), + LangfuseSpanAttributes.UPDATE_TRACE_KEYS.value: json.dumps( + ["key1", "key2"] + ), LangfuseSpanAttributes.DEBUG_LANGFUSE.value: True, } @@ -191,7 +229,9 @@ class TestLangfuseOtelIntegration: for call in mock_safe_set_attribute.call_args_list } - assert actual == expected, "Mismatch between expected and actual OTEL attribute mapping." + assert ( + actual == expected + ), "Mismatch between expected and actual OTEL attribute mapping." def test_set_langfuse_specific_attributes_with_content(self): """Test that _set_langfuse_specific_attributes correctly sets observation.output with regular content response.""" @@ -200,15 +240,15 @@ class TestLangfuseOtelIntegration: # Create response with content response_obj = ModelResponse( - id='chatcmpl-test', - model='gpt-4o', + id="chatcmpl-test", + model="gpt-4o", choices=[ Choices( - finish_reason='stop', + finish_reason="stop", message={ "role": "assistant", - "content": "The weather in Tokyo is sunny." - } + "content": "The weather in Tokyo is sunny.", + }, ) ], ) @@ -217,20 +257,21 @@ class TestLangfuseOtelIntegration: "messages": [{"role": "user", "content": "What's the weather in Tokyo?"}], } - with patch('litellm.integrations.arize._utils.safe_set_attribute') as mock_safe_set_attribute: - LangfuseOtelLogger._set_langfuse_specific_attributes(MagicMock(), kwargs, response_obj) + with patch( + "litellm.integrations.arize._utils.safe_set_attribute" + ) as mock_safe_set_attribute: + LangfuseOtelLogger._set_langfuse_specific_attributes( + MagicMock(), kwargs, response_obj + ) expect_output = { LangfuseSpanAttributes.OBSERVATION_INPUT.value: [ - { - "role": "user", - "content": "What's the weather in Tokyo?" - } + {"role": "user", "content": "What's the weather in Tokyo?"} ], LangfuseSpanAttributes.OBSERVATION_OUTPUT.value: { "role": "assistant", - "content": "The weather in Tokyo is sunny." - } + "content": "The weather in Tokyo is sunny.", + }, } # Flatten the actual calls into {key: value} @@ -239,8 +280,9 @@ class TestLangfuseOtelIntegration: for call in mock_safe_set_attribute.call_args_list } - assert actual == expect_output, "Mismatch in observation input/output OTEL attributes." - + assert ( + actual == expect_output + ), "Mismatch in observation input/output OTEL attributes." def test_set_langfuse_specific_attributes_with_tool_calls(self): """Test that _set_langfuse_specific_attributes correctly sets observation.output with tool calls in Langfuse format.""" @@ -254,42 +296,44 @@ class TestLangfuseOtelIntegration: # Create response with tool calls response_obj = ModelResponse( - id='chatcmpl-test', - model='gpt-4o', + id="chatcmpl-test", + model="gpt-4o", choices=[ Choices( - finish_reason='tool_calls', + finish_reason="tool_calls", message={ "role": "assistant", "content": None, "tool_calls": [ ChatCompletionMessageToolCall( function=Function( - arguments='{"location":"Tokyo"}', - name='get_weather' + arguments='{"location":"Tokyo"}', name="get_weather" ), - id='call_123', - type='function' + id="call_123", + type="function", ) - ] - } + ], + }, ) ], ) - with patch('litellm.integrations.arize._utils.safe_set_attribute') as mock_safe_set_attribute: - LangfuseOtelLogger._set_langfuse_specific_attributes(MagicMock(), {}, - response_obj) + with patch( + "litellm.integrations.arize._utils.safe_set_attribute" + ) as mock_safe_set_attribute: + LangfuseOtelLogger._set_langfuse_specific_attributes( + MagicMock(), {}, response_obj + ) expected = { LangfuseSpanAttributes.OBSERVATION_OUTPUT.value: [ - { - "id": "chatcmpl-test", - "name": "get_weather", - "arguments": {"location": "Tokyo"}, - "call_id": "call_123", - "type": "function_call" - } + { + "id": "chatcmpl-test", + "name": "get_weather", + "arguments": {"location": "Tokyo"}, + "call_id": "call_123", + "type": "function_call", + } ] } @@ -298,8 +342,9 @@ class TestLangfuseOtelIntegration: call.args[1]: json.loads(call.args[2]) for call in mock_safe_set_attribute.call_args_list } - assert actual == expected, "Mismatch in observation output OTEL attribute for tool calls." - + assert ( + actual == expected + ), "Mismatch in observation output OTEL attribute for tool calls." def test_construct_dynamic_otel_headers_with_langfuse_keys(self): """Test that construct_dynamic_otel_headers creates proper auth headers when langfuse keys are provided.""" @@ -307,28 +352,27 @@ class TestLangfuseOtelIntegration: # Create dynamic params with langfuse keys dynamic_params = StandardCallbackDynamicParams( - langfuse_public_key="test_public_key", - langfuse_secret_key="test_secret_key" + langfuse_public_key="test_public_key", langfuse_secret_key="test_secret_key" ) - + logger = LangfuseOtelLogger() result = logger.construct_dynamic_otel_headers(dynamic_params) - + # Should return a dict with otlp_auth_headers assert result is not None assert "Authorization" in result - + # The auth header should contain the basic auth format auth_header = result["Authorization"] assert auth_header.startswith("Basic ") - + # Verify the header format by decoding import base64 # Extract the base64 part from "Authorization=Basic " base64_part = auth_header.replace("Basic ", "") decoded = base64.b64decode(base64_part).decode() - + assert decoded == "test_public_key:test_secret_key" def test_construct_dynamic_otel_headers_empty_params(self): @@ -337,24 +381,28 @@ class TestLangfuseOtelIntegration: # Create dynamic params without langfuse keys dynamic_params = StandardCallbackDynamicParams() - + logger = LangfuseOtelLogger() result = logger.construct_dynamic_otel_headers(dynamic_params) - + # Should return an empty dict assert result == {} - + def test_get_langfuse_otel_config_with_otel_host_priority(self): """LANGFUSE_OTEL_HOST should take priority over LANGFUSE_HOST.""" - with patch.dict(os.environ, { - 'LANGFUSE_PUBLIC_KEY': 'test_public_key', - 'LANGFUSE_SECRET_KEY': 'test_secret_key', - 'LANGFUSE_HOST': 'https://should-not-be-used.com', - 'LANGFUSE_OTEL_HOST': 'https://otel-host.com' - }, clear=False): - _ = LangfuseOtelLogger.get_langfuse_otel_config() - - assert os.environ.get("OTEL_EXPORTER_OTLP_ENDPOINT") == "https://otel-host.com/api/public/otel" + with patch.dict( + os.environ, + { + "LANGFUSE_PUBLIC_KEY": "test_public_key", + "LANGFUSE_SECRET_KEY": "test_secret_key", + "LANGFUSE_HOST": "https://should-not-be-used.com", + "LANGFUSE_OTEL_HOST": "https://otel-host.com", + }, + clear=False, + ): + config = LangfuseOtelLogger.get_langfuse_otel_config() + assert isinstance(config, LangfuseOtelConfig) + # Endpoint assertion removed as side effect is gone class TestLangfuseOtelResponsesAPI: @@ -369,46 +417,52 @@ class TestLangfuseOtelResponsesAPI: output=[ { "type": "message", - "content": [{"type": "text", "text": "Hello from responses API"}] + "content": [{"type": "text", "text": "Hello from responses API"}], } ], parallel_tool_calls=False, tool_choice="auto", tools=[], - top_p=1.0 + top_p=1.0, ) - + # Create kwargs with metadata that should be logged test_metadata = { - "user_id": "test123", - "session_id": "abc456", + "user_id": "test123", + "session_id": "abc456", "custom_field": "test_value", "generation_name": "responses_test_generation", - "trace_name": "responses_api_trace" + "trace_name": "responses_api_trace", } - + kwargs = { "call_type": "responses", "messages": [{"role": "user", "content": "Hello"}], "model": "gpt-4o", "optional_params": {}, - "litellm_params": {"metadata": test_metadata} + "litellm_params": {"metadata": test_metadata}, } - + mock_span = MagicMock() - + from litellm.integrations.langfuse.langfuse_otel_attributes import ( LangfuseLLMObsOTELAttributes, ) - - with patch('litellm.integrations.arize._utils.set_attributes') as mock_set_attributes: - with patch('litellm.integrations.arize._utils.safe_set_attribute') as mock_safe_set_attribute: + + with patch( + "litellm.integrations.arize._utils.set_attributes" + ) as mock_set_attributes: + with patch( + "litellm.integrations.arize._utils.safe_set_attribute" + ) as mock_safe_set_attribute: logger = LangfuseOtelLogger() logger.set_langfuse_otel_attributes(mock_span, kwargs, mock_response) - + # Verify that set_attributes was called for general attributes - mock_set_attributes.assert_called_once_with(mock_span, kwargs, mock_response, LangfuseLLMObsOTELAttributes) - + mock_set_attributes.assert_called_once_with( + mock_span, kwargs, mock_response, LangfuseLLMObsOTELAttributes + ) + # Verify that Langfuse-specific attributes were set mock_safe_set_attribute.assert_any_call( mock_span, "langfuse.generation.name", "responses_test_generation" @@ -421,29 +475,30 @@ class TestLangfuseOtelResponsesAPI: """Test that metadata is correctly extracted from ResponsesAPI kwargs.""" # Clean up any existing module mocks import sys + if "litellm.integrations.langfuse.langfuse" in sys.modules: original_module = sys.modules["litellm.integrations.langfuse.langfuse"] - + test_metadata = { "user_id": "responses_user_123", - "session_id": "responses_session_456", + "session_id": "responses_session_456", "custom_metadata": {"key": "value"}, "generation_name": "responses_generation", - "trace_id": "custom_trace_id" + "trace_id": "custom_trace_id", } - + kwargs = { "call_type": "responses", "model": "gpt-4o", - "litellm_params": {"metadata": test_metadata} + "litellm_params": {"metadata": test_metadata}, } - + extracted_metadata = LangfuseOtelLogger._extract_langfuse_metadata(kwargs) - + # Verify all expected metadata was extracted (may have additional fields from header enrichment) for key, value in test_metadata.items(): assert extracted_metadata[key] == value - + assert extracted_metadata["user_id"] == "responses_user_123" assert extracted_metadata["generation_name"] == "responses_generation" assert extracted_metadata["trace_id"] == "custom_trace_id" @@ -457,39 +512,61 @@ class TestLangfuseOtelResponsesAPI: "trace_user_id": "resp_user_456", "session_id": "resp_session_789", "tags": ["responses", "api", "test"], - "trace_metadata": {"source": "responses_api", "version": "1.0"} + "trace_metadata": {"source": "responses_api", "version": "1.0"}, } - - kwargs = { - "call_type": "responses", - "litellm_params": {"metadata": metadata} - } - + + kwargs = {"call_type": "responses", "litellm_params": {"metadata": metadata}} + mock_span = MagicMock() - - with patch('litellm.integrations.arize._utils.safe_set_attribute') as mock_safe_set_attribute: + + with patch( + "litellm.integrations.arize._utils.safe_set_attribute" + ) as mock_safe_set_attribute: LangfuseOtelLogger._set_langfuse_specific_attributes(mock_span, kwargs, {}) - + # Verify specific attributes were set from litellm.types.integrations.langfuse_otel import LangfuseSpanAttributes - + expected_calls = [ - (mock_span, LangfuseSpanAttributes.GENERATION_NAME.value, "responses_gen"), + ( + mock_span, + LangfuseSpanAttributes.GENERATION_NAME.value, + "responses_gen", + ), (mock_span, LangfuseSpanAttributes.GENERATION_ID.value, "resp_gen_123"), (mock_span, LangfuseSpanAttributes.TRACE_NAME.value, "responses_trace"), - (mock_span, LangfuseSpanAttributes.TRACE_USER_ID.value, "resp_user_456"), - (mock_span, LangfuseSpanAttributes.SESSION_ID.value, "resp_session_789"), - (mock_span, LangfuseSpanAttributes.TAGS.value, json.dumps(["responses", "api", "test"])), - (mock_span, LangfuseSpanAttributes.TRACE_METADATA.value, - json.dumps({"source": "responses_api", "version": "1.0"})) + ( + mock_span, + LangfuseSpanAttributes.TRACE_USER_ID.value, + "resp_user_456", + ), + ( + mock_span, + LangfuseSpanAttributes.SESSION_ID.value, + "resp_session_789", + ), + ( + mock_span, + LangfuseSpanAttributes.TAGS.value, + json.dumps(["responses", "api", "test"]), + ), + ( + mock_span, + LangfuseSpanAttributes.TRACE_METADATA.value, + json.dumps({"source": "responses_api", "version": "1.0"}), + ), ] - + for expected_call in expected_calls: mock_safe_set_attribute.assert_any_call(*expected_call) def test_responses_api_with_output(self): """Test Langfuse OTEL logger with Responses API output (reasoning + message).""" - from openai.types.responses import ResponseReasoningItem, ResponseOutputMessage, ResponseOutputText + from openai.types.responses import ( + ResponseReasoningItem, + ResponseOutputMessage, + ResponseOutputText, + ) from openai.types.responses.response_reasoning_item import Summary from litellm.types.integrations.langfuse_otel import LangfuseSpanAttributes @@ -504,9 +581,9 @@ class TestLangfuseOtelResponsesAPI: summary=[ Summary( text="Let me analyze this problem step by step...", - type="summary_text" + type="summary_text", ) - ] + ], ), ResponseOutputMessage( id="msg-001", @@ -519,26 +596,33 @@ class TestLangfuseOtelResponsesAPI: text="The weather in San Francisco is sunny, 20°C.", type="output_text", ) - ] - ) - ] + ], + ), + ], ) kwargs = { "call_type": "responses", - "messages": [{"role": "user", "content": "What's the weather in San Francisco?"}], + "messages": [ + {"role": "user", "content": "What's the weather in San Francisco?"} + ], "model": "gpt-4o", "optional_params": {}, } mock_span = MagicMock() - with patch('litellm.integrations.arize._utils.safe_set_attribute') as mock_safe_set_attribute: - LangfuseOtelLogger._set_langfuse_specific_attributes(mock_span, kwargs, response_obj) + with patch( + "litellm.integrations.arize._utils.safe_set_attribute" + ) as mock_safe_set_attribute: + LangfuseOtelLogger._set_langfuse_specific_attributes( + mock_span, kwargs, response_obj + ) # Verify observation output was set output_calls = [ - call for call in mock_safe_set_attribute.call_args_list + call + for call in mock_safe_set_attribute.call_args_list if call.args[1] == LangfuseSpanAttributes.OBSERVATION_OUTPUT.value ] @@ -552,11 +636,17 @@ class TestLangfuseOtelResponsesAPI: # Verify reasoning summary assert output_data[0]["role"] == "reasoning_summary" - assert output_data[0]["content"] == "Let me analyze this problem step by step..." + assert ( + output_data[0]["content"] + == "Let me analyze this problem step by step..." + ) # Verify message assert output_data[1]["role"] == "assistant" - assert output_data[1]["content"] == "The weather in San Francisco is sunny, 20°C." + assert ( + output_data[1]["content"] + == "The weather in San Francisco is sunny, 20°C." + ) def test_responses_api_with_function_calls(self): """Test Langfuse OTEL logger with Responses API function_call output.""" @@ -574,26 +664,33 @@ class TestLangfuseOtelResponsesAPI: name="get_weather", call_id="call-abc", arguments='{"location": "San Francisco", "unit": "celsius"}', - status="completed" + status="completed", ) - ] + ], ) kwargs = { "call_type": "responses", - "messages": [{"role": "user", "content": "What's the weather in San Francisco?"}], + "messages": [ + {"role": "user", "content": "What's the weather in San Francisco?"} + ], "model": "gpt-4o", "optional_params": {}, } mock_span = MagicMock() - with patch('litellm.integrations.arize._utils.safe_set_attribute') as mock_safe_set_attribute: - LangfuseOtelLogger._set_langfuse_specific_attributes(mock_span, kwargs, response_obj) + with patch( + "litellm.integrations.arize._utils.safe_set_attribute" + ) as mock_safe_set_attribute: + LangfuseOtelLogger._set_langfuse_specific_attributes( + mock_span, kwargs, response_obj + ) # Verify observation output was set output_calls = [ - call for call in mock_safe_set_attribute.call_args_list + call + for call in mock_safe_set_attribute.call_args_list if call.args[1] == LangfuseSpanAttributes.OBSERVATION_OUTPUT.value ] @@ -615,4 +712,4 @@ class TestLangfuseOtelResponsesAPI: if __name__ == "__main__": - pytest.main([__file__]) \ No newline at end of file + pytest.main([__file__]) From 4d86236cd6edffc10c66b63fa5b71ed25d593810 Mon Sep 17 00:00:00 2001 From: Harshit Jain Date: Wed, 28 Jan 2026 07:51:31 +0530 Subject: [PATCH 02/10] fix: Langfuse otel handle --- .../integrations/langfuse/langfuse_otel.py | 13 ++--- litellm/integrations/opentelemetry.py | 11 +++- .../initialize_dynamic_callback_params.py | 53 ++++++++++++++++--- .../test_dynamic_otel_keys.py | 52 ++++++++++++++++++ 4 files changed, 115 insertions(+), 14 deletions(-) create mode 100644 tests/logging_callback_tests/test_dynamic_otel_keys.py diff --git a/litellm/integrations/langfuse/langfuse_otel.py b/litellm/integrations/langfuse/langfuse_otel.py index 4038745eb5a..6380c3c7b66 100644 --- a/litellm/integrations/langfuse/langfuse_otel.py +++ b/litellm/integrations/langfuse/langfuse_otel.py @@ -8,9 +8,8 @@ from litellm.integrations.arize import _utils from litellm.integrations.langfuse.langfuse_otel_attributes import ( LangfuseLLMObsOTELAttributes, ) -from litellm.integrations.opentelemetry import OpenTelemetry +from litellm.integrations.opentelemetry import OpenTelemetry, OpenTelemetryConfig from litellm.types.integrations.langfuse_otel import ( - LangfuseOtelConfig, LangfuseSpanAttributes, ) from litellm.types.utils import StandardCallbackDynamicParams @@ -313,7 +312,7 @@ class LangfuseOtelLogger(OpenTelemetry): ) @staticmethod - def get_langfuse_otel_config() -> LangfuseOtelConfig: + def get_langfuse_otel_config() -> "OpenTelemetryConfig": """ Retrieves the Langfuse OpenTelemetry configuration based on environment variables. @@ -323,7 +322,7 @@ class LangfuseOtelLogger(OpenTelemetry): LANGFUSE_HOST: Optional. Custom Langfuse host URL. Defaults to US cloud. Returns: - LangfuseOtelConfig: A Pydantic model containing Langfuse OTEL configuration. + OpenTelemetryConfig: A Pydantic model containing Langfuse OTEL configuration. Raises: ValueError: If required keys are missing. @@ -359,8 +358,10 @@ class LangfuseOtelLogger(OpenTelemetry): # os.environ["OTEL_EXPORTER_OTLP_ENDPOINT"] = endpoint # os.environ["OTEL_EXPORTER_OTLP_HEADERS"] = otlp_auth_headers - return LangfuseOtelConfig( - otlp_auth_headers=otlp_auth_headers, protocol="otlp_http" + return OpenTelemetryConfig( + exporter="otlp_http", + endpoint=endpoint, + headers=otlp_auth_headers, ) @staticmethod diff --git a/litellm/integrations/opentelemetry.py b/litellm/integrations/opentelemetry.py index a8a7fa77b3d..29be248fe85 100644 --- a/litellm/integrations/opentelemetry.py +++ b/litellm/integrations/opentelemetry.py @@ -665,7 +665,10 @@ class OpenTelemetry(CustomLogger): kwargs, response_obj, start_time, end_time, span ) # Ensure proxy-request parent span is annotated with the actual operation kind - if parent_span is not None and parent_span.name == LITELLM_PROXY_REQUEST_SPAN_NAME: + if ( + parent_span is not None + and parent_span.name == LITELLM_PROXY_REQUEST_SPAN_NAME + ): self.set_attributes(parent_span, kwargs, response_obj) else: # Do not create primary span (keep hierarchy shallow when parent exists) @@ -994,10 +997,13 @@ class OpenTelemetry(CustomLogger): # TODO: Refactor to use the proper OTEL Logs API instead of directly creating SDK LogRecords from opentelemetry._logs import SeverityNumber, get_logger, get_logger_provider + try: from opentelemetry.sdk._logs import LogRecord as SdkLogRecord # type: ignore[attr-defined] # OTEL < 1.39.0 except ImportError: - from opentelemetry.sdk._logs._internal import LogRecord as SdkLogRecord # OTEL >= 1.39.0 + from opentelemetry.sdk._logs._internal import ( + LogRecord as SdkLogRecord, + ) # OTEL >= 1.39.0 otel_logger = get_logger(LITELLM_LOGGER_NAME) @@ -1708,6 +1714,7 @@ class OpenTelemetry(CustomLogger): def set_raw_request_attributes(self, span: Span, kwargs, response_obj): try: + self.set_attributes(span, kwargs, response_obj) kwargs.get("optional_params", {}) litellm_params = kwargs.get("litellm_params", {}) or {} custom_llm_provider = litellm_params.get("custom_llm_provider", "Unknown") diff --git a/litellm/litellm_core_utils/initialize_dynamic_callback_params.py b/litellm/litellm_core_utils/initialize_dynamic_callback_params.py index c425319b4d4..78846f8e82c 100644 --- a/litellm/litellm_core_utils/initialize_dynamic_callback_params.py +++ b/litellm/litellm_core_utils/initialize_dynamic_callback_params.py @@ -1,8 +1,34 @@ from typing import Dict, Optional - from litellm.secret_managers.main import get_secret_str from litellm.types.utils import StandardCallbackDynamicParams +# Hardcoded list of supported callback params to avoid runtime inspection issues with TypedDict +_supported_callback_params = [ + "langfuse_public_key", + "langfuse_secret", + "langfuse_secret_key", + "langfuse_host", + "langfuse_prompt_version", + "gcs_bucket_name", + "gcs_path_service_account", + "langsmith_api_key", + "langsmith_project", + "langsmith_base_url", + "langsmith_sampling_rate", + "langsmith_tenant_id", + "humanloop_api_key", + "arize_api_key", + "arize_space_key", + "arize_space_id", + "posthog_api_key", + "posthog_host", + "braintrust_api_key", + "braintrust_project", + "braintrust_host", + "slack_webhook_url", + "lunary_public_key", +] + def initialize_standard_callback_dynamic_params( kwargs: Optional[Dict] = None, @@ -15,13 +41,10 @@ def initialize_standard_callback_dynamic_params( standard_callback_dynamic_params = StandardCallbackDynamicParams() if kwargs: - _supported_callback_params = ( - StandardCallbackDynamicParams.__annotations__.keys() - ) - + # 1. Check top-level kwargs for param in _supported_callback_params: if param in kwargs: - _param_value = kwargs.pop(param) + _param_value = kwargs.get(param) if ( _param_value is not None and isinstance(_param_value, str) @@ -30,4 +53,22 @@ def initialize_standard_callback_dynamic_params( _param_value = get_secret_str(secret_name=_param_value) standard_callback_dynamic_params[param] = _param_value # type: ignore + # 2. Fallback: check "metadata" or "litellm_params" -> "metadata" + metadata = (kwargs.get("metadata") or {}).copy() + litellm_params = kwargs.get("litellm_params") or {} + if isinstance(litellm_params, dict): + metadata.update(litellm_params.get("metadata") or {}) + + if isinstance(metadata, dict): + for param in _supported_callback_params: + if param not in standard_callback_dynamic_params and param in metadata: + _param_value = metadata.get(param) + if ( + _param_value is not None + and isinstance(_param_value, str) + and "os.environ/" in _param_value + ): + _param_value = get_secret_str(secret_name=_param_value) + standard_callback_dynamic_params[param] = _param_value # type: ignore + return standard_callback_dynamic_params diff --git a/tests/logging_callback_tests/test_dynamic_otel_keys.py b/tests/logging_callback_tests/test_dynamic_otel_keys.py new file mode 100644 index 00000000000..2a463fddc0d --- /dev/null +++ b/tests/logging_callback_tests/test_dynamic_otel_keys.py @@ -0,0 +1,52 @@ +import sys +import os + +sys.path.insert(0, os.path.abspath("../..")) + +from litellm.litellm_core_utils.initialize_dynamic_callback_params import ( + initialize_standard_callback_dynamic_params, +) + + +def test_dynamic_key_extraction_from_metadata(): + """ + Test extraction of langfuse keys from metadata in kwargs. + This simulates a Proxy request where keys are passed in metadata. + """ + kwargs = { + "metadata": { + "langfuse_public_key": "pk-test", + "langfuse_secret_key": "sk-test", + "langfuse_host": "https://test.langfuse.com", + } + } + + params = initialize_standard_callback_dynamic_params(kwargs) + + assert params.get("langfuse_public_key") == "pk-test" + assert params.get("langfuse_secret_key") == "sk-test" + assert params.get("langfuse_host") == "https://test.langfuse.com" + + +def test_dynamic_key_extraction_from_litellm_params_metadata(): + """ + Test extraction of langfuse keys from litellm_params.metadata. + """ + kwargs = { + "litellm_params": { + "metadata": { + "langfuse_public_key": "pk-litellm", + "langfuse_secret_key": "sk-litellm", + } + } + } + + params = initialize_standard_callback_dynamic_params(kwargs) + + assert params.get("langfuse_public_key") == "pk-litellm" + assert params.get("langfuse_secret_key") == "sk-litellm" + + +if __name__ == "__main__": + test_dynamic_key_extraction_from_metadata() + test_dynamic_key_extraction_from_litellm_params_metadata() From 6a045ee48fbc2e0ec09d2ce3360ccfa74d09b4fc Mon Sep 17 00:00:00 2001 From: Harshit Jain Date: Wed, 28 Jan 2026 08:36:51 +0530 Subject: [PATCH 03/10] fix lint errors mypy --- litellm/integrations/langfuse/langfuse_otel.py | 9 --------- 1 file changed, 9 deletions(-) diff --git a/litellm/integrations/langfuse/langfuse_otel.py b/litellm/integrations/langfuse/langfuse_otel.py index 6380c3c7b66..8955d3619f7 100644 --- a/litellm/integrations/langfuse/langfuse_otel.py +++ b/litellm/integrations/langfuse/langfuse_otel.py @@ -17,17 +17,8 @@ from litellm.types.utils import StandardCallbackDynamicParams if TYPE_CHECKING: from opentelemetry.trace import Span as _Span - from litellm.integrations.opentelemetry import ( - OpenTelemetryConfig as _OpenTelemetryConfig, - ) - from litellm.types.integrations.arize import Protocol as _Protocol - - Protocol = _Protocol - OpenTelemetryConfig = _OpenTelemetryConfig Span = Union[_Span, Any] else: - Protocol = Any - OpenTelemetryConfig = Any Span = Any From ac6fe0198f63a70723c96db222912278046790e9 Mon Sep 17 00:00:00 2001 From: Harshit Jain Date: Wed, 4 Feb 2026 08:11:31 +0530 Subject: [PATCH 04/10] passing all test case --- .../integrations/test_langfuse_otel.py | 18 +++++++++--------- 1 file changed, 9 insertions(+), 9 deletions(-) diff --git a/tests/test_litellm/integrations/test_langfuse_otel.py b/tests/test_litellm/integrations/test_langfuse_otel.py index 88602813902..ba4a096be24 100644 --- a/tests/test_litellm/integrations/test_langfuse_otel.py +++ b/tests/test_litellm/integrations/test_langfuse_otel.py @@ -5,7 +5,7 @@ from unittest.mock import MagicMock, patch import pytest from litellm.integrations.langfuse.langfuse_otel import LangfuseOtelLogger -from litellm.types.integrations.langfuse_otel import LangfuseOtelConfig +from litellm.integrations.opentelemetry import OpenTelemetryConfig from litellm.types.llms.openai import ResponsesAPIResponse @@ -33,9 +33,9 @@ class TestLangfuseOtelIntegration: config = LangfuseOtelLogger.get_langfuse_otel_config() - assert isinstance(config, LangfuseOtelConfig) - assert config.protocol == "otlp_http" - assert "Authorization=Basic" in config.otlp_auth_headers + assert isinstance(config, OpenTelemetryConfig) + assert config.exporter == "otlp_http" + assert "Authorization=Basic" in config.headers # Note: We no longer set os.environ explicitly to avoid leakage # assert os.environ.get("OTEL_EXPORTER_OTLP_ENDPOINT") == "https://us.cloud.langfuse.com/api/public/otel" # assert "Authorization=Basic" in os.environ.get("OTEL_EXPORTER_OTLP_HEADERS", "") @@ -62,7 +62,7 @@ class TestLangfuseOtelIntegration: ): config = LangfuseOtelLogger.get_langfuse_otel_config() # Endpoint assertion removed as side effect is gone - assert isinstance(config, LangfuseOtelConfig) + assert isinstance(config, OpenTelemetryConfig) def test_get_langfuse_otel_config_with_custom_host(self): """Test config with custom host.""" @@ -77,7 +77,7 @@ class TestLangfuseOtelIntegration: ): config = LangfuseOtelLogger.get_langfuse_otel_config() # Endpoint assertion removed as side effect is gone - assert isinstance(config, LangfuseOtelConfig) + assert isinstance(config, OpenTelemetryConfig) def test_get_langfuse_otel_config_with_host_no_protocol(self): """Test config with custom host without protocol.""" @@ -92,7 +92,7 @@ class TestLangfuseOtelIntegration: ): config = LangfuseOtelLogger.get_langfuse_otel_config() # Endpoint assertion removed as side effect is gone - assert isinstance(config, LangfuseOtelConfig) + assert isinstance(config, OpenTelemetryConfig) def test_set_langfuse_otel_attributes(self): """Test that set_langfuse_otel_attributes calls the Arize utils function.""" @@ -401,7 +401,7 @@ class TestLangfuseOtelIntegration: clear=False, ): config = LangfuseOtelLogger.get_langfuse_otel_config() - assert isinstance(config, LangfuseOtelConfig) + assert isinstance(config, OpenTelemetryConfig) # Endpoint assertion removed as side effect is gone @@ -477,7 +477,7 @@ class TestLangfuseOtelResponsesAPI: import sys if "litellm.integrations.langfuse.langfuse" in sys.modules: - original_module = sys.modules["litellm.integrations.langfuse.langfuse"] + sys.modules["litellm.integrations.langfuse.langfuse"] test_metadata = { "user_id": "responses_user_123", From acfddf204985c8c662ca8be7a8639a8556962ad1 Mon Sep 17 00:00:00 2001 From: Harshit Jain <48647625+Harshit28j@users.noreply.github.com> Date: Mon, 16 Feb 2026 19:07:22 +0530 Subject: [PATCH 05/10] Update litellm/litellm_core_utils/litellm_logging.py Co-authored-by: greptile-apps[bot] <165735046+greptile-apps[bot]@users.noreply.github.com> --- litellm/litellm_core_utils/litellm_logging.py | 14 +++++--------- 1 file changed, 5 insertions(+), 9 deletions(-) diff --git a/litellm/litellm_core_utils/litellm_logging.py b/litellm/litellm_core_utils/litellm_logging.py index 0ee33d0c3b0..ed8ef994b0f 100644 --- a/litellm/litellm_core_utils/litellm_logging.py +++ b/litellm/litellm_core_utils/litellm_logging.py @@ -3137,15 +3137,11 @@ class Logging(LiteLLMLoggingBaseClass): else: combined = list(dynamic_success_callbacks) + list(global_callbacks) - verbose_logger.debug( - f"[LANGFUSE DEBUG] Combined callbacks BEFORE filtering: {[self._get_callback_name(cb) for cb in combined]}" - ) - verbose_logger.debug( - f"[LANGFUSE DEBUG] Dynamic callbacks: {[self._get_callback_name(cb) for cb in (dynamic_success_callbacks or [])]}" - ) - verbose_logger.debug( - f"[LANGFUSE DEBUG] Global callbacks: {[self._get_callback_name(cb) for cb in global_callbacks]}" - ) + if verbose_logger.isEnabledFor(logging.DEBUG): + verbose_logger.debug( + "Combined callbacks BEFORE filtering: %s", + [self._get_callback_name(cb) for cb in combined], + ) # Filter duplicate Langfuse loggers to prevent trace leakage # Only keep ONE Langfuse logger per request (prefer dynamic over global) From b753d8e413ce72597c1860170017c5f53deffcbb Mon Sep 17 00:00:00 2001 From: Harshit Jain <48647625+Harshit28j@users.noreply.github.com> Date: Mon, 16 Feb 2026 19:08:42 +0530 Subject: [PATCH 06/10] Update litellm/litellm_core_utils/litellm_logging.py Co-authored-by: greptile-apps[bot] <165735046+greptile-apps[bot]@users.noreply.github.com> --- litellm/litellm_core_utils/litellm_logging.py | 11 ++++++----- 1 file changed, 6 insertions(+), 5 deletions(-) diff --git a/litellm/litellm_core_utils/litellm_logging.py b/litellm/litellm_core_utils/litellm_logging.py index ed8ef994b0f..b7b336dbbe9 100644 --- a/litellm/litellm_core_utils/litellm_logging.py +++ b/litellm/litellm_core_utils/litellm_logging.py @@ -3152,11 +3152,12 @@ class Logging(LiteLLMLoggingBaseClass): cb_name = self._get_callback_name(cb) # Check if this is a Langfuse logger (vanilla or OTEL) - is_langfuse = ( - cb_name.lower() - in ["langfuse", "langfuselogger", "langfuse_otel", "langfuseotellogger"] - or "langfuse" in cb_name.lower() - ) + is_langfuse = cb_name.lower() in [ + "langfuse", + "langfuselogger", + "langfuse_otel", + "langfuseotellogger", + ] if is_langfuse: if langfuse_logger_found is None: From ca4029a715d1e142c29ada1b1cb8df91bd3112e0 Mon Sep 17 00:00:00 2001 From: Harshit Jain <48647625+Harshit28j@users.noreply.github.com> Date: Mon, 16 Feb 2026 13:53:26 +0000 Subject: [PATCH 07/10] fix req changes --- litellm/litellm_core_utils/litellm_logging.py | 16 +++++++++++++--- 1 file changed, 13 insertions(+), 3 deletions(-) diff --git a/litellm/litellm_core_utils/litellm_logging.py b/litellm/litellm_core_utils/litellm_logging.py index b7b336dbbe9..b2aee6bd2a0 100644 --- a/litellm/litellm_core_utils/litellm_logging.py +++ b/litellm/litellm_core_utils/litellm_logging.py @@ -3184,9 +3184,19 @@ class Logging(LiteLLMLoggingBaseClass): f"[LANGFUSE DEBUG] Langfuse logger kept: {self._get_callback_name(langfuse_logger_found)}" ) - # Don't use set() - it breaks deduplication for callback instances - # The filtering logic above already handles deduplication - return filtered + # After Langfuse filtering, deduplicate remaining callbacks + seen = set() + final = [] + for cb in filtered: + cb_id = id(cb) if not isinstance(cb, str) else cb + if cb_id not in seen: + seen.add(cb_id) + final.append(cb) + else: + verbose_logger.debug( + f"LiteLLM Logging: Skipping duplicate callback: {self._get_callback_name(cb)}" + ) + return final def _remove_internal_litellm_callbacks(self, callbacks: List) -> List: """ From 0341b6fa2bdfb22705356dd70d952ba2dfa7da36 Mon Sep 17 00:00:00 2001 From: Harshit Jain <48647625+Harshit28j@users.noreply.github.com> Date: Mon, 16 Feb 2026 19:34:13 +0530 Subject: [PATCH 08/10] Update litellm/litellm_core_utils/litellm_logging.py Co-authored-by: greptile-apps[bot] <165735046+greptile-apps[bot]@users.noreply.github.com> --- litellm/litellm_core_utils/litellm_logging.py | 11 +++++++---- 1 file changed, 7 insertions(+), 4 deletions(-) diff --git a/litellm/litellm_core_utils/litellm_logging.py b/litellm/litellm_core_utils/litellm_logging.py index b2aee6bd2a0..f5b12b9e0f7 100644 --- a/litellm/litellm_core_utils/litellm_logging.py +++ b/litellm/litellm_core_utils/litellm_logging.py @@ -3176,12 +3176,15 @@ class Logging(LiteLLMLoggingBaseClass): # Keep all non-Langfuse callbacks filtered.append(cb) - verbose_logger.debug( - f"[LANGFUSE DEBUG] Filtered callbacks AFTER filtering: {[self._get_callback_name(cb) for cb in filtered]}" - ) + if verbose_logger.isEnabledFor(logging.DEBUG): + verbose_logger.debug( + "[LANGFUSE DEBUG] Filtered callbacks AFTER filtering: %s", + [self._get_callback_name(cb) for cb in filtered], + ) if langfuse_logger_found: verbose_logger.debug( - f"[LANGFUSE DEBUG] Langfuse logger kept: {self._get_callback_name(langfuse_logger_found)}" + "[LANGFUSE DEBUG] Langfuse logger kept: %s", + self._get_callback_name(langfuse_logger_found), ) # After Langfuse filtering, deduplicate remaining callbacks From d61f3ac463954eff434c0837867c57f0de72d28f Mon Sep 17 00:00:00 2001 From: Harshit Jain Date: Fri, 20 Feb 2026 23:00:17 +0530 Subject: [PATCH 09/10] fix: add missing loggin module to avoid failure in logs --- litellm/litellm_core_utils/litellm_logging.py | 1 + 1 file changed, 1 insertion(+) diff --git a/litellm/litellm_core_utils/litellm_logging.py b/litellm/litellm_core_utils/litellm_logging.py index f5b12b9e0f7..59749ef33d8 100644 --- a/litellm/litellm_core_utils/litellm_logging.py +++ b/litellm/litellm_core_utils/litellm_logging.py @@ -4,6 +4,7 @@ import copy import datetime import json +import logging import os import re import subprocess From 1cc185eb1654556b1db808c5885095806895f67e Mon Sep 17 00:00:00 2001 From: Harshit28j Date: Sun, 22 Feb 2026 19:53:18 +0530 Subject: [PATCH 10/10] fix: update opentelemetry with greptile review --- litellm/integrations/opentelemetry.py | 10 + .../integrations/test_opentelemetry.py | 347 ++++++++++++++---- 2 files changed, 290 insertions(+), 67 deletions(-) diff --git a/litellm/integrations/opentelemetry.py b/litellm/integrations/opentelemetry.py index c861555296a..7cdd338c4f7 100644 --- a/litellm/integrations/opentelemetry.py +++ b/litellm/integrations/opentelemetry.py @@ -1234,6 +1234,15 @@ class OpenTelemetry(CustomLogger): ) _parent_context, parent_otel_span = self._get_span_context(kwargs) + # CRITICAL FIX: For langfuse_otel, ALWAYS create primary spans + # Don't use parent spans from other providers as they cause trace corruption + is_langfuse_otel = ( + hasattr(self, "callback_name") and self.callback_name == "langfuse_otel" + ) + if is_langfuse_otel: + parent_otel_span = None # Ignore parent spans from other providers + _parent_context = None + # Decide whether to create a primary span # Always create if no parent span exists (backward compatibility) # OR if USE_OTEL_LITELLM_REQUEST_SPAN is explicitly enabled @@ -1274,6 +1283,7 @@ class OpenTelemetry(CustomLogger): # However, proxy-created spans should be closed here if ( parent_otel_span is not None + and hasattr(parent_otel_span, "name") and parent_otel_span.name == LITELLM_PROXY_REQUEST_SPAN_NAME ): parent_otel_span.end(end_time=self._to_ns(end_time)) diff --git a/tests/test_litellm/integrations/test_opentelemetry.py b/tests/test_litellm/integrations/test_opentelemetry.py index 3da9d9857c9..96b0fae5b84 100644 --- a/tests/test_litellm/integrations/test_opentelemetry.py +++ b/tests/test_litellm/integrations/test_opentelemetry.py @@ -41,7 +41,9 @@ class TestOpenTelemetryGuardrails(unittest.TestCase): } # Create a kwargs dict with standard_logging_object containing guardrail information - kwargs = {"standard_logging_object": {"guardrail_information": [ guardrail_info ]}} + kwargs = { + "standard_logging_object": {"guardrail_information": [guardrail_info]} + } # Call the method otel._create_guardrail_span(kwargs=kwargs, context=None) @@ -195,11 +197,13 @@ class TestOpenTelemetryProviderInitialization(unittest.TestCase): # Assert: The existing provider should still be active current_provider = trace.get_tracer_provider() - assert current_provider is existing_provider, ( - "Existing TracerProvider should be respected and not overridden" - ) + assert ( + current_provider is existing_provider + ), "Existing TracerProvider should be respected and not overridden" - @patch.dict(os.environ, {"LITELLM_OTEL_INTEGRATION_ENABLE_METRICS": "true"}, clear=True) + @patch.dict( + os.environ, {"LITELLM_OTEL_INTEGRATION_ENABLE_METRICS": "true"}, clear=True + ) def test_init_metrics_respects_existing_meter_provider(self): """ Unit test: _init_metrics() should respect existing MeterProvider. @@ -221,11 +225,13 @@ class TestOpenTelemetryProviderInitialization(unittest.TestCase): # Assert: The existing provider should still be active current_provider = metrics.get_meter_provider() - assert current_provider is existing_provider, ( - "Existing MeterProvider should be respected and not overridden" - ) + assert ( + current_provider is existing_provider + ), "Existing MeterProvider should be respected and not overridden" - @patch.dict(os.environ, {"LITELLM_OTEL_INTEGRATION_ENABLE_EVENTS": "true"}, clear=True) + @patch.dict( + os.environ, {"LITELLM_OTEL_INTEGRATION_ENABLE_EVENTS": "true"}, clear=True + ) def test_init_logs_respects_existing_logger_provider(self): """ Unit test: _init_logs() should respect existing LoggerProvider. @@ -247,9 +253,9 @@ class TestOpenTelemetryProviderInitialization(unittest.TestCase): # Assert: The existing provider should still be active current_provider = get_logger_provider() - assert current_provider is existing_provider, ( - "Existing LoggerProvider should be respected and not overridden" - ) + assert ( + current_provider is existing_provider + ), "Existing LoggerProvider should be respected and not overridden" class TestOpenTelemetry(unittest.TestCase): @@ -287,7 +293,9 @@ class TestOpenTelemetry(unittest.TestCase): self.assertEqual(config.exporter, "otlp_http") # When exporter is explicitly set to something other than console, should not override - config_grpc = OpenTelemetryConfig(exporter="grpc", endpoint="https://otel-collector.example.com:443") + config_grpc = OpenTelemetryConfig( + exporter="grpc", endpoint="https://otel-collector.example.com:443" + ) self.assertEqual(config_grpc.exporter, "grpc") # When no endpoint is set, should keep console as default @@ -365,7 +373,9 @@ class TestOpenTelemetry(unittest.TestCase): } # Create a kwargs dict with standard_logging_object containing guardrail information - kwargs = {"standard_logging_object": {"guardrail_information": [ guardrail_info ]}} + kwargs = { + "standard_logging_object": {"guardrail_information": [guardrail_info]} + } # Call the method otel._create_guardrail_span(kwargs=kwargs, context=None) @@ -723,7 +733,6 @@ class TestOpenTelemetry(unittest.TestCase): # But other attributes from OTEL_RESOURCE_ATTRIBUTES should still be present self.assertEqual(attributes.get("extra.attr"), "extra-value") - def test_handle_success_spans_only(self): # make sure neither events nor metrics is on os.environ.pop("LITELLM_OTEL_INTEGRATION_ENABLE_EVENTS", None) @@ -790,7 +799,9 @@ class TestOpenTelemetry(unittest.TestCase): logs = log_exporter.get_finished_logs() self.assertFalse(logs, "Did not expect any logs") - @patch.dict(os.environ, {"LITELLM_OTEL_INTEGRATION_ENABLE_METRICS": "true"}, clear=True) + @patch.dict( + os.environ, {"LITELLM_OTEL_INTEGRATION_ENABLE_METRICS": "true"}, clear=True + ) def test_handle_success_spans_and_metrics(self): # ─── build in‐memory OTEL providers/exporters ───────────────────────────── span_exporter = InMemorySpanExporter() @@ -1458,9 +1469,7 @@ class TestOpenTelemetryExternalSpan(unittest.TestCase): """Set up common test fixtures""" self.span_exporter = InMemorySpanExporter() self.tracer_provider = TracerProvider() - self.tracer_provider.add_span_processor( - SimpleSpanProcessor(self.span_exporter) - ) + self.tracer_provider.add_span_processor(SimpleSpanProcessor(self.span_exporter)) # Don't set global tracer provider - instead, get tracers directly from our provider # This avoids "Overriding of current TracerProvider is not allowed" warnings @@ -1514,7 +1523,7 @@ class TestOpenTelemetryExternalSpan(unittest.TestCase): self.assertTrue( parent_span.is_recording(), - "External span should be recording before completion calls" + "External span should be recording before completion calls", ) # First completion call @@ -1525,7 +1534,7 @@ class TestOpenTelemetryExternalSpan(unittest.TestCase): # Verify parent span is still recording self.assertTrue( parent_span.is_recording(), - "External span should still be recording after first completion" + "External span should still be recording after first completion", ) # Second completion call @@ -1536,7 +1545,7 @@ class TestOpenTelemetryExternalSpan(unittest.TestCase): # Verify parent span is still recording self.assertTrue( parent_span.is_recording(), - "External span should still be recording after second completion" + "External span should still be recording after second completion", ) # After exiting context, verify spans @@ -1547,23 +1556,25 @@ class TestOpenTelemetryExternalSpan(unittest.TestCase): self.assertEqual( span.context.trace_id, parent_trace_id, - f"Span {span.name} should have same trace_id as parent" + f"Span {span.name} should have same trace_id as parent", ) # Should have external_parent_span parent_spans = self._get_spans_by_name("external_parent_span") - self.assertEqual(len(parent_spans), 1, "Should have exactly one external_parent_span") + self.assertEqual( + len(parent_spans), 1, "Should have exactly one external_parent_span" + ) # Verify LiteLLM set attributes on external parent span parent_span_finished = parent_spans[0] self.assertIsNotNone( parent_span_finished.attributes, - "Parent span should have attributes set by LiteLLM" + "Parent span should have attributes set by LiteLLM", ) self.assertIn( "gen_ai.request.model", parent_span_finished.attributes, - "Parent span should have model attribute from LiteLLM" + "Parent span should have model attribute from LiteLLM", ) # Should have raw_gen_ai_request spans (if message_logging is on) @@ -1575,7 +1586,7 @@ class TestOpenTelemetryExternalSpan(unittest.TestCase): self.assertEqual( len(litellm_spans), 0, - "Should NOT have litellm_request spans when USE_OTEL_LITELLM_REQUEST_SPAN=false" + "Should NOT have litellm_request spans when USE_OTEL_LITELLM_REQUEST_SPAN=false", ) # Verify raw_gen_ai_request spans are direct children of external span @@ -1583,7 +1594,7 @@ class TestOpenTelemetryExternalSpan(unittest.TestCase): self.assertEqual( raw_span.parent.span_id if raw_span.parent else None, parent_span_id, - f"raw_gen_ai_request should be direct child of external_parent_span" + "raw_gen_ai_request should be direct child of external_parent_span", ) @patch.dict(os.environ, {"USE_OTEL_LITELLM_REQUEST_SPAN": "true"}, clear=False) @@ -1619,7 +1630,7 @@ class TestOpenTelemetryExternalSpan(unittest.TestCase): # Verify parent span is still recording self.assertTrue( parent_span.is_recording(), - "External span should still be recording after first completion" + "External span should still be recording after first completion", ) # Second completion call @@ -1630,7 +1641,7 @@ class TestOpenTelemetryExternalSpan(unittest.TestCase): # Verify parent span is still recording self.assertTrue( parent_span.is_recording(), - "External span should still be recording after second completion" + "External span should still be recording after second completion", ) # After exiting context, verify spans @@ -1641,7 +1652,7 @@ class TestOpenTelemetryExternalSpan(unittest.TestCase): self.assertEqual( span.context.trace_id, parent_trace_id, - f"Span {span.name} should have same trace_id as parent" + f"Span {span.name} should have same trace_id as parent", ) # Should have litellm_request spans (USE_OTEL_LITELLM_REQUEST_SPAN=true) @@ -1649,7 +1660,7 @@ class TestOpenTelemetryExternalSpan(unittest.TestCase): self.assertEqual( len(litellm_spans), 2, - "Should have 2 litellm_request spans when USE_OTEL_LITELLM_REQUEST_SPAN=true" + "Should have 2 litellm_request spans when USE_OTEL_LITELLM_REQUEST_SPAN=true", ) # Verify litellm_request spans are children of external span @@ -1657,7 +1668,7 @@ class TestOpenTelemetryExternalSpan(unittest.TestCase): self.assertEqual( litellm_span.parent.span_id if litellm_span.parent else None, parent_span_id, - "litellm_request should be child of external_parent_span" + "litellm_request should be child of external_parent_span", ) # Verify raw_gen_ai_request spans (if present) are children of litellm_request @@ -1668,7 +1679,7 @@ class TestOpenTelemetryExternalSpan(unittest.TestCase): self.assertIn( raw_span.parent.span_id if raw_span.parent else None, litellm_span_ids, - "raw_gen_ai_request should be child of litellm_request" + "raw_gen_ai_request should be child of litellm_request", ) @patch.dict(os.environ, {"USE_OTEL_LITELLM_REQUEST_SPAN": "false"}, clear=False) @@ -1706,7 +1717,7 @@ class TestOpenTelemetryExternalSpan(unittest.TestCase): # Verify parent span is still recording after each call self.assertTrue( parent_span.is_recording(), - f"External span should still be recording after completion #{i+1}" + f"External span should still be recording after completion #{i+1}", ) # Verify all spans have the same trace_id @@ -1715,19 +1726,21 @@ class TestOpenTelemetryExternalSpan(unittest.TestCase): self.assertEqual( span.context.trace_id, parent_trace_id, - f"All spans should belong to the same trace" + "All spans should belong to the same trace", ) # Should have the external parent span parent_spans = self._get_spans_by_name("external_parent_span") - self.assertEqual(len(parent_spans), 1, "Should have exactly one external_parent_span") + self.assertEqual( + len(parent_spans), 1, "Should have exactly one external_parent_span" + ) # Verify LiteLLM set attributes on external parent span parent_span_finished = parent_spans[0] self.assertIn( "gen_ai.request.model", parent_span_finished.attributes, - "Parent span should have model attribute from LiteLLM" + "Parent span should have model attribute from LiteLLM", ) @patch.dict(os.environ, {"USE_OTEL_LITELLM_REQUEST_SPAN": "false"}, clear=False) @@ -1759,7 +1772,9 @@ class TestOpenTelemetryExternalSpan(unittest.TestCase): # Verify the span is in global context current_span = trace.get_current_span() - self.assertEqual(current_span, parent_span, "Span should be in global context") + self.assertEqual( + current_span, parent_span, "Span should be in global context" + ) # Make completion call start_time = datetime.utcnow() @@ -1769,7 +1784,7 @@ class TestOpenTelemetryExternalSpan(unittest.TestCase): # Verify parent span is still recording self.assertTrue( parent_span.is_recording(), - "External span from global context should not be closed" + "External span from global context should not be closed", ) # Verify trace structure @@ -1778,7 +1793,7 @@ class TestOpenTelemetryExternalSpan(unittest.TestCase): self.assertEqual( span.context.trace_id, parent_trace_id, - "All spans should have the same trace_id" + "All spans should have the same trace_id", ) @patch.dict(os.environ, {"USE_OTEL_LITELLM_REQUEST_SPAN": "false"}, clear=False) @@ -1793,7 +1808,9 @@ class TestOpenTelemetryExternalSpan(unittest.TestCase): """ # Initialize OpenTelemetry otel = OpenTelemetry(tracer_provider=self.tracer_provider) - otel.message_logging = True # Enable message logging to get raw_gen_ai_request spans + otel.message_logging = ( + True # Enable message logging to get raw_gen_ai_request spans + ) # Load test data kwargs, response_obj = self._create_test_kwargs_and_response() @@ -1824,7 +1841,7 @@ class TestOpenTelemetryExternalSpan(unittest.TestCase): self.assertEqual( raw_span.parent.span_id if raw_span.parent else None, parent_span_id, - "raw_gen_ai_request should be child of external_parent_span" + "raw_gen_ai_request should be child of external_parent_span", ) @patch.dict(os.environ, {"USE_OTEL_LITELLM_REQUEST_SPAN": "false"}, clear=False) @@ -1864,7 +1881,7 @@ class TestOpenTelemetryExternalSpan(unittest.TestCase): # Verify parent span is still recording self.assertTrue( parent_span.is_recording(), - "External span should still be recording even after failure" + "External span should still be recording even after failure", ) # Verify trace structure @@ -1875,40 +1892,42 @@ class TestOpenTelemetryExternalSpan(unittest.TestCase): self.assertEqual( span.context.trace_id, parent_trace_id, - "All spans should have the same trace_id even on failure" + "All spans should have the same trace_id even on failure", ) # Should have external_parent_span parent_spans = self._get_spans_by_name("external_parent_span") - self.assertEqual(len(parent_spans), 1, "Should have exactly one external_parent_span") + self.assertEqual( + len(parent_spans), 1, "Should have exactly one external_parent_span" + ) # Verify LiteLLM set attributes on external parent span even on failure parent_span_finished = parent_spans[0] self.assertIn( "gen_ai.request.model", parent_span_finished.attributes, - "Parent span should have model attribute from LiteLLM even on failure" + "Parent span should have model attribute from LiteLLM even on failure", ) class TestOpenTelemetrySemanticConventions138(unittest.TestCase): """ Test suite for OpenTelemetry 1.38 Semantic Conventions compliance. - - These tests verify that LiteLLM emits span attributes following the + + These tests verify that LiteLLM emits span attributes following the OpenTelemetry GenAI semantic conventions v1.38, including: - gen_ai.input.messages (JSON string with parts array) - gen_ai.output.messages (JSON string with parts array) - gen_ai.usage.input_tokens / output_tokens (new naming) - gen_ai.response.finish_reasons (JSON array) - + See: https://github.com/BerriAI/litellm/issues/17794 """ def test_input_messages_uses_parts_structure(self): """ Test that gen_ai.input.messages uses the OTEL 1.38 parts array structure. - + Expected format: [{"role": "user", "parts": [{"type": "text", "content": "Hello"}]}] """ @@ -1943,14 +1962,19 @@ class TestOpenTelemetrySemanticConventions138(unittest.TestCase): # Find the call that set gen_ai.input.messages input_messages_calls = [ - call for call in mock_span.set_attribute.call_args_list + call + for call in mock_span.set_attribute.call_args_list if call[0][0] == "gen_ai.input.messages" ] - self.assertEqual(len(input_messages_calls), 1, "Should have exactly one gen_ai.input.messages attribute") - + self.assertEqual( + len(input_messages_calls), + 1, + "Should have exactly one gen_ai.input.messages attribute", + ) + input_messages_value = input_messages_calls[0][0][1] parsed = json.loads(input_messages_value) - + # Verify structure self.assertIsInstance(parsed, list) self.assertEqual(len(parsed), 1) @@ -1962,7 +1986,7 @@ class TestOpenTelemetrySemanticConventions138(unittest.TestCase): def test_output_messages_uses_parts_structure(self): """ Test that gen_ai.output.messages uses the OTEL 1.38 parts array structure. - + Expected format: [{"role": "assistant", "parts": [{"type": "text", "content": "Hi!"}], "finish_reason": "stop"}] """ @@ -1997,14 +2021,19 @@ class TestOpenTelemetrySemanticConventions138(unittest.TestCase): # Find the call that set gen_ai.output.messages output_messages_calls = [ - call for call in mock_span.set_attribute.call_args_list + call + for call in mock_span.set_attribute.call_args_list if call[0][0] == "gen_ai.output.messages" ] - self.assertEqual(len(output_messages_calls), 1, "Should have exactly one gen_ai.output.messages attribute") - + self.assertEqual( + len(output_messages_calls), + 1, + "Should have exactly one gen_ai.output.messages attribute", + ) + output_messages_value = output_messages_calls[0][0][1] parsed = json.loads(output_messages_value) - + # Verify structure self.assertIsInstance(parsed, list) self.assertEqual(len(parsed), 1) @@ -2039,7 +2068,11 @@ class TestOpenTelemetrySemanticConventions138(unittest.TestCase): "id": "test-response-id", "model": "gpt-4", "choices": [], - "usage": {"prompt_tokens": 100, "completion_tokens": 50, "total_tokens": 150}, + "usage": { + "prompt_tokens": 100, + "completion_tokens": 50, + "total_tokens": 150, + }, } otel.set_attributes(span=mock_span, kwargs=kwargs, response_obj=response_obj) @@ -2052,7 +2085,7 @@ class TestOpenTelemetrySemanticConventions138(unittest.TestCase): def test_finish_reasons_is_json_array(self): """ Test that gen_ai.response.finish_reasons is a proper JSON array. - + Expected: '["stop"]' (not "['stop']") """ otel = OpenTelemetry() @@ -2074,7 +2107,10 @@ class TestOpenTelemetrySemanticConventions138(unittest.TestCase): "id": "test-response-id", "model": "gpt-4", "choices": [ - {"finish_reason": "stop", "message": {"role": "assistant", "content": "Hi"}}, + { + "finish_reason": "stop", + "message": {"role": "assistant", "content": "Hi"}, + }, ], "usage": {"prompt_tokens": 10, "completion_tokens": 20, "total_tokens": 30}, } @@ -2083,13 +2119,18 @@ class TestOpenTelemetrySemanticConventions138(unittest.TestCase): # Find the call that set gen_ai.response.finish_reasons finish_reasons_calls = [ - call for call in mock_span.set_attribute.call_args_list + call + for call in mock_span.set_attribute.call_args_list if call[0][0] == "gen_ai.response.finish_reasons" ] - self.assertEqual(len(finish_reasons_calls), 1, "Should have exactly one gen_ai.response.finish_reasons attribute") - + self.assertEqual( + len(finish_reasons_calls), + 1, + "Should have exactly one gen_ai.response.finish_reasons attribute", + ) + finish_reasons_value = finish_reasons_calls[0][0][1] - + # Verify it's valid JSON (not Python repr) parsed = json.loads(finish_reasons_value) self.assertEqual(parsed, ["stop"]) @@ -2123,3 +2164,175 @@ class TestOpenTelemetrySemanticConventions138(unittest.TestCase): otel.set_attributes(span=mock_span, kwargs=kwargs, response_obj=response_obj) mock_span.set_attribute.assert_any_call("gen_ai.operation.name", "chat") + + def test_handle_failure_langfuse_otel_nulls_parent_span(self): + """ + For langfuse_otel, _handle_failure should ignore parent spans from other providers + and create a root-level error span (symmetric with _handle_success). + """ + span_exporter = InMemorySpanExporter() + tracer_provider = TracerProvider() + tracer_provider.add_span_processor(SimpleSpanProcessor(span_exporter)) + + otel = OpenTelemetry( + callback_name="langfuse_otel", + tracer_provider=tracer_provider, + ) + otel.tracer = tracer_provider.get_tracer("litellm") + + other_tracer = tracer_provider.get_tracer("other_provider") + other_span = other_tracer.start_span("other_provider_span") + + start = datetime.utcnow() + end = start + timedelta(seconds=1) + + kwargs = { + "model": "gpt-4", + "messages": [{"role": "user", "content": "Hello"}], + "optional_params": {}, + "litellm_params": { + "custom_llm_provider": "openai", + "metadata": {"litellm_parent_otel_span": other_span}, + }, + "standard_logging_object": { + "id": "test-id", + "call_type": "completion", + "metadata": {}, + }, + "exception": Exception("test error"), + } + + otel._handle_failure(kwargs, None, start, end) + + other_span.end() + + spans = span_exporter.get_finished_spans() + failure_spans = [s for s in spans if s.name != "other_provider_span"] + + self.assertTrue(failure_spans, "Expected at least one failure span") + for span in failure_spans: + self.assertIsNone( + span.parent, + f"langfuse_otel failure span should be a root span, but has parent: {span.parent}", + ) + + def test_handle_failure_non_langfuse_preserves_parent_span(self): + """ + For non-langfuse_otel callbacks, _handle_failure should still use parent spans normally. + """ + span_exporter = InMemorySpanExporter() + tracer_provider = TracerProvider() + tracer_provider.add_span_processor(SimpleSpanProcessor(span_exporter)) + + otel = OpenTelemetry(tracer_provider=tracer_provider) + otel.tracer = tracer_provider.get_tracer("litellm") + + parent_span = otel.tracer.start_span("parent_span") + + start = datetime.utcnow() + end = start + timedelta(seconds=1) + + kwargs = { + "model": "gpt-4", + "messages": [{"role": "user", "content": "Hello"}], + "optional_params": {}, + "litellm_params": { + "custom_llm_provider": "openai", + "metadata": {"litellm_parent_otel_span": parent_span}, + }, + "standard_logging_object": { + "id": "test-id", + "call_type": "completion", + "metadata": {}, + }, + "exception": Exception("test error"), + } + + with patch.dict(os.environ, {"USE_OTEL_LITELLM_REQUEST_SPAN": "true"}): + otel._handle_failure(kwargs, None, start, end) + + parent_span.end() + + spans = span_exporter.get_finished_spans() + child_spans = [s for s in spans if s.name != "parent_span"] + + self.assertTrue(child_spans, "Expected at least one child failure span") + for span in child_spans: + self.assertIsNotNone( + span.parent, + "Non-langfuse_otel failure span should have a parent", + ) + + def test_handle_failure_hasattr_guard_on_parent_name(self): + """ + _handle_failure should not raise AttributeError when parent_otel_span + lacks a 'name' attribute (e.g., NonRecordingSpan). + """ + otel = OpenTelemetry() + otel.tracer = MagicMock() + mock_span = MagicMock() + otel.tracer.start_span.return_value = mock_span + parent_without_name = MagicMock() + del parent_without_name.name + + start = datetime.utcnow() + end = start + timedelta(seconds=1) + + kwargs = { + "model": "gpt-4", + "messages": [{"role": "user", "content": "Hello"}], + "optional_params": {}, + "litellm_params": { + "custom_llm_provider": "openai", + "metadata": {"litellm_parent_otel_span": parent_without_name}, + }, + "standard_logging_object": { + "id": "test-id", + "call_type": "completion", + "metadata": {}, + }, + } + + try: + otel._handle_failure(kwargs, None, start, end) + except AttributeError as e: + self.fail( + f"_handle_failure raised AttributeError on parent span without 'name': {e}" + ) + + def test_handle_failure_creates_error_span(self): + """ + _handle_failure should create a span with ERROR status. + """ + span_exporter = InMemorySpanExporter() + tracer_provider = TracerProvider() + tracer_provider.add_span_processor(SimpleSpanProcessor(span_exporter)) + + otel = OpenTelemetry(tracer_provider=tracer_provider) + otel.tracer = tracer_provider.get_tracer("litellm") + + start = datetime.utcnow() + end = start + timedelta(seconds=1) + + kwargs = { + "model": "gpt-4", + "messages": [{"role": "user", "content": "Hello"}], + "optional_params": {}, + "litellm_params": {"custom_llm_provider": "openai"}, + "standard_logging_object": { + "id": "test-id", + "call_type": "completion", + "metadata": {}, + }, + "exception": Exception("test error"), + } + + otel._handle_failure(kwargs, None, start, end) + + spans = span_exporter.get_finished_spans() + self.assertTrue(spans, "Expected at least one span") + + from opentelemetry.trace import StatusCode + + error_spans = [s for s in spans if s.status.status_code == StatusCode.ERROR] + self.assertTrue(error_spans, "Expected at least one span with ERROR status")