Fix async failure path not logging to Langfuse (proxy bug)

The proxy uses async_failure_handler → LangfusePromptManagement.async_log_failure_event(),
which silently returned when standard_logging_object was None. This meant failed LLM calls
never created traces in Langfuse. Remove the early return and fall back to extracting the
error message from kwargs["exception"] when standard_logging_object is unavailable.

Co-Authored-By: Claude Opus 4.6 <noreply@anthropic.com>
This commit is contained in:
Harshit28j 2026-03-02 20:20:27 +05:30
parent 83cab3f54a
commit 016a4fd608
2 changed files with 136 additions and 3 deletions

View file

@ -338,14 +338,17 @@ class LangfusePromptManagement(LangFuseLogger, PromptManagementBase, CustomLogge
Optional[StandardLoggingPayload],
kwargs.get("standard_logging_object", None),
)
if standard_logging_object is None:
return
status_message = str(kwargs.get("exception", "Unknown error"))
if standard_logging_object is not None:
status_message = standard_logging_object.get(
"error_str", status_message
)
langfuse_logger_to_use.log_event_on_langfuse(
start_time=start_time,
end_time=end_time,
response_obj=None,
user_id=kwargs.get("user", None),
status_message=standard_logging_object["error_str"],
status_message=status_message,
level="ERROR",
kwargs=kwargs,
)

View file

@ -669,6 +669,136 @@ def test_failure_handler_langfuse_kwargs_excludes_original_response():
litellm.failure_callback = original_failure_callback
@pytest.mark.asyncio
async def test_async_log_failure_event_logs_to_langfuse():
"""
Test that LangfusePromptManagement.async_log_failure_event() calls
log_event_on_langfuse with level=ERROR even when standard_logging_object
is present. This is the code path the proxy uses for failed LLM calls.
"""
from litellm.integrations.langfuse.langfuse_prompt_management import (
LangfusePromptManagement,
)
mock_langfuse_module = MagicMock()
mock_langfuse_module.version.__version__ = "3.0.0"
with patch.dict(
"os.environ",
{
"LANGFUSE_SECRET_KEY": "test-secret",
"LANGFUSE_PUBLIC_KEY": "test-public",
"LANGFUSE_HOST": "https://test.langfuse.com",
},
), patch.dict("sys.modules", {"langfuse": mock_langfuse_module}):
prompt_mgmt = LangfusePromptManagement()
# Mock the langfuse logger returned by get_langfuse_logger_for_request
mock_logger = MagicMock()
mock_logger.log_event_on_langfuse.return_value = {
"trace_id": "mock-trace",
"generation_id": "mock-gen",
}
with patch(
"litellm.integrations.langfuse.langfuse_prompt_management.LangFuseHandler"
) as mock_handler:
mock_handler.get_langfuse_logger_for_request.return_value = mock_logger
kwargs = {
"litellm_params": {
"metadata": {"session_id": "test-session-fail"},
},
"litellm_call_id": "call-fail-123",
"user": "test-user",
"exception": Exception("API error: model not found"),
"standard_logging_object": {
"error_str": "API error: model not found",
"trace_id": "std-trace-fail",
"metadata": {},
},
}
await prompt_mgmt.async_log_failure_event(
kwargs=kwargs,
response_obj=None,
start_time=datetime.datetime.utcnow(),
end_time=datetime.datetime.utcnow(),
)
# Verify log_event_on_langfuse was called
assert mock_logger.log_event_on_langfuse.called, (
"log_event_on_langfuse was not called for failure event"
)
call_kwargs = mock_logger.log_event_on_langfuse.call_args[1]
assert call_kwargs["level"] == "ERROR"
assert call_kwargs["status_message"] == "API error: model not found"
assert call_kwargs["response_obj"] is None
@pytest.mark.asyncio
async def test_async_log_failure_event_works_without_standard_logging_object():
"""
Test that async_log_failure_event() still logs to Langfuse even when
standard_logging_object is None (e.g. when get_standard_logging_object_payload
threw an exception). This is the critical fix before, it silently returned.
"""
from litellm.integrations.langfuse.langfuse_prompt_management import (
LangfusePromptManagement,
)
mock_langfuse_module = MagicMock()
mock_langfuse_module.version.__version__ = "3.0.0"
with patch.dict(
"os.environ",
{
"LANGFUSE_SECRET_KEY": "test-secret",
"LANGFUSE_PUBLIC_KEY": "test-public",
"LANGFUSE_HOST": "https://test.langfuse.com",
},
), patch.dict("sys.modules", {"langfuse": mock_langfuse_module}):
prompt_mgmt = LangfusePromptManagement()
mock_logger = MagicMock()
mock_logger.log_event_on_langfuse.return_value = {
"trace_id": "mock-trace",
"generation_id": "mock-gen",
}
with patch(
"litellm.integrations.langfuse.langfuse_prompt_management.LangFuseHandler"
) as mock_handler:
mock_handler.get_langfuse_logger_for_request.return_value = mock_logger
kwargs = {
"litellm_params": {
"metadata": {"session_id": "test-session-no-slo"},
},
"litellm_call_id": "call-no-slo-456",
"user": "test-user",
"exception": Exception("InternalServerError: something broke"),
"standard_logging_object": None, # This is the key — it's None
}
await prompt_mgmt.async_log_failure_event(
kwargs=kwargs,
response_obj=None,
start_time=datetime.datetime.utcnow(),
end_time=datetime.datetime.utcnow(),
)
# CRITICAL: log_event_on_langfuse MUST still be called
assert mock_logger.log_event_on_langfuse.called, (
"log_event_on_langfuse was NOT called when standard_logging_object "
"is None — failure trace would be silently dropped"
)
call_kwargs = mock_logger.log_event_on_langfuse.call_args[1]
assert call_kwargs["level"] == "ERROR"
# Falls back to exception from kwargs
assert "InternalServerError" in call_kwargs["status_message"]
def test_max_langfuse_clients_limit():
"""
Test that the max langfuse clients limit is respected when initializing multiple clients