From c98b125851aa646367ea835709fc026273ca9c7f Mon Sep 17 00:00:00 2001 From: Ishaan Jaff Date: Thu, 6 Nov 2025 13:53:53 -0800 Subject: [PATCH] [Feat] OTEL - Log Cost Breakdown on OTEL Logger (#16334) * add gen_ai cost metrics * TestOpenTelemetryCostBreakdown * fix QA check * validate_redacted_message_span_attributes --- litellm/integrations/opentelemetry.py | 11 +++ litellm/llms/xai/responses/transformation.py | 2 +- ...odel_prices_and_context_window_backup.json | 2 - .../test_otel_logging.py | 8 +- .../integrations/test_opentelemetry.py | 92 +++++++++++++++++++ 5 files changed, 109 insertions(+), 6 deletions(-) diff --git a/litellm/integrations/opentelemetry.py b/litellm/integrations/opentelemetry.py index 9a17244a06d..53b7825b3d3 100644 --- a/litellm/integrations/opentelemetry.py +++ b/litellm/integrations/opentelemetry.py @@ -10,6 +10,7 @@ from litellm.litellm_core_utils.safe_json_dumps import safe_dumps from litellm.types.services import ServiceLoggerPayload from litellm.types.utils import ( ChatCompletionMessageToolCall, + CostBreakdown, Function, StandardCallbackDynamicParams, StandardLoggingPayload, @@ -1076,6 +1077,16 @@ class OpenTelemetry(CustomLogger): self.safe_set_attribute( span=span, key="hidden_params", value=safe_dumps(hidden_params) ) + # Cost breakdown tracking + cost_breakdown: Optional[CostBreakdown] = standard_logging_payload.get("cost_breakdown") + if cost_breakdown: + for key, value in cost_breakdown.items(): + if value is not None: + self.safe_set_attribute( + span=span, + key=f"gen_ai.cost.{key}", + value=value, + ) ############################################# ########## LLM Request Attributes ########### ############################################# diff --git a/litellm/llms/xai/responses/transformation.py b/litellm/llms/xai/responses/transformation.py index 5767177b7a6..bd422c8d81e 100644 --- a/litellm/llms/xai/responses/transformation.py +++ b/litellm/llms/xai/responses/transformation.py @@ -85,7 +85,7 @@ class XAIResponsesAPIConfig(OpenAIResponsesAPIConfig): # XAI supports code_interpreter but doesn't use the container field # Keep only the type field verbose_logger.debug( - f"XAI: Transforming code_interpreter tool, removing container field" + "XAI: Transforming code_interpreter tool, removing container field" ) transformed_tools.append({"type": "code_interpreter"}) else: diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 49236741fff..5b901946b39 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -3704,7 +3704,6 @@ "output_cost_per_token": 2.75e-05, "source": "https://azure.microsoft.com/en-us/blog/grok-4-is-now-available-in-azure-ai-foundry-unlock-frontier-intelligence-and-business-ready-capabilities/", "supports_function_calling": true, - "supports_reasoning": true, "supports_response_schema": true, "supports_tool_choice": true, "supports_web_search": true @@ -3732,7 +3731,6 @@ "mode": "chat", "source": "https://techcommunity.microsoft.com/blog/azure-ai-foundry-blog/announcing-the-grok-4-fast-models-from-xai-now-available-in-azure-ai-foundry/4456701", "supports_function_calling": true, - "supports_reasoning": true, "supports_response_schema": true, "supports_tool_choice": true, "supports_web_search": true diff --git a/tests/logging_callback_tests/test_otel_logging.py b/tests/logging_callback_tests/test_otel_logging.py index ecbcc8734f5..8d1da0439d3 100644 --- a/tests/logging_callback_tests/test_otel_logging.py +++ b/tests/logging_callback_tests/test_otel_logging.py @@ -281,11 +281,13 @@ def validate_redacted_message_span_attributes(span): _all_attributes ), f"Missing required attributes: {required_set - _all_attributes}" - # Check that any additional attributes are metadata fields (start with "metadata.") + # Check that any additional attributes are metadata fields (start with "metadata.") or cost fields non_required_attrs = _all_attributes - required_set for attr in non_required_attrs: - assert attr.startswith("metadata.") or attr.startswith( - "hidden_params" + assert ( + attr.startswith("metadata.") + or attr.startswith("hidden_params") + or attr.startswith("gen_ai.cost.") ), f"Non-metadata attribute found: {attr}" pass diff --git a/tests/test_litellm/integrations/test_opentelemetry.py b/tests/test_litellm/integrations/test_opentelemetry.py index 2774d168701..58fccf5e42d 100644 --- a/tests/test_litellm/integrations/test_opentelemetry.py +++ b/tests/test_litellm/integrations/test_opentelemetry.py @@ -79,6 +79,98 @@ class TestOpenTelemetryGuardrails(unittest.TestCase): otel.tracer.start_span.assert_not_called() +class TestOpenTelemetryCostBreakdown(unittest.TestCase): + def test_cost_breakdown_emitted_to_otel_span(self): + """ + Test that cost breakdown from StandardLoggingPayload is emitted to OpenTelemetry span attributes. + """ + otel = OpenTelemetry() + mock_span = MagicMock() + + cost_breakdown = { + "input_cost": 0.001, + "output_cost": 0.002, + "total_cost": 0.003, + "tool_usage_cost": 0.0001, + "original_cost": 0.004, + "discount_percent": 0.25, + "discount_amount": 0.001, + } + + kwargs = { + "model": "gpt-4", + "messages": [{"role": "user", "content": "Hello"}], + "optional_params": {}, + "litellm_params": {"custom_llm_provider": "openai"}, + "standard_logging_object": { + "id": "test-id", + "call_type": "completion", + "metadata": {}, + "cost_breakdown": cost_breakdown, + }, + } + + response_obj = { + "id": "test-response-id", + "model": "gpt-4", + "choices": [], + "usage": {"prompt_tokens": 10, "completion_tokens": 20, "total_tokens": 30}, + } + + otel.set_attributes(span=mock_span, kwargs=kwargs, response_obj=response_obj) + + mock_span.set_attribute.assert_any_call("gen_ai.cost.input_cost", 0.001) + mock_span.set_attribute.assert_any_call("gen_ai.cost.output_cost", 0.002) + mock_span.set_attribute.assert_any_call("gen_ai.cost.total_cost", 0.003) + mock_span.set_attribute.assert_any_call("gen_ai.cost.tool_usage_cost", 0.0001) + mock_span.set_attribute.assert_any_call("gen_ai.cost.original_cost", 0.004) + mock_span.set_attribute.assert_any_call("gen_ai.cost.discount_percent", 0.25) + mock_span.set_attribute.assert_any_call("gen_ai.cost.discount_amount", 0.001) + + def test_cost_breakdown_with_partial_fields(self): + """ + Test that cost breakdown works correctly when only some fields are present. + """ + otel = OpenTelemetry() + mock_span = MagicMock() + + cost_breakdown = { + "input_cost": 0.001, + "output_cost": 0.002, + "total_cost": 0.003, + } + + kwargs = { + "model": "gpt-4", + "messages": [{"role": "user", "content": "Hello"}], + "optional_params": {}, + "litellm_params": {"custom_llm_provider": "openai"}, + "standard_logging_object": { + "id": "test-id", + "call_type": "completion", + "metadata": {}, + "cost_breakdown": cost_breakdown, + }, + } + + response_obj = { + "id": "test-response-id", + "model": "gpt-4", + "choices": [], + "usage": {"prompt_tokens": 10, "completion_tokens": 20, "total_tokens": 30}, + } + + otel.set_attributes(span=mock_span, kwargs=kwargs, response_obj=response_obj) + + mock_span.set_attribute.assert_any_call("gen_ai.cost.input_cost", 0.001) + mock_span.set_attribute.assert_any_call("gen_ai.cost.output_cost", 0.002) + mock_span.set_attribute.assert_any_call("gen_ai.cost.total_cost", 0.003) + + call_args_list = [call[0] for call in mock_span.set_attribute.call_args_list] + assert ("gen_ai.cost.tool_usage_cost", 0.0001) not in call_args_list + assert ("gen_ai.cost.original_cost", 0.004) not in call_args_list + + class TestOpenTelemetry(unittest.TestCase): POLL_INTERVAL = 0.05 POLL_TIMEOUT = 2.0