mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-11 03:38:38 +00:00
fix(otel, arize): report unknown cost instead of a fake 0 on pricing failure
When LiteLLM cannot price a call (model newer than the bundled cost map), the standard logging payload gets response_cost = 0.0 together with response_cost_failure_debug_info. The span writers only skipped None, so OTel v2 spans carried litellm.cost.total = 0 and Arize/Phoenix spans carried llm.cost.total = 0 — indistinguishable from a genuinely free model. Langfuse prefers an ingested cost over its own price knowledge, so every call to an unpriced model showed $0 there (issue #44186). When response_cost_failure_debug_info is present, the cost is now reported as unknown (None) instead of 0 in both the OTel v2 span payload builders and the Arize attribute writer. A genuine 0 without the failure marker (free model) is still emitted as 0.
This commit is contained in:
parent
5e6ce0f6db
commit
70cb219dfa
4 changed files with 139 additions and 127 deletions
|
|
@ -897,6 +897,14 @@ def _set_response_cost_attr(span: "Span", standard_logging_payload) -> None:
|
|||
cost: Final = standard_logging_payload.get("response_cost")
|
||||
if cost is None:
|
||||
return
|
||||
# A 0 cost is ambiguous between "genuinely free" and "pricing failed".
|
||||
# When the pricing-failure debug info is present, omit the attributes
|
||||
# rather than reporting a fake $0 that cost backends cannot distinguish
|
||||
# from a free model (issue #44186) — e.g. Langfuse prefers an ingested
|
||||
# cost over its own price knowledge, so a fake 0 hides real spend there.
|
||||
failure_debug: Final = standard_logging_payload.get("response_cost_failure_debug_info")
|
||||
if failure_debug:
|
||||
return
|
||||
try:
|
||||
cost_value: Final = float(cost)
|
||||
except (TypeError, ValueError):
|
||||
|
|
|
|||
|
|
@ -472,7 +472,14 @@ class LLMCallSpanData:
|
|||
usage=LLMUsage.from_standard_logging_payload(payload),
|
||||
finish_reasons=finish_reasons,
|
||||
error=_parse_error(payload),
|
||||
response_cost=as_float(payload.get("response_cost")),
|
||||
# response_cost = 0 is ambiguous between a free model and a
|
||||
# pricing failure; when the pricing-failure debug info is
|
||||
# present, report None (unknown) instead of a fake 0 (issue
|
||||
# #44186) — e.g. Langfuse prefers an ingested cost over its own
|
||||
# price knowledge, so a fake 0 hides real spend there.
|
||||
response_cost=(
|
||||
as_float(payload.get("response_cost")) if not payload.get("response_cost_failure_debug_info") else None
|
||||
),
|
||||
cost=LLMCost.from_breakdown(cast("Mapping[str, object] | None", payload.get("cost_breakdown"))),
|
||||
server=ServerInfo.from_api_base(context.api_base),
|
||||
identity=context.identity,
|
||||
|
|
@ -566,7 +573,10 @@ class MCPToolCallSpanData:
|
|||
_json_or_none(meta.get("result")) if capture_content and meta.get("result") is not None else None
|
||||
),
|
||||
error=_parse_error(payload),
|
||||
response_cost=as_float(payload.get("response_cost")),
|
||||
# Same 0-vs-pricing-failure ambiguity as the LLM span above.
|
||||
response_cost=(
|
||||
as_float(payload.get("response_cost")) if not payload.get("response_cost_failure_debug_info") else None
|
||||
),
|
||||
identity=RequestContext.from_standard_logging_payload(payload).identity,
|
||||
)
|
||||
|
||||
|
|
|
|||
|
|
@ -70,9 +70,7 @@ def test_arize_set_attributes():
|
|||
# Simulated LLM response object
|
||||
response_obj = ModelResponse(
|
||||
usage={"total_tokens": 100, "completion_tokens": 60, "prompt_tokens": 40},
|
||||
choices=[
|
||||
Choices(message={"role": "assistant", "content": "Basic Response Content"})
|
||||
],
|
||||
choices=[Choices(message={"role": "assistant", "content": "Basic Response Content"})],
|
||||
model="gpt-4o",
|
||||
id="chatcmpl-ID",
|
||||
)
|
||||
|
|
@ -89,9 +87,7 @@ def test_arize_set_attributes():
|
|||
assert span.set_attribute.call_count == 26
|
||||
|
||||
# Metadata attached to the span
|
||||
span.set_attribute.assert_any_call(
|
||||
SpanAttributes.METADATA, json.dumps({"key_1": "value_1", "key_2": None})
|
||||
)
|
||||
span.set_attribute.assert_any_call(SpanAttributes.METADATA, json.dumps({"key_1": "value_1", "key_2": None}))
|
||||
|
||||
# Basic LLM information
|
||||
span.set_attribute.assert_any_call(SpanAttributes.LLM_MODEL_NAME, "gpt-4o")
|
||||
|
|
@ -114,16 +110,12 @@ def test_arize_set_attributes():
|
|||
span.set_attribute.assert_any_call(SpanAttributes.OPENINFERENCE_SPAN_KIND, "LLM")
|
||||
# And TOOL must never be written for an LLM chat completion call.
|
||||
span_kind_writes = [
|
||||
c.args[1]
|
||||
for c in span.set_attribute.call_args_list
|
||||
if c.args[0] == SpanAttributes.OPENINFERENCE_SPAN_KIND
|
||||
c.args[1] for c in span.set_attribute.call_args_list if c.args[0] == SpanAttributes.OPENINFERENCE_SPAN_KIND
|
||||
]
|
||||
assert "TOOL" not in span_kind_writes
|
||||
|
||||
# Request message content and metadata
|
||||
span.set_attribute.assert_any_call(
|
||||
SpanAttributes.INPUT_VALUE, "Basic Request Content"
|
||||
)
|
||||
span.set_attribute.assert_any_call(SpanAttributes.INPUT_VALUE, "Basic Request Content")
|
||||
span.set_attribute.assert_any_call(
|
||||
f"{SpanAttributes.LLM_INPUT_MESSAGES}.0.{MessageAttributes.MESSAGE_ROLE}",
|
||||
"user",
|
||||
|
|
@ -134,9 +126,7 @@ def test_arize_set_attributes():
|
|||
)
|
||||
|
||||
# Tool call definitions and function names
|
||||
span.set_attribute.assert_any_call(
|
||||
f"{SpanAttributes.LLM_TOOLS}.0.name", "get_weather"
|
||||
)
|
||||
span.set_attribute.assert_any_call(f"{SpanAttributes.LLM_TOOLS}.0.name", "get_weather")
|
||||
span.set_attribute.assert_any_call(
|
||||
f"{SpanAttributes.LLM_TOOLS}.0.description",
|
||||
"Fetches weather details.",
|
||||
|
|
@ -146,26 +136,20 @@ def test_arize_set_attributes():
|
|||
json.dumps(
|
||||
{
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"location": {"type": "string", "description": "City name"}
|
||||
},
|
||||
"properties": {"location": {"type": "string", "description": "City name"}},
|
||||
"required": ["location"],
|
||||
}
|
||||
),
|
||||
)
|
||||
|
||||
# Invocation parameters
|
||||
span.set_attribute.assert_any_call(
|
||||
SpanAttributes.LLM_INVOCATION_PARAMETERS, '{"user": "test_user"}'
|
||||
)
|
||||
span.set_attribute.assert_any_call(SpanAttributes.LLM_INVOCATION_PARAMETERS, '{"user": "test_user"}')
|
||||
|
||||
# User ID
|
||||
span.set_attribute.assert_any_call(SpanAttributes.USER_ID, "test_user")
|
||||
|
||||
# Output message content
|
||||
span.set_attribute.assert_any_call(
|
||||
SpanAttributes.OUTPUT_VALUE, "Basic Response Content"
|
||||
)
|
||||
span.set_attribute.assert_any_call(SpanAttributes.OUTPUT_VALUE, "Basic Response Content")
|
||||
span.set_attribute.assert_any_call(
|
||||
f"{SpanAttributes.LLM_OUTPUT_MESSAGES}.0.{MessageAttributes.MESSAGE_ROLE}",
|
||||
"assistant",
|
||||
|
|
@ -228,9 +212,7 @@ def test_arize_set_attributes_responses_api():
|
|||
ResponseReasoningItem(
|
||||
id="reasoning-001",
|
||||
type="reasoning",
|
||||
summary=[
|
||||
Summary(text="First, I need to analyze...", type="summary_text")
|
||||
],
|
||||
summary=[Summary(text="First, I need to analyze...", type="summary_text")],
|
||||
),
|
||||
ResponseOutputMessage(
|
||||
id="msg-001",
|
||||
|
|
@ -277,9 +259,7 @@ def test_arize_set_attributes_responses_api():
|
|||
span.set_attribute.assert_any_call(SpanAttributes.LLM_TOKEN_COUNT_TOTAL, 370)
|
||||
span.set_attribute.assert_any_call(SpanAttributes.LLM_TOKEN_COUNT_COMPLETION, 250)
|
||||
span.set_attribute.assert_any_call(SpanAttributes.LLM_TOKEN_COUNT_PROMPT, 120)
|
||||
span.set_attribute.assert_any_call(
|
||||
SpanAttributes.LLM_TOKEN_COUNT_COMPLETION_DETAILS_REASONING, 180
|
||||
)
|
||||
span.set_attribute.assert_any_call(SpanAttributes.LLM_TOKEN_COUNT_COMPLETION_DETAILS_REASONING, 180)
|
||||
|
||||
|
||||
def test_set_usage_outputs_pydantic_completion_usage():
|
||||
|
|
@ -327,9 +307,7 @@ def test_set_usage_outputs_pydantic_completion_usage():
|
|||
span.set_attribute.assert_any_call(SpanAttributes.LLM_TOKEN_COUNT_PROMPT, 40)
|
||||
span.set_attribute.assert_any_call(SpanAttributes.LLM_TOKEN_COUNT_COMPLETION, 60)
|
||||
# reasoning_tokens for chat completions live in completion_tokens_details
|
||||
span.set_attribute.assert_any_call(
|
||||
SpanAttributes.LLM_TOKEN_COUNT_COMPLETION_DETAILS_REASONING, 25
|
||||
)
|
||||
span.set_attribute.assert_any_call(SpanAttributes.LLM_TOKEN_COUNT_COMPLETION_DETAILS_REASONING, 25)
|
||||
|
||||
|
||||
def test_set_usage_outputs_pydantic_response_api_usage():
|
||||
|
|
@ -362,9 +340,7 @@ def test_set_usage_outputs_pydantic_response_api_usage():
|
|||
span.set_attribute.assert_any_call(SpanAttributes.LLM_TOKEN_COUNT_TOTAL, 370)
|
||||
span.set_attribute.assert_any_call(SpanAttributes.LLM_TOKEN_COUNT_PROMPT, 120)
|
||||
span.set_attribute.assert_any_call(SpanAttributes.LLM_TOKEN_COUNT_COMPLETION, 250)
|
||||
span.set_attribute.assert_any_call(
|
||||
SpanAttributes.LLM_TOKEN_COUNT_COMPLETION_DETAILS_REASONING, 180
|
||||
)
|
||||
span.set_attribute.assert_any_call(SpanAttributes.LLM_TOKEN_COUNT_COMPLETION_DETAILS_REASONING, 180)
|
||||
|
||||
|
||||
class TestArizeLogger(CustomLogger):
|
||||
|
|
@ -375,16 +351,12 @@ class TestArizeLogger(CustomLogger):
|
|||
|
||||
def __init__(self, *args, **kwargs):
|
||||
super().__init__(*args, **kwargs)
|
||||
self.standard_callback_dynamic_params: Optional[
|
||||
StandardCallbackDynamicParams
|
||||
] = None
|
||||
self.standard_callback_dynamic_params: Optional[StandardCallbackDynamicParams] = None
|
||||
|
||||
async def async_log_success_event(self, kwargs, response_obj, start_time, end_time):
|
||||
# Capture dynamic params and print them for verification
|
||||
print("logged kwargs", json.dumps(kwargs, indent=4, default=str))
|
||||
self.standard_callback_dynamic_params = kwargs.get(
|
||||
"standard_callback_dynamic_params"
|
||||
)
|
||||
self.standard_callback_dynamic_params = kwargs.get("standard_callback_dynamic_params")
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
|
|
@ -410,14 +382,8 @@ async def test_arize_dynamic_params():
|
|||
|
||||
# Assert dynamic parameters were received in the callback
|
||||
assert test_arize_logger.standard_callback_dynamic_params is not None
|
||||
assert (
|
||||
test_arize_logger.standard_callback_dynamic_params.get("arize_api_key")
|
||||
== "test_api_key_dynamic"
|
||||
)
|
||||
assert (
|
||||
test_arize_logger.standard_callback_dynamic_params.get("arize_space_key")
|
||||
== "test_space_key_dynamic"
|
||||
)
|
||||
assert test_arize_logger.standard_callback_dynamic_params.get("arize_api_key") == "test_api_key_dynamic"
|
||||
assert test_arize_logger.standard_callback_dynamic_params.get("arize_space_key") == "test_space_key_dynamic"
|
||||
|
||||
|
||||
def test_construct_dynamic_arize_headers():
|
||||
|
|
@ -428,9 +394,7 @@ def test_construct_dynamic_arize_headers():
|
|||
from litellm.types.utils import StandardCallbackDynamicParams
|
||||
|
||||
# Test with all parameters present
|
||||
dynamic_params_full = StandardCallbackDynamicParams(
|
||||
arize_api_key="test_api_key", arize_space_id="test_space_id"
|
||||
)
|
||||
dynamic_params_full = StandardCallbackDynamicParams(arize_api_key="test_api_key", arize_space_id="test_space_id")
|
||||
arize_logger = ArizeLogger()
|
||||
|
||||
headers = arize_logger.construct_dynamic_otel_headers(dynamic_params_full)
|
||||
|
|
@ -438,9 +402,7 @@ def test_construct_dynamic_arize_headers():
|
|||
assert headers == expected_headers
|
||||
|
||||
# Test with only space_id
|
||||
dynamic_params_space_id_only = StandardCallbackDynamicParams(
|
||||
arize_space_id="test_space_id"
|
||||
)
|
||||
dynamic_params_space_id_only = StandardCallbackDynamicParams(arize_space_id="test_space_id")
|
||||
|
||||
headers = arize_logger.construct_dynamic_otel_headers(dynamic_params_space_id_only)
|
||||
expected_headers = {"arize-space-id": "test_space_id"}
|
||||
|
|
@ -456,9 +418,7 @@ def test_construct_dynamic_arize_headers():
|
|||
dynamic_params_space_key_and_api_key = StandardCallbackDynamicParams(
|
||||
arize_space_key="test_space_key", arize_api_key="test_api_key"
|
||||
)
|
||||
headers = arize_logger.construct_dynamic_otel_headers(
|
||||
dynamic_params_space_key_and_api_key
|
||||
)
|
||||
headers = arize_logger.construct_dynamic_otel_headers(dynamic_params_space_key_and_api_key)
|
||||
expected_headers = {"arize-space-id": "test_space_key", "api_key": "test_api_key"}
|
||||
|
||||
|
||||
|
|
@ -528,9 +488,7 @@ def test_arize_emits_no_cache_tokens_when_absent():
|
|||
from litellm.integrations.arize._utils import _set_usage_outputs
|
||||
|
||||
span = MagicMock()
|
||||
response_obj = {
|
||||
"usage": {"total_tokens": 10, "completion_tokens": 4, "prompt_tokens": 6}
|
||||
}
|
||||
response_obj = {"usage": {"total_tokens": 10, "completion_tokens": 4, "prompt_tokens": 6}}
|
||||
_set_usage_outputs(span, response_obj, SpanAttributes)
|
||||
attrs = _collect_calls(span)
|
||||
assert SpanAttributes.LLM_TOKEN_COUNT_PROMPT_DETAILS_CACHE_READ not in attrs
|
||||
|
|
@ -542,14 +500,8 @@ def test_passthrough_call_type_resolves_to_llm_span_kind():
|
|||
from litellm.integrations._types.open_inference import OpenInferenceSpanKindValues
|
||||
from litellm.integrations.arize._utils import _infer_open_inference_span_kind
|
||||
|
||||
assert (
|
||||
_infer_open_inference_span_kind("allm_passthrough_route")
|
||||
== OpenInferenceSpanKindValues.LLM.value
|
||||
)
|
||||
assert (
|
||||
_infer_open_inference_span_kind("llm_passthrough_route")
|
||||
== OpenInferenceSpanKindValues.LLM.value
|
||||
)
|
||||
assert _infer_open_inference_span_kind("allm_passthrough_route") == OpenInferenceSpanKindValues.LLM.value
|
||||
assert _infer_open_inference_span_kind("llm_passthrough_route") == OpenInferenceSpanKindValues.LLM.value
|
||||
|
||||
|
||||
def test_arize_chat_completion_with_tools_stays_llm_span_kind():
|
||||
|
|
@ -605,9 +557,7 @@ def test_arize_chat_completion_with_tools_stays_llm_span_kind():
|
|||
|
||||
ArizeLogger.set_arize_attributes(span, kwargs, response_obj)
|
||||
span_kind_writes = [
|
||||
c.args[1]
|
||||
for c in span.set_attribute.call_args_list
|
||||
if c.args[0] == SpanAttributes.OPENINFERENCE_SPAN_KIND
|
||||
c.args[1] for c in span.set_attribute.call_args_list if c.args[0] == SpanAttributes.OPENINFERENCE_SPAN_KIND
|
||||
]
|
||||
assert span_kind_writes, "span.kind must be written"
|
||||
assert all(v == "LLM" for v in span_kind_writes)
|
||||
|
|
@ -659,13 +609,8 @@ def test_arize_emits_assistant_tool_calls_on_output_message():
|
|||
attrs = _collect_calls(span)
|
||||
base = f"{SpanAttributes.LLM_OUTPUT_MESSAGES}.0.{MessageAttributes.MESSAGE_TOOL_CALLS}.0"
|
||||
assert attrs[f"{base}.{ToolCallAttributes.TOOL_CALL_ID}"] == "call_abc"
|
||||
assert (
|
||||
attrs[f"{base}.{ToolCallAttributes.TOOL_CALL_FUNCTION_NAME}"] == "get_weather"
|
||||
)
|
||||
assert (
|
||||
attrs[f"{base}.{ToolCallAttributes.TOOL_CALL_FUNCTION_ARGUMENTS_JSON}"]
|
||||
== '{"location": "SF"}'
|
||||
)
|
||||
assert attrs[f"{base}.{ToolCallAttributes.TOOL_CALL_FUNCTION_NAME}"] == "get_weather"
|
||||
assert attrs[f"{base}.{ToolCallAttributes.TOOL_CALL_FUNCTION_ARGUMENTS_JSON}"] == '{"location": "SF"}'
|
||||
|
||||
|
||||
def test_arize_output_value_falls_back_to_tool_calls_summary():
|
||||
|
|
@ -818,9 +763,7 @@ def test_arize_emits_tool_call_id_and_name_on_input_tool_message():
|
|||
assert attrs[f"{assistant_base}.{ToolCallAttributes.TOOL_CALL_ID}"] == "call_abc"
|
||||
# Tool message at index 2
|
||||
tool_prefix = f"{SpanAttributes.LLM_INPUT_MESSAGES}.2"
|
||||
assert (
|
||||
attrs[f"{tool_prefix}.{MessageAttributes.MESSAGE_TOOL_CALL_ID}"] == "call_abc"
|
||||
)
|
||||
assert attrs[f"{tool_prefix}.{MessageAttributes.MESSAGE_TOOL_CALL_ID}"] == "call_abc"
|
||||
assert attrs[f"{tool_prefix}.{MessageAttributes.MESSAGE_NAME}"] == "get_weather"
|
||||
|
||||
|
||||
|
|
@ -866,10 +809,7 @@ def test_arize_emits_multimodal_input_contents():
|
|||
assert attrs[f"{base}.0.message_content.type"] == "text"
|
||||
assert attrs[f"{base}.0.message_content.text"] == "What is in this image?"
|
||||
assert attrs[f"{base}.1.message_content.type"] == "image"
|
||||
assert (
|
||||
attrs[f"{base}.1.message_content.image.image.url"]
|
||||
== "https://example.com/cat.png"
|
||||
)
|
||||
assert attrs[f"{base}.1.message_content.image.image.url"] == "https://example.com/cat.png"
|
||||
|
||||
|
||||
def test_arize_emits_session_and_user_attrs_from_metadata():
|
||||
|
|
@ -974,11 +914,7 @@ def test_arize_does_not_overwrite_user_id_from_optional_params():
|
|||
id="r2",
|
||||
)
|
||||
ArizeLogger.set_arize_attributes(span, kwargs, response_obj)
|
||||
user_id_writes = [
|
||||
c.args[1]
|
||||
for c in span.set_attribute.call_args_list
|
||||
if c.args[0] == SpanAttributes.USER_ID
|
||||
]
|
||||
user_id_writes = [c.args[1] for c in span.set_attribute.call_args_list if c.args[0] == SpanAttributes.USER_ID]
|
||||
assert "from_metadata" not in user_id_writes
|
||||
|
||||
|
||||
|
|
@ -1013,6 +949,72 @@ def test_arize_emits_response_cost():
|
|||
assert attrs["llm.response.cost"] == 0.0012345 # legacy key still emitted
|
||||
|
||||
|
||||
def test_arize_omits_cost_when_pricing_failed():
|
||||
"""response_cost = 0 plus pricing-failure debug info means LiteLLM could
|
||||
not price the call — omit the cost attributes entirely: backends cannot
|
||||
tell a fake 0 from a free model (issue #44186)."""
|
||||
from unittest.mock import MagicMock
|
||||
|
||||
from litellm.types.utils import Choices, ModelResponse
|
||||
|
||||
span = MagicMock()
|
||||
kwargs = {
|
||||
"model": "gpt-4o",
|
||||
"messages": [{"role": "user", "content": "hi"}],
|
||||
"standard_logging_object": {
|
||||
"model_parameters": {},
|
||||
"metadata": {},
|
||||
"call_type": "completion",
|
||||
"response_cost": 0.0,
|
||||
"response_cost_failure_debug_info": {"error_str": "model not in cost map"},
|
||||
},
|
||||
"optional_params": {},
|
||||
"litellm_params": {"custom_llm_provider": "openai"},
|
||||
}
|
||||
response_obj = ModelResponse(
|
||||
usage={"total_tokens": 4, "completion_tokens": 2, "prompt_tokens": 2},
|
||||
choices=[Choices(message={"role": "assistant", "content": "hello"})],
|
||||
model="gpt-4o",
|
||||
id="r4",
|
||||
)
|
||||
ArizeLogger.set_arize_attributes(span, kwargs, response_obj)
|
||||
attrs = _collect_calls(span)
|
||||
assert "llm.cost.total" not in attrs
|
||||
assert "llm.response.cost" not in attrs
|
||||
|
||||
|
||||
def test_arize_emits_zero_cost_for_free_model():
|
||||
"""A genuine 0 (free model, no pricing-failure info) must still be
|
||||
emitted — the guard only fires on the pricing-failure marker."""
|
||||
from unittest.mock import MagicMock
|
||||
|
||||
from litellm.types.utils import Choices, ModelResponse
|
||||
|
||||
span = MagicMock()
|
||||
kwargs = {
|
||||
"model": "gpt-4o",
|
||||
"messages": [{"role": "user", "content": "hi"}],
|
||||
"standard_logging_object": {
|
||||
"model_parameters": {},
|
||||
"metadata": {},
|
||||
"call_type": "completion",
|
||||
"response_cost": 0.0,
|
||||
},
|
||||
"optional_params": {},
|
||||
"litellm_params": {"custom_llm_provider": "openai"},
|
||||
}
|
||||
response_obj = ModelResponse(
|
||||
usage={"total_tokens": 4, "completion_tokens": 2, "prompt_tokens": 2},
|
||||
choices=[Choices(message={"role": "assistant", "content": "hello"})],
|
||||
model="gpt-4o",
|
||||
id="r5",
|
||||
)
|
||||
ArizeLogger.set_arize_attributes(span, kwargs, response_obj)
|
||||
attrs = _collect_calls(span)
|
||||
assert attrs["llm.cost.total"] == 0.0
|
||||
assert attrs["llm.response.cost"] == 0.0
|
||||
|
||||
|
||||
def test_arize_passthrough_bedrock_anthropic_normalization():
|
||||
"""Bedrock-Anthropic passthrough: input/output text must be set so the
|
||||
span renders something other than raw provider attrs."""
|
||||
|
|
@ -1048,9 +1050,7 @@ def test_arize_passthrough_bedrock_anthropic_normalization():
|
|||
"complete_input_dict": {
|
||||
"anthropic_version": "bedrock-2023-05-31",
|
||||
"max_tokens": 64,
|
||||
"messages": [
|
||||
{"role": "user", "content": "What is the capital of France?"}
|
||||
],
|
||||
"messages": [{"role": "user", "content": "What is the capital of France?"}],
|
||||
}
|
||||
},
|
||||
"standard_logging_object": {
|
||||
|
|
@ -1068,19 +1068,13 @@ def test_arize_passthrough_bedrock_anthropic_normalization():
|
|||
assert attrs[SpanAttributes.INPUT_VALUE] == "What is the capital of France?"
|
||||
msg0 = f"{SpanAttributes.LLM_INPUT_MESSAGES}.0"
|
||||
assert attrs[f"{msg0}.{MessageAttributes.MESSAGE_ROLE}"] == "user"
|
||||
assert (
|
||||
attrs[f"{msg0}.{MessageAttributes.MESSAGE_CONTENT}"]
|
||||
== "What is the capital of France?"
|
||||
)
|
||||
assert attrs[f"{msg0}.{MessageAttributes.MESSAGE_CONTENT}"] == "What is the capital of France?"
|
||||
|
||||
# Output rendering (Anthropic content[].text)
|
||||
assert attrs[SpanAttributes.OUTPUT_VALUE] == "The capital of France is Paris."
|
||||
out0 = f"{SpanAttributes.LLM_OUTPUT_MESSAGES}.0"
|
||||
assert attrs[f"{out0}.{MessageAttributes.MESSAGE_ROLE}"] == "assistant"
|
||||
assert (
|
||||
attrs[f"{out0}.{MessageAttributes.MESSAGE_CONTENT}"]
|
||||
== "The capital of France is Paris."
|
||||
)
|
||||
assert attrs[f"{out0}.{MessageAttributes.MESSAGE_CONTENT}"] == "The capital of France is Paris."
|
||||
|
||||
# Token counts (Bedrock input_tokens/output_tokens) — extracted via
|
||||
# coercion of the non-dict response.
|
||||
|
|
@ -1089,9 +1083,7 @@ def test_arize_passthrough_bedrock_anthropic_normalization():
|
|||
|
||||
# Span kind defended even though the call_type is a passthrough variant.
|
||||
span_kind_writes = [
|
||||
c.args[1]
|
||||
for c in span.set_attribute.call_args_list
|
||||
if c.args[0] == SpanAttributes.OPENINFERENCE_SPAN_KIND
|
||||
c.args[1] for c in span.set_attribute.call_args_list if c.args[0] == SpanAttributes.OPENINFERENCE_SPAN_KIND
|
||||
]
|
||||
assert span_kind_writes # at least one
|
||||
assert all(v == "LLM" for v in span_kind_writes)
|
||||
|
|
@ -1109,11 +1101,7 @@ def test_arize_passthrough_call_type_does_not_run_on_chat_completion():
|
|||
span = MagicMock()
|
||||
_maybe_normalize_passthrough(
|
||||
span,
|
||||
{
|
||||
"additional_args": {
|
||||
"complete_input_dict": {"messages": [{"role": "user", "content": "x"}]}
|
||||
}
|
||||
},
|
||||
{"additional_args": {"complete_input_dict": {"messages": [{"role": "user", "content": "x"}]}}},
|
||||
{"choices": [{"message": {"role": "assistant", "content": "y"}}]},
|
||||
{"choices": [{"message": {"role": "assistant", "content": "y"}}]},
|
||||
{"call_type": "completion"},
|
||||
|
|
@ -1133,11 +1121,7 @@ def test_arize_passthrough_skipped_when_message_redaction_enabled():
|
|||
span = MagicMock()
|
||||
kwargs = {
|
||||
"additional_args": {
|
||||
"complete_input_dict": {
|
||||
"messages": [
|
||||
{"role": "user", "content": "Patient John Doe, SSN 123-45-6789"}
|
||||
]
|
||||
}
|
||||
"complete_input_dict": {"messages": [{"role": "user", "content": "Patient John Doe, SSN 123-45-6789"}]}
|
||||
},
|
||||
# Enables redaction via the dynamic-param path inside
|
||||
# should_redact_message_logging(), without touching globals.
|
||||
|
|
@ -1211,9 +1195,7 @@ def test_arize_mcp_call_tool_result_does_not_break_attribute_setting():
|
|||
"optional_params": {},
|
||||
"litellm_params": {"custom_llm_provider": "mcp"},
|
||||
}
|
||||
response_obj = CallToolResult(
|
||||
content=[TextContent(type="text", text="sunny, 21C")], isError=False
|
||||
)
|
||||
response_obj = CallToolResult(content=[TextContent(type="text", text="sunny, 21C")], isError=False)
|
||||
|
||||
ArizeLogger.set_arize_attributes(span, kwargs, response_obj)
|
||||
|
||||
|
|
@ -1295,9 +1277,7 @@ def test_arize_mcp_tool_span_renders_name_input_and_output():
|
|||
from mcp.types import CallToolResult, TextContent
|
||||
|
||||
span = MagicMock()
|
||||
response_obj = CallToolResult(
|
||||
content=[TextContent(type="text", text="sunny, 21C")], isError=False
|
||||
)
|
||||
response_obj = CallToolResult(content=[TextContent(type="text", text="sunny, 21C")], isError=False)
|
||||
|
||||
ArizeLogger.set_arize_attributes(span, _mcp_kwargs(), response_obj)
|
||||
|
||||
|
|
@ -1336,9 +1316,7 @@ def test_arize_mcp_tool_span_respects_message_redaction():
|
|||
from mcp.types import CallToolResult, TextContent
|
||||
|
||||
span = MagicMock()
|
||||
response_obj = CallToolResult(
|
||||
content=[TextContent(type="text", text="SSN 123-45-6789")], isError=False
|
||||
)
|
||||
response_obj = CallToolResult(content=[TextContent(type="text", text="SSN 123-45-6789")], isError=False)
|
||||
|
||||
ArizeLogger.set_arize_attributes(
|
||||
span,
|
||||
|
|
|
|||
|
|
@ -90,6 +90,22 @@ def test_llm_call_span_cost_breakdown():
|
|||
assert f"{LiteLLM.COST_PREFIX}margin_total_amount" not in a
|
||||
|
||||
|
||||
def test_llm_call_span_pricing_failure_reports_unknown_cost():
|
||||
"""A 0 response_cost alongside pricing-failure debug info means LiteLLM
|
||||
could not price the call — report unknown (None), not a fake 0 that cost
|
||||
backends cannot distinguish from a free model (issue #44186)."""
|
||||
data = LLMCallSpanData.from_standard_logging_payload(
|
||||
_payload(response_cost=0.0, response_cost_failure_debug_info={"error_str": "model not in cost map"})
|
||||
)
|
||||
assert data.response_cost is None
|
||||
|
||||
|
||||
def test_llm_call_span_zero_cost_without_failure_info_is_kept():
|
||||
"""A genuine 0 (free model) without pricing-failure info stays 0."""
|
||||
data = LLMCallSpanData.from_standard_logging_payload(_payload(response_cost=0.0))
|
||||
assert data.response_cost == 0.0
|
||||
|
||||
|
||||
def test_tracer_scope_carries_litellm_version():
|
||||
from litellm._version import version as litellm_version
|
||||
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue