mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-10 03:28:53 +00:00
Add MCP semantic conventions to otelv2 (#29468)
* Add MCP semantic conventions to otelv2
Emit OpenTelemetry GenAI MCP tool-call spans from the v2 logger. A closed
call_mcp_tool request now produces a CLIENT span named "tools/call {tool}"
carrying mcp.method.name, gen_ai.operation.name=execute_tool, gen_ai.tool.name,
the upstream server name, and (opt-in, content-gated) tool arguments/result.
Adds the MCP and JSON-RPC attribute vocabulary to the semconv module, an
MCPToolCallSpanData payload built from StandardLoggingMCPToolCall, an
MCP_TOOL_CALL span role, and mapper support.
* Complete the MCP span-attribute vocabulary in otelv2 semconv
Add the remaining OTel GenAI MCP semconv attribute keys: gen_ai.prompt.name,
the network.* transport keys with their well-known NetworkTransport values, and
the client.* peer keys for MCP server spans. A test pins the full vocabulary so
a dropped or renamed key fails loudly.
* Populate mcp.session.id on MCP tool-call spans
Capture the mcp-session-id header (case-insensitively) at the tool-call entry
point and thread it through StandardLoggingMCPToolCall into the span, so spans
for stateful MCP sessions carry mcp.session.id. Stateless calls have no such
header and the attribute is simply absent.
* Test that stateless MCP calls omit mcp.session.id
---------
Co-authored-by: Claude <noreply@anthropic.com>
This commit is contained in:
parent
84c4c12f90
commit
b98a656254
13 changed files with 563 additions and 16 deletions
|
|
@ -32,20 +32,28 @@ from litellm.integrations.otel.model.payloads import (
|
|||
LLMCallSpanData,
|
||||
LLMRequestParams,
|
||||
LLMUsage,
|
||||
MCPToolCallSpanData,
|
||||
ProxyRequestSpanData,
|
||||
ServerInfo,
|
||||
ServiceSpanData,
|
||||
SpanError,
|
||||
is_mcp_tool_call,
|
||||
)
|
||||
from litellm.integrations.otel.model.semconv import (
|
||||
DB,
|
||||
HTTP,
|
||||
MCP,
|
||||
Client,
|
||||
Error,
|
||||
GenAI,
|
||||
GenAIOperation,
|
||||
GenAIProvider,
|
||||
HTTP,
|
||||
JsonRpc,
|
||||
LiteLLM,
|
||||
MCPMethod,
|
||||
Metric,
|
||||
Network,
|
||||
NetworkTransport,
|
||||
Server,
|
||||
resolve_operation,
|
||||
resolve_provider,
|
||||
|
|
@ -69,13 +77,19 @@ __all__ = [
|
|||
"BAGGAGE_PROMOTED_KEYS",
|
||||
"DB",
|
||||
"DEFAULT_BAGGAGE_METADATA_KEYS",
|
||||
"Client",
|
||||
"Error",
|
||||
"GenAI",
|
||||
"GenAIOperation",
|
||||
"GenAIProvider",
|
||||
"HTTP",
|
||||
"JsonRpc",
|
||||
"LiteLLM",
|
||||
"MCP",
|
||||
"MCPMethod",
|
||||
"Metric",
|
||||
"Network",
|
||||
"NetworkTransport",
|
||||
"Server",
|
||||
"resolve_operation",
|
||||
"resolve_provider",
|
||||
|
|
@ -92,11 +106,13 @@ __all__ = [
|
|||
"LLMCallSpanData",
|
||||
"LLMRequestParams",
|
||||
"LLMUsage",
|
||||
"MCPToolCallSpanData",
|
||||
"ProxyRequestSpanData",
|
||||
"RequestContext",
|
||||
"RequestIdentity",
|
||||
"ServerInfo",
|
||||
"ServiceSpanData",
|
||||
"SpanError",
|
||||
"is_mcp_tool_call",
|
||||
"promoted_baggage",
|
||||
]
|
||||
|
|
|
|||
|
|
@ -13,6 +13,7 @@ from litellm.integrations.otel.mappers.base import AttributeMapper, SpanData
|
|||
from litellm.integrations.otel.model.payloads import (
|
||||
GuardrailSpanData,
|
||||
LLMCallSpanData,
|
||||
MCPToolCallSpanData,
|
||||
ServiceSpanData,
|
||||
)
|
||||
from litellm.integrations.otel.plumbing.providers import to_otel_span_kind
|
||||
|
|
@ -22,6 +23,7 @@ from litellm.integrations.otel.model.spans import (
|
|||
SpanRole,
|
||||
guardrail_span_name,
|
||||
llm_call_span_name,
|
||||
mcp_tool_call_span_name,
|
||||
service_span_name,
|
||||
)
|
||||
|
||||
|
|
@ -30,6 +32,7 @@ from litellm.integrations.otel.model.spans import (
|
|||
# have no builder here.
|
||||
_NAME_BUILDERS: dict[SpanRole, Callable[..., str]] = {
|
||||
SpanRole.LLM_CALL: llm_call_span_name,
|
||||
SpanRole.MCP_TOOL_CALL: mcp_tool_call_span_name,
|
||||
SpanRole.GUARDRAIL: guardrail_span_name,
|
||||
# DB_CALL and SERVICE are both built from ServiceSpanData; they differ only in
|
||||
# span kind (CLIENT vs INTERNAL) and attribute vocabulary, not in naming.
|
||||
|
|
@ -121,10 +124,14 @@ class SpanEmitter:
|
|||
Return the span, or ``None`` if it was deduplicated away. ``tracer``
|
||||
overrides the bound tracer for this span, used for per-request routing.
|
||||
"""
|
||||
# Only LLM-call spans carry a dedup key; LLM-call and service spans
|
||||
# carry an ``error`` field. ``isinstance`` narrows the type for mypy and
|
||||
# keeps the engine free of duck-typed attribute reads.
|
||||
dedup_key = data.identity.call_id if isinstance(data, LLMCallSpanData) else None
|
||||
# LLM-call and MCP tool-call spans carry a dedup key (their request's
|
||||
# call id), so a sync+async double-firing coalesces. ``isinstance`` narrows
|
||||
# the type for mypy and keeps the engine free of duck-typed attribute reads.
|
||||
dedup_key = (
|
||||
data.identity.call_id
|
||||
if isinstance(data, (LLMCallSpanData, MCPToolCallSpanData))
|
||||
else None
|
||||
)
|
||||
if self._seen(dedup_key, role):
|
||||
return None
|
||||
span = self.start_span(
|
||||
|
|
@ -160,7 +167,15 @@ class SpanEmitter:
|
|||
span.set_attribute(key, value)
|
||||
error = (
|
||||
data.error
|
||||
if isinstance(data, (LLMCallSpanData, ServiceSpanData, GuardrailSpanData))
|
||||
if isinstance(
|
||||
data,
|
||||
(
|
||||
LLMCallSpanData,
|
||||
MCPToolCallSpanData,
|
||||
ServiceSpanData,
|
||||
GuardrailSpanData,
|
||||
),
|
||||
)
|
||||
else None
|
||||
)
|
||||
if error and (error.error_type or error.message):
|
||||
|
|
|
|||
|
|
@ -31,8 +31,10 @@ from litellm.integrations.otel.model.metadata import (
|
|||
from litellm.integrations.otel.model.payloads import (
|
||||
GuardrailSpanData,
|
||||
LLMCallSpanData,
|
||||
MCPToolCallSpanData,
|
||||
ServiceSpanData,
|
||||
SpanError,
|
||||
is_mcp_tool_call,
|
||||
)
|
||||
from litellm.integrations.otel.plumbing.providers import (
|
||||
build_tracer_provider,
|
||||
|
|
@ -43,7 +45,10 @@ from litellm.integrations.otel.model.spans import SpanRole, span_role_for_servic
|
|||
from litellm.integrations.otel.model.utils import to_ns
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from litellm.types.utils import StandardLoggingGuardrailInformation
|
||||
from litellm.types.utils import (
|
||||
StandardLoggingGuardrailInformation,
|
||||
StandardLoggingPayload,
|
||||
)
|
||||
|
||||
LITELLM_TRACER_NAME = "litellm"
|
||||
|
||||
|
|
@ -200,11 +205,53 @@ class OpenTelemetryV2(CustomLogger):
|
|||
self._open_llm_calls.popitem(last=False)
|
||||
|
||||
async def async_log_success_event(self, kwargs, response_obj, start_time, end_time):
|
||||
if self._emit_mcp_tool_call(kwargs, start_time, end_time):
|
||||
return
|
||||
self._close_llm_call(kwargs, start_time, end_time)
|
||||
|
||||
async def async_log_failure_event(self, kwargs, response_obj, start_time, end_time):
|
||||
if self._emit_mcp_tool_call(kwargs, start_time, end_time):
|
||||
return
|
||||
self._close_llm_call(kwargs, start_time, end_time)
|
||||
|
||||
def _emit_mcp_tool_call(
|
||||
self,
|
||||
kwargs: Mapping[str, Any],
|
||||
start_time: datetime | float | None,
|
||||
end_time: datetime | float | None,
|
||||
) -> bool:
|
||||
"""Emit an MCP tool-call span when the closed request was a tool call.
|
||||
|
||||
MCP tool calls reach the success/failure callbacks like any other request
|
||||
(with ``call_type`` ``call_mcp_tool``), but they are not LLM calls and have
|
||||
no ``pre_call`` carrier — so they get their own CLIENT span here, parented
|
||||
to the request's server span. Returns whether it handled the event, so the
|
||||
caller skips the LLM-call path. The whole span is emitted at once (there is
|
||||
no boundary to open it at), deduped on the call id by the emitter.
|
||||
"""
|
||||
raw_payload = kwargs.get("standard_logging_object")
|
||||
if not raw_payload or not is_mcp_tool_call(
|
||||
cast(Mapping[str, object], raw_payload)
|
||||
):
|
||||
return False
|
||||
payload = cast("StandardLoggingPayload", raw_payload)
|
||||
data = MCPToolCallSpanData.from_standard_logging_payload(
|
||||
payload, capture_content=self.config.capture_span_content
|
||||
)
|
||||
# A stray LLM carrier from a ``pre_call`` that mis-fired for this id would
|
||||
# otherwise linger until evicted; drop it so it's neither leaked nor closed
|
||||
# as a phantom LLM span.
|
||||
if data.identity.call_id:
|
||||
self._open_llm_calls.pop(data.identity.call_id, None)
|
||||
self._emitter.emit(
|
||||
SpanRole.MCP_TOOL_CALL,
|
||||
data,
|
||||
parent_context=resolve_request_span_context(),
|
||||
start_time_ns=to_ns(start_time),
|
||||
end_time_ns=to_ns(end_time),
|
||||
)
|
||||
return True
|
||||
|
||||
def _close_llm_call(
|
||||
self,
|
||||
kwargs: Mapping[str, Any],
|
||||
|
|
|
|||
|
|
@ -7,6 +7,7 @@ from typing_extensions import Protocol, runtime_checkable
|
|||
from litellm.integrations.otel.model.payloads import (
|
||||
GuardrailSpanData,
|
||||
LLMCallSpanData,
|
||||
MCPToolCallSpanData,
|
||||
ServiceSpanData,
|
||||
)
|
||||
|
||||
|
|
@ -21,7 +22,7 @@ AttributeMap = dict[str, AttrValue]
|
|||
# The closed set of span-data types the engine routes through the mapper chain.
|
||||
# Server spans (PROXY_REQUEST + management routes) belong to the mounted FastAPI
|
||||
# instrumentor, not the mapper chain.
|
||||
SpanData = LLMCallSpanData | GuardrailSpanData | ServiceSpanData
|
||||
SpanData = LLMCallSpanData | MCPToolCallSpanData | GuardrailSpanData | ServiceSpanData
|
||||
|
||||
|
||||
@runtime_checkable
|
||||
|
|
|
|||
|
|
@ -14,10 +14,18 @@ from litellm.integrations.otel.mappers.utils import collect, drop_none
|
|||
from litellm.integrations.otel.model.payloads import (
|
||||
GuardrailSpanData,
|
||||
LLMCallSpanData,
|
||||
MCPToolCallSpanData,
|
||||
ServiceSpanData,
|
||||
ToolDefinition,
|
||||
)
|
||||
from litellm.integrations.otel.model.semconv import DB, Error, GenAI, LiteLLM, Server
|
||||
from litellm.integrations.otel.model.semconv import (
|
||||
DB,
|
||||
MCP,
|
||||
Error,
|
||||
GenAI,
|
||||
LiteLLM,
|
||||
Server,
|
||||
)
|
||||
from litellm.integrations.otel.model.spans import db_system
|
||||
|
||||
|
||||
|
|
@ -64,6 +72,18 @@ class GenAIMapper:
|
|||
"parameters": lambda t: t.parameters_json or None,
|
||||
}
|
||||
|
||||
_MCP_ATTRS: dict[str, Callable[[MCPToolCallSpanData], AttrValue | None]] = {
|
||||
GenAI.OPERATION_NAME: lambda d: d.operation.value,
|
||||
MCP.METHOD_NAME: lambda d: d.method,
|
||||
MCP.SESSION_ID: lambda d: d.session_id,
|
||||
GenAI.TOOL_NAME: lambda d: d.tool_name or None,
|
||||
GenAI.TOOL_CALL_ARGUMENTS: lambda d: d.arguments_json,
|
||||
GenAI.TOOL_CALL_RESULT: lambda d: d.result_json,
|
||||
LiteLLM.MCP_SERVER_NAME: lambda d: d.server_name,
|
||||
LiteLLM.CALL_ID: lambda d: d.identity.call_id or None,
|
||||
f"{LiteLLM.COST_PREFIX}total": lambda d: d.response_cost,
|
||||
}
|
||||
|
||||
_GUARDRAIL_ATTRS: dict[str, Callable[[GuardrailSpanData], AttrValue | None]] = {
|
||||
LiteLLM.GUARDRAIL_NAME: lambda d: d.guardrail_name,
|
||||
LiteLLM.GUARDRAIL_MODE: lambda d: d.mode,
|
||||
|
|
@ -92,6 +112,8 @@ class GenAIMapper:
|
|||
match data:
|
||||
case LLMCallSpanData():
|
||||
return self._llm_call(data)
|
||||
case MCPToolCallSpanData():
|
||||
return collect(self._MCP_ATTRS, data)
|
||||
case GuardrailSpanData():
|
||||
return self._guardrail(data)
|
||||
case ServiceSpanData():
|
||||
|
|
|
|||
|
|
@ -14,6 +14,7 @@ from litellm.integrations.otel.model.metadata import (
|
|||
)
|
||||
from litellm.integrations.otel.model.semconv import (
|
||||
GenAIOperation,
|
||||
MCPMethod,
|
||||
resolve_operation,
|
||||
resolve_provider,
|
||||
)
|
||||
|
|
@ -35,11 +36,13 @@ __all__ = [
|
|||
"LLMCallSpanData",
|
||||
"LLMRequestParams",
|
||||
"LLMUsage",
|
||||
"MCPToolCallSpanData",
|
||||
"ProxyRequestSpanData",
|
||||
"ServerInfo",
|
||||
"ServiceSpanData",
|
||||
"SpanError",
|
||||
"ToolDefinition",
|
||||
"is_mcp_tool_call",
|
||||
]
|
||||
|
||||
if TYPE_CHECKING:
|
||||
|
|
@ -309,6 +312,66 @@ class LLMCallSpanData:
|
|||
)
|
||||
|
||||
|
||||
# --- the MCP tool-call model ------------------------------------------------- #
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class MCPToolCallSpanData:
|
||||
"""One MCP ``tools/call`` execution, parsed from a closed request's payload.
|
||||
|
||||
The proxy is an MCP *client* to the upstream server it forwards the call to,
|
||||
so this is a CLIENT span. ``arguments_json``/``result_json`` are the tool's
|
||||
input/output — sensitive content, so they're only retained when content
|
||||
capture is enabled, mirroring ``LLMCallSpanData``'s message bodies.
|
||||
"""
|
||||
|
||||
operation: GenAIOperation
|
||||
method: str
|
||||
tool_name: str
|
||||
server_name: str | None
|
||||
session_id: str | None
|
||||
arguments_json: str | None
|
||||
result_json: str | None
|
||||
error: SpanError | None
|
||||
response_cost: float | None
|
||||
identity: RequestIdentity
|
||||
|
||||
@classmethod
|
||||
def from_standard_logging_payload(
|
||||
cls, payload: "StandardLoggingPayload", capture_content: bool = False
|
||||
) -> "MCPToolCallSpanData":
|
||||
meta = cast(Mapping[str, object], payload.get("mcp_tool_call_metadata") or {})
|
||||
return cls(
|
||||
operation=resolve_operation(as_str(payload.get("call_type"))),
|
||||
method=MCPMethod.TOOLS_CALL.value,
|
||||
tool_name=as_str(meta.get("name")) or "",
|
||||
server_name=as_str(meta.get("mcp_server_name")),
|
||||
session_id=as_str(meta.get("mcp_session_id")),
|
||||
arguments_json=(
|
||||
_json_or_none(meta.get("arguments"))
|
||||
if capture_content and meta.get("arguments") is not None
|
||||
else None
|
||||
),
|
||||
result_json=(
|
||||
_json_or_none(meta.get("result"))
|
||||
if capture_content and meta.get("result") is not None
|
||||
else None
|
||||
),
|
||||
error=_parse_error(payload),
|
||||
response_cost=as_float(payload.get("response_cost")),
|
||||
identity=RequestContext.from_standard_logging_payload(payload).identity,
|
||||
)
|
||||
|
||||
|
||||
def is_mcp_tool_call(payload: Mapping[str, object]) -> bool:
|
||||
"""Whether a closed request's payload is an MCP tool call rather than an LLM
|
||||
call — true when the MCP gateway stamped its tool-call metadata, or the call
|
||||
type says so on a path that hasn't populated the metadata yet."""
|
||||
return bool(payload.get("mcp_tool_call_metadata")) or (
|
||||
payload.get("call_type") == "call_mcp_tool"
|
||||
)
|
||||
|
||||
|
||||
# --- service event_metadata sanitization ------------------------------------ #
|
||||
|
||||
# Substrings (case-insensitive) of keys that must never reach a span: secrets,
|
||||
|
|
|
|||
|
|
@ -16,7 +16,7 @@ class GenAIOperation(str, Enum):
|
|||
GENERATE_CONTENT = "generate_content"
|
||||
CREATE_AGENT = "create_agent" # reserved for future agent spans
|
||||
INVOKE_AGENT = "invoke_agent" # reserved for future agent spans
|
||||
EXECUTE_TOOL = "execute_tool" # reserved for future tool spans
|
||||
EXECUTE_TOOL = "execute_tool" # MCP tool-call spans
|
||||
|
||||
|
||||
class GenAIProvider(str, Enum):
|
||||
|
|
@ -38,6 +38,16 @@ class GenAIProvider(str, Enum):
|
|||
IBM_WATSONX_AI = "ibm.watsonx.ai"
|
||||
|
||||
|
||||
class MCPMethod(str, Enum):
|
||||
"""Well-known values for ``mcp.method.name`` that litellm's MCP gateway
|
||||
serves. The value is the JSON-RPC method exactly as it travels on the wire."""
|
||||
|
||||
TOOLS_CALL = "tools/call"
|
||||
TOOLS_LIST = "tools/list"
|
||||
PROMPTS_GET = "prompts/get"
|
||||
PROMPTS_LIST = "prompts/list"
|
||||
|
||||
|
||||
class GenAI:
|
||||
"""Canonical OTel GenAI span-attribute keys."""
|
||||
|
||||
|
|
@ -68,11 +78,68 @@ class GenAI:
|
|||
SYSTEM_INSTRUCTIONS: Final = "gen_ai.system_instructions"
|
||||
OUTPUT_TYPE: Final = "gen_ai.output.type"
|
||||
CONVERSATION_ID: Final = "gen_ai.conversation.id"
|
||||
# agent / tool (reserved)
|
||||
# agent (reserved)
|
||||
AGENT_ID: Final = "gen_ai.agent.id"
|
||||
AGENT_NAME: Final = "gen_ai.agent.name"
|
||||
# tool / tool-call (stamped on MCP tool-call spans). Arguments and result are
|
||||
# the tool's input/output payloads — sensitive, so they're opt-in and gated by
|
||||
# the same content-capture mode as prompt/response content.
|
||||
TOOL_NAME: Final = "gen_ai.tool.name"
|
||||
TOOL_CALL_ID: Final = "gen_ai.tool.call.id"
|
||||
TOOL_CALL_ARGUMENTS: Final = "gen_ai.tool.call.arguments"
|
||||
TOOL_CALL_RESULT: Final = "gen_ai.tool.call.result"
|
||||
# prompt (MCP ``prompts/get`` etc.)
|
||||
PROMPT_NAME: Final = "gen_ai.prompt.name"
|
||||
|
||||
|
||||
class MCP:
|
||||
"""OTel GenAI MCP (Model Context Protocol) span-attribute keys.
|
||||
|
||||
``METHOD_NAME`` is the only key litellm populates from a closed request today;
|
||||
the rest are part of the convention's vocabulary and are stamped when the
|
||||
corresponding signal (session, protocol version, resource) is available.
|
||||
"""
|
||||
|
||||
METHOD_NAME: Final = "mcp.method.name"
|
||||
SESSION_ID: Final = "mcp.session.id"
|
||||
PROTOCOL_VERSION: Final = "mcp.protocol.version"
|
||||
RESOURCE_URI: Final = "mcp.resource.uri"
|
||||
|
||||
|
||||
class JsonRpc:
|
||||
"""JSON-RPC keys carried on MCP spans. The error/status code lives in the
|
||||
``rpc.*`` namespace per semconv, not ``jsonrpc.*``."""
|
||||
|
||||
REQUEST_ID: Final = "jsonrpc.request.id"
|
||||
PROTOCOL_VERSION: Final = "jsonrpc.protocol.version"
|
||||
RESPONSE_STATUS_CODE: Final = "rpc.response.status_code"
|
||||
|
||||
|
||||
class NetworkTransport(str, Enum):
|
||||
"""Well-known values for ``network.transport``."""
|
||||
|
||||
TCP = "tcp"
|
||||
UDP = "udp"
|
||||
QUIC = "quic"
|
||||
UNIX = "unix"
|
||||
PIPE = "pipe"
|
||||
|
||||
|
||||
class Network:
|
||||
"""OTel network keys, recommended on MCP spans to describe the transport
|
||||
carrying the JSON-RPC messages (stdio pipe, HTTP, websocket, …)."""
|
||||
|
||||
PROTOCOL_NAME: Final = "network.protocol.name"
|
||||
PROTOCOL_VERSION: Final = "network.protocol.version"
|
||||
TRANSPORT: Final = "network.transport"
|
||||
|
||||
|
||||
class Client:
|
||||
"""Peer (client) network keys, stamped on MCP *server* spans the same way
|
||||
``server.*`` is stamped on client spans."""
|
||||
|
||||
ADDRESS: Final = "client.address"
|
||||
PORT: Final = "client.port"
|
||||
|
||||
|
||||
class Error:
|
||||
|
|
@ -137,6 +204,10 @@ class LiteLLM:
|
|||
SERVICE_NAME: Final = "litellm.service.name"
|
||||
SERVICE_CALL_TYPE: Final = "litellm.service.call_type"
|
||||
PREPROCESSING_MS: Final = "litellm.preprocessing.duration_ms"
|
||||
# The logical name of the MCP server a tool call was routed to. There is no
|
||||
# semconv key for an MCP server's *name* (the convention uses ``server.address``
|
||||
# for its network location), so it lives under the vendor namespace.
|
||||
MCP_SERVER_NAME: Final = "litellm.mcp.server.name"
|
||||
|
||||
|
||||
class Metric:
|
||||
|
|
@ -179,6 +250,7 @@ _OPERATION_BY_CALL_TYPE: dict[str, GenAIOperation] = {
|
|||
"aembedding": GenAIOperation.EMBEDDINGS,
|
||||
"responses": GenAIOperation.CHAT,
|
||||
"aresponses": GenAIOperation.CHAT,
|
||||
"call_mcp_tool": GenAIOperation.EXECUTE_TOOL,
|
||||
}
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -46,6 +46,7 @@ if TYPE_CHECKING:
|
|||
from litellm.integrations.otel.model.payloads import (
|
||||
GuardrailSpanData,
|
||||
LLMCallSpanData,
|
||||
MCPToolCallSpanData,
|
||||
ProxyRequestSpanData,
|
||||
ServiceSpanData,
|
||||
)
|
||||
|
|
@ -54,6 +55,7 @@ if TYPE_CHECKING:
|
|||
class SpanRole(str, Enum):
|
||||
PROXY_REQUEST = "proxy_request"
|
||||
LLM_CALL = "llm_call"
|
||||
MCP_TOOL_CALL = "mcp_tool_call"
|
||||
GUARDRAIL = "guardrail"
|
||||
DB_CALL = "db_call"
|
||||
SERVICE = "service"
|
||||
|
|
@ -81,6 +83,11 @@ SPAN_REGISTRY: dict[SpanRole, SpanSpec] = {
|
|||
SpanRole.LLM_CALL: SpanSpec(
|
||||
SpanRole.LLM_CALL, LiteLLMSpanKind.CLIENT, parent=SpanRole.PROXY_REQUEST
|
||||
),
|
||||
# The proxy is an MCP client to the upstream server it dispatches the tool
|
||||
# call to, so this is a CLIENT span, sibling of the LLM call under the request.
|
||||
SpanRole.MCP_TOOL_CALL: SpanSpec(
|
||||
SpanRole.MCP_TOOL_CALL, LiteLLMSpanKind.CLIENT, parent=SpanRole.PROXY_REQUEST
|
||||
),
|
||||
SpanRole.GUARDRAIL: SpanSpec(
|
||||
SpanRole.GUARDRAIL, LiteLLMSpanKind.INTERNAL, parent=SpanRole.PROXY_REQUEST
|
||||
),
|
||||
|
|
@ -165,6 +172,11 @@ def llm_call_span_name(data: "LLMCallSpanData") -> str:
|
|||
return f"{data.operation.value} {model}".strip()
|
||||
|
||||
|
||||
def mcp_tool_call_span_name(data: "MCPToolCallSpanData") -> str:
|
||||
"""``"{mcp.method.name} {tool}"`` e.g. ``"tools/call get-weather"`` (MCP semconv)."""
|
||||
return f"{data.method} {data.tool_name}".strip()
|
||||
|
||||
|
||||
def proxy_request_span_name(data: "ProxyRequestSpanData") -> str:
|
||||
"""``"{method} {route}"`` (HTTP semconv)."""
|
||||
return f"{data.http_method} {data.route}".strip()
|
||||
|
|
|
|||
|
|
@ -145,6 +145,19 @@ _SESSION_MANAGERS_INITIALIZED = False
|
|||
_INITIALIZATION_LOCK = asyncio.Lock()
|
||||
|
||||
|
||||
def _mcp_session_id_from_headers(
|
||||
raw_headers: Optional[Dict[str, str]],
|
||||
) -> Optional[str]:
|
||||
"""The ``mcp-session-id`` of a stateful MCP session, read case-insensitively
|
||||
from the request headers. ``None`` for stateless calls (no such header)."""
|
||||
if not raw_headers:
|
||||
return None
|
||||
for key, value in raw_headers.items():
|
||||
if isinstance(key, str) and key.lower() == "mcp-session-id":
|
||||
return value or None
|
||||
return None
|
||||
|
||||
|
||||
if MCP_AVAILABLE:
|
||||
from mcp.server import Server
|
||||
from mcp.server.lowlevel.server import NotificationOptions
|
||||
|
|
@ -2324,6 +2337,7 @@ if MCP_AVAILABLE:
|
|||
name=original_tool_name, # Use original name for logging
|
||||
arguments=arguments,
|
||||
server_name=server_name,
|
||||
session_id=_mcp_session_id_from_headers(raw_headers),
|
||||
)
|
||||
)
|
||||
litellm_logging_obj: Optional[LiteLLMLoggingObj] = kwargs.get(
|
||||
|
|
@ -2695,8 +2709,10 @@ if MCP_AVAILABLE:
|
|||
name: str,
|
||||
arguments: Dict[str, Any],
|
||||
server_name: Optional[str],
|
||||
session_id: Optional[str] = None,
|
||||
) -> StandardLoggingMCPToolCall:
|
||||
mcp_server = global_mcp_server_manager._get_mcp_server_from_tool_name(name)
|
||||
namespaced_tool_name = f"{server_name}/{name}" if server_name else name
|
||||
if mcp_server:
|
||||
mcp_info = mcp_server.mcp_info or {}
|
||||
return StandardLoggingMCPToolCall(
|
||||
|
|
@ -2704,13 +2720,15 @@ if MCP_AVAILABLE:
|
|||
arguments=arguments,
|
||||
mcp_server_name=mcp_info.get("server_name"),
|
||||
mcp_server_logo_url=mcp_info.get("logo_url"),
|
||||
namespaced_tool_name=f"{server_name}/{name}" if server_name else name,
|
||||
namespaced_tool_name=namespaced_tool_name,
|
||||
mcp_session_id=session_id,
|
||||
)
|
||||
else:
|
||||
return StandardLoggingMCPToolCall(
|
||||
name=name,
|
||||
arguments=arguments,
|
||||
namespaced_tool_name=f"{server_name}/{name}" if server_name else name,
|
||||
namespaced_tool_name=namespaced_tool_name,
|
||||
mcp_session_id=session_id,
|
||||
)
|
||||
|
||||
async def _handle_managed_mcp_tool(
|
||||
|
|
|
|||
|
|
@ -2580,6 +2580,12 @@ class StandardLoggingMCPToolCall(TypedDict, total=False):
|
|||
Cost per query for the MCP server tool call
|
||||
"""
|
||||
|
||||
mcp_session_id: Optional[str]
|
||||
"""
|
||||
The MCP `mcp-session-id` of the stateful session this tool call ran in, when
|
||||
the client is driving a stateful session. Absent for stateless calls.
|
||||
"""
|
||||
|
||||
|
||||
class StandardLoggingVectorStoreRequest(TypedDict, total=False):
|
||||
"""
|
||||
|
|
|
|||
|
|
@ -238,6 +238,128 @@ def test_idempotent_on_repeat_callback():
|
|||
assert len(exporter.get_finished_spans()) == 1
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# MCP tool-call spans
|
||||
# --------------------------------------------------------------------------- #
|
||||
|
||||
|
||||
def _mcp_payload(**overrides):
|
||||
payload = {
|
||||
"call_type": "call_mcp_tool",
|
||||
"status": "success",
|
||||
"litellm_call_id": "mcp_1",
|
||||
"response_cost": 0.01,
|
||||
"metadata": {"user_api_key_team_id": "t1"},
|
||||
"hidden_params": {},
|
||||
"mcp_tool_call_metadata": {
|
||||
"name": "get_weather",
|
||||
"arguments": {"city": "Paris"},
|
||||
"result": {"temp_c": 21},
|
||||
"mcp_server_name": "weather-mcp",
|
||||
"mcp_session_id": "sess-abc123",
|
||||
},
|
||||
}
|
||||
payload.update(overrides)
|
||||
return payload
|
||||
|
||||
|
||||
def _logger_capturing():
|
||||
from litellm.integrations.otel.model.config import CaptureMessageContent
|
||||
|
||||
cfg = OpenTelemetryV2Config(
|
||||
exporter="in_memory",
|
||||
legacy_compat=False,
|
||||
capture_message_content=CaptureMessageContent.SPAN_ONLY,
|
||||
)
|
||||
exporter = InMemorySpanExporter()
|
||||
tracer_provider = providers.build_tracer_provider(cfg, exporter=exporter)
|
||||
return OpenTelemetryV2(config=cfg, tracer_provider=tracer_provider), exporter
|
||||
|
||||
|
||||
def test_mcp_tool_call_emits_client_span():
|
||||
"""A closed MCP tool call becomes a CLIENT span named ``tools/call {tool}``,
|
||||
carrying the MCP semconv method/operation and the vendor server name."""
|
||||
logger, exporter = _logger()
|
||||
kwargs = {"standard_logging_object": _mcp_payload()}
|
||||
asyncio.run(logger.async_log_success_event(kwargs, None, None, None))
|
||||
(span,) = exporter.get_finished_spans()
|
||||
assert span.name == "tools/call get_weather"
|
||||
assert span.kind is SpanKind.CLIENT
|
||||
assert span.attributes["mcp.method.name"] == "tools/call"
|
||||
assert span.attributes["mcp.session.id"] == "sess-abc123"
|
||||
assert span.attributes[GenAI.OPERATION_NAME] == "execute_tool"
|
||||
assert span.attributes["gen_ai.tool.name"] == "get_weather"
|
||||
assert span.attributes[LiteLLM.MCP_SERVER_NAME] == "weather-mcp"
|
||||
assert span.attributes[LiteLLM.CALL_ID] == "mcp_1"
|
||||
assert span.status.status_code is StatusCode.UNSET
|
||||
# Tool I/O is content: withheld while capture is off (the default).
|
||||
assert "gen_ai.tool.call.arguments" not in span.attributes
|
||||
assert "gen_ai.tool.call.result" not in span.attributes
|
||||
|
||||
|
||||
def test_mcp_tool_call_stateless_omits_session_id():
|
||||
"""A stateless MCP call carries no ``mcp-session-id``, so the span must omit
|
||||
``mcp.session.id`` rather than stamping an empty or ``None`` value."""
|
||||
logger, exporter = _logger()
|
||||
payload = _mcp_payload()
|
||||
del payload["mcp_tool_call_metadata"]["mcp_session_id"]
|
||||
asyncio.run(
|
||||
logger.async_log_success_event(
|
||||
{"standard_logging_object": payload}, None, None, None
|
||||
)
|
||||
)
|
||||
(span,) = exporter.get_finished_spans()
|
||||
assert "mcp.session.id" not in span.attributes
|
||||
assert span.attributes["mcp.method.name"] == "tools/call"
|
||||
|
||||
|
||||
def test_mcp_tool_call_is_not_logged_as_llm_call():
|
||||
"""The MCP branch must short-circuit the LLM-call path: even if ``pre_call``
|
||||
opened a stray carrier for this id, the result is one MCP span, never an LLM
|
||||
``chat`` span."""
|
||||
logger, exporter = _logger()
|
||||
kwargs = {"standard_logging_object": _mcp_payload()}
|
||||
logger.log_pre_api_call(model="MCP: get_weather", messages=[], kwargs=kwargs)
|
||||
asyncio.run(logger.async_log_success_event(kwargs, None, None, None))
|
||||
(span,) = exporter.get_finished_spans()
|
||||
assert span.attributes["mcp.method.name"] == "tools/call"
|
||||
assert "gen_ai.request.model" not in span.attributes
|
||||
|
||||
|
||||
def test_mcp_tool_call_captures_io_when_enabled():
|
||||
logger, exporter = _logger_capturing()
|
||||
kwargs = {"standard_logging_object": _mcp_payload()}
|
||||
asyncio.run(logger.async_log_success_event(kwargs, None, None, None))
|
||||
(span,) = exporter.get_finished_spans()
|
||||
assert '"Paris"' in span.attributes["gen_ai.tool.call.arguments"]
|
||||
assert "21" in span.attributes["gen_ai.tool.call.result"]
|
||||
|
||||
|
||||
def test_mcp_tool_call_failure_marks_error():
|
||||
logger, exporter = _logger()
|
||||
payload = _mcp_payload(
|
||||
status="failure",
|
||||
error_information={"error_class": "MCPError", "error_message": "upstream 500"},
|
||||
)
|
||||
asyncio.run(
|
||||
logger.async_log_failure_event(
|
||||
{"standard_logging_object": payload}, None, None, None
|
||||
)
|
||||
)
|
||||
(span,) = exporter.get_finished_spans()
|
||||
assert span.name == "tools/call get_weather"
|
||||
assert span.status.status_code is StatusCode.ERROR
|
||||
assert span.attributes["error.type"] == "MCPError"
|
||||
|
||||
|
||||
def test_mcp_tool_call_deduped_on_repeat():
|
||||
logger, exporter = _logger()
|
||||
kwargs = {"standard_logging_object": _mcp_payload()}
|
||||
asyncio.run(logger.async_log_success_event(kwargs, None, None, None))
|
||||
asyncio.run(logger.async_log_success_event(kwargs, None, None, None))
|
||||
assert len(exporter.get_finished_spans()) == 1
|
||||
|
||||
|
||||
def test_pre_call_idempotent_keeps_first_span():
|
||||
"""A retried call may re-enter ``pre_call`` with the same call id; the first
|
||||
span (with the true start time) is kept, not replaced."""
|
||||
|
|
|
|||
|
|
@ -1,8 +1,6 @@
|
|||
"""Tests for the OTel v2 sources of truth: span registry, semconv keys, config,
|
||||
and the typed StandardLoggingPayload adapter. These need no OTel SDK."""
|
||||
|
||||
import pytest
|
||||
|
||||
from litellm.integrations.otel import (
|
||||
BAGGAGE_PROMOTED_KEYS,
|
||||
DB,
|
||||
|
|
@ -92,11 +90,14 @@ def test_registry_hierarchy_shape():
|
|||
# guardrail runs before the LLM call exists, so it's a sibling of it.
|
||||
assert set(child_roles(SpanRole.PROXY_REQUEST)) == {
|
||||
SpanRole.LLM_CALL,
|
||||
SpanRole.MCP_TOOL_CALL,
|
||||
SpanRole.GUARDRAIL,
|
||||
SpanRole.DB_CALL,
|
||||
SpanRole.SERVICE,
|
||||
}
|
||||
assert SPAN_REGISTRY[SpanRole.LLM_CALL].kind is LiteLLMSpanKind.CLIENT
|
||||
# The proxy is an MCP client to the upstream tool server: CLIENT span.
|
||||
assert SPAN_REGISTRY[SpanRole.MCP_TOOL_CALL].kind is LiteLLMSpanKind.CLIENT
|
||||
assert SPAN_REGISTRY[SpanRole.PROXY_REQUEST].kind is LiteLLMSpanKind.SERVER
|
||||
assert SPAN_REGISTRY[SpanRole.GUARDRAIL].parent is SpanRole.PROXY_REQUEST
|
||||
# An outbound datastore call is a CLIENT span; an internal service is INTERNAL.
|
||||
|
|
@ -121,14 +122,52 @@ def _all_constants(cls):
|
|||
|
||||
|
||||
def test_attribute_keys_are_unique_across_namespaces():
|
||||
from litellm.integrations.otel import MCP, Client, JsonRpc, Network
|
||||
|
||||
# prefixes are allowed to be substrings; exact keys must not collide.
|
||||
exact = set()
|
||||
for cls in (GenAI, Error, Server, HTTP, DB):
|
||||
for cls in (GenAI, Error, Server, HTTP, DB, MCP, JsonRpc, Network, Client):
|
||||
for key in _all_constants(cls):
|
||||
assert key not in exact, f"duplicate attribute key {key}"
|
||||
exact.add(key)
|
||||
|
||||
|
||||
def test_mcp_attribute_vocabulary_is_complete():
|
||||
"""Every span-attribute key the OTel GenAI MCP semconv defines has a constant.
|
||||
|
||||
Pins the vocabulary so a dropped or renamed key fails here rather than
|
||||
silently emitting a non-conformant attribute name.
|
||||
"""
|
||||
from litellm.integrations.otel import MCP, Client, JsonRpc, Network
|
||||
|
||||
defined = set()
|
||||
for cls in (GenAI, Error, Server, MCP, JsonRpc, Network, Client):
|
||||
defined |= _all_constants(cls)
|
||||
required = {
|
||||
"mcp.method.name",
|
||||
"mcp.session.id",
|
||||
"mcp.protocol.version",
|
||||
"mcp.resource.uri",
|
||||
"jsonrpc.request.id",
|
||||
"jsonrpc.protocol.version",
|
||||
"rpc.response.status_code",
|
||||
"gen_ai.operation.name",
|
||||
"gen_ai.tool.name",
|
||||
"gen_ai.tool.call.arguments",
|
||||
"gen_ai.tool.call.result",
|
||||
"gen_ai.prompt.name",
|
||||
"error.type",
|
||||
"server.address",
|
||||
"server.port",
|
||||
"client.address",
|
||||
"client.port",
|
||||
"network.protocol.name",
|
||||
"network.protocol.version",
|
||||
"network.transport",
|
||||
}
|
||||
assert required <= defined, f"missing MCP semconv keys: {required - defined}"
|
||||
|
||||
|
||||
def test_provider_resolution():
|
||||
assert resolve_provider("openai") == "openai"
|
||||
assert resolve_provider("bedrock") == "aws.bedrock"
|
||||
|
|
@ -143,6 +182,101 @@ def test_operation_resolution():
|
|||
assert resolve_operation("aembedding") is GenAIOperation.EMBEDDINGS
|
||||
assert resolve_operation("atext_completion") is GenAIOperation.TEXT_COMPLETION
|
||||
assert resolve_operation(None) is GenAIOperation.CHAT
|
||||
# An MCP tool call is an ``execute_tool`` operation, not a chat completion.
|
||||
assert resolve_operation("call_mcp_tool") is GenAIOperation.EXECUTE_TOOL
|
||||
|
||||
|
||||
# --- MCP tool-call (source of truth #1/#2/#3) ------------------------------- #
|
||||
|
||||
|
||||
def _mcp_payload(capture=False, **overrides):
|
||||
payload = {
|
||||
"call_type": "call_mcp_tool",
|
||||
"status": "success",
|
||||
"litellm_call_id": "mcp_call_1",
|
||||
"response_cost": 0.01,
|
||||
"metadata": {"user_api_key_team_id": "t1"},
|
||||
"hidden_params": {},
|
||||
"mcp_tool_call_metadata": {
|
||||
"name": "get_weather",
|
||||
"arguments": {"city": "Paris"},
|
||||
"result": {"temp_c": 21},
|
||||
"mcp_server_name": "weather-mcp",
|
||||
"mcp_session_id": "sess-abc123",
|
||||
},
|
||||
}
|
||||
payload.update(overrides)
|
||||
return payload
|
||||
|
||||
|
||||
def test_mcp_method_values_match_wire_format():
|
||||
from litellm.integrations.otel import MCP, MCPMethod
|
||||
|
||||
assert MCPMethod.TOOLS_CALL.value == "tools/call"
|
||||
assert MCPMethod.TOOLS_LIST.value == "tools/list"
|
||||
assert MCP.METHOD_NAME == "mcp.method.name"
|
||||
|
||||
|
||||
def test_mcp_tool_call_adapter_extracts_fields():
|
||||
from litellm.integrations.otel import MCPToolCallSpanData
|
||||
|
||||
data = MCPToolCallSpanData.from_standard_logging_payload(_mcp_payload())
|
||||
assert data.operation is GenAIOperation.EXECUTE_TOOL
|
||||
assert data.method == "tools/call"
|
||||
assert data.tool_name == "get_weather"
|
||||
assert data.server_name == "weather-mcp"
|
||||
assert data.session_id == "sess-abc123"
|
||||
assert data.response_cost == 0.01
|
||||
assert data.identity.call_id == "mcp_call_1"
|
||||
assert data.identity.team_id == "t1"
|
||||
assert data.error is None
|
||||
|
||||
|
||||
def test_mcp_tool_call_content_gated_off_by_default():
|
||||
# Arguments and result are sensitive tool I/O: withheld unless content capture
|
||||
# is explicitly enabled, exactly like prompt/response bodies.
|
||||
from litellm.integrations.otel import MCPToolCallSpanData
|
||||
|
||||
off = MCPToolCallSpanData.from_standard_logging_payload(_mcp_payload())
|
||||
assert off.arguments_json is None and off.result_json is None
|
||||
|
||||
on = MCPToolCallSpanData.from_standard_logging_payload(
|
||||
_mcp_payload(), capture_content=True
|
||||
)
|
||||
assert on.arguments_json is not None and '"Paris"' in on.arguments_json
|
||||
assert on.result_json is not None and "21" in on.result_json
|
||||
|
||||
|
||||
def test_mcp_tool_call_failure_path():
|
||||
from litellm.integrations.otel import MCPToolCallSpanData
|
||||
|
||||
data = MCPToolCallSpanData.from_standard_logging_payload(
|
||||
_mcp_payload(
|
||||
status="failure",
|
||||
error_information={"error_class": "MCPError", "error_message": "boom"},
|
||||
)
|
||||
)
|
||||
assert data.error is not None
|
||||
assert data.error.error_type == "MCPError"
|
||||
assert data.error.message == "boom"
|
||||
|
||||
|
||||
def test_is_mcp_tool_call_detection():
|
||||
from litellm.integrations.otel import is_mcp_tool_call
|
||||
|
||||
assert is_mcp_tool_call(_mcp_payload()) is True
|
||||
# call_type alone is enough even before the gateway stamps its metadata.
|
||||
assert is_mcp_tool_call({"call_type": "call_mcp_tool"}) is True
|
||||
assert is_mcp_tool_call({"call_type": "acompletion"}) is False
|
||||
assert is_mcp_tool_call({}) is False
|
||||
|
||||
|
||||
def test_mcp_tool_call_span_name():
|
||||
from litellm.integrations.otel import MCPToolCallSpanData
|
||||
from litellm.integrations.otel.model.spans import mcp_tool_call_span_name
|
||||
|
||||
data = MCPToolCallSpanData.from_standard_logging_payload(_mcp_payload())
|
||||
assert mcp_tool_call_span_name(data) == "tools/call get_weather"
|
||||
|
||||
|
||||
# --- typed adapter (source of truth #3) ------------------------------------- #
|
||||
|
|
|
|||
|
|
@ -0,0 +1,19 @@
|
|||
"""The MCP ``mcp-session-id`` is captured for tool-call logging so the otel span
|
||||
can carry ``mcp.session.id``. Guards the header read against casing and absence."""
|
||||
|
||||
from litellm.proxy._experimental.mcp_server.server import _mcp_session_id_from_headers
|
||||
|
||||
|
||||
def test_reads_session_id_case_insensitively():
|
||||
# Clients send varied casing (``Mcp-Session-Id``, ``mcp-session-id``); all resolve.
|
||||
assert _mcp_session_id_from_headers({"mcp-session-id": "s1"}) == "s1"
|
||||
assert _mcp_session_id_from_headers({"Mcp-Session-Id": "s2"}) == "s2"
|
||||
assert _mcp_session_id_from_headers({"MCP-SESSION-ID": "s3"}) == "s3"
|
||||
|
||||
|
||||
def test_stateless_call_has_no_session_id():
|
||||
# No header (stateless request) and an empty value both yield None, not "".
|
||||
assert _mcp_session_id_from_headers({"authorization": "Bearer x"}) is None
|
||||
assert _mcp_session_id_from_headers({"mcp-session-id": ""}) is None
|
||||
assert _mcp_session_id_from_headers(None) is None
|
||||
assert _mcp_session_id_from_headers({}) is None
|
||||
Loading…
Add table
Reference in a new issue