mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-06 02:48:13 +00:00
feat(traces): tracing development seed (#44363)
* feat(dev): seed linked tracing and spend fixtures * chore(dev): use OpenAI model in tracing config * chore(dev): align tracing credentials with UI E2E * fix(dev): update fixture seeder query scope * feat(dev): seed linked tracing and spend fixtures * chore(dev): use OpenAI model in tracing config * chore(dev): align tracing credentials with UI E2E * fix(dev): update fixture seeder query scope * wip * wip * wip * chore(trace): checkpoint ongoing Rust migration * refactor(trace): group Python bridge under trace package * refactor(traces): read span conventions through a Convention trait Each span format (Claude Code, LangSmith, OpenInference, gen_ai) now lives under normalize/convention/ as a unit struct implementing Convention, owning both its detection and its extraction. Precedence is one ordered registry instead of an if-chain in mod.rs that reached into each module differently. The modules now share one way to read attributes: present() for the first non-empty key and Payload for a text that also reports the key it consumed, replacing three different idioms and the &mut Vec threaded through payload readers. Instrumentation::adjust returns a new Extraction instead of mutating one, with each SDK rule as its own function, and the LangChain middleware suffix list exists once. * feat(trace): export Rust-owned wire schemas and enforce contract bounds * fix(trace): bound quoted counts in ClickHouse wire schemas * feat(trace): generate Python wire contracts with datamodel-code-generator * test(trace): validate migrated callers and generated contracts at the native boundary * refactor(traces): rename normalization convention to format * fix(traces): reconcile spend evidence and preserve unknown costs * feat(traces): normalize additional telemetry formats * test(traces): cover captured normalization fixtures * refactor(traces): isolate SDK normalization rules * feat(tracing): seed all trace exports for local dashboard * fix(clickhouse): preserve custom LiteLLM request metadata * docs(traces): define normalization module boundaries * docs(traces): define resolution and OTLP boundaries * fix(ui): normalize nullable trace message names * refactor(traces): split resolver modules and cover resolution behavior * test(traces): replace normalization snapshots with behavior assertions * fix(ui): align dashboard API contracts with generated types * refactor(traces): type normalization and storage boundaries * fix(traces): seed captured SDK spend and preserve provider identities * wip * test(traces): verify guide discovery and content ordering --------- Co-authored-by: Yujong Lee <yujong@berri.ai> Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
parent
d260765652
commit
e340e546e2
256 changed files with 41414 additions and 5194 deletions
4
.github/workflows/test-litellm-ui-unit.yml
vendored
4
.github/workflows/test-litellm-ui-unit.yml
vendored
|
|
@ -49,6 +49,10 @@ jobs:
|
|||
if: steps.changes.outputs.decision != 'skip'
|
||||
run: npm ci
|
||||
|
||||
- name: Check UI production source types
|
||||
if: steps.changes.outputs.decision != 'skip'
|
||||
run: npm run typecheck
|
||||
|
||||
- name: Run UI type tests (Vitest)
|
||||
if: steps.changes.outputs.decision != 'skip'
|
||||
env:
|
||||
|
|
|
|||
12
.github/workflows/test-rust.yml
vendored
12
.github/workflows/test-rust.yml
vendored
|
|
@ -5,6 +5,8 @@ on:
|
|||
paths:
|
||||
- "litellm-rust/**"
|
||||
- "litellm/rust_bridge/**"
|
||||
- "scripts/generate_trace_types.py"
|
||||
- "scripts/trace_codegen/**"
|
||||
- "tests/test_litellm_rust/**"
|
||||
- "litellm/integrations/custom_logger.py"
|
||||
- "litellm/litellm_core_utils/litellm_logging.py"
|
||||
|
|
@ -32,6 +34,8 @@ on:
|
|||
paths:
|
||||
- "litellm-rust/**"
|
||||
- "litellm/rust_bridge/**"
|
||||
- "scripts/generate_trace_types.py"
|
||||
- "scripts/trace_codegen/**"
|
||||
- "tests/test_litellm_rust/**"
|
||||
- "litellm/integrations/custom_logger.py"
|
||||
- "litellm/litellm_core_utils/litellm_logging.py"
|
||||
|
|
@ -85,7 +89,7 @@ jobs:
|
|||
cache-on-failure: true
|
||||
save-if: ${{ github.ref == 'refs/heads/main' }}
|
||||
|
||||
- run: cargo clippy --workspace --all-targets --locked -- -D warnings
|
||||
- run: cargo clippy --workspace --all-targets --locked --features litellm-traces/schema,litellm-traces-clickhouse/schema -- -D warnings
|
||||
|
||||
rust-test:
|
||||
runs-on: ubuntu-latest
|
||||
|
|
@ -124,7 +128,11 @@ jobs:
|
|||
cache-on-failure: true
|
||||
save-if: ${{ github.ref == 'refs/heads/main' }}
|
||||
|
||||
- run: cargo nextest run --workspace --locked
|
||||
- name: Check generated trace contracts
|
||||
working-directory: .
|
||||
run: uv run scripts/generate_trace_types.py --check
|
||||
|
||||
- run: cargo nextest run --workspace --locked --features litellm-traces/schema,litellm-traces-clickhouse/schema
|
||||
|
||||
- run: cargo test --workspace --doc --locked
|
||||
|
||||
|
|
|
|||
|
|
@ -2,5 +2,6 @@ FROM python:3.12-slim
|
|||
WORKDIR /app
|
||||
RUN pip install --no-cache-dir httpx==0.28.1 pydantic==2.11.7
|
||||
COPY litellm/proxy/lens/__init__.py litellm/proxy/lens/models.py litellm/proxy/lens/trace_store.py litellm/proxy/lens/analysis.py litellm/proxy/lens/worker.py /app/lens/
|
||||
COPY litellm/proxy/lens/prompts/ /app/lens/prompts/
|
||||
USER 65532:65532
|
||||
CMD ["python", "-m", "lens.worker"]
|
||||
|
|
|
|||
|
|
@ -4,5 +4,8 @@
|
|||
!litellm/proxy/lens/
|
||||
!litellm/proxy/lens/__init__.py
|
||||
!litellm/proxy/lens/models.py
|
||||
!litellm/proxy/lens/trace_store.py
|
||||
!litellm/proxy/lens/analysis.py
|
||||
!litellm/proxy/lens/worker.py
|
||||
!litellm/proxy/lens/prompts/
|
||||
!litellm/proxy/lens/prompts/**
|
||||
|
|
|
|||
|
|
@ -7,7 +7,8 @@ services:
|
|||
target: runtime
|
||||
command: ["--config", "/app/tracing-config.yaml", "--port", "4000"]
|
||||
environment:
|
||||
LITELLM_MASTER_KEY: local-tracing-master-key
|
||||
LITELLM_MASTER_KEY: sk-1234
|
||||
LITELLM_DANGEROUSLY_PERMIT_WEAK_OR_UNSET_MASTER_KEY: "true"
|
||||
LITELLM_SALT_KEY: sk-local-tracing-salt-key
|
||||
DATABASE_URL: postgresql://litellm:litellm@db:5432/litellm
|
||||
STORE_MODEL_IN_DB: "True"
|
||||
|
|
|
|||
10
litellm-rust/Cargo.lock
generated
10
litellm-rust/Cargo.lock
generated
|
|
@ -4453,15 +4453,20 @@ dependencies = [
|
|||
name = "litellm-traces"
|
||||
version = "0.1.0"
|
||||
dependencies = [
|
||||
"askama",
|
||||
"criterion",
|
||||
"indexmap 2.14.0",
|
||||
"litellm-llms-types",
|
||||
"macro_rules_attribute",
|
||||
"opentelemetry-proto",
|
||||
"prost",
|
||||
"rstest",
|
||||
"schemars 1.2.2",
|
||||
"serde",
|
||||
"serde_json",
|
||||
"strum",
|
||||
"thiserror 2.0.19",
|
||||
"time",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
|
|
@ -4469,15 +4474,19 @@ name = "litellm-traces-clickhouse"
|
|||
version = "0.1.0"
|
||||
dependencies = [
|
||||
"askama",
|
||||
"base64 0.22.1",
|
||||
"flate2",
|
||||
"futures-util",
|
||||
"hmac 0.12.1",
|
||||
"jsonschema",
|
||||
"litellm-http",
|
||||
"litellm-migrate",
|
||||
"litellm-storage-clickhouse",
|
||||
"litellm-traces",
|
||||
"macro_rules_attribute",
|
||||
"moka",
|
||||
"rstest",
|
||||
"schemars 1.2.2",
|
||||
"serde",
|
||||
"serde_json",
|
||||
"sha2 0.10.9",
|
||||
|
|
@ -4486,6 +4495,7 @@ dependencies = [
|
|||
"thiserror 2.0.19",
|
||||
"time",
|
||||
"tokio",
|
||||
"tracing",
|
||||
"url",
|
||||
"wiremock",
|
||||
]
|
||||
|
|
|
|||
|
|
@ -45,8 +45,7 @@ mod _native {
|
|||
use crate::routes::token_counter::TokenCounter;
|
||||
#[pymodule_export]
|
||||
use crate::routes::traces::{
|
||||
NativeTraceConfig, NativeTraceStorage, trace_decode_otlp, trace_encode_error,
|
||||
trace_normalized_field_definitions,
|
||||
NativeTraceConfig, NativeTraceStorage, trace_encode_error, trace_span_rows,
|
||||
};
|
||||
#[cfg(feature = "huggingface")]
|
||||
#[pymodule_export]
|
||||
|
|
@ -114,9 +113,8 @@ mod tests {
|
|||
"NativeDiagnosticProcessor",
|
||||
"NativeTraceConfig",
|
||||
"NativeTraceStorage",
|
||||
"trace_decode_otlp",
|
||||
"trace_encode_error",
|
||||
"trace_normalized_field_definitions",
|
||||
"trace_span_rows",
|
||||
"TokenCounter",
|
||||
"Tokenizer",
|
||||
"gil_stats",
|
||||
|
|
|
|||
|
|
@ -1,14 +1,13 @@
|
|||
use std::collections::BTreeMap;
|
||||
|
||||
use litellm_host_python::{FromPythonCache, ToPythonCache};
|
||||
use litellm_http::ClientVariant;
|
||||
use litellm_traces::{QueryScope, ReadQuery, Shared};
|
||||
use litellm_traces::{QueryScope, ReadQuery, Tenant, query::named::ReadAccessParams};
|
||||
use litellm_traces_clickhouse::{Config, Error, InsertTable, Parameter, QueryReaders};
|
||||
use prost::Message;
|
||||
use pyo3::{
|
||||
exceptions::{PyOverflowError, PyRuntimeError, PyValueError},
|
||||
prelude::*,
|
||||
types::{PyBytes, PyDict, PyList, PyMapping, PyString},
|
||||
types::PyBytes,
|
||||
};
|
||||
|
||||
#[derive(Message)]
|
||||
|
|
@ -36,14 +35,20 @@ fn map_error_ref(error: &Error) -> PyErr {
|
|||
use litellm_storage_clickhouse::Error as StorageError;
|
||||
|
||||
match error {
|
||||
Error::Decode(litellm_traces::Error::TooLarge) | Error::InsertTooLarge => {
|
||||
PyOverflowError::new_err(error.to_string())
|
||||
}
|
||||
Error::InvalidRow
|
||||
| Error::InvalidTable
|
||||
| Error::InvalidCursor(_)
|
||||
| Error::AmbiguousTrace
|
||||
| Error::Decode(_)
|
||||
| Error::InvalidSchema
|
||||
| Error::InvalidQuery
|
||||
| Error::InvalidParameters
|
||||
| Error::InvalidScope => PyValueError::new_err(error.to_string()),
|
||||
Error::InsertTooLarge => PyOverflowError::new_err(error.to_string()),
|
||||
Error::SchemaFailed(_)
|
||||
Error::Task
|
||||
| Error::SchemaFailed(_)
|
||||
| Error::SchemaTransport
|
||||
| Error::MissingSecret
|
||||
| Error::Busy
|
||||
|
|
@ -87,9 +92,15 @@ pub struct NativeTraceConfig {
|
|||
#[pymethods]
|
||||
impl NativeTraceConfig {
|
||||
#[new]
|
||||
fn new(database: String, url: &str, retention_days: u32) -> PyResult<Self> {
|
||||
fn new(
|
||||
database: String,
|
||||
url: &str,
|
||||
retention_days: u32,
|
||||
max_attribute_value_bytes: usize,
|
||||
) -> PyResult<Self> {
|
||||
Ok(Self {
|
||||
inner: Config::new(database, url, retention_days).map_err(map_error)?,
|
||||
inner: Config::new(database, url, retention_days, max_attribute_value_bytes)
|
||||
.map_err(map_error)?,
|
||||
})
|
||||
}
|
||||
}
|
||||
|
|
@ -137,7 +148,9 @@ impl NativeTraceStorage {
|
|||
&self,
|
||||
py: Python<'py>,
|
||||
table: &str,
|
||||
#[pyo3(from_py_with = insert_rows_from_py)] rows: Vec<litellm_traces_clickhouse::InsertRow>,
|
||||
#[pyo3(from_py_with = litellm_host_python::from_py_argument)] rows: Vec<
|
||||
BTreeMap<String, serde_json::Value>,
|
||||
>,
|
||||
) -> PyResult<Bound<'py, PyAny>> {
|
||||
let table = InsertTable::parse(table).map_err(map_error)?;
|
||||
let client = crate::http::host_client(py, ClientVariant::NoRedirect)?;
|
||||
|
|
@ -146,13 +159,156 @@ impl NativeTraceStorage {
|
|||
crate::execution::run_async(
|
||||
py,
|
||||
async move {
|
||||
litellm_traces_clickhouse::insert_rows(&client, &connection, &database, table, rows)
|
||||
.await
|
||||
},
|
||||
map_error,
|
||||
)
|
||||
}
|
||||
|
||||
fn ingest<'py>(
|
||||
&self,
|
||||
py: Python<'py>,
|
||||
payload: &[u8],
|
||||
content_type: Option<String>,
|
||||
#[pyo3(from_py_with = litellm_host_python::from_py_argument)] tenant: Tenant,
|
||||
) -> PyResult<Bound<'py, PyAny>> {
|
||||
let payload = payload.to_vec();
|
||||
let max_value_bytes = self.config.max_attribute_value_bytes();
|
||||
let client = crate::http::host_client(py, ClientVariant::NoRedirect)?;
|
||||
let connection = self.config.storage().writer().clone();
|
||||
let database = self.config.storage().database().to_owned();
|
||||
crate::execution::run_async(
|
||||
py,
|
||||
async move {
|
||||
let rows = tokio::task::spawn_blocking(move || {
|
||||
litellm_traces::decode_otlp(&payload, content_type.as_deref()).map(|spans| {
|
||||
litellm_traces_clickhouse::span_rows(spans, &tenant, max_value_bytes)
|
||||
})
|
||||
})
|
||||
.await
|
||||
.map_err(|_| Error::Task)??;
|
||||
let count = rows.len();
|
||||
litellm_traces_clickhouse::insert_shared_rows(
|
||||
&client,
|
||||
&connection,
|
||||
&database,
|
||||
table,
|
||||
InsertTable::OtelTraces,
|
||||
rows,
|
||||
)
|
||||
.await?;
|
||||
Ok(count)
|
||||
},
|
||||
map_error,
|
||||
)
|
||||
}
|
||||
|
||||
#[pyo3(signature = (scope, start_ms, end_ms, cursor, limit))]
|
||||
fn list_traces<'py>(
|
||||
&self,
|
||||
py: Python<'py>,
|
||||
#[pyo3(from_py_with = litellm_host_python::from_py_argument)] scope: ReadAccessParams,
|
||||
start_ms: i64,
|
||||
end_ms: i64,
|
||||
cursor: Option<String>,
|
||||
limit: u32,
|
||||
) -> PyResult<Bound<'py, PyAny>> {
|
||||
let client = crate::http::host_client(py, ClientVariant::NoRedirect)?;
|
||||
let connection = self.config.storage().reader().clone();
|
||||
crate::execution::run_async(
|
||||
py,
|
||||
async move {
|
||||
litellm_traces_clickhouse::list_traces(
|
||||
&client,
|
||||
&connection,
|
||||
&scope,
|
||||
start_ms,
|
||||
end_ms,
|
||||
cursor.as_deref(),
|
||||
limit,
|
||||
)
|
||||
.await
|
||||
},
|
||||
map_error,
|
||||
)
|
||||
}
|
||||
|
||||
fn get_trace<'py>(
|
||||
&self,
|
||||
py: Python<'py>,
|
||||
trace_id: String,
|
||||
#[pyo3(from_py_with = litellm_host_python::from_py_argument)] scope: ReadAccessParams,
|
||||
trace_ref: String,
|
||||
) -> PyResult<Bound<'py, PyAny>> {
|
||||
let client = crate::http::host_client(py, ClientVariant::NoRedirect)?;
|
||||
let connection = self.config.storage().reader().clone();
|
||||
crate::execution::run_async(
|
||||
py,
|
||||
async move {
|
||||
litellm_traces_clickhouse::get_trace(
|
||||
&client,
|
||||
&connection,
|
||||
&scope,
|
||||
&trace_id,
|
||||
&trace_ref,
|
||||
)
|
||||
.await
|
||||
},
|
||||
map_error,
|
||||
)
|
||||
}
|
||||
|
||||
fn get_span<'py>(
|
||||
&self,
|
||||
py: Python<'py>,
|
||||
trace_id: String,
|
||||
span_id: String,
|
||||
#[pyo3(from_py_with = litellm_host_python::from_py_argument)] scope: ReadAccessParams,
|
||||
trace_ref: String,
|
||||
) -> PyResult<Bound<'py, PyAny>> {
|
||||
let client = crate::http::host_client(py, ClientVariant::NoRedirect)?;
|
||||
let connection = self.config.storage().reader().clone();
|
||||
crate::execution::run_async(
|
||||
py,
|
||||
async move {
|
||||
litellm_traces_clickhouse::get_span(
|
||||
&client,
|
||||
&connection,
|
||||
&scope,
|
||||
&trace_id,
|
||||
&span_id,
|
||||
&trace_ref,
|
||||
)
|
||||
.await
|
||||
},
|
||||
map_error,
|
||||
)
|
||||
}
|
||||
|
||||
#[pyo3(signature = (trace_id, span_id, scope, trace_ref, cursor))]
|
||||
fn get_span_error<'py>(
|
||||
&self,
|
||||
py: Python<'py>,
|
||||
trace_id: String,
|
||||
span_id: String,
|
||||
#[pyo3(from_py_with = litellm_host_python::from_py_argument)] scope: ReadAccessParams,
|
||||
trace_ref: String,
|
||||
cursor: Option<String>,
|
||||
) -> PyResult<Bound<'py, PyAny>> {
|
||||
let client = crate::http::host_client(py, ClientVariant::NoRedirect)?;
|
||||
let connection = self.config.storage().reader().clone();
|
||||
crate::execution::run_async(
|
||||
py,
|
||||
async move {
|
||||
litellm_traces_clickhouse::get_span_error(
|
||||
&client,
|
||||
&connection,
|
||||
&scope,
|
||||
&trace_id,
|
||||
&span_id,
|
||||
&trace_ref,
|
||||
cursor.as_deref(),
|
||||
)
|
||||
.await
|
||||
},
|
||||
map_error,
|
||||
|
|
@ -232,101 +388,23 @@ impl NativeTraceStorage {
|
|||
}
|
||||
}
|
||||
|
||||
/// The `otel_traces` rows an export would be stored as, without writing them.
|
||||
#[pyfunction]
|
||||
pub fn trace_decode_otlp<'py>(
|
||||
pub fn trace_span_rows<'py>(
|
||||
py: Python<'py>,
|
||||
body: &[u8],
|
||||
content_type: Option<&str>,
|
||||
#[pyo3(from_py_with = litellm_host_python::from_py_argument)] tenant: Tenant,
|
||||
max_attribute_value_bytes: usize,
|
||||
) -> PyResult<Bound<'py, PyAny>> {
|
||||
let spans = py
|
||||
.detach(|| litellm_traces::decode_otlp(body, content_type))
|
||||
.map_err(|error| match error {
|
||||
litellm_traces::Error::TooLarge => PyOverflowError::new_err(error.to_string()),
|
||||
_ => PyValueError::new_err(error.to_string()),
|
||||
})?;
|
||||
spans_to_py(py, &spans).map(Bound::into_any)
|
||||
}
|
||||
|
||||
fn insert_rows_from_py(
|
||||
value: &Bound<'_, PyAny>,
|
||||
) -> PyResult<Vec<litellm_traces_clickhouse::InsertRow>> {
|
||||
let mut resources = FromPythonCache::default();
|
||||
value
|
||||
.try_iter()?
|
||||
.map(|row| {
|
||||
let row = row?;
|
||||
let mut fields = BTreeMap::new();
|
||||
for item in row.cast::<PyMapping>()?.items()?.iter() {
|
||||
let (key, value): (String, Bound<'_, PyAny>) = item.extract()?;
|
||||
let converted = if matches!(
|
||||
key.as_str(),
|
||||
"ResourceAttributes" | "ScopeName" | "ScopeVersion"
|
||||
) {
|
||||
resources
|
||||
.get_or_try_insert_with(&value, |value| {
|
||||
litellm_host_python::from_py_argument::<serde_json::Value>(value)
|
||||
.map(Shared::new)
|
||||
})?
|
||||
.clone()
|
||||
} else {
|
||||
Shared::new(litellm_host_python::from_py_argument(&value)?)
|
||||
};
|
||||
fields.insert(key, converted);
|
||||
}
|
||||
Ok(fields)
|
||||
let rows = py
|
||||
.detach(|| {
|
||||
litellm_traces::decode_otlp(body, content_type).map(|spans| {
|
||||
litellm_traces_clickhouse::span_rows(spans, &tenant, max_attribute_value_bytes)
|
||||
})
|
||||
})
|
||||
.collect()
|
||||
}
|
||||
|
||||
fn spans_to_py<'py>(
|
||||
py: Python<'py>,
|
||||
spans: &[litellm_traces::DecodedSpan],
|
||||
) -> PyResult<Bound<'py, PyList>> {
|
||||
let mut resources = ToPythonCache::default();
|
||||
let mut scopes = ToPythonCache::default();
|
||||
let result = PyList::empty(py);
|
||||
for span in spans {
|
||||
let resource = resources
|
||||
.get_or_try_insert_with(span.resource_attributes.as_ref(), |value| {
|
||||
litellm_host_python::Pythonized(value).into_pyobject(py)
|
||||
})?;
|
||||
let row = PyDict::new(py);
|
||||
row.set_item("trace_id", &span.trace_id)?;
|
||||
row.set_item("span_id", &span.span_id)?;
|
||||
row.set_item("parent_span_id", &span.parent_span_id)?;
|
||||
row.set_item("trace_state", &span.trace_state)?;
|
||||
row.set_item("name", &span.name)?;
|
||||
row.set_item("kind", &span.kind)?;
|
||||
row.set_item("resource_attributes", resource)?;
|
||||
for (key, value) in [
|
||||
("scope_name", &span.scope_name),
|
||||
("scope_version", &span.scope_version),
|
||||
] {
|
||||
let value = scopes.get_or_try_insert_with(value.as_ref(), |value| {
|
||||
Ok(PyString::new(py, value).into_any())
|
||||
})?;
|
||||
row.set_item(key, value)?;
|
||||
}
|
||||
row.set_item("attributes", &span.attributes)?;
|
||||
row.set_item("start_ns", span.start_ns)?;
|
||||
row.set_item("end_ns", span.end_ns)?;
|
||||
row.set_item("status_code", &span.status_code)?;
|
||||
row.set_item("status_message", &span.status_message)?;
|
||||
row.set_item(
|
||||
"events",
|
||||
litellm_host_python::Pythonized(&span.events).into_pyobject(py)?,
|
||||
)?;
|
||||
row.set_item(
|
||||
"normalized",
|
||||
litellm_host_python::Pythonized(&span.normalized).into_pyobject(py)?,
|
||||
)?;
|
||||
row.set_item(
|
||||
"consumed_attributes",
|
||||
litellm_host_python::Pythonized(&span.consumed_attributes).into_pyobject(py)?,
|
||||
)?;
|
||||
result.append(row)?;
|
||||
}
|
||||
Ok(result)
|
||||
.map_err(|error| map_error(error.into()))?;
|
||||
litellm_host_python::Pythonized(rows).into_pyobject(py)
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
|
|
@ -383,50 +461,20 @@ mod tests {
|
|||
}
|
||||
|
||||
#[rstest]
|
||||
fn insert_projection_preserves_identity_without_merging_equal_resources() {
|
||||
#[case::decode_budget(Error::Decode(litellm_traces::Error::TooLarge), "OverflowError")]
|
||||
#[case::invalid_export(Error::Decode(litellm_traces::Error::InvalidPayload), "ValueError")]
|
||||
#[case::cursor(Error::InvalidCursor("trace"), "ValueError")]
|
||||
#[case::ambiguous(Error::AmbiguousTrace, "ValueError")]
|
||||
fn trace_read_and_ingest_failures_preserve_public_exception_types(
|
||||
#[case] error: Error,
|
||||
#[case] exception_name: &str,
|
||||
) {
|
||||
Python::initialize();
|
||||
Python::attach(|py| {
|
||||
let resource = PyDict::new(py);
|
||||
resource.set_item("service.name", "shared").unwrap();
|
||||
let equal_resource = resource.copy().unwrap();
|
||||
let rows = PyList::empty(py);
|
||||
for value in [&resource, &resource, &equal_resource] {
|
||||
let row = PyDict::new(py);
|
||||
row.set_item("ResourceAttributes", value).unwrap();
|
||||
rows.append(row).unwrap();
|
||||
}
|
||||
let projected = insert_rows_from_py(rows.as_any()).unwrap();
|
||||
assert!(Shared::shares_storage_with(
|
||||
&projected[0]["ResourceAttributes"],
|
||||
&projected[1]["ResourceAttributes"]
|
||||
));
|
||||
assert!(!Shared::shares_storage_with(
|
||||
&projected[0]["ResourceAttributes"],
|
||||
&projected[2]["ResourceAttributes"]
|
||||
));
|
||||
assert_eq!(projected[0], projected[2]);
|
||||
});
|
||||
}
|
||||
|
||||
#[rstest]
|
||||
fn shared_conversion_preserves_every_decoded_field() {
|
||||
Python::initialize();
|
||||
Python::attach(|py| {
|
||||
let spans = litellm_traces::decode_otlp(
|
||||
include_bytes!("../../../../../tests/test_litellm/tracing/fixtures/langsmith_deep_agent_export.json"),
|
||||
Some("application/json"),
|
||||
).unwrap();
|
||||
let expected = litellm_host_python::Pythonized(&spans)
|
||||
.into_pyobject(py)
|
||||
.unwrap();
|
||||
let actual = spans_to_py(py, &spans).unwrap();
|
||||
assert!(actual.eq(expected).unwrap());
|
||||
assert_eq!(
|
||||
map_error(error).get_type(py).name().unwrap(),
|
||||
exception_name
|
||||
);
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
#[pyfunction]
|
||||
pub fn trace_normalized_field_definitions<'py>(py: Python<'py>) -> PyResult<Bound<'py, PyAny>> {
|
||||
litellm_host_python::Pythonized(litellm_traces_clickhouse::NORMALIZED_FIELD_DEFINITIONS)
|
||||
.into_pyobject(py)
|
||||
}
|
||||
|
|
|
|||
|
|
@ -5,8 +5,14 @@ edition.workspace = true
|
|||
license.workspace = true
|
||||
repository.workspace = true
|
||||
|
||||
[features]
|
||||
schema = ["dep:schemars", "litellm-traces/schema"]
|
||||
|
||||
[dependencies]
|
||||
macro_rules_attribute.workspace = true
|
||||
schemars = { workspace = true, optional = true }
|
||||
askama.workspace = true
|
||||
base64.workspace = true
|
||||
flate2.workspace = true
|
||||
futures-util.workspace = true
|
||||
hmac = "0.12.1"
|
||||
|
|
@ -22,10 +28,17 @@ strum.workspace = true
|
|||
thiserror.workspace = true
|
||||
time = { workspace = true, features = ["formatting"] }
|
||||
tokio.workspace = true
|
||||
tracing.workspace = true
|
||||
url.workspace = true
|
||||
|
||||
[dev-dependencies]
|
||||
jsonschema = { version = "0.55.1", default-features = false }
|
||||
litellm-http = { workspace = true, features = ["test-support"] }
|
||||
rstest.workspace = true
|
||||
testcontainers-modules = { version = "0.15.0", features = ["clickhouse"] }
|
||||
wiremock.workspace = true
|
||||
|
||||
[[bin]]
|
||||
name = "export-traces-clickhouse-schema"
|
||||
path = "src/bin/export_schema.rs"
|
||||
required-features = ["schema"]
|
||||
|
|
|
|||
|
|
@ -0,0 +1,5 @@
|
|||
ALTER TABLE {database}.otel_traces
|
||||
ADD COLUMN IF NOT EXISTS WrapperCandidate Bool DEFAULT false AFTER ObservationType,
|
||||
ADD COLUMN IF NOT EXISTS CallKeys Array(String) DEFAULT [] AFTER LiteLLMRequestId,
|
||||
ADD COLUMN IF NOT EXISTS CallEvidence LowCardinality(String) DEFAULT '' AFTER CallKeys,
|
||||
ADD COLUMN IF NOT EXISTS ToolCallId String DEFAULT '' AFTER Output
|
||||
|
|
@ -0,0 +1,2 @@
|
|||
ALTER TABLE {database}.otel_traces
|
||||
ADD COLUMN IF NOT EXISTS AgentMetadata String DEFAULT '{}' CODEC(ZSTD(3))
|
||||
|
|
@ -0,0 +1 @@
|
|||
ALTER TABLE {database}.spend_logs MODIFY COLUMN spend Nullable(Float64) DEFAULT NULL
|
||||
|
|
@ -1,10 +1,21 @@
|
|||
SELECT request_id, response_id, team_id, api_key, user, spend,
|
||||
SELECT request_id, response_id, upstream_response_id, trace_id, span_id, team_id, api_key, user, spend,
|
||||
toUnixTimestamp64Milli(start_time) AS start_ms
|
||||
FROM spend_logs FINAL
|
||||
FROM (
|
||||
SELECT *,
|
||||
-- A chat request served through the Responses API returns the upstream `resp_` id to the
|
||||
-- client but logs LiteLLM's managed `resp_<base64>` id, which embeds it.
|
||||
if(startsWith(response_id, 'resp_'),
|
||||
extract(tryBase64Decode(substring(response_id, 6)), 'response_id:([^;]+)'),
|
||||
'') AS upstream_response_id
|
||||
FROM spend_logs FINAL
|
||||
WHERE start_time >= fromUnixTimestamp64Milli({start_ms:Int64})
|
||||
AND start_time < fromUnixTimestamp64Milli({end_ms:Int64})
|
||||
AND ({all_teams:UInt8} = 1
|
||||
OR ({user_id:String} != '' AND user = {user_id:String})
|
||||
OR has({team_ids:Array(String)}, team_id))
|
||||
)
|
||||
WHERE response_id IN {response_ids:Array(String)}
|
||||
AND start_time >= fromUnixTimestamp64Milli({start_ms:Int64})
|
||||
AND start_time < fromUnixTimestamp64Milli({end_ms:Int64})
|
||||
AND ({all_teams:UInt8} = 1
|
||||
OR ({user_id:String} != '' AND user = {user_id:String})
|
||||
OR has({team_ids:Array(String)}, team_id))
|
||||
OR upstream_response_id IN {response_ids:Array(String)}
|
||||
OR request_id IN {request_ids:Array(String)}
|
||||
OR (trace_id != '' AND trace_id IN {trace_ids:Array(String)})
|
||||
ORDER BY start_time DESC
|
||||
|
|
|
|||
|
|
@ -0,0 +1,24 @@
|
|||
SELECT o.TraceId AS trace_id, o.SpanId AS span_id, o.ParentSpanId AS parent_span_id, o.SpanName AS name,
|
||||
o.ObservationType AS type, toUInt8(o.WrapperCandidate) AS wrapper_candidate, o.AgentName AS agent,
|
||||
o.Framework AS framework, o.StatusCode AS status,
|
||||
substringUTF8(o.StatusMessage, 1, 128) AS status_message,
|
||||
lengthUTF8(o.StatusMessage) > 128 AS error_truncated,
|
||||
toUnixTimestamp64Nano(o.Timestamp) AS start_ns, o.Duration AS duration_ns,
|
||||
o.ServiceName AS service, o.InputPreview AS input_preview, o.Model AS model,
|
||||
o.InputTokens AS input_tokens, o.OutputTokens AS output_tokens,
|
||||
o.LiteLLMRequestId AS litellm_request_id,
|
||||
o.CallKeys AS call_keys, o.CallEvidence AS call_evidence,
|
||||
-- Rows written before ToolCallId keep the call id only in their attributes.
|
||||
if(o.ToolCallId != '' OR o.ObservationType != 'tool', o.ToolCallId,
|
||||
coalesce(nullIf(o.SpanAttributes['gen_ai.tool.call.id'], ''), nullIf(o.SpanAttributes['tool.id'], ''), ''))
|
||||
AS tool_call_id,
|
||||
o.UserId AS user_id, o.TeamId AS team_id, o.ApiKeyHash AS api_key_hash
|
||||
FROM otel_traces AS o
|
||||
WHERE o.Timestamp >= fromUnixTimestamp64Milli({start_ms:Int64})
|
||||
AND o.Timestamp < fromUnixTimestamp64Milli({end_ms:Int64})
|
||||
AND ({all_teams:UInt8} = 1
|
||||
OR ({user_id:String} != '' AND o.UserId = {user_id:String})
|
||||
OR has({team_ids:Array(String)}, o.TeamId))
|
||||
AND hex(SHA256(concat(o.TeamId, char(0), o.ApiKeyHash, char(0), o.TraceId))) IN {trace_refs:Array(String)}
|
||||
ORDER BY o.Timestamp, o.EngineReceivedMs, o.StatusMessage
|
||||
LIMIT 1 BY o.TeamId, o.ApiKeyHash, o.TraceId, o.SpanId
|
||||
|
|
@ -1,5 +1,5 @@
|
|||
SELECT o.SpanId AS span_id, o.ParentSpanId AS parent_span_id, o.SpanName AS name,
|
||||
o.ObservationType AS type, o.AgentName AS agent,
|
||||
SELECT o.TraceId AS trace_id, o.SpanId AS span_id, o.ParentSpanId AS parent_span_id, o.SpanName AS name,
|
||||
o.ObservationType AS type, toUInt8(o.WrapperCandidate) AS wrapper_candidate, o.AgentName AS agent,
|
||||
o.Framework AS framework, o.StatusCode AS status,
|
||||
substringUTF8(o.StatusMessage, 1, 128) AS status_message,
|
||||
lengthUTF8(o.StatusMessage) > 128 AS error_truncated,
|
||||
|
|
@ -7,6 +7,11 @@ SELECT o.SpanId AS span_id, o.ParentSpanId AS parent_span_id, o.SpanName AS name
|
|||
o.ServiceName AS service, o.InputPreview AS input_preview, o.Model AS model,
|
||||
o.InputTokens AS input_tokens, o.OutputTokens AS output_tokens,
|
||||
o.LiteLLMRequestId AS litellm_request_id,
|
||||
o.CallKeys AS call_keys, o.CallEvidence AS call_evidence,
|
||||
-- Rows written before ToolCallId keep the call id only in their attributes.
|
||||
if(o.ToolCallId != '' OR o.ObservationType != 'tool', o.ToolCallId,
|
||||
coalesce(nullIf(o.SpanAttributes['gen_ai.tool.call.id'], ''), nullIf(o.SpanAttributes['tool.id'], ''), ''))
|
||||
AS tool_call_id,
|
||||
o.UserId AS user_id, o.TeamId AS team_id, o.ApiKeyHash AS api_key_hash
|
||||
FROM otel_traces AS o
|
||||
WHERE o.TraceId = {trace_id:String}
|
||||
|
|
|
|||
|
|
@ -0,0 +1,6 @@
|
|||
fn main() {
|
||||
println!(
|
||||
"{}",
|
||||
serde_json::to_string_pretty(&litellm_traces_clickhouse::wire_schema::schemas()).unwrap()
|
||||
);
|
||||
}
|
||||
|
|
@ -5,14 +5,21 @@ use litellm_storage_clickhouse::Storage;
|
|||
pub struct Config {
|
||||
storage: Storage,
|
||||
retention_days: u32,
|
||||
max_attribute_value_bytes: usize,
|
||||
}
|
||||
|
||||
impl Config {
|
||||
pub fn new(database: String, url: &str, retention_days: u32) -> Result<Self, Error> {
|
||||
pub fn new(
|
||||
database: String,
|
||||
url: &str,
|
||||
retention_days: u32,
|
||||
max_attribute_value_bytes: usize,
|
||||
) -> Result<Self, Error> {
|
||||
super::schema_statements(&database, retention_days)?;
|
||||
Ok(Self {
|
||||
storage: Storage::new(database, url)?,
|
||||
retention_days,
|
||||
max_attribute_value_bytes,
|
||||
})
|
||||
}
|
||||
|
||||
|
|
@ -23,4 +30,9 @@ impl Config {
|
|||
pub fn retention_days(&self) -> u32 {
|
||||
self.retention_days
|
||||
}
|
||||
|
||||
/// Stored span attribute and payload values longer than this are truncated with a marker.
|
||||
pub fn max_attribute_value_bytes(&self) -> usize {
|
||||
self.max_attribute_value_bytes
|
||||
}
|
||||
}
|
||||
|
|
|
|||
|
|
@ -30,6 +30,14 @@ pub enum Error {
|
|||
ProvisionFailed(u16),
|
||||
#[error("ClickHouse reader provisioning transport failed")]
|
||||
ProvisionTransport,
|
||||
#[error("Invalid {0} cursor")]
|
||||
InvalidCursor(&'static str),
|
||||
#[error("Multiple traces have this ID; provide trace_ref")]
|
||||
AmbiguousTrace,
|
||||
#[error(transparent)]
|
||||
Decode(#[from] litellm_traces::Error),
|
||||
#[error("trace ingestion task failed")]
|
||||
Task,
|
||||
#[error(transparent)]
|
||||
Storage(#[from] litellm_storage_clickhouse::Error),
|
||||
#[error(transparent)]
|
||||
|
|
|
|||
|
|
@ -1,11 +1,27 @@
|
|||
macro_rules_attribute::attribute_alias! {
|
||||
#[apply(wire_type)] =
|
||||
#[derive(serde::Serialize, serde::Deserialize)]
|
||||
#[cfg_attr(feature = "schema", derive(schemars::JsonSchema))];
|
||||
#[apply(response_type)] =
|
||||
#[derive(serde::Serialize)]
|
||||
#[cfg_attr(feature = "schema", derive(schemars::JsonSchema))];
|
||||
#[apply(request_type)] =
|
||||
#[derive(serde::Deserialize)]
|
||||
#[cfg_attr(feature = "schema", derive(schemars::JsonSchema))];
|
||||
}
|
||||
|
||||
mod config;
|
||||
mod error;
|
||||
mod insert;
|
||||
pub mod query;
|
||||
mod query_access;
|
||||
mod reads;
|
||||
mod schema;
|
||||
mod span_row;
|
||||
mod sql;
|
||||
mod table;
|
||||
#[cfg(feature = "schema")]
|
||||
pub mod wire_schema;
|
||||
|
||||
pub use config::Config;
|
||||
pub use error::Error;
|
||||
|
|
@ -14,8 +30,10 @@ pub use litellm_storage_clickhouse::{Connection, Parameter};
|
|||
pub use litellm_traces::{QueryScope, ReadQuery};
|
||||
pub use query::{QueryHelp, execute_read, query_help, query_sql};
|
||||
pub use query_access::QueryReaders;
|
||||
pub use reads::{get_span, get_span_error, get_trace, list_traces};
|
||||
pub use schema::{
|
||||
NORMALIZED_FIELD_DEFINITIONS, NormalizedFieldDefinition, ensure_schema, schema_statements,
|
||||
};
|
||||
pub use span_row::span_rows;
|
||||
pub use sql::execute_named_read;
|
||||
pub use table::TraceTable;
|
||||
|
|
|
|||
|
|
@ -6,6 +6,7 @@ use futures_util::{
|
|||
stream::{self, TryStreamExt},
|
||||
};
|
||||
use litellm_http::Client;
|
||||
use litellm_traces::query::guide::{Example, QueryGuide, Section};
|
||||
use serde::{Deserialize, Serialize, Serializer};
|
||||
use serde_json::Value;
|
||||
use strum::IntoEnumIterator;
|
||||
|
|
@ -39,21 +40,24 @@ struct MetadataRow {
|
|||
metadata: String,
|
||||
}
|
||||
|
||||
#[derive(Deserialize)]
|
||||
#[macro_rules_attribute::apply(request_type)]
|
||||
struct AttributeRow {
|
||||
key: String,
|
||||
}
|
||||
|
||||
#[derive(Clone, Debug, Eq, Ord, PartialEq, PartialOrd, Serialize)]
|
||||
#[macro_rules_attribute::apply(response_type)]
|
||||
#[derive(Clone, Debug, Eq, Ord, PartialEq, PartialOrd)]
|
||||
#[serde(untagged)]
|
||||
enum PathPart {
|
||||
Key(String),
|
||||
Index(usize),
|
||||
}
|
||||
|
||||
#[derive(Clone, Copy, Debug, Eq, Ord, PartialEq, PartialOrd, Serialize, strum::Display)]
|
||||
#[macro_rules_attribute::apply(response_type)]
|
||||
#[derive(Clone, Copy, Debug, Eq, Ord, PartialEq, PartialOrd, strum::Display)]
|
||||
#[serde(rename_all = "lowercase")]
|
||||
#[strum(serialize_all = "lowercase")]
|
||||
#[cfg_attr(feature = "schema", schemars(rename = "MetadataValueType"))]
|
||||
enum JsonKind {
|
||||
Array,
|
||||
Boolean,
|
||||
|
|
@ -78,19 +82,23 @@ impl JsonKind {
|
|||
}
|
||||
}
|
||||
|
||||
#[derive(Clone, Copy, Debug, Serialize, strum::Display)]
|
||||
#[macro_rules_attribute::apply(response_type)]
|
||||
#[derive(Clone, Copy, Debug, strum::Display)]
|
||||
enum MapValueType {
|
||||
String,
|
||||
}
|
||||
|
||||
#[derive(Serialize)]
|
||||
#[macro_rules_attribute::apply(response_type)]
|
||||
#[cfg_attr(feature = "schema", schemars(deny_unknown_fields))]
|
||||
#[cfg_attr(feature = "schema", schemars(rename = "TraceQueryMetadataField"))]
|
||||
struct MetadataField {
|
||||
path: Vec<PathPart>,
|
||||
types: BTreeSet<JsonKind>,
|
||||
expression: String,
|
||||
}
|
||||
|
||||
#[derive(Deserialize, Serialize)]
|
||||
#[macro_rules_attribute::apply(wire_type)]
|
||||
#[cfg_attr(feature = "schema", schemars(rename = "TraceQueryColumn"))]
|
||||
struct ColumnSchema {
|
||||
name: String,
|
||||
#[serde(rename = "type")]
|
||||
|
|
@ -99,7 +107,9 @@ struct ColumnSchema {
|
|||
details: BTreeMap<String, Value>,
|
||||
}
|
||||
|
||||
#[derive(Serialize)]
|
||||
#[macro_rules_attribute::apply(response_type)]
|
||||
#[cfg_attr(feature = "schema", schemars(deny_unknown_fields))]
|
||||
#[cfg_attr(feature = "schema", schemars(rename = "TraceQueryTable"))]
|
||||
struct TableSchema {
|
||||
name: TraceTable,
|
||||
columns: Vec<ColumnSchema>,
|
||||
|
|
@ -114,6 +124,38 @@ enum Discovery<T> {
|
|||
Unavailable(String),
|
||||
}
|
||||
|
||||
#[cfg(feature = "schema")]
|
||||
impl<T: schemars::JsonSchema> schemars::JsonSchema for Discovery<T> {
|
||||
fn schema_name() -> std::borrow::Cow<'static, str> {
|
||||
format!("Discovery{}", T::schema_name()).into()
|
||||
}
|
||||
|
||||
fn json_schema(generator: &mut schemars::SchemaGenerator) -> schemars::Schema {
|
||||
let mut schema = T::json_schema(generator);
|
||||
schema
|
||||
.as_object_mut()
|
||||
.unwrap()
|
||||
.get_mut("properties")
|
||||
.unwrap()
|
||||
.as_object_mut()
|
||||
.unwrap()
|
||||
.insert(
|
||||
"error".into(),
|
||||
serde_json::json!({"type": ["string", "null"], "default": null}),
|
||||
);
|
||||
schema
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(feature = "schema")]
|
||||
pub(crate) fn help_schema() -> schemars::Schema {
|
||||
schemars::generate::SchemaSettings::draft2020_12()
|
||||
.for_serialize()
|
||||
.with_transform(litellm_traces::schema::integer_bounds)
|
||||
.into_generator()
|
||||
.into_root_schema_for::<QueryHelp>()
|
||||
}
|
||||
|
||||
impl<T: Serialize + Unobserved> Serialize for Discovery<T> {
|
||||
fn serialize<S: Serializer>(&self, serializer: S) -> Result<S::Ok, S::Error> {
|
||||
#[derive(Serialize)]
|
||||
|
|
@ -133,7 +175,7 @@ impl<T: Serialize + Unobserved> Serialize for Discovery<T> {
|
|||
}
|
||||
}
|
||||
|
||||
#[derive(Serialize)]
|
||||
#[macro_rules_attribute::apply(response_type)]
|
||||
struct MetadataSample {
|
||||
fields: Vec<MetadataField>,
|
||||
sampled_rows: usize,
|
||||
|
|
@ -152,7 +194,9 @@ impl Unobserved for MetadataSample {
|
|||
}
|
||||
}
|
||||
|
||||
#[derive(Serialize)]
|
||||
#[macro_rules_attribute::apply(response_type)]
|
||||
#[cfg_attr(feature = "schema", schemars(deny_unknown_fields))]
|
||||
#[cfg_attr(feature = "schema", schemars(rename = "TraceQueryMetadata"))]
|
||||
struct MetadataCatalog {
|
||||
table: TraceTable,
|
||||
column: &'static str,
|
||||
|
|
@ -162,7 +206,9 @@ struct MetadataCatalog {
|
|||
scope: &'static str,
|
||||
}
|
||||
|
||||
#[derive(Serialize)]
|
||||
#[macro_rules_attribute::apply(response_type)]
|
||||
#[cfg_attr(feature = "schema", schemars(deny_unknown_fields))]
|
||||
#[cfg_attr(feature = "schema", schemars(rename = "TraceQueryAttributeField"))]
|
||||
struct AttributeField {
|
||||
key: String,
|
||||
#[serde(rename = "type")]
|
||||
|
|
@ -170,7 +216,7 @@ struct AttributeField {
|
|||
expression: String,
|
||||
}
|
||||
|
||||
#[derive(Serialize)]
|
||||
#[macro_rules_attribute::apply(response_type)]
|
||||
struct AttributeSample {
|
||||
fields: Vec<AttributeField>,
|
||||
truncated: bool,
|
||||
|
|
@ -185,7 +231,9 @@ impl Unobserved for AttributeSample {
|
|||
}
|
||||
}
|
||||
|
||||
#[derive(Serialize)]
|
||||
#[macro_rules_attribute::apply(response_type)]
|
||||
#[cfg_attr(feature = "schema", schemars(deny_unknown_fields))]
|
||||
#[cfg_attr(feature = "schema", schemars(rename = "TraceQueryAttributes"))]
|
||||
struct AttributeCatalog {
|
||||
table: TraceTable,
|
||||
column: &'static str,
|
||||
|
|
@ -195,7 +243,9 @@ struct AttributeCatalog {
|
|||
scope: &'static str,
|
||||
}
|
||||
|
||||
#[derive(Serialize)]
|
||||
#[macro_rules_attribute::apply(response_type)]
|
||||
#[cfg_attr(feature = "schema", schemars(deny_unknown_fields))]
|
||||
#[cfg_attr(feature = "schema", schemars(rename = "TraceQueryNormalizedField"))]
|
||||
struct NormalizedField {
|
||||
table: TraceTable,
|
||||
name: &'static str,
|
||||
|
|
@ -217,7 +267,9 @@ impl From<&NormalizedFieldDefinition> for NormalizedField {
|
|||
}
|
||||
}
|
||||
|
||||
#[derive(Serialize)]
|
||||
#[macro_rules_attribute::apply(response_type)]
|
||||
#[cfg_attr(feature = "schema", schemars(deny_unknown_fields))]
|
||||
#[cfg_attr(feature = "schema", schemars(rename = "TraceQueryRelationship"))]
|
||||
struct Relationship {
|
||||
left: &'static str,
|
||||
right: &'static str,
|
||||
|
|
@ -228,11 +280,13 @@ struct Relationship {
|
|||
const RELATIONSHIPS: [Relationship; 1] = [Relationship {
|
||||
left: "otel_traces.LiteLLMRequestId",
|
||||
right: "spend_logs.response_id",
|
||||
additional_predicates: "otel_traces.TeamId = spend_logs.team_id AND (otel_traces.TeamId != '' OR (otel_traces.UserId != '' AND otel_traces.UserId = spend_logs.user) OR (otel_traces.ApiKeyHash != '' AND otel_traces.ApiKeyHash = spend_logs.api_key))",
|
||||
meaning: "The normalized ID is the response ID, not request_id. Cached requests can share response_id; joins may return multiple spend rows",
|
||||
additional_predicates: "otel_traces.TeamId = spend_logs.team_id AND ((otel_traces.UserId != '' AND otel_traces.UserId = spend_logs.user) OR (otel_traces.ApiKeyHash != '' AND otel_traces.ApiKeyHash = spend_logs.api_key))",
|
||||
meaning: "LiteLLMRequestId contains the first normalized request or provider response ID. This relationship matches response IDs only; CallKeys retains all typed identifiers. Cached requests can share response_id; joins may return multiple spend rows",
|
||||
}];
|
||||
|
||||
#[derive(Serialize)]
|
||||
#[macro_rules_attribute::apply(response_type)]
|
||||
#[cfg_attr(feature = "schema", schemars(deny_unknown_fields))]
|
||||
#[cfg_attr(feature = "schema", schemars(rename = "TraceQueryHelp"))]
|
||||
pub struct QueryHelp {
|
||||
dialect: &'static str,
|
||||
access: &'static str,
|
||||
|
|
@ -242,8 +296,10 @@ pub struct QueryHelp {
|
|||
metadata: MetadataCatalog,
|
||||
attributes: Vec<AttributeCatalog>,
|
||||
relationships: &'static [Relationship],
|
||||
examples: [guide::Example; 5],
|
||||
gotchas: [String; 11],
|
||||
#[cfg_attr(feature = "schema", schemars(with = "Vec<Example>"))]
|
||||
examples: [Example; 9],
|
||||
#[cfg_attr(feature = "schema", schemars(with = "Vec<String>"))]
|
||||
gotchas: [String; 13],
|
||||
guide: String,
|
||||
}
|
||||
|
||||
|
|
@ -423,13 +479,33 @@ pub async fn query_help(client: &Client, connection: &Connection) -> Result<Quer
|
|||
attributes: &attributes,
|
||||
limits: &READER_LIMITS,
|
||||
};
|
||||
let bodies = guide.sections()?;
|
||||
let sections = [
|
||||
"Live ClickHouse schema",
|
||||
"Normalized span fields",
|
||||
"Observed LLM call metadata",
|
||||
"Observed span and resource attributes",
|
||||
]
|
||||
.into_iter()
|
||||
.zip(&bodies)
|
||||
.map(|(title, body)| Section { title, body })
|
||||
.collect::<Vec<_>>();
|
||||
let examples = guide.examples()?;
|
||||
let gotchas = guide.gotchas()?;
|
||||
let rendered = QueryGuide {
|
||||
sections: §ions,
|
||||
examples: &examples,
|
||||
gotchas: &gotchas,
|
||||
}
|
||||
.render()
|
||||
.map_err(|_| Error::InvalidResponse)?;
|
||||
Ok(QueryHelp {
|
||||
dialect: "ClickHouse SQL",
|
||||
access: "Request-log visibility enforced by ClickHouse row policies; proxy admins see all rows, users see their own rows and permitted teams",
|
||||
response: "ClickHouse JSON envelope: meta, data, rows, statistics; 64-bit integers may be strings",
|
||||
examples: guide.examples()?,
|
||||
gotchas: guide.gotchas()?,
|
||||
guide: guide::render(&guide)?,
|
||||
examples,
|
||||
gotchas,
|
||||
guide: rendered,
|
||||
normalized_fields: NORMALIZED_FIELD_DEFINITIONS
|
||||
.iter()
|
||||
.map(NormalizedField::from)
|
||||
|
|
@ -447,6 +523,33 @@ mod tests {
|
|||
use rstest::rstest;
|
||||
use serde_json::json;
|
||||
|
||||
#[cfg(feature = "schema")]
|
||||
#[rstest]
|
||||
#[case::observed(false)]
|
||||
#[case::unavailable(true)]
|
||||
fn discovery_serialization_matches_its_schema(#[case] unavailable: bool) {
|
||||
let discovery = if unavailable {
|
||||
Discovery::Unavailable("discovery failed".into())
|
||||
} else {
|
||||
Discovery::Observed(MetadataSample::unobserved())
|
||||
};
|
||||
let catalog = MetadataCatalog {
|
||||
table: TraceTable::SpendLogs,
|
||||
column: "metadata",
|
||||
discovery,
|
||||
sample_sql: METADATA_SQL,
|
||||
scope: METADATA_SCOPE,
|
||||
};
|
||||
let schema = schemars::generate::SchemaSettings::draft2020_12()
|
||||
.for_serialize()
|
||||
.into_generator()
|
||||
.into_root_schema_for::<MetadataCatalog>();
|
||||
let serialized = serde_json::to_value(&catalog).unwrap();
|
||||
assert!(jsonschema::is_valid(schema.as_value(), &serialized));
|
||||
assert_eq!(serialized.get("error").is_some(), unavailable);
|
||||
assert!(serialized["fields"].is_array());
|
||||
}
|
||||
|
||||
#[rstest]
|
||||
fn metadata_discovery_preserves_mixed_types_and_reports_invalid_rows() {
|
||||
let sample = [
|
||||
|
|
|
|||
|
|
@ -1,11 +1,15 @@
|
|||
use askama::Template;
|
||||
use serde::Serialize;
|
||||
use litellm_traces::query::guide::Example;
|
||||
|
||||
use super::{AttributeCatalog, Discovery, MetadataCatalog, TableSchema};
|
||||
use crate::{Error, NormalizedFieldDefinition, query_access::ReaderLimits};
|
||||
|
||||
#[derive(Template)]
|
||||
#[template(path = "query_help.jinja", escape = "none", blocks = [
|
||||
"live_schema",
|
||||
"normalized_fields",
|
||||
"metadata",
|
||||
"attributes",
|
||||
"recent_spans_name",
|
||||
"recent_spans_sql",
|
||||
"custom_metadata_name",
|
||||
|
|
@ -16,6 +20,16 @@ use crate::{Error, NormalizedFieldDefinition, query_access::ReaderLimits};
|
|||
"correlated_calls_sql",
|
||||
"discover_keys_name",
|
||||
"discover_keys_sql",
|
||||
"recent_spend_name",
|
||||
"recent_spend_sql",
|
||||
"model_spend_name",
|
||||
"model_spend_sql",
|
||||
"trace_spend_name",
|
||||
"trace_spend_sql",
|
||||
"unmatched_spans_name",
|
||||
"unmatched_spans_sql",
|
||||
"missing_spend",
|
||||
"partial_spend",
|
||||
"time_window",
|
||||
"reader_limits",
|
||||
"reader_profile",
|
||||
|
|
@ -36,14 +50,17 @@ pub(super) struct QueryGuide<'a> {
|
|||
pub limits: &'a ReaderLimits,
|
||||
}
|
||||
|
||||
#[derive(Serialize)]
|
||||
pub(super) struct Example {
|
||||
name: String,
|
||||
sql: String,
|
||||
}
|
||||
|
||||
impl QueryGuide<'_> {
|
||||
pub fn examples(&self) -> Result<[Example; 5], Error> {
|
||||
pub fn sections(&self) -> Result<[String; 4], Error> {
|
||||
Ok([
|
||||
render(&self.as_live_schema())?,
|
||||
render(&self.as_normalized_fields())?,
|
||||
render(&self.as_metadata())?,
|
||||
render(&self.as_attributes())?,
|
||||
])
|
||||
}
|
||||
|
||||
pub fn examples(&self) -> Result<[Example; 9], Error> {
|
||||
Ok([
|
||||
Example {
|
||||
name: render(&self.as_recent_spans_name())?,
|
||||
|
|
@ -65,10 +82,26 @@ impl QueryGuide<'_> {
|
|||
name: render(&self.as_discover_keys_name())?,
|
||||
sql: render(&self.as_discover_keys_sql())?,
|
||||
},
|
||||
Example {
|
||||
name: render(&self.as_recent_spend_name())?,
|
||||
sql: render(&self.as_recent_spend_sql())?,
|
||||
},
|
||||
Example {
|
||||
name: render(&self.as_model_spend_name())?,
|
||||
sql: render(&self.as_model_spend_sql())?,
|
||||
},
|
||||
Example {
|
||||
name: render(&self.as_trace_spend_name())?,
|
||||
sql: render(&self.as_trace_spend_sql())?,
|
||||
},
|
||||
Example {
|
||||
name: render(&self.as_unmatched_spans_name())?,
|
||||
sql: render(&self.as_unmatched_spans_sql())?,
|
||||
},
|
||||
])
|
||||
}
|
||||
|
||||
pub fn gotchas(&self) -> Result<[String; 11], Error> {
|
||||
pub fn gotchas(&self) -> Result<[String; 13], Error> {
|
||||
Ok([
|
||||
render(&self.as_time_window())?,
|
||||
render(&self.as_reader_limits())?,
|
||||
|
|
@ -79,6 +112,8 @@ impl QueryGuide<'_> {
|
|||
render(&self.as_literal_keys())?,
|
||||
render(&self.as_time_units())?,
|
||||
render(&self.as_spend_totals())?,
|
||||
render(&self.as_missing_spend())?,
|
||||
render(&self.as_partial_spend())?,
|
||||
render(&self.as_trace_rollups())?,
|
||||
render(&self.as_sampling())?,
|
||||
])
|
||||
|
|
|
|||
|
|
@ -1,27 +1,72 @@
|
|||
use litellm_storage_clickhouse::Query;
|
||||
use serde::{Deserialize, Serialize};
|
||||
|
||||
#[derive(Debug, Deserialize, Serialize)]
|
||||
pub const LENS_QUERIES: [litellm_traces::ReadQuery; 5] = [
|
||||
litellm_traces::ReadQuery::Availability,
|
||||
litellm_traces::ReadQuery::Agents,
|
||||
litellm_traces::ReadQuery::Sample,
|
||||
litellm_traces::ReadQuery::Content,
|
||||
litellm_traces::ReadQuery::Evidence,
|
||||
];
|
||||
|
||||
#[macro_rules_attribute::apply(wire_type)]
|
||||
#[derive(Debug)]
|
||||
#[serde(rename_all = "lowercase")]
|
||||
pub enum ExecutionSource {
|
||||
Traces,
|
||||
Requests,
|
||||
Both,
|
||||
}
|
||||
|
||||
#[macro_rules_attribute::apply(wire_type)]
|
||||
#[derive(Debug)]
|
||||
#[serde(rename_all = "lowercase")]
|
||||
pub enum ContentSource {
|
||||
Traces,
|
||||
Requests,
|
||||
}
|
||||
|
||||
#[macro_rules_attribute::apply(wire_type)]
|
||||
#[derive(Debug)]
|
||||
#[cfg_attr(feature = "schema", schemars(deny_unknown_fields))]
|
||||
pub struct LensAccessParams {
|
||||
#[serde(deserialize_with = "super::number::deserialize")]
|
||||
pub all_teams: u8,
|
||||
#[serde(
|
||||
deserialize_with = "super::number::boolean",
|
||||
serialize_with = "litellm_traces::wire::serialize_flag"
|
||||
)]
|
||||
#[cfg_attr(
|
||||
feature = "schema",
|
||||
schemars(schema_with = "litellm_traces::schema::flag")
|
||||
)]
|
||||
pub all_teams: bool,
|
||||
pub team: String,
|
||||
pub key_hash: String,
|
||||
}
|
||||
|
||||
pub struct LensAvailability;
|
||||
|
||||
#[derive(Debug, Deserialize, Serialize)]
|
||||
#[macro_rules_attribute::apply(wire_type)]
|
||||
#[derive(Debug)]
|
||||
#[serde(deny_unknown_fields)]
|
||||
pub struct LensAvailabilityParams {
|
||||
#[serde(flatten)]
|
||||
pub access: LensAccessParams,
|
||||
}
|
||||
|
||||
#[derive(Debug, Deserialize, Serialize)]
|
||||
#[macro_rules_attribute::apply(wire_type)]
|
||||
#[derive(Debug)]
|
||||
#[cfg_attr(feature = "schema", schemars(rename = "ActivityAvailability"))]
|
||||
pub struct LensAvailabilityRow {
|
||||
#[serde(deserialize_with = "super::number::deserialize")]
|
||||
#[serde(default, deserialize_with = "super::number::flag")]
|
||||
#[cfg_attr(
|
||||
feature = "schema",
|
||||
schemars(schema_with = "crate::wire_schema::boolean_flag")
|
||||
)]
|
||||
pub traces: u8,
|
||||
#[serde(deserialize_with = "super::number::deserialize")]
|
||||
#[serde(default, deserialize_with = "super::number::flag")]
|
||||
#[cfg_attr(
|
||||
feature = "schema",
|
||||
schemars(schema_with = "crate::wire_schema::boolean_flag")
|
||||
)]
|
||||
pub requests: u8,
|
||||
}
|
||||
|
||||
|
|
@ -34,13 +79,17 @@ impl Query for LensAvailability {
|
|||
|
||||
pub struct LensAgents;
|
||||
|
||||
#[derive(Debug, Deserialize, Serialize)]
|
||||
#[macro_rules_attribute::apply(wire_type)]
|
||||
#[derive(Debug)]
|
||||
#[serde(deny_unknown_fields)]
|
||||
pub struct LensAgentsParams {
|
||||
#[serde(flatten)]
|
||||
pub access: LensAccessParams,
|
||||
}
|
||||
|
||||
#[derive(Debug, Deserialize, Serialize)]
|
||||
#[macro_rules_attribute::apply(wire_type)]
|
||||
#[derive(Debug)]
|
||||
#[cfg_attr(feature = "schema", schemars(rename = "AgentRow"))]
|
||||
pub struct LensAgentsRow {
|
||||
pub agent_name: String,
|
||||
}
|
||||
|
|
@ -54,11 +103,13 @@ impl Query for LensAgents {
|
|||
|
||||
pub struct LensSample;
|
||||
|
||||
#[derive(Debug, Deserialize, Serialize)]
|
||||
#[macro_rules_attribute::apply(wire_type)]
|
||||
#[derive(Debug)]
|
||||
#[serde(deny_unknown_fields)]
|
||||
pub struct LensSampleParams {
|
||||
#[serde(flatten)]
|
||||
pub access: LensAccessParams,
|
||||
pub source: String,
|
||||
pub source: ExecutionSource,
|
||||
#[serde(deserialize_with = "super::number::deserialize")]
|
||||
pub start: u64,
|
||||
#[serde(deserialize_with = "super::number::deserialize")]
|
||||
|
|
@ -71,9 +122,14 @@ pub struct LensSampleParams {
|
|||
pub execution_ids: Vec<String>,
|
||||
#[serde(deserialize_with = "super::number::deserialize")]
|
||||
pub sample_cap: u64,
|
||||
#[serde(deserialize_with = "super::number::deserialize")]
|
||||
#[serde(deserialize_with = "super::number::percent")]
|
||||
#[cfg_attr(feature = "schema", schemars(range(min = 0, max = 100)))]
|
||||
pub sample_percent: f64,
|
||||
#[serde(deserialize_with = "super::number::deserialize")]
|
||||
#[serde(deserialize_with = "super::number::flag")]
|
||||
#[cfg_attr(
|
||||
feature = "schema",
|
||||
schemars(schema_with = "litellm_traces::schema::flag")
|
||||
)]
|
||||
pub preview: u8,
|
||||
pub after: String,
|
||||
#[serde(deserialize_with = "super::number::deserialize")]
|
||||
|
|
@ -82,26 +138,49 @@ pub struct LensSampleParams {
|
|||
pub offset: u64,
|
||||
}
|
||||
|
||||
#[derive(Debug, Deserialize, Serialize)]
|
||||
#[macro_rules_attribute::apply(wire_type)]
|
||||
#[derive(Debug)]
|
||||
#[cfg_attr(feature = "schema", schemars(rename = "ExecutionRow"))]
|
||||
pub struct LensSampleRow {
|
||||
pub source: String,
|
||||
pub source: ContentSource,
|
||||
pub trace_id: String,
|
||||
pub team_id: String,
|
||||
#[serde(default)]
|
||||
pub trace_ref: String,
|
||||
pub name: String,
|
||||
pub start_time: String,
|
||||
#[serde(deserialize_with = "super::number::deserialize")]
|
||||
#[cfg_attr(
|
||||
feature = "schema",
|
||||
schemars(schema_with = "crate::wire_schema::u64_number")
|
||||
)]
|
||||
pub span_count: u64,
|
||||
#[serde(deserialize_with = "super::number::deserialize")]
|
||||
#[serde(deserialize_with = "super::number::flag")]
|
||||
#[cfg_attr(
|
||||
feature = "schema",
|
||||
schemars(schema_with = "crate::wire_schema::flag_number")
|
||||
)]
|
||||
pub root_seen: u8,
|
||||
#[serde(default)]
|
||||
pub service: String,
|
||||
#[serde(default)]
|
||||
pub attributes: Vec<(String, String)>,
|
||||
#[serde(deserialize_with = "super::number::deserialize")]
|
||||
#[cfg_attr(
|
||||
feature = "schema",
|
||||
schemars(schema_with = "crate::wire_schema::u64_number")
|
||||
)]
|
||||
pub eligible: u64,
|
||||
#[serde(deserialize_with = "super::number::deserialize")]
|
||||
#[cfg_attr(feature = "schema", schemars(skip))]
|
||||
pub position: u64,
|
||||
#[serde(deserialize_with = "super::number::deserialize")]
|
||||
#[serde(default, deserialize_with = "super::number::deserialize")]
|
||||
#[cfg_attr(
|
||||
feature = "schema",
|
||||
schemars(schema_with = "crate::wire_schema::selected")
|
||||
)]
|
||||
pub selected: f64,
|
||||
#[serde(default)]
|
||||
pub selection_key: String,
|
||||
}
|
||||
|
||||
|
|
@ -114,11 +193,13 @@ impl Query for LensSample {
|
|||
|
||||
pub struct LensContent;
|
||||
|
||||
#[derive(Debug, Deserialize, Serialize)]
|
||||
#[macro_rules_attribute::apply(wire_type)]
|
||||
#[derive(Debug)]
|
||||
#[serde(deny_unknown_fields)]
|
||||
pub struct LensContentParams {
|
||||
#[serde(flatten)]
|
||||
pub access: LensAccessParams,
|
||||
pub source: String,
|
||||
pub source: ContentSource,
|
||||
pub id: String,
|
||||
pub record_team: String,
|
||||
pub trace_ref: String,
|
||||
|
|
@ -127,14 +208,20 @@ pub struct LensContentParams {
|
|||
pub offset: u32,
|
||||
}
|
||||
|
||||
#[derive(Debug, Deserialize, Serialize)]
|
||||
#[macro_rules_attribute::apply(wire_type)]
|
||||
#[derive(Debug)]
|
||||
#[cfg_attr(feature = "schema", schemars(rename = "PartRow"))]
|
||||
pub struct LensContentRow {
|
||||
pub span_id: String,
|
||||
pub parent_span_id: String,
|
||||
pub name: String,
|
||||
pub kind: String,
|
||||
pub content: String,
|
||||
#[serde(deserialize_with = "super::number::deserialize")]
|
||||
#[serde(deserialize_with = "super::number::flag")]
|
||||
#[cfg_attr(
|
||||
feature = "schema",
|
||||
schemars(schema_with = "crate::wire_schema::flag_number")
|
||||
)]
|
||||
pub truncated: u8,
|
||||
}
|
||||
|
||||
|
|
@ -147,11 +234,13 @@ impl Query for LensContent {
|
|||
|
||||
pub struct LensEvidence;
|
||||
|
||||
#[derive(Debug, Deserialize, Serialize)]
|
||||
#[macro_rules_attribute::apply(wire_type)]
|
||||
#[derive(Debug)]
|
||||
#[serde(deny_unknown_fields)]
|
||||
pub struct LensEvidenceParams {
|
||||
#[serde(flatten)]
|
||||
pub access: LensAccessParams,
|
||||
pub source: String,
|
||||
pub source: ContentSource,
|
||||
pub id: String,
|
||||
pub record_team: String,
|
||||
pub trace_ref: String,
|
||||
|
|
@ -159,9 +248,15 @@ pub struct LensEvidenceParams {
|
|||
pub quote: String,
|
||||
}
|
||||
|
||||
#[derive(Debug, Deserialize, Serialize)]
|
||||
#[macro_rules_attribute::apply(wire_type)]
|
||||
#[derive(Debug)]
|
||||
#[cfg_attr(feature = "schema", schemars(rename = "CountRow"))]
|
||||
pub struct LensEvidenceRow {
|
||||
#[serde(deserialize_with = "super::number::deserialize")]
|
||||
#[cfg_attr(
|
||||
feature = "schema",
|
||||
schemars(schema_with = "crate::wire_schema::u64_number")
|
||||
)]
|
||||
pub count: u64,
|
||||
}
|
||||
|
||||
|
|
|
|||
|
|
@ -42,7 +42,8 @@ struct ListTracesRowEncoding {
|
|||
pub name: String,
|
||||
pub service: String,
|
||||
pub input_preview: String,
|
||||
pub status: String,
|
||||
#[serde(serialize_with = "litellm_traces::wire::serialize_status")]
|
||||
pub status: litellm_traces::SpanStatus,
|
||||
#[serde(deserialize_with = "super::number::deserialize")]
|
||||
pub start_ms: i64,
|
||||
#[serde(deserialize_with = "super::number::deserialize")]
|
||||
|
|
@ -79,18 +80,30 @@ pub use contracts::TraceSpansParams;
|
|||
#[derive(Deserialize, Serialize)]
|
||||
#[serde(remote = "contracts::TraceSpansRow")]
|
||||
struct TraceSpansRowEncoding {
|
||||
#[serde(default)]
|
||||
pub trace_id: String,
|
||||
pub span_id: String,
|
||||
pub parent_span_id: String,
|
||||
pub name: String,
|
||||
#[serde(rename = "type")]
|
||||
pub kind: String,
|
||||
pub kind: litellm_traces::ObservationType,
|
||||
#[serde(
|
||||
default,
|
||||
deserialize_with = "super::number::boolean",
|
||||
serialize_with = "litellm_traces::wire::serialize_flag"
|
||||
)]
|
||||
pub wrapper_candidate: bool,
|
||||
pub agent: String,
|
||||
#[serde(default)]
|
||||
pub framework: String,
|
||||
pub status: String,
|
||||
#[serde(serialize_with = "litellm_traces::wire::serialize_status")]
|
||||
pub status: litellm_traces::SpanStatus,
|
||||
pub status_message: String,
|
||||
#[serde(deserialize_with = "super::number::deserialize")]
|
||||
pub error_truncated: u8,
|
||||
#[serde(
|
||||
deserialize_with = "super::number::boolean",
|
||||
serialize_with = "litellm_traces::wire::serialize_flag"
|
||||
)]
|
||||
pub error_truncated: bool,
|
||||
#[serde(deserialize_with = "super::number::deserialize")]
|
||||
pub start_ns: i64,
|
||||
#[serde(deserialize_with = "super::number::deserialize")]
|
||||
|
|
@ -103,6 +116,16 @@ struct TraceSpansRowEncoding {
|
|||
#[serde(deserialize_with = "super::number::deserialize")]
|
||||
pub output_tokens: u32,
|
||||
pub litellm_request_id: String,
|
||||
#[serde(default)]
|
||||
pub call_keys: Vec<litellm_traces::CallKey>,
|
||||
#[serde(
|
||||
default,
|
||||
deserialize_with = "litellm_traces::wire::evidence",
|
||||
serialize_with = "litellm_traces::wire::serialize_evidence"
|
||||
)]
|
||||
pub call_evidence: Option<litellm_traces::CallEvidenceKind>,
|
||||
#[serde(default)]
|
||||
pub tool_call_id: String,
|
||||
pub team_id: String,
|
||||
pub api_key_hash: String,
|
||||
pub user_id: String,
|
||||
|
|
@ -158,6 +181,8 @@ struct SpendByResponseIdsParamsEncoding {
|
|||
#[serde(flatten)]
|
||||
pub access: contracts::ReadAccessParams,
|
||||
pub response_ids: Vec<String>,
|
||||
pub request_ids: Vec<String>,
|
||||
pub trace_ids: Vec<String>,
|
||||
#[serde(deserialize_with = "super::number::deserialize")]
|
||||
pub start_ms: i64,
|
||||
#[serde(deserialize_with = "super::number::deserialize")]
|
||||
|
|
@ -180,11 +205,14 @@ impl From<contracts::SpendByResponseIdsParams> for SpendByResponseIdsParams {
|
|||
struct SpendByResponseIdsRowEncoding {
|
||||
pub request_id: String,
|
||||
pub response_id: String,
|
||||
pub upstream_response_id: String,
|
||||
pub trace_id: String,
|
||||
pub span_id: String,
|
||||
pub team_id: String,
|
||||
pub api_key: String,
|
||||
pub user: String,
|
||||
#[serde(deserialize_with = "super::number::deserialize")]
|
||||
pub spend: f64,
|
||||
#[serde(deserialize_with = "super::number::optional_finite")]
|
||||
pub spend: Option<f64>,
|
||||
#[serde(deserialize_with = "super::number::deserialize")]
|
||||
pub start_ms: i64,
|
||||
}
|
||||
|
|
@ -203,6 +231,38 @@ impl Query for ListTraces {
|
|||
const SQL: &'static str = include_str!("../../query/list_traces.sql");
|
||||
}
|
||||
|
||||
#[derive(Deserialize, Serialize)]
|
||||
#[serde(remote = "contracts::TracePageSpansParams")]
|
||||
struct TracePageSpansParamsEncoding {
|
||||
#[serde(flatten)]
|
||||
pub access: contracts::ReadAccessParams,
|
||||
pub trace_refs: Vec<String>,
|
||||
#[serde(deserialize_with = "super::number::deserialize")]
|
||||
pub start_ms: i64,
|
||||
#[serde(deserialize_with = "super::number::deserialize")]
|
||||
pub end_ms: i64,
|
||||
}
|
||||
|
||||
#[derive(Debug, Deserialize, Serialize)]
|
||||
pub struct TracePageSpansParams(
|
||||
#[serde(with = "TracePageSpansParamsEncoding")] pub contracts::TracePageSpansParams,
|
||||
);
|
||||
|
||||
impl From<contracts::TracePageSpansParams> for TracePageSpansParams {
|
||||
fn from(value: contracts::TracePageSpansParams) -> Self {
|
||||
Self(value)
|
||||
}
|
||||
}
|
||||
|
||||
pub struct TracePageSpans;
|
||||
|
||||
impl Query for TracePageSpans {
|
||||
type Params = TracePageSpansParams;
|
||||
type Row = TraceSpansRow;
|
||||
|
||||
const SQL: &'static str = include_str!("../../query/trace_page_spans.sql");
|
||||
}
|
||||
|
||||
pub struct TraceSpans;
|
||||
|
||||
impl Query for TraceSpans {
|
||||
|
|
@ -279,11 +339,11 @@ mod tests {
|
|||
#[case::quoted(true)]
|
||||
fn rows_decode_into_neutral_contracts(#[case] quoted: bool) {
|
||||
round_trip::<ListTracesRow>(
|
||||
json!({"trace_id": "trace", "trace_ref": "ref", "team_id": "team", "api_key_hash": "key", "user_id": "user", "name": "agent", "service": "service", "input_preview": "input", "status": "ok", "start_ms": -1, "duration_ms": 20, "span_count": u64::MAX, "agent_count": 1, "agent_invocations": 2, "agent_names": ["agent"], "frameworks": ["claude-agent-sdk"], "llm_calls": 3, "tool_calls": 4, "input_tokens": 5, "output_tokens": 6, "models": ["model"], "error_count": 0, "request_ids": ["request"]}),
|
||||
json!({"trace_id": "trace", "trace_ref": "ref", "team_id": "team", "api_key_hash": "key", "user_id": "user", "name": "agent", "service": "service", "input_preview": "input", "status": "STATUS_CODE_OK", "start_ms": -1, "duration_ms": 20, "span_count": u64::MAX, "agent_count": 1, "agent_invocations": 2, "agent_names": ["agent"], "frameworks": ["claude-agent-sdk"], "llm_calls": 3, "tool_calls": 4, "input_tokens": 5, "output_tokens": 6, "models": ["model"], "error_count": 0, "request_ids": ["request"]}),
|
||||
quoted,
|
||||
);
|
||||
round_trip::<TraceSpansRow>(
|
||||
json!({"span_id": "span", "parent_span_id": "parent", "name": "agent", "type": "agent", "agent": "agent", "framework": "claude-agent-sdk", "status": "error", "status_message": "error", "error_truncated": 1, "start_ns": -1, "duration_ns": u64::MAX, "service": "service", "input_preview": "input", "model": "model", "input_tokens": u32::MAX, "output_tokens": 6, "litellm_request_id": "request", "team_id": "team", "api_key_hash": "key", "user_id": "user"}),
|
||||
json!({"trace_id": "trace", "span_id": "span", "parent_span_id": "parent", "name": "agent", "type": "agent", "wrapper_candidate": 1, "agent": "agent", "framework": "claude-agent-sdk", "status": "STATUS_CODE_ERROR", "status_message": "error", "error_truncated": 1, "start_ns": -1, "duration_ns": u64::MAX, "service": "service", "input_preview": "input", "model": "model", "input_tokens": u32::MAX, "output_tokens": 6, "litellm_request_id": "request", "call_keys": ["provider_response:request"], "call_evidence": "complete", "tool_call_id": "call", "team_id": "team", "api_key_hash": "key", "user_id": "user"}),
|
||||
quoted,
|
||||
);
|
||||
round_trip::<SpanDetailRow>(
|
||||
|
|
@ -295,7 +355,7 @@ mod tests {
|
|||
quoted,
|
||||
);
|
||||
round_trip::<SpendByResponseIdsRow>(
|
||||
json!({"request_id": "request", "response_id": "response", "team_id": "team", "api_key": "key", "user": "user", "spend": 0.125, "start_ms": -1}),
|
||||
json!({"request_id": "request", "response_id": "response", "upstream_response_id": "upstream", "trace_id": "trace", "span_id": "span", "team_id": "team", "api_key": "key", "user": "user", "spend": 0.125, "start_ms": -1}),
|
||||
quoted,
|
||||
);
|
||||
}
|
||||
|
|
@ -313,8 +373,36 @@ mod tests {
|
|||
quoted,
|
||||
);
|
||||
round_trip::<SpendByResponseIdsParams>(
|
||||
json!({"all_teams": 0, "user_id": "user", "team_ids": ["team-a", "team-b"], "response_ids": ["response"], "start_ms": -1, "end_ms": 10}),
|
||||
json!({"all_teams": 0, "user_id": "user", "team_ids": ["team-a", "team-b"], "response_ids": ["response"], "request_ids": ["request"], "trace_ids": ["trace"], "start_ms": -1, "end_ms": 10}),
|
||||
quoted,
|
||||
);
|
||||
}
|
||||
#[rstest]
|
||||
#[case::unknown(json!(null), None)]
|
||||
#[case::free(json!(0), Some(0.0))]
|
||||
#[case::paid(json!("0.125"), Some(0.125))]
|
||||
fn spend_rows_preserve_unknown_and_known_cost(
|
||||
#[case] cost: serde_json::Value,
|
||||
#[case] expected: Option<f64>,
|
||||
) {
|
||||
let row: SpendByResponseIdsRow = serde_json::from_value(json!({
|
||||
"request_id": "request", "response_id": "response", "upstream_response_id": "",
|
||||
"trace_id": "trace", "span_id": "span", "team_id": "team", "api_key": "key",
|
||||
"user": "user", "spend": cost, "start_ms": 0
|
||||
}))
|
||||
.unwrap();
|
||||
assert_eq!(row.0.spend, expected);
|
||||
}
|
||||
#[rstest]
|
||||
#[case::nan(json!("NaN"))]
|
||||
#[case::infinity(json!("1e999"))]
|
||||
#[case::boolean(json!(true))]
|
||||
fn spend_rows_reject_invalid_cost(#[case] cost: serde_json::Value) {
|
||||
let row = serde_json::from_value::<SpendByResponseIdsRow>(json!({
|
||||
"request_id": "request", "response_id": "response", "upstream_response_id": "",
|
||||
"trace_id": "trace", "span_id": "span", "team_id": "team", "api_key": "key",
|
||||
"user": "user", "spend": cost, "start_ms": 0
|
||||
}));
|
||||
assert!(row.is_err());
|
||||
}
|
||||
}
|
||||
|
|
|
|||
|
|
@ -18,11 +18,95 @@ where
|
|||
.map_err(serde::de::Error::custom)
|
||||
}
|
||||
|
||||
pub(super) fn optional_finite<'de, D: Deserializer<'de>>(
|
||||
deserializer: D,
|
||||
) -> Result<Option<f64>, D::Error> {
|
||||
let value = Option::<serde_json::Value>::deserialize(deserializer)?;
|
||||
let Some(value) = value else {
|
||||
return Ok(None);
|
||||
};
|
||||
let number: f64 = deserialize(value).map_err(serde::de::Error::custom)?;
|
||||
if number.is_finite() {
|
||||
Ok(Some(number))
|
||||
} else {
|
||||
Err(serde::de::Error::custom("expected finite spend"))
|
||||
}
|
||||
}
|
||||
|
||||
pub(super) fn flag<'de, D: Deserializer<'de>>(deserializer: D) -> Result<u8, D::Error> {
|
||||
match deserialize(deserializer)? {
|
||||
value @ 0..=1 => Ok(value),
|
||||
_ => Err(serde::de::Error::custom("expected 0 or 1")),
|
||||
}
|
||||
}
|
||||
|
||||
pub(super) fn percent<'de, D: Deserializer<'de>>(deserializer: D) -> Result<f64, D::Error> {
|
||||
let value: f64 = deserialize(deserializer)?;
|
||||
if value.is_finite() && (0.0..=100.0).contains(&value) {
|
||||
Ok(value)
|
||||
} else {
|
||||
Err(serde::de::Error::custom(
|
||||
"expected a finite percentage between 0 and 100",
|
||||
))
|
||||
}
|
||||
}
|
||||
|
||||
pub(super) fn boolean<'de, D: Deserializer<'de>>(deserializer: D) -> Result<bool, D::Error> {
|
||||
flag(deserializer).map(|value| value == 1)
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use crate::query::named::SpanErrorRow;
|
||||
use rstest::rstest;
|
||||
|
||||
#[rstest]
|
||||
#[case::flag_zero(serde_json::json!(0), true)]
|
||||
#[case::flag_one(serde_json::json!("1"), true)]
|
||||
#[case::invalid_flag(serde_json::json!(2), false)]
|
||||
fn access_rejects_non_boolean_flags(#[case] value: serde_json::Value, #[case] valid: bool) {
|
||||
let parameters = serde_json::json!({"all_teams": value, "team": "team", "key_hash": ""});
|
||||
assert_eq!(
|
||||
serde_json::from_value::<crate::query::lens::LensAccessParams>(parameters).is_ok(),
|
||||
valid
|
||||
);
|
||||
}
|
||||
|
||||
#[rstest]
|
||||
#[case::zero(serde_json::json!(0), true)]
|
||||
#[case::hundred(serde_json::json!("100"), true)]
|
||||
#[case::negative(serde_json::json!(-0.1), false)]
|
||||
#[case::too_large(serde_json::json!(100.1), false)]
|
||||
#[case::nan(serde_json::json!("NaN"), false)]
|
||||
fn sampling_rejects_invalid_percentages(#[case] value: serde_json::Value, #[case] valid: bool) {
|
||||
let parameters = serde_json::json!({
|
||||
"all_teams": 0, "team": "team", "key_hash": "", "source": "both", "start": 0, "end": 1,
|
||||
"agent_name": "", "service": "", "filter_keys": [], "filter_values": [], "selected_team": "",
|
||||
"execution_ids": [], "sample_cap": 0, "sample_percent": value, "preview": 0, "after": "",
|
||||
"limit": 10, "offset": 0
|
||||
});
|
||||
assert_eq!(
|
||||
serde_json::from_value::<crate::query::lens::LensSampleParams>(parameters).is_ok(),
|
||||
valid
|
||||
);
|
||||
}
|
||||
|
||||
#[rstest]
|
||||
#[case::trace("traces", true)]
|
||||
#[case::request("requests", true)]
|
||||
#[case::both("both", false)]
|
||||
#[case::unknown("unknown", false)]
|
||||
fn content_rejects_unsupported_sources(#[case] source: &str, #[case] valid: bool) {
|
||||
let parameters = serde_json::json!({
|
||||
"all_teams": 0, "team": "team", "key_hash": "", "source": source, "id": "id",
|
||||
"record_team": "team", "trace_ref": "", "cursor": "", "offset": 0
|
||||
});
|
||||
assert_eq!(
|
||||
serde_json::from_value::<crate::query::lens::LensContentParams>(parameters).is_ok(),
|
||||
valid
|
||||
);
|
||||
}
|
||||
|
||||
#[rstest]
|
||||
#[case::quoted_max(serde_json::json!(u64::MAX.to_string()), Some(u64::MAX))]
|
||||
#[case::unquoted_max(serde_json::json!(u64::MAX), Some(u64::MAX))]
|
||||
|
|
|
|||
368
litellm-rust/crates/traces-clickhouse/src/reads.rs
Normal file
368
litellm-rust/crates/traces-clickhouse/src/reads.rs
Normal file
|
|
@ -0,0 +1,368 @@
|
|||
//! Scoped trace reads: the trace list, one trace resolved with its spend, and span payloads.
|
||||
|
||||
use std::collections::HashMap;
|
||||
|
||||
use base64::{Engine, engine::general_purpose::URL_SAFE};
|
||||
use litellm_http::Client;
|
||||
use litellm_storage_clickhouse::fetch;
|
||||
use litellm_traces::{
|
||||
SpanDetail, SpanErrorPage, SpendLookup, Trace, TracePage, listed_summary,
|
||||
query::named as contracts, resolve_trace, to_ui_content,
|
||||
};
|
||||
use serde::{Deserialize, Serialize};
|
||||
|
||||
use crate::{
|
||||
Connection, Error,
|
||||
query::named::{
|
||||
ListTraces, ListTracesParams, ReadAccessParams, SpanDetail as SpanDetailQuery,
|
||||
SpanDetailParams, SpanError, SpanErrorParams, SpendByResponseIds, SpendByResponseIdsParams,
|
||||
TraceIdentity, TraceIdentityParams, TracePageSpans, TracePageSpansParams, TraceSpans,
|
||||
TraceSpansParams,
|
||||
},
|
||||
};
|
||||
|
||||
const NANOS_PER_MS: i64 = 1_000_000;
|
||||
const SPEND_WINDOW_MS: i64 = 30 * 60 * 1000;
|
||||
|
||||
fn encode_cursor<T: Serialize>(position: &T) -> String {
|
||||
URL_SAFE.encode(serde_json::to_vec(position).unwrap_or_default())
|
||||
}
|
||||
|
||||
fn decode_cursor<T: for<'de> Deserialize<'de>>(
|
||||
cursor: &str,
|
||||
kind: &'static str,
|
||||
) -> Result<T, Error> {
|
||||
URL_SAFE
|
||||
.decode(cursor)
|
||||
.ok()
|
||||
.and_then(|json| serde_json::from_slice(&json).ok())
|
||||
.ok_or(Error::InvalidCursor(kind))
|
||||
}
|
||||
|
||||
fn trace_position(cursor: Option<&str>) -> Result<(i64, String), Error> {
|
||||
let Some(cursor) = cursor.filter(|cursor| !cursor.is_empty()) else {
|
||||
return Ok((0, String::new()));
|
||||
};
|
||||
match decode_cursor::<(i64, String)>(cursor, "trace")? {
|
||||
(start_ms, trace_ref) if start_ms > 0 && !trace_ref.is_empty() => Ok((start_ms, trace_ref)),
|
||||
_ => Err(Error::InvalidCursor("trace")),
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Deserialize, Serialize)]
|
||||
struct ErrorPosition {
|
||||
offset: u64,
|
||||
version: String,
|
||||
}
|
||||
|
||||
fn error_position(cursor: Option<&str>) -> Result<Option<ErrorPosition>, Error> {
|
||||
let Some(cursor) = cursor else {
|
||||
return Ok(None);
|
||||
};
|
||||
let position = decode_cursor::<ErrorPosition>(cursor, "diagnostic")?;
|
||||
let valid_version = position.version.len() == 64
|
||||
&& position
|
||||
.version
|
||||
.bytes()
|
||||
.all(|byte| byte.is_ascii_digit() || (b'A'..=b'F').contains(&byte));
|
||||
if i64::try_from(position.offset).is_err() || !valid_version {
|
||||
return Err(Error::InvalidCursor("diagnostic"));
|
||||
}
|
||||
Ok(Some(position))
|
||||
}
|
||||
|
||||
/// The stored run a trace id names for this caller; ids can repeat across tenants and runs.
|
||||
async fn reference(
|
||||
client: &Client,
|
||||
connection: &Connection,
|
||||
access: &ReadAccessParams,
|
||||
trace_id: &str,
|
||||
trace_ref: &str,
|
||||
) -> Result<Option<String>, Error> {
|
||||
if !trace_ref.is_empty() {
|
||||
return Ok(Some(trace_ref.to_owned()));
|
||||
}
|
||||
let params = TraceIdentityParams {
|
||||
access: access.clone(),
|
||||
trace_id: trace_id.to_owned(),
|
||||
};
|
||||
let mut identities = fetch::<TraceIdentity>(client, connection, ¶ms).await?;
|
||||
if identities.len() > 1 {
|
||||
return Err(Error::AmbiguousTrace);
|
||||
}
|
||||
Ok(identities.pop().map(|identity| identity.trace_ref))
|
||||
}
|
||||
|
||||
/// Spend records behind the spans' calls. A failed lookup leaves cost unknown instead of failing
|
||||
/// the read.
|
||||
async fn spend(
|
||||
client: &Client,
|
||||
connection: &Connection,
|
||||
access: &ReadAccessParams,
|
||||
rows: &[contracts::TraceSpansRow],
|
||||
) -> Vec<contracts::SpendByResponseIdsRow> {
|
||||
let lookup = SpendLookup::new(rows);
|
||||
let (Some(start_ns), Some(end_ns)) = (
|
||||
rows.iter().map(|row| row.start_ns).min(),
|
||||
rows.iter()
|
||||
.map(|row| row.start_ns.saturating_add_unsigned(row.duration_ns))
|
||||
.max(),
|
||||
) else {
|
||||
return Vec::new();
|
||||
};
|
||||
if lookup.is_empty() {
|
||||
return Vec::new();
|
||||
}
|
||||
let params = SpendByResponseIdsParams::from(contracts::SpendByResponseIdsParams {
|
||||
access: access.clone(),
|
||||
response_ids: lookup.response_ids,
|
||||
request_ids: lookup.request_ids,
|
||||
trace_ids: lookup.trace_ids,
|
||||
start_ms: start_ns.div_euclid(NANOS_PER_MS) - SPEND_WINDOW_MS,
|
||||
end_ms: end_ns.div_euclid(NANOS_PER_MS) + SPEND_WINDOW_MS,
|
||||
});
|
||||
match fetch::<SpendByResponseIds>(client, connection, ¶ms).await {
|
||||
Ok(rows) => rows.into_iter().map(|row| row.0).collect(),
|
||||
Err(error) => {
|
||||
tracing::warn!(%error, "trace spend lookup unavailable");
|
||||
Vec::new()
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
pub async fn list_traces(
|
||||
client: &Client,
|
||||
connection: &Connection,
|
||||
access: &ReadAccessParams,
|
||||
start_ms: i64,
|
||||
end_ms: i64,
|
||||
cursor: Option<&str>,
|
||||
limit: u32,
|
||||
) -> Result<TracePage, Error> {
|
||||
let (cursor_ms, cursor_trace_id) = trace_position(cursor)?;
|
||||
let params = ListTracesParams::from(contracts::ListTracesParams {
|
||||
access: access.clone(),
|
||||
start_ms,
|
||||
end_ms,
|
||||
cursor_ms,
|
||||
cursor_trace_id,
|
||||
limit,
|
||||
});
|
||||
let page: Vec<contracts::ListTracesRow> = fetch::<ListTraces>(client, connection, ¶ms)
|
||||
.await?
|
||||
.into_iter()
|
||||
.map(|row| row.0)
|
||||
.collect();
|
||||
let next_cursor = page
|
||||
.last()
|
||||
.filter(|_| page.len() == limit as usize)
|
||||
.map(|last| encode_cursor(&(last.start_ms, &last.trace_ref)));
|
||||
let (Some(page_start), Some(page_end)) = (
|
||||
page.iter().map(|row| row.start_ms).min(),
|
||||
page.iter().map(|row| row.start_ms + row.duration_ms).max(),
|
||||
) else {
|
||||
return Ok(TracePage {
|
||||
data: Vec::new(),
|
||||
next_cursor,
|
||||
});
|
||||
};
|
||||
let span_params = TracePageSpansParams::from(contracts::TracePageSpansParams {
|
||||
access: access.clone(),
|
||||
trace_refs: page.iter().map(|row| row.trace_ref.clone()).collect(),
|
||||
start_ms: page_start,
|
||||
end_ms: page_end + 1,
|
||||
});
|
||||
let span_rows: Vec<contracts::TraceSpansRow> =
|
||||
fetch::<TracePageSpans>(client, connection, &span_params)
|
||||
.await?
|
||||
.into_iter()
|
||||
.map(|row| row.0)
|
||||
.collect();
|
||||
let spend_rows = spend(client, connection, access, &span_rows).await;
|
||||
let mut by_trace: HashMap<(String, String, String), Vec<contracts::TraceSpansRow>> =
|
||||
HashMap::new();
|
||||
for span in span_rows {
|
||||
let key = (
|
||||
span.team_id.clone(),
|
||||
span.api_key_hash.clone(),
|
||||
span.trace_id.clone(),
|
||||
);
|
||||
by_trace.entry(key).or_default().push(span);
|
||||
}
|
||||
let data = page
|
||||
.iter()
|
||||
.map(|row| {
|
||||
let spans = by_trace
|
||||
.get(&(
|
||||
row.team_id.clone(),
|
||||
row.api_key_hash.clone(),
|
||||
row.trace_id.clone(),
|
||||
))
|
||||
.map(Vec::as_slice)
|
||||
.unwrap_or_default();
|
||||
resolve_trace(&row.trace_id, &row.trace_ref, spans, &spend_rows)
|
||||
.map_or_else(|| listed_summary(row), |trace| trace.summary)
|
||||
})
|
||||
.collect();
|
||||
Ok(TracePage { data, next_cursor })
|
||||
}
|
||||
|
||||
pub async fn get_trace(
|
||||
client: &Client,
|
||||
connection: &Connection,
|
||||
access: &ReadAccessParams,
|
||||
trace_id: &str,
|
||||
trace_ref: &str,
|
||||
) -> Result<Option<Trace>, Error> {
|
||||
let Some(trace_ref) = reference(client, connection, access, trace_id, trace_ref).await? else {
|
||||
return Ok(None);
|
||||
};
|
||||
let params = TraceSpansParams {
|
||||
access: access.clone(),
|
||||
trace_id: trace_id.to_owned(),
|
||||
trace_ref: trace_ref.clone(),
|
||||
};
|
||||
let rows: Vec<contracts::TraceSpansRow> = fetch::<TraceSpans>(client, connection, ¶ms)
|
||||
.await?
|
||||
.into_iter()
|
||||
.map(|row| row.0)
|
||||
.collect();
|
||||
if rows.is_empty() {
|
||||
return Ok(None);
|
||||
}
|
||||
let spend_rows = spend(client, connection, access, &rows).await;
|
||||
Ok(resolve_trace(trace_id, &trace_ref, &rows, &spend_rows))
|
||||
}
|
||||
|
||||
pub async fn get_span(
|
||||
client: &Client,
|
||||
connection: &Connection,
|
||||
access: &ReadAccessParams,
|
||||
trace_id: &str,
|
||||
span_id: &str,
|
||||
trace_ref: &str,
|
||||
) -> Result<Option<SpanDetail>, Error> {
|
||||
let Some(trace_ref) = reference(client, connection, access, trace_id, trace_ref).await? else {
|
||||
return Ok(None);
|
||||
};
|
||||
let params = SpanDetailParams {
|
||||
access: access.clone(),
|
||||
trace_id: trace_id.to_owned(),
|
||||
trace_ref,
|
||||
span_id: span_id.to_owned(),
|
||||
};
|
||||
let row = fetch::<SpanDetailQuery>(client, connection, ¶ms)
|
||||
.await?
|
||||
.into_iter()
|
||||
.next();
|
||||
Ok(row.map(|row| SpanDetail {
|
||||
input_ui: to_ui_content(&row.input),
|
||||
output_ui: to_ui_content(&row.output),
|
||||
span_id: row.span_id,
|
||||
input: row.input,
|
||||
output: row.output,
|
||||
attributes: row.attributes,
|
||||
}))
|
||||
}
|
||||
|
||||
pub async fn get_span_error(
|
||||
client: &Client,
|
||||
connection: &Connection,
|
||||
access: &ReadAccessParams,
|
||||
trace_id: &str,
|
||||
span_id: &str,
|
||||
trace_ref: &str,
|
||||
cursor: Option<&str>,
|
||||
) -> Result<Option<SpanErrorPage>, Error> {
|
||||
let position = error_position(cursor)?;
|
||||
let Some(trace_ref) = reference(client, connection, access, trace_id, trace_ref).await? else {
|
||||
return Ok(None);
|
||||
};
|
||||
let offset = position.as_ref().map_or(0, |position| position.offset);
|
||||
let params = SpanErrorParams::from(contracts::SpanErrorParams {
|
||||
access: access.clone(),
|
||||
trace_id: trace_id.to_owned(),
|
||||
trace_ref,
|
||||
span_id: span_id.to_owned(),
|
||||
error_offset: offset,
|
||||
error_version: position
|
||||
.map(|position| position.version)
|
||||
.unwrap_or_default(),
|
||||
});
|
||||
let Some(row) = fetch::<SpanError>(client, connection, ¶ms)
|
||||
.await?
|
||||
.into_iter()
|
||||
.next()
|
||||
else {
|
||||
return Ok(None);
|
||||
};
|
||||
let row = row.0;
|
||||
let next_offset = offset + row.message.chars().count() as u64;
|
||||
let next_cursor = (next_offset < row.total_chars).then(|| {
|
||||
encode_cursor(&ErrorPosition {
|
||||
offset: next_offset,
|
||||
version: row.version,
|
||||
})
|
||||
});
|
||||
Ok(Some(SpanErrorPage {
|
||||
span_id: row.span_id,
|
||||
message: row.message,
|
||||
total_chars: row.total_chars,
|
||||
next_cursor,
|
||||
}))
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use rstest::rstest;
|
||||
|
||||
use super::*;
|
||||
|
||||
#[rstest]
|
||||
fn trace_cursor_round_trips_the_last_listed_run() {
|
||||
let cursor = encode_cursor(&(1_790_742_989_377_i64, "4bad42b84e9de3ba46fc870185f8f023"));
|
||||
assert_eq!(
|
||||
trace_position(Some(&cursor)).unwrap(),
|
||||
(
|
||||
1_790_742_989_377,
|
||||
"4bad42b84e9de3ba46fc870185f8f023".to_owned()
|
||||
)
|
||||
);
|
||||
assert_eq!(trace_position(None).unwrap(), (0, String::new()));
|
||||
assert_eq!(trace_position(Some("")).unwrap(), (0, String::new()));
|
||||
}
|
||||
|
||||
#[rstest]
|
||||
#[case::not_base64("abc")]
|
||||
#[case::not_json("bm90LWpzb24=")]
|
||||
#[case::numeric_reference("WzEsIDJd")]
|
||||
#[case::zero_start("WzAsICJ0Il0=")]
|
||||
fn malformed_trace_cursors_are_rejected(#[case] cursor: &str) {
|
||||
assert!(matches!(
|
||||
trace_position(Some(cursor)),
|
||||
Err(Error::InvalidCursor("trace"))
|
||||
));
|
||||
}
|
||||
|
||||
#[rstest]
|
||||
#[case::not_base64("garbage")]
|
||||
#[case::missing_fields("e30=")]
|
||||
#[case::not_an_object("WzEsMl0=")]
|
||||
fn malformed_diagnostic_cursors_are_rejected(#[case] cursor: &str) {
|
||||
assert!(matches!(
|
||||
error_position(Some(cursor)),
|
||||
Err(Error::InvalidCursor("diagnostic"))
|
||||
));
|
||||
}
|
||||
|
||||
#[rstest]
|
||||
#[case::lowercase_version("a".repeat(64))]
|
||||
#[case::short_version("A".repeat(63))]
|
||||
fn diagnostic_cursor_requires_a_content_version(#[case] version: String) {
|
||||
let cursor = encode_cursor(&ErrorPosition { offset: 1, version });
|
||||
assert!(matches!(
|
||||
error_position(Some(&cursor)),
|
||||
Err(Error::InvalidCursor("diagnostic"))
|
||||
));
|
||||
}
|
||||
}
|
||||
|
|
@ -78,12 +78,18 @@ pub struct NormalizedFieldDefinition {
|
|||
pub meaning: &'static str,
|
||||
}
|
||||
|
||||
pub const NORMALIZED_FIELD_DEFINITIONS: [NormalizedFieldDefinition; 9] = [
|
||||
pub const NORMALIZED_FIELD_DEFINITIONS: [NormalizedFieldDefinition; 15] = [
|
||||
NormalizedFieldDefinition {
|
||||
name: "observation_type",
|
||||
clickhouse_column: "ObservationType",
|
||||
clickhouse_type: "LowCardinality(String)",
|
||||
meaning: "Agent, LLM, tool, chain, or framework span",
|
||||
meaning: "Operation recorded by the span, including agent, model, tool, retrieval and evaluation steps",
|
||||
},
|
||||
NormalizedFieldDefinition {
|
||||
name: "wrapper_candidate",
|
||||
clickhouse_column: "WrapperCandidate",
|
||||
clickhouse_type: "Bool",
|
||||
meaning: "Span may only wrap the operation it names; the trace graph decides",
|
||||
},
|
||||
NormalizedFieldDefinition {
|
||||
name: "agent_name",
|
||||
|
|
@ -97,12 +103,30 @@ pub const NORMALIZED_FIELD_DEFINITIONS: [NormalizedFieldDefinition; 9] = [
|
|||
clickhouse_type: "LowCardinality(String)",
|
||||
meaning: "Agent framework or SDK that emitted this span, e.g. claude-agent-sdk",
|
||||
},
|
||||
NormalizedFieldDefinition {
|
||||
name: "agent_metadata",
|
||||
clickhouse_column: "AgentMetadata",
|
||||
clickhouse_type: "String",
|
||||
meaning: "Typed agent metadata as JSON, including thread, subagent, runtime and repository identity",
|
||||
},
|
||||
NormalizedFieldDefinition {
|
||||
name: "litellm_request_id",
|
||||
clickhouse_column: "LiteLLMRequestId",
|
||||
clickhouse_type: "String",
|
||||
meaning: "LiteLLM response ID used to link a span to a spend log",
|
||||
},
|
||||
NormalizedFieldDefinition {
|
||||
name: "call_keys",
|
||||
clickhouse_column: "CallKeys",
|
||||
clickhouse_type: "Array(String)",
|
||||
meaning: "Model requests the span accounts for, as kind:id (litellm_request, provider_response, transport)",
|
||||
},
|
||||
NormalizedFieldDefinition {
|
||||
name: "call_evidence",
|
||||
clickhouse_column: "CallEvidence",
|
||||
clickhouse_type: "LowCardinality(String)",
|
||||
meaning: "Whether CallKeys are all of the span's requests: complete, partial or unknown",
|
||||
},
|
||||
NormalizedFieldDefinition {
|
||||
name: "model",
|
||||
clickhouse_column: "Model",
|
||||
|
|
@ -127,10 +151,22 @@ pub const NORMALIZED_FIELD_DEFINITIONS: [NormalizedFieldDefinition; 9] = [
|
|||
clickhouse_type: "String",
|
||||
meaning: "Normalized input payload",
|
||||
},
|
||||
NormalizedFieldDefinition {
|
||||
name: "input_preview",
|
||||
clickhouse_column: "InputPreview",
|
||||
clickhouse_type: "String",
|
||||
meaning: "Latest user message of the input, else the input's first characters",
|
||||
},
|
||||
NormalizedFieldDefinition {
|
||||
name: "output",
|
||||
clickhouse_column: "Output",
|
||||
clickhouse_type: "String",
|
||||
meaning: "Normalized output payload",
|
||||
},
|
||||
NormalizedFieldDefinition {
|
||||
name: "tool_call_id",
|
||||
clickhouse_column: "ToolCallId",
|
||||
clickhouse_type: "String",
|
||||
meaning: "Tool call the span executes, shared by instrumentations recording the same call",
|
||||
},
|
||||
];
|
||||
|
|
|
|||
217
litellm-rust/crates/traces-clickhouse/src/span_row.rs
Normal file
217
litellm-rust/crates/traces-clickhouse/src/span_row.rs
Normal file
|
|
@ -0,0 +1,217 @@
|
|||
//! Decoded spans as `otel_traces` rows: payloads capped, the sending tenant stamped over whatever
|
||||
//! the export claimed, and resource maps shared across the rows that came from one resource.
|
||||
|
||||
use std::collections::{BTreeMap, HashMap};
|
||||
|
||||
use litellm_traces::{
|
||||
CallEvidence, CallKey, DecodedEvent, DecodedSpan, Shared, SharedIdentity, Tenant,
|
||||
truncate_messages, truncate_value,
|
||||
};
|
||||
use serde::Serialize;
|
||||
use serde_json::{Map, Value};
|
||||
|
||||
use crate::InsertRow;
|
||||
|
||||
/// Converts each distinct shared source once; keeping the source pins its identity.
|
||||
struct SharedValues<T>(HashMap<SharedIdentity, (Shared<T>, Shared<Value>)>);
|
||||
|
||||
impl<T: Clone> SharedValues<T> {
|
||||
fn new() -> Self {
|
||||
Self(HashMap::new())
|
||||
}
|
||||
|
||||
fn get(&mut self, source: &Shared<T>, convert: impl FnOnce(&T) -> Value) -> Shared<Value> {
|
||||
self.0
|
||||
.entry(source.identity())
|
||||
.or_insert_with(|| (source.clone(), Shared::new(convert(source))))
|
||||
.1
|
||||
.clone()
|
||||
}
|
||||
}
|
||||
|
||||
fn stamped(attributes: &BTreeMap<String, String>, tenant: &Tenant) -> Value {
|
||||
let mut stamped: Map<String, Value> = attributes
|
||||
.iter()
|
||||
.map(|(key, value)| (key.clone(), Value::from(value.as_str())))
|
||||
.collect();
|
||||
for (key, value) in [
|
||||
("litellm.team_id", &tenant.team_id),
|
||||
("litellm.api_key_hash", &tenant.api_key_hash),
|
||||
("litellm.org_id", &tenant.org_id),
|
||||
("litellm.user_id", &tenant.user_id),
|
||||
] {
|
||||
stamped.insert(key.to_owned(), Value::from(value.as_str()));
|
||||
}
|
||||
Value::Object(stamped)
|
||||
}
|
||||
|
||||
fn exception_message(events: &[DecodedEvent]) -> String {
|
||||
events
|
||||
.iter()
|
||||
.find(|event| event.name == "exception")
|
||||
.and_then(|event| {
|
||||
event
|
||||
.attributes
|
||||
.get("exception.message")
|
||||
.filter(|message| !message.is_empty())
|
||||
.or_else(|| event.attributes.get("exception.type"))
|
||||
})
|
||||
.cloned()
|
||||
.unwrap_or_default()
|
||||
}
|
||||
|
||||
fn json<T: Serialize>(value: T) -> Value {
|
||||
serde_json::to_value(value).unwrap_or(Value::Null)
|
||||
}
|
||||
|
||||
fn present_fields<T: Serialize>(value: &T) -> String {
|
||||
match json(value) {
|
||||
Value::Object(fields) => Value::Object(
|
||||
fields
|
||||
.into_iter()
|
||||
.filter(|(_, value)| !value.is_null())
|
||||
.collect(),
|
||||
)
|
||||
.to_string(),
|
||||
other => other.to_string(),
|
||||
}
|
||||
}
|
||||
|
||||
pub fn span_rows(
|
||||
spans: Vec<DecodedSpan>,
|
||||
tenant: &Tenant,
|
||||
max_value_bytes: usize,
|
||||
) -> Vec<InsertRow> {
|
||||
let mut resources = SharedValues::new();
|
||||
let mut scopes = SharedValues::new();
|
||||
spans
|
||||
.into_iter()
|
||||
.map(|span| {
|
||||
let normalized = span.normalized;
|
||||
let service = span
|
||||
.resource_attributes
|
||||
.get("service.name")
|
||||
.cloned()
|
||||
.unwrap_or_default();
|
||||
let status_message = if span.status_message.is_empty() {
|
||||
exception_message(&span.events)
|
||||
} else {
|
||||
span.status_message
|
||||
};
|
||||
let attributes: Map<String, Value> = span
|
||||
.attributes
|
||||
.into_iter()
|
||||
.filter(|(key, _)| !span.consumed_attributes.contains(&key.as_str()))
|
||||
.map(|(key, value)| (key, Value::String(truncate_value(value, max_value_bytes))))
|
||||
.collect();
|
||||
let shared = [
|
||||
(
|
||||
"ResourceAttributes",
|
||||
resources.get(&span.resource_attributes, |attributes| {
|
||||
stamped(attributes, tenant)
|
||||
}),
|
||||
),
|
||||
(
|
||||
"ScopeName",
|
||||
scopes.get(&span.scope_name, |name| Value::from(name.as_str())),
|
||||
),
|
||||
(
|
||||
"ScopeVersion",
|
||||
scopes.get(&span.scope_version, |version| Value::from(version.as_str())),
|
||||
),
|
||||
];
|
||||
let owned = [
|
||||
("Timestamp", json(span.start_ns)),
|
||||
("TraceId", Value::String(span.trace_id)),
|
||||
("SpanId", Value::String(span.span_id)),
|
||||
("ParentSpanId", Value::String(span.parent_span_id)),
|
||||
("TraceState", Value::String(span.trace_state)),
|
||||
("SpanName", Value::String(span.name)),
|
||||
("SpanKind", Value::String(span.kind)),
|
||||
("ServiceName", Value::String(service)),
|
||||
("SpanAttributes", Value::Object(attributes)),
|
||||
("Duration", json(span.end_ns - span.start_ns)),
|
||||
("StatusCode", Value::String(span.status_code)),
|
||||
("StatusMessage", Value::String(status_message)),
|
||||
("TeamId", Value::from(tenant.team_id.as_str())),
|
||||
("ApiKeyHash", Value::from(tenant.api_key_hash.as_str())),
|
||||
("UserId", Value::from(tenant.user_id.as_str())),
|
||||
("ObservationType", json(normalized.observation_type)),
|
||||
(
|
||||
"WrapperCandidate",
|
||||
Value::Bool(normalized.wrapper_candidate),
|
||||
),
|
||||
(
|
||||
"AgentName",
|
||||
Value::String(normalized.agent_name.unwrap_or_default()),
|
||||
),
|
||||
(
|
||||
"Framework",
|
||||
Value::String(
|
||||
normalized
|
||||
.framework
|
||||
.map(|integration| integration.to_string())
|
||||
.unwrap_or_default(),
|
||||
),
|
||||
),
|
||||
(
|
||||
"AgentMetadata",
|
||||
Value::String(present_fields(&normalized.agent_metadata)),
|
||||
),
|
||||
(
|
||||
"LiteLLMRequestId",
|
||||
Value::String(request_id(&normalized.calls).to_owned()),
|
||||
),
|
||||
(
|
||||
"CallKeys",
|
||||
json(
|
||||
normalized
|
||||
.calls
|
||||
.key_set()
|
||||
.into_iter()
|
||||
.flatten()
|
||||
.collect::<Vec<_>>(),
|
||||
),
|
||||
),
|
||||
("CallEvidence", json(normalized.calls.kind())),
|
||||
("Model", Value::String(normalized.model.unwrap_or_default())),
|
||||
("InputTokens", Value::from(normalized.input_tokens)),
|
||||
("OutputTokens", Value::from(normalized.output_tokens)),
|
||||
(
|
||||
"Input",
|
||||
Value::String(truncate_messages(normalized.input, max_value_bytes)),
|
||||
),
|
||||
("InputPreview", Value::String(normalized.input_preview)),
|
||||
(
|
||||
"Output",
|
||||
Value::String(truncate_value(normalized.output, max_value_bytes)),
|
||||
),
|
||||
(
|
||||
"ToolCallId",
|
||||
Value::String(normalized.tool_call_id.unwrap_or_default()),
|
||||
),
|
||||
];
|
||||
shared
|
||||
.into_iter()
|
||||
.chain(
|
||||
owned
|
||||
.into_iter()
|
||||
.map(|(column, value)| (column, Shared::new(value))),
|
||||
)
|
||||
.map(|(column, value)| (column.to_owned(), value))
|
||||
.collect()
|
||||
})
|
||||
.collect()
|
||||
}
|
||||
|
||||
fn request_id(evidence: &CallEvidence) -> &str {
|
||||
evidence
|
||||
.key_set()
|
||||
.into_iter()
|
||||
.flatten()
|
||||
.find_map(|key| match key {
|
||||
CallKey::LiteLlmRequest(id) | CallKey::ProviderResponse(id) => Some(id.as_str()),
|
||||
CallKey::Transport => None,
|
||||
})
|
||||
.unwrap_or_default()
|
||||
}
|
||||
|
|
@ -19,6 +19,9 @@ pub async fn execute_named_read(
|
|||
named_json::<TraceIdentity>(client, connection, parameters).await
|
||||
}
|
||||
ReadQuery::TraceSpans => named_json::<TraceSpans>(client, connection, parameters).await,
|
||||
ReadQuery::TracePageSpans => {
|
||||
named_json::<TracePageSpans>(client, connection, parameters).await
|
||||
}
|
||||
ReadQuery::SpanDetail => named_json::<SpanDetail>(client, connection, parameters).await,
|
||||
ReadQuery::SpanError => named_json::<SpanError>(client, connection, parameters).await,
|
||||
ReadQuery::SpendByResponseIds => {
|
||||
|
|
|
|||
|
|
@ -1,12 +1,7 @@
|
|||
#[macro_rules_attribute::apply(response_type)]
|
||||
#[cfg_attr(feature = "schema", schemars(rename = "TraceTableName"))]
|
||||
#[derive(
|
||||
Clone,
|
||||
Copy,
|
||||
Debug,
|
||||
serde::Serialize,
|
||||
strum::Display,
|
||||
strum::AsRefStr,
|
||||
strum::EnumIter,
|
||||
strum::IntoStaticStr,
|
||||
Clone, Copy, Debug, strum::Display, strum::AsRefStr, strum::EnumIter, strum::IntoStaticStr,
|
||||
)]
|
||||
#[serde(rename_all = "snake_case")]
|
||||
#[strum(serialize_all = "snake_case")]
|
||||
|
|
|
|||
119
litellm-rust/crates/traces-clickhouse/src/wire_schema.rs
Normal file
119
litellm-rust/crates/traces-clickhouse/src/wire_schema.rs
Normal file
|
|
@ -0,0 +1,119 @@
|
|||
use std::collections::BTreeMap;
|
||||
|
||||
use schemars::{JsonSchema, Schema, SchemaGenerator, generate::SchemaSettings};
|
||||
use serde_json::json;
|
||||
|
||||
use crate::query::lens;
|
||||
|
||||
fn quoted_u64() -> Schema {
|
||||
let upper = u64::MAX.to_string();
|
||||
let alternatives = upper
|
||||
.char_indices()
|
||||
.filter_map(|(index, digit)| {
|
||||
let lower = if index == 0 { '1' } else { '0' };
|
||||
if digit <= lower {
|
||||
return None;
|
||||
}
|
||||
Some(format!(
|
||||
"{}[{}-{}][0-9]{{{}}}",
|
||||
&upper[..index],
|
||||
lower,
|
||||
char::from(digit as u8 - 1),
|
||||
upper.len() - index - 1
|
||||
))
|
||||
})
|
||||
.collect::<Vec<_>>()
|
||||
.join("|");
|
||||
json!({
|
||||
"type": "string",
|
||||
"pattern": format!("^(?:0|[1-9][0-9]{{0,{}}}|{alternatives}|{upper})$", upper.len() - 2),
|
||||
})
|
||||
.try_into()
|
||||
.unwrap()
|
||||
}
|
||||
|
||||
fn numeric_wire(normalized: Schema, python_type: String) -> Schema {
|
||||
json!({
|
||||
"anyOf": [normalized, quoted_u64()],
|
||||
"x-python-normalized": {"type": python_type, "minimum": 0, "maximum": u64::MAX},
|
||||
})
|
||||
.try_into()
|
||||
.unwrap()
|
||||
}
|
||||
|
||||
pub(crate) fn u64_number(generator: &mut SchemaGenerator) -> Schema {
|
||||
numeric_wire(u64::json_schema(generator), "int".to_owned())
|
||||
}
|
||||
|
||||
pub(crate) fn flag_number(_: &mut SchemaGenerator) -> Schema {
|
||||
json!({
|
||||
"anyOf": [{"type": "integer", "enum": [0, 1]}, {"type": "string", "enum": ["0", "1"]}],
|
||||
"x-python-normalized": {"type": "int", "minimum": 0, "maximum": 1}
|
||||
})
|
||||
.try_into()
|
||||
.unwrap()
|
||||
}
|
||||
|
||||
pub(crate) fn boolean_flag(_: &mut SchemaGenerator) -> Schema {
|
||||
json!({
|
||||
"anyOf": [{"type": "boolean"}, {"type": "integer", "enum": [0, 1]}, {"type": "string", "enum": ["0", "1"]}],
|
||||
"default": false,
|
||||
"x-python-normalized": {"type": "bool"}
|
||||
}).try_into().unwrap()
|
||||
}
|
||||
|
||||
pub(crate) fn selected(generator: &mut SchemaGenerator) -> Schema {
|
||||
u64_number(generator)
|
||||
}
|
||||
|
||||
fn received<T: JsonSchema>() -> Schema {
|
||||
SchemaSettings::draft2020_12()
|
||||
.for_deserialize()
|
||||
.with_transform(litellm_traces::schema::integer_bounds)
|
||||
.into_generator()
|
||||
.into_root_schema_for::<T>()
|
||||
}
|
||||
|
||||
pub fn schemas() -> BTreeMap<&'static str, Schema> {
|
||||
BTreeMap::from([
|
||||
("ReadQueryName", json!({"$schema": "https://json-schema.org/draft/2020-12/schema", "title": "ReadQueryName", "type": "string", "enum": lens::LENS_QUERIES.map(|query| query.to_string())}).try_into().unwrap()),
|
||||
("LensAccessParams", received::<lens::LensAccessParams>()),
|
||||
("LensSampleParams", received::<lens::LensSampleParams>()),
|
||||
("LensContentParams", received::<lens::LensContentParams>()),
|
||||
("LensEvidenceParams", received::<lens::LensEvidenceParams>()),
|
||||
(
|
||||
"ActivityAvailability",
|
||||
received::<lens::LensAvailabilityRow>(),
|
||||
),
|
||||
("ExecutionRow", received::<lens::LensSampleRow>()),
|
||||
("PartRow", received::<lens::LensContentRow>()),
|
||||
("CountRow", received::<lens::LensEvidenceRow>()),
|
||||
("AgentRow", received::<lens::LensAgentsRow>()),
|
||||
("TraceQueryHelp", crate::query::help_schema()),
|
||||
])
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use rstest::rstest;
|
||||
|
||||
#[rstest]
|
||||
#[case::zero(json!(0), true)]
|
||||
#[case::quoted_zero(json!("0"), true)]
|
||||
#[case::maximum(json!(u64::MAX), true)]
|
||||
#[case::quoted_maximum(json!(u64::MAX.to_string()), true)]
|
||||
#[case::negative(json!(-1), false)]
|
||||
#[case::overflow(json!((u128::from(u64::MAX) + 1).to_string()), false)]
|
||||
#[case::fraction(json!(1.5), false)]
|
||||
fn count_schema_enforces_the_native_range(
|
||||
#[case] value: serde_json::Value,
|
||||
#[case] valid: bool,
|
||||
) {
|
||||
let schema = received::<lens::LensEvidenceRow>();
|
||||
assert_eq!(
|
||||
jsonschema::is_valid(schema.as_value(), &json!({"count": value})),
|
||||
valid
|
||||
);
|
||||
}
|
||||
}
|
||||
|
|
@ -1,65 +1,268 @@
|
|||
Trace SQL query guide
|
||||
|
||||
Live ClickHouse schema
|
||||
{% for table in tables %}
|
||||
{% block live_schema -%}
|
||||
{% for table in tables -%}
|
||||
{{ table.name }}
|
||||
{% for column in table.columns %}{{ column.name }}: {{ column.kind }}
|
||||
{% endfor %}{% endfor %}
|
||||
Normalized span fields
|
||||
{% for field in normalized_fields %}{{ field.name }}: otel_traces.{{ field.clickhouse_column }} ({{ field.clickhouse_type }})
|
||||
{{ field.meaning }}
|
||||
{% for column in table.columns -%}
|
||||
{{ column.name }}: {{ column.kind }}
|
||||
{% endfor %}
|
||||
Observed LLM call metadata
|
||||
{% endfor -%}
|
||||
{%- endblock %}
|
||||
|
||||
{% block normalized_fields -%}
|
||||
{% for field in normalized_fields -%}
|
||||
{{ field.name }}: otel_traces.{{ field.clickhouse_column }} ({{ field.clickhouse_type }})
|
||||
{{ field.meaning }}
|
||||
{% endfor -%}
|
||||
{%- endblock %}
|
||||
|
||||
{% block metadata -%}
|
||||
{{ metadata.scope }}
|
||||
{% match metadata.discovery %}{% when Discovery::Unavailable(error) %}Metadata discovery unavailable: {{ error }}
|
||||
{% when Discovery::Observed(sample) %}Sampled rows: {{ sample.sampled_rows }}; invalid JSON rows: {{ sample.invalid_json_rows }}; truncated: {{ sample.truncated }}
|
||||
{% if sample.fields.is_empty() %}No metadata paths found in the sampled rows
|
||||
{% else %}{% for field in sample.fields %}{{ field.expression }}: {% for kind in field.types %}{{ kind }} {% endfor %}
|
||||
{% endfor %}{% endif %}{% endmatch %}
|
||||
Observed span and resource attributes
|
||||
{% for catalog in attributes %}{{ catalog.table }}.{{ catalog.column }}
|
||||
Sampling SQL:
|
||||
{{ metadata.sample_sql }}
|
||||
{% match metadata.discovery -%}
|
||||
{% when Discovery::Unavailable(error) -%}
|
||||
Metadata discovery unavailable: {{ error }}
|
||||
{% when Discovery::Observed(sample) -%}
|
||||
Sampled rows: {{ sample.sampled_rows }}; invalid JSON rows: {{ sample.invalid_json_rows }}; truncated: {{ sample.truncated }}
|
||||
{% if sample.fields.is_empty() -%}
|
||||
No metadata paths found in the sampled rows
|
||||
{% else -%}
|
||||
{% for field in sample.fields -%}
|
||||
{{ field.expression }}: {{ field.types|join(", ") }}
|
||||
{% endfor -%}
|
||||
{% endif -%}
|
||||
{% endmatch -%}
|
||||
{%- endblock %}
|
||||
|
||||
{% block attributes -%}
|
||||
{% for catalog in attributes -%}
|
||||
{{ catalog.table }}.{{ catalog.column }}
|
||||
{{ catalog.scope }}
|
||||
{% match catalog.discovery %}{% when Discovery::Unavailable(error) %}Attribute discovery unavailable: {{ error }}
|
||||
{% when Discovery::Observed(sample) %}{% if sample.fields.is_empty() %}No attribute keys found in the sampled spans
|
||||
{% else %}{% for field in sample.fields %}{{ field.expression }}: {{ field.kind }}
|
||||
{% endfor %}{% endif %}{% endmatch %}{% endfor %}
|
||||
Examples
|
||||
Discovery SQL:
|
||||
{{ catalog.discovery_sql }}
|
||||
{% match catalog.discovery -%}
|
||||
{% when Discovery::Unavailable(error) -%}
|
||||
Attribute discovery unavailable: {{ error }}
|
||||
{% when Discovery::Observed(sample) -%}
|
||||
Truncated: {{ sample.truncated }}
|
||||
{% if sample.fields.is_empty() -%}
|
||||
No attribute keys found in the sampled spans
|
||||
{% else -%}
|
||||
{% for field in sample.fields -%}
|
||||
{{ field.expression }}: {{ field.kind }}
|
||||
{% endfor -%}
|
||||
{% endif -%}
|
||||
{% endmatch %}
|
||||
{% endfor -%}
|
||||
{%- endblock %}
|
||||
|
||||
{% block recent_spans_name %}Recent normalized LLM spans{% endblock %}
|
||||
{% block recent_spans_sql %}SELECT TraceId, SpanId, Model, InputTokens, OutputTokens, Duration / 1000000 AS duration_ms FROM otel_traces WHERE Timestamp >= now() - INTERVAL 1 DAY AND ObservationType = 'llm' ORDER BY Timestamp DESC LIMIT 100{% endblock %}
|
||||
{% block recent_spans_name -%}
|
||||
Recent normalized LLM spans
|
||||
{%- endblock %}
|
||||
|
||||
{% block custom_metadata_name %}Find calls by custom metadata{% endblock %}
|
||||
{% block custom_metadata_sql %}SELECT request_id, response_id, model, spend, JSONExtractString(metadata, 'project') AS project FROM spend_logs FINAL WHERE start_time >= now() - INTERVAL 1 DAY AND JSONHas(metadata, 'project') AND JSONExtractString(metadata, 'project') = 'example' ORDER BY start_time DESC LIMIT 100{% endblock %}
|
||||
{% block recent_spans_sql -%}
|
||||
SELECT
|
||||
TraceId, SpanId, Model, InputTokens, OutputTokens,
|
||||
Duration / 1000000 AS duration_ms
|
||||
FROM otel_traces
|
||||
WHERE Timestamp >= now() - INTERVAL 1 DAY
|
||||
AND ObservationType = 'llm'
|
||||
ORDER BY Timestamp DESC
|
||||
LIMIT 100
|
||||
{%- endblock %}
|
||||
|
||||
{% block nested_metadata_name %}Nested metadata with unknown types{% endblock %}
|
||||
{% block nested_metadata_sql %}SELECT request_id, JSONType(metadata, 'labels', 'priority') AS type, JSONExtractRaw(metadata, 'labels', 'priority') AS value FROM spend_logs FINAL WHERE start_time >= now() - INTERVAL 1 DAY AND JSONHas(metadata, 'labels', 'priority') LIMIT 100{% endblock %}
|
||||
{% block custom_metadata_name -%}
|
||||
Find calls by custom metadata
|
||||
{%- endblock %}
|
||||
|
||||
{% block correlated_calls_name %}Traces correlated with LLM call metadata{% endblock %}
|
||||
{% block correlated_calls_sql %}SELECT t.TraceId, t.SpanId, s.request_id, s.spend, s.metadata FROM otel_traces AS t INNER JOIN (SELECT * FROM spend_logs FINAL WHERE start_time >= now() - INTERVAL 1 DAY) AS s ON t.LiteLLMRequestId = s.response_id AND t.TeamId = s.team_id AND (t.TeamId != '' OR (t.UserId != '' AND t.UserId = s.user) OR (t.ApiKeyHash != '' AND t.ApiKeyHash = s.api_key)) WHERE t.Timestamp >= now() - INTERVAL 1 DAY AND t.LiteLLMRequestId != '' AND JSONExtractString(s.metadata, 'project') = 'example' LIMIT 100{% endblock %}
|
||||
{% block custom_metadata_sql -%}
|
||||
SELECT
|
||||
request_id, response_id, model, spend, JSONExtractString(metadata, 'project') AS project
|
||||
FROM spend_logs FINAL
|
||||
WHERE start_time >= now() - INTERVAL 1 DAY
|
||||
AND JSONHas(metadata, 'project')
|
||||
AND JSONExtractString(metadata, 'project') = 'example'
|
||||
ORDER BY start_time DESC
|
||||
LIMIT 100
|
||||
{%- endblock %}
|
||||
|
||||
{% block discover_keys_name %}Discover metadata keys over a different window{% endblock %}
|
||||
{% block discover_keys_sql %}SELECT DISTINCT arrayJoin(JSONExtractKeys(metadata)) AS key FROM spend_logs FINAL WHERE start_time >= now() - INTERVAL 30 DAY ORDER BY key LIMIT 200{% endblock %}
|
||||
{% block nested_metadata_name -%}
|
||||
Nested metadata with unknown types
|
||||
{%- endblock %}
|
||||
|
||||
Gotchas
|
||||
{% block nested_metadata_sql -%}
|
||||
SELECT
|
||||
request_id,
|
||||
JSONType(metadata, 'labels', 'priority') AS type,
|
||||
JSONExtractRaw(metadata, 'labels', 'priority') AS value
|
||||
FROM spend_logs FINAL
|
||||
WHERE start_time >= now() - INTERVAL 1 DAY
|
||||
AND JSONHas(metadata, 'labels', 'priority')
|
||||
LIMIT 100
|
||||
{%- endblock %}
|
||||
|
||||
{% block time_window %}Always bound Timestamp or start_time and use LIMIT; add TeamId/ApiKeyHash or team_id/api_key filters when investigating one tenant{% endblock %}
|
||||
{% block correlated_calls_name -%}
|
||||
Traces correlated with LLM call metadata
|
||||
{%- endblock %}
|
||||
|
||||
{% block reader_limits %}The reader enforces {{ limits.result_rows }} result rows, {{ limits.result_mib() }} MiB response bytes, {{ limits.memory_mib() }} MiB memory and a {{ limits.execution_seconds }} second query limit; exceeding limits fails instead of returning partial results{% endblock %}
|
||||
{% block correlated_calls_sql -%}
|
||||
SELECT t.TraceId, t.SpanId, s.request_id, s.spend, s.metadata
|
||||
FROM otel_traces AS t
|
||||
INNER JOIN (
|
||||
SELECT *
|
||||
FROM spend_logs FINAL
|
||||
WHERE start_time >= now() - INTERVAL 1 DAY
|
||||
) AS s
|
||||
ON t.LiteLLMRequestId = s.response_id
|
||||
AND t.TeamId = s.team_id
|
||||
AND ((t.UserId != '' AND t.UserId = s.user)
|
||||
OR (t.ApiKeyHash != '' AND t.ApiKeyHash = s.api_key))
|
||||
WHERE t.Timestamp >= now() - INTERVAL 1 DAY
|
||||
AND t.LiteLLMRequestId != ''
|
||||
LIMIT 100
|
||||
{%- endblock %}
|
||||
|
||||
{% block reader_profile %}LiteLLM provisions SELECT-only readers from the configured ClickHouse connection and enforces request-log visibility through row policies. Callers see their own user rows and permitted teams. Provisioning requires CREATE USER, ALTER USER, CREATE ROW POLICY, and GRANT SELECT permissions{% endblock %}
|
||||
{% block discover_keys_name -%}
|
||||
Discover metadata keys over a different window
|
||||
{%- endblock %}
|
||||
|
||||
{% block output_format %}Do not add FORMAT clauses; the endpoint requires ClickHouse JSON output{% endblock %}
|
||||
{% block discover_keys_sql -%}
|
||||
SELECT
|
||||
DISTINCT arrayJoin(JSONExtractKeys(metadata)) AS key
|
||||
FROM spend_logs FINAL
|
||||
WHERE start_time >= now() - INTERVAL 30 DAY
|
||||
ORDER BY key
|
||||
LIMIT 200
|
||||
{%- endblock %}
|
||||
|
||||
{% block json_values %}metadata is a JSON-encoded String; use JSONHas before typed extraction to distinguish missing values from empty strings, zero and false{% endblock %}
|
||||
{% block recent_spend_name -%}
|
||||
Recent spend records
|
||||
{%- endblock %}
|
||||
|
||||
{% block map_values %}SpanAttributes and ResourceAttributes are Map(String, String); missing map keys return an empty string, so use mapContains for existence checks{% endblock %}
|
||||
{% block recent_spend_sql -%}
|
||||
SELECT
|
||||
request_id, response_id, trace_id, span_id, model, spend,
|
||||
prompt_tokens, completion_tokens, status,
|
||||
JSONExtractBool(metadata, 'synthetic_spend') AS synthetic_spend
|
||||
FROM spend_logs FINAL
|
||||
WHERE start_time >= now() - INTERVAL 1 DAY
|
||||
ORDER BY start_time DESC, request_id
|
||||
LIMIT 100
|
||||
{%- endblock %}
|
||||
|
||||
{% block literal_keys %}Use the discovered path components as separate JSONExtract arguments; a dot inside a key is literal, not a path separator{% endblock %}
|
||||
{% block model_spend_name -%}
|
||||
Spend and tokens by model
|
||||
{%- endblock %}
|
||||
|
||||
{% block time_units %}Duration is nanoseconds; Timestamp has nanosecond precision, spend start_time has millisecond precision{% endblock %}
|
||||
{% block model_spend_sql -%}
|
||||
SELECT
|
||||
team_id, model, requests, unknown_cost_requests,
|
||||
if(unknown_cost_requests = 0, recorded_spend, NULL) AS spend,
|
||||
input_tokens, output_tokens
|
||||
FROM (
|
||||
SELECT
|
||||
team_id, model, count() AS requests,
|
||||
countIf(isNull(spend) OR NOT isFinite(spend)) AS unknown_cost_requests,
|
||||
sum(spend) AS recorded_spend,
|
||||
sum(prompt_tokens) AS input_tokens,
|
||||
sum(completion_tokens) AS output_tokens
|
||||
FROM spend_logs FINAL
|
||||
WHERE start_time >= now() - INTERVAL 1 DAY
|
||||
GROUP BY team_id, model
|
||||
)
|
||||
ORDER BY team_id, model
|
||||
LIMIT 100
|
||||
{%- endblock %}
|
||||
|
||||
{% block spend_totals %}Use spend_logs FINAL to collapse replacement rows before totals. Shared response IDs and multiple spans can multiply costs in joins; require one spend match per response ID and ownership before aggregating. Missing IDs or costs leave totals unknown{% endblock %}
|
||||
{% block trace_spend_name -%}
|
||||
Recorded spend by trace
|
||||
{%- endblock %}
|
||||
|
||||
{% block trace_rollups %}agent_traces_by_key uses SimpleAggregateFunction columns; group by TeamId, ApiKeyHash and TraceId, using min(StartTs), max(EndTs), sum(SpanCount) and groupUniqArrayArray(Models). Do not use Merge combinators{% endblock %}
|
||||
{% block trace_spend_sql -%}
|
||||
SELECT
|
||||
team_id, api_key, trace_id, count() AS requests,
|
||||
countIf(isNull(spend) OR NOT isFinite(spend)) AS unknown_cost_requests,
|
||||
if(unknown_cost_requests = 0, sum(spend), NULL) AS recorded_spend
|
||||
FROM spend_logs FINAL
|
||||
WHERE start_time >= now() - INTERVAL 1 DAY
|
||||
AND trace_id != ''
|
||||
GROUP BY team_id, api_key, trace_id
|
||||
ORDER BY team_id, api_key, trace_id
|
||||
LIMIT 100
|
||||
{%- endblock %}
|
||||
|
||||
{% block sampling %}Discovery is sampled, contains no metadata values, and is not an exhaustive schema. Edit the supplied discovery SQL for older data or nested JSONExtractKeys(metadata, 'parent'){% endblock %}
|
||||
{% block unmatched_spans_name -%}
|
||||
LLM spans without a direct spend match
|
||||
{%- endblock %}
|
||||
|
||||
{% block unmatched_spans_sql -%}
|
||||
SELECT
|
||||
t.TraceId, t.SpanId, t.Model, t.LiteLLMRequestId,
|
||||
t.InputTokens, t.OutputTokens
|
||||
FROM otel_traces AS t
|
||||
LEFT ANTI JOIN (
|
||||
SELECT *
|
||||
FROM spend_logs FINAL
|
||||
WHERE start_time >= now() - INTERVAL 1 DAY
|
||||
) AS s
|
||||
ON t.TeamId = s.team_id
|
||||
AND ((t.UserId != '' AND t.UserId = s.user)
|
||||
OR (t.ApiKeyHash != '' AND t.ApiKeyHash = s.api_key))
|
||||
AND t.LiteLLMRequestId != ''
|
||||
AND (t.LiteLLMRequestId = s.response_id OR t.LiteLLMRequestId = s.request_id)
|
||||
WHERE t.Timestamp >= now() - INTERVAL 1 DAY
|
||||
AND t.ObservationType = 'llm'
|
||||
ORDER BY t.Timestamp DESC, t.SpanId
|
||||
LIMIT 100
|
||||
{%- endblock %}
|
||||
|
||||
{% block time_window -%}
|
||||
Always bound Timestamp or start_time and use LIMIT; add TeamId/ApiKeyHash or team_id/api_key filters when investigating one tenant
|
||||
{%- endblock %}
|
||||
|
||||
{% block reader_limits -%}
|
||||
The reader enforces {{ limits.result_rows }} result rows, {{ limits.result_mib() }} MiB response bytes, {{ limits.memory_mib() }} MiB memory and a {{ limits.execution_seconds }} second query limit; exceeding limits fails instead of returning partial results
|
||||
{%- endblock %}
|
||||
|
||||
{% block reader_profile -%}
|
||||
LiteLLM provisions SELECT-only readers from the configured ClickHouse connection and enforces request-log visibility through row policies. Callers see their own user rows and permitted teams. Provisioning requires CREATE USER, ALTER USER, CREATE ROW POLICY, and GRANT SELECT permissions
|
||||
{%- endblock %}
|
||||
|
||||
{% block output_format -%}
|
||||
Do not add FORMAT clauses; the endpoint requires ClickHouse JSON output
|
||||
{%- endblock %}
|
||||
|
||||
{% block json_values -%}
|
||||
metadata is a JSON-encoded String; use JSONHas before typed extraction to distinguish missing values from empty strings, zero and false
|
||||
{%- endblock %}
|
||||
|
||||
{% block map_values -%}
|
||||
SpanAttributes and ResourceAttributes are Map(String, String); missing map keys return an empty string, so use mapContains for existence checks
|
||||
{%- endblock %}
|
||||
|
||||
{% block literal_keys -%}
|
||||
Use the discovered path components as separate JSONExtract arguments; a dot inside a key is literal, not a path separator
|
||||
{%- endblock %}
|
||||
|
||||
{% block time_units -%}
|
||||
Duration is nanoseconds; Timestamp has nanosecond precision, spend start_time has millisecond precision
|
||||
{%- endblock %}
|
||||
|
||||
{% block missing_spend -%}
|
||||
Token usage does not establish billed spend. OTLP exports without companion spend_logs rows have unknown cost; synthetic fixture spend is marked by metadata.synthetic_spend
|
||||
{%- endblock %}
|
||||
|
||||
{% block partial_spend -%}
|
||||
Recorded spend by trace totals only requests whose spend_logs.trace_id is populated. Direct ID joins do not resolve every CallKeys entry, managed Responses IDs, or transport correlation. Use the trace detail API for resolved totals; unmatched spans are a starting point for investigation
|
||||
{%- endblock %}
|
||||
|
||||
{% block spend_totals -%}
|
||||
Use spend_logs FINAL to collapse replacement rows before totals. Shared response IDs and multiple spans can multiply costs in joins; require one spend match per response ID and ownership before aggregating. Missing IDs or costs leave totals unknown
|
||||
{%- endblock %}
|
||||
|
||||
{% block trace_rollups -%}
|
||||
agent_traces_by_key uses SimpleAggregateFunction columns; group by TeamId, ApiKeyHash and TraceId, using min(StartTs), max(EndTs), sum(SpanCount) and groupUniqArrayArray(Models). Do not use Merge combinators
|
||||
{%- endblock %}
|
||||
|
||||
{% block sampling -%}
|
||||
Discovery is sampled, contains no metadata values, and is not an exhaustive schema. Edit the supplied discovery SQL for older data or nested JSONExtractKeys(metadata, 'parent')
|
||||
{%- endblock %}
|
||||
|
|
|
|||
|
|
@ -8,6 +8,18 @@ Raw OTLP exports live in `crates/traces/tests/fixtures/query_*.json`. The seeded
|
|||
|
||||
The swarm capture has handoff spans marked ERROR with `ParentCommand` exception events and a root with UNSET status. These are exported diagnostic statuses, which do not establish a failed execution. The tests preserve incoming statuses and check root status separately from the count of error spans, deriving both from the decoded export. They do not infer an execution outcome from exception text, framework names, successful model calls, or output presence. Framework-specific interpretation of control-flow exceptions belongs in the instrumentation integration
|
||||
|
||||
For a local dashboard with linked requests and traces, run `bash scripts/run_tracing_proxy_local.sh --seed` from the repository root and open `http://127.0.0.1:4002/ui/`. Log in as `admin` with password `sk-1234`, matching the UI E2E harness. The launcher keeps the proxy running until Ctrl-C and leaves the database volumes intact
|
||||
|
||||
`deeplite_swarm_spend_logs.jsonl` pairs every LLM span in the swarm export with a ClickHouse spend row. Response IDs, trace and span IDs, token counts, input, output, and timestamps come from the export. Messages and responses use the chat completion format supported by the request viewer. Spend is synthetic, set to $0.01 per request and marked in metadata, because the export does not include actual billed costs. These rows are stored here because `traces-clickhouse` owns the spend row schema
|
||||
|
||||
The simple and swarm exports for all twelve SDK examples were captured on 2026-10-03 against port 4002 using `openai/gpt-6-luna`. Each export has a matching `<name>_spend_logs.jsonl` with actual proxy spend, usage, request and response IDs, messages, and timestamps. Authorization headers, provider cookies, organization and project identifiers, and local paths were redacted. OTLP identifiers and enums use their canonical JSON encodings. `metadata.fixture_capture` identifies the associated export and whether model spans contain sufficient identity to join spend
|
||||
|
||||
The LlamaIndex captures contain provider IDs inside `output.value.raw.id`. Regression tests require normalization to retain those call keys and trace cost resolution to count nested model spans once. The Claude captures use the SDK example's local gateway adapter, which supplies the actual Anthropic message ID in the `request-id` response header. The two `claude_agent_sdk_missing_request_id_*` exports retain the earlier behavior: real spend rows exist, but model spans contain no matching call IDs, so trace spend remains unknown
|
||||
|
||||
`scripts/seed_tracing_fixtures.py` replays every JSON export in `crates/traces/tests/fixtures` through `POST /v1/traces`, then inserts all companion spend rows into ClickHouse through the production storage API and into Postgres through Prisma. The Requests table reads Postgres, while trace costs and Lens read ClickHouse. It shifts each capture into the current time window, keeping span, event, and paired spend timestamps aligned. The split `query_*.json` exports share a time shift and ID namespace to preserve cross-file parent links. Other captures get separate ID namespaces to avoid collisions between fixtures. It assigns fresh linked IDs for each run, including provider IDs inside managed response IDs, and reads the authenticated tenant from the ingested spans before stamping spend rows. The command exits unsuccessfully if any trace detail API result differs from its captured spend total or expected unknown cost. Exports without companion spend rows retain missing costs and do not create Requests entries
|
||||
|
||||
`tests/test_litellm_rust/test_traces.py` ingests these exports and spend rows into an isolated ClickHouse container, then checks trace detail costs and spend queries through the real FastAPI endpoints. Every SQL example returned by `/v1/traces/query/help` is executed through `/v1/traces/query`, including missing costs, free requests, replacement rows, and tenant ownership cases
|
||||
|
||||
`spend_logs.jsonl` contains spend insert rows with millisecond timestamps, including two versions of one request. Replace this small placeholder dataset when the actual data is available. The query fixture applies production migrations, then removes TTL from its isolated database so fixed timestamps do not expire. Background merges are stopped so rollup aggregation and `FINAL` deduplication are exercised on unmerged data. Retention behavior stays covered by the migration tests
|
||||
|
||||
Curated SQL lives in `tests/queries/*.sql`. Each query has a matching `.expected.json` containing ordered result rows for `admin`, `team`, `key`, and `other_team` readers. Update the exports and expected results together. Add a named case in `tests/queries.rs` for each new query. Assertions compare only result data, excluding server statistics and execution timing
|
||||
|
|
|
|||
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
5
litellm-rust/crates/traces-clickhouse/tests/fixtures/claude_agent_sdk_swarm_spend_logs.jsonl
vendored
Normal file
5
litellm-rust/crates/traces-clickhouse/tests/fixtures/claude_agent_sdk_swarm_spend_logs.jsonl
vendored
Normal file
File diff suppressed because one or more lines are too long
1
litellm-rust/crates/traces-clickhouse/tests/fixtures/crewai_simple_spend_logs.jsonl
vendored
Normal file
1
litellm-rust/crates/traces-clickhouse/tests/fixtures/crewai_simple_spend_logs.jsonl
vendored
Normal file
File diff suppressed because one or more lines are too long
3
litellm-rust/crates/traces-clickhouse/tests/fixtures/crewai_swarm_spend_logs.jsonl
vendored
Normal file
3
litellm-rust/crates/traces-clickhouse/tests/fixtures/crewai_swarm_spend_logs.jsonl
vendored
Normal file
File diff suppressed because one or more lines are too long
1
litellm-rust/crates/traces-clickhouse/tests/fixtures/deepagents_simple_spend_logs.jsonl
vendored
Normal file
1
litellm-rust/crates/traces-clickhouse/tests/fixtures/deepagents_simple_spend_logs.jsonl
vendored
Normal file
File diff suppressed because one or more lines are too long
5
litellm-rust/crates/traces-clickhouse/tests/fixtures/deepagents_swarm_spend_logs.jsonl
vendored
Normal file
5
litellm-rust/crates/traces-clickhouse/tests/fixtures/deepagents_swarm_spend_logs.jsonl
vendored
Normal file
File diff suppressed because one or more lines are too long
7
litellm-rust/crates/traces-clickhouse/tests/fixtures/deeplite_swarm_spend_logs.jsonl
vendored
Normal file
7
litellm-rust/crates/traces-clickhouse/tests/fixtures/deeplite_swarm_spend_logs.jsonl
vendored
Normal file
File diff suppressed because one or more lines are too long
1
litellm-rust/crates/traces-clickhouse/tests/fixtures/google_adk_simple_spend_logs.jsonl
vendored
Normal file
1
litellm-rust/crates/traces-clickhouse/tests/fixtures/google_adk_simple_spend_logs.jsonl
vendored
Normal file
File diff suppressed because one or more lines are too long
5
litellm-rust/crates/traces-clickhouse/tests/fixtures/google_adk_swarm_spend_logs.jsonl
vendored
Normal file
5
litellm-rust/crates/traces-clickhouse/tests/fixtures/google_adk_swarm_spend_logs.jsonl
vendored
Normal file
File diff suppressed because one or more lines are too long
1
litellm-rust/crates/traces-clickhouse/tests/fixtures/langchain_simple_spend_logs.jsonl
vendored
Normal file
1
litellm-rust/crates/traces-clickhouse/tests/fixtures/langchain_simple_spend_logs.jsonl
vendored
Normal file
File diff suppressed because one or more lines are too long
5
litellm-rust/crates/traces-clickhouse/tests/fixtures/langchain_swarm_spend_logs.jsonl
vendored
Normal file
5
litellm-rust/crates/traces-clickhouse/tests/fixtures/langchain_swarm_spend_logs.jsonl
vendored
Normal file
File diff suppressed because one or more lines are too long
1
litellm-rust/crates/traces-clickhouse/tests/fixtures/langgraph_simple_spend_logs.jsonl
vendored
Normal file
1
litellm-rust/crates/traces-clickhouse/tests/fixtures/langgraph_simple_spend_logs.jsonl
vendored
Normal file
File diff suppressed because one or more lines are too long
2
litellm-rust/crates/traces-clickhouse/tests/fixtures/langgraph_swarm_spend_logs.jsonl
vendored
Normal file
2
litellm-rust/crates/traces-clickhouse/tests/fixtures/langgraph_swarm_spend_logs.jsonl
vendored
Normal file
File diff suppressed because one or more lines are too long
1
litellm-rust/crates/traces-clickhouse/tests/fixtures/llamaindex_simple_spend_logs.jsonl
vendored
Normal file
1
litellm-rust/crates/traces-clickhouse/tests/fixtures/llamaindex_simple_spend_logs.jsonl
vendored
Normal file
File diff suppressed because one or more lines are too long
3
litellm-rust/crates/traces-clickhouse/tests/fixtures/llamaindex_swarm_spend_logs.jsonl
vendored
Normal file
3
litellm-rust/crates/traces-clickhouse/tests/fixtures/llamaindex_swarm_spend_logs.jsonl
vendored
Normal file
File diff suppressed because one or more lines are too long
1
litellm-rust/crates/traces-clickhouse/tests/fixtures/openai_agents_simple_spend_logs.jsonl
vendored
Normal file
1
litellm-rust/crates/traces-clickhouse/tests/fixtures/openai_agents_simple_spend_logs.jsonl
vendored
Normal file
File diff suppressed because one or more lines are too long
5
litellm-rust/crates/traces-clickhouse/tests/fixtures/openai_agents_swarm_spend_logs.jsonl
vendored
Normal file
5
litellm-rust/crates/traces-clickhouse/tests/fixtures/openai_agents_swarm_spend_logs.jsonl
vendored
Normal file
File diff suppressed because one or more lines are too long
1
litellm-rust/crates/traces-clickhouse/tests/fixtures/opentelemetry_simple_spend_logs.jsonl
vendored
Normal file
1
litellm-rust/crates/traces-clickhouse/tests/fixtures/opentelemetry_simple_spend_logs.jsonl
vendored
Normal file
File diff suppressed because one or more lines are too long
2
litellm-rust/crates/traces-clickhouse/tests/fixtures/opentelemetry_swarm_spend_logs.jsonl
vendored
Normal file
2
litellm-rust/crates/traces-clickhouse/tests/fixtures/opentelemetry_swarm_spend_logs.jsonl
vendored
Normal file
File diff suppressed because one or more lines are too long
1
litellm-rust/crates/traces-clickhouse/tests/fixtures/pydantic_ai_simple_spend_logs.jsonl
vendored
Normal file
1
litellm-rust/crates/traces-clickhouse/tests/fixtures/pydantic_ai_simple_spend_logs.jsonl
vendored
Normal file
File diff suppressed because one or more lines are too long
5
litellm-rust/crates/traces-clickhouse/tests/fixtures/pydantic_ai_swarm_spend_logs.jsonl
vendored
Normal file
5
litellm-rust/crates/traces-clickhouse/tests/fixtures/pydantic_ai_swarm_spend_logs.jsonl
vendored
Normal file
File diff suppressed because one or more lines are too long
1
litellm-rust/crates/traces-clickhouse/tests/fixtures/strands_simple_spend_logs.jsonl
vendored
Normal file
1
litellm-rust/crates/traces-clickhouse/tests/fixtures/strands_simple_spend_logs.jsonl
vendored
Normal file
File diff suppressed because one or more lines are too long
5
litellm-rust/crates/traces-clickhouse/tests/fixtures/strands_swarm_spend_logs.jsonl
vendored
Normal file
5
litellm-rust/crates/traces-clickhouse/tests/fixtures/strands_swarm_spend_logs.jsonl
vendored
Normal file
File diff suppressed because one or more lines are too long
1
litellm-rust/crates/traces-clickhouse/tests/fixtures/vercel_ai_sdk_simple_spend_logs.jsonl
vendored
Normal file
1
litellm-rust/crates/traces-clickhouse/tests/fixtures/vercel_ai_sdk_simple_spend_logs.jsonl
vendored
Normal file
File diff suppressed because one or more lines are too long
5
litellm-rust/crates/traces-clickhouse/tests/fixtures/vercel_ai_sdk_swarm_spend_logs.jsonl
vendored
Normal file
5
litellm-rust/crates/traces-clickhouse/tests/fixtures/vercel_ai_sdk_swarm_spend_logs.jsonl
vendored
Normal file
File diff suppressed because one or more lines are too long
|
|
@ -101,7 +101,7 @@ async fn schema_supports_span_rollups_and_spend_joins(
|
|||
&reader,
|
||||
&litellm_traces_clickhouse::query::named::SpanDetailParams {
|
||||
access: litellm_traces_clickhouse::query::named::ReadAccessParams {
|
||||
all_teams: 0,
|
||||
all_teams: false,
|
||||
user_id: String::new(),
|
||||
team_ids: vec!["team-1".into()],
|
||||
},
|
||||
|
|
@ -148,6 +148,8 @@ async fn schema_supports_span_rollups_and_spend_joins(
|
|||
"response_ids".into(),
|
||||
Parameter::Strings(vec!["response-1".into()]),
|
||||
),
|
||||
("request_ids".into(), Parameter::Strings(Vec::new())),
|
||||
("trace_ids".into(), Parameter::Strings(Vec::new())),
|
||||
("all_teams".into(), Parameter::Integer(0)),
|
||||
("user_id".into(), Parameter::Text(String::new())),
|
||||
("team_ids".into(), Parameter::Strings(vec!["team-1".into()])),
|
||||
|
|
@ -236,6 +238,44 @@ async fn normalized_fields_match_clickhouse_catalog(
|
|||
Ok(())
|
||||
}
|
||||
|
||||
#[rstest]
|
||||
#[tokio::test]
|
||||
async fn agent_metadata_is_stored_and_queryable(
|
||||
#[future(awt)] database: TestResult<ClickHouseDatabase>,
|
||||
) -> TestResult {
|
||||
let ready = database?;
|
||||
ensure_schema(
|
||||
&ready.client,
|
||||
&Connection::writer(&ready.url)?,
|
||||
"trace_test",
|
||||
7,
|
||||
)
|
||||
.await?;
|
||||
let metadata = serde_json::json!({"thread_id": "thread-1", "ls_subagent_id": "agent-1"});
|
||||
let timestamp = time::OffsetDateTime::now_utc().unix_timestamp_nanos() as i64;
|
||||
insert_rows(
|
||||
&ready,
|
||||
"otel_traces",
|
||||
vec![BTreeMap::from([
|
||||
("Timestamp".into(), timestamp.into()),
|
||||
("TraceId".into(), "trace-1".into()),
|
||||
("SpanId".into(), "span-1".into()),
|
||||
("AgentMetadata".into(), metadata.to_string().into()),
|
||||
])],
|
||||
)
|
||||
.await?;
|
||||
let response = read_json(
|
||||
&ready,
|
||||
"SELECT JSONExtractString(AgentMetadata, 'thread_id') AS thread_id, JSONExtractString(AgentMetadata, 'ls_subagent_id') AS subagent_id FROM trace_test.otel_traces WHERE TraceId = 'trace-1'",
|
||||
).await?;
|
||||
assert_eq!(response["data"][0]["thread_id"], metadata["thread_id"]);
|
||||
assert_eq!(
|
||||
response["data"][0]["subagent_id"],
|
||||
metadata["ls_subagent_id"]
|
||||
);
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[rstest]
|
||||
#[tokio::test]
|
||||
async fn insert_rejects_unknown_columns_even_if_url_requests_skipping_them(
|
||||
|
|
@ -796,7 +836,7 @@ async fn lens_filters_reads_and_evidence_keep_reused_trace_ids_separate(
|
|||
}))?]).await?;
|
||||
}
|
||||
let connection = Connection::configured(&database.url, "trace_test", "default", "")?;
|
||||
let parameters = BTreeMap::from([
|
||||
let sample_parameters = BTreeMap::from([
|
||||
("source".into(), Parameter::Text("traces".into())),
|
||||
("all_teams".into(), Parameter::Integer(1)),
|
||||
("team".into(), Parameter::Text(String::new())),
|
||||
|
|
@ -833,7 +873,7 @@ async fn lens_filters_reads_and_evidence_keep_reused_trace_ids_separate(
|
|||
&database.client,
|
||||
&connection,
|
||||
ReadQuery::Sample,
|
||||
¶meters,
|
||||
&sample_parameters,
|
||||
)
|
||||
.await?,
|
||||
)?;
|
||||
|
|
@ -856,13 +896,12 @@ async fn lens_filters_reads_and_evidence_keep_reused_trace_ids_separate(
|
|||
.await?,
|
||||
)?;
|
||||
assert_eq!(identities["data"].as_array().map(Vec::len), Some(2));
|
||||
let user_params = identity_params
|
||||
.into_iter()
|
||||
.chain([
|
||||
("team_ids".into(), Parameter::Strings(vec![])),
|
||||
("user_id".into(), Parameter::Text("one".into())),
|
||||
])
|
||||
.collect();
|
||||
let user_params = BTreeMap::from([
|
||||
("trace_id".into(), Parameter::Text("shared".into())),
|
||||
("all_teams".into(), Parameter::Integer(0)),
|
||||
("user_id".into(), Parameter::Text("one".into())),
|
||||
("team_ids".into(), Parameter::Strings(vec![])),
|
||||
]);
|
||||
let identity: serde_json::Value = serde_json::from_str(
|
||||
&execute_named_read(
|
||||
&database.client,
|
||||
|
|
@ -878,23 +917,23 @@ async fn lens_filters_reads_and_evidence_keep_reused_trace_ids_separate(
|
|||
.any(|row| row["trace_ref"] == identity["data"][0]["trace_ref"])
|
||||
);
|
||||
let first_ref = rows[0]["trace_ref"].as_str().expect("reference");
|
||||
let read_parameters: BTreeMap<_, _> = parameters
|
||||
.into_iter()
|
||||
.chain([
|
||||
("id".into(), Parameter::Text("shared".into())),
|
||||
("record_team".into(), Parameter::Text("team".into())),
|
||||
("trace_ref".into(), Parameter::Text(first_ref.into())),
|
||||
("cursor".into(), Parameter::Text(String::new())),
|
||||
("offset".into(), Parameter::Integer(1)),
|
||||
("span".into(), Parameter::Text("root".into())),
|
||||
])
|
||||
.collect();
|
||||
let content_parameters = BTreeMap::from([
|
||||
("all_teams".into(), Parameter::Integer(1)),
|
||||
("team".into(), Parameter::Text(String::new())),
|
||||
("key_hash".into(), Parameter::Text(String::new())),
|
||||
("source".into(), Parameter::Text("traces".into())),
|
||||
("id".into(), Parameter::Text("shared".into())),
|
||||
("record_team".into(), Parameter::Text("team".into())),
|
||||
("trace_ref".into(), Parameter::Text(first_ref.into())),
|
||||
("cursor".into(), Parameter::Text(String::new())),
|
||||
("offset".into(), Parameter::Integer(1)),
|
||||
]);
|
||||
let content: serde_json::Value = serde_json::from_str(
|
||||
&execute_named_read(
|
||||
&database.client,
|
||||
&connection,
|
||||
ReadQuery::Content,
|
||||
&read_parameters,
|
||||
&content_parameters,
|
||||
)
|
||||
.await?,
|
||||
)?;
|
||||
|
|
@ -905,10 +944,17 @@ async fn lens_filters_reads_and_evidence_keep_reused_trace_ids_separate(
|
|||
} else {
|
||||
"timeout"
|
||||
};
|
||||
let evidence_parameters = read_parameters
|
||||
.into_iter()
|
||||
.chain([("quote".into(), Parameter::Text(opposite.into()))])
|
||||
.collect();
|
||||
let evidence_parameters = BTreeMap::from([
|
||||
("all_teams".into(), Parameter::Integer(1)),
|
||||
("team".into(), Parameter::Text(String::new())),
|
||||
("key_hash".into(), Parameter::Text(String::new())),
|
||||
("source".into(), Parameter::Text("traces".into())),
|
||||
("id".into(), Parameter::Text("shared".into())),
|
||||
("record_team".into(), Parameter::Text("team".into())),
|
||||
("trace_ref".into(), Parameter::Text(first_ref.into())),
|
||||
("span".into(), Parameter::Text("root".into())),
|
||||
("quote".into(), Parameter::Text(opposite.into())),
|
||||
]);
|
||||
let evidence: serde_json::Value = serde_json::from_str(
|
||||
&execute_named_read(
|
||||
&database.client,
|
||||
|
|
@ -1317,7 +1363,7 @@ async fn lens_agent_discovery_and_selection_preserve_scope(
|
|||
.await?;
|
||||
}
|
||||
let connection = Connection::configured(&database.url, "trace_test", "default", "")?;
|
||||
let scope_parameters = BTreeMap::from([
|
||||
let agent_parameters = BTreeMap::from([
|
||||
("all_teams".into(), Parameter::Integer(0)),
|
||||
("team".into(), Parameter::Text("alpha".into())),
|
||||
("key_hash".into(), Parameter::Text("one".into())),
|
||||
|
|
@ -1327,7 +1373,7 @@ async fn lens_agent_discovery_and_selection_preserve_scope(
|
|||
&database.client,
|
||||
&connection,
|
||||
ReadQuery::Agents,
|
||||
&scope_parameters,
|
||||
&agent_parameters,
|
||||
)
|
||||
.await?,
|
||||
)?;
|
||||
|
|
@ -1337,53 +1383,58 @@ async fn lens_agent_discovery_and_selection_preserve_scope(
|
|||
{"agent_name": "research_agent"}, {"agent_name": "support_agent"}
|
||||
])
|
||||
);
|
||||
let parameters = scope_parameters
|
||||
.into_iter()
|
||||
.chain([
|
||||
("source".into(), Parameter::Text("traces".into())),
|
||||
(
|
||||
"start".into(),
|
||||
Parameter::Integer(timestamp / 1_000_000 - 1000),
|
||||
),
|
||||
(
|
||||
"end".into(),
|
||||
Parameter::Integer(timestamp / 1_000_000 + 1000),
|
||||
),
|
||||
("service".into(), Parameter::Text("shared-app".into())),
|
||||
(
|
||||
"agent_name".into(),
|
||||
Parameter::Text("research_agent".into()),
|
||||
),
|
||||
("filter_keys".into(), Parameter::Strings(vec![])),
|
||||
("filter_values".into(), Parameter::Strings(vec![])),
|
||||
("limit".into(), Parameter::Integer(100)),
|
||||
("offset".into(), Parameter::Integer(0)),
|
||||
("after".into(), Parameter::Text(String::new())),
|
||||
("sample_percent".into(), Parameter::Text("100".into())),
|
||||
("sample_cap".into(), Parameter::Integer(0)),
|
||||
("preview".into(), Parameter::Integer(1)),
|
||||
("selected_team".into(), Parameter::Text(String::new())),
|
||||
("execution_ids".into(), Parameter::Strings(vec![])),
|
||||
])
|
||||
.collect::<BTreeMap<_, _>>();
|
||||
let sample_parameters = BTreeMap::from([
|
||||
("all_teams".into(), Parameter::Integer(0)),
|
||||
("team".into(), Parameter::Text("alpha".into())),
|
||||
("key_hash".into(), Parameter::Text("one".into())),
|
||||
("source".into(), Parameter::Text("traces".into())),
|
||||
(
|
||||
"start".into(),
|
||||
Parameter::Integer(timestamp / 1_000_000 - 1000),
|
||||
),
|
||||
(
|
||||
"end".into(),
|
||||
Parameter::Integer(timestamp / 1_000_000 + 1000),
|
||||
),
|
||||
("service".into(), Parameter::Text("shared-app".into())),
|
||||
(
|
||||
"agent_name".into(),
|
||||
Parameter::Text("research_agent".into()),
|
||||
),
|
||||
("filter_keys".into(), Parameter::Strings(vec![])),
|
||||
("filter_values".into(), Parameter::Strings(vec![])),
|
||||
("limit".into(), Parameter::Integer(100)),
|
||||
("offset".into(), Parameter::Integer(0)),
|
||||
("after".into(), Parameter::Text(String::new())),
|
||||
("sample_percent".into(), Parameter::Text("100".into())),
|
||||
("sample_cap".into(), Parameter::Integer(0)),
|
||||
("preview".into(), Parameter::Integer(1)),
|
||||
("selected_team".into(), Parameter::Text(String::new())),
|
||||
("execution_ids".into(), Parameter::Strings(vec![])),
|
||||
]);
|
||||
let sample: serde_json::Value = serde_json::from_str(
|
||||
&execute_named_read(
|
||||
&database.client,
|
||||
&connection,
|
||||
ReadQuery::Sample,
|
||||
¶meters,
|
||||
&sample_parameters,
|
||||
)
|
||||
.await?,
|
||||
)?;
|
||||
assert_eq!(sample["data"].as_array().expect("rows").len(), 1);
|
||||
assert_eq!(sample["data"][0]["trace_id"], "research");
|
||||
assert_eq!(sample["data"][0]["span_count"], 2);
|
||||
let availability_parameters = BTreeMap::from([
|
||||
("all_teams".into(), Parameter::Integer(0)),
|
||||
("team".into(), Parameter::Text("alpha".into())),
|
||||
("key_hash".into(), Parameter::Text("one".into())),
|
||||
]);
|
||||
let available: serde_json::Value = serde_json::from_str(
|
||||
&execute_named_read(
|
||||
&database.client,
|
||||
&connection,
|
||||
ReadQuery::Availability,
|
||||
¶meters,
|
||||
&availability_parameters,
|
||||
)
|
||||
.await?,
|
||||
)?;
|
||||
|
|
@ -1427,7 +1478,7 @@ async fn query_help_discovers_live_schema_and_runs_its_examples(
|
|||
.await?;
|
||||
let metadata = serde_json::json!({
|
||||
"project": "example", "labels": {"priority": 3, "enabled": true},
|
||||
"dotted.key": "literal", "quote'\\key": null, "items": [{"name": "first"}],
|
||||
"dotted.key": "private-metadata-value", "quote'\\key": null, "items": [{"name": "first"}],
|
||||
"<custom>&{{key}}": {"nested.key": true}
|
||||
});
|
||||
insert_rows(
|
||||
|
|
@ -1435,7 +1486,7 @@ async fn query_help_discovers_live_schema_and_runs_its_examples(
|
|||
"spend_logs",
|
||||
vec![serde_json::from_value(serde_json::json!({
|
||||
"request_id": "request-1", "response_id": "response-1", "team_id": "team-1",
|
||||
"api_key": "key-1", "metadata": metadata.to_string(), "spend": 0.25,
|
||||
"api_key": "key-1", "trace_id": "trace-1", "metadata": metadata.to_string(), "spend": 0.25,
|
||||
"start_time": timestamp / 1_000_000, "end_time": timestamp / 1_000_000 + 100
|
||||
}))?],
|
||||
)
|
||||
|
|
@ -1501,9 +1552,31 @@ async fn query_help_discovers_live_schema_and_runs_its_examples(
|
|||
)));
|
||||
}
|
||||
}
|
||||
for gotcha in help["gotchas"].as_array().ok_or("missing gotchas")? {
|
||||
assert!(guide.contains(gotcha.as_str().ok_or("gotcha text")?));
|
||||
let gotchas = help["gotchas"].as_array().ok_or("missing gotchas")?;
|
||||
let gotcha_positions = gotchas
|
||||
.iter()
|
||||
.map(|gotcha| {
|
||||
guide
|
||||
.find(gotcha.as_str().expect("gotcha text"))
|
||||
.expect("rendered gotcha")
|
||||
})
|
||||
.collect::<Vec<_>>();
|
||||
assert!(gotcha_positions.windows(2).all(|pair| pair[0] < pair[1]));
|
||||
assert!(
|
||||
guide.contains(
|
||||
help["metadata"]["sample_sql"]
|
||||
.as_str()
|
||||
.ok_or("sampling SQL")?
|
||||
)
|
||||
);
|
||||
for catalog in help["attributes"].as_array().ok_or("attributes")? {
|
||||
assert!(guide.contains(catalog["discovery_sql"].as_str().ok_or("discovery SQL")?));
|
||||
assert!(guide.contains(&format!("Truncated: {}", catalog["truncated"])));
|
||||
}
|
||||
assert_eq!(
|
||||
guide.contains("No attribute keys found in the sampled spans"),
|
||||
!populated
|
||||
);
|
||||
let tables = help["tables"].as_array().ok_or("missing tables")?;
|
||||
assert_eq!(tables.len(), 3);
|
||||
let columns = tables[0]["columns"].as_array().ok_or("missing columns")?;
|
||||
|
|
@ -1557,6 +1630,7 @@ async fn query_help_discovers_live_schema_and_runs_its_examples(
|
|||
.any(|field| field["path"] == serde_json::json!(["items", 1, "name"]))
|
||||
);
|
||||
assert!(guide.contains("CustomColumn: String"));
|
||||
assert!(!guide.contains("private-metadata-value"));
|
||||
assert!(guide.contains("JSONExtractRaw(metadata, '<custom>&{{key}}', 'nested.key')"));
|
||||
assert!(guide.contains("SpanAttributes['custom.tag']"));
|
||||
assert!(guide.contains("ResourceAttributes['custom.resource']"));
|
||||
|
|
@ -1575,7 +1649,24 @@ async fn query_help_discovers_live_schema_and_runs_its_examples(
|
|||
assert_ne!(values["data"][0]["value"], "");
|
||||
}
|
||||
}
|
||||
for example in help["examples"].as_array().ok_or("missing examples")? {
|
||||
let examples = help["examples"].as_array().ok_or("missing examples")?;
|
||||
let example_positions = examples
|
||||
.iter()
|
||||
.map(|example| {
|
||||
let rendered = format!(
|
||||
"{}\n{}",
|
||||
example["name"].as_str().expect("name"),
|
||||
example["sql"].as_str().expect("SQL")
|
||||
);
|
||||
guide.find(&rendered).expect("rendered example")
|
||||
})
|
||||
.collect::<Vec<_>>();
|
||||
assert!(example_positions.windows(2).all(|pair| pair[0] < pair[1]));
|
||||
assert!(
|
||||
example_positions.last().ok_or("last example")?
|
||||
< gotcha_positions.first().ok_or("first gotcha")?
|
||||
);
|
||||
for example in examples {
|
||||
let sql = example["sql"].as_str().ok_or("missing example SQL")?;
|
||||
assert!(guide.contains(example["name"].as_str().ok_or("missing example name")?));
|
||||
assert!(guide.contains(sql));
|
||||
|
|
@ -1592,7 +1683,7 @@ async fn query_help_discovers_live_schema_and_runs_its_examples(
|
|||
let values: serde_json::Value = serde_json::from_str(&body)?;
|
||||
assert_eq!(
|
||||
values["data"].as_array().ok_or("missing data")?.is_empty(),
|
||||
!populated,
|
||||
!populated || example["name"] == "LLM spans without a direct spend match",
|
||||
"{sql}"
|
||||
);
|
||||
if populated && example["name"] == "Traces correlated with LLM call metadata" {
|
||||
|
|
@ -1666,6 +1757,18 @@ async fn query_help_preserves_schema_and_guide_when_discovery_hits_reader_limits
|
|||
guide.contains("Attribute discovery unavailable:"),
|
||||
span_rows > 1
|
||||
);
|
||||
assert!(!guide.contains("No metadata paths found in the sampled rows"));
|
||||
assert!(!guide.contains("No attribute keys found in the sampled spans"));
|
||||
assert!(
|
||||
guide.contains(
|
||||
help["metadata"]["sample_sql"]
|
||||
.as_str()
|
||||
.ok_or("sampling SQL")?
|
||||
)
|
||||
);
|
||||
for catalog in help["attributes"].as_array().ok_or("attributes")? {
|
||||
assert!(guide.contains(catalog["discovery_sql"].as_str().ok_or("discovery SQL")?));
|
||||
}
|
||||
for (catalog, unavailable) in [
|
||||
(&help["metadata"], spend_rows > 1),
|
||||
(&help["attributes"][0], span_rows > 1),
|
||||
|
|
@ -1681,28 +1784,86 @@ async fn query_help_preserves_schema_and_guide_when_discovery_hits_reader_limits
|
|||
Ok(())
|
||||
}
|
||||
|
||||
#[rstest]
|
||||
#[tokio::test]
|
||||
async fn query_help_displays_discovery_truncation(
|
||||
#[future(awt)] database: TestResult<ClickHouseDatabase>,
|
||||
) -> TestResult {
|
||||
let database = database?;
|
||||
let writer = Connection::writer(&database.url)?;
|
||||
ensure_schema(&database.client, &writer, "trace_test", 7).await?;
|
||||
execute_write(
|
||||
&database,
|
||||
"INSERT INTO trace_test.otel_traces (Timestamp, TraceId, SpanId, SpanAttributes, ResourceAttributes) \
|
||||
SELECT now64(9), 'trace', 'span', \
|
||||
mapFromArrays(arrayMap(x -> concat('key-', toString(x)), range(1000)), arrayMap(x -> 'value', range(1000))) AS attributes, \
|
||||
attributes FROM numbers(1)",
|
||||
)
|
||||
.await?;
|
||||
execute_write(
|
||||
&database,
|
||||
"INSERT INTO trace_test.spend_logs (request_id, start_time, end_time, metadata) \
|
||||
SELECT toString(number), now64(3), now64(3), '{\"key\":true}' FROM numbers(1000)",
|
||||
)
|
||||
.await?;
|
||||
let reader = Connection::configured(&database.url, "trace_test", "default", "")?;
|
||||
let help = serde_json::to_value(
|
||||
litellm_traces_clickhouse::query_help(&database.client, &reader).await?,
|
||||
)?;
|
||||
let guide = help["guide"].as_str().ok_or("guide")?;
|
||||
assert_eq!(help["metadata"]["truncated"], true);
|
||||
assert!(guide.contains("truncated: true"));
|
||||
for catalog in help["attributes"].as_array().ok_or("attributes")? {
|
||||
assert_eq!(catalog["truncated"], true);
|
||||
let displayed = format!(
|
||||
"{}.{}",
|
||||
catalog["table"].as_str().ok_or("table")?,
|
||||
catalog["column"].as_str().ok_or("column")?
|
||||
);
|
||||
let section = guide.split(&displayed).nth(1).ok_or("attribute section")?;
|
||||
assert!(
|
||||
section
|
||||
.split("\n\n")
|
||||
.next()
|
||||
.ok_or("catalog body")?
|
||||
.contains("Truncated: true")
|
||||
);
|
||||
for field in catalog["fields"].as_array().ok_or("fields")? {
|
||||
assert!(section.contains(field["expression"].as_str().ok_or("expression")?));
|
||||
}
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[rstest]
|
||||
fn field_definitions_match_serialized_normalized_span() {
|
||||
use litellm_traces::decode_otlp;
|
||||
use litellm_traces::{Tenant, decode_otlp};
|
||||
use litellm_traces_clickhouse::span_rows;
|
||||
use std::collections::BTreeSet;
|
||||
let spans = decode_otlp(
|
||||
br#"{"resourceSpans":[{"scopeSpans":[{"spans":[{"traceId":"11111111111111111111111111111111","spanId":"2222222222222222","name":"root"}]}]}]}"#,
|
||||
Some("application/json"),
|
||||
)
|
||||
.expect("valid OTLP");
|
||||
let fields = &spans[0].normalized;
|
||||
let serialized = serde_json::to_value(fields).expect("serializable fields");
|
||||
let keys: BTreeSet<_> = serialized
|
||||
let tenant = Tenant {
|
||||
team_id: "team".into(),
|
||||
api_key_hash: "key".into(),
|
||||
..Tenant::default()
|
||||
};
|
||||
let rows = span_rows(spans, &tenant, 64 * 1024);
|
||||
let row =
|
||||
serde_json::to_value(rows.first().expect("storage row")).expect("serializable storage row");
|
||||
let keys: BTreeSet<_> = row
|
||||
.as_object()
|
||||
.expect("field object")
|
||||
.expect("storage row object")
|
||||
.keys()
|
||||
.map(String::as_str)
|
||||
.collect();
|
||||
let mapped: BTreeSet<_> = NORMALIZED_FIELD_DEFINITIONS
|
||||
.iter()
|
||||
.map(|field| field.name)
|
||||
.map(|field| field.clickhouse_column)
|
||||
.collect();
|
||||
assert_eq!(keys, mapped);
|
||||
assert!(mapped.is_subset(&keys));
|
||||
}
|
||||
|
||||
#[rstest]
|
||||
|
|
@ -1743,6 +1904,8 @@ async fn named_and_sql_readers_share_request_log_visibility(
|
|||
"api_key_hash": legacy_key.unwrap_or_default(),
|
||||
}))?,
|
||||
response_ids: vec!["shared-response".into()],
|
||||
request_ids: Vec::new(),
|
||||
trace_ids: Vec::new(),
|
||||
start_ms: timestamp / 1_000_000 - 1,
|
||||
end_ms: timestamp / 1_000_000 + 1,
|
||||
});
|
||||
|
|
@ -1813,7 +1976,7 @@ async fn rollup_cost_completeness_preserves_missing_ids_and_fails_closed_for_his
|
|||
let params = litellm_traces_clickhouse::query::named::ListTracesParams::from(
|
||||
litellm_traces::query::named::ListTracesParams {
|
||||
access: litellm_traces::query::named::ReadAccessParams {
|
||||
all_teams: 0,
|
||||
all_teams: false,
|
||||
user_id: "".into(),
|
||||
team_ids: vec!["team".into()],
|
||||
},
|
||||
|
|
@ -1835,7 +1998,7 @@ async fn rollup_cost_completeness_preserves_missing_ids_and_fails_closed_for_his
|
|||
access: litellm_traces::query::named::ReadAccessParams {
|
||||
user_id: "owner".into(),
|
||||
team_ids: vec![],
|
||||
all_teams: 0,
|
||||
all_teams: false,
|
||||
},
|
||||
..params.0
|
||||
},
|
||||
|
|
@ -1932,7 +2095,7 @@ async fn agent_final_answer_preserves_visibility_and_trace_ownership(
|
|||
&reader,
|
||||
&SpanDetailParams {
|
||||
access: ReadAccessParams {
|
||||
all_teams,
|
||||
all_teams: all_teams == 1,
|
||||
user_id: user.into(),
|
||||
team_ids: teams.into_iter().map(str::to_owned).collect(),
|
||||
},
|
||||
|
|
@ -1951,3 +2114,55 @@ async fn agent_final_answer_preserves_visibility_and_trace_ownership(
|
|||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[rstest]
|
||||
#[tokio::test]
|
||||
async fn nullable_spend_upgrade_preserves_existing_costs_and_unknown_new_costs(
|
||||
#[future(awt)] database: TestResult<ClickHouseDatabase>,
|
||||
) -> TestResult {
|
||||
let database = database?;
|
||||
let writer = Connection::writer(&database.url)?;
|
||||
let timestamp = (time::OffsetDateTime::now_utc().unix_timestamp_nanos() / 1_000_000) as i64;
|
||||
let statements = schema_statements("trace_test", 7)?;
|
||||
for statement in &statements[..statements.len() - 1] {
|
||||
execute_write(&database, statement).await?;
|
||||
}
|
||||
let legacy = serde_json::from_value(serde_json::json!({
|
||||
"request_id": "legacy", "response_id": "legacy-response", "spend": 0.25,
|
||||
"start_time": timestamp, "end_time": timestamp + 100
|
||||
}))?;
|
||||
insert_rows(&database, "spend_logs", vec![legacy]).await?;
|
||||
ensure_schema(&database.client, &writer, "trace_test", 7).await?;
|
||||
ensure_schema(&database.client, &writer, "trace_test", 7).await?;
|
||||
let unknown = serde_json::from_value(serde_json::json!({
|
||||
"request_id": "unknown", "response_id": "unknown-response", "spend": null,
|
||||
"start_time": timestamp, "end_time": timestamp + 100
|
||||
}))?;
|
||||
let free = serde_json::from_value(serde_json::json!({
|
||||
"request_id": "free", "response_id": "free-response", "spend": 0.0,
|
||||
"start_time": timestamp, "end_time": timestamp + 100
|
||||
}))?;
|
||||
insert_rows(&database, "spend_logs", vec![unknown, free]).await?;
|
||||
let result = read_json(
|
||||
&database,
|
||||
"SELECT request_id, spend FROM trace_test.spend_logs FINAL ORDER BY request_id",
|
||||
)
|
||||
.await?;
|
||||
#[derive(Debug, serde::Deserialize)]
|
||||
struct CostRow {
|
||||
request_id: String,
|
||||
spend: Option<f64>,
|
||||
}
|
||||
let rows: Vec<CostRow> = serde_json::from_value(result["data"].clone())?;
|
||||
assert_eq!(
|
||||
rows.iter()
|
||||
.map(|row| (row.request_id.as_str(), row.spend))
|
||||
.collect::<Vec<_>>(),
|
||||
vec![
|
||||
("free", Some(0.0)),
|
||||
("legacy", Some(0.25)),
|
||||
("unknown", None)
|
||||
]
|
||||
);
|
||||
Ok(())
|
||||
}
|
||||
|
|
|
|||
|
|
@ -144,11 +144,11 @@ async fn typed_queries_read_normalized_spans_and_keep_trace_identities_separate(
|
|||
);
|
||||
assert_eq!(
|
||||
(
|
||||
spans[1].0.kind.as_str(),
|
||||
spans[1].0.kind,
|
||||
spans[1].0.input_tokens,
|
||||
spans[1].0.output_tokens
|
||||
),
|
||||
("llm", 12, 6)
|
||||
(litellm_traces::ObservationType::Llm, 12, 6)
|
||||
);
|
||||
assert_eq!(spans[2].0.status_message, "lookup timed out");
|
||||
Ok(())
|
||||
|
|
@ -225,7 +225,13 @@ async fn captured_deeplite_exports_round_trip_through_clickhouse(
|
|||
.filter(|span| span.parent_span_id.is_empty())
|
||||
.collect::<Vec<_>>();
|
||||
assert_eq!(roots.len(), 1);
|
||||
assert_eq!(traces[0].0.status, roots[0].status_code);
|
||||
assert_eq!(
|
||||
traces[0].0.status,
|
||||
serde_json::from_value::<litellm_traces::SpanStatus>(serde_json::json!(
|
||||
roots[0].status_code
|
||||
))
|
||||
.unwrap()
|
||||
);
|
||||
assert_eq!(
|
||||
traces[0].0.error_count,
|
||||
decoded
|
||||
|
|
@ -246,7 +252,13 @@ async fn captured_deeplite_exports_round_trip_through_clickhouse(
|
|||
assert_eq!(row.duration_ns, span.end_ns - span.start_ns);
|
||||
assert_eq!(row.input_tokens, span.normalized.input_tokens);
|
||||
assert_eq!(row.output_tokens, span.normalized.output_tokens);
|
||||
assert_eq!(row.status, span.status_code);
|
||||
assert_eq!(
|
||||
row.status,
|
||||
serde_json::from_value::<litellm_traces::SpanStatus>(serde_json::json!(
|
||||
span.status_code
|
||||
))
|
||||
.unwrap()
|
||||
);
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
|
|
|||
|
|
@ -139,11 +139,29 @@ fn span_row(span: &DecodedSpan, team: &str, key: &str) -> BTreeMap<String, Value
|
|||
"ObservationType".into(),
|
||||
json!(span.normalized.observation_type),
|
||||
),
|
||||
("AgentName".into(), json!(span.normalized.agent_name)),
|
||||
("Model".into(), json!(span.normalized.model)),
|
||||
(
|
||||
"AgentName".into(),
|
||||
json!(span.normalized.agent_name.as_deref().unwrap_or_default()),
|
||||
),
|
||||
(
|
||||
"Model".into(),
|
||||
json!(span.normalized.model.as_deref().unwrap_or_default()),
|
||||
),
|
||||
(
|
||||
"LiteLLMRequestId".into(),
|
||||
json!(span.normalized.litellm_request_id),
|
||||
json!(
|
||||
span.normalized
|
||||
.calls
|
||||
.key_set()
|
||||
.into_iter()
|
||||
.flatten()
|
||||
.find_map(|key| match key {
|
||||
litellm_traces::CallKey::LiteLlmRequest(id)
|
||||
| litellm_traces::CallKey::ProviderResponse(id) => Some(id.as_str()),
|
||||
litellm_traces::CallKey::Transport => None,
|
||||
})
|
||||
.unwrap_or_default()
|
||||
),
|
||||
),
|
||||
("InputTokens".into(), json!(span.normalized.input_tokens)),
|
||||
("OutputTokens".into(), json!(span.normalized.output_tokens)),
|
||||
|
|
|
|||
216
litellm-rust/crates/traces-clickhouse/tests/span_rows.rs
Normal file
216
litellm-rust/crates/traces-clickhouse/tests/span_rows.rs
Normal file
|
|
@ -0,0 +1,216 @@
|
|||
use litellm_traces::{Shared, Tenant, decode_otlp};
|
||||
use litellm_traces_clickhouse::{NORMALIZED_FIELD_DEFINITIONS, span_rows};
|
||||
use rstest::{fixture, rstest};
|
||||
use serde_json::{Value, json};
|
||||
|
||||
const MAX_VALUE_BYTES: usize = 64 * 1024;
|
||||
|
||||
#[fixture]
|
||||
fn tenant() -> Tenant {
|
||||
Tenant {
|
||||
team_id: "team-a".into(),
|
||||
api_key_hash: "key-a".into(),
|
||||
org_id: "org-a".into(),
|
||||
user_id: "user-a".into(),
|
||||
}
|
||||
}
|
||||
|
||||
fn attribute(key: &str, value: &str) -> Value {
|
||||
json!({"key": key, "value": {"stringValue": value}})
|
||||
}
|
||||
|
||||
fn span(span_id: &str, attributes: Vec<Value>, extra: Value) -> Value {
|
||||
let mut span = json!({
|
||||
"traceId": "01".repeat(16),
|
||||
"spanId": span_id,
|
||||
"name": "operation",
|
||||
"startTimeUnixNano": "1000",
|
||||
"endTimeUnixNano": "5000",
|
||||
"attributes": attributes,
|
||||
});
|
||||
span.as_object_mut()
|
||||
.unwrap()
|
||||
.extend(extra.as_object().unwrap().clone());
|
||||
span
|
||||
}
|
||||
|
||||
fn export(resources: Vec<(Vec<Value>, Vec<Value>)>) -> Vec<u8> {
|
||||
let resource_spans: Vec<Value> = resources
|
||||
.into_iter()
|
||||
.map(|(attributes, spans)| {
|
||||
json!({
|
||||
"resource": {"attributes": attributes},
|
||||
"scopeSpans": [{"scope": {"name": "scope", "version": "1"}, "spans": spans}],
|
||||
})
|
||||
})
|
||||
.collect();
|
||||
json!({"resourceSpans": resource_spans})
|
||||
.to_string()
|
||||
.into_bytes()
|
||||
}
|
||||
|
||||
fn rows(body: &[u8], tenant: &Tenant, max_value_bytes: usize) -> Vec<Value> {
|
||||
let spans = decode_otlp(body, Some("application/json")).unwrap();
|
||||
span_rows(spans, tenant, max_value_bytes)
|
||||
.iter()
|
||||
.map(|row| serde_json::to_value(row).unwrap())
|
||||
.collect()
|
||||
}
|
||||
|
||||
#[rstest]
|
||||
fn tenant_overwrites_claimed_identity_and_resources_stay_shared_per_group(tenant: Tenant) {
|
||||
let spoofed = vec![
|
||||
attribute("service.name", "svc"),
|
||||
attribute("litellm.team_id", "spoofed-team"),
|
||||
attribute("litellm.user_id", "spoofed-user"),
|
||||
];
|
||||
let body = export(vec![
|
||||
(
|
||||
spoofed.clone(),
|
||||
vec![
|
||||
span(&"02".repeat(8), vec![], json!({})),
|
||||
span(&"03".repeat(8), vec![], json!({})),
|
||||
],
|
||||
),
|
||||
(spoofed, vec![span(&"04".repeat(8), vec![], json!({}))]),
|
||||
]);
|
||||
let spans = decode_otlp(&body, Some("application/json")).unwrap();
|
||||
let stored = span_rows(spans, &tenant, MAX_VALUE_BYTES);
|
||||
let resource = |index: usize| &stored[index]["ResourceAttributes"];
|
||||
|
||||
assert!(Shared::shares_storage_with(resource(0), resource(1)));
|
||||
assert!(!Shared::shares_storage_with(resource(0), resource(2)));
|
||||
assert_eq!(resource(0), resource(2));
|
||||
assert_eq!(
|
||||
**resource(0),
|
||||
json!({
|
||||
"service.name": "svc",
|
||||
"litellm.team_id": "team-a",
|
||||
"litellm.user_id": "user-a",
|
||||
"litellm.api_key_hash": "key-a",
|
||||
"litellm.org_id": "org-a",
|
||||
})
|
||||
);
|
||||
for row in &stored {
|
||||
assert_eq!(
|
||||
(&*row["TeamId"], &*row["ApiKeyHash"], &*row["UserId"]),
|
||||
(&json!("team-a"), &json!("key-a"), &json!("user-a"))
|
||||
);
|
||||
assert_eq!(*row["ServiceName"], json!("svc"));
|
||||
}
|
||||
}
|
||||
|
||||
#[rstest]
|
||||
#[case::exception_event("", json!("customer acme-404 not found"))]
|
||||
#[case::status_message_wins("boom", json!("boom"))]
|
||||
fn status_message_falls_back_to_the_exception_event(
|
||||
tenant: Tenant,
|
||||
#[case] status_message: &str,
|
||||
#[case] expected: Value,
|
||||
) {
|
||||
let exported = span(
|
||||
&"02".repeat(8),
|
||||
vec![],
|
||||
json!({
|
||||
"status": {"code": 2, "message": status_message},
|
||||
"events": [{"name": "exception", "timeUnixNano": "2000", "attributes": [
|
||||
attribute("exception.type", "KeyError"),
|
||||
attribute("exception.message", "customer acme-404 not found"),
|
||||
]}],
|
||||
}),
|
||||
);
|
||||
let row = &rows(
|
||||
&export(vec![(vec![], vec![exported])]),
|
||||
&tenant,
|
||||
MAX_VALUE_BYTES,
|
||||
)[0];
|
||||
assert_eq!(row["StatusCode"], "STATUS_CODE_ERROR");
|
||||
assert_eq!(row["StatusMessage"], expected);
|
||||
}
|
||||
|
||||
#[rstest]
|
||||
fn consumed_payloads_leave_span_attributes_and_long_values_are_capped(tenant: Tenant) {
|
||||
let messages = json!([
|
||||
{"role": "system", "content": "be brief"},
|
||||
{"role": "user", "content": "x".repeat(300)},
|
||||
{"role": "user", "content": "latest question"},
|
||||
]);
|
||||
let exported = span(
|
||||
&"02".repeat(8),
|
||||
vec![
|
||||
attribute("gen_ai.operation.name", "chat"),
|
||||
attribute("gen_ai.input.messages", &messages.to_string()),
|
||||
attribute(
|
||||
"gen_ai.output.messages",
|
||||
&json!([{"role": "assistant", "content": "y".repeat(300)}]).to_string(),
|
||||
),
|
||||
attribute("custom.blob", &"z".repeat(300)),
|
||||
],
|
||||
json!({}),
|
||||
);
|
||||
let row = &rows(&export(vec![(vec![], vec![exported])]), &tenant, 200)[0];
|
||||
let attributes = row["SpanAttributes"].as_object().unwrap();
|
||||
assert!(!attributes.contains_key("gen_ai.input.messages"));
|
||||
assert!(!attributes.contains_key("gen_ai.output.messages"));
|
||||
assert_eq!(
|
||||
attributes["custom.blob"],
|
||||
format!("{}…[truncated 100 bytes]", "z".repeat(200))
|
||||
);
|
||||
let input = row["Input"].as_str().unwrap();
|
||||
let kept: Vec<Value> = serde_json::from_str(input).unwrap();
|
||||
assert!(input.len() <= 200);
|
||||
assert_eq!(kept[0]["content"], "be brief");
|
||||
assert_eq!(kept.last().unwrap()["content"], "latest question");
|
||||
assert!(row["Output"].as_str().unwrap().contains("…[truncated "));
|
||||
assert_eq!(row["ObservationType"], "llm");
|
||||
}
|
||||
|
||||
#[rstest]
|
||||
fn rows_carry_every_normalized_column(tenant: Tenant) {
|
||||
let row = &rows(
|
||||
&export(vec![(
|
||||
vec![],
|
||||
vec![span(&"02".repeat(8), vec![], json!({}))],
|
||||
)]),
|
||||
&tenant,
|
||||
MAX_VALUE_BYTES,
|
||||
)[0];
|
||||
for field in NORMALIZED_FIELD_DEFINITIONS {
|
||||
assert!(
|
||||
row.get(field.clickhouse_column).is_some(),
|
||||
"{}",
|
||||
field.clickhouse_column
|
||||
);
|
||||
}
|
||||
assert_eq!(row["Duration"], 4000);
|
||||
assert_eq!(row["AgentMetadata"], "{}");
|
||||
}
|
||||
|
||||
#[rstest]
|
||||
fn absent_identity_fields_are_empty_only_in_storage(tenant: Tenant) {
|
||||
let body = export(vec![(
|
||||
vec![],
|
||||
vec![span(&"02".repeat(8), vec![], json!({}))],
|
||||
)]);
|
||||
let decoded = decode_otlp(&body, Some("application/json")).unwrap();
|
||||
let normalized = &decoded[0].normalized;
|
||||
assert_eq!(normalized.agent_name, None);
|
||||
assert_eq!(normalized.framework, None);
|
||||
assert_eq!(normalized.model, None);
|
||||
assert_eq!(normalized.tool_call_id, None);
|
||||
let stored = span_rows(decoded, &tenant, MAX_VALUE_BYTES);
|
||||
let row = serde_json::to_value(&stored[0]).unwrap();
|
||||
assert_eq!(
|
||||
[
|
||||
"AgentName",
|
||||
"Framework",
|
||||
"Model",
|
||||
"ToolCallId",
|
||||
"LiteLLMRequestId"
|
||||
]
|
||||
.map(|column| row[column].clone()),
|
||||
[""; 5].map(|value| json!(value)),
|
||||
);
|
||||
assert_eq!(row["CallKeys"], json!([]));
|
||||
assert_eq!(row["CallEvidence"], "unknown");
|
||||
}
|
||||
|
|
@ -5,14 +5,22 @@ edition.workspace = true
|
|||
license.workspace = true
|
||||
repository.workspace = true
|
||||
|
||||
[features]
|
||||
schema = ["dep:schemars"]
|
||||
|
||||
[dependencies]
|
||||
askama.workspace = true
|
||||
macro_rules_attribute.workspace = true
|
||||
schemars = { workspace = true, optional = true }
|
||||
indexmap = { version = "2", features = ["serde"] }
|
||||
litellm-llms-types.workspace = true
|
||||
opentelemetry-proto = { workspace = true, features = ["gen-tonic-messages", "trace", "with-serde"] }
|
||||
prost.workspace = true
|
||||
serde = { workspace = true, features = ["rc"] }
|
||||
serde_json.workspace = true
|
||||
serde_json = { workspace = true, features = ["preserve_order"] }
|
||||
strum.workspace = true
|
||||
thiserror.workspace = true
|
||||
time.workspace = true
|
||||
|
||||
[dev-dependencies]
|
||||
criterion.workspace = true
|
||||
|
|
@ -21,3 +29,8 @@ rstest.workspace = true
|
|||
[[bench]]
|
||||
name = "resource-fanout"
|
||||
harness = false
|
||||
|
||||
[[bin]]
|
||||
name = "export-traces-schema"
|
||||
path = "src/bin/export_schema.rs"
|
||||
required-features = ["schema"]
|
||||
|
|
|
|||
6
litellm-rust/crates/traces/src/bin/export_schema.rs
Normal file
6
litellm-rust/crates/traces/src/bin/export_schema.rs
Normal file
|
|
@ -0,0 +1,6 @@
|
|||
fn main() {
|
||||
println!(
|
||||
"{}",
|
||||
serde_json::to_string_pretty(&litellm_traces::schema::schemas()).unwrap()
|
||||
);
|
||||
}
|
||||
|
|
@ -15,3 +15,7 @@ pub struct InvalidScope;
|
|||
#[derive(Debug, thiserror::Error)]
|
||||
#[error("unknown ClickHouse read query")]
|
||||
pub struct InvalidQuery;
|
||||
|
||||
#[derive(Debug, thiserror::Error)]
|
||||
#[error("invalid trace call key")]
|
||||
pub struct InvalidCallKey;
|
||||
|
|
|
|||
|
|
@ -1,13 +1,43 @@
|
|||
macro_rules_attribute::attribute_alias! {
|
||||
#[apply(wire_type)] =
|
||||
#[derive(serde::Serialize, serde::Deserialize)]
|
||||
#[cfg_attr(feature = "schema", derive(schemars::JsonSchema))];
|
||||
#[apply(response_type)] =
|
||||
#[derive(serde::Serialize)]
|
||||
#[cfg_attr(feature = "schema", derive(schemars::JsonSchema))];
|
||||
#[apply(request_type)] =
|
||||
#[derive(serde::Deserialize)]
|
||||
#[cfg_attr(feature = "schema", derive(schemars::JsonSchema))];
|
||||
}
|
||||
|
||||
mod error;
|
||||
mod normalize;
|
||||
mod otlp;
|
||||
pub mod query;
|
||||
mod query_access;
|
||||
mod resolve;
|
||||
#[cfg(feature = "schema")]
|
||||
pub mod schema;
|
||||
mod shared;
|
||||
mod tenant;
|
||||
mod truncate;
|
||||
mod ui;
|
||||
mod view;
|
||||
pub mod wire;
|
||||
|
||||
pub use error::{Error, InvalidQuery, InvalidScope};
|
||||
pub use normalize::{NormalizedSpan, ObservationType};
|
||||
pub use otlp::{DecodedSpan, decode_otlp};
|
||||
pub use error::{Error, InvalidCallKey, InvalidQuery, InvalidScope};
|
||||
pub use normalize::{
|
||||
AgentMetadata, AgentType, CallEvidence, CallEvidenceKind, CallKey, Integration, NormalizedSpan,
|
||||
ObservationType,
|
||||
};
|
||||
pub use otlp::{DecodedEvent, DecodedSpan, decode_otlp};
|
||||
pub use query::ReadQuery;
|
||||
pub use query_access::QueryScope;
|
||||
pub use resolve::{SpendLookup, iso_time, listed_summary, resolve_trace};
|
||||
pub use shared::{Shared, SharedIdentity};
|
||||
pub use tenant::Tenant;
|
||||
pub use truncate::{truncate_messages, truncate_value};
|
||||
pub use ui::{ChatRole, UiContent, UiField, UiMessage, UiToolCall, to_ui_content};
|
||||
pub use view::{
|
||||
AgentNode, Span, SpanDetail, SpanErrorPage, SpanStatus, Trace, TracePage, TraceSummary,
|
||||
};
|
||||
|
|
|
|||
7
litellm-rust/crates/traces/src/normalize/AGENTS.md
Normal file
7
litellm-rust/crates/traces/src/normalize/AGENTS.md
Normal file
|
|
@ -0,0 +1,7 @@
|
|||
- Normalize one decoded span at a time: `format/` extracts recorded facts, then `instrumentation/` applies SDK semantics
|
||||
- Own normalized span types, role and call evidence, shared message conversion in `messages.rs`, and metadata extraction in `metadata.rs`
|
||||
- Keep wire-format parsing in `format/` and SDK-specific interpretation in `instrumentation/`; share message helpers instead of duplicating payload parsing
|
||||
- Preserve format precedence, attribute alias precedence, token validation, and consumed-attribute tracking
|
||||
- Leave wrapper resolution, cross-span ownership, and spend attribution to `resolve/`; related spans can arrive in separate exports
|
||||
- Keep OTLP decoding in `otlp/`, storage in `traces-clickhouse`, and Python conversion in `python-bridge`
|
||||
- Test observable normalization through the public API in `tests/normalize.rs` and `tests/normalization_formats.rs`; keep private-helper tests inline
|
||||
11
litellm-rust/crates/traces/src/normalize/format/AGENTS.md
Normal file
11
litellm-rust/crates/traces/src/normalize/format/AGENTS.md
Normal file
|
|
@ -0,0 +1,11 @@
|
|||
- Read a span's recorded convention into `Extraction`: facts, an optional display name, and consumed attributes
|
||||
- Own convention detection, attribute aliases, payload shapes, model and token fields, tool-call IDs, and explicitly recorded roles
|
||||
- Preserve first-match format precedence in `mod.rs`, with GenAI as the fallback; use shared alias and token helpers from the parent module
|
||||
- Track the source attributes selected for payload extraction so normalization retains unconsumed data
|
||||
- Reuse `../messages.rs` for canonical messages, indexed attributes, and event payloads; keep SDK behavior in `../instrumentation/`
|
||||
- Leave cross-span wrapper resolution, ownership, and spend attribution to `resolve/`
|
||||
- Extend `tests/normalization_formats.rs` for parsing changes, including mixed conventions, fallbacks, and malformed payloads
|
||||
- Consult the convention specifications when changing mappings:
|
||||
- [OpenInference](https://github.com/Arize-ai/openinference/tree/main/spec)
|
||||
- [OpenTelemetry GenAI](https://opentelemetry.io/docs/specs/semconv/registry/attributes/gen-ai/index.md)
|
||||
- [LangSmith OTLP](https://docs.langchain.com/langsmith/trace-with-opentelemetry.md)
|
||||
|
|
@ -2,14 +2,18 @@ use std::collections::BTreeMap;
|
|||
|
||||
use serde_json::{Map, Value, json};
|
||||
|
||||
use super::{NormalizedSpan, ObservationType, SpanNormalizer, attr, first, tokens};
|
||||
use crate::{Error, otlp::DecodedEvent};
|
||||
use super::{Extraction, Format, SpanFacts};
|
||||
use crate::{
|
||||
Error,
|
||||
normalize::{
|
||||
CLAUDE_CODE_AGENT, CLAUDE_CODE_SCOPE, CallEvidence, CallKey, ObservationType, RoleEvidence,
|
||||
SpanContext, attr, present, tokens,
|
||||
},
|
||||
otlp::DecodedEvent,
|
||||
};
|
||||
|
||||
pub(crate) const CLAUDE_CODE_SCOPE: &str = "com.anthropic.claude_code.tracing";
|
||||
pub(crate) const CLAUDE_CODE_AGENT: &str = "claude-code";
|
||||
const AGENT_SDK_FRAMEWORK: &str = "claude-agent-sdk";
|
||||
|
||||
pub(super) struct ClaudeCodeNormalizer;
|
||||
/// Claude Code's built-in tracing, identified by its instrumentation scope.
|
||||
pub(crate) struct ClaudeCode;
|
||||
|
||||
enum SpanType {
|
||||
Interaction,
|
||||
|
|
@ -33,14 +37,13 @@ fn span_type(name: &str, attributes: &BTreeMap<String, String>) -> SpanType {
|
|||
}
|
||||
}
|
||||
|
||||
fn framework(attributes: &BTreeMap<String, String>) -> &'static str {
|
||||
if attr(attributes, "query_source_safe") == "sdk"
|
||||
|| attr(attributes, "system_prompt_preview").contains("cc_entrypoint=sdk")
|
||||
{
|
||||
AGENT_SDK_FRAMEWORK
|
||||
} else {
|
||||
CLAUDE_CODE_AGENT
|
||||
}
|
||||
/// `agent:custom:search_agent` -> `search_agent`: the subagent a request ran for.
|
||||
fn subagent(attributes: &BTreeMap<String, String>) -> Option<&str> {
|
||||
let mut parts = attr(attributes, "query_source")
|
||||
.strip_prefix("agent:")?
|
||||
.splitn(2, ':');
|
||||
let (_kind, name) = (parts.next()?, parts.next()?);
|
||||
(!name.is_empty()).then_some(name)
|
||||
}
|
||||
|
||||
fn split_header(text: &str) -> Option<(&str, &str)> {
|
||||
|
|
@ -154,68 +157,69 @@ fn input_tokens(attributes: &BTreeMap<String, String>) -> Result<u32, Error> {
|
|||
})
|
||||
}
|
||||
|
||||
impl SpanNormalizer for ClaudeCodeNormalizer {
|
||||
fn matches(&self, scope_name: &str, _attributes: &BTreeMap<String, String>) -> bool {
|
||||
scope_name == CLAUDE_CODE_SCOPE
|
||||
impl Format for ClaudeCode {
|
||||
fn matches(&self, context: &SpanContext<'_>) -> bool {
|
||||
context.scope == CLAUDE_CODE_SCOPE
|
||||
}
|
||||
|
||||
fn consumed_attributes(&self, attributes: &BTreeMap<String, String>) -> [&'static str; 2] {
|
||||
match span_type("", attributes) {
|
||||
SpanType::Interaction => ["user_prompt", ""],
|
||||
SpanType::LlmRequest => ["new_context", "response.model_output"],
|
||||
SpanType::Tool if tool_arguments(attributes).is_some() => ["tool_input", ""],
|
||||
SpanType::Tool | SpanType::Other => ["", ""],
|
||||
}
|
||||
}
|
||||
|
||||
fn display_name(&self, attributes: &BTreeMap<String, String>) -> Option<String> {
|
||||
let tool_name = attr(attributes, "tool_name");
|
||||
(matches!(span_type("", attributes), SpanType::Tool) && !tool_name.is_empty())
|
||||
.then(|| tool_name.to_owned())
|
||||
}
|
||||
|
||||
fn normalize(
|
||||
&self,
|
||||
name: &str,
|
||||
_parent_span_id: &str,
|
||||
attributes: &BTreeMap<String, String>,
|
||||
events: &[DecodedEvent],
|
||||
) -> Result<NormalizedSpan, Error> {
|
||||
let base = NormalizedSpan {
|
||||
observation_type: ObservationType::Framework,
|
||||
agent_name: CLAUDE_CODE_AGENT.to_owned(),
|
||||
framework: framework(attributes).to_owned(),
|
||||
litellm_request_id: String::new(),
|
||||
model: String::new(),
|
||||
input_tokens: 0,
|
||||
output_tokens: 0,
|
||||
input: String::new(),
|
||||
output: String::new(),
|
||||
fn extract(&self, context: &SpanContext<'_>) -> Result<Extraction, Error> {
|
||||
let attributes = context.attributes;
|
||||
let kind = span_type(context.name, attributes);
|
||||
let base = SpanFacts {
|
||||
role: Some(RoleEvidence::Declared(ObservationType::Framework)),
|
||||
agent_name: Some(CLAUDE_CODE_AGENT.to_owned()),
|
||||
tool_call_id: present(attributes, &["gen_ai.tool.call.id"]),
|
||||
..SpanFacts::default()
|
||||
};
|
||||
Ok(match span_type(name, attributes) {
|
||||
SpanType::Interaction => NormalizedSpan {
|
||||
observation_type: ObservationType::Agent,
|
||||
input: user_prompt(attributes),
|
||||
..base
|
||||
let (facts, consumed): (SpanFacts, Vec<&'static str>) = match kind {
|
||||
SpanType::Interaction => (
|
||||
SpanFacts {
|
||||
role: Some(RoleEvidence::Declared(ObservationType::Agent)),
|
||||
input: user_prompt(attributes),
|
||||
..base
|
||||
},
|
||||
vec!["user_prompt"],
|
||||
),
|
||||
SpanType::LlmRequest => (
|
||||
SpanFacts {
|
||||
role: Some(RoleEvidence::Declared(ObservationType::Llm)),
|
||||
agent_name: Some(subagent(attributes).unwrap_or(CLAUDE_CODE_AGENT).to_owned()),
|
||||
model: present(attributes, &["model", "gen_ai.request.model"]),
|
||||
input_tokens: input_tokens(attributes)?,
|
||||
output_tokens: tokens(attributes, "output_tokens")?,
|
||||
input: llm_input(attributes),
|
||||
output: llm_output(attributes),
|
||||
calls: present(attributes, &["gen_ai.response.id", "request_id"])
|
||||
.map_or(CallEvidence::Unknown, |id| {
|
||||
CallEvidence::complete(CallKey::ProviderResponse(id))
|
||||
}),
|
||||
..base
|
||||
},
|
||||
vec!["new_context", "response.model_output"],
|
||||
),
|
||||
SpanType::Tool => (
|
||||
SpanFacts {
|
||||
role: Some(RoleEvidence::Declared(ObservationType::Tool)),
|
||||
input: tool_input(attributes),
|
||||
output: tool_output(attributes, context.events),
|
||||
..base
|
||||
},
|
||||
if tool_arguments(attributes).is_some() {
|
||||
vec!["tool_input"]
|
||||
} else {
|
||||
Vec::new()
|
||||
},
|
||||
),
|
||||
SpanType::Other => (base, Vec::new()),
|
||||
};
|
||||
Ok(Extraction {
|
||||
facts,
|
||||
display_name: if matches!(kind, SpanType::Tool) {
|
||||
present(attributes, &["tool_name"])
|
||||
} else {
|
||||
None
|
||||
},
|
||||
SpanType::LlmRequest => NormalizedSpan {
|
||||
observation_type: ObservationType::Llm,
|
||||
litellm_request_id: first(attributes, "gen_ai.response.id", "request_id")
|
||||
.to_owned(),
|
||||
model: first(attributes, "model", "gen_ai.request.model").to_owned(),
|
||||
input_tokens: input_tokens(attributes)?,
|
||||
output_tokens: tokens(attributes, "output_tokens")?,
|
||||
input: llm_input(attributes),
|
||||
output: llm_output(attributes),
|
||||
..base
|
||||
},
|
||||
SpanType::Tool => NormalizedSpan {
|
||||
observation_type: ObservationType::Tool,
|
||||
input: tool_input(attributes),
|
||||
output: tool_output(attributes, events),
|
||||
..base
|
||||
},
|
||||
SpanType::Other => base,
|
||||
consumed_attributes: consumed,
|
||||
})
|
||||
}
|
||||
}
|
||||
|
|
@ -227,8 +231,35 @@ mod tests {
|
|||
use rstest::rstest;
|
||||
use serde_json::Value;
|
||||
|
||||
use super::{CLAUDE_CODE_SCOPE, ClaudeCodeNormalizer, SpanNormalizer};
|
||||
use crate::{Error, normalize::ObservationType, otlp::DecodedEvent};
|
||||
use super::CLAUDE_CODE_SCOPE;
|
||||
use crate::{
|
||||
Error,
|
||||
normalize::{Normalization, NormalizedSpan, ObservationType},
|
||||
otlp::DecodedEvent,
|
||||
};
|
||||
|
||||
fn normalization(
|
||||
name: &str,
|
||||
attributes: &BTreeMap<String, String>,
|
||||
events: &[DecodedEvent],
|
||||
) -> Result<Normalization, Error> {
|
||||
crate::normalize::normalize(&crate::normalize::SpanContext {
|
||||
scope: CLAUDE_CODE_SCOPE,
|
||||
name,
|
||||
parent_span_id: "parent",
|
||||
attributes,
|
||||
events,
|
||||
resource_attributes: &BTreeMap::new(),
|
||||
})
|
||||
}
|
||||
|
||||
fn normalize(
|
||||
name: &str,
|
||||
attributes: &BTreeMap<String, String>,
|
||||
events: &[DecodedEvent],
|
||||
) -> Result<NormalizedSpan, Error> {
|
||||
normalization(name, attributes, events).map(|normalization| normalization.span)
|
||||
}
|
||||
|
||||
fn attributes(pairs: &[(&str, &str)]) -> BTreeMap<String, String> {
|
||||
pairs
|
||||
|
|
@ -239,19 +270,17 @@ mod tests {
|
|||
|
||||
#[rstest]
|
||||
fn tool_without_detailed_input_lists_known_arguments() {
|
||||
let span = ClaudeCodeNormalizer
|
||||
.normalize(
|
||||
"claude_code.tool",
|
||||
"parent",
|
||||
&attributes(&[
|
||||
("span.type", "tool"),
|
||||
("tool_name", "Bash"),
|
||||
("full_command", "git status"),
|
||||
("bash_argv0", "git"),
|
||||
]),
|
||||
&[],
|
||||
)
|
||||
.expect("valid span");
|
||||
let span = normalize(
|
||||
"claude_code.tool",
|
||||
&attributes(&[
|
||||
("span.type", "tool"),
|
||||
("tool_name", "Bash"),
|
||||
("full_command", "git status"),
|
||||
("bash_argv0", "git"),
|
||||
]),
|
||||
&[],
|
||||
)
|
||||
.expect("valid span");
|
||||
let input: Value = serde_json::from_str(&span.input).expect("argument object");
|
||||
assert_eq!(input["command"], "git status");
|
||||
assert_eq!(input["bash_argv0"], "git");
|
||||
|
|
@ -266,14 +295,13 @@ mod tests {
|
|||
("tool_input", "[TOOL INPUT: Read]\nnot json"),
|
||||
("file_path", "/workspace/a.py"),
|
||||
]);
|
||||
let span = ClaudeCodeNormalizer
|
||||
.normalize("claude_code.tool", "parent", &attrs, &[])
|
||||
.expect("valid span");
|
||||
let span = normalize("claude_code.tool", &attrs, &[]).expect("valid span");
|
||||
let input: Value = serde_json::from_str(&span.input).expect("argument object");
|
||||
assert_eq!(input["file_path"], "/workspace/a.py");
|
||||
assert!(
|
||||
!ClaudeCodeNormalizer
|
||||
.consumed_attributes(&attrs)
|
||||
!normalization("claude_code.tool", &attrs, &[])
|
||||
.expect("valid span")
|
||||
.consumed_attributes
|
||||
.contains(&"tool_input")
|
||||
);
|
||||
}
|
||||
|
|
@ -295,45 +323,40 @@ mod tests {
|
|||
#[case] events: Vec<DecodedEvent>,
|
||||
#[case] expected: &str,
|
||||
) {
|
||||
let span = ClaudeCodeNormalizer
|
||||
.normalize(
|
||||
"claude_code.tool",
|
||||
"parent",
|
||||
&attributes(&[
|
||||
("span.type", "tool"),
|
||||
("new_context", "[TOOL RESULT: Bash]\n{\"stdout\":\"ctx\"}"),
|
||||
]),
|
||||
&events,
|
||||
)
|
||||
.expect("valid span");
|
||||
let span = normalize(
|
||||
"claude_code.tool",
|
||||
&attributes(&[
|
||||
("span.type", "tool"),
|
||||
("new_context", "[TOOL RESULT: Bash]\n{\"stdout\":\"ctx\"}"),
|
||||
]),
|
||||
&events,
|
||||
)
|
||||
.expect("valid span");
|
||||
assert_eq!(span.output, expected);
|
||||
}
|
||||
|
||||
#[rstest]
|
||||
fn llm_tool_result_context_becomes_tool_message() {
|
||||
let span = ClaudeCodeNormalizer
|
||||
.normalize(
|
||||
"claude_code.llm_request",
|
||||
"parent",
|
||||
&attributes(&[
|
||||
("span.type", "llm_request"),
|
||||
("new_context", "[TOOL RESULT: toolu_1]\n1\timport os"),
|
||||
]),
|
||||
&[],
|
||||
)
|
||||
.expect("valid span");
|
||||
let span = normalize(
|
||||
"claude_code.llm_request",
|
||||
&attributes(&[
|
||||
("span.type", "llm_request"),
|
||||
("new_context", "[TOOL RESULT: toolu_1]\n1\timport os"),
|
||||
]),
|
||||
&[],
|
||||
)
|
||||
.expect("valid span");
|
||||
let input: Value = serde_json::from_str(&span.input).expect("messages");
|
||||
assert_eq!(input[0]["role"], "tool");
|
||||
assert_eq!(input[0]["content"], "1\timport os");
|
||||
assert_eq!(span.output, "");
|
||||
assert_eq!(span.framework, "claude-code");
|
||||
assert_eq!(span.framework, Some(crate::Integration::ClaudeCode));
|
||||
}
|
||||
|
||||
#[rstest]
|
||||
fn llm_token_sum_overflow_is_rejected() {
|
||||
let result = ClaudeCodeNormalizer.normalize(
|
||||
let result = normalize(
|
||||
"claude_code.llm_request",
|
||||
"parent",
|
||||
&attributes(&[
|
||||
("span.type", "llm_request"),
|
||||
("input_tokens", "4294967295"),
|
||||
|
|
@ -358,10 +381,7 @@ mod tests {
|
|||
} else {
|
||||
attributes(&[("span.type", kind)])
|
||||
};
|
||||
let span = ClaudeCodeNormalizer
|
||||
.normalize(name, "parent", &attrs, &[])
|
||||
.expect("valid span");
|
||||
let span = normalize(name, &attrs, &[]).expect("valid span");
|
||||
assert_eq!(span.observation_type, expected);
|
||||
assert!(ClaudeCodeNormalizer.matches(CLAUDE_CODE_SCOPE, &attrs));
|
||||
}
|
||||
}
|
||||
132
litellm-rust/crates/traces/src/normalize/format/genai.rs
Normal file
132
litellm-rust/crates/traces/src/normalize/format/genai.rs
Normal file
|
|
@ -0,0 +1,132 @@
|
|||
use super::{Extraction, Format, Payload, SpanFacts};
|
||||
use crate::{
|
||||
Error,
|
||||
normalize::{
|
||||
ObservationType, RoleEvidence, SpanContext, attr, messages, present, select_attribute,
|
||||
usage_tokens,
|
||||
},
|
||||
};
|
||||
|
||||
/// OpenTelemetry GenAI semantic conventions: the fallback, since any span may carry `gen_ai.*`.
|
||||
pub(crate) struct GenAi;
|
||||
|
||||
#[derive(strum::EnumString)]
|
||||
#[strum(serialize_all = "snake_case")]
|
||||
pub(crate) enum Operation {
|
||||
CreateAgent,
|
||||
InvokeAgent,
|
||||
InvokeWorkflow,
|
||||
Chat,
|
||||
#[strum(serialize = "text_completion", serialize = "completion")]
|
||||
TextCompletion,
|
||||
GenerateContent,
|
||||
ExecuteTool,
|
||||
#[strum(serialize = "embeddings", serialize = "embedding")]
|
||||
Embeddings,
|
||||
Retrieval,
|
||||
}
|
||||
|
||||
impl Operation {
|
||||
pub(crate) fn from_context(context: &SpanContext<'_>) -> Option<Self> {
|
||||
Self::try_from(attr(context.attributes, "gen_ai.operation.name")).ok()
|
||||
}
|
||||
|
||||
fn role(self) -> ObservationType {
|
||||
match self {
|
||||
Self::InvokeAgent => ObservationType::Agent,
|
||||
Self::CreateAgent => ObservationType::Framework,
|
||||
Self::InvokeWorkflow => ObservationType::Chain,
|
||||
Self::Chat | Self::TextCompletion | Self::GenerateContent => ObservationType::Llm,
|
||||
Self::ExecuteTool => ObservationType::Tool,
|
||||
Self::Embeddings => ObservationType::Embedding,
|
||||
Self::Retrieval => ObservationType::Retriever,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
const INPUT_KEYS: [&str; 4] = [
|
||||
"gen_ai.input.messages",
|
||||
"gen_ai.tool.call.arguments",
|
||||
"gen_ai.retrieval.query.text",
|
||||
"gen_ai.prompt",
|
||||
];
|
||||
|
||||
const OUTPUT_KEYS: [&str; 4] = [
|
||||
"gen_ai.output.messages",
|
||||
"gen_ai.tool.call.result",
|
||||
"gen_ai.retrieval.documents",
|
||||
"gen_ai.completion",
|
||||
];
|
||||
|
||||
/// The messages key comes first and is put in the common format; other payloads stay as recorded.
|
||||
fn payload(context: &SpanContext<'_>, keys: &[&'static str]) -> Payload {
|
||||
let Some(attribute) = select_attribute(context.attributes, keys) else {
|
||||
let prefix = if keys[0] == INPUT_KEYS[0] {
|
||||
"gen_ai.prompt"
|
||||
} else {
|
||||
"gen_ai.completion"
|
||||
};
|
||||
let indexed = messages::indexed(context.attributes, prefix);
|
||||
return Payload {
|
||||
text: indexed
|
||||
.or_else(|| {
|
||||
let events: Vec<_> = context
|
||||
.events
|
||||
.iter()
|
||||
.filter_map(|event| {
|
||||
let encoded = attr(&event.attributes, "gen_ai.event.content");
|
||||
let value = serde_json::from_str(encoded).unwrap_or_else(|_| {
|
||||
serde_json::to_value(&event.attributes).unwrap_or_default()
|
||||
});
|
||||
messages::event_message(&event.name, &value)
|
||||
})
|
||||
.collect();
|
||||
messages::event_payload(&events, keys[0] == OUTPUT_KEYS[0])
|
||||
})
|
||||
.unwrap_or_default(),
|
||||
consumed: None,
|
||||
};
|
||||
};
|
||||
Payload {
|
||||
text: if attribute.source == keys[0] {
|
||||
messages::canonical(attribute.text)
|
||||
} else {
|
||||
attribute.text.to_owned()
|
||||
},
|
||||
consumed: Some(attribute.source),
|
||||
}
|
||||
}
|
||||
|
||||
impl Format for GenAi {
|
||||
fn matches(&self, _context: &SpanContext<'_>) -> bool {
|
||||
true
|
||||
}
|
||||
|
||||
fn extract(&self, context: &SpanContext<'_>) -> Result<Extraction, Error> {
|
||||
let attributes = context.attributes;
|
||||
let (input_tokens, output_tokens) = usage_tokens(attributes)?;
|
||||
let input = payload(context, &INPUT_KEYS);
|
||||
let output = payload(context, &OUTPUT_KEYS);
|
||||
Ok(Extraction {
|
||||
facts: SpanFacts {
|
||||
role: Operation::from_context(context)
|
||||
.map(|operation| RoleEvidence::Declared(operation.role())),
|
||||
model: present(
|
||||
attributes,
|
||||
&["gen_ai.request.model", "gen_ai.response.model"],
|
||||
),
|
||||
input_tokens,
|
||||
output_tokens,
|
||||
input: input.text,
|
||||
output: output.text,
|
||||
tool_call_id: present(attributes, &["gen_ai.tool.call.id"]),
|
||||
..SpanFacts::default()
|
||||
},
|
||||
display_name: None,
|
||||
consumed_attributes: [input.consumed, output.consumed]
|
||||
.into_iter()
|
||||
.flatten()
|
||||
.collect(),
|
||||
})
|
||||
}
|
||||
}
|
||||
267
litellm-rust/crates/traces/src/normalize/format/langsmith.rs
Normal file
267
litellm-rust/crates/traces/src/normalize/format/langsmith.rs
Normal file
|
|
@ -0,0 +1,267 @@
|
|||
use std::collections::BTreeMap;
|
||||
|
||||
use serde::{
|
||||
Deserialize, Deserializer,
|
||||
de::{DeserializeOwned, IgnoredAny},
|
||||
};
|
||||
use serde_json::Value;
|
||||
|
||||
use super::{Extraction, Format, SpanFacts, genai::GenAi};
|
||||
use crate::{
|
||||
Error,
|
||||
normalize::{
|
||||
CallEvidence, ObservationType, RoleEvidence, SpanContext, attr,
|
||||
messages::{RawMessage, encode, langchain_result},
|
||||
},
|
||||
};
|
||||
|
||||
/// LangSmith's OpenTelemetry exporter: spans carry `langsmith.span.kind`.
|
||||
pub(crate) struct LangSmith;
|
||||
|
||||
enum MessageBatch {
|
||||
Flat(Vec<RawMessage>),
|
||||
Nested(Vec<Vec<RawMessage>>),
|
||||
}
|
||||
|
||||
impl<'de> Deserialize<'de> for MessageBatch {
|
||||
fn deserialize<D: Deserializer<'de>>(deserializer: D) -> Result<Self, D::Error> {
|
||||
let value = Value::deserialize(deserializer)?;
|
||||
let Value::Array(items) = value else {
|
||||
return Err(serde::de::Error::custom("messages must be an array"));
|
||||
};
|
||||
let parse = |items: Vec<Value>| {
|
||||
items
|
||||
.into_iter()
|
||||
.filter_map(|item| serde_json::from_value(item).ok())
|
||||
.collect()
|
||||
};
|
||||
Ok(if items.first().is_some_and(Value::is_array) {
|
||||
Self::Nested(
|
||||
items
|
||||
.into_iter()
|
||||
.filter_map(|item| item.as_array().cloned())
|
||||
.map(parse)
|
||||
.collect(),
|
||||
)
|
||||
} else {
|
||||
Self::Flat(parse(items))
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
fn lenient<'de, D: Deserializer<'de>, T: DeserializeOwned>(
|
||||
deserializer: D,
|
||||
) -> Result<Option<T>, D::Error> {
|
||||
let value = Value::deserialize(deserializer)?;
|
||||
Ok(serde_json::from_value(value).ok())
|
||||
}
|
||||
|
||||
impl MessageBatch {
|
||||
fn first_batch(&self) -> &[RawMessage] {
|
||||
match self {
|
||||
Self::Flat(messages) => messages,
|
||||
Self::Nested(batches) => batches.first().map(Vec::as_slice).unwrap_or_default(),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Default, Deserialize)]
|
||||
struct Payload {
|
||||
#[serde(default, deserialize_with = "lenient")]
|
||||
messages: Option<MessageBatch>,
|
||||
}
|
||||
|
||||
#[derive(Deserialize)]
|
||||
struct Command {
|
||||
update: CommandUpdate,
|
||||
}
|
||||
|
||||
#[derive(Deserialize)]
|
||||
struct CommandUpdate {
|
||||
messages: Vec<Value>,
|
||||
}
|
||||
|
||||
#[derive(Deserialize)]
|
||||
struct ContentValue {
|
||||
content: Value,
|
||||
}
|
||||
|
||||
#[derive(Deserialize)]
|
||||
struct WrappedOutput {
|
||||
output: Value,
|
||||
#[serde(flatten)]
|
||||
_other: BTreeMap<String, IgnoredAny>,
|
||||
}
|
||||
|
||||
struct SpanIo {
|
||||
input: String,
|
||||
output: String,
|
||||
calls: CallEvidence,
|
||||
}
|
||||
|
||||
fn normalized_messages(messages: &[RawMessage]) -> String {
|
||||
encode(
|
||||
&messages
|
||||
.iter()
|
||||
.map(RawMessage::normalized)
|
||||
.collect::<Vec<_>>(),
|
||||
)
|
||||
}
|
||||
|
||||
fn tool_output(raw_completion: &str) -> String {
|
||||
let completion = serde_json::from_str::<Value>(raw_completion).unwrap_or(Value::Null);
|
||||
let raw = WrappedOutput::deserialize(&completion)
|
||||
.map(|wrapped| wrapped.output)
|
||||
.unwrap_or(completion);
|
||||
let selected = Command::deserialize(&raw)
|
||||
.ok()
|
||||
.and_then(|command| command.update.messages.into_iter().last())
|
||||
.unwrap_or(raw);
|
||||
let output = ContentValue::deserialize(&selected)
|
||||
.map(|message| message.content)
|
||||
.unwrap_or(selected);
|
||||
output
|
||||
.as_str()
|
||||
.map(str::to_owned)
|
||||
.unwrap_or_else(|| encode(&output))
|
||||
}
|
||||
|
||||
fn span_io(kind: ObservationType, attributes: &BTreeMap<String, String>) -> SpanIo {
|
||||
let raw_prompt = attr(attributes, "gen_ai.prompt");
|
||||
let raw_completion = attr(attributes, "gen_ai.completion");
|
||||
let prompt = serde_json::from_str::<Payload>(raw_prompt).unwrap_or_default();
|
||||
if kind == ObservationType::Llm
|
||||
&& serde_json::from_str::<Value>(raw_completion).is_ok_and(|value| value.is_object())
|
||||
{
|
||||
let input = prompt.messages.as_ref().map_or_else(
|
||||
|| "[]".to_owned(),
|
||||
|messages| normalized_messages(messages.first_batch()),
|
||||
);
|
||||
let result = serde_json::from_str::<Value>(raw_completion)
|
||||
.ok()
|
||||
.and_then(|value| langchain_result(&value));
|
||||
return match result {
|
||||
Some(result) if result.first.is_some() => SpanIo {
|
||||
input,
|
||||
output: result.first.as_ref().map(encode).unwrap_or_default(),
|
||||
calls: result.calls,
|
||||
},
|
||||
_ => SpanIo {
|
||||
input,
|
||||
output: raw_completion.to_owned(),
|
||||
calls: result.map_or(CallEvidence::Unknown, |result| result.calls),
|
||||
},
|
||||
};
|
||||
}
|
||||
if kind == ObservationType::Tool {
|
||||
return SpanIo {
|
||||
input: raw_prompt.to_owned(),
|
||||
output: tool_output(raw_completion),
|
||||
calls: CallEvidence::Unknown,
|
||||
};
|
||||
}
|
||||
|
||||
SpanIo {
|
||||
input: raw_prompt.to_owned(),
|
||||
output: raw_completion.to_owned(),
|
||||
calls: CallEvidence::Unknown,
|
||||
}
|
||||
}
|
||||
|
||||
impl Format for LangSmith {
|
||||
fn matches(&self, context: &SpanContext<'_>) -> bool {
|
||||
context.scope == "langsmith" || context.attributes.contains_key("langsmith.span.kind")
|
||||
}
|
||||
|
||||
fn extract(&self, context: &SpanContext<'_>) -> Result<Extraction, Error> {
|
||||
let attributes = context.attributes;
|
||||
let base = GenAi.extract(context)?;
|
||||
let observation_type = ObservationType::try_from(attr(attributes, "langsmith.span.kind"))
|
||||
.unwrap_or(ObservationType::Chain);
|
||||
let io = span_io(observation_type, attributes);
|
||||
Ok(Extraction {
|
||||
facts: SpanFacts {
|
||||
role: Some(RoleEvidence::Declared(observation_type)),
|
||||
input: if attr(attributes, "gen_ai.prompt").is_empty() {
|
||||
String::new()
|
||||
} else {
|
||||
io.input
|
||||
},
|
||||
output: if attr(attributes, "gen_ai.completion").is_empty() {
|
||||
String::new()
|
||||
} else {
|
||||
io.output
|
||||
},
|
||||
calls: io.calls,
|
||||
..SpanFacts::default()
|
||||
}
|
||||
.or(base.facts),
|
||||
display_name: None,
|
||||
consumed_attributes: base.consumed_attributes,
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use std::collections::BTreeMap;
|
||||
|
||||
use rstest::rstest;
|
||||
use serde_json::{Value, json};
|
||||
|
||||
use super::{CallEvidence, ObservationType, span_io};
|
||||
use crate::normalize::CallKey;
|
||||
|
||||
#[rstest]
|
||||
fn malformed_messages_preserve_valid_input_and_response_id() {
|
||||
let attributes = BTreeMap::from([
|
||||
(
|
||||
"gen_ai.prompt".to_owned(),
|
||||
r#"{"messages":[[{"kwargs":{"type":"human","content":"hello"}},null]]}"#.to_owned(),
|
||||
),
|
||||
(
|
||||
"gen_ai.completion".to_owned(),
|
||||
r#"{"messages":"unexpected","generations":[[{"message":{"kwargs":{"type":"ai","content":"hi","response_metadata":{"id":"response-1"}}}}]]}"#.to_owned(),
|
||||
),
|
||||
]);
|
||||
let io = span_io(ObservationType::Llm, &attributes);
|
||||
let input: Value = serde_json::from_str(&io.input).expect("normalized input");
|
||||
assert_eq!(input.as_array().expect("messages").len(), 1);
|
||||
assert_eq!(input[0]["content"], "hello");
|
||||
assert_eq!(
|
||||
io.calls,
|
||||
CallEvidence::complete(CallKey::ProviderResponse("response-1".to_owned()))
|
||||
);
|
||||
}
|
||||
|
||||
#[rstest]
|
||||
#[case::null(r#"{"output":null}"#, Value::Null)]
|
||||
#[case::string(r#""answer""#, json!("answer"))]
|
||||
#[case::wrapped_string(r#"{"output":"answer","other":7}"#, json!("answer"))]
|
||||
#[case::repeated_output(r#"{"output":"first","output":"last"}"#, json!("last"))]
|
||||
#[case::wrapped_content(r#"{"output":{"content":"answer"}}"#, json!("answer"))]
|
||||
#[case::last_command_message(r#"{"output":{"update":{"messages":[{"content":"first"},{"content":"last"}]}}}"#, json!("last"))]
|
||||
#[case::direct_command(r#"{"update":{"messages":[{"content":"answer"}]}}"#, json!("answer"))]
|
||||
#[case::empty_command(r#"{"update":{"messages":[]}}"#, json!({"update":{"messages":[]}}))]
|
||||
#[case::arbitrary_object(r#"{"result":7}"#, json!({"result":7}))]
|
||||
#[case::arbitrary_array(r#"[1,2]"#, json!([1,2]))]
|
||||
#[case::malformed("not-json", Value::Null)]
|
||||
fn tool_outputs_preserve_content_and_fallbacks(
|
||||
#[case] completion: &str,
|
||||
#[case] expected: Value,
|
||||
) {
|
||||
let attributes = BTreeMap::from([("gen_ai.completion".to_owned(), completion.to_owned())]);
|
||||
let io = span_io(ObservationType::Tool, &attributes);
|
||||
match expected {
|
||||
Value::String(text) => assert_eq!(io.output, text),
|
||||
value => assert_eq!(serde_json::from_str::<Value>(&io.output).unwrap(), value),
|
||||
}
|
||||
}
|
||||
|
||||
#[rstest]
|
||||
fn absent_llm_messages_render_as_an_empty_list() {
|
||||
let attributes = BTreeMap::from([("gen_ai.completion".to_owned(), "{}".to_owned())]);
|
||||
let io = span_io(ObservationType::Llm, &attributes);
|
||||
assert_eq!(io.input, "[]");
|
||||
}
|
||||
}
|
||||
64
litellm-rust/crates/traces/src/normalize/format/logfire.rs
Normal file
64
litellm-rust/crates/traces/src/normalize/format/logfire.rs
Normal file
|
|
@ -0,0 +1,64 @@
|
|||
use serde::Deserialize;
|
||||
use serde_json::Value;
|
||||
|
||||
use super::{Extraction, Format, SpanFacts, genai::GenAi};
|
||||
use crate::{
|
||||
Error,
|
||||
normalize::{SpanContext, messages, select_attribute},
|
||||
};
|
||||
|
||||
pub(crate) struct Logfire;
|
||||
|
||||
impl Format for Logfire {
|
||||
fn matches(&self, context: &SpanContext<'_>) -> bool {
|
||||
context.attributes.contains_key("all_messages_events")
|
||||
|| ((context.scope.starts_with("logfire") || context.scope == "pydantic-ai")
|
||||
&& (context.attributes.contains_key("events")
|
||||
|| context.attributes.contains_key("prompt")))
|
||||
}
|
||||
|
||||
fn extract(&self, context: &SpanContext<'_>) -> Result<Extraction, Error> {
|
||||
let base = GenAi.extract(context)?;
|
||||
let input = base
|
||||
.facts
|
||||
.input
|
||||
.is_empty()
|
||||
.then(|| select_attribute(context.attributes, &["prompt"]))
|
||||
.flatten();
|
||||
let output = base
|
||||
.facts
|
||||
.output
|
||||
.is_empty()
|
||||
.then(|| select_attribute(context.attributes, &["final_result"]))
|
||||
.flatten();
|
||||
let recorded = select_attribute(context.attributes, &["all_messages_events", "events"]);
|
||||
let values = recorded
|
||||
.as_ref()
|
||||
.and_then(|value| serde_json::from_str::<Vec<Value>>(value.text).ok())
|
||||
.unwrap_or_default();
|
||||
let events: Vec<_> = values
|
||||
.iter()
|
||||
.filter_map(|value| messages::EventMessage::deserialize(value).ok()?.recorded())
|
||||
.collect();
|
||||
Ok(Extraction {
|
||||
facts: base.facts.or(SpanFacts {
|
||||
input: input
|
||||
.as_ref()
|
||||
.map(|value| messages::canonical(value.text))
|
||||
.or_else(|| messages::event_payload(&events, false))
|
||||
.unwrap_or_default(),
|
||||
output: output
|
||||
.as_ref()
|
||||
.map(|value| value.text.to_owned())
|
||||
.or_else(|| messages::event_payload(&events, true))
|
||||
.unwrap_or_default(),
|
||||
..SpanFacts::default()
|
||||
}),
|
||||
display_name: base.display_name,
|
||||
consumed_attributes: base.consumed_attributes,
|
||||
}
|
||||
.consuming(input)
|
||||
.consuming(output)
|
||||
.consuming(recorded))
|
||||
}
|
||||
}
|
||||
133
litellm-rust/crates/traces/src/normalize/format/mod.rs
Normal file
133
litellm-rust/crates/traces/src/normalize/format/mod.rs
Normal file
|
|
@ -0,0 +1,133 @@
|
|||
//! Step one of normalization: what a span records, read in the format it was recorded in.
|
||||
|
||||
use super::{AttributeText, CallEvidence, RoleEvidence, SpanContext};
|
||||
use crate::Error;
|
||||
|
||||
pub(crate) mod claude_code;
|
||||
pub(crate) mod genai;
|
||||
pub(crate) mod langsmith;
|
||||
pub(crate) mod logfire;
|
||||
pub(crate) mod openinference;
|
||||
pub(crate) mod traceloop;
|
||||
pub(crate) mod vercel;
|
||||
|
||||
/// What a span records, read in its convention's format.
|
||||
#[derive(Debug, Default)]
|
||||
pub(crate) struct SpanFacts {
|
||||
pub role: Option<RoleEvidence>,
|
||||
pub agent_name: Option<String>,
|
||||
pub model: Option<String>,
|
||||
pub input_tokens: u32,
|
||||
pub output_tokens: u32,
|
||||
pub input: String,
|
||||
pub output: String,
|
||||
pub tool_call_id: Option<String>,
|
||||
pub calls: CallEvidence,
|
||||
/// Set when the latest user message is not simply read from `input`.
|
||||
pub input_preview: Option<String>,
|
||||
}
|
||||
|
||||
impl SpanFacts {
|
||||
pub(crate) fn or(self, fallback: Self) -> Self {
|
||||
Self {
|
||||
role: self.role.or(fallback.role),
|
||||
agent_name: self.agent_name.or(fallback.agent_name),
|
||||
model: self.model.or(fallback.model),
|
||||
input_tokens: if self.input_tokens == 0 {
|
||||
fallback.input_tokens
|
||||
} else {
|
||||
self.input_tokens
|
||||
},
|
||||
output_tokens: if self.output_tokens == 0 {
|
||||
fallback.output_tokens
|
||||
} else {
|
||||
self.output_tokens
|
||||
},
|
||||
input: if self.input.is_empty() {
|
||||
fallback.input
|
||||
} else {
|
||||
self.input
|
||||
},
|
||||
output: if self.output.is_empty() {
|
||||
fallback.output
|
||||
} else {
|
||||
self.output
|
||||
},
|
||||
tool_call_id: self.tool_call_id.or(fallback.tool_call_id),
|
||||
calls: if self.calls == CallEvidence::Unknown {
|
||||
fallback.calls
|
||||
} else {
|
||||
self.calls
|
||||
},
|
||||
input_preview: self.input_preview.or(fallback.input_preview),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// A convention's complete reading of a span, including which attributes it consumed.
|
||||
pub(crate) struct Extraction {
|
||||
pub facts: SpanFacts,
|
||||
pub display_name: Option<String>,
|
||||
pub consumed_attributes: Vec<&'static str>,
|
||||
}
|
||||
|
||||
impl Extraction {
|
||||
pub(crate) fn consuming(self, attribute: Option<AttributeText<'_>>) -> Self {
|
||||
Self {
|
||||
consumed_attributes: self
|
||||
.consumed_attributes
|
||||
.into_iter()
|
||||
.chain(attribute.map(|value| value.source))
|
||||
.collect(),
|
||||
..self
|
||||
}
|
||||
}
|
||||
|
||||
pub(crate) fn map_facts(self, adjust: impl FnOnce(SpanFacts) -> SpanFacts) -> Self {
|
||||
Self {
|
||||
facts: adjust(self.facts),
|
||||
..self
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// A payload read from one attribute, which the extraction then reports as consumed.
|
||||
#[derive(Default)]
|
||||
pub(crate) struct Payload {
|
||||
pub text: String,
|
||||
pub consumed: Option<&'static str>,
|
||||
}
|
||||
|
||||
impl From<AttributeText<'_>> for Payload {
|
||||
fn from(attribute: AttributeText<'_>) -> Self {
|
||||
Self {
|
||||
text: attribute.text.to_owned(),
|
||||
consumed: Some(attribute.source),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// A span format: whether a span is recorded in it, and what the span then records.
|
||||
pub(crate) trait Format {
|
||||
fn matches(&self, context: &SpanContext<'_>) -> bool;
|
||||
fn extract(&self, context: &SpanContext<'_>) -> Result<Extraction, Error>;
|
||||
}
|
||||
|
||||
/// In precedence order. `gen_ai` accepts every span, so it is last.
|
||||
const FORMATS: [&dyn Format; 7] = [
|
||||
&claude_code::ClaudeCode,
|
||||
&langsmith::LangSmith,
|
||||
&openinference::OpenInference,
|
||||
&traceloop::Traceloop,
|
||||
&vercel::Vercel,
|
||||
&logfire::Logfire,
|
||||
&genai::GenAi,
|
||||
];
|
||||
|
||||
pub(crate) fn extract(context: &SpanContext<'_>) -> Result<Extraction, Error> {
|
||||
FORMATS
|
||||
.into_iter()
|
||||
.find(|format| format.matches(context))
|
||||
.unwrap_or(&genai::GenAi)
|
||||
.extract(context)
|
||||
}
|
||||
128
litellm-rust/crates/traces/src/normalize/format/openinference.rs
Normal file
128
litellm-rust/crates/traces/src/normalize/format/openinference.rs
Normal file
|
|
@ -0,0 +1,128 @@
|
|||
use std::collections::BTreeMap;
|
||||
|
||||
use litellm_llms_types::recognized::Recognized;
|
||||
use serde::{Deserialize, de::IgnoredAny};
|
||||
use serde_json::Value;
|
||||
|
||||
use super::{Extraction, Format, Payload, SpanFacts};
|
||||
use crate::{
|
||||
Error,
|
||||
normalize::{
|
||||
CallEvidence, CallKey, ObservationType, RoleEvidence, SpanContext, attr, messages, present,
|
||||
select_attribute, tokens, usage_tokens,
|
||||
},
|
||||
};
|
||||
|
||||
/// Arize OpenInference: spans carry `openinference.span.kind`.
|
||||
pub(crate) struct OpenInference;
|
||||
|
||||
#[derive(Deserialize)]
|
||||
struct ResponseIdentity {
|
||||
#[serde(default, deserialize_with = "messages::present")]
|
||||
id: Option<Recognized<String>>,
|
||||
#[serde(flatten)]
|
||||
_other: BTreeMap<String, IgnoredAny>,
|
||||
}
|
||||
|
||||
#[derive(Deserialize)]
|
||||
struct ProviderResponse {
|
||||
raw: Option<Recognized<ResponseIdentity>>,
|
||||
#[serde(flatten)]
|
||||
response: ResponseIdentity,
|
||||
}
|
||||
|
||||
impl ProviderResponse {
|
||||
fn id(&self) -> Option<&str> {
|
||||
let identity = match &self.response.id {
|
||||
Some(id) => return id.known().map(String::as_str),
|
||||
None => self.raw.as_ref()?.known()?,
|
||||
};
|
||||
identity.id.as_ref()?.known().map(String::as_str)
|
||||
}
|
||||
}
|
||||
|
||||
fn role(context: &SpanContext<'_>) -> Option<RoleEvidence> {
|
||||
let root = context.parent_span_id.is_empty();
|
||||
match ObservationType::try_from(attr(context.attributes, "openinference.span.kind")) {
|
||||
// A root chain (crew kickoff, workflow run) may be the agent run or only wrap its agents.
|
||||
Ok(ObservationType::Chain) if root => {
|
||||
Some(RoleEvidence::WrapperCandidate(ObservationType::Agent))
|
||||
}
|
||||
Ok(kind) => Some(RoleEvidence::Declared(kind)),
|
||||
_ if root => None,
|
||||
_ => Some(RoleEvidence::Declared(ObservationType::Chain)),
|
||||
}
|
||||
}
|
||||
|
||||
/// LLM instrumentations record the provider response as `output.value`: a raw response is one
|
||||
/// request (`id`); a LangChain `LLMResult` carries one per prompt.
|
||||
fn calls(output: &str) -> CallEvidence {
|
||||
let Ok(value) = serde_json::from_str::<Value>(output) else {
|
||||
return CallEvidence::Unknown;
|
||||
};
|
||||
if let Ok(response) = ProviderResponse::deserialize(&value)
|
||||
&& let Some(id) = response.id()
|
||||
{
|
||||
return CallEvidence::complete(CallKey::ProviderResponse(id.to_owned()));
|
||||
}
|
||||
messages::langchain_result(&value).map_or(CallEvidence::Unknown, |result| result.calls)
|
||||
}
|
||||
|
||||
/// `llm.<direction>_messages.*` when the instrumentation flattened the messages, else `raw`.
|
||||
fn payload(context: &SpanContext<'_>, flattened: &str, raw: &'static str) -> Payload {
|
||||
if let Some(conversation) = messages::flattened(context.attributes, flattened) {
|
||||
return Payload {
|
||||
text: messages::encode(&conversation),
|
||||
consumed: None,
|
||||
};
|
||||
}
|
||||
select_attribute(context.attributes, &[raw])
|
||||
.map(Payload::from)
|
||||
.unwrap_or_default()
|
||||
}
|
||||
|
||||
/// OpenInference's own count when recorded, else the `gen_ai.usage.*` one.
|
||||
fn token_count(attributes: &BTreeMap<String, String>, key: &str, usage: u32) -> Result<u32, Error> {
|
||||
if attributes.contains_key(key) {
|
||||
tokens(attributes, key)
|
||||
} else {
|
||||
Ok(usage)
|
||||
}
|
||||
}
|
||||
|
||||
impl Format for OpenInference {
|
||||
fn matches(&self, context: &SpanContext<'_>) -> bool {
|
||||
context.attributes.contains_key("openinference.span.kind")
|
||||
}
|
||||
|
||||
fn extract(&self, context: &SpanContext<'_>) -> Result<Extraction, Error> {
|
||||
let attributes = context.attributes;
|
||||
let (usage_input, usage_output) = usage_tokens(attributes)?;
|
||||
let role = role(context);
|
||||
let input = payload(context, "llm.input_messages", "input.value");
|
||||
let output = payload(context, "llm.output_messages", "output.value");
|
||||
Ok(Extraction {
|
||||
facts: SpanFacts {
|
||||
role,
|
||||
agent_name: present(attributes, &["agent.name"]),
|
||||
model: present(attributes, &["llm.model_name", "embedding.model_name"]),
|
||||
input_tokens: token_count(attributes, "llm.token_count.prompt", usage_input)?,
|
||||
output_tokens: token_count(attributes, "llm.token_count.completion", usage_output)?,
|
||||
input: input.text,
|
||||
output: output.text,
|
||||
tool_call_id: present(attributes, &["tool.id"]),
|
||||
calls: if role == Some(RoleEvidence::Declared(ObservationType::Llm)) {
|
||||
calls(attr(attributes, "output.value"))
|
||||
} else {
|
||||
CallEvidence::Unknown
|
||||
},
|
||||
input_preview: None,
|
||||
},
|
||||
display_name: None,
|
||||
consumed_attributes: [input.consumed, output.consumed]
|
||||
.into_iter()
|
||||
.flatten()
|
||||
.collect(),
|
||||
})
|
||||
}
|
||||
}
|
||||
57
litellm-rust/crates/traces/src/normalize/format/traceloop.rs
Normal file
57
litellm-rust/crates/traces/src/normalize/format/traceloop.rs
Normal file
|
|
@ -0,0 +1,57 @@
|
|||
use super::{Extraction, Format, SpanFacts, genai::GenAi};
|
||||
use crate::{
|
||||
Error,
|
||||
normalize::{
|
||||
ObservationType, RoleEvidence, SpanContext, attr, messages, present, select_attribute,
|
||||
},
|
||||
};
|
||||
|
||||
pub(crate) struct Traceloop;
|
||||
|
||||
impl Format for Traceloop {
|
||||
fn matches(&self, context: &SpanContext<'_>) -> bool {
|
||||
context
|
||||
.attributes
|
||||
.keys()
|
||||
.any(|key| key.starts_with("traceloop."))
|
||||
}
|
||||
|
||||
fn extract(&self, context: &SpanContext<'_>) -> Result<Extraction, Error> {
|
||||
let base = GenAi.extract(context)?;
|
||||
let role = match attr(context.attributes, "traceloop.span.kind") {
|
||||
"agent" => Some(ObservationType::Agent),
|
||||
"tool" => Some(ObservationType::Tool),
|
||||
"workflow" | "task" => Some(ObservationType::Chain),
|
||||
_ => match present(
|
||||
context.attributes,
|
||||
&["traceloop.llm.request.type", "llm.request.type"],
|
||||
)
|
||||
.as_deref()
|
||||
{
|
||||
Some("embedding" | "embeddings") => Some(ObservationType::Embedding),
|
||||
Some("chat" | "completion") => Some(ObservationType::Llm),
|
||||
_ => None,
|
||||
},
|
||||
};
|
||||
let input = select_attribute(context.attributes, &["traceloop.entity.input"]);
|
||||
let output = select_attribute(context.attributes, &["traceloop.entity.output"]);
|
||||
Ok(Extraction {
|
||||
facts: SpanFacts {
|
||||
role: role.map(RoleEvidence::Declared),
|
||||
input: input
|
||||
.as_ref()
|
||||
.map_or(String::new(), |value| messages::canonical(value.text)),
|
||||
output: output
|
||||
.as_ref()
|
||||
.map_or(String::new(), |value| messages::canonical(value.text)),
|
||||
..SpanFacts::default()
|
||||
}
|
||||
.or(base.facts),
|
||||
display_name: present(context.attributes, &["traceloop.entity.name"])
|
||||
.or(base.display_name),
|
||||
consumed_attributes: base.consumed_attributes,
|
||||
}
|
||||
.consuming(input)
|
||||
.consuming(output))
|
||||
}
|
||||
}
|
||||
171
litellm-rust/crates/traces/src/normalize/format/vercel.rs
Normal file
171
litellm-rust/crates/traces/src/normalize/format/vercel.rs
Normal file
|
|
@ -0,0 +1,171 @@
|
|||
use serde::{Deserialize, Serialize};
|
||||
use serde_json::Value;
|
||||
|
||||
use super::{Extraction, Format, SpanFacts, genai::GenAi};
|
||||
use crate::{
|
||||
Error,
|
||||
normalize::{
|
||||
ObservationType, RoleEvidence, SpanContext, attr, messages, present, select_attribute,
|
||||
token_alias,
|
||||
},
|
||||
};
|
||||
|
||||
pub(crate) struct Vercel;
|
||||
|
||||
#[derive(Deserialize)]
|
||||
struct Prompt {
|
||||
messages: Option<Value>,
|
||||
prompt: Option<String>,
|
||||
system: Option<String>,
|
||||
}
|
||||
|
||||
#[derive(Deserialize, Serialize)]
|
||||
struct ToolCall {
|
||||
#[serde(rename(deserialize = "toolCallId"))]
|
||||
id: String,
|
||||
#[serde(rename(deserialize = "toolName"))]
|
||||
name: String,
|
||||
#[serde(alias = "args", alias = "input")]
|
||||
arguments: Value,
|
||||
}
|
||||
|
||||
fn prompt(raw: &str) -> String {
|
||||
let Ok(value) = serde_json::from_str::<Prompt>(raw) else {
|
||||
return messages::canonical(raw);
|
||||
};
|
||||
let content = value.messages.unwrap_or_else(|| {
|
||||
Value::Array(
|
||||
value
|
||||
.prompt
|
||||
.into_iter()
|
||||
.map(|text| serde_json::json!({"role": "user", "content": text}))
|
||||
.collect(),
|
||||
)
|
||||
});
|
||||
let conversation: Vec<Value> = value
|
||||
.system
|
||||
.into_iter()
|
||||
.map(|text| serde_json::json!({"role": "system", "content": text}))
|
||||
.chain(content.as_array().into_iter().flatten().cloned())
|
||||
.collect();
|
||||
if conversation.is_empty() {
|
||||
return raw.to_owned();
|
||||
}
|
||||
messages::canonical(&messages::encode(&conversation))
|
||||
}
|
||||
|
||||
impl Format for Vercel {
|
||||
fn matches(&self, context: &SpanContext<'_>) -> bool {
|
||||
context.attributes.contains_key("ai.operationId")
|
||||
|| (context.scope == "ai"
|
||||
&& context.attributes.keys().any(|key| key.starts_with("ai.")))
|
||||
}
|
||||
|
||||
fn extract(&self, context: &SpanContext<'_>) -> Result<Extraction, Error> {
|
||||
let base = GenAi.extract(context)?;
|
||||
let operation = attr(context.attributes, "ai.operationId");
|
||||
let role = match operation {
|
||||
"ai.toolCall" => Some(ObservationType::Tool),
|
||||
"ai.embed" | "ai.embedMany" | "ai.embed.doEmbed" | "ai.embedMany.doEmbed" => {
|
||||
Some(ObservationType::Embedding)
|
||||
}
|
||||
"ai.generateText"
|
||||
| "ai.streamText"
|
||||
| "ai.generateObject"
|
||||
| "ai.streamObject"
|
||||
| "ai.generateText.doGenerate"
|
||||
| "ai.streamText.doStream"
|
||||
| "ai.generateObject.doGenerate"
|
||||
| "ai.streamObject.doStream" => Some(ObservationType::Llm),
|
||||
_ => None,
|
||||
};
|
||||
let input = base
|
||||
.facts
|
||||
.input
|
||||
.is_empty()
|
||||
.then(|| {
|
||||
select_attribute(
|
||||
context.attributes,
|
||||
&[
|
||||
"ai.toolCall.args",
|
||||
"ai.prompt.messages",
|
||||
"ai.prompt",
|
||||
"ai.value",
|
||||
"ai.values",
|
||||
],
|
||||
)
|
||||
})
|
||||
.flatten();
|
||||
let output = base
|
||||
.facts
|
||||
.output
|
||||
.is_empty()
|
||||
.then(|| {
|
||||
select_attribute(
|
||||
context.attributes,
|
||||
&[
|
||||
"ai.toolCall.result",
|
||||
"ai.response.object",
|
||||
"ai.response.text",
|
||||
"ai.embeddings",
|
||||
"ai.embedding",
|
||||
],
|
||||
)
|
||||
})
|
||||
.flatten();
|
||||
let calls = base
|
||||
.facts
|
||||
.output
|
||||
.is_empty()
|
||||
.then(|| select_attribute(context.attributes, &["ai.response.toolCalls"]))
|
||||
.flatten();
|
||||
let response = calls
|
||||
.as_ref()
|
||||
.and_then(|value| serde_json::from_str::<Vec<ToolCall>>(value.text).ok());
|
||||
let legacy_output = match response {
|
||||
Some(calls) => messages::canonical(&messages::encode(&serde_json::json!([{
|
||||
"role": "assistant", "content": output.as_ref().map_or("", |value| value.text), "tool_calls": calls,
|
||||
}]))),
|
||||
None => output
|
||||
.as_ref()
|
||||
.map_or(String::new(), |value| value.text.to_owned()),
|
||||
};
|
||||
Ok(Extraction {
|
||||
facts: base.facts.or(SpanFacts {
|
||||
role: role.map(RoleEvidence::Declared),
|
||||
model: present(context.attributes, &["ai.model.id"]),
|
||||
input_tokens: token_alias(
|
||||
context.attributes,
|
||||
&[
|
||||
"gen_ai.usage.input_tokens",
|
||||
"gen_ai.usage.prompt_tokens",
|
||||
"ai.usage.promptTokens",
|
||||
"ai.usage.tokens",
|
||||
],
|
||||
)?,
|
||||
output_tokens: token_alias(
|
||||
context.attributes,
|
||||
&[
|
||||
"gen_ai.usage.output_tokens",
|
||||
"gen_ai.usage.completion_tokens",
|
||||
"ai.usage.completionTokens",
|
||||
],
|
||||
)?,
|
||||
input: input
|
||||
.as_ref()
|
||||
.map_or(String::new(), |value| match value.source {
|
||||
"ai.prompt" | "ai.prompt.messages" => prompt(value.text),
|
||||
_ => value.text.to_owned(),
|
||||
}),
|
||||
output: legacy_output,
|
||||
tool_call_id: present(context.attributes, &["ai.toolCall.id"]),
|
||||
..SpanFacts::default()
|
||||
}),
|
||||
display_name: present(context.attributes, &["ai.toolCall.name"]).or(base.display_name),
|
||||
consumed_attributes: base.consumed_attributes,
|
||||
}
|
||||
.consuming(input)
|
||||
.consuming(output)
|
||||
.consuming(calls))
|
||||
}
|
||||
}
|
||||
|
|
@ -1,65 +0,0 @@
|
|||
use std::collections::BTreeMap;
|
||||
|
||||
use super::{NormalizedSpan, ObservationType, SpanNormalizer, attr, first, usage_tokens};
|
||||
use crate::{Error, otlp::DecodedEvent};
|
||||
|
||||
pub(super) struct GenAiNormalizer;
|
||||
|
||||
impl SpanNormalizer for GenAiNormalizer {
|
||||
fn matches(&self, _scope_name: &str, _attributes: &BTreeMap<String, String>) -> bool {
|
||||
true
|
||||
}
|
||||
|
||||
fn consumed_attributes(&self, attributes: &BTreeMap<String, String>) -> [&'static str; 2] {
|
||||
[
|
||||
if attr(attributes, "gen_ai.input.messages").is_empty() {
|
||||
"gen_ai.tool.call.arguments"
|
||||
} else {
|
||||
"gen_ai.input.messages"
|
||||
},
|
||||
if attr(attributes, "gen_ai.output.messages").is_empty() {
|
||||
"gen_ai.tool.call.result"
|
||||
} else {
|
||||
"gen_ai.output.messages"
|
||||
},
|
||||
]
|
||||
}
|
||||
|
||||
fn normalize(
|
||||
&self,
|
||||
_name: &str,
|
||||
parent_span_id: &str,
|
||||
attributes: &BTreeMap<String, String>,
|
||||
_events: &[DecodedEvent],
|
||||
) -> Result<NormalizedSpan, Error> {
|
||||
let (input_tokens, output_tokens) = usage_tokens(attributes)?;
|
||||
let observation_type = match attr(attributes, "gen_ai.operation.name") {
|
||||
"invoke_agent" => ObservationType::Agent,
|
||||
"chat" | "text_completion" | "generate_content" => ObservationType::Llm,
|
||||
"execute_tool" => ObservationType::Tool,
|
||||
_ if parent_span_id.is_empty() => ObservationType::Agent,
|
||||
_ => ObservationType::Chain,
|
||||
};
|
||||
Ok(NormalizedSpan {
|
||||
observation_type,
|
||||
agent_name: attr(attributes, "gen_ai.agent.name").to_owned(),
|
||||
framework: String::new(),
|
||||
litellm_request_id: attr(attributes, "gen_ai.response.id").to_owned(),
|
||||
model: first(attributes, "gen_ai.request.model", "gen_ai.response.model").to_owned(),
|
||||
input_tokens,
|
||||
output_tokens,
|
||||
input: first(
|
||||
attributes,
|
||||
"gen_ai.input.messages",
|
||||
"gen_ai.tool.call.arguments",
|
||||
)
|
||||
.to_owned(),
|
||||
output: first(
|
||||
attributes,
|
||||
"gen_ai.output.messages",
|
||||
"gen_ai.tool.call.result",
|
||||
)
|
||||
.to_owned(),
|
||||
})
|
||||
}
|
||||
}
|
||||
|
|
@ -0,0 +1,7 @@
|
|||
- Interpret extracted facts using known behavior of the SDK or instrumentor that emitted the span
|
||||
- Own SDK detection, integration identity, agent naming, role adjustments, input previews, and call-evidence guarantees
|
||||
- Require positive SDK evidence before applying a rule; preserve detection precedence when scopes overlap
|
||||
- Mark call evidence complete only when the emitting contract guarantees which calls the span represents, never from the number of IDs found
|
||||
- Keep attribute conventions and payload decoding in `../format/`; reuse `../messages.rs` for message and state conversion
|
||||
- Emit role and call evidence for `resolve/`; do not infer wrappers, ownership, or spend from spans outside the current context
|
||||
- Add regression cases to the existing public normalization tests for SDK behavior and ambiguous or unmatched input
|
||||
|
|
@ -0,0 +1,12 @@
|
|||
use super::{ObservationType, RoleEvidence, SpanFacts};
|
||||
|
||||
pub(super) fn adjust(facts: SpanFacts) -> SpanFacts {
|
||||
if facts.agent_name.as_deref() != Some("Agent") {
|
||||
return facts;
|
||||
}
|
||||
SpanFacts {
|
||||
role: Some(RoleEvidence::WrapperCandidate(ObservationType::Agent)),
|
||||
agent_name: None,
|
||||
..facts
|
||||
}
|
||||
}
|
||||
|
|
@ -0,0 +1,60 @@
|
|||
use super::{
|
||||
Integration, ObservationType, RoleEvidence, Rule, SpanContext, SpanFacts, attr, present,
|
||||
};
|
||||
use crate::normalize::{CLAUDE_CODE_AGENT, CLAUDE_CODE_SCOPE};
|
||||
use std::collections::BTreeMap;
|
||||
|
||||
pub(super) const SCOPE: &str = CLAUDE_CODE_SCOPE;
|
||||
|
||||
pub(super) fn adjust(context: &SpanContext<'_>, facts: SpanFacts) -> SpanFacts {
|
||||
if attr(context.attributes, "parent.source") != "env"
|
||||
|| facts.role != Some(RoleEvidence::Declared(ObservationType::Agent))
|
||||
{
|
||||
return facts;
|
||||
}
|
||||
SpanFacts {
|
||||
role: Some(RoleEvidence::WrapperCandidate(ObservationType::Agent)),
|
||||
..facts
|
||||
}
|
||||
}
|
||||
|
||||
fn framework(attributes: &BTreeMap<String, String>) -> Integration {
|
||||
if attr(attributes, "query_source_safe") == "sdk"
|
||||
|| attr(attributes, "system_prompt_preview").contains("cc_entrypoint=sdk")
|
||||
{
|
||||
Integration::ClaudeAgentSdk
|
||||
} else {
|
||||
Integration::ClaudeCode
|
||||
}
|
||||
}
|
||||
|
||||
pub(super) struct ClaudeCode;
|
||||
|
||||
impl Rule for ClaudeCode {
|
||||
fn matches(&self, context: &SpanContext<'_>) -> bool {
|
||||
context.scope == SCOPE
|
||||
}
|
||||
fn integration(&self, context: &SpanContext<'_>) -> Option<Integration> {
|
||||
Some(framework(context.attributes))
|
||||
}
|
||||
fn adjust(
|
||||
&self,
|
||||
context: &SpanContext<'_>,
|
||||
extraction: super::Extraction,
|
||||
) -> super::Extraction {
|
||||
extraction.map_facts(|facts| adjust(context, facts))
|
||||
}
|
||||
|
||||
fn agent_name(&self, context: &SpanContext<'_>, recorded: Option<String>) -> Option<String> {
|
||||
match (
|
||||
present(context.resource_attributes, &["gen_ai.agent.name"]),
|
||||
recorded.as_deref(),
|
||||
) {
|
||||
(Some(name), None | Some(CLAUDE_CODE_AGENT)) => Some(name),
|
||||
(None, Some(CLAUDE_CODE_AGENT)) => {
|
||||
present(context.resource_attributes, &["service.name"]).or(recorded)
|
||||
}
|
||||
_ => recorded,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
|
@ -0,0 +1,10 @@
|
|||
use super::{SpanFacts, messages};
|
||||
|
||||
pub(super) const SCOPE: &str = "gcp.vertex.agent";
|
||||
|
||||
pub(super) fn adjust(facts: SpanFacts) -> SpanFacts {
|
||||
SpanFacts {
|
||||
input_preview: messages::state_preview(&facts.input, "new_message").or(facts.input_preview),
|
||||
..facts
|
||||
}
|
||||
}
|
||||
|
|
@ -0,0 +1,22 @@
|
|||
use super::{Integration, Rule, SpanContext, present};
|
||||
|
||||
const SCOPE: &str = "hermes-otel-plugin";
|
||||
|
||||
pub(super) struct Hermes;
|
||||
|
||||
impl Rule for Hermes {
|
||||
fn matches(&self, context: &SpanContext<'_>) -> bool {
|
||||
context.scope == SCOPE
|
||||
}
|
||||
|
||||
fn integration(&self, _: &SpanContext<'_>) -> Option<Integration> {
|
||||
None
|
||||
}
|
||||
|
||||
fn agent_name(&self, context: &SpanContext<'_>, recorded: Option<String>) -> Option<String> {
|
||||
if recorded.as_deref() == Some("hermes-agent") {
|
||||
return present(context.resource_attributes, &["gen_ai.agent.name"]).or(recorded);
|
||||
}
|
||||
recorded
|
||||
}
|
||||
}
|
||||
|
|
@ -0,0 +1,38 @@
|
|||
use super::{CallEvidence, CallKey, ObservationType, RoleEvidence, SpanContext, SpanFacts};
|
||||
use super::{Integration, Rule};
|
||||
|
||||
const SCOPES: [&str; 7] = [
|
||||
"opentelemetry.instrumentation.httpx",
|
||||
"opentelemetry.instrumentation.requests",
|
||||
"opentelemetry.instrumentation.aiohttp_client",
|
||||
"opentelemetry.instrumentation.urllib3",
|
||||
"opentelemetry.instrumentation.urllib",
|
||||
"@opentelemetry/instrumentation-http",
|
||||
"@opentelemetry/instrumentation-undici",
|
||||
];
|
||||
|
||||
pub(super) fn matches(context: &SpanContext<'_>) -> bool {
|
||||
SCOPES.contains(&context.scope)
|
||||
}
|
||||
|
||||
pub(super) fn adjust(facts: SpanFacts) -> SpanFacts {
|
||||
SpanFacts {
|
||||
role: Some(RoleEvidence::Declared(ObservationType::Framework)),
|
||||
calls: CallEvidence::complete(CallKey::Transport),
|
||||
..facts
|
||||
}
|
||||
}
|
||||
|
||||
pub(super) struct HttpClient;
|
||||
|
||||
impl Rule for HttpClient {
|
||||
fn matches(&self, context: &SpanContext<'_>) -> bool {
|
||||
matches(context)
|
||||
}
|
||||
fn integration(&self, _: &SpanContext<'_>) -> Option<Integration> {
|
||||
None
|
||||
}
|
||||
fn adjust(&self, _: &SpanContext<'_>, extraction: super::Extraction) -> super::Extraction {
|
||||
extraction.map_facts(adjust)
|
||||
}
|
||||
}
|
||||
|
|
@ -0,0 +1,80 @@
|
|||
use super::{
|
||||
AgentMetadata, Integration, ObservationType, RoleEvidence, SpanContext, SpanFacts, attr,
|
||||
messages,
|
||||
};
|
||||
use crate::normalize::present;
|
||||
use std::collections::BTreeMap;
|
||||
|
||||
pub(super) fn adjust(context: &SpanContext<'_>, facts: SpanFacts) -> SpanFacts {
|
||||
let middleware = !context.parent_span_id.is_empty() && is_langchain_middleware(context.name);
|
||||
SpanFacts {
|
||||
role: if middleware {
|
||||
Some(RoleEvidence::Declared(ObservationType::Framework))
|
||||
} else {
|
||||
facts.role
|
||||
},
|
||||
input_preview: messages::state_preview(&facts.input, "messages"),
|
||||
..facts
|
||||
}
|
||||
}
|
||||
|
||||
pub(super) fn agent_name(context: &SpanContext<'_>, metadata: &AgentMetadata) -> Option<String> {
|
||||
let node = attr(context.attributes, "graph.node.id");
|
||||
if !node.is_empty() {
|
||||
return Some(node.to_owned());
|
||||
}
|
||||
(metadata.ls_integration == Some(Integration::Langgraph)
|
||||
&& context.name != "LangGraph"
|
||||
&& !is_langchain_middleware(context.name))
|
||||
.then(|| context.name.to_owned())
|
||||
}
|
||||
|
||||
const MIDDLEWARE_SUFFIXES: [&str; 6] = [
|
||||
".wrap_model_call",
|
||||
".wrap_tool_call",
|
||||
".before_agent",
|
||||
".after_agent",
|
||||
".before_model",
|
||||
".after_model",
|
||||
];
|
||||
|
||||
pub(super) fn is_langchain_middleware(name: &str) -> bool {
|
||||
MIDDLEWARE_SUFFIXES
|
||||
.iter()
|
||||
.any(|suffix| name.ends_with(suffix))
|
||||
}
|
||||
|
||||
fn span_type(
|
||||
name: &str,
|
||||
parent_span_id: &str,
|
||||
attributes: &BTreeMap<String, String>,
|
||||
) -> ObservationType {
|
||||
match ObservationType::try_from(attr(attributes, "langsmith.span.kind")) {
|
||||
Ok(kind) if kind != ObservationType::Chain => kind,
|
||||
_ if parent_span_id.is_empty()
|
||||
|| name == attr(attributes, "langsmith.metadata.lc_agent_name") =>
|
||||
{
|
||||
ObservationType::Agent
|
||||
}
|
||||
_ if is_langchain_middleware(name) => ObservationType::Framework,
|
||||
_ => ObservationType::Chain,
|
||||
}
|
||||
}
|
||||
|
||||
pub(super) fn langsmith(context: &SpanContext<'_>, facts: SpanFacts) -> SpanFacts {
|
||||
let kind = span_type(context.name, context.parent_span_id, context.attributes);
|
||||
let input = (kind == ObservationType::Agent)
|
||||
.then(|| messages::state_conversation(&facts.input))
|
||||
.flatten();
|
||||
let output = (kind == ObservationType::Agent)
|
||||
.then(|| messages::state_conversation(&facts.output))
|
||||
.flatten()
|
||||
.and_then(|conversation| conversation.last().map(messages::encode));
|
||||
SpanFacts {
|
||||
role: Some(RoleEvidence::Declared(kind)),
|
||||
agent_name: present(context.attributes, &["langsmith.metadata.lc_agent_name"]),
|
||||
input: input.map_or(facts.input, |conversation| messages::encode(&conversation)),
|
||||
output: output.unwrap_or(facts.output),
|
||||
..facts
|
||||
}
|
||||
}
|
||||
|
|
@ -0,0 +1,58 @@
|
|||
use super::{ObservationType, RoleEvidence, SpanContext, SpanFacts, attr};
|
||||
use serde_json::Value;
|
||||
|
||||
pub(super) fn adjust(context: &SpanContext<'_>, facts: SpanFacts) -> SpanFacts {
|
||||
let agent = context
|
||||
.name
|
||||
.ends_with(".run_agent_step")
|
||||
.then(|| current_agent_name(attr(context.attributes, "input.value")))
|
||||
.flatten();
|
||||
let role = match agent {
|
||||
Some(_) => Some(RoleEvidence::Declared(ObservationType::Agent)),
|
||||
None if context.name.ends_with("._prepare_chat_with_tools") => {
|
||||
Some(RoleEvidence::Declared(ObservationType::Chain))
|
||||
}
|
||||
None => facts.role,
|
||||
};
|
||||
let engine_state = context.parent_span_id.is_empty() && has_key(&facts.input, "start_event");
|
||||
SpanFacts {
|
||||
role,
|
||||
agent_name: agent.map(str::to_owned).or(facts.agent_name),
|
||||
input_preview: if engine_state {
|
||||
Some(String::new())
|
||||
} else {
|
||||
facts.input_preview
|
||||
},
|
||||
..facts
|
||||
}
|
||||
}
|
||||
|
||||
fn current_agent_name(input: &str) -> Option<&str> {
|
||||
let (_, rest) = input.split_once("current_agent_name='")?;
|
||||
let (agent, _) = rest.split_once('\'')?;
|
||||
(!agent.is_empty()).then_some(agent)
|
||||
}
|
||||
|
||||
fn has_key(input: &str, key: &str) -> bool {
|
||||
serde_json::from_str::<serde_json::Map<String, Value>>(input)
|
||||
.is_ok_and(|object| object.contains_key(key))
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use rstest::rstest;
|
||||
|
||||
use super::current_agent_name;
|
||||
|
||||
#[rstest]
|
||||
#[case::named("ev=current_agent_name='delegate'", Some("delegate"))]
|
||||
#[case::missing("ev=other", None)]
|
||||
#[case::empty("current_agent_name=''", None)]
|
||||
#[case::unterminated("current_agent_name='delegate", None)]
|
||||
fn agent_name_requires_a_complete_nonempty_value(
|
||||
#[case] input: &str,
|
||||
#[case] expected: Option<&str>,
|
||||
) {
|
||||
assert_eq!(current_agent_name(input), expected);
|
||||
}
|
||||
}
|
||||
261
litellm-rust/crates/traces/src/normalize/instrumentation/mod.rs
Normal file
261
litellm-rust/crates/traces/src/normalize/instrumentation/mod.rs
Normal file
|
|
@ -0,0 +1,261 @@
|
|||
//! What is known about the SDK that emitted a span, applied to its convention's [`SpanFacts`].
|
||||
//! Each rule needs positive evidence from that SDK; anything less stays a [`RoleEvidence`] for the
|
||||
//! trace graph to settle.
|
||||
|
||||
use super::{
|
||||
AgentMetadata, AgentType, CallEvidence, CallKey, Integration, Normalization, NormalizedSpan,
|
||||
ObservationType, RoleEvidence, SpanContext, attr,
|
||||
format::{Extraction, SpanFacts},
|
||||
messages, present, select_attribute,
|
||||
};
|
||||
|
||||
const OPENINFERENCE_PREFIX: &str = "openinference.instrumentation.";
|
||||
|
||||
pub(super) mod claude_agent_sdk;
|
||||
pub(super) mod claude_code;
|
||||
pub(super) mod google_adk;
|
||||
pub(super) mod hermes;
|
||||
pub(super) mod http_client;
|
||||
pub(super) mod langchain;
|
||||
pub(super) mod llama_index;
|
||||
pub(super) mod pydantic_ai;
|
||||
|
||||
pub(super) trait Rule: Sync {
|
||||
fn matches(&self, context: &SpanContext<'_>) -> bool;
|
||||
fn integration(&self, context: &SpanContext<'_>) -> Option<Integration>;
|
||||
fn agent_name(&self, _: &SpanContext<'_>, recorded: Option<String>) -> Option<String> {
|
||||
recorded
|
||||
}
|
||||
fn adjust(&self, _: &SpanContext<'_>, extraction: Extraction) -> Extraction {
|
||||
extraction
|
||||
}
|
||||
}
|
||||
|
||||
struct Scoped {
|
||||
scope: &'static str,
|
||||
integration: Integration,
|
||||
prefix: bool,
|
||||
}
|
||||
|
||||
impl Rule for Scoped {
|
||||
fn matches(&self, context: &SpanContext<'_>) -> bool {
|
||||
if self.prefix {
|
||||
context.scope.starts_with(self.scope)
|
||||
} else {
|
||||
context.scope == self.scope
|
||||
}
|
||||
}
|
||||
|
||||
fn integration(&self, _: &SpanContext<'_>) -> Option<Integration> {
|
||||
Some(self.integration.clone())
|
||||
}
|
||||
}
|
||||
|
||||
struct OpenInference;
|
||||
|
||||
impl Rule for OpenInference {
|
||||
fn matches(&self, context: &SpanContext<'_>) -> bool {
|
||||
context.scope.starts_with(OPENINFERENCE_PREFIX)
|
||||
}
|
||||
|
||||
fn integration(&self, context: &SpanContext<'_>) -> Option<Integration> {
|
||||
context
|
||||
.scope
|
||||
.strip_prefix(OPENINFERENCE_PREFIX)
|
||||
.filter(|name| !name.is_empty())
|
||||
.map(|name| Integration::from(name.replace('_', "-")))
|
||||
}
|
||||
|
||||
fn agent_name(&self, context: &SpanContext<'_>, recorded: Option<String>) -> Option<String> {
|
||||
recorded.filter(|name| {
|
||||
name != "Agent" || self.integration(context) != Some(Integration::ClaudeAgentSdk)
|
||||
})
|
||||
}
|
||||
|
||||
fn adjust(&self, context: &SpanContext<'_>, extraction: Extraction) -> Extraction {
|
||||
match self.integration(context) {
|
||||
Some(Integration::Langchain) => {
|
||||
extraction.map_facts(|facts| langchain::adjust(context, facts))
|
||||
}
|
||||
Some(Integration::LlamaIndex) => {
|
||||
extraction.map_facts(|facts| llama_index::adjust(context, facts))
|
||||
}
|
||||
Some(Integration::ClaudeAgentSdk) => extraction.map_facts(claude_agent_sdk::adjust),
|
||||
Some(Integration::GoogleAdk) => extraction.map_facts(google_adk::adjust),
|
||||
_ => extraction,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
const RULES: [&dyn Rule; 9] = [
|
||||
&claude_code::ClaudeCode,
|
||||
&hermes::Hermes,
|
||||
&OpenInference,
|
||||
&http_client::HttpClient,
|
||||
&pydantic_ai::PydanticAi,
|
||||
&Scoped {
|
||||
scope: google_adk::SCOPE,
|
||||
integration: Integration::GoogleAdk,
|
||||
prefix: false,
|
||||
},
|
||||
&Scoped {
|
||||
scope: "gen_ai",
|
||||
integration: Integration::VercelAiSdk,
|
||||
prefix: false,
|
||||
},
|
||||
&Scoped {
|
||||
scope: "ai",
|
||||
integration: Integration::VercelAiSdk,
|
||||
prefix: false,
|
||||
},
|
||||
&Scoped {
|
||||
scope: "strands.",
|
||||
integration: Integration::Strands,
|
||||
prefix: true,
|
||||
},
|
||||
];
|
||||
|
||||
pub(super) struct Instrumentation(Option<&'static dyn Rule>);
|
||||
|
||||
impl Instrumentation {
|
||||
pub(super) fn detect(context: &SpanContext<'_>) -> Self {
|
||||
Self(RULES.into_iter().find(|rule| rule.matches(context)))
|
||||
}
|
||||
|
||||
fn adjust(&self, context: &SpanContext<'_>, extraction: Extraction) -> Extraction {
|
||||
match self.0 {
|
||||
Some(rule) => rule.adjust(context, extraction),
|
||||
None => extraction,
|
||||
}
|
||||
}
|
||||
|
||||
pub(super) fn interpret(
|
||||
&self,
|
||||
context: &SpanContext<'_>,
|
||||
extraction: Extraction,
|
||||
metadata: AgentMetadata,
|
||||
) -> Normalization {
|
||||
let prepared = if context.scope != claude_code::SCOPE
|
||||
&& (context.scope == "langsmith"
|
||||
|| context.attributes.contains_key("langsmith.span.kind"))
|
||||
{
|
||||
extraction.map_facts(|facts| langchain::langsmith(context, facts))
|
||||
} else {
|
||||
extraction
|
||||
};
|
||||
let Extraction {
|
||||
facts,
|
||||
display_name,
|
||||
consumed_attributes,
|
||||
} = self.adjust(
|
||||
context,
|
||||
prepared.map_facts(|facts| with_response_id(context, facts)),
|
||||
);
|
||||
let role = match (facts.role, metadata.ls_agent_type) {
|
||||
(
|
||||
None
|
||||
| Some(RoleEvidence::Declared(ObservationType::Agent | ObservationType::Chain))
|
||||
| Some(RoleEvidence::WrapperCandidate(ObservationType::Agent)),
|
||||
Some(agent_type),
|
||||
) => Some(RoleEvidence::Declared(match agent_type {
|
||||
AgentType::Root | AgentType::Subagent => ObservationType::Agent,
|
||||
AgentType::Middleware | AgentType::Compaction => ObservationType::Framework,
|
||||
})),
|
||||
(role, _) => role,
|
||||
};
|
||||
let (observation_type, wrapper_candidate) = match role.unwrap_or(RoleEvidence::Unspecified)
|
||||
{
|
||||
RoleEvidence::Declared(kind) => (kind, false),
|
||||
RoleEvidence::WrapperCandidate(kind) => (kind, true),
|
||||
// An unlabelled root may be the agent run itself, or only wrap the agents below it.
|
||||
RoleEvidence::Unspecified if context.parent_span_id.is_empty() => {
|
||||
(ObservationType::Agent, true)
|
||||
}
|
||||
RoleEvidence::Unspecified => (ObservationType::Chain, false),
|
||||
};
|
||||
let recorded_name =
|
||||
recorded_agent_name(context, facts.agent_name, observation_type, &metadata);
|
||||
let sdk_name = match self.0 {
|
||||
Some(rule) => rule.agent_name(context, recorded_name),
|
||||
None => recorded_name,
|
||||
};
|
||||
let agent_name =
|
||||
sdk_name.or_else(|| present(context.resource_attributes, &["gen_ai.agent.name"]));
|
||||
let framework = metadata
|
||||
.ls_integration
|
||||
.clone()
|
||||
.or_else(|| self.0.and_then(|rule| rule.integration(context)));
|
||||
let model = facts.model.or_else(|| metadata.ls_model_name.clone());
|
||||
let display_name = if observation_type == ObservationType::Tool {
|
||||
display_name.or_else(|| metadata.ls_tool_name.clone())
|
||||
} else {
|
||||
display_name
|
||||
};
|
||||
let input_preview = facts
|
||||
.input_preview
|
||||
.unwrap_or_else(|| messages::input_preview(&facts.input));
|
||||
Normalization {
|
||||
span: NormalizedSpan {
|
||||
observation_type,
|
||||
wrapper_candidate,
|
||||
agent_name,
|
||||
framework,
|
||||
agent_metadata: metadata,
|
||||
calls: facts.calls,
|
||||
model,
|
||||
input_tokens: facts.input_tokens,
|
||||
output_tokens: facts.output_tokens,
|
||||
input: facts.input,
|
||||
input_preview,
|
||||
output: facts.output,
|
||||
tool_call_id: facts.tool_call_id,
|
||||
},
|
||||
display_name,
|
||||
consumed_attributes: consumed_attributes.into_boxed_slice(),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// `gen_ai.response.id` names one provider response, whichever convention recorded it.
|
||||
fn with_response_id(context: &SpanContext<'_>, facts: SpanFacts) -> SpanFacts {
|
||||
match present(context.attributes, &["gen_ai.response.id"]) {
|
||||
Some(id) => SpanFacts {
|
||||
calls: facts.calls.with(CallKey::ProviderResponse(id)),
|
||||
..facts
|
||||
},
|
||||
None => facts,
|
||||
}
|
||||
}
|
||||
|
||||
fn recorded_agent_name(
|
||||
context: &SpanContext<'_>,
|
||||
extracted: Option<String>,
|
||||
observation_type: ObservationType,
|
||||
metadata: &AgentMetadata,
|
||||
) -> Option<String> {
|
||||
if let Some(name) = extracted {
|
||||
return Some(name);
|
||||
}
|
||||
let attributes = context.attributes;
|
||||
let explicit = [
|
||||
attr(attributes, "gen_ai.agent.name"),
|
||||
attr(attributes, "agent.name"),
|
||||
attr(attributes, "openclaw.agent"),
|
||||
]
|
||||
.into_iter()
|
||||
.find(|value| !value.is_empty());
|
||||
if let Some(value) = explicit {
|
||||
return Some(value.to_owned());
|
||||
}
|
||||
if let Some(name) = metadata
|
||||
.lc_agent_name
|
||||
.as_ref()
|
||||
.or(metadata.ls_subagent_type.as_ref())
|
||||
{
|
||||
return Some(name.clone());
|
||||
}
|
||||
if observation_type == ObservationType::Agent {
|
||||
return langchain::agent_name(context, metadata);
|
||||
}
|
||||
None
|
||||
}
|
||||
|
|
@ -0,0 +1,57 @@
|
|||
use super::{Extraction, SpanContext, SpanFacts, messages, select_attribute};
|
||||
use super::{Integration, Rule};
|
||||
use crate::normalize::format::genai::Operation;
|
||||
|
||||
pub(super) const SCOPE: &str = "pydantic-ai";
|
||||
|
||||
pub(super) fn adjust(context: &SpanContext<'_>, extraction: Extraction) -> Extraction {
|
||||
if !matches!(
|
||||
Operation::from_context(context),
|
||||
Some(Operation::InvokeAgent)
|
||||
) {
|
||||
return extraction;
|
||||
}
|
||||
let input = extraction
|
||||
.facts
|
||||
.input
|
||||
.is_empty()
|
||||
.then(|| select_attribute(context.attributes, &["pydantic_ai.all_messages"]))
|
||||
.flatten();
|
||||
let output = extraction
|
||||
.facts
|
||||
.output
|
||||
.is_empty()
|
||||
.then(|| select_attribute(context.attributes, &["final_result"]))
|
||||
.flatten();
|
||||
let fallback = SpanFacts {
|
||||
input: input
|
||||
.as_ref()
|
||||
.map_or(String::new(), |payload| messages::canonical(payload.text)),
|
||||
output: output
|
||||
.as_ref()
|
||||
.map_or(String::new(), |payload| payload.text.to_owned()),
|
||||
..SpanFacts::default()
|
||||
};
|
||||
extraction
|
||||
.map_facts(|facts| facts.or(fallback))
|
||||
.consuming(input)
|
||||
.consuming(output)
|
||||
}
|
||||
|
||||
pub(super) struct PydanticAi;
|
||||
|
||||
impl Rule for PydanticAi {
|
||||
fn matches(&self, context: &SpanContext<'_>) -> bool {
|
||||
context.scope == SCOPE
|
||||
}
|
||||
fn integration(&self, _: &SpanContext<'_>) -> Option<Integration> {
|
||||
Some(Integration::PydanticAi)
|
||||
}
|
||||
fn adjust(
|
||||
&self,
|
||||
context: &SpanContext<'_>,
|
||||
extraction: super::Extraction,
|
||||
) -> super::Extraction {
|
||||
adjust(context, extraction)
|
||||
}
|
||||
}
|
||||
|
|
@ -1,470 +0,0 @@
|
|||
use std::{collections::BTreeMap, io};
|
||||
|
||||
use indexmap::IndexMap;
|
||||
use serde::{Deserialize, Deserializer, Serialize, de::DeserializeOwned};
|
||||
use serde_json::{Value, ser::Formatter};
|
||||
|
||||
use super::{NormalizedSpan, ObservationType, SpanNormalizer, attr, usage_tokens};
|
||||
use crate::{Error, otlp::DecodedEvent};
|
||||
|
||||
pub(super) struct LangSmithNormalizer;
|
||||
|
||||
#[derive(Deserialize)]
|
||||
#[serde(untagged)]
|
||||
enum MessageContent {
|
||||
Text(String),
|
||||
Blocks(Vec<ContentBlock>),
|
||||
Other(Value),
|
||||
}
|
||||
|
||||
impl MessageContent {
|
||||
fn display_text(&self) -> String {
|
||||
match self {
|
||||
Self::Text(text) => text.clone(),
|
||||
Self::Blocks(blocks) => blocks
|
||||
.iter()
|
||||
.filter_map(|block| match block {
|
||||
ContentBlock::Text { text } => Some(text.as_str()),
|
||||
ContentBlock::Hidden(kind) => match kind {
|
||||
HiddenBlock::Reasoning
|
||||
| HiddenBlock::Thinking
|
||||
| HiddenBlock::RedactedThinking
|
||||
| HiddenBlock::FunctionCall
|
||||
| HiddenBlock::ToolUse
|
||||
| HiddenBlock::ToolCall => None,
|
||||
},
|
||||
})
|
||||
.collect::<Vec<_>>()
|
||||
.join("\n\n"),
|
||||
Self::Other(value) => encode(value),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Deserialize)]
|
||||
#[serde(untagged)]
|
||||
enum ContentBlock {
|
||||
Text { text: String },
|
||||
Hidden(HiddenBlock),
|
||||
}
|
||||
|
||||
#[derive(Deserialize)]
|
||||
#[serde(tag = "type", rename_all = "snake_case")]
|
||||
enum HiddenBlock {
|
||||
Reasoning,
|
||||
Thinking,
|
||||
RedactedThinking,
|
||||
FunctionCall,
|
||||
ToolUse,
|
||||
ToolCall,
|
||||
}
|
||||
|
||||
#[derive(Deserialize, Serialize)]
|
||||
#[serde(transparent)]
|
||||
struct RawToolCall(IndexMap<String, Value>);
|
||||
|
||||
#[derive(Deserialize)]
|
||||
struct ResponseMetadata {
|
||||
id: Option<String>,
|
||||
}
|
||||
|
||||
#[derive(Deserialize)]
|
||||
struct RawMessage {
|
||||
kwargs: Option<Box<RawMessage>>,
|
||||
#[serde(rename = "type")]
|
||||
kind: Option<String>,
|
||||
role: Option<String>,
|
||||
content: Option<MessageContent>,
|
||||
tool_calls: Option<Vec<RawToolCall>>,
|
||||
name: Option<Value>,
|
||||
response_metadata: Option<ResponseMetadata>,
|
||||
}
|
||||
|
||||
impl RawMessage {
|
||||
fn unwrapped(&self) -> &Self {
|
||||
self.kwargs.as_deref().unwrap_or(self)
|
||||
}
|
||||
|
||||
fn normalized(&self) -> NormalizedMessage<'_> {
|
||||
let fields = self.unwrapped();
|
||||
let raw_role = fields
|
||||
.kind
|
||||
.as_deref()
|
||||
.filter(|role| !role.is_empty())
|
||||
.or_else(|| fields.role.as_deref().filter(|role| !role.is_empty()))
|
||||
.unwrap_or_default();
|
||||
let role = match raw_role {
|
||||
"human" => "user",
|
||||
"ai" => "assistant",
|
||||
other => other,
|
||||
};
|
||||
NormalizedMessage {
|
||||
role,
|
||||
content: fields
|
||||
.content
|
||||
.as_ref()
|
||||
.map_or_else(String::new, MessageContent::display_text),
|
||||
tool_calls: fields
|
||||
.tool_calls
|
||||
.as_deref()
|
||||
.filter(|calls| !calls.is_empty()),
|
||||
name: (role == "tool")
|
||||
.then_some(fields.name.as_ref())
|
||||
.flatten()
|
||||
.filter(|name| !name.is_null() && name != &&Value::String(String::new())),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Serialize)]
|
||||
struct NormalizedMessage<'a> {
|
||||
role: &'a str,
|
||||
content: String,
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
tool_calls: Option<&'a [RawToolCall]>,
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
name: Option<&'a Value>,
|
||||
}
|
||||
|
||||
enum MessageBatch {
|
||||
Flat(Vec<RawMessage>),
|
||||
Nested(Vec<Vec<RawMessage>>),
|
||||
}
|
||||
|
||||
impl<'de> Deserialize<'de> for MessageBatch {
|
||||
fn deserialize<D: Deserializer<'de>>(deserializer: D) -> Result<Self, D::Error> {
|
||||
let value = Value::deserialize(deserializer)?;
|
||||
let Value::Array(items) = value else {
|
||||
return Err(serde::de::Error::custom("messages must be an array"));
|
||||
};
|
||||
let parse = |items: Vec<Value>| {
|
||||
items
|
||||
.into_iter()
|
||||
.filter_map(|item| serde_json::from_value(item).ok())
|
||||
.collect()
|
||||
};
|
||||
Ok(if items.first().is_some_and(Value::is_array) {
|
||||
Self::Nested(
|
||||
items
|
||||
.into_iter()
|
||||
.filter_map(|item| item.as_array().cloned())
|
||||
.map(parse)
|
||||
.collect(),
|
||||
)
|
||||
} else {
|
||||
Self::Flat(parse(items))
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
fn lenient<'de, D: Deserializer<'de>, T: DeserializeOwned>(
|
||||
deserializer: D,
|
||||
) -> Result<Option<T>, D::Error> {
|
||||
let value = Value::deserialize(deserializer)?;
|
||||
Ok(serde_json::from_value(value).ok())
|
||||
}
|
||||
|
||||
impl MessageBatch {
|
||||
fn first_batch(&self) -> &[RawMessage] {
|
||||
match self {
|
||||
Self::Flat(messages) => messages,
|
||||
Self::Nested(batches) => batches.first().map(Vec::as_slice).unwrap_or_default(),
|
||||
}
|
||||
}
|
||||
|
||||
fn agent_messages(&self) -> &[RawMessage] {
|
||||
match self {
|
||||
Self::Flat(messages) => messages,
|
||||
Self::Nested(_) => &[],
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Deserialize)]
|
||||
struct GenerationMessage {
|
||||
kwargs: Option<RawMessage>,
|
||||
}
|
||||
|
||||
#[derive(Deserialize)]
|
||||
struct Generation {
|
||||
message: Option<GenerationMessage>,
|
||||
}
|
||||
|
||||
#[derive(Default, Deserialize)]
|
||||
struct Payload {
|
||||
#[serde(default, deserialize_with = "lenient")]
|
||||
messages: Option<MessageBatch>,
|
||||
#[serde(default, deserialize_with = "lenient")]
|
||||
generations: Option<Vec<Vec<Generation>>>,
|
||||
}
|
||||
|
||||
#[derive(Deserialize)]
|
||||
struct Command {
|
||||
update: CommandUpdate,
|
||||
}
|
||||
|
||||
#[derive(Deserialize)]
|
||||
struct CommandUpdate {
|
||||
messages: Vec<Value>,
|
||||
}
|
||||
|
||||
#[derive(Deserialize)]
|
||||
struct ContentValue {
|
||||
content: Value,
|
||||
}
|
||||
|
||||
struct SpanIo {
|
||||
input: String,
|
||||
output: String,
|
||||
request_id: String,
|
||||
}
|
||||
|
||||
struct PythonJsonFormatter;
|
||||
|
||||
impl Formatter for PythonJsonFormatter {
|
||||
fn begin_array_value<W: ?Sized + io::Write>(
|
||||
&mut self,
|
||||
writer: &mut W,
|
||||
first: bool,
|
||||
) -> io::Result<()> {
|
||||
if first {
|
||||
Ok(())
|
||||
} else {
|
||||
writer.write_all(b", ")
|
||||
}
|
||||
}
|
||||
|
||||
fn begin_object_key<W: ?Sized + io::Write>(
|
||||
&mut self,
|
||||
writer: &mut W,
|
||||
first: bool,
|
||||
) -> io::Result<()> {
|
||||
if first {
|
||||
Ok(())
|
||||
} else {
|
||||
writer.write_all(b", ")
|
||||
}
|
||||
}
|
||||
|
||||
fn begin_object_value<W: ?Sized + io::Write>(&mut self, writer: &mut W) -> io::Result<()> {
|
||||
writer.write_all(b": ")
|
||||
}
|
||||
}
|
||||
|
||||
fn encode<T: Serialize>(value: &T) -> String {
|
||||
let mut output = Vec::new();
|
||||
let mut serializer = serde_json::Serializer::with_formatter(&mut output, PythonJsonFormatter);
|
||||
if value.serialize(&mut serializer).is_err() {
|
||||
return String::new();
|
||||
}
|
||||
String::from_utf8(output).unwrap_or_default()
|
||||
}
|
||||
|
||||
fn normalized_messages(messages: &[RawMessage]) -> String {
|
||||
encode(
|
||||
&messages
|
||||
.iter()
|
||||
.map(RawMessage::normalized)
|
||||
.collect::<Vec<_>>(),
|
||||
)
|
||||
}
|
||||
|
||||
fn span_type(
|
||||
name: &str,
|
||||
parent_span_id: &str,
|
||||
attributes: &BTreeMap<String, String>,
|
||||
) -> ObservationType {
|
||||
match attr(attributes, "langsmith.span.kind") {
|
||||
"llm" => ObservationType::Llm,
|
||||
"tool" => ObservationType::Tool,
|
||||
_ if parent_span_id.is_empty()
|
||||
|| name == attr(attributes, "langsmith.metadata.lc_agent_name") =>
|
||||
{
|
||||
ObservationType::Agent
|
||||
}
|
||||
_ if [
|
||||
".wrap_model_call",
|
||||
".wrap_tool_call",
|
||||
".before_agent",
|
||||
".after_agent",
|
||||
".before_model",
|
||||
".after_model",
|
||||
]
|
||||
.iter()
|
||||
.any(|suffix| name.ends_with(suffix)) =>
|
||||
{
|
||||
ObservationType::Framework
|
||||
}
|
||||
_ => ObservationType::Chain,
|
||||
}
|
||||
}
|
||||
|
||||
fn tool_output(raw_completion: &str) -> String {
|
||||
let completion = serde_json::from_str::<Value>(raw_completion).unwrap_or(Value::Null);
|
||||
let raw = completion.get("output").cloned().unwrap_or(completion);
|
||||
let selected = serde_json::from_value::<Command>(raw.clone())
|
||||
.ok()
|
||||
.and_then(|command| command.update.messages.into_iter().last())
|
||||
.unwrap_or(raw);
|
||||
let output = serde_json::from_value::<ContentValue>(selected.clone())
|
||||
.map(|message| message.content)
|
||||
.unwrap_or(selected);
|
||||
output
|
||||
.as_str()
|
||||
.map(str::to_owned)
|
||||
.unwrap_or_else(|| encode(&output))
|
||||
}
|
||||
|
||||
fn span_io(kind: ObservationType, attributes: &BTreeMap<String, String>) -> SpanIo {
|
||||
let raw_prompt = attr(attributes, "gen_ai.prompt");
|
||||
let raw_completion = attr(attributes, "gen_ai.completion");
|
||||
let prompt = serde_json::from_str::<Payload>(raw_prompt).unwrap_or_default();
|
||||
let completion = serde_json::from_str::<Payload>(raw_completion).unwrap_or_default();
|
||||
if kind == ObservationType::Llm
|
||||
&& serde_json::from_str::<Value>(raw_completion).is_ok_and(|value| value.is_object())
|
||||
{
|
||||
let input = prompt.messages.as_ref().map_or_else(
|
||||
|| "[]".to_owned(),
|
||||
|messages| normalized_messages(messages.first_batch()),
|
||||
);
|
||||
let generation = completion
|
||||
.generations
|
||||
.as_ref()
|
||||
.and_then(|batches| batches.first())
|
||||
.and_then(|batch| batch.first())
|
||||
.and_then(|generation| generation.message.as_ref())
|
||||
.and_then(|message| message.kwargs.as_ref());
|
||||
if let Some(generation) = generation {
|
||||
let id = generation
|
||||
.response_metadata
|
||||
.as_ref()
|
||||
.and_then(|metadata| metadata.id.as_deref())
|
||||
.unwrap_or_default()
|
||||
.to_owned();
|
||||
return SpanIo {
|
||||
input,
|
||||
output: encode(&generation.normalized()),
|
||||
request_id: id,
|
||||
};
|
||||
}
|
||||
return SpanIo {
|
||||
input,
|
||||
output: raw_completion.to_owned(),
|
||||
request_id: String::new(),
|
||||
};
|
||||
}
|
||||
if kind == ObservationType::Tool {
|
||||
return SpanIo {
|
||||
input: raw_prompt.to_owned(),
|
||||
output: tool_output(raw_completion),
|
||||
request_id: String::new(),
|
||||
};
|
||||
}
|
||||
if kind == ObservationType::Agent {
|
||||
let input = prompt
|
||||
.messages
|
||||
.as_ref()
|
||||
.filter(|messages| !messages.agent_messages().is_empty())
|
||||
.map_or_else(
|
||||
|| raw_prompt.to_owned(),
|
||||
|messages| normalized_messages(messages.agent_messages()),
|
||||
);
|
||||
let output = completion
|
||||
.messages
|
||||
.as_ref()
|
||||
.and_then(|messages| messages.agent_messages().last())
|
||||
.map_or_else(
|
||||
|| raw_completion.to_owned(),
|
||||
|message| encode(&message.normalized()),
|
||||
);
|
||||
return SpanIo {
|
||||
input,
|
||||
output,
|
||||
request_id: String::new(),
|
||||
};
|
||||
}
|
||||
SpanIo {
|
||||
input: raw_prompt.to_owned(),
|
||||
output: raw_completion.to_owned(),
|
||||
request_id: String::new(),
|
||||
}
|
||||
}
|
||||
|
||||
impl SpanNormalizer for LangSmithNormalizer {
|
||||
fn matches(&self, scope_name: &str, attributes: &BTreeMap<String, String>) -> bool {
|
||||
scope_name == "langsmith" || attributes.contains_key("langsmith.span.kind")
|
||||
}
|
||||
|
||||
fn consumed_attributes(&self, _attributes: &BTreeMap<String, String>) -> [&'static str; 2] {
|
||||
["gen_ai.prompt", "gen_ai.completion"]
|
||||
}
|
||||
|
||||
fn normalize(
|
||||
&self,
|
||||
name: &str,
|
||||
parent_span_id: &str,
|
||||
attributes: &BTreeMap<String, String>,
|
||||
_events: &[DecodedEvent],
|
||||
) -> Result<NormalizedSpan, Error> {
|
||||
let (input_tokens, output_tokens) = usage_tokens(attributes)?;
|
||||
let observation_type = span_type(name, parent_span_id, attributes);
|
||||
let io = span_io(observation_type, attributes);
|
||||
Ok(NormalizedSpan {
|
||||
observation_type,
|
||||
agent_name: attr(attributes, "langsmith.metadata.lc_agent_name").to_owned(),
|
||||
framework: String::new(),
|
||||
litellm_request_id: io.request_id,
|
||||
model: attr(attributes, "gen_ai.request.model").to_owned(),
|
||||
input_tokens,
|
||||
output_tokens,
|
||||
input: io.input,
|
||||
output: io.output,
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use std::collections::BTreeMap;
|
||||
|
||||
use rstest::rstest;
|
||||
use serde_json::Value;
|
||||
|
||||
use super::{ObservationType, span_io};
|
||||
|
||||
#[rstest]
|
||||
fn malformed_messages_preserve_valid_input_and_response_id() {
|
||||
let attributes = BTreeMap::from([
|
||||
(
|
||||
"gen_ai.prompt".to_owned(),
|
||||
r#"{"messages":[[{"kwargs":{"type":"human","content":"hello"}},null]]}"#.to_owned(),
|
||||
),
|
||||
(
|
||||
"gen_ai.completion".to_owned(),
|
||||
r#"{"messages":"unexpected","generations":[[{"message":{"kwargs":{"type":"ai","content":"hi","response_metadata":{"id":"response-1"}}}}]]}"#.to_owned(),
|
||||
),
|
||||
]);
|
||||
let io = span_io(ObservationType::Llm, &attributes);
|
||||
let input: Value = serde_json::from_str(&io.input).expect("normalized input");
|
||||
assert_eq!(input.as_array().expect("messages").len(), 1);
|
||||
assert_eq!(input[0]["content"], "hello");
|
||||
assert_eq!(io.request_id, "response-1");
|
||||
}
|
||||
|
||||
#[rstest]
|
||||
fn explicit_null_tool_output_is_preserved() {
|
||||
let attributes = BTreeMap::from([(
|
||||
"gen_ai.completion".to_owned(),
|
||||
r#"{"output":null}"#.to_owned(),
|
||||
)]);
|
||||
let io = span_io(ObservationType::Tool, &attributes);
|
||||
assert_eq!(io.output, "null");
|
||||
}
|
||||
|
||||
#[rstest]
|
||||
fn absent_llm_messages_render_as_an_empty_list() {
|
||||
let attributes = BTreeMap::from([("gen_ai.completion".to_owned(), "{}".to_owned())]);
|
||||
let io = span_io(ObservationType::Llm, &attributes);
|
||||
assert_eq!(io.input, "[]");
|
||||
}
|
||||
}
|
||||
627
litellm-rust/crates/traces/src/normalize/messages.rs
Normal file
627
litellm-rust/crates/traces/src/normalize/messages.rs
Normal file
|
|
@ -0,0 +1,627 @@
|
|||
//! The common message format normalizers emit for span input and output: a JSON array of
|
||||
//! `{role, content, tool_calls?, name?}` that the UI renders as a conversation.
|
||||
|
||||
use indexmap::IndexMap;
|
||||
use serde::{Deserialize, Deserializer, Serialize};
|
||||
use serde_json::{Value, ser::Formatter};
|
||||
use std::{
|
||||
collections::{BTreeMap, BTreeSet},
|
||||
io,
|
||||
};
|
||||
|
||||
use litellm_llms_types::{formats::chat_completions::ChatMessageContent, recognized::Recognized};
|
||||
|
||||
use super::{CallEvidence, CallKey, attr};
|
||||
|
||||
/// Characters of a span's input kept for list views.
|
||||
pub(super) const PREVIEW_CHARS: usize = 240;
|
||||
|
||||
/// Content blocks that carry no display text: reasoning and the model's own tool requests.
|
||||
pub(crate) const HIDDEN_BLOCK_TYPES: [&str; 6] = [
|
||||
"reasoning",
|
||||
"thinking",
|
||||
"redacted_thinking",
|
||||
"function_call",
|
||||
"tool_use",
|
||||
"tool_call",
|
||||
];
|
||||
|
||||
fn display_text(content: &Recognized<ChatMessageContent>) -> String {
|
||||
match content {
|
||||
Recognized::Known(ChatMessageContent::Text(text)) => text.clone(),
|
||||
Recognized::Known(ChatMessageContent::Parts(blocks)) => blocks
|
||||
.iter()
|
||||
.filter(|block| {
|
||||
!block
|
||||
.get("type")
|
||||
.and_then(Value::as_str)
|
||||
.is_some_and(|kind| HIDDEN_BLOCK_TYPES.contains(&kind))
|
||||
})
|
||||
.filter_map(|block| block.get("text").and_then(Value::as_str))
|
||||
.collect::<Vec<_>>()
|
||||
.join("\n\n"),
|
||||
Recognized::Unrecognized(value) => encode(value),
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Clone, Deserialize, Serialize)]
|
||||
#[serde(transparent)]
|
||||
pub(super) struct ToolCall(IndexMap<String, Value>);
|
||||
|
||||
#[derive(Deserialize)]
|
||||
pub(super) struct ResponseMetadata {
|
||||
pub id: Option<String>,
|
||||
}
|
||||
|
||||
#[derive(Deserialize)]
|
||||
#[serde(untagged)]
|
||||
pub(crate) enum MessagePayload<T> {
|
||||
Single {
|
||||
#[serde(flatten)]
|
||||
message: T,
|
||||
},
|
||||
Batch(Vec<T>),
|
||||
}
|
||||
|
||||
impl<T> MessagePayload<T> {
|
||||
pub(crate) fn into_messages(self) -> Vec<T> {
|
||||
match self {
|
||||
Self::Single { message } => vec![message],
|
||||
Self::Batch(messages) => messages,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
pub(super) fn present<'de, D, T>(deserializer: D) -> Result<Option<T>, D::Error>
|
||||
where
|
||||
D: Deserializer<'de>,
|
||||
T: Deserialize<'de>,
|
||||
{
|
||||
T::deserialize(deserializer).map(Some)
|
||||
}
|
||||
|
||||
#[derive(Default, Deserialize)]
|
||||
struct EventFields {
|
||||
#[serde(default, deserialize_with = "present")]
|
||||
role: Option<Value>,
|
||||
#[serde(default, deserialize_with = "present")]
|
||||
content: Option<Value>,
|
||||
#[serde(default, deserialize_with = "present")]
|
||||
tool_calls: Option<Value>,
|
||||
#[serde(flatten)]
|
||||
indexed: BTreeMap<String, Value>,
|
||||
}
|
||||
|
||||
#[derive(Deserialize)]
|
||||
pub(super) struct EventMessage {
|
||||
#[serde(rename = "event.name")]
|
||||
name: Option<Recognized<String>>,
|
||||
#[serde(default, deserialize_with = "present")]
|
||||
message: Option<Recognized<EventFields>>,
|
||||
#[serde(rename = "message.role", default, deserialize_with = "present")]
|
||||
role: Option<Value>,
|
||||
#[serde(rename = "message.content", default, deserialize_with = "present")]
|
||||
content: Option<Value>,
|
||||
#[serde(flatten)]
|
||||
body: EventFields,
|
||||
}
|
||||
|
||||
impl EventMessage {
|
||||
pub(super) fn recorded(&self) -> Option<(bool, Value)> {
|
||||
self.normalized(self.name.as_ref()?.known()?)
|
||||
}
|
||||
|
||||
fn normalized(&self, name: &str) -> Option<(bool, Value)> {
|
||||
let (output, role) = match name {
|
||||
"gen_ai.system.message" => (false, "system"),
|
||||
"gen_ai.user.message" | "gen_ai.content.prompt" => (false, "user"),
|
||||
"gen_ai.assistant.message" | "gen_ai.choice" | "gen_ai.content.completion" => {
|
||||
(true, "assistant")
|
||||
}
|
||||
"gen_ai.tool.message" => (true, "tool"),
|
||||
_ => return None,
|
||||
};
|
||||
let empty = EventFields::default();
|
||||
let body = match &self.message {
|
||||
Some(Recognized::Known(message)) => message,
|
||||
Some(Recognized::Unrecognized(_)) => &empty,
|
||||
None => &self.body,
|
||||
};
|
||||
let content = body.content.as_ref().or(self.content.as_ref());
|
||||
let calls = event_tool_calls(body);
|
||||
if content.is_none() && calls.is_none() {
|
||||
return None;
|
||||
}
|
||||
Some((
|
||||
output,
|
||||
serde_json::json!({
|
||||
"role": body.role.as_ref().or(self.role.as_ref()).cloned().unwrap_or(Value::from(role)),
|
||||
"content": content.cloned().unwrap_or(Value::from("")),
|
||||
"tool_calls": calls,
|
||||
}),
|
||||
))
|
||||
}
|
||||
}
|
||||
|
||||
/// One part of an OpenTelemetry GenAI (`type` + `content`) or Gemini (`text`) message.
|
||||
#[derive(Deserialize)]
|
||||
struct Part {
|
||||
#[serde(rename = "type")]
|
||||
kind: Option<String>,
|
||||
content: Option<Value>,
|
||||
text: Option<String>,
|
||||
id: Option<Value>,
|
||||
name: Option<String>,
|
||||
arguments: Option<Value>,
|
||||
response: Option<Value>,
|
||||
}
|
||||
|
||||
/// A message as instrumentations record it: OpenAI chat (`role` + `content`), LangChain
|
||||
/// (`type`, wrapped in `kwargs` by `dumpd` or `data` by `messages_to_dict`), or OpenTelemetry
|
||||
/// GenAI and Gemini (`role` + `parts`).
|
||||
#[derive(Deserialize)]
|
||||
pub(super) struct RawMessage {
|
||||
kwargs: Option<Box<RawMessage>>,
|
||||
data: Option<Box<RawMessage>>,
|
||||
#[serde(rename = "type")]
|
||||
kind: Option<String>,
|
||||
role: Option<String>,
|
||||
content: Option<Recognized<ChatMessageContent>>,
|
||||
parts: Option<Vec<Part>>,
|
||||
tool_calls: Option<Vec<ToolCall>>,
|
||||
name: Option<Value>,
|
||||
pub response_metadata: Option<ResponseMetadata>,
|
||||
}
|
||||
|
||||
#[derive(Serialize)]
|
||||
pub(super) struct Message {
|
||||
role: String,
|
||||
content: String,
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
tool_calls: Option<Vec<ToolCall>>,
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
name: Option<Value>,
|
||||
}
|
||||
|
||||
impl RawMessage {
|
||||
pub(super) fn unwrapped(&self) -> &Self {
|
||||
self.kwargs
|
||||
.as_deref()
|
||||
.or(self.data.as_deref())
|
||||
.unwrap_or(self)
|
||||
}
|
||||
|
||||
fn role(&self) -> &str {
|
||||
let fields = self.unwrapped();
|
||||
let raw = fields
|
||||
.kind
|
||||
.as_deref()
|
||||
.filter(|role| !role.is_empty())
|
||||
.or_else(|| fields.role.as_deref().filter(|role| !role.is_empty()))
|
||||
.or_else(|| self.kind.as_deref().filter(|role| !role.is_empty()))
|
||||
.unwrap_or_default();
|
||||
match raw {
|
||||
"human" => "user",
|
||||
"ai" | "model" => "assistant",
|
||||
other => other,
|
||||
}
|
||||
}
|
||||
|
||||
fn is_message(&self) -> bool {
|
||||
let fields = self.unwrapped();
|
||||
!self.role().is_empty()
|
||||
&& (fields.content.is_some() || fields.parts.is_some() || fields.tool_calls.is_some())
|
||||
}
|
||||
|
||||
pub(super) fn normalized(&self) -> Message {
|
||||
let fields = self.unwrapped();
|
||||
let role = self.role().to_owned();
|
||||
let parts = fields.parts.as_deref().unwrap_or_default();
|
||||
let content = match &fields.content {
|
||||
Some(content) => display_text(content),
|
||||
None => parts
|
||||
.iter()
|
||||
.filter_map(Part::text)
|
||||
.collect::<Vec<_>>()
|
||||
.join("\n\n"),
|
||||
};
|
||||
let tool_calls = fields
|
||||
.tool_calls
|
||||
.clone()
|
||||
.unwrap_or_else(|| parts.iter().filter_map(Part::tool_call).collect());
|
||||
Message {
|
||||
name: (role == "tool")
|
||||
.then_some(fields.name.clone())
|
||||
.flatten()
|
||||
.filter(|name| !name.is_null() && name != &Value::String(String::new())),
|
||||
role,
|
||||
content,
|
||||
tool_calls: (!tool_calls.is_empty()).then_some(tool_calls),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl Part {
|
||||
fn text(&self) -> Option<String> {
|
||||
match self.kind.as_deref().unwrap_or("text") {
|
||||
"text" => self
|
||||
.text
|
||||
.clone()
|
||||
.or_else(|| self.content.as_ref().map(display_value)),
|
||||
"tool_call_response" => self.response.as_ref().map(display_value),
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
|
||||
fn tool_call(&self) -> Option<ToolCall> {
|
||||
(self.kind.as_deref() == Some("tool_call")).then(|| {
|
||||
ToolCall(IndexMap::from([
|
||||
(
|
||||
"name".to_owned(),
|
||||
Value::from(self.name.clone().unwrap_or_default()),
|
||||
),
|
||||
(
|
||||
"arguments".to_owned(),
|
||||
self.arguments.clone().unwrap_or(Value::Null),
|
||||
),
|
||||
("id".to_owned(), self.id.clone().unwrap_or(Value::Null)),
|
||||
]))
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
fn display_value(value: &Value) -> String {
|
||||
value.as_str().map_or_else(|| encode(value), str::to_owned)
|
||||
}
|
||||
|
||||
/// The conversation `value` holds: an array of messages or a single message.
|
||||
pub(super) fn parse(value: &Value) -> Option<Vec<Message>> {
|
||||
let raw = MessagePayload::<RawMessage>::deserialize(value)
|
||||
.ok()?
|
||||
.into_messages();
|
||||
(!raw.is_empty() && raw.iter().all(RawMessage::is_message))
|
||||
.then(|| raw.iter().map(RawMessage::normalized).collect())
|
||||
}
|
||||
|
||||
/// OpenInference's flattened `<prefix>.<i>.message.{role,content,contents,tool_calls}` attributes.
|
||||
pub(super) fn flattened(
|
||||
attributes: &BTreeMap<String, String>,
|
||||
prefix: &str,
|
||||
) -> Option<Vec<Message>> {
|
||||
let messages: Vec<Message> = (0..)
|
||||
.map(|index| format!("{prefix}.{index}.message."))
|
||||
.take_while(|message| {
|
||||
attributes
|
||||
.keys()
|
||||
.any(|key| key.starts_with(message.as_str()))
|
||||
})
|
||||
.map(|message| {
|
||||
let field = |name: &str| attr(attributes, &format!("{message}{name}")).to_owned();
|
||||
let content = if field("content").is_empty() {
|
||||
(0..)
|
||||
.map(|part| field(&format!("contents.{part}.message_content.text")))
|
||||
.take_while(|text| !text.is_empty())
|
||||
.collect::<Vec<_>>()
|
||||
.join("\n\n")
|
||||
} else {
|
||||
field("content")
|
||||
};
|
||||
let tool_calls: Vec<ToolCall> = (0..)
|
||||
.map(|call| format!("tool_calls.{call}.tool_call."))
|
||||
.take_while(|call| !field(&format!("{call}function.name")).is_empty())
|
||||
.map(|call| {
|
||||
ToolCall(IndexMap::from([
|
||||
(
|
||||
"name".to_owned(),
|
||||
Value::from(field(&format!("{call}function.name"))),
|
||||
),
|
||||
(
|
||||
"arguments".to_owned(),
|
||||
Value::from(field(&format!("{call}function.arguments"))),
|
||||
),
|
||||
("id".to_owned(), Value::from(field(&format!("{call}id")))),
|
||||
]))
|
||||
})
|
||||
.collect();
|
||||
Message {
|
||||
role: field("role"),
|
||||
content,
|
||||
tool_calls: (!tool_calls.is_empty()).then_some(tool_calls),
|
||||
name: Some(field("name"))
|
||||
.filter(|name| !name.is_empty())
|
||||
.map(Value::from),
|
||||
}
|
||||
})
|
||||
.collect();
|
||||
(!messages.is_empty()).then_some(messages)
|
||||
}
|
||||
|
||||
pub(super) fn indexed(attributes: &BTreeMap<String, String>, prefix: &str) -> Option<String> {
|
||||
let indices: BTreeSet<usize> = attributes
|
||||
.keys()
|
||||
.filter_map(|key| {
|
||||
key.strip_prefix(prefix)?
|
||||
.strip_prefix('.')?
|
||||
.split('.')
|
||||
.next()?
|
||||
.parse()
|
||||
.ok()
|
||||
})
|
||||
.collect();
|
||||
let values: Vec<Value> = indices
|
||||
.into_iter()
|
||||
.filter_map(|index| {
|
||||
let base = format!("{prefix}.{index}.");
|
||||
let fields = Value::Object(
|
||||
attributes
|
||||
.range(base.clone()..)
|
||||
.take_while(|(key, _)| key.starts_with(&base))
|
||||
.filter_map(|(key, value)| {
|
||||
let suffix = key.strip_prefix(&base)?;
|
||||
Some((
|
||||
suffix.strip_prefix("message.").unwrap_or(suffix).to_owned(),
|
||||
Value::from(value.clone()),
|
||||
))
|
||||
})
|
||||
.collect(),
|
||||
);
|
||||
let message = EventFields::deserialize(&fields).ok()?;
|
||||
let calls = event_tool_calls(&message);
|
||||
if message.content.is_none() && calls.is_none() {
|
||||
return None;
|
||||
}
|
||||
Some(serde_json::json!({
|
||||
"role": message.role?,
|
||||
"content": message.content.unwrap_or(Value::from("")),
|
||||
"tool_calls": calls,
|
||||
}))
|
||||
})
|
||||
.collect();
|
||||
(!values.is_empty()).then(|| canonical(&encode(&values)))
|
||||
}
|
||||
|
||||
fn event_tool_calls(value: &EventFields) -> Option<Value> {
|
||||
if let Some(calls) = &value.tool_calls {
|
||||
return Some(calls.clone());
|
||||
}
|
||||
let indices: BTreeSet<usize> = value
|
||||
.indexed
|
||||
.keys()
|
||||
.filter_map(|key| {
|
||||
key.strip_prefix("tool_calls.")?
|
||||
.split('.')
|
||||
.next()?
|
||||
.parse()
|
||||
.ok()
|
||||
})
|
||||
.collect();
|
||||
let calls: Vec<Value> = indices
|
||||
.into_iter()
|
||||
.filter_map(|index| {
|
||||
let prefix = format!("tool_calls.{index}");
|
||||
Some(serde_json::json!({
|
||||
"id": value.indexed.get(&format!("{prefix}.id")),
|
||||
"name": value.indexed.get(&format!("{prefix}.function.name"))?,
|
||||
"arguments": value.indexed.get(&format!("{prefix}.function.arguments")),
|
||||
}))
|
||||
})
|
||||
.collect();
|
||||
(!calls.is_empty()).then_some(Value::Array(calls))
|
||||
}
|
||||
|
||||
pub(super) fn event_message(name: &str, value: &Value) -> Option<(bool, Value)> {
|
||||
EventMessage::deserialize(value).ok()?.normalized(name)
|
||||
}
|
||||
|
||||
pub(super) fn event_payload(events: &[(bool, Value)], output: bool) -> Option<String> {
|
||||
let values: Vec<&Value> = events
|
||||
.iter()
|
||||
.filter(|(direction, _)| *direction == output)
|
||||
.map(|(_, value)| value)
|
||||
.collect();
|
||||
(!values.is_empty()).then(|| canonical(&encode(&values)))
|
||||
}
|
||||
|
||||
/// The latest user message with text.
|
||||
pub(super) fn preview(messages: &[Message]) -> String {
|
||||
messages
|
||||
.iter()
|
||||
.rev()
|
||||
.find(|message| message.role == "user" && !message.content.is_empty())
|
||||
.map_or("", |message| message.content.as_str())
|
||||
.chars()
|
||||
.take(PREVIEW_CHARS)
|
||||
.collect()
|
||||
}
|
||||
|
||||
/// The latest user message when `input` is a conversation, else the input itself.
|
||||
pub(super) fn input_preview(input: &str) -> String {
|
||||
match serde_json::from_str::<Value>(input)
|
||||
.ok()
|
||||
.and_then(|value| parse(&value))
|
||||
{
|
||||
Some(messages) => preview(&messages),
|
||||
None => input.chars().take(PREVIEW_CHARS).collect(),
|
||||
}
|
||||
}
|
||||
|
||||
/// `raw` in the common format when it holds a conversation, else unchanged.
|
||||
pub(super) fn canonical(raw: &str) -> String {
|
||||
serde_json::from_str::<Value>(raw)
|
||||
.ok()
|
||||
.and_then(|value| parse(&value))
|
||||
.map_or_else(|| raw.to_owned(), |messages| encode(&messages))
|
||||
}
|
||||
|
||||
#[derive(Deserialize)]
|
||||
struct LlmOutput {
|
||||
id: Option<String>,
|
||||
}
|
||||
|
||||
#[derive(Deserialize)]
|
||||
struct Generation {
|
||||
message: RawMessage,
|
||||
}
|
||||
|
||||
#[derive(Deserialize)]
|
||||
struct LlmResult {
|
||||
generations: Vec<Recognized<Vec<Recognized<Generation>>>>,
|
||||
llm_output: Option<LlmOutput>,
|
||||
}
|
||||
|
||||
/// A LangChain `LLMResult`'s first generation and the requests behind it.
|
||||
pub(super) struct Generations {
|
||||
pub first: Option<Message>,
|
||||
pub calls: CallEvidence,
|
||||
}
|
||||
|
||||
/// LangChain `LLMResult`: `generations[prompt][candidate]`. Each prompt is one provider request,
|
||||
/// whose candidates share its response id (`response_metadata.id`; `llm_output.id` for a single
|
||||
/// prompt). The evidence is complete only when every prompt yields exactly one id and no entry
|
||||
/// failed to parse.
|
||||
pub(super) fn langchain_result(value: &Value) -> Option<Generations> {
|
||||
let result = LlmResult::deserialize(value).ok()?;
|
||||
let mut complete = true;
|
||||
let mut first = None;
|
||||
let mut keys = BTreeSet::new();
|
||||
for prompt in &result.generations {
|
||||
let Recognized::Known(candidates) = prompt else {
|
||||
complete = false;
|
||||
continue;
|
||||
};
|
||||
let mut ids = BTreeSet::new();
|
||||
for candidate in candidates {
|
||||
match candidate {
|
||||
Recognized::Known(generation) => {
|
||||
let message = generation.message.unwrapped();
|
||||
if let Some(id) = message
|
||||
.response_metadata
|
||||
.as_ref()
|
||||
.and_then(|metadata| metadata.id.clone())
|
||||
{
|
||||
ids.insert(id);
|
||||
}
|
||||
if first.is_none() {
|
||||
first = Some(generation.message.normalized());
|
||||
}
|
||||
}
|
||||
Recognized::Unrecognized(_) => complete = false,
|
||||
}
|
||||
}
|
||||
if ids.is_empty()
|
||||
&& result.generations.len() == 1
|
||||
&& let Some(id) = result
|
||||
.llm_output
|
||||
.as_ref()
|
||||
.and_then(|output| output.id.clone())
|
||||
{
|
||||
ids.insert(id);
|
||||
}
|
||||
complete &= ids.len() == 1;
|
||||
keys.extend(ids.into_iter().map(CallKey::ProviderResponse));
|
||||
}
|
||||
let calls = match (keys.is_empty(), complete && !result.generations.is_empty()) {
|
||||
(true, _) => CallEvidence::Unknown,
|
||||
(false, true) => CallEvidence::Complete(keys),
|
||||
(false, false) => CallEvidence::Partial(keys),
|
||||
};
|
||||
Some(Generations { first, calls })
|
||||
}
|
||||
|
||||
struct PythonJsonFormatter;
|
||||
|
||||
impl Formatter for PythonJsonFormatter {
|
||||
fn begin_array_value<W: ?Sized + io::Write>(
|
||||
&mut self,
|
||||
writer: &mut W,
|
||||
first: bool,
|
||||
) -> io::Result<()> {
|
||||
if first {
|
||||
Ok(())
|
||||
} else {
|
||||
writer.write_all(b", ")
|
||||
}
|
||||
}
|
||||
|
||||
fn begin_object_key<W: ?Sized + io::Write>(
|
||||
&mut self,
|
||||
writer: &mut W,
|
||||
first: bool,
|
||||
) -> io::Result<()> {
|
||||
if first {
|
||||
Ok(())
|
||||
} else {
|
||||
writer.write_all(b", ")
|
||||
}
|
||||
}
|
||||
|
||||
fn begin_object_value<W: ?Sized + io::Write>(&mut self, writer: &mut W) -> io::Result<()> {
|
||||
writer.write_all(b": ")
|
||||
}
|
||||
}
|
||||
|
||||
pub(crate) fn encode<T: Serialize>(value: &T) -> String {
|
||||
let mut output = Vec::new();
|
||||
let mut serializer = serde_json::Serializer::with_formatter(&mut output, PythonJsonFormatter);
|
||||
if value.serialize(&mut serializer).is_err() {
|
||||
return String::new();
|
||||
}
|
||||
String::from_utf8(output).unwrap_or_default()
|
||||
}
|
||||
|
||||
pub(super) fn state_preview(input: &str, key: &str) -> Option<String> {
|
||||
let object = serde_json::from_str::<serde_json::Map<String, Value>>(input).ok()?;
|
||||
let conversation = parse(object.get(key)?)?;
|
||||
Some(preview(&conversation))
|
||||
}
|
||||
|
||||
pub(super) fn state_conversation(input: &str) -> Option<Vec<Message>> {
|
||||
let value: Value = serde_json::from_str(input).ok()?;
|
||||
let items = value.get("messages")?.as_array()?;
|
||||
if items.first().is_some_and(Value::is_array) {
|
||||
return None;
|
||||
}
|
||||
let conversation: Vec<Message> = items
|
||||
.iter()
|
||||
.filter_map(|item| RawMessage::deserialize(item).ok())
|
||||
.map(|message| message.normalized())
|
||||
.collect();
|
||||
(!conversation.is_empty()).then_some(conversation)
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use rstest::rstest;
|
||||
|
||||
use super::{state_conversation, state_preview};
|
||||
use serde_json::Value;
|
||||
|
||||
#[rstest]
|
||||
#[case::latest_user(r#"{"messages":[{"role":"user","content":"first"},{"role":"assistant","content":"reply"},{"role":"user","content":"last"}]}"#, Some("last"))]
|
||||
#[case::malformed("not-json", None)]
|
||||
#[case::missing("{}", None)]
|
||||
#[case::not_messages(r#"{"messages":[{"role":"user"}]}"#, None)]
|
||||
fn state_preview_requires_a_valid_conversation(
|
||||
#[case] input: &str,
|
||||
#[case] expected: Option<&str>,
|
||||
) {
|
||||
assert_eq!(state_preview(input, "messages").as_deref(), expected);
|
||||
}
|
||||
|
||||
#[rstest]
|
||||
#[case::lenient_flat(
|
||||
r#"{"messages":[null,{"type":"human","content":"hello"}]}"#,
|
||||
Some(r#"[{"role":"user","content":"hello"}]"#)
|
||||
)]
|
||||
#[case::nested(r#"{"messages":[[{"role":"user","content":"hello"}]]}"#, None)]
|
||||
#[case::empty(r#"{"messages":[]}"#, None)]
|
||||
fn state_conversation_preserves_flat_batch_semantics(
|
||||
#[case] input: &str,
|
||||
#[case] expected: Option<&str>,
|
||||
) {
|
||||
let observed =
|
||||
state_conversation(input).map(|messages| serde_json::to_value(messages).unwrap());
|
||||
let expected_value = expected.map(|value| serde_json::from_str::<Value>(value).unwrap());
|
||||
assert_eq!(observed, expected_value);
|
||||
}
|
||||
}
|
||||
194
litellm-rust/crates/traces/src/normalize/metadata.rs
Normal file
194
litellm-rust/crates/traces/src/normalize/metadata.rs
Normal file
|
|
@ -0,0 +1,194 @@
|
|||
use std::collections::BTreeMap;
|
||||
|
||||
use serde::{Deserialize, Deserializer, Serialize, de::DeserializeOwned};
|
||||
use serde_json::{Map, Value};
|
||||
|
||||
use super::{SpanContext, attr};
|
||||
|
||||
#[derive(Clone, Copy, Debug, Deserialize, Eq, PartialEq, Serialize)]
|
||||
#[serde(rename_all = "snake_case")]
|
||||
pub enum AgentType {
|
||||
Root,
|
||||
Subagent,
|
||||
Middleware,
|
||||
Compaction,
|
||||
}
|
||||
|
||||
#[derive(
|
||||
Clone, Debug, Eq, PartialEq, Serialize, Deserialize, strum::EnumString, strum::Display,
|
||||
)]
|
||||
#[serde(from = "String", into = "String")]
|
||||
#[strum(serialize_all = "kebab-case")]
|
||||
pub enum Integration {
|
||||
ClaudeCode,
|
||||
ClaudeAgentSdk,
|
||||
OpenaiCodex,
|
||||
DeepagentsCode,
|
||||
Cursor,
|
||||
Pi,
|
||||
Opencode,
|
||||
Copilot,
|
||||
Langchain,
|
||||
Langgraph,
|
||||
Deepagents,
|
||||
Autogen,
|
||||
Crewai,
|
||||
GoogleAdk,
|
||||
LlamaIndex,
|
||||
Mastra,
|
||||
MicrosoftAgentFramework,
|
||||
OpenaiAgents,
|
||||
PydanticAi,
|
||||
SemanticKernel,
|
||||
Strands,
|
||||
VercelAiSdk,
|
||||
Instructor,
|
||||
N8n,
|
||||
Temporal,
|
||||
#[strum(default)]
|
||||
Other(String),
|
||||
}
|
||||
|
||||
impl From<String> for Integration {
|
||||
fn from(value: String) -> Self {
|
||||
Self::from(value.as_str())
|
||||
}
|
||||
}
|
||||
|
||||
impl From<Integration> for String {
|
||||
fn from(value: Integration) -> Self {
|
||||
value.to_string()
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Debug, Default, Deserialize, Eq, PartialEq, Serialize)]
|
||||
#[serde(default)]
|
||||
pub struct AgentMetadata {
|
||||
#[serde(deserialize_with = "optional")]
|
||||
pub lc_agent_name: Option<String>,
|
||||
#[serde(deserialize_with = "optional")]
|
||||
pub ls_integration: Option<Integration>,
|
||||
#[serde(deserialize_with = "optional")]
|
||||
pub ls_agent_type: Option<AgentType>,
|
||||
#[serde(deserialize_with = "optional")]
|
||||
pub ls_agent_purpose: Option<String>,
|
||||
#[serde(deserialize_with = "optional")]
|
||||
pub ls_agent_runtime: Option<String>,
|
||||
#[serde(deserialize_with = "optional")]
|
||||
pub ls_agent_version: Option<String>,
|
||||
#[serde(deserialize_with = "optional")]
|
||||
pub ls_trace_schema_version: Option<String>,
|
||||
#[serde(deserialize_with = "optional")]
|
||||
pub thread_id: Option<String>,
|
||||
#[serde(deserialize_with = "optional")]
|
||||
pub ls_subagent_id: Option<String>,
|
||||
#[serde(deserialize_with = "optional")]
|
||||
pub ls_subagent_type: Option<String>,
|
||||
#[serde(deserialize_with = "optional")]
|
||||
pub ls_tool_name: Option<String>,
|
||||
#[serde(deserialize_with = "optional")]
|
||||
pub ls_model_name: Option<String>,
|
||||
#[serde(deserialize_with = "optional")]
|
||||
pub ls_provider: Option<String>,
|
||||
#[serde(deserialize_with = "optional")]
|
||||
pub git_branch: Option<String>,
|
||||
#[serde(deserialize_with = "optional")]
|
||||
pub git_commit_sha: Option<String>,
|
||||
#[serde(deserialize_with = "optional")]
|
||||
pub git_repo_url: Option<String>,
|
||||
#[serde(deserialize_with = "optional")]
|
||||
pub working_directory: Option<String>,
|
||||
}
|
||||
|
||||
impl AgentMetadata {
|
||||
pub(crate) fn byte_len(&self) -> usize {
|
||||
let strings = [
|
||||
&self.lc_agent_name,
|
||||
&self.ls_agent_purpose,
|
||||
&self.ls_agent_runtime,
|
||||
&self.ls_agent_version,
|
||||
&self.ls_trace_schema_version,
|
||||
&self.thread_id,
|
||||
&self.ls_subagent_id,
|
||||
&self.ls_subagent_type,
|
||||
&self.ls_tool_name,
|
||||
&self.ls_model_name,
|
||||
&self.ls_provider,
|
||||
&self.git_branch,
|
||||
&self.git_commit_sha,
|
||||
&self.git_repo_url,
|
||||
&self.working_directory,
|
||||
];
|
||||
strings
|
||||
.into_iter()
|
||||
.filter_map(Option::as_ref)
|
||||
.map(String::len)
|
||||
.sum::<usize>()
|
||||
+ self
|
||||
.ls_integration
|
||||
.as_ref()
|
||||
.map_or(0, |integration| integration.to_string().len())
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(strum::EnumString, strum::IntoStaticStr)]
|
||||
#[strum(serialize_all = "snake_case")]
|
||||
enum MetadataField {
|
||||
LcAgentName,
|
||||
LsIntegration,
|
||||
LsAgentType,
|
||||
LsAgentPurpose,
|
||||
LsAgentRuntime,
|
||||
#[strum(serialize = "ls_agent_runtime_version", to_string = "ls_agent_version")]
|
||||
LsAgentVersion,
|
||||
LsTraceSchemaVersion,
|
||||
ThreadId,
|
||||
LsSubagentId,
|
||||
LsSubagentType,
|
||||
LsToolName,
|
||||
LsModelName,
|
||||
LsProvider,
|
||||
GitBranch,
|
||||
GitCommitSha,
|
||||
#[strum(serialize = "repository_url", to_string = "git_repo_url")]
|
||||
GitRepoUrl,
|
||||
#[strum(serialize = "cwd", to_string = "working_directory")]
|
||||
WorkingDirectory,
|
||||
}
|
||||
|
||||
fn optional<'de, D: Deserializer<'de>, T: DeserializeOwned>(
|
||||
deserializer: D,
|
||||
) -> Result<Option<T>, D::Error> {
|
||||
let value = Value::deserialize(deserializer)?;
|
||||
Ok(serde_json::from_value(value).ok())
|
||||
}
|
||||
|
||||
fn field(key: &str, value: impl FnOnce() -> Value) -> Option<(String, Value)> {
|
||||
let canonical: &'static str = MetadataField::try_from(key).ok()?.into();
|
||||
let value = value();
|
||||
if value.is_null() || value.as_str().is_some_and(str::is_empty) {
|
||||
return None;
|
||||
}
|
||||
Some((canonical.to_owned(), value))
|
||||
}
|
||||
|
||||
pub(super) fn extract(context: &SpanContext<'_>) -> AgentMetadata {
|
||||
let nested = serde_json::from_str::<Map<String, Value>>(attr(context.attributes, "metadata"))
|
||||
.unwrap_or_default();
|
||||
let values: BTreeMap<String, Value> = nested
|
||||
.into_iter()
|
||||
.filter_map(|(key, value)| field(&key, || value))
|
||||
.chain(
|
||||
context
|
||||
.attributes
|
||||
.iter()
|
||||
.filter_map(|(key, value)| field(key, || Value::String(value.clone()))),
|
||||
)
|
||||
.chain(context.attributes.iter().filter_map(|(key, value)| {
|
||||
field(key.strip_prefix("langsmith.metadata.")?, || {
|
||||
Value::String(value.clone())
|
||||
})
|
||||
}))
|
||||
.collect();
|
||||
serde_json::from_value(Value::Object(values.into_iter().collect())).unwrap_or_default()
|
||||
}
|
||||
|
|
@ -1,76 +1,249 @@
|
|||
use std::collections::BTreeMap;
|
||||
//! Span normalization in two steps: a [`format::Format`] extracts what a span records in its format,
|
||||
//! then an [`Instrumentation`] interprets those facts with what is known about the SDK that emitted
|
||||
//! it. Relationships between spans (wrappers, ownership, spend) are resolved later, over the whole
|
||||
//! trace, because parents and children can arrive in separate exports.
|
||||
|
||||
use std::{
|
||||
collections::{BTreeMap, BTreeSet},
|
||||
fmt,
|
||||
str::FromStr,
|
||||
};
|
||||
|
||||
use crate::{Error, otlp::DecodedEvent};
|
||||
use serde::{Deserialize, Serialize};
|
||||
use serde::{Deserialize, Serialize, Serializer};
|
||||
|
||||
#[derive(Clone, Copy, Debug, Eq, PartialEq, Serialize)]
|
||||
mod format;
|
||||
mod instrumentation;
|
||||
mod messages;
|
||||
mod metadata;
|
||||
|
||||
pub(crate) const CLAUDE_CODE_SCOPE: &str = "com.anthropic.claude_code.tracing";
|
||||
pub(crate) const CLAUDE_CODE_AGENT: &str = "claude-code";
|
||||
use instrumentation::Instrumentation;
|
||||
pub(crate) use messages::{HIDDEN_BLOCK_TYPES, MessagePayload, encode};
|
||||
pub use metadata::{AgentMetadata, AgentType, Integration};
|
||||
|
||||
#[macro_rules_attribute::apply(wire_type)]
|
||||
#[derive(Clone, Copy, Debug, Eq, PartialEq, strum::EnumString)]
|
||||
#[serde(rename_all = "lowercase")]
|
||||
#[strum(serialize_all = "lowercase", ascii_case_insensitive)]
|
||||
#[cfg_attr(feature = "schema", schemars(rename = "SpanType"))]
|
||||
pub enum ObservationType {
|
||||
Agent,
|
||||
Llm,
|
||||
Tool,
|
||||
Chain,
|
||||
Framework,
|
||||
Retriever,
|
||||
Embedding,
|
||||
Reranker,
|
||||
Guardrail,
|
||||
Evaluator,
|
||||
Prompt,
|
||||
Decision,
|
||||
}
|
||||
|
||||
/// A model request a span stands for, by the identifier its instrumentation recorded.
|
||||
#[derive(Clone, Debug, Deserialize, Eq, Ord, PartialEq, PartialOrd)]
|
||||
#[serde(try_from = "String")]
|
||||
pub enum CallKey {
|
||||
/// LiteLLM's own id for the request (`spend_logs.request_id`).
|
||||
LiteLlmRequest(String),
|
||||
/// The provider response id returned to the caller (`spend_logs.response_id`).
|
||||
ProviderResponse(String),
|
||||
/// The span is the HTTP request itself; LiteLLM logs its `traceparent` span id.
|
||||
Transport,
|
||||
}
|
||||
|
||||
impl fmt::Display for CallKey {
|
||||
fn fmt(&self, formatter: &mut fmt::Formatter<'_>) -> fmt::Result {
|
||||
match self {
|
||||
Self::LiteLlmRequest(id) => write!(formatter, "litellm_request:{id}"),
|
||||
Self::ProviderResponse(id) => write!(formatter, "provider_response:{id}"),
|
||||
Self::Transport => formatter.write_str("transport:"),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl FromStr for CallKey {
|
||||
type Err = crate::InvalidCallKey;
|
||||
|
||||
fn from_str(encoded: &str) -> Result<Self, Self::Err> {
|
||||
match encoded.split_once(':') {
|
||||
Some(("provider_response", id)) if !id.is_empty() => {
|
||||
Ok(Self::ProviderResponse(id.to_owned()))
|
||||
}
|
||||
Some(("litellm_request", id)) if !id.is_empty() => {
|
||||
Ok(Self::LiteLlmRequest(id.to_owned()))
|
||||
}
|
||||
Some(("transport", "")) => Ok(Self::Transport),
|
||||
_ => Err(crate::InvalidCallKey),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl TryFrom<String> for CallKey {
|
||||
type Error = crate::InvalidCallKey;
|
||||
|
||||
fn try_from(value: String) -> Result<Self, Self::Error> {
|
||||
value.parse()
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Clone, Copy, Debug, Deserialize, Eq, PartialEq, Serialize)]
|
||||
#[serde(rename_all = "lowercase")]
|
||||
pub enum CallEvidenceKind {
|
||||
Unknown,
|
||||
Partial,
|
||||
Complete,
|
||||
}
|
||||
|
||||
impl Serialize for CallKey {
|
||||
fn serialize<S: Serializer>(&self, serializer: S) -> Result<S::Ok, S::Error> {
|
||||
serializer.collect_str(self)
|
||||
}
|
||||
}
|
||||
|
||||
/// Which model requests a span accounts for. `Complete` comes only from an instrumentation's known
|
||||
/// contract (one chat span is one response), never from how many ids happened to be found.
|
||||
#[derive(Clone, Debug, Default, Eq, PartialEq, Serialize)]
|
||||
pub enum CallEvidence {
|
||||
#[default]
|
||||
Unknown,
|
||||
Partial(BTreeSet<CallKey>),
|
||||
Complete(BTreeSet<CallKey>),
|
||||
}
|
||||
|
||||
impl CallEvidence {
|
||||
pub(crate) fn row_keys(row: &crate::query::named::TraceSpansRow) -> BTreeSet<CallKey> {
|
||||
if row.call_keys.is_empty() && !row.litellm_request_id.is_empty() {
|
||||
BTreeSet::from([CallKey::ProviderResponse(row.litellm_request_id.clone())])
|
||||
} else {
|
||||
row.call_keys.iter().cloned().collect()
|
||||
}
|
||||
}
|
||||
|
||||
pub(crate) fn from_row(row: &crate::query::named::TraceSpansRow) -> Self {
|
||||
let kind = row
|
||||
.call_evidence
|
||||
.unwrap_or(if row.litellm_request_id.is_empty() {
|
||||
CallEvidenceKind::Unknown
|
||||
} else {
|
||||
CallEvidenceKind::Complete
|
||||
});
|
||||
match kind {
|
||||
CallEvidenceKind::Complete => Self::Complete(Self::row_keys(row)),
|
||||
CallEvidenceKind::Partial => Self::Partial(Self::row_keys(row)),
|
||||
CallEvidenceKind::Unknown => Self::Unknown,
|
||||
}
|
||||
}
|
||||
|
||||
pub(crate) fn complete(key: CallKey) -> Self {
|
||||
Self::Complete(BTreeSet::from([key]))
|
||||
}
|
||||
|
||||
/// The same evidence with one more key: an id named outside the convention adds to what the
|
||||
/// convention found, but says nothing about completeness.
|
||||
fn with(self, key: CallKey) -> Self {
|
||||
match self {
|
||||
Self::Unknown => Self::complete(key),
|
||||
Self::Partial(keys) => Self::Partial(keys.into_iter().chain([key]).collect()),
|
||||
Self::Complete(keys) => Self::Complete(keys.into_iter().chain([key]).collect()),
|
||||
}
|
||||
}
|
||||
|
||||
pub fn key_set(&self) -> Option<&BTreeSet<CallKey>> {
|
||||
match self {
|
||||
Self::Unknown => None,
|
||||
Self::Partial(keys) | Self::Complete(keys) => Some(keys),
|
||||
}
|
||||
}
|
||||
|
||||
pub fn kind(&self) -> CallEvidenceKind {
|
||||
match self {
|
||||
Self::Unknown => CallEvidenceKind::Unknown,
|
||||
Self::Partial(_) => CallEvidenceKind::Partial,
|
||||
Self::Complete(_) => CallEvidenceKind::Complete,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// What a span says about its own role. A `WrapperCandidate` may only wrap the real operation
|
||||
/// (a crew kickoff around its agents); the trace graph decides.
|
||||
#[derive(Clone, Copy, Debug, Eq, PartialEq)]
|
||||
pub enum RoleEvidence {
|
||||
Unspecified,
|
||||
Declared(ObservationType),
|
||||
WrapperCandidate(ObservationType),
|
||||
}
|
||||
|
||||
pub(crate) struct SpanContext<'a> {
|
||||
pub scope: &'a str,
|
||||
pub name: &'a str,
|
||||
pub parent_span_id: &'a str,
|
||||
pub attributes: &'a BTreeMap<String, String>,
|
||||
pub events: &'a [DecodedEvent],
|
||||
pub resource_attributes: &'a BTreeMap<String, String>,
|
||||
}
|
||||
|
||||
#[derive(Debug, Serialize)]
|
||||
pub struct NormalizedSpan {
|
||||
pub observation_type: ObservationType,
|
||||
pub agent_name: String,
|
||||
pub framework: String,
|
||||
pub litellm_request_id: String,
|
||||
pub model: String,
|
||||
pub wrapper_candidate: bool,
|
||||
pub agent_name: Option<String>,
|
||||
pub framework: Option<Integration>,
|
||||
pub agent_metadata: AgentMetadata,
|
||||
pub calls: CallEvidence,
|
||||
pub model: Option<String>,
|
||||
pub input_tokens: u32,
|
||||
pub output_tokens: u32,
|
||||
pub input: String,
|
||||
pub input_preview: String,
|
||||
pub output: String,
|
||||
pub tool_call_id: Option<String>,
|
||||
}
|
||||
|
||||
pub(crate) struct Normalization {
|
||||
pub span: NormalizedSpan,
|
||||
pub display_name: Option<String>,
|
||||
pub consumed_attributes: [&'static str; 2],
|
||||
pub consumed_attributes: Box<[&'static str]>,
|
||||
}
|
||||
|
||||
trait SpanNormalizer {
|
||||
fn matches(&self, scope_name: &str, attributes: &BTreeMap<String, String>) -> bool;
|
||||
fn consumed_attributes(&self, attributes: &BTreeMap<String, String>) -> [&'static str; 2];
|
||||
fn normalize(
|
||||
&self,
|
||||
name: &str,
|
||||
parent_span_id: &str,
|
||||
attributes: &BTreeMap<String, String>,
|
||||
events: &[DecodedEvent],
|
||||
) -> Result<NormalizedSpan, Error>;
|
||||
fn display_name(&self, _attributes: &BTreeMap<String, String>) -> Option<String> {
|
||||
None
|
||||
}
|
||||
pub(crate) fn normalize(context: &SpanContext<'_>) -> Result<Normalization, Error> {
|
||||
let extraction = format::extract(context)?;
|
||||
Ok(Instrumentation::detect(context).interpret(context, extraction, metadata::extract(context)))
|
||||
}
|
||||
|
||||
mod claude_code;
|
||||
mod genai;
|
||||
mod langsmith;
|
||||
mod openinference;
|
||||
/// An attribute's text together with the key it came from, so consumption follows extraction.
|
||||
pub(crate) struct AttributeText<'a> {
|
||||
pub source: &'static str,
|
||||
pub text: &'a str,
|
||||
}
|
||||
|
||||
use claude_code::ClaudeCodeNormalizer;
|
||||
pub(crate) use claude_code::{CLAUDE_CODE_AGENT, CLAUDE_CODE_SCOPE};
|
||||
use genai::GenAiNormalizer;
|
||||
use langsmith::LangSmithNormalizer;
|
||||
use openinference::OpenInferenceNormalizer;
|
||||
/// The first of `keys` that is recorded and not empty.
|
||||
fn select_attribute<'a>(
|
||||
attributes: &'a BTreeMap<String, String>,
|
||||
keys: &[&'static str],
|
||||
) -> Option<AttributeText<'a>> {
|
||||
keys.iter().copied().find_map(|source| {
|
||||
attributes
|
||||
.get(source)
|
||||
.filter(|text| !text.is_empty())
|
||||
.map(|text| AttributeText {
|
||||
source,
|
||||
text: text.as_str(),
|
||||
})
|
||||
})
|
||||
}
|
||||
|
||||
fn present(attributes: &BTreeMap<String, String>, keys: &[&'static str]) -> Option<String> {
|
||||
select_attribute(attributes, keys).map(|attribute| attribute.text.to_owned())
|
||||
}
|
||||
|
||||
fn attr<'a>(attributes: &'a BTreeMap<String, String>, key: &str) -> &'a str {
|
||||
attributes.get(key).map(String::as_str).unwrap_or_default()
|
||||
}
|
||||
|
||||
fn first<'a>(attributes: &'a BTreeMap<String, String>, left: &str, right: &str) -> &'a str {
|
||||
let value = attr(attributes, left);
|
||||
if value.is_empty() {
|
||||
attr(attributes, right)
|
||||
} else {
|
||||
value
|
||||
}
|
||||
}
|
||||
|
||||
fn tokens(attributes: &BTreeMap<String, String>, key: &str) -> Result<u32, Error> {
|
||||
let value = attr(attributes, key).trim();
|
||||
if value.is_empty() {
|
||||
|
|
@ -91,119 +264,41 @@ fn tokens(attributes: &BTreeMap<String, String>, key: &str) -> Result<u32, Error
|
|||
}
|
||||
}
|
||||
|
||||
fn token_alias(attributes: &BTreeMap<String, String>, keys: &[&'static str]) -> Result<u32, Error> {
|
||||
select_attribute(attributes, keys)
|
||||
.map_or(Ok(0), |attribute| tokens(attributes, attribute.source))
|
||||
}
|
||||
|
||||
fn usage_tokens(attributes: &BTreeMap<String, String>) -> Result<(u32, u32), Error> {
|
||||
Ok((
|
||||
tokens(attributes, "gen_ai.usage.input_tokens")?,
|
||||
tokens(attributes, "gen_ai.usage.output_tokens")?,
|
||||
token_alias(
|
||||
attributes,
|
||||
&["gen_ai.usage.input_tokens", "gen_ai.usage.prompt_tokens"],
|
||||
)?,
|
||||
token_alias(
|
||||
attributes,
|
||||
&[
|
||||
"gen_ai.usage.output_tokens",
|
||||
"gen_ai.usage.completion_tokens",
|
||||
],
|
||||
)?,
|
||||
))
|
||||
}
|
||||
|
||||
#[derive(Default, Deserialize)]
|
||||
struct AgentMetadata {
|
||||
#[serde(default)]
|
||||
lc_agent_name: String,
|
||||
#[serde(default)]
|
||||
ls_integration: String,
|
||||
}
|
||||
|
||||
fn recorded_agent_name(
|
||||
name: &str,
|
||||
attributes: &BTreeMap<String, String>,
|
||||
span: &NormalizedSpan,
|
||||
) -> String {
|
||||
let explicit = [
|
||||
span.agent_name.as_str(),
|
||||
attr(attributes, "gen_ai.agent.name"),
|
||||
attr(attributes, "agent.name"),
|
||||
attr(attributes, "openclaw.agent"),
|
||||
]
|
||||
.into_iter()
|
||||
.find(|value| !value.is_empty());
|
||||
if let Some(value) = explicit {
|
||||
return value.to_owned();
|
||||
}
|
||||
let metadata =
|
||||
serde_json::from_str::<AgentMetadata>(attr(attributes, "metadata")).unwrap_or_default();
|
||||
if !metadata.lc_agent_name.is_empty() {
|
||||
return metadata.lc_agent_name;
|
||||
}
|
||||
if span.observation_type == ObservationType::Agent {
|
||||
let node = attr(attributes, "graph.node.id");
|
||||
if !node.is_empty() {
|
||||
return node.to_owned();
|
||||
}
|
||||
if metadata.ls_integration == "langgraph" && name != "LangGraph" && !is_middleware(name) {
|
||||
return name.to_owned();
|
||||
}
|
||||
}
|
||||
String::new()
|
||||
}
|
||||
|
||||
fn is_middleware(name: &str) -> bool {
|
||||
[
|
||||
".wrap_model_call",
|
||||
".wrap_tool_call",
|
||||
".before_agent",
|
||||
".after_agent",
|
||||
".before_model",
|
||||
".after_model",
|
||||
]
|
||||
.iter()
|
||||
.any(|suffix| name.ends_with(suffix))
|
||||
}
|
||||
|
||||
pub fn normalize(
|
||||
scope_name: &str,
|
||||
name: &str,
|
||||
parent_span_id: &str,
|
||||
attributes: &BTreeMap<String, String>,
|
||||
events: &[DecodedEvent],
|
||||
) -> Result<Normalization, Error> {
|
||||
let normalizers: [&dyn SpanNormalizer; 4] = [
|
||||
&ClaudeCodeNormalizer,
|
||||
&LangSmithNormalizer,
|
||||
&OpenInferenceNormalizer,
|
||||
&GenAiNormalizer,
|
||||
];
|
||||
let normalizer = normalizers
|
||||
.into_iter()
|
||||
.find(|normalizer| normalizer.matches(scope_name, attributes))
|
||||
.expect("GenAI fallback always matches");
|
||||
let span = normalizer.normalize(name, parent_span_id, attributes, events)?;
|
||||
let agent_name = recorded_agent_name(name, attributes, &span);
|
||||
let observation_type = if !parent_span_id.is_empty()
|
||||
&& scope_name == "openinference.instrumentation.langchain"
|
||||
&& is_middleware(name)
|
||||
{
|
||||
ObservationType::Framework
|
||||
} else {
|
||||
span.observation_type
|
||||
};
|
||||
Ok(Normalization {
|
||||
span: NormalizedSpan {
|
||||
agent_name,
|
||||
observation_type,
|
||||
..span
|
||||
},
|
||||
display_name: normalizer.display_name(attributes),
|
||||
consumed_attributes: normalizer.consumed_attributes(attributes),
|
||||
})
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use std::collections::BTreeMap;
|
||||
|
||||
use rstest::rstest;
|
||||
|
||||
use super::{ObservationType, normalize};
|
||||
use super::{ObservationType, SpanContext, normalize};
|
||||
|
||||
#[rstest]
|
||||
#[case::langsmith("langsmith", [("langsmith.span.kind", "llm"), ("openinference.span.kind", "TOOL")], ObservationType::Llm)]
|
||||
#[case::openinference("other", [("openinference.span.kind", "LLM"), ("gen_ai.operation.name", "execute_tool")], ObservationType::Llm)]
|
||||
#[case::genai("other", [("gen_ai.operation.name", "execute_tool"), ("gen_ai.usage.input_tokens", "7")], ObservationType::Tool)]
|
||||
#[case::claude_code("com.anthropic.claude_code.tracing", [("span.type", "llm_request"), ("openinference.span.kind", "TOOL")], ObservationType::Llm)]
|
||||
fn convention_dispatch_preserves_precedence(
|
||||
fn format_dispatch_preserves_precedence(
|
||||
#[case] scope: &str,
|
||||
#[case] attributes: [(&str, &str); 2],
|
||||
#[case] expected: ObservationType,
|
||||
|
|
@ -212,9 +307,16 @@ mod tests {
|
|||
.into_iter()
|
||||
.map(|(key, value)| (key.to_owned(), value.to_owned()))
|
||||
.collect();
|
||||
let fields = normalize(scope, "step", "parent", &attributes, &[])
|
||||
.expect("valid tokens")
|
||||
.span;
|
||||
let fields = normalize(&SpanContext {
|
||||
scope,
|
||||
name: "step",
|
||||
parent_span_id: "parent",
|
||||
attributes: &attributes,
|
||||
events: &[],
|
||||
resource_attributes: &BTreeMap::new(),
|
||||
})
|
||||
.expect("valid tokens")
|
||||
.span;
|
||||
assert_eq!(fields.observation_type, expected);
|
||||
if expected == ObservationType::Tool {
|
||||
assert_eq!(fields.input_tokens, 7);
|
||||
|
|
@ -225,9 +327,16 @@ mod tests {
|
|||
fn token_counts_accept_surrounding_whitespace() {
|
||||
let attributes =
|
||||
BTreeMap::from([("gen_ai.usage.input_tokens".to_owned(), " 7 ".to_owned())]);
|
||||
let fields = normalize("", "root", "", &attributes, &[])
|
||||
.expect("valid tokens")
|
||||
.span;
|
||||
let fields = normalize(&SpanContext {
|
||||
scope: "",
|
||||
name: "root",
|
||||
parent_span_id: "",
|
||||
attributes: &attributes,
|
||||
events: &[],
|
||||
resource_attributes: &BTreeMap::new(),
|
||||
})
|
||||
.expect("valid tokens")
|
||||
.span;
|
||||
assert_eq!(fields.input_tokens, 7);
|
||||
}
|
||||
|
||||
|
|
@ -237,6 +346,16 @@ mod tests {
|
|||
fn token_counts_outside_storage_range_are_rejected(#[case] value: &str) {
|
||||
let attributes =
|
||||
BTreeMap::from([("gen_ai.usage.input_tokens".to_owned(), value.to_owned())]);
|
||||
assert!(normalize("", "root", "", &attributes, &[]).is_err());
|
||||
assert!(
|
||||
normalize(&SpanContext {
|
||||
scope: "",
|
||||
name: "root",
|
||||
parent_span_id: "",
|
||||
attributes: &attributes,
|
||||
events: &[],
|
||||
resource_attributes: &BTreeMap::new()
|
||||
})
|
||||
.is_err()
|
||||
);
|
||||
}
|
||||
}
|
||||
|
|
|
|||
|
|
@ -1,55 +0,0 @@
|
|||
use std::collections::BTreeMap;
|
||||
|
||||
use super::{NormalizedSpan, ObservationType, SpanNormalizer, attr, tokens, usage_tokens};
|
||||
use crate::{Error, otlp::DecodedEvent};
|
||||
|
||||
pub(super) struct OpenInferenceNormalizer;
|
||||
|
||||
impl SpanNormalizer for OpenInferenceNormalizer {
|
||||
fn matches(&self, _scope_name: &str, attributes: &BTreeMap<String, String>) -> bool {
|
||||
attributes.contains_key("openinference.span.kind")
|
||||
}
|
||||
|
||||
fn consumed_attributes(&self, _attributes: &BTreeMap<String, String>) -> [&'static str; 2] {
|
||||
["input.value", "output.value"]
|
||||
}
|
||||
|
||||
fn normalize(
|
||||
&self,
|
||||
_name: &str,
|
||||
parent_span_id: &str,
|
||||
attributes: &BTreeMap<String, String>,
|
||||
_events: &[DecodedEvent],
|
||||
) -> Result<NormalizedSpan, Error> {
|
||||
let (usage_input, usage_output) = usage_tokens(attributes)?;
|
||||
let observation_type = match attr(attributes, "openinference.span.kind")
|
||||
.to_ascii_uppercase()
|
||||
.as_str()
|
||||
{
|
||||
"AGENT" => ObservationType::Agent,
|
||||
"LLM" => ObservationType::Llm,
|
||||
"TOOL" => ObservationType::Tool,
|
||||
_ if parent_span_id.is_empty() => ObservationType::Agent,
|
||||
_ => ObservationType::Chain,
|
||||
};
|
||||
Ok(NormalizedSpan {
|
||||
observation_type,
|
||||
agent_name: attr(attributes, "agent.name").to_owned(),
|
||||
framework: String::new(),
|
||||
litellm_request_id: String::new(),
|
||||
model: attr(attributes, "llm.model_name").to_owned(),
|
||||
input_tokens: if attributes.contains_key("llm.token_count.prompt") {
|
||||
tokens(attributes, "llm.token_count.prompt")?
|
||||
} else {
|
||||
usage_input
|
||||
},
|
||||
output_tokens: if attributes.contains_key("llm.token_count.completion") {
|
||||
tokens(attributes, "llm.token_count.completion")?
|
||||
} else {
|
||||
usage_output
|
||||
},
|
||||
input: attr(attributes, "input.value").to_owned(),
|
||||
output: attr(attributes, "output.value").to_owned(),
|
||||
})
|
||||
}
|
||||
}
|
||||
8
litellm-rust/crates/traces/src/otlp/AGENTS.md
Normal file
8
litellm-rust/crates/traces/src/otlp/AGENTS.md
Normal file
|
|
@ -0,0 +1,8 @@
|
|||
- Decode OTLP JSON and protobuf exports into validated `DecodedSpan` values through `decode_otlp`
|
||||
- Keep media-type dispatch and wire decoding in `wire.rs`, structural and allocation budgets in `limits.rs`, attribute conversion in `attributes.rs`, and span flattening in `span.rs`
|
||||
- Validate span and link IDs, timestamp ranges and ordering, and collection limits before producing decoded spans
|
||||
- Preserve preflight depth and node limits for both encodings and account for decoded allocations, including normalized payloads
|
||||
- Share resource attributes and scope identity across sibling spans through `Shared`; account for copies when a build cannot share storage
|
||||
- Delegate semantic interpretation to `../normalize/`; retain raw attributes and carry consumed-attribute tracking alongside normalized output
|
||||
- Keep HTTP routing, decompression, storage writes, and trace-wide resolution outside this module; return the crate's typed decoding errors
|
||||
- Extend `tests/otlp.rs` with public decoding regressions for both encodings, malformed input, budget enforcement, and shared resource identity
|
||||
|
|
@ -32,7 +32,7 @@ pub struct DecodedSpan {
|
|||
pub status_message: String,
|
||||
pub events: Vec<DecodedEvent>,
|
||||
pub normalized: NormalizedSpan,
|
||||
pub consumed_attributes: [&'static str; 2],
|
||||
pub consumed_attributes: Box<[&'static str]>,
|
||||
}
|
||||
|
||||
pub fn decode_otlp(body: &[u8], content_type: Option<&str>) -> Result<Vec<DecodedSpan>, Error> {
|
||||
|
|
|
|||
|
|
@ -12,7 +12,7 @@ use super::{
|
|||
};
|
||||
use crate::{
|
||||
Error, Shared,
|
||||
normalize::{CLAUDE_CODE_AGENT, CLAUDE_CODE_SCOPE, normalize},
|
||||
normalize::{SpanContext, normalize},
|
||||
};
|
||||
|
||||
pub(super) fn flatten(request: ExportTraceServiceRequest) -> Result<Vec<DecodedSpan>, Error> {
|
||||
|
|
@ -141,39 +141,40 @@ fn decoded_span(
|
|||
})
|
||||
})
|
||||
.collect::<Result<Vec<_>, Error>>()?;
|
||||
let normalization = normalize(
|
||||
scope_name.as_ref(),
|
||||
&span.name,
|
||||
&parent_span_id,
|
||||
&span_attributes,
|
||||
&events,
|
||||
)?;
|
||||
let resource_agent_name = resource_attributes
|
||||
.get("gen_ai.agent.name")
|
||||
.filter(|name| !name.is_empty());
|
||||
let agent_name = match (resource_agent_name, normalization.span.agent_name.as_str()) {
|
||||
(Some(name), "") => name.clone(),
|
||||
(Some(name), "hermes-agent") if scope_name.as_ref() == "hermes-otel-plugin" => name.clone(),
|
||||
(Some(name), CLAUDE_CODE_AGENT) if scope_name.as_ref() == CLAUDE_CODE_SCOPE => name.clone(),
|
||||
(None, CLAUDE_CODE_AGENT) if scope_name.as_ref() == CLAUDE_CODE_SCOPE => {
|
||||
resource_attributes
|
||||
.get("service.name")
|
||||
.filter(|name| !name.is_empty())
|
||||
.map_or_else(|| CLAUDE_CODE_AGENT.to_owned(), Clone::clone)
|
||||
}
|
||||
(_, name) => name.to_owned(),
|
||||
};
|
||||
let normalized = crate::normalize::NormalizedSpan {
|
||||
agent_name,
|
||||
..normalization.span
|
||||
};
|
||||
let normalization = normalize(&SpanContext {
|
||||
scope: scope_name.as_ref(),
|
||||
name: &span.name,
|
||||
parent_span_id: &parent_span_id,
|
||||
attributes: &span_attributes,
|
||||
events: &events,
|
||||
resource_attributes: resource_attributes.as_ref(),
|
||||
})?;
|
||||
let normalized = normalization.span;
|
||||
budget.consume(
|
||||
normalized.input.len()
|
||||
+ normalized.output.len()
|
||||
+ normalized.agent_name.len()
|
||||
+ normalized.framework.len()
|
||||
+ normalized.litellm_request_id.len()
|
||||
+ normalized.model.len()
|
||||
+ normalized.agent_name.as_ref().map_or(0, String::len)
|
||||
+ normalized
|
||||
.framework
|
||||
.as_ref()
|
||||
.map_or(0, |integration| match integration {
|
||||
crate::Integration::Other(name) => name.len(),
|
||||
_ => 0,
|
||||
})
|
||||
+ normalized.agent_metadata.byte_len()
|
||||
+ normalized
|
||||
.calls
|
||||
.key_set()
|
||||
.into_iter()
|
||||
.flatten()
|
||||
.map(|key| match key {
|
||||
crate::CallKey::LiteLlmRequest(id) | crate::CallKey::ProviderResponse(id) => {
|
||||
id.len() + size_of::<crate::CallKey>()
|
||||
}
|
||||
crate::CallKey::Transport => size_of::<crate::CallKey>(),
|
||||
})
|
||||
.sum::<usize>()
|
||||
+ normalized.model.as_ref().map_or(0, String::len)
|
||||
+ normalization.display_name.as_ref().map_or(0, String::len),
|
||||
)?;
|
||||
Ok(DecodedSpan {
|
||||
|
|
|
|||
|
|
@ -1,3 +1,4 @@
|
|||
pub mod guide;
|
||||
pub mod named;
|
||||
|
||||
#[derive(Clone, Copy, Debug, Eq, PartialEq, strum::EnumString, strum::Display, strum::AsRefStr)]
|
||||
|
|
@ -5,6 +6,7 @@ pub mod named;
|
|||
pub enum ReadQuery {
|
||||
ListTraces,
|
||||
TraceSpans,
|
||||
TracePageSpans,
|
||||
TraceIdentity,
|
||||
SpanDetail,
|
||||
SpanError,
|
||||
|
|
|
|||
27
litellm-rust/crates/traces/src/query/guide.rs
Normal file
27
litellm-rust/crates/traces/src/query/guide.rs
Normal file
|
|
@ -0,0 +1,27 @@
|
|||
use askama::Template;
|
||||
|
||||
#[macro_rules_attribute::apply(response_type)]
|
||||
#[cfg_attr(feature = "schema", schemars(rename = "TraceQueryExample"))]
|
||||
pub struct Example {
|
||||
pub name: String,
|
||||
pub sql: String,
|
||||
}
|
||||
|
||||
pub struct Section<'a> {
|
||||
pub title: &'a str,
|
||||
pub body: &'a str,
|
||||
}
|
||||
|
||||
#[derive(Template)]
|
||||
#[template(path = "query_help.jinja", escape = "none")]
|
||||
pub struct QueryGuide<'a> {
|
||||
pub sections: &'a [Section<'a>],
|
||||
pub examples: &'a [Example],
|
||||
pub gotchas: &'a [String],
|
||||
}
|
||||
|
||||
impl QueryGuide<'_> {
|
||||
pub fn render(&self) -> Result<String, askama::Error> {
|
||||
Template::render(self)
|
||||
}
|
||||
}
|
||||
|
|
@ -1,9 +1,16 @@
|
|||
use serde::{Deserialize, Serialize};
|
||||
use std::collections::BTreeMap;
|
||||
|
||||
#[derive(Debug, Deserialize, Serialize)]
|
||||
#[macro_rules_attribute::apply(wire_type)]
|
||||
#[derive(Clone, Debug)]
|
||||
#[cfg_attr(feature = "schema", schemars(rename = "TraceScope"))]
|
||||
pub struct ReadAccessParams {
|
||||
pub all_teams: u8,
|
||||
#[serde(
|
||||
deserialize_with = "crate::wire::flag",
|
||||
serialize_with = "crate::wire::serialize_flag"
|
||||
)]
|
||||
#[cfg_attr(feature = "schema", schemars(schema_with = "crate::schema::flag"))]
|
||||
pub all_teams: bool,
|
||||
pub user_id: String,
|
||||
pub team_ids: Vec<String>,
|
||||
}
|
||||
|
|
@ -29,7 +36,8 @@ pub struct ListTracesRow {
|
|||
pub name: String,
|
||||
pub service: String,
|
||||
pub input_preview: String,
|
||||
pub status: String,
|
||||
#[serde(serialize_with = "crate::wire::serialize_status")]
|
||||
pub status: crate::SpanStatus,
|
||||
pub start_ms: i64,
|
||||
pub duration_ms: i64,
|
||||
pub span_count: u64,
|
||||
|
|
@ -58,17 +66,30 @@ pub struct TraceSpansParams {
|
|||
|
||||
#[derive(Debug, Deserialize, Serialize)]
|
||||
pub struct TraceSpansRow {
|
||||
#[serde(default)]
|
||||
pub trace_id: String,
|
||||
pub span_id: String,
|
||||
pub parent_span_id: String,
|
||||
pub name: String,
|
||||
#[serde(rename = "type")]
|
||||
pub kind: String,
|
||||
pub kind: crate::ObservationType,
|
||||
#[serde(
|
||||
default,
|
||||
deserialize_with = "crate::wire::flag",
|
||||
serialize_with = "crate::wire::serialize_flag"
|
||||
)]
|
||||
pub wrapper_candidate: bool,
|
||||
pub agent: String,
|
||||
#[serde(default)]
|
||||
pub framework: String,
|
||||
pub status: String,
|
||||
#[serde(serialize_with = "crate::wire::serialize_status")]
|
||||
pub status: crate::SpanStatus,
|
||||
pub status_message: String,
|
||||
pub error_truncated: u8,
|
||||
#[serde(
|
||||
deserialize_with = "crate::wire::flag",
|
||||
serialize_with = "crate::wire::serialize_flag"
|
||||
)]
|
||||
pub error_truncated: bool,
|
||||
pub start_ns: i64,
|
||||
pub duration_ns: u64,
|
||||
pub service: String,
|
||||
|
|
@ -77,11 +98,30 @@ pub struct TraceSpansRow {
|
|||
pub input_tokens: u32,
|
||||
pub output_tokens: u32,
|
||||
pub litellm_request_id: String,
|
||||
#[serde(default)]
|
||||
pub call_keys: Vec<crate::CallKey>,
|
||||
#[serde(
|
||||
default,
|
||||
deserialize_with = "crate::wire::evidence",
|
||||
serialize_with = "crate::wire::serialize_evidence"
|
||||
)]
|
||||
pub call_evidence: Option<crate::CallEvidenceKind>,
|
||||
#[serde(default)]
|
||||
pub tool_call_id: String,
|
||||
pub team_id: String,
|
||||
pub api_key_hash: String,
|
||||
pub user_id: String,
|
||||
}
|
||||
|
||||
#[derive(Debug, Deserialize, Serialize)]
|
||||
pub struct TracePageSpansParams {
|
||||
#[serde(flatten)]
|
||||
pub access: ReadAccessParams,
|
||||
pub trace_refs: Vec<String>,
|
||||
pub start_ms: i64,
|
||||
pub end_ms: i64,
|
||||
}
|
||||
|
||||
#[derive(Debug, Deserialize, Serialize)]
|
||||
pub struct SpanDetailParams {
|
||||
#[serde(flatten)]
|
||||
|
|
@ -123,6 +163,8 @@ pub struct SpendByResponseIdsParams {
|
|||
#[serde(flatten)]
|
||||
pub access: ReadAccessParams,
|
||||
pub response_ids: Vec<String>,
|
||||
pub request_ids: Vec<String>,
|
||||
pub trace_ids: Vec<String>,
|
||||
pub start_ms: i64,
|
||||
pub end_ms: i64,
|
||||
}
|
||||
|
|
@ -131,10 +173,13 @@ pub struct SpendByResponseIdsParams {
|
|||
pub struct SpendByResponseIdsRow {
|
||||
pub request_id: String,
|
||||
pub response_id: String,
|
||||
pub upstream_response_id: String,
|
||||
pub trace_id: String,
|
||||
pub span_id: String,
|
||||
pub team_id: String,
|
||||
pub api_key: String,
|
||||
pub user: String,
|
||||
pub spend: f64,
|
||||
pub spend: Option<f64>,
|
||||
pub start_ms: i64,
|
||||
}
|
||||
|
||||
|
|
|
|||
|
|
@ -1,11 +1,12 @@
|
|||
use serde::{Deserialize, Serialize};
|
||||
|
||||
use crate::InvalidScope;
|
||||
|
||||
#[derive(Clone, Debug, Deserialize, Serialize)]
|
||||
#[macro_rules_attribute::apply(wire_type)]
|
||||
#[derive(Clone, Debug)]
|
||||
#[serde(tag = "kind", rename_all = "snake_case", deny_unknown_fields)]
|
||||
pub enum QueryScope {
|
||||
#[cfg_attr(feature = "schema", schemars(title = "AllQueryScope"))]
|
||||
All,
|
||||
#[cfg_attr(feature = "schema", schemars(title = "OwnedQueryScope"))]
|
||||
Owned {
|
||||
user_id: String,
|
||||
team_ids: Vec<String>,
|
||||
|
|
|
|||
Some files were not shown because too many files have changed in this diff Show more
Loading…
Add table
Reference in a new issue