From 03ba79d261af22800fc3c603d3e0b610f700656b Mon Sep 17 00:00:00 2001 From: ishaan-berri <155045088+ishaan-berri@users.noreply.github.com> Date: Wed, 7 Oct 2026 20:14:30 -0700 Subject: [PATCH] feat(lens): show end-user feedback on traces, stored in ClickHouse (#45171) * feat(lens): add lens_feedback ClickHouse table for human trace scores ReplacingMergeTree(UpdatedAt, IsDeleted) keyed like agent_traces_by_key so feedback joins traces in sort order, with a bloom filter on TraceId, a score CHECK, and the same retention as traces Co-Authored-By: Claude Opus 5.5 * feat(lens): add scoped ClickHouse reads for trace feedback feedback_target resolves a visible trace's team, key and trace_ref before a write; feedback lists the latest live entry per author; feedback_summary returns count, average and lowest score for a batch of traces Co-Authored-By: Claude Opus 5.5 * test(lens): pin the Claude session to trace id hash shared with feedback Co-Authored-By: Claude Opus 5.5 * feat(lens): expose feedback queries to the Python trace bridge Co-Authored-By: Claude Opus 5.5 * feat(lens): add typed feedback models and ClickHouse feedback store Co-Authored-By: Claude Opus 5.5 * feat(lens): add /lens/feedback API for 0-10 human scores on traces PUT saves or replaces the caller's score and comment, GET lists every entry, DELETE removes the caller's, POST /summary feeds the trace list. Callers can target a trace_id or a Claude Code session_id Co-Authored-By: Claude Opus 5.5 * feat(lens-ui): add typed feedback calls to the traces API Co-Authored-By: Claude Opus 5.5 * feat(lens-ui): flag rated traces and filter the list by feedback A Feedback column shows each run's average score and rating count, marked red when any score is 4 or below, and a filter narrows to rated or low-score runs through the URL Co-Authored-By: Claude Opus 5.5 * feat(lens-ui): show and edit human feedback inside a trace Co-Authored-By: Claude Opus 5.5 * feat(lens): let app keys post end-user feedback on their own traces Any key can now PUT/DELETE feedback on traces its team or key sent, naming the end user in an optional user field. Reading feedback stays with Lens admins Co-Authored-By: Claude Opus 5.5 * feat(lens-ui): show end-user feedback first in a run and flag low-scored rows Replace the edit popover and feedback filter with a read-only view: the list shows each run's score in a fixed-width badge and tints the row red when a user scored it 4 or below, and opening a run shows what users said with their score right under the header Co-Authored-By: Claude Opus 5.5 --------- Co-authored-by: Claude Opus 5.5 --- litellm-rust/crates/lens/src/lib.rs | 20 +- litellm-rust/crates/lens/tests/receiver.rs | 70 ++- .../migrations/0017_lens_feedback.sql | 18 + .../traces-clickhouse/query/lens_feedback.sql | 12 + .../query/lens_feedback_summary.sql | 9 + .../query/lens_feedback_target.sql | 9 + .../crates/traces-clickhouse/src/insert.rs | 3 + .../traces-clickhouse/src/query/lens.rs | 110 ++++- .../crates/traces-clickhouse/src/schema.rs | 3 +- .../crates/traces-clickhouse/src/sql.rs | 7 + .../traces-clickhouse/src/wire_schema.rs | 12 + .../traces-clickhouse/tests/lens_feedback.rs | 408 ++++++++++++++++++ .../traces-clickhouse/tests/migrations.rs | 24 +- litellm-rust/crates/traces/src/query.rs | 3 + litellm-rust/crates/traces/tests/otlp.rs | 3 + litellm-rust/crates/traces/tests/query.rs | 3 + litellm/constants.py | 2 + litellm/proxy/_types.py | 3 + litellm/proxy/lens/feedback_endpoints.py | 124 ++++++ litellm/proxy/lens/feedback_models.py | 37 ++ litellm/proxy/lens/feedback_repository.py | 186 ++++++++ litellm/proxy/proxy_server.py | 2 + litellm/rust_bridge/trace/generated/models.py | 179 ++++++++ litellm/rust_bridge/trace/generated/types.py | 12 +- litellm/rust_bridge/trace/queries.py | 15 + litellm/tracing/remote.py | 8 +- .../traces-clickhouse/FeedbackRow.json | 53 +++ .../traces-clickhouse/FeedbackSummaryRow.json | 62 +++ .../traces-clickhouse/FeedbackTargetRow.json | 21 + .../traces-clickhouse/LensFeedbackParams.json | 34 ++ .../LensFeedbackSummaryParams.json | 33 ++ .../LensFeedbackTargetParams.json | 34 ++ .../traces-clickhouse/ReadQueryName.json | 5 +- .../proxy/lens/test_feedback_endpoints.py | 323 ++++++++++++++ tests/unit/tracing/test_remote.py | 17 +- .../lens/LensSetup.integration.test.tsx | 1 + .../lens/LensWorkspace.integration.test.tsx | 2 + .../lens/data/demo/createLensDemo.ts | 34 ++ .../src/components/lens/traces/api.ts | 16 + .../FeedbackPanel.integration.test.tsx | 88 ++++ .../traces/detail/feedback/FeedbackPanel.tsx | 87 ++++ .../traces/detail/feedback/feedback.test.ts | 31 ++ .../lens/traces/detail/feedback/feedback.ts | 18 + .../lens/traces/detail/run/RunView.tsx | 2 + .../AgentTracesSection.integration.test.tsx | 71 ++- .../lens/traces/list/AgentTracesSection.tsx | 3 + .../traces/list/AgentTracesTable.test.tsx | 44 ++ .../lens/traces/list/AgentTracesTable.tsx | 169 +++++--- .../useTraceFeedback.integration.test.tsx | 35 ++ .../lens/traces/list/useTraceFeedback.test.ts | 21 + .../lens/traces/list/useTraceFeedback.ts | 54 +++ .../src/components/lens/traces/types.ts | 5 + ui/litellm-dashboard/src/lib/http/schema.d.ts | 250 +++++++++++ 53 files changed, 2712 insertions(+), 83 deletions(-) create mode 100644 litellm-rust/crates/traces-clickhouse/migrations/0017_lens_feedback.sql create mode 100644 litellm-rust/crates/traces-clickhouse/query/lens_feedback.sql create mode 100644 litellm-rust/crates/traces-clickhouse/query/lens_feedback_summary.sql create mode 100644 litellm-rust/crates/traces-clickhouse/query/lens_feedback_target.sql create mode 100644 litellm-rust/crates/traces-clickhouse/tests/lens_feedback.rs create mode 100644 litellm/proxy/lens/feedback_endpoints.py create mode 100644 litellm/proxy/lens/feedback_models.py create mode 100644 litellm/proxy/lens/feedback_repository.py create mode 100644 scripts/trace_codegen/schemas/traces-clickhouse/FeedbackRow.json create mode 100644 scripts/trace_codegen/schemas/traces-clickhouse/FeedbackSummaryRow.json create mode 100644 scripts/trace_codegen/schemas/traces-clickhouse/FeedbackTargetRow.json create mode 100644 scripts/trace_codegen/schemas/traces-clickhouse/LensFeedbackParams.json create mode 100644 scripts/trace_codegen/schemas/traces-clickhouse/LensFeedbackSummaryParams.json create mode 100644 scripts/trace_codegen/schemas/traces-clickhouse/LensFeedbackTargetParams.json create mode 100644 tests/unit/proxy/lens/test_feedback_endpoints.py create mode 100644 ui/litellm-dashboard/src/components/lens/traces/detail/feedback/FeedbackPanel.integration.test.tsx create mode 100644 ui/litellm-dashboard/src/components/lens/traces/detail/feedback/FeedbackPanel.tsx create mode 100644 ui/litellm-dashboard/src/components/lens/traces/detail/feedback/feedback.test.ts create mode 100644 ui/litellm-dashboard/src/components/lens/traces/detail/feedback/feedback.ts create mode 100644 ui/litellm-dashboard/src/components/lens/traces/list/useTraceFeedback.integration.test.tsx create mode 100644 ui/litellm-dashboard/src/components/lens/traces/list/useTraceFeedback.test.ts create mode 100644 ui/litellm-dashboard/src/components/lens/traces/list/useTraceFeedback.ts diff --git a/litellm-rust/crates/lens/src/lib.rs b/litellm-rust/crates/lens/src/lib.rs index 887f7f571fb..a3038a0f975 100644 --- a/litellm-rust/crates/lens/src/lib.rs +++ b/litellm-rust/crates/lens/src/lib.rs @@ -116,6 +116,7 @@ pub fn router(state: Arc) -> Router { Router::new() .route("/internal/read", post(read)) .route("/internal/spend", post(spend)) + .route("/internal/feedback", post(feedback)) .route("/internal/credentials", post(credentials)) .route("/internal/status", get(status)), ) @@ -233,6 +234,23 @@ async fn spend( AppState(state): AppState>, headers: HeaderMap, body: Body, +) -> Result { + insert(state, headers, body, InsertTable::SpendLogs).await +} + +async fn feedback( + AppState(state): AppState>, + headers: HeaderMap, + body: Body, +) -> Result { + insert(state, headers, body, InsertTable::LensFeedback).await +} + +async fn insert( + state: Arc, + headers: HeaderMap, + body: Body, + table: InsertTable, ) -> Result { auth::authorize_service(&headers, &state.service_token)?; state.require_storage()?; @@ -256,7 +274,7 @@ async fn spend( &state.storage.client, state.storage.config.storage().writer(), state.storage.config.storage().database(), - InsertTable::SpendLogs, + table, rows, ) .await?; diff --git a/litellm-rust/crates/lens/tests/receiver.rs b/litellm-rust/crates/lens/tests/receiver.rs index 1679fc2540d..bda1cecbbb3 100644 --- a/litellm-rust/crates/lens/tests/receiver.rs +++ b/litellm-rust/crates/lens/tests/receiver.rs @@ -116,6 +116,74 @@ async fn agent_picker_query_preserves_scope_through_the_internal_read_route() { assert_eq!(response.json::().await.unwrap(), result); } +#[rstest] +#[tokio::test] +async fn feedback_summary_query_preserves_scope_through_the_internal_read_route() { + let store = MockServer::start().await; + let result = json!({"data": [{ + "trace_id": "1234567890abcdef1234567890abcdef", "trace_ref": "REF", + "count": "2", "average": 5.5, "lowest": "2" + }]}); + Mock::given(method("POST")) + .and(body_string_contains("FROM lens_feedback FINAL")) + .and(query_param("param_all_teams", "0")) + .and(query_param("param_team", "feedback-team")) + .and(query_param( + "param_trace_ids", + "['1234567890abcdef1234567890abcdef']", + )) + .respond_with(ResponseTemplate::new(200).set_body_json(&result)) + .expect(1) + .mount(&store) + .await; + let server = serve(&store.uri(), true).await; + let response = http_client() + .unwrap() + .post(format!("{}/internal/read", server.url)) + .bearer_auth(SERVICE_TOKEN) + .json(&json!({ + "operation": "query", "name": "feedback_summary", "parameters": { + "all_teams": 0, "team": "feedback-team", "key_hash": "", + "trace_ids": ["1234567890abcdef1234567890abcdef"] + } + })) + .send() + .await + .unwrap(); + assert_eq!(response.status(), 200); + assert_eq!(response.json::().await.unwrap(), result); +} + +#[rstest] +#[tokio::test] +async fn feedback_rows_are_written_to_the_feedback_table() { + let store = MockServer::start().await; + Mock::given(method("POST")) + .and(query_param( + "query", + "INSERT INTO `litellm`.lens_feedback FORMAT JSONEachRow", + )) + .respond_with(ResponseTemplate::new(200)) + .expect(1) + .mount(&store) + .await; + let server = serve(&store.uri(), true).await; + let response = http_client() + .unwrap() + .post(format!("{}/internal/feedback", server.url)) + .bearer_auth(SERVICE_TOKEN) + .json(&json!([{ + "TeamId": "team", "ApiKeyHash": "", "TraceId": "1234567890abcdef1234567890abcdef", + "Author": "customer-1042", "Score": 2, "Comment": "wrong command", + "CreatedAt": "2026-10-07T21:57:01.414Z", "UpdatedAt": "2026-10-07T21:57:01.414Z", + "IsDeleted": 0 + }])) + .send() + .await + .unwrap(); + assert_eq!(response.status(), 204); +} + #[rstest] #[tokio::test] async fn ingestion_confirms_storage_and_overwrites_exporter_tenant() { @@ -297,7 +365,7 @@ async fn ingestion_key_cannot_read_or_export_gateway_records() { let store = MockServer::start().await; let server = serve(&store.uri(), true).await; let client = http_client().unwrap(); - for path in ["/internal/read", "/internal/spend"] { + for path in ["/internal/read", "/internal/spend", "/internal/feedback"] { let response = client .post(format!("{}{path}", server.url)) .bearer_auth(KEY) diff --git a/litellm-rust/crates/traces-clickhouse/migrations/0017_lens_feedback.sql b/litellm-rust/crates/traces-clickhouse/migrations/0017_lens_feedback.sql new file mode 100644 index 00000000000..9cef629e223 --- /dev/null +++ b/litellm-rust/crates/traces-clickhouse/migrations/0017_lens_feedback.sql @@ -0,0 +1,18 @@ +CREATE TABLE IF NOT EXISTS {database}.lens_feedback +( + TeamId LowCardinality(String), + ApiKeyHash String, + TraceId String CODEC(ZSTD(1)), + Author String, + Score UInt8, + Comment String CODEC(ZSTD(3)), + CreatedAt DateTime64(3), + UpdatedAt DateTime64(3), + IsDeleted UInt8, + EngineReceivedMs UInt64 DEFAULT 0, + INDEX idx_trace_id TraceId TYPE bloom_filter(0.001) GRANULARITY 1, + CONSTRAINT score_range CHECK Score <= 10 +) +ENGINE = ReplacingMergeTree(UpdatedAt, IsDeleted) +ORDER BY (TeamId, ApiKeyHash, TraceId, Author) +SETTINGS materialize_ttl_recalculate_only = 1, non_replicated_deduplication_window = 1000 diff --git a/litellm-rust/crates/traces-clickhouse/query/lens_feedback.sql b/litellm-rust/crates/traces-clickhouse/query/lens_feedback.sql new file mode 100644 index 00000000000..11540af6ddf --- /dev/null +++ b/litellm-rust/crates/traces-clickhouse/query/lens_feedback.sql @@ -0,0 +1,12 @@ +SELECT TraceId AS trace_id, + hex(SHA256(concat(TeamId, char(0), ApiKeyHash, char(0), TraceId))) AS trace_ref, + Author AS author, Score AS score, Comment AS comment, + formatDateTime(CreatedAt, '%FT%T.%fZ', 'UTC') AS created_at, + formatDateTime(UpdatedAt, '%FT%T.%fZ', 'UTC') AS updated_at +FROM lens_feedback FINAL +WHERE TraceId = {trace_id:String} + AND hex(SHA256(concat(TeamId, char(0), ApiKeyHash, char(0), TraceId))) = {trace_ref:String} + AND ({all_teams:UInt8}=1 OR TeamId={team:String}) + AND ({key_hash:String}='' OR ApiKeyHash={key_hash:String}) + AND IsDeleted = 0 +ORDER BY UpdatedAt DESC, Author diff --git a/litellm-rust/crates/traces-clickhouse/query/lens_feedback_summary.sql b/litellm-rust/crates/traces-clickhouse/query/lens_feedback_summary.sql new file mode 100644 index 00000000000..8786ecdfa42 --- /dev/null +++ b/litellm-rust/crates/traces-clickhouse/query/lens_feedback_summary.sql @@ -0,0 +1,9 @@ +SELECT TraceId AS trace_id, + hex(SHA256(concat(TeamId, char(0), ApiKeyHash, char(0), TraceId))) AS trace_ref, + count() AS count, avg(Score) AS average, min(Score) AS lowest +FROM lens_feedback FINAL +WHERE TraceId IN {trace_ids:Array(String)} + AND ({all_teams:UInt8}=1 OR TeamId={team:String}) + AND ({key_hash:String}='' OR ApiKeyHash={key_hash:String}) + AND IsDeleted = 0 +GROUP BY TeamId, ApiKeyHash, TraceId diff --git a/litellm-rust/crates/traces-clickhouse/query/lens_feedback_target.sql b/litellm-rust/crates/traces-clickhouse/query/lens_feedback_target.sql new file mode 100644 index 00000000000..e2b7f96d90b --- /dev/null +++ b/litellm-rust/crates/traces-clickhouse/query/lens_feedback_target.sql @@ -0,0 +1,9 @@ +SELECT TeamId AS team_id, ApiKeyHash AS key_hash, + hex(SHA256(concat(TeamId, char(0), ApiKeyHash, char(0), TraceId))) AS trace_ref +FROM agent_traces_by_key +WHERE TraceId = {trace_id:String} + AND ({all_teams:UInt8}=1 OR TeamId={team:String}) + AND ({key_hash:String}='' OR ApiKeyHash={key_hash:String}) + AND ({trace_ref:String}='' OR hex(SHA256(concat(TeamId, char(0), ApiKeyHash, char(0), TraceId)))={trace_ref:String}) +GROUP BY TeamId, ApiKeyHash, TraceId +LIMIT 2 diff --git a/litellm-rust/crates/traces-clickhouse/src/insert.rs b/litellm-rust/crates/traces-clickhouse/src/insert.rs index d45cc8d53b8..ed8db188c14 100644 --- a/litellm-rust/crates/traces-clickhouse/src/insert.rs +++ b/litellm-rust/crates/traces-clickhouse/src/insert.rs @@ -33,6 +33,7 @@ pub type InsertRow = BTreeMap>; pub enum InsertTable { OtelTraces, SpendLogs, + LensFeedback, } impl InsertTable { @@ -40,6 +41,7 @@ impl InsertTable { match value { "otel_traces" => Ok(Self::OtelTraces), "spend_logs" => Ok(Self::SpendLogs), + "lens_feedback" => Ok(Self::LensFeedback), _ => Err(Error::InvalidTable), } } @@ -48,6 +50,7 @@ impl InsertTable { match self { Self::OtelTraces => "otel_traces", Self::SpendLogs => "spend_logs", + Self::LensFeedback => "lens_feedback", } } } diff --git a/litellm-rust/crates/traces-clickhouse/src/query/lens.rs b/litellm-rust/crates/traces-clickhouse/src/query/lens.rs index fcca4fe6f96..97b74de4c20 100644 --- a/litellm-rust/crates/traces-clickhouse/src/query/lens.rs +++ b/litellm-rust/crates/traces-clickhouse/src/query/lens.rs @@ -6,13 +6,16 @@ const SAMPLE_READ_LIMITS: ReadLimits = ReadLimits { ..litellm_storage_clickhouse::READ_LIMITS }; -pub const LENS_QUERIES: [litellm_traces::ReadQuery; 6] = [ +pub const LENS_QUERIES: [litellm_traces::ReadQuery; 9] = [ litellm_traces::ReadQuery::TraceAgents, litellm_traces::ReadQuery::Availability, litellm_traces::ReadQuery::Agents, litellm_traces::ReadQuery::Sample, litellm_traces::ReadQuery::Content, litellm_traces::ReadQuery::Evidence, + litellm_traces::ReadQuery::FeedbackTarget, + litellm_traces::ReadQuery::Feedback, + litellm_traces::ReadQuery::FeedbackSummary, ]; #[macro_rules_attribute::apply(wire_type)] @@ -339,3 +342,108 @@ impl Query for LensEvidence { const SQL: &'static str = include_str!("../../query/lens_evidence.sql"); } + +pub struct LensFeedbackTarget; + +#[macro_rules_attribute::apply(wire_type)] +#[derive(Debug)] +#[serde(deny_unknown_fields)] +pub struct LensFeedbackTargetParams { + #[serde(flatten)] + pub access: LensAccessParams, + pub trace_id: String, + pub trace_ref: String, +} + +#[macro_rules_attribute::apply(wire_type)] +#[derive(Debug)] +#[cfg_attr(feature = "schema", schemars(rename = "FeedbackTargetRow"))] +pub struct LensFeedbackTargetRow { + pub team_id: String, + pub key_hash: String, + pub trace_ref: String, +} + +impl Query for LensFeedbackTarget { + type Params = LensFeedbackTargetParams; + type Row = LensFeedbackTargetRow; + + const SQL: &'static str = include_str!("../../query/lens_feedback_target.sql"); +} + +pub struct LensFeedback; + +#[macro_rules_attribute::apply(wire_type)] +#[derive(Debug)] +#[serde(deny_unknown_fields)] +pub struct LensFeedbackParams { + #[serde(flatten)] + pub access: LensAccessParams, + pub trace_id: String, + pub trace_ref: String, +} + +#[macro_rules_attribute::apply(wire_type)] +#[derive(Debug)] +#[cfg_attr(feature = "schema", schemars(rename = "FeedbackRow"))] +pub struct LensFeedbackRow { + pub trace_id: String, + pub trace_ref: String, + pub author: String, + #[serde(deserialize_with = "super::number::deserialize")] + #[cfg_attr( + feature = "schema", + schemars(schema_with = "crate::wire_schema::u64_number") + )] + pub score: u64, + pub comment: String, + pub created_at: String, + pub updated_at: String, +} + +impl Query for LensFeedback { + type Params = LensFeedbackParams; + type Row = LensFeedbackRow; + + const SQL: &'static str = include_str!("../../query/lens_feedback.sql"); +} + +pub struct LensFeedbackSummary; + +#[macro_rules_attribute::apply(wire_type)] +#[derive(Debug)] +#[serde(deny_unknown_fields)] +pub struct LensFeedbackSummaryParams { + #[serde(flatten)] + pub access: LensAccessParams, + pub trace_ids: Vec, +} + +#[macro_rules_attribute::apply(wire_type)] +#[derive(Debug)] +#[cfg_attr(feature = "schema", schemars(rename = "FeedbackSummaryRow"))] +pub struct LensFeedbackSummaryRow { + pub trace_id: String, + pub trace_ref: String, + #[serde(deserialize_with = "super::number::deserialize")] + #[cfg_attr( + feature = "schema", + schemars(schema_with = "crate::wire_schema::u64_number") + )] + pub count: u64, + #[serde(deserialize_with = "super::number::deserialize")] + pub average: f64, + #[serde(deserialize_with = "super::number::deserialize")] + #[cfg_attr( + feature = "schema", + schemars(schema_with = "crate::wire_schema::u64_number") + )] + pub lowest: u64, +} + +impl Query for LensFeedbackSummary { + type Params = LensFeedbackSummaryParams; + type Row = LensFeedbackSummaryRow; + + const SQL: &'static str = include_str!("../../query/lens_feedback_summary.sql"); +} diff --git a/litellm-rust/crates/traces-clickhouse/src/schema.rs b/litellm-rust/crates/traces-clickhouse/src/schema.rs index 562dbb976c2..5eb8bf0b973 100644 --- a/litellm-rust/crates/traces-clickhouse/src/schema.rs +++ b/litellm-rust/crates/traces-clickhouse/src/schema.rs @@ -14,10 +14,11 @@ static MIGRATOR: Migrator = Migrator { ..sqlx::migrate!("./migrations") }; -const RETENTION: [(&str, &str); 3] = [ +const RETENTION: [(&str, &str); 4] = [ ("otel_traces", "toDateTime(Timestamp)"), ("agent_traces_by_key", "toDateTime(StartTs)"), ("spend_logs", "toDateTime(start_time)"), + ("lens_feedback", "toDateTime(CreatedAt)"), ]; fn validate_schema(database: &str, retention_days: u32) -> Result<(), Error> { diff --git a/litellm-rust/crates/traces-clickhouse/src/sql.rs b/litellm-rust/crates/traces-clickhouse/src/sql.rs index 1f41f0f6f43..613159cf06d 100644 --- a/litellm-rust/crates/traces-clickhouse/src/sql.rs +++ b/litellm-rust/crates/traces-clickhouse/src/sql.rs @@ -37,6 +37,13 @@ pub async fn execute_named_read( ReadQuery::Sample => named_json::(client, connection, parameters).await, ReadQuery::Content => named_json::(client, connection, parameters).await, ReadQuery::Evidence => named_json::(client, connection, parameters).await, + ReadQuery::FeedbackTarget => { + named_json::(client, connection, parameters).await + } + ReadQuery::Feedback => named_json::(client, connection, parameters).await, + ReadQuery::FeedbackSummary => { + named_json::(client, connection, parameters).await + } } } diff --git a/litellm-rust/crates/traces-clickhouse/src/wire_schema.rs b/litellm-rust/crates/traces-clickhouse/src/wire_schema.rs index f7a52a9edd0..408d224f67b 100644 --- a/litellm-rust/crates/traces-clickhouse/src/wire_schema.rs +++ b/litellm-rust/crates/traces-clickhouse/src/wire_schema.rs @@ -81,6 +81,18 @@ pub fn schemas() -> BTreeMap<&'static str, Schema> { ("LensSampleParams", received::()), ("LensContentParams", received::()), ("LensEvidenceParams", received::()), + ( + "LensFeedbackTargetParams", + received::(), + ), + ("LensFeedbackParams", received::()), + ( + "LensFeedbackSummaryParams", + received::(), + ), + ("FeedbackTargetRow", received::()), + ("FeedbackRow", received::()), + ("FeedbackSummaryRow", received::()), ( "ActivityAvailability", received::(), diff --git a/litellm-rust/crates/traces-clickhouse/tests/lens_feedback.rs b/litellm-rust/crates/traces-clickhouse/tests/lens_feedback.rs new file mode 100644 index 00000000000..b2efea71823 --- /dev/null +++ b/litellm-rust/crates/traces-clickhouse/tests/lens_feedback.rs @@ -0,0 +1,408 @@ +use std::collections::BTreeMap; + +use litellm_traces_clickhouse::{ + Connection, InsertTable, Parameter, ReadQuery, ensure_schema, execute_named_read, execute_read, + insert_rows, +}; +use rstest::rstest; +use serde_json::{Value, json}; +use time::{Duration, OffsetDateTime, format_description::well_known::Rfc3339}; + +mod support; + +use support::{ClickHouseDatabase, TestResult, database}; + +struct Feedback<'a> { + team: &'a str, + key: &'a str, + trace: &'a str, + author: &'a str, + score: u8, + comment: &'a str, + edited_after_seconds: i64, + deleted: bool, +} + +const SAVED: Feedback<'static> = Feedback { + team: "team-a", + key: "key-a", + trace: "trace-1", + author: "alice", + score: 3, + comment: "missed the file", + edited_after_seconds: 0, + deleted: false, +}; + +fn now() -> TestResult { + Ok(OffsetDateTime::now_utc().replace_millisecond(0)?) +} + +fn iso(at: OffsetDateTime) -> TestResult { + Ok(at.format(&Rfc3339)?) +} + +fn row(feedback: &Feedback, created: OffsetDateTime) -> TestResult> { + let updated = created + Duration::seconds(feedback.edited_after_seconds); + Ok(serde_json::from_value(json!({ + "TeamId": feedback.team, "ApiKeyHash": feedback.key, "TraceId": feedback.trace, + "Author": feedback.author, "Score": feedback.score, "Comment": feedback.comment, + "CreatedAt": iso(created)?, "UpdatedAt": iso(updated)?, + "IsDeleted": u8::from(feedback.deleted) + }))?) +} + +async fn ready(database: &ClickHouseDatabase) -> TestResult { + let writer = Connection::writer(&database.url)?; + ensure_schema(&database.client, &writer, "trace_test", 7).await?; + Ok(writer) +} + +async fn save( + database: &ClickHouseDatabase, + writer: &Connection, + created: OffsetDateTime, + feedback: &[Feedback<'_>], +) -> TestResult { + let rows = feedback + .iter() + .map(|entry| row(entry, created)) + .collect::>>()?; + insert_rows( + &database.client, + writer, + "trace_test", + InsertTable::LensFeedback, + rows, + ) + .await?; + Ok(()) +} + +fn access(team: &str) -> BTreeMap { + BTreeMap::from([ + ( + "all_teams".into(), + Parameter::Integer(i64::from(team.is_empty())), + ), + ("team".into(), Parameter::Text(team.into())), + ("key_hash".into(), Parameter::Text(String::new())), + ]) +} + +async fn read( + database: &ClickHouseDatabase, + query: ReadQuery, + parameters: BTreeMap, +) -> TestResult> { + let reader = Connection::configured(&database.url, "trace_test", "default", "")?; + let body: Value = serde_json::from_str( + &execute_named_read(&database.client, &reader, query, ¶meters).await?, + )?; + Ok(body["data"] + .as_array() + .cloned() + .ok_or_else(|| format!("no data in {body}"))?) +} + +async fn trace_ref( + database: &ClickHouseDatabase, + team: &str, + key: &str, + trace: &str, +) -> TestResult { + let reader = Connection::configured(&database.url, "trace_test", "default", "")?; + let sql = format!( + "SELECT hex(SHA256(concat('{team}', char(0), '{key}', char(0), '{trace}'))) AS trace_ref" + ); + let response: Value = serde_json::from_str( + &execute_read(&database.client, &reader, &sql, &BTreeMap::new()).await?, + )?; + Ok(response["data"][0]["trace_ref"] + .as_str() + .unwrap_or_default() + .to_owned()) +} + +async fn feedback( + database: &ClickHouseDatabase, + team: &str, + trace: &str, + trace_ref: &str, +) -> TestResult> { + let mut parameters = access(team); + parameters.insert("trace_id".into(), Parameter::Text(trace.into())); + parameters.insert("trace_ref".into(), Parameter::Text(trace_ref.into())); + read(database, ReadQuery::Feedback, parameters).await +} + +fn timestamp(row: &Value, field: &str) -> TestResult { + Ok(OffsetDateTime::parse( + row[field].as_str().unwrap_or_default(), + &Rfc3339, + )?) +} + +#[rstest] +#[tokio::test] +async fn a_later_save_replaces_the_authors_feedback_and_keeps_other_authors( + #[future(awt)] database: TestResult, +) -> TestResult { + let database = database?; + let writer = ready(&database).await?; + let created = now()?; + save(&database, &writer, created, &[SAVED]).await?; + save( + &database, + &writer, + created, + &[ + Feedback { + score: 8, + comment: "fine after retry", + edited_after_seconds: 300, + ..SAVED + }, + Feedback { + author: "bob", + score: 10, + comment: "", + ..SAVED + }, + ], + ) + .await?; + let reference = trace_ref(&database, "team-a", "key-a", "trace-1").await?; + + let rows = feedback(&database, "", "trace-1", &reference).await?; + + let shown: Vec<(&str, u64, &str)> = rows + .iter() + .map(|row| { + ( + row["author"].as_str().unwrap_or_default(), + row["score"].as_u64().unwrap_or(99), + row["comment"].as_str().unwrap_or_default(), + ) + }) + .collect(); + assert_eq!(shown, [("alice", 8, "fine after retry"), ("bob", 10, "")]); + assert_eq!(timestamp(&rows[0], "created_at")?, created); + assert_eq!( + timestamp(&rows[0], "updated_at")?, + created + Duration::seconds(300) + ); + Ok(()) +} + +#[rstest] +#[tokio::test] +async fn a_deleted_version_hides_the_authors_feedback( + #[future(awt)] database: TestResult, +) -> TestResult { + let database = database?; + let writer = ready(&database).await?; + let created = now()?; + save( + &database, + &writer, + created, + &[ + SAVED, + Feedback { + author: "bob", + score: 6, + ..SAVED + }, + ], + ) + .await?; + save( + &database, + &writer, + created, + &[Feedback { + deleted: true, + edited_after_seconds: 540, + ..SAVED + }], + ) + .await?; + let reference = trace_ref(&database, "team-a", "key-a", "trace-1").await?; + + let authors: Vec = feedback(&database, "", "trace-1", &reference) + .await? + .into_iter() + .map(|row| row["author"].clone()) + .collect(); + + assert_eq!(authors, [json!("bob")]); + Ok(()) +} + +#[rstest] +#[tokio::test] +async fn summaries_aggregate_live_feedback_per_trace_and_respect_team_scope( + #[future(awt)] database: TestResult, +) -> TestResult { + let database = database?; + let writer = ready(&database).await?; + save( + &database, + &writer, + now()?, + &[ + Feedback { score: 2, ..SAVED }, + Feedback { + author: "bob", + score: 7, + ..SAVED + }, + Feedback { + author: "carol", + score: 0, + deleted: true, + ..SAVED + }, + Feedback { + team: "team-b", + key: "key-b", + trace: "trace-2", + score: 9, + ..SAVED + }, + ], + ) + .await?; + let summary = |team: &str| { + let mut parameters = access(team); + parameters.insert( + "trace_ids".into(), + Parameter::Strings(vec![ + "trace-1".into(), + "trace-2".into(), + "trace-unrated".into(), + ]), + ); + parameters + }; + + let everyone = read(&database, ReadQuery::FeedbackSummary, summary("")).await?; + let team_a = read(&database, ReadQuery::FeedbackSummary, summary("team-a")).await?; + + let by_trace: BTreeMap<&str, (u64, f64, u64)> = everyone + .iter() + .map(|row| { + ( + row["trace_id"].as_str().unwrap_or_default(), + ( + row["count"].as_u64().unwrap_or(0), + row["average"].as_f64().unwrap_or(-1.0), + row["lowest"].as_u64().unwrap_or(99), + ), + ) + }) + .collect(); + assert_eq!( + by_trace, + BTreeMap::from([("trace-1", (2, 4.5, 2)), ("trace-2", (1, 9.0, 9))]) + ); + assert_eq!( + team_a + .iter() + .map(|row| row["trace_id"].clone()) + .collect::>(), + [json!("trace-1")] + ); + Ok(()) +} + +#[rstest] +#[tokio::test] +async fn feedback_for_another_teams_trace_is_not_readable( + #[future(awt)] database: TestResult, +) -> TestResult { + let database = database?; + let writer = ready(&database).await?; + save(&database, &writer, now()?, &[SAVED]).await?; + let reference = trace_ref(&database, "team-a", "key-a", "trace-1").await?; + + assert!( + feedback(&database, "team-b", "trace-1", &reference) + .await? + .is_empty() + ); + assert_eq!( + feedback(&database, "team-a", "trace-1", &reference) + .await? + .len(), + 1 + ); + Ok(()) +} + +#[rstest] +#[tokio::test] +async fn the_table_rejects_scores_above_ten( + #[future(awt)] database: TestResult, +) -> TestResult { + let database = database?; + let writer = ready(&database).await?; + + let rejected = save( + &database, + &writer, + now()?, + &[Feedback { score: 11, ..SAVED }], + ) + .await; + + assert!(rejected.is_err()); + Ok(()) +} + +#[rstest] +#[tokio::test] +async fn target_resolves_team_and_key_for_a_visible_trace_only( + #[future(awt)] database: TestResult, +) -> TestResult { + let database = database?; + let writer = ready(&database).await?; + let span: BTreeMap = serde_json::from_value(json!({ + "Timestamp": now()?.unix_timestamp_nanos() as i64, "TraceId": "trace-1", "SpanId": "root", + "ParentSpanId": "", "ServiceName": "agent", "SpanName": "run", + "ResourceAttributes": {"litellm.team_id": "team-a", "litellm.api_key_hash": "key-a"} + }))?; + insert_rows( + &database.client, + &writer, + "trace_test", + InsertTable::OtelTraces, + vec![span], + ) + .await?; + let target = |team: &str| { + let mut parameters = access(team); + parameters.insert("trace_id".into(), Parameter::Text("trace-1".into())); + parameters.insert("trace_ref".into(), Parameter::Text(String::new())); + parameters + }; + + let visible = read(&database, ReadQuery::FeedbackTarget, target("team-a")).await?; + let hidden = read(&database, ReadQuery::FeedbackTarget, target("team-b")).await?; + + assert_eq!(visible.len(), 1); + assert_eq!( + ( + visible[0]["team_id"].as_str(), + visible[0]["key_hash"].as_str() + ), + (Some("team-a"), Some("key-a")) + ); + assert_eq!( + visible[0]["trace_ref"].as_str().map(str::to_owned), + Some(trace_ref(&database, "team-a", "key-a", "trace-1").await?) + ); + assert!(hidden.is_empty()); + Ok(()) +} diff --git a/litellm-rust/crates/traces-clickhouse/tests/migrations.rs b/litellm-rust/crates/traces-clickhouse/tests/migrations.rs index 9a3cd4b3050..a00f34a0002 100644 --- a/litellm-rust/crates/traces-clickhouse/tests/migrations.rs +++ b/litellm-rust/crates/traces-clickhouse/tests/migrations.rs @@ -335,10 +335,10 @@ async fn concurrent_schema_setup_succeeds( &database, "SELECT count() AS tables FROM system.tables \ WHERE database = 'trace_test' AND name IN \ - ('otel_traces', 'agent_traces_by_key', 'spend_logs')", + ('otel_traces', 'agent_traces_by_key', 'spend_logs', 'lens_feedback')", ) .await?; - assert_eq!(tables["data"][0]["tables"].as_u64(), Some(3)); + assert_eq!(tables["data"][0]["tables"].as_u64(), Some(4)); assert_eq!( migration_ledger_versions(&database).await?, migration_versions() @@ -1040,6 +1040,7 @@ async fn retention_changes_materialize_existing_rows_and_remain_idempotent( tables["data"], serde_json::json!([ {"name": "agent_traces_by_key"}, + {"name": "lens_feedback"}, {"name": "otel_traces"}, {"name": "spend_logs"} ]) @@ -1057,7 +1058,13 @@ async fn retention_changes_materialize_existing_rows_and_remain_idempotent( "start_time": old_timestamp_ms, "end_time": old_timestamp_ms + 1000 }))?; insert_rows(&database, "otel_traces", vec![span]).await?; + let old_iso = old_time.format(&time::format_description::well_known::Rfc3339)?; + let feedback = serde_json::from_value(serde_json::json!({ + "TeamId": "team-1", "ApiKeyHash": "", "TraceId": "expired", "Author": "admin", + "Score": 4, "Comment": "", "CreatedAt": old_iso, "UpdatedAt": old_iso, "IsDeleted": 0 + }))?; insert_rows(&database, "spend_logs", vec![spend]).await?; + insert_rows(&database, "lens_feedback", vec![feedback]).await?; assert_eq!(table_rows(&database, "agent_traces_by_key").await?, 1); ensure_schema(&database.client, &writer, "trace_test", 14).await?; let deadline = tokio::time::Instant::now() + Duration::from_secs(60); @@ -1087,6 +1094,8 @@ async fn retention_changes_materialize_existing_rows_and_remain_idempotent( ) .await?; execute_write(&database, "OPTIMIZE TABLE trace_test.spend_logs FINAL").await?; + execute_write(&database, "OPTIMIZE TABLE trace_test.lens_feedback FINAL").await?; + assert_eq!(table_rows(&database, "lens_feedback").await?, 0); assert_eq!(table_rows(&database, "otel_traces").await?, 0); assert_eq!(table_rows(&database, "agent_traces_by_key").await?, 0); assert_eq!(table_rows(&database, "spend_logs").await?, 0); @@ -1109,7 +1118,7 @@ async fn retention_reconciliation_updates_each_table_ttl( &database, "SELECT name, create_table_query FROM system.tables \ WHERE database = 'trace_test' AND name IN \ - ('otel_traces', 'agent_traces_by_key', 'spend_logs') ORDER BY name", + ('otel_traces', 'agent_traces_by_key', 'spend_logs', 'lens_feedback') ORDER BY name", ) .await?; let ttl_queries = ttl_queries["data"].as_array().expect("retention tables"); @@ -1118,7 +1127,12 @@ async fn retention_reconciliation_updates_each_table_ttl( .iter() .map(|row| row["name"].as_str().expect("table name")) .collect::>(), - ["agent_traces_by_key", "otel_traces", "spend_logs"] + [ + "agent_traces_by_key", + "lens_feedback", + "otel_traces", + "spend_logs" + ] ); for row in ttl_queries { let query = row["create_table_query"] @@ -1135,7 +1149,7 @@ async fn retention_reconciliation_updates_each_table_ttl( &database, "SELECT name, create_table_query FROM system.tables \ WHERE database = 'trace_test' AND name IN \ - ('otel_traces', 'agent_traces_by_key', 'spend_logs') ORDER BY name", + ('otel_traces', 'agent_traces_by_key', 'spend_logs', 'lens_feedback') ORDER BY name", ) .await?; for row in ttl_queries["data"].as_array().expect("retention tables") { diff --git a/litellm-rust/crates/traces/src/query.rs b/litellm-rust/crates/traces/src/query.rs index e2653e15b61..26eaa33d5ac 100644 --- a/litellm-rust/crates/traces/src/query.rs +++ b/litellm-rust/crates/traces/src/query.rs @@ -17,6 +17,9 @@ pub enum ReadQuery { Sample, Content, Evidence, + FeedbackTarget, + Feedback, + FeedbackSummary, } impl ReadQuery { diff --git a/litellm-rust/crates/traces/tests/otlp.rs b/litellm-rust/crates/traces/tests/otlp.rs index 68a25fb028b..1b817a0bef5 100644 --- a/litellm-rust/crates/traces/tests/otlp.rs +++ b/litellm-rust/crates/traces/tests/otlp.rs @@ -1650,6 +1650,9 @@ fn session_capture_joins_native_logs_and_traces_across_turns_without_changing_sp let first = litellm_traces::decode_otlp_logs(&logs.encode_to_vec(), None).unwrap(); let second = decode_otlp(&request.encode_to_vec(), None).unwrap(); assert_eq!(first[0].trace_id, second[0].trace_id); + // Lens feedback resolves session ids the same way; keep in sync with + // litellm/proxy/lens/feedback_repository.py::session_trace_id. + assert_eq!(second[0].trace_id, "5fddf060372c8501dca4f331b9da882b"); assert_eq!( first[0].attributes["lens.original_trace_id"], "01".repeat(16) diff --git a/litellm-rust/crates/traces/tests/query.rs b/litellm-rust/crates/traces/tests/query.rs index 67f422d6997..7c125b01379 100644 --- a/litellm-rust/crates/traces/tests/query.rs +++ b/litellm-rust/crates/traces/tests/query.rs @@ -14,6 +14,9 @@ use rstest::rstest; #[case::sample("sample", ReadQuery::Sample)] #[case::content("content", ReadQuery::Content)] #[case::evidence("evidence", ReadQuery::Evidence)] +#[case::feedback_target("feedback_target", ReadQuery::FeedbackTarget)] +#[case::feedback("feedback", ReadQuery::Feedback)] +#[case::feedback_summary("feedback_summary", ReadQuery::FeedbackSummary)] fn names_select_the_public_query(#[case] name: &str, #[case] query: ReadQuery) { assert_eq!(ReadQuery::parse(name).unwrap(), query); assert_eq!(query.as_ref(), name); diff --git a/litellm/constants.py b/litellm/constants.py index b09da3d6e7d..c4adc0de22f 100644 --- a/litellm/constants.py +++ b/litellm/constants.py @@ -64,6 +64,8 @@ AGENT_TRACING_AGENT_LIST_LIMIT: Final = get_env_int("AGENT_TRACING_AGENT_LIST_LI LENS_DATASET_MAX_CASES: Final = get_env_int("LENS_DATASET_MAX_CASES", 200) LENS_DATASET_MAX_CASE_CHARS: Final = get_env_int("LENS_DATASET_MAX_CASE_CHARS", 20_000) LENS_DATASET_TRACE_PAGE_SIZE: Final = 500 +LENS_FEEDBACK_MAX_COMMENT_CHARS: Final = 10_000 +LENS_FEEDBACK_MAX_SCORE: Final = 10 DEFAULT_S3_FLUSH_INTERVAL_SECONDS: Final = int(os.getenv("DEFAULT_S3_FLUSH_INTERVAL_SECONDS", 10)) DEFAULT_S3_BATCH_SIZE: Final = int(os.getenv("DEFAULT_S3_BATCH_SIZE", 512)) DEFAULT_S3_MAX_CONCURRENT_UPLOADS: Final = int(os.getenv("DEFAULT_S3_MAX_CONCURRENT_UPLOADS", "16")) diff --git a/litellm/proxy/_types.py b/litellm/proxy/_types.py index d4229c825b6..ff36f3f78db 100644 --- a/litellm/proxy/_types.py +++ b/litellm/proxy/_types.py @@ -549,6 +549,8 @@ class LiteLLMRoutes(enum.Enum): "/lens/{lens_id}/executions/{execution_id}", "/lens/{lens_id}/cancel", "/lens/{lens_id}/findings/{finding_id}", + "/lens/feedback", + "/lens/feedback/summary", "/lens/preview/sample", "/lens/workers/register", "/lens/workers/{worker_id}", @@ -1064,6 +1066,7 @@ class LiteLLMRoutes(enum.Enum): admin_viewer_routes = ( [ "/lens/traces/findings", + "/lens/feedback/summary", "/user/list", "/user/available_users", "/user/available_roles", diff --git a/litellm/proxy/lens/feedback_endpoints.py b/litellm/proxy/lens/feedback_endpoints.py new file mode 100644 index 00000000000..f9479873a84 --- /dev/null +++ b/litellm/proxy/lens/feedback_endpoints.py @@ -0,0 +1,124 @@ +from datetime import datetime, timezone +from typing import Annotated, Final, TypeAlias + +from fastapi import APIRouter, Depends, HTTPException, Query, Response +from pydantic import Field, model_validator + +from litellm.proxy._types import LitellmUserRoles, UserAPIKeyAuth +from litellm.proxy.lens.endpoints import Auth, user_scope +from litellm.proxy.lens.feedback_models import ( + Feedback, + FeedbackInput, + TraceFeedback, + TraceFeedbackRequest, + TraceFeedbackSummary, +) +from litellm.proxy.lens.feedback_repository import ( + ClickHouseFeedbackStore, + FeedbackStore, + FeedbackWrite, + session_trace_id, +) +from litellm.proxy.lens.models import Record, Scope, TraceIdentity +from litellm.proxy.tracing_runtime import provide_storage +from litellm.rust_bridge.trace.storage import ClickHouseStorage + +router: Final = APIRouter(prefix="/lens/feedback", tags=["Lens"]) +TRACE_NOT_FOUND: Final = "Trace not found" + + +class FeedbackTarget(Record): + trace_id: str | None = Field(default=None, min_length=1, max_length=128) + session_id: str | None = Field(default=None, min_length=1, max_length=512) + trace_ref: str = Field(default="", max_length=512) + + @model_validator(mode="after") + def one_target(self) -> "FeedbackTarget": + if (self.trace_id is None) == (self.session_id is None): + raise ValueError("Pass exactly one of trace_id or session_id") + return self + + def trace(self) -> TraceIdentity: + trace_id: Final = self.trace_id if self.trace_id is not None else session_trace_id(self.session_id or "") + return TraceIdentity(trace_id=trace_id, trace_ref=self.trace_ref) + + +class FeedbackSubmission(FeedbackTarget, FeedbackInput): + pass + + +class FeedbackDeletion(FeedbackTarget): + user: str = Field(default="", max_length=256) + + +def write_scope(auth: UserAPIKeyAuth) -> Scope: + if auth.user_role == LitellmUserRoles.PROXY_ADMIN: + return Scope(all_teams=True) + if auth.user_role == LitellmUserRoles.PROXY_ADMIN_VIEW_ONLY: + raise HTTPException(403, "Admin viewers cannot write feedback") + return Scope(team_id=auth.team_id or "", api_key_hash="" if auth.team_id else auth.token or "") + + +def feedback_store( + storage: Annotated[ClickHouseStorage | None, Depends(provide_storage)], +) -> FeedbackStore: + if storage is None: + raise HTTPException(501, "Lens feedback needs agent tracing. Configure the Lens service and LITELLM_LENS_URL.") + return ClickHouseFeedbackStore(storage) + + +def stored_now() -> datetime: + now: Final = datetime.now(timezone.utc) + return now.replace(microsecond=now.microsecond // 1000 * 1000) + + +def author(auth: UserAPIKeyAuth, user: str) -> str: + identity: Final = user or auth.user_id or auth.token + if not identity: + raise HTTPException(422, "Name the user who left this feedback") + return identity + + +Store: TypeAlias = Annotated[FeedbackStore, Depends(feedback_store)] +Now: TypeAlias = Annotated[datetime, Depends(stored_now)] +Target: TypeAlias = Annotated[FeedbackTarget, Query()] +Deletion: TypeAlias = Annotated[FeedbackDeletion, Query()] + + +@router.get("", response_model=TraceFeedback) +async def read_feedback(target: Target, auth: Auth, store: Store) -> TraceFeedback: + feedback: Final = await store.for_trace(user_scope(auth), target.trace()) + if feedback is None: + raise HTTPException(404, TRACE_NOT_FOUND) + return feedback + + +@router.put("", response_model=Feedback) +async def submit_feedback(body: FeedbackSubmission, auth: Auth, store: Store, now: Now) -> Feedback: + saved: Final = await store.upsert( + write_scope(auth), + FeedbackWrite( + trace=body.trace(), + author=author(auth, body.user), + feedback=FeedbackInput(score=body.score, comment=body.comment), + at=now, + ), + ) + if saved is None: + raise HTTPException(404, TRACE_NOT_FOUND) + return saved + + +@router.delete("", status_code=204) +async def delete_feedback(target: Deletion, auth: Auth, store: Store, now: Now) -> Response: + deleted: Final = await store.delete(write_scope(auth), target.trace(), author(auth, target.user), now) + if deleted is None: + raise HTTPException(404, TRACE_NOT_FOUND) + if not deleted: + raise HTTPException(404, "No feedback from this user on this trace") + return Response(status_code=204) + + +@router.post("/summary", response_model=tuple[TraceFeedbackSummary, ...]) +async def feedback_summary(body: TraceFeedbackRequest, auth: Auth, store: Store) -> tuple[TraceFeedbackSummary, ...]: + return await store.summaries(user_scope(auth), body.traces) diff --git a/litellm/proxy/lens/feedback_models.py b/litellm/proxy/lens/feedback_models.py new file mode 100644 index 00000000000..31429c9dadc --- /dev/null +++ b/litellm/proxy/lens/feedback_models.py @@ -0,0 +1,37 @@ +from datetime import datetime +from typing import Annotated, TypeAlias + +from pydantic import Field + +from litellm.constants import LENS_FEEDBACK_MAX_COMMENT_CHARS, LENS_FEEDBACK_MAX_SCORE +from litellm.proxy.lens.models import Record, TraceIdentity + +FeedbackScore: TypeAlias = Annotated[int, Field(ge=0, le=LENS_FEEDBACK_MAX_SCORE)] + + +class FeedbackInput(Record): + score: FeedbackScore + comment: str = Field(default="", max_length=LENS_FEEDBACK_MAX_COMMENT_CHARS) + user: str = Field(default="", max_length=256) + + +class Feedback(TraceIdentity): + score: FeedbackScore + comment: str + author: str + created_at: datetime + updated_at: datetime + + +class TraceFeedback(TraceIdentity): + feedback: tuple[Feedback, ...] + + +class TraceFeedbackSummary(TraceIdentity): + count: int = Field(ge=0) + average: float | None = Field(ge=0, le=LENS_FEEDBACK_MAX_SCORE) + lowest: FeedbackScore | None + + +class TraceFeedbackRequest(Record): + traces: tuple[TraceIdentity, ...] = Field(min_length=1, max_length=500) diff --git a/litellm/proxy/lens/feedback_repository.py b/litellm/proxy/lens/feedback_repository.py new file mode 100644 index 00000000000..9ee1af164e2 --- /dev/null +++ b/litellm/proxy/lens/feedback_repository.py @@ -0,0 +1,186 @@ +import hashlib +from collections.abc import Mapping +from datetime import datetime, timezone +from typing import Final, Protocol + +from litellm.proxy.lens.feedback_models import Feedback, FeedbackInput, TraceFeedback, TraceFeedbackSummary +from litellm.proxy.lens.models import Record, Scope, TraceIdentity +from litellm.proxy.lens.sources import access_parameters +from litellm.rust_bridge.trace.generated.models import ( + FeedbackRow, + FeedbackSummaryRow, + LensFeedbackParams, + LensFeedbackSummaryParams, + LensFeedbackTargetParams, +) +from litellm.rust_bridge.trace.queries import LENS_FEEDBACK, LENS_FEEDBACK_SUMMARY, LENS_FEEDBACK_TARGET +from litellm.rust_bridge.trace.storage import ClickHouseStorage + +FEEDBACK_TABLE: Final = "lens_feedback" + + +def session_trace_id(session_id: str) -> str: + return hashlib.sha256(f"litellm.claude.session.v1\0{session_id}".encode()).digest()[:16].hex() + + +class FeedbackWrite(Record): + trace: TraceIdentity + author: str + feedback: FeedbackInput + at: datetime + + +class FeedbackStore(Protocol): + async def for_trace(self, scope: Scope, trace: TraceIdentity) -> TraceFeedback | None: ... + async def upsert(self, scope: Scope, write: FeedbackWrite) -> Feedback | None: ... + async def delete(self, scope: Scope, trace: TraceIdentity, author: str, at: datetime) -> bool | None: ... + async def summaries(self, scope: Scope, traces: tuple[TraceIdentity, ...]) -> tuple[TraceFeedbackSummary, ...]: ... + + +class _Target(Record): + trace_id: str + trace_ref: str + team_id: str + key_hash: str + + +def _iso(at: datetime) -> str: + return at.astimezone(timezone.utc).isoformat(timespec="milliseconds").replace("+00:00", "Z") + + +def _feedback(row: FeedbackRow) -> Feedback: + return Feedback( + trace_id=row.trace_id, + trace_ref=row.trace_ref, + score=row.score, + comment=row.comment, + author=row.author, + created_at=datetime.fromisoformat(row.created_at), + updated_at=datetime.fromisoformat(row.updated_at), + ) + + +class ClickHouseFeedbackStore: + def __init__(self, storage: ClickHouseStorage) -> None: + self.storage: Final = storage + + async def _target(self, scope: Scope, trace: TraceIdentity) -> _Target | None: + rows: Final = await self.storage.query( + LENS_FEEDBACK_TARGET, + LensFeedbackTargetParams( + **access_parameters(scope).model_dump(), trace_id=trace.trace_id, trace_ref=trace.trace_ref + ), + ) + if len(rows) != 1: + return None + return _Target( + trace_id=trace.trace_id, trace_ref=rows[0].trace_ref, team_id=rows[0].team_id, key_hash=rows[0].key_hash + ) + + async def _rows(self, scope: Scope, trace: TraceIdentity) -> tuple[Feedback, ...]: + rows: Final = await self.storage.query( + LENS_FEEDBACK, + LensFeedbackParams( + **access_parameters(scope).model_dump(), trace_id=trace.trace_id, trace_ref=trace.trace_ref + ), + ) + return tuple(_feedback(row) for row in rows) + + async def for_trace(self, scope: Scope, trace: TraceIdentity) -> TraceFeedback | None: + target: Final = await self._target(scope, trace) + if target is None: + return None + identity: Final = TraceIdentity(trace_id=target.trace_id, trace_ref=target.trace_ref) + return TraceFeedback( + trace_id=identity.trace_id, trace_ref=identity.trace_ref, feedback=await self._rows(scope, identity) + ) + + async def _write(self, target: _Target, author: str, values: Mapping[str, object]) -> None: + await self.storage.insert_rows( + FEEDBACK_TABLE, + ( + { + "TeamId": target.team_id, + "ApiKeyHash": target.key_hash, + "TraceId": target.trace_id, + "Author": author, + **values, + }, + ), + ) + + async def upsert(self, scope: Scope, write: FeedbackWrite) -> Feedback | None: + target: Final = await self._target(scope, write.trace) + if target is None: + return None + identity: Final = TraceIdentity(trace_id=target.trace_id, trace_ref=target.trace_ref) + previous: Final = next((f for f in await self._rows(scope, identity) if f.author == write.author), None) + created: Final = previous.created_at if previous else write.at + await self._write( + target, + write.author, + { + "Score": write.feedback.score, + "Comment": write.feedback.comment, + "CreatedAt": _iso(created), + "UpdatedAt": _iso(write.at), + "IsDeleted": 0, + }, + ) + return Feedback( + trace_id=target.trace_id, + trace_ref=target.trace_ref, + score=write.feedback.score, + comment=write.feedback.comment, + author=write.author, + created_at=created, + updated_at=write.at, + ) + + async def delete(self, scope: Scope, trace: TraceIdentity, author: str, at: datetime) -> bool | None: + target: Final = await self._target(scope, trace) + if target is None: + return None + identity: Final = TraceIdentity(trace_id=target.trace_id, trace_ref=target.trace_ref) + previous: Final = next((f for f in await self._rows(scope, identity) if f.author == author), None) + if previous is None: + return False + await self._write( + target, + author, + { + "Score": previous.score, + "Comment": "", + "CreatedAt": _iso(previous.created_at), + "UpdatedAt": _iso(at), + "IsDeleted": 1, + }, + ) + return True + + async def summaries(self, scope: Scope, traces: tuple[TraceIdentity, ...]) -> tuple[TraceFeedbackSummary, ...]: + rows: Final = await self.storage.query( + LENS_FEEDBACK_SUMMARY, + LensFeedbackSummaryParams( + **access_parameters(scope).model_dump(), trace_ids=sorted({t.trace_id for t in traces}) + ), + ) + return tuple(summary for trace in traces for summary in _summaries(trace, rows)) + + +def _summaries(trace: TraceIdentity, rows: tuple[FeedbackSummaryRow, ...]) -> tuple[TraceFeedbackSummary, ...]: + matched: Final = tuple( + row for row in rows if row.trace_id == trace.trace_id and trace.trace_ref in ("", row.trace_ref) + ) + if not matched: + return ( + TraceFeedbackSummary( + trace_id=trace.trace_id, trace_ref=trace.trace_ref, count=0, average=None, lowest=None + ), + ) + return tuple( + TraceFeedbackSummary( + trace_id=row.trace_id, trace_ref=row.trace_ref, count=row.count, average=row.average, lowest=row.lowest + ) + for row in matched + ) diff --git a/litellm/proxy/proxy_server.py b/litellm/proxy/proxy_server.py index 8afacda2e4c..aa086e25411 100644 --- a/litellm/proxy/proxy_server.py +++ b/litellm/proxy/proxy_server.py @@ -595,6 +595,7 @@ from litellm.proxy.hooks.proxy_track_cost_callback import ( # noqa: F401, RUF10 from litellm.proxy.image_endpoints.endpoints import router as image_router from litellm.proxy.lens.dataset_endpoints import router as lens_dataset_router from litellm.proxy.lens.endpoints import router as lens_router +from litellm.proxy.lens.feedback_endpoints import router as lens_feedback_router from litellm.proxy.lens.repository import WriterDatabase from litellm.proxy.lens.signal_repository import SignalRepository from litellm.proxy.lens.signals import ( @@ -20307,6 +20308,7 @@ app.include_router(tag_management_router) app.include_router(workflow_management_router) app.include_router(memory_router) app.include_router(lens_dataset_router) +app.include_router(lens_feedback_router) app.include_router(lens_router) app.include_router(plugin_router) app.include_router(cost_tracking_settings_router) diff --git a/litellm/rust_bridge/trace/generated/models.py b/litellm/rust_bridge/trace/generated/models.py index 7a748102991..c9600cc545a 100644 --- a/litellm/rust_bridge/trace/generated/models.py +++ b/litellm/rust_bridge/trace/generated/models.py @@ -192,6 +192,141 @@ class ExecutionRow(LiteLLMBaseModel): selection_key: str = "" +Score: TypeAlias = Annotated[ + int, + Field( + ..., + ge=0, + json_schema_extra={ + "x-python-normalized": { + "type": "int", + "minimum": 0, + "maximum": 18446744073709551615, + } + }, + le=18446744073709551615, + ), +] + + +Score1: TypeAlias = Annotated[ + str, + Field( + ..., + json_schema_extra={ + "x-python-normalized": { + "type": "int", + "minimum": 0, + "maximum": 18446744073709551615, + } + }, + pattern="^(?:0|[1-9][0-9]{0,18}|1[0-7][0-9]{18}|18[0-3][0-9]{17}|184[0-3][0-9]{16}|1844[0-5][0-9]{15}|18446[0-6][0-9]{14}|184467[0-3][0-9]{13}|1844674[0-3][0-9]{12}|184467440[0-6][0-9]{10}|1844674407[0-2][0-9]{9}|18446744073[0-6][0-9]{8}|1844674407370[0-8][0-9]{6}|18446744073709[0-4][0-9]{5}|184467440737095[0-4][0-9]{4}|1844674407370955[0-0][0-9]{3}|18446744073709551[0-5][0-9]{2}|184467440737095516[0-0][0-9]{1}|1844674407370955161[0-4][0-9]{0}|18446744073709551615)$", + ), +] + + +class FeedbackRow(LiteLLMBaseModel): + model_config = ConfigDict( + frozen=True, + ) + + trace_id: str + trace_ref: str + author: str + score: int = Field(..., ge=0, le=18446744073709551615) + comment: str + created_at: str + updated_at: str + + +Count2: TypeAlias = Annotated[ + int, + Field( + ..., + ge=0, + json_schema_extra={ + "x-python-normalized": { + "type": "int", + "minimum": 0, + "maximum": 18446744073709551615, + } + }, + le=18446744073709551615, + ), +] + + +Count3: TypeAlias = Annotated[ + str, + Field( + ..., + json_schema_extra={ + "x-python-normalized": { + "type": "int", + "minimum": 0, + "maximum": 18446744073709551615, + } + }, + pattern="^(?:0|[1-9][0-9]{0,18}|1[0-7][0-9]{18}|18[0-3][0-9]{17}|184[0-3][0-9]{16}|1844[0-5][0-9]{15}|18446[0-6][0-9]{14}|184467[0-3][0-9]{13}|1844674[0-3][0-9]{12}|184467440[0-6][0-9]{10}|1844674407[0-2][0-9]{9}|18446744073[0-6][0-9]{8}|1844674407370[0-8][0-9]{6}|18446744073709[0-4][0-9]{5}|184467440737095[0-4][0-9]{4}|1844674407370955[0-0][0-9]{3}|18446744073709551[0-5][0-9]{2}|184467440737095516[0-0][0-9]{1}|1844674407370955161[0-4][0-9]{0}|18446744073709551615)$", + ), +] + + +Lowest: TypeAlias = Annotated[ + int, + Field( + ..., + ge=0, + json_schema_extra={ + "x-python-normalized": { + "type": "int", + "minimum": 0, + "maximum": 18446744073709551615, + } + }, + le=18446744073709551615, + ), +] + + +Lowest1: TypeAlias = Annotated[ + str, + Field( + ..., + json_schema_extra={ + "x-python-normalized": { + "type": "int", + "minimum": 0, + "maximum": 18446744073709551615, + } + }, + pattern="^(?:0|[1-9][0-9]{0,18}|1[0-7][0-9]{18}|18[0-3][0-9]{17}|184[0-3][0-9]{16}|1844[0-5][0-9]{15}|18446[0-6][0-9]{14}|184467[0-3][0-9]{13}|1844674[0-3][0-9]{12}|184467440[0-6][0-9]{10}|1844674407[0-2][0-9]{9}|18446744073[0-6][0-9]{8}|1844674407370[0-8][0-9]{6}|18446744073709[0-4][0-9]{5}|184467440737095[0-4][0-9]{4}|1844674407370955[0-0][0-9]{3}|18446744073709551[0-5][0-9]{2}|184467440737095516[0-0][0-9]{1}|1844674407370955161[0-4][0-9]{0}|18446744073709551615)$", + ), +] + + +class FeedbackSummaryRow(LiteLLMBaseModel): + model_config = ConfigDict( + frozen=True, + ) + + trace_id: str + trace_ref: str + count: int = Field(..., ge=0, le=18446744073709551615) + average: float + lowest: int = Field(..., ge=0, le=18446744073709551615) + + +class FeedbackTargetRow(LiteLLMBaseModel): + model_config = ConfigDict( + frozen=True, + ) + + team_id: str + key_hash: str + trace_ref: str + + class LensAccessParams(LiteLLMBaseModel): model_config = ConfigDict( extra="forbid", @@ -239,6 +374,44 @@ class LensEvidenceParams(LiteLLMBaseModel): quote: str +class LensFeedbackParams(LiteLLMBaseModel): + model_config = ConfigDict( + extra="forbid", + frozen=True, + ) + + all_teams: Literal[0, 1] + team: str + key_hash: str + trace_id: str + trace_ref: str + + +class LensFeedbackSummaryParams(LiteLLMBaseModel): + model_config = ConfigDict( + extra="forbid", + frozen=True, + ) + + all_teams: Literal[0, 1] + team: str + key_hash: str + trace_ids: tuple[str, ...] + + +class LensFeedbackTargetParams(LiteLLMBaseModel): + model_config = ConfigDict( + extra="forbid", + frozen=True, + ) + + all_teams: Literal[0, 1] + team: str + key_hash: str + trace_id: str + trace_ref: str + + ExecutionSource: TypeAlias = Literal["traces", "requests", "both"] @@ -555,9 +728,15 @@ TraceWireModels: TypeAlias = Annotated[ | AgentRow | CountRow | ExecutionRow + | FeedbackRow + | FeedbackSummaryRow + | FeedbackTargetRow | LensAccessParams | LensContentParams | LensEvidenceParams + | LensFeedbackParams + | LensFeedbackSummaryParams + | LensFeedbackTargetParams | LensSampleParams | PartRow | TraceAgentRow diff --git a/litellm/rust_bridge/trace/generated/types.py b/litellm/rust_bridge/trace/generated/types.py index 12f38da6e28..16f22bc0b95 100644 --- a/litellm/rust_bridge/trace/generated/types.py +++ b/litellm/rust_bridge/trace/generated/types.py @@ -90,7 +90,17 @@ class TraceScope(typing_extensions.TypedDict): team_ids: ReadOnly[tuple[str, ...]] -ReadQueryName: TypeAlias = Literal["trace_agents", "availability", "agents", "sample", "content", "evidence"] +ReadQueryName: TypeAlias = Literal[ + "trace_agents", + "availability", + "agents", + "sample", + "content", + "evidence", + "feedback_target", + "feedback", + "feedback_summary", +] class UIFields(typing_extensions.TypedDict): diff --git a/litellm/rust_bridge/trace/queries.py b/litellm/rust_bridge/trace/queries.py index 8d1ee978b2b..5bfc2303057 100644 --- a/litellm/rust_bridge/trace/queries.py +++ b/litellm/rust_bridge/trace/queries.py @@ -11,9 +11,15 @@ from .generated.models import ( AgentRow, CountRow, ExecutionRow, + FeedbackRow, + FeedbackSummaryRow, + FeedbackTargetRow, LensAccessParams, LensContentParams, LensEvidenceParams, + LensFeedbackParams, + LensFeedbackSummaryParams, + LensFeedbackTargetParams, LensSampleParams, PartRow, TraceAgentRow, @@ -74,3 +80,12 @@ LENS_CONTENT: Final[ReadQuery[LensContentParams, PartRow]] = ReadQuery( LENS_EVIDENCE: Final[ReadQuery[LensEvidenceParams, CountRow]] = ReadQuery( "evidence", LensEvidenceParams, TypeAdapter(QueryResponse[CountRow]) ) +LENS_FEEDBACK_TARGET: Final[ReadQuery[LensFeedbackTargetParams, FeedbackTargetRow]] = ReadQuery( + "feedback_target", LensFeedbackTargetParams, TypeAdapter(QueryResponse[FeedbackTargetRow]) +) +LENS_FEEDBACK: Final[ReadQuery[LensFeedbackParams, FeedbackRow]] = ReadQuery( + "feedback", LensFeedbackParams, TypeAdapter(QueryResponse[FeedbackRow]) +) +LENS_FEEDBACK_SUMMARY: Final[ReadQuery[LensFeedbackSummaryParams, FeedbackSummaryRow]] = ReadQuery( + "feedback_summary", LensFeedbackSummaryParams, TypeAdapter(QueryResponse[FeedbackSummaryRow]) +) diff --git a/litellm/tracing/remote.py b/litellm/tracing/remote.py index 5d71557b8b2..187300d6e23 100644 --- a/litellm/tracing/remote.py +++ b/litellm/tracing/remote.py @@ -16,6 +16,7 @@ from litellm.rust_bridge.trace.generated.types import QueryScope, ReadQueryName, MAX_RESPONSE_BYTES: Final = 64 * 1024 * 1024 _JSON: Final[TypeAdapter[JsonValue]] = TypeAdapter(JsonValue) +_INSERT_PATHS: Final[Mapping[str, str]] = {"spend_logs": "/internal/spend", "lens_feedback": "/internal/feedback"} @dataclass(frozen=True, slots=True, repr=False) @@ -125,9 +126,10 @@ class RemoteTraceStore: return _ReadFailure.INVALID_RESPONSE async def insert_rows(self, table: str, rows: Sequence[Mapping[str, object]]) -> None: - if table != "spend_logs": - raise ValueError("Lens only accepts gateway request records on this endpoint") - response: Final = await self.client.post("/internal/spend", json=tuple(dict(row) for row in rows)) + path: Final = _INSERT_PATHS.get(table) + if path is None: + raise ValueError("Lens only accepts gateway request records and feedback on this endpoint") + response: Final = await self.client.post(path, json=tuple(dict(row) for row in rows)) response.raise_for_status() async def ingest( diff --git a/scripts/trace_codegen/schemas/traces-clickhouse/FeedbackRow.json b/scripts/trace_codegen/schemas/traces-clickhouse/FeedbackRow.json new file mode 100644 index 00000000000..a107aea47e3 --- /dev/null +++ b/scripts/trace_codegen/schemas/traces-clickhouse/FeedbackRow.json @@ -0,0 +1,53 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "properties": { + "author": { + "type": "string" + }, + "comment": { + "type": "string" + }, + "created_at": { + "type": "string" + }, + "score": { + "anyOf": [ + { + "format": "uint64", + "maximum": 18446744073709551615, + "minimum": 0, + "type": "integer" + }, + { + "pattern": "^(?:0|[1-9][0-9]{0,18}|1[0-7][0-9]{18}|18[0-3][0-9]{17}|184[0-3][0-9]{16}|1844[0-5][0-9]{15}|18446[0-6][0-9]{14}|184467[0-3][0-9]{13}|1844674[0-3][0-9]{12}|184467440[0-6][0-9]{10}|1844674407[0-2][0-9]{9}|18446744073[0-6][0-9]{8}|1844674407370[0-8][0-9]{6}|18446744073709[0-4][0-9]{5}|184467440737095[0-4][0-9]{4}|1844674407370955[0-0][0-9]{3}|18446744073709551[0-5][0-9]{2}|184467440737095516[0-0][0-9]{1}|1844674407370955161[0-4][0-9]{0}|18446744073709551615)$", + "type": "string" + } + ], + "x-python-normalized": { + "maximum": 18446744073709551615, + "minimum": 0, + "type": "int" + } + }, + "trace_id": { + "type": "string" + }, + "trace_ref": { + "type": "string" + }, + "updated_at": { + "type": "string" + } + }, + "required": [ + "trace_id", + "trace_ref", + "author", + "score", + "comment", + "created_at", + "updated_at" + ], + "title": "FeedbackRow", + "type": "object" +} diff --git a/scripts/trace_codegen/schemas/traces-clickhouse/FeedbackSummaryRow.json b/scripts/trace_codegen/schemas/traces-clickhouse/FeedbackSummaryRow.json new file mode 100644 index 00000000000..53f0e7e306a --- /dev/null +++ b/scripts/trace_codegen/schemas/traces-clickhouse/FeedbackSummaryRow.json @@ -0,0 +1,62 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "properties": { + "average": { + "format": "double", + "type": "number" + }, + "count": { + "anyOf": [ + { + "format": "uint64", + "maximum": 18446744073709551615, + "minimum": 0, + "type": "integer" + }, + { + "pattern": "^(?:0|[1-9][0-9]{0,18}|1[0-7][0-9]{18}|18[0-3][0-9]{17}|184[0-3][0-9]{16}|1844[0-5][0-9]{15}|18446[0-6][0-9]{14}|184467[0-3][0-9]{13}|1844674[0-3][0-9]{12}|184467440[0-6][0-9]{10}|1844674407[0-2][0-9]{9}|18446744073[0-6][0-9]{8}|1844674407370[0-8][0-9]{6}|18446744073709[0-4][0-9]{5}|184467440737095[0-4][0-9]{4}|1844674407370955[0-0][0-9]{3}|18446744073709551[0-5][0-9]{2}|184467440737095516[0-0][0-9]{1}|1844674407370955161[0-4][0-9]{0}|18446744073709551615)$", + "type": "string" + } + ], + "x-python-normalized": { + "maximum": 18446744073709551615, + "minimum": 0, + "type": "int" + } + }, + "lowest": { + "anyOf": [ + { + "format": "uint64", + "maximum": 18446744073709551615, + "minimum": 0, + "type": "integer" + }, + { + "pattern": "^(?:0|[1-9][0-9]{0,18}|1[0-7][0-9]{18}|18[0-3][0-9]{17}|184[0-3][0-9]{16}|1844[0-5][0-9]{15}|18446[0-6][0-9]{14}|184467[0-3][0-9]{13}|1844674[0-3][0-9]{12}|184467440[0-6][0-9]{10}|1844674407[0-2][0-9]{9}|18446744073[0-6][0-9]{8}|1844674407370[0-8][0-9]{6}|18446744073709[0-4][0-9]{5}|184467440737095[0-4][0-9]{4}|1844674407370955[0-0][0-9]{3}|18446744073709551[0-5][0-9]{2}|184467440737095516[0-0][0-9]{1}|1844674407370955161[0-4][0-9]{0}|18446744073709551615)$", + "type": "string" + } + ], + "x-python-normalized": { + "maximum": 18446744073709551615, + "minimum": 0, + "type": "int" + } + }, + "trace_id": { + "type": "string" + }, + "trace_ref": { + "type": "string" + } + }, + "required": [ + "trace_id", + "trace_ref", + "count", + "average", + "lowest" + ], + "title": "FeedbackSummaryRow", + "type": "object" +} diff --git a/scripts/trace_codegen/schemas/traces-clickhouse/FeedbackTargetRow.json b/scripts/trace_codegen/schemas/traces-clickhouse/FeedbackTargetRow.json new file mode 100644 index 00000000000..17b9adb2d68 --- /dev/null +++ b/scripts/trace_codegen/schemas/traces-clickhouse/FeedbackTargetRow.json @@ -0,0 +1,21 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "properties": { + "key_hash": { + "type": "string" + }, + "team_id": { + "type": "string" + }, + "trace_ref": { + "type": "string" + } + }, + "required": [ + "team_id", + "key_hash", + "trace_ref" + ], + "title": "FeedbackTargetRow", + "type": "object" +} diff --git a/scripts/trace_codegen/schemas/traces-clickhouse/LensFeedbackParams.json b/scripts/trace_codegen/schemas/traces-clickhouse/LensFeedbackParams.json new file mode 100644 index 00000000000..63794197547 --- /dev/null +++ b/scripts/trace_codegen/schemas/traces-clickhouse/LensFeedbackParams.json @@ -0,0 +1,34 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "additionalProperties": false, + "properties": { + "all_teams": { + "enum": [ + 0, + 1 + ], + "type": "integer" + }, + "key_hash": { + "type": "string" + }, + "team": { + "type": "string" + }, + "trace_id": { + "type": "string" + }, + "trace_ref": { + "type": "string" + } + }, + "required": [ + "all_teams", + "team", + "key_hash", + "trace_id", + "trace_ref" + ], + "title": "LensFeedbackParams", + "type": "object" +} diff --git a/scripts/trace_codegen/schemas/traces-clickhouse/LensFeedbackSummaryParams.json b/scripts/trace_codegen/schemas/traces-clickhouse/LensFeedbackSummaryParams.json new file mode 100644 index 00000000000..dde3eb83612 --- /dev/null +++ b/scripts/trace_codegen/schemas/traces-clickhouse/LensFeedbackSummaryParams.json @@ -0,0 +1,33 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "additionalProperties": false, + "properties": { + "all_teams": { + "enum": [ + 0, + 1 + ], + "type": "integer" + }, + "key_hash": { + "type": "string" + }, + "team": { + "type": "string" + }, + "trace_ids": { + "items": { + "type": "string" + }, + "type": "array" + } + }, + "required": [ + "all_teams", + "team", + "key_hash", + "trace_ids" + ], + "title": "LensFeedbackSummaryParams", + "type": "object" +} diff --git a/scripts/trace_codegen/schemas/traces-clickhouse/LensFeedbackTargetParams.json b/scripts/trace_codegen/schemas/traces-clickhouse/LensFeedbackTargetParams.json new file mode 100644 index 00000000000..4e5e475ce27 --- /dev/null +++ b/scripts/trace_codegen/schemas/traces-clickhouse/LensFeedbackTargetParams.json @@ -0,0 +1,34 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "additionalProperties": false, + "properties": { + "all_teams": { + "enum": [ + 0, + 1 + ], + "type": "integer" + }, + "key_hash": { + "type": "string" + }, + "team": { + "type": "string" + }, + "trace_id": { + "type": "string" + }, + "trace_ref": { + "type": "string" + } + }, + "required": [ + "all_teams", + "team", + "key_hash", + "trace_id", + "trace_ref" + ], + "title": "LensFeedbackTargetParams", + "type": "object" +} diff --git a/scripts/trace_codegen/schemas/traces-clickhouse/ReadQueryName.json b/scripts/trace_codegen/schemas/traces-clickhouse/ReadQueryName.json index 1a3fb1151f0..33d66e64a35 100644 --- a/scripts/trace_codegen/schemas/traces-clickhouse/ReadQueryName.json +++ b/scripts/trace_codegen/schemas/traces-clickhouse/ReadQueryName.json @@ -6,7 +6,10 @@ "agents", "sample", "content", - "evidence" + "evidence", + "feedback_target", + "feedback", + "feedback_summary" ], "title": "ReadQueryName", "type": "string" diff --git a/tests/unit/proxy/lens/test_feedback_endpoints.py b/tests/unit/proxy/lens/test_feedback_endpoints.py new file mode 100644 index 00000000000..3ee7c79f169 --- /dev/null +++ b/tests/unit/proxy/lens/test_feedback_endpoints.py @@ -0,0 +1,323 @@ +import hashlib +from collections.abc import Mapping, Sequence +from datetime import datetime, timedelta, timezone +from typing import Final + +import pytest +from fastapi import HTTPException +from pydantic import BaseModel, ValidationError + +from litellm.proxy._types import LitellmUserRoles, UserAPIKeyAuth +from litellm.proxy.lens.feedback_endpoints import ( + FeedbackDeletion, + FeedbackSubmission, + FeedbackTarget, + delete_feedback, + feedback_summary, + read_feedback, + submit_feedback, +) +from litellm.proxy.lens.feedback_models import TraceFeedbackRequest +from litellm.proxy.lens.feedback_repository import FEEDBACK_TABLE, ClickHouseFeedbackStore, session_trace_id +from litellm.proxy.lens.models import TraceIdentity +from litellm.rust_bridge.trace.generated.models import ( + FeedbackRow, + FeedbackSummaryRow, + FeedbackTargetRow, + LensFeedbackParams, + LensFeedbackSummaryParams, + LensFeedbackTargetParams, +) +from litellm.rust_bridge.trace.queries import ReadQuery +from litellm.rust_bridge.trace.storage import ClickHouseStorage + +ADMIN: Final = UserAPIKeyAuth(user_role=LitellmUserRoles.PROXY_ADMIN, user_id="admin") +OTHER_ADMIN: Final = UserAPIKeyAuth(user_role=LitellmUserRoles.PROXY_ADMIN, user_id="other") +VIEWER: Final = UserAPIKeyAuth(user_role=LitellmUserRoles.PROXY_ADMIN_VIEW_ONLY, user_id="viewer") +INTERNAL: Final = UserAPIKeyAuth(user_role=LitellmUserRoles.INTERNAL_USER, user_id="dev") +TEAM_APP: Final = UserAPIKeyAuth(user_role=LitellmUserRoles.INTERNAL_USER, team_id="team-a", token="app-key") +SOLO_APP: Final = UserAPIKeyAuth(user_role=LitellmUserRoles.INTERNAL_USER, token="key-solo") +T0: Final = datetime(2026, 3, 1, 12, 0, tzinfo=timezone.utc) + + +def ref(team: str, key: str, trace: str) -> str: + return hashlib.sha256(f"{team}\0{key}\0{trace}".encode()).hexdigest().upper() + + +class FakeClickHouse(ClickHouseStorage): + """Mirrors lens_feedback: ReplacingMergeTree(UpdatedAt, IsDeleted) keyed by team, key, trace, author.""" + + def __init__(self, traces: Mapping[str, tuple[tuple[str, str], ...]]) -> None: + self.traces: Final = traces + self.rows: Final[list[Mapping[str, object]]] = [] # mutable-ok: stands in for the table + + async def insert_rows(self, table: str, rows: Sequence[Mapping[str, object]]) -> None: + assert table == FEEDBACK_TABLE + self.rows.extend(rows) + + def _visible( + self, team: str, key: str, access: LensFeedbackTargetParams | LensFeedbackParams | LensFeedbackSummaryParams + ) -> bool: + if access.all_teams: + return True + return team == access.team and (access.key_hash == "" or key == access.key_hash) + + def _latest(self) -> tuple[Mapping[str, object], ...]: + newest: Final = {} # mutable-ok: emulates FINAL collapse + for row in sorted(self.rows, key=lambda r: str(r["UpdatedAt"])): + newest[(row["TeamId"], row["ApiKeyHash"], row["TraceId"], row["Author"])] = row + return tuple(r for r in newest.values() if r["IsDeleted"] == 0) + + async def query(self, query: ReadQuery[BaseModel, object], parameters: BaseModel) -> tuple[object, ...]: # pyright: ignore[reportIncompatibleMethodOverride] # fake dispatches on the concrete params type + match parameters: + case LensFeedbackTargetParams(): + return tuple( + FeedbackTargetRow(team_id=team, key_hash=key, trace_ref=ref(team, key, parameters.trace_id)) + for team, key in self.traces.get(parameters.trace_id, ()) + if self._visible(team, key, parameters) + and parameters.trace_ref in ("", ref(team, key, parameters.trace_id)) + ) + case LensFeedbackParams(): + return tuple( + FeedbackRow( + trace_id=str(r["TraceId"]), + trace_ref=ref(str(r["TeamId"]), str(r["ApiKeyHash"]), str(r["TraceId"])), + author=str(r["Author"]), + score=int(str(r["Score"])), + comment=str(r["Comment"]), + created_at=str(r["CreatedAt"]), + updated_at=str(r["UpdatedAt"]), + ) + for r in self._latest() + if r["TraceId"] == parameters.trace_id + and ref(str(r["TeamId"]), str(r["ApiKeyHash"]), str(r["TraceId"])) == parameters.trace_ref + and self._visible(str(r["TeamId"]), str(r["ApiKeyHash"]), parameters) + ) + case LensFeedbackSummaryParams(): + live = tuple( + r + for r in self._latest() + if r["TraceId"] in parameters.trace_ids + and self._visible(str(r["TeamId"]), str(r["ApiKeyHash"]), parameters) + ) + keys = sorted({(str(r["TeamId"]), str(r["ApiKeyHash"]), str(r["TraceId"])) for r in live}) + return tuple( + FeedbackSummaryRow( + trace_id=trace, + trace_ref=ref(team, key, trace), + count=len(scores), + average=sum(scores) / len(scores), + lowest=min(scores), + ) + for team, key, trace in keys + for scores in [ + [ + int(str(r["Score"])) + for r in live + if (r["TeamId"], r["ApiKeyHash"], r["TraceId"]) == (team, key, trace) + ] + ] + ) + case _: + raise AssertionError(f"unexpected query {query.name}") + + +def store(**traces: tuple[tuple[str, str], ...]) -> ClickHouseFeedbackStore: + return ClickHouseFeedbackStore(FakeClickHouse(traces or {"t1": (("team-a", "key-a"),)})) + + +def submission(score: int, comment: str = "", **fields: str) -> FeedbackSubmission: + target: Final = {} if "trace_id" in fields or "session_id" in fields else {"trace_id": "t1"} + return FeedbackSubmission.model_validate({"score": score, "comment": comment, **target, **fields}) + + +@pytest.mark.asyncio +async def test_resubmitting_replaces_the_authors_feedback_and_keeps_other_authors() -> None: + feedback: Final = store() + first: Final = await submit_feedback(submission(3, "wrong file"), ADMIN, feedback, T0) + await submit_feedback(submission(9, "great"), OTHER_ADMIN, feedback, T0) + second: Final = await submit_feedback( + submission(7, "fixed after retry"), ADMIN, feedback, T0 + timedelta(minutes=5) + ) + + listed: Final = await read_feedback(FeedbackTarget(trace_id="t1"), VIEWER, feedback) + + assert (first.created_at, second.created_at, second.updated_at) == (T0, T0, T0 + timedelta(minutes=5)) + assert {f.author: f.created_at for f in listed.feedback}["admin"] == T0 + assert listed.trace_ref == ref("team-a", "key-a", "t1") + assert {(f.author, f.score, f.comment) for f in listed.feedback} == { + ("admin", 7, "fixed after retry"), + ("other", 9, "great"), + } + + +@pytest.mark.asyncio +async def test_session_id_resolves_to_the_trace_lens_derives_at_ingest() -> None: + session_trace: Final = session_trace_id("session-one") + feedback: Final = store(**{session_trace: (("team-a", "key-a"),)}) + + saved: Final = await submit_feedback(submission(4, session_id="session-one"), ADMIN, feedback, T0) + listed: Final = await read_feedback(FeedbackTarget(session_id="session-one"), ADMIN, feedback) + + assert saved.trace_id == session_trace + assert listed.trace_ref == ref("team-a", "key-a", session_trace) + assert [f.score for f in listed.feedback] == [4] + + +def test_session_trace_id_matches_the_rust_ingest_hash() -> None: + # Pinned in litellm-rust/crates/traces/tests/otlp.rs (session_capture_joins_native_logs_...). + assert session_trace_id("session-one") == "5fddf060372c8501dca4f331b9da882b" + + +@pytest.mark.asyncio +async def test_unknown_traces_are_not_found_and_write_nothing() -> None: + feedback: Final = store() + + with pytest.raises(HTTPException) as write: + await submit_feedback(submission(5, trace_id="missing"), ADMIN, feedback, T0) + with pytest.raises(HTTPException) as read: + await read_feedback(FeedbackTarget(trace_id="missing"), ADMIN, feedback) + + assert (write.value.status_code, read.value.status_code) == (404, 404) + assert isinstance(feedback.storage, FakeClickHouse) and feedback.storage.rows == [] + + +@pytest.mark.parametrize("score", (-1, 11)) +def test_scores_outside_zero_to_ten_are_rejected(score: int) -> None: + with pytest.raises(ValidationError): + submission(score) + + +@pytest.mark.parametrize("target", ({}, {"trace_id": "t1", "session_id": "s1"})) +def test_target_needs_exactly_one_of_trace_or_session(target: dict[str, str]) -> None: + with pytest.raises(ValidationError): + FeedbackTarget.model_validate(target) + + +@pytest.mark.asyncio +async def test_viewers_can_read_but_not_write_and_non_admins_cannot_read_in_lens() -> None: + feedback: Final = store() + await read_feedback(FeedbackTarget(trace_id="t1"), VIEWER, feedback) + + with pytest.raises(HTTPException) as write: + await submit_feedback(submission(5), VIEWER, feedback, T0) + with pytest.raises(HTTPException) as read: + await read_feedback(FeedbackTarget(trace_id="t1"), INTERNAL, feedback) + + assert (write.value.status_code, read.value.status_code) == (403, 403) + + +@pytest.mark.asyncio +async def test_tenant_comes_from_the_trace_and_author_defaults_to_the_caller() -> None: + feedback: Final = store() + saved: Final = await submit_feedback(submission(5), OTHER_ADMIN, feedback, T0) + + assert saved.author == "other" + assert isinstance(feedback.storage, FakeClickHouse) + assert {(r["TeamId"], r["ApiKeyHash"], r["Author"]) for r in feedback.storage.rows} == { + ("team-a", "key-a", "other") + } + with pytest.raises(ValidationError): + FeedbackSubmission.model_validate({"trace_id": "t1", "score": 5, "team_id": "someone-else"}) + + +@pytest.mark.asyncio +async def test_delete_hides_only_the_callers_feedback() -> None: + feedback: Final = store() + await submit_feedback(submission(2), ADMIN, feedback, T0) + await submit_feedback(submission(8), OTHER_ADMIN, feedback, T0) + + await delete_feedback(FeedbackDeletion(trace_id="t1"), ADMIN, feedback, T0 + timedelta(minutes=9)) + with pytest.raises(HTTPException) as missing: + await delete_feedback(FeedbackDeletion(trace_id="t1"), ADMIN, feedback, T0 + timedelta(minutes=9)) + + remaining: Final = await read_feedback(FeedbackTarget(trace_id="t1"), ADMIN, feedback) + assert missing.value.status_code == 404 + assert [f.author for f in remaining.feedback] == ["other"] + + +@pytest.mark.asyncio +async def test_summary_flags_rated_traces_and_leaves_unrated_ones_empty() -> None: + feedback: Final = store(t1=(("team-a", "key-a"),), t2=(("team-a", "key-a"),)) + await submit_feedback(submission(2), ADMIN, feedback, T0) + await submit_feedback(submission(8), OTHER_ADMIN, feedback, T0) + rated: Final = TraceIdentity(trace_id="t1", trace_ref=ref("team-a", "key-a", "t1")) + unrated: Final = TraceIdentity(trace_id="t2", trace_ref=ref("team-a", "key-a", "t2")) + + summaries: Final = await feedback_summary(TraceFeedbackRequest(traces=(rated, unrated)), VIEWER, feedback) + + assert {s.trace_id: (s.count, s.average, s.lowest) for s in summaries} == { + "t1": (2, 5.0, 2), + "t2": (0, None, None), + } + + +@pytest.mark.asyncio +async def test_a_trace_id_shared_by_two_keys_needs_its_trace_ref() -> None: + feedback: Final = store(t1=(("team-a", "key-a"), ("team-a", "key-b"))) + + with pytest.raises(HTTPException) as ambiguous: + await submit_feedback(submission(5), ADMIN, feedback, T0) + saved: Final = await submit_feedback( + submission(5, trace_id="t1", trace_ref=ref("team-a", "key-b", "t1")), ADMIN, feedback, T0 + ) + + assert ambiguous.value.status_code == 404 + assert saved.trace_ref == ref("team-a", "key-b", "t1") + + +@pytest.mark.asyncio +async def test_summary_without_trace_ref_reports_the_rated_trace_with_its_resolved_ref() -> None: + feedback: Final = store() + await submit_feedback(submission(6), ADMIN, feedback, T0) + + summaries: Final = await feedback_summary( + TraceFeedbackRequest(traces=(TraceIdentity(trace_id="t1"),)), VIEWER, feedback + ) + + assert [(s.trace_ref, s.count, s.lowest) for s in summaries] == [(ref("team-a", "key-a", "t1"), 1, 6)] + + +@pytest.mark.asyncio +async def test_an_app_key_records_its_end_users_feedback_on_its_own_teams_trace() -> None: + feedback: Final = store() + await submit_feedback(submission(2, "It ignored my file", user="customer-1"), TEAM_APP, feedback, T0) + await submit_feedback(submission(9, "Perfect", user="customer-2"), TEAM_APP, feedback, T0) + await submit_feedback( + submission(4, "Better after retry", user="customer-1"), TEAM_APP, feedback, T0 + timedelta(minutes=1) + ) + + listed: Final = await read_feedback(FeedbackTarget(trace_id="t1"), ADMIN, feedback) + + assert {(f.author, f.score, f.comment) for f in listed.feedback} == { + ("customer-1", 4, "Better after retry"), + ("customer-2", 9, "Perfect"), + } + + +@pytest.mark.asyncio +async def test_an_app_key_cannot_write_feedback_on_another_tenants_trace() -> None: + feedback: Final = store(t1=(("team-b", "key-b"),), solo=(("", "key-solo"),), other=(("", "key-other"),)) + + with pytest.raises(HTTPException) as other_team: + await submit_feedback(submission(5, user="customer-1"), TEAM_APP, feedback, T0) + with pytest.raises(HTTPException) as other_key: + await submit_feedback(submission(5, trace_id="other", user="customer-1"), SOLO_APP, feedback, T0) + saved: Final = await submit_feedback(submission(5, trace_id="solo", user="customer-1"), SOLO_APP, feedback, T0) + + assert (other_team.value.status_code, other_key.value.status_code) == (404, 404) + assert saved.author == "customer-1" + + +@pytest.mark.asyncio +async def test_an_app_can_remove_one_end_users_feedback() -> None: + feedback: Final = store() + await submit_feedback(submission(2, user="customer-1"), TEAM_APP, feedback, T0) + await submit_feedback(submission(9, user="customer-2"), TEAM_APP, feedback, T0) + + await delete_feedback( + FeedbackDeletion(trace_id="t1", user="customer-1"), TEAM_APP, feedback, T0 + timedelta(minutes=1) + ) + + listed: Final = await read_feedback(FeedbackTarget(trace_id="t1"), ADMIN, feedback) + assert [f.author for f in listed.feedback] == ["customer-2"] diff --git a/tests/unit/tracing/test_remote.py b/tests/unit/tracing/test_remote.py index 2572c2a2044..92922dbd3ca 100644 --- a/tests/unit/tracing/test_remote.py +++ b/tests/unit/tracing/test_remote.py @@ -168,7 +168,7 @@ async def test_gateway_cannot_relay_otlp_or_write_arbitrary_tables() -> None: await store.ensure_schema() with pytest.raises(RuntimeError, match="directly"): await store.ingest(b"{}", "application/json", {}) - with pytest.raises(ValueError, match="request records"): + with pytest.raises(ValueError, match="request records and feedback"): await store.insert_rows("otel_traces", ()) @@ -185,3 +185,18 @@ async def test_request_records_use_the_internal_service_endpoint() -> None: request: Final = requests.get_nowait() assert request.url.path == "/internal/spend" assert json.loads(request.content) == [{"request_id": "r"}] + + +@pytest.mark.asyncio +async def test_feedback_rows_use_the_internal_feedback_endpoint() -> None: + requests: Final = asyncio.Queue[httpx.Request]() + + def accept(request: httpx.Request) -> httpx.Response: + requests.put_nowait(request) + return httpx.Response(204) + + async with httpx.AsyncClient(base_url="http://lens", transport=httpx.MockTransport(accept)) as client: + await RemoteTraceStore(client).insert_rows("lens_feedback", ({"TraceId": "t", "Score": 2},)) + request: Final = requests.get_nowait() + assert request.url.path == "/internal/feedback" + assert json.loads(request.content) == [{"TraceId": "t", "Score": 2}] diff --git a/ui/litellm-dashboard/src/components/lens/LensSetup.integration.test.tsx b/ui/litellm-dashboard/src/components/lens/LensSetup.integration.test.tsx index 1a81235b361..e3b79d04496 100644 --- a/ui/litellm-dashboard/src/components/lens/LensSetup.integration.test.tsx +++ b/ui/litellm-dashboard/src/components/lens/LensSetup.integration.test.tsx @@ -40,6 +40,7 @@ function serve({ enabled = false, traces = false, requests = false, connected = return Response.json({ agents: traces ? rollUpAgents([data.runs[0].trace.summary]) : [] }); if (path === "/lens/activity/available") return Response.json({ traces, requests }); if (path === "/lens/traces/findings") return Response.json([]); + if (path === "/lens/feedback/summary") return Response.json([]); if (path === "/lens" && method === "POST") { const saved = { ...data.lenses[0], settings: { ...data.lenses[0].settings, ...(body as object) } }; list.mockResolvedValue({ lenses: [saved], workers: [worker()], tracing_enabled: true }); diff --git a/ui/litellm-dashboard/src/components/lens/LensWorkspace.integration.test.tsx b/ui/litellm-dashboard/src/components/lens/LensWorkspace.integration.test.tsx index f3891376882..41bf0fc6730 100644 --- a/ui/litellm-dashboard/src/components/lens/LensWorkspace.integration.test.tsx +++ b/ui/litellm-dashboard/src/components/lens/LensWorkspace.integration.test.tsx @@ -178,6 +178,7 @@ describe("Lens interactive demo", () => { if (path.endsWith("/runs")) return Response.json(saved.jobs); if (path === "/v1/traces") return Response.json({ data: data.runs.map((run) => run.trace.summary) }); if (path === "/lens/traces/findings") return Response.json([]); + if (path === "/lens/feedback/summary") return Response.json([]); return Response.json({ data: [], traces: true, requests: false }); }); renderWithProviders(, { @@ -358,6 +359,7 @@ describe("Lens interactive demo", () => { if (path.startsWith("/lens/preview")) return Response.json({ eligible: 0, selected: 0, executions: [] }); if (path === "/v1/traces") return Response.json({ data: [createLensDemoData().runs[0].trace.summary] }); if (path === "/lens/traces/findings") return Response.json([]); + if (path === "/lens/feedback/summary") return Response.json([]); return Response.json({ data: [], traces: true, requests: false }); }); renderWithProviders(, { diff --git a/ui/litellm-dashboard/src/components/lens/data/demo/createLensDemo.ts b/ui/litellm-dashboard/src/components/lens/data/demo/createLensDemo.ts index 48c993eb6ed..1cc2d672f81 100644 --- a/ui/litellm-dashboard/src/components/lens/data/demo/createLensDemo.ts +++ b/ui/litellm-dashboard/src/components/lens/data/demo/createLensDemo.ts @@ -1,6 +1,7 @@ import { ApiError } from "@/lib/http/client"; import type { TracesApi } from "@/components/lens/traces/api"; import { rollUpAgents } from "@/components/lens/agents/agentRollup"; +import type { Feedback, TraceSummary } from "@/components/lens/traces/types"; import type { LensServices } from "../LensServices"; import type { LensApi } from "../service"; import { demoDatasetsApi } from "./demoDatasets"; @@ -50,6 +51,22 @@ function demoLensApi(data: LensDemoData): LensApi { }; } +const DEMO_FEEDBACK = [ + { score: 3, comment: "It edited the wrong file and I had to ask twice.", author: "customer-1042" }, + { score: 9, comment: "Exactly what I asked for.", author: "customer-2210" }, +] as const; + +function demoFeedback(runs: LensDemoData["runs"]): Feedback[] { + return DEMO_FEEDBACK.flatMap((entry, index) => { + const summary: TraceSummary | undefined = runs[index]?.trace.summary; + if (!summary) return []; + const at = summary.start_time; + return [ + { ...entry, trace_id: summary.trace_id, trace_ref: summary.trace_ref ?? "", created_at: at, updated_at: at }, + ]; + }); +} + const summariesIn = (data: LensDemoData, startMs: number, endMs: number) => data.runs .map((item) => item.trace.summary) @@ -57,6 +74,8 @@ const summariesIn = (data: LensDemoData, startMs: number, endMs: number) => function demoTracesApi(data: LensDemoData): TracesApi { const run = (traceId: string) => data.runs.find(({ trace }) => trace.summary.trace_id === traceId); + const feedback = demoFeedback(data.runs); + const feedbackFor = (traceId: string) => feedback.filter((entry) => entry.trace_id === traceId); return { live: false, handoff: (traceId, spanId) => { @@ -96,6 +115,21 @@ function demoTracesApi(data: LensDemoData): TracesApi { }), signals: async (traces) => traces.map((trace) => ({ ...trace, status: "unclassified" as const, flags: [], model: "", classified_at: null })), + feedbackSummary: async (traces) => + traces.map((trace) => { + const scores = feedbackFor(trace.trace_id).map((entry) => entry.score); + return { + trace_id: trace.trace_id, + trace_ref: trace.trace_ref ?? "", + count: scores.length, + average: scores.length ? scores.reduce((sum, score) => sum + score, 0) / scores.length : null, + lowest: scores.length ? Math.min(...scores) : null, + }; + }), + feedback: async (traceId) => { + const summary = (await found(run(traceId))).trace.summary; + return { trace_id: traceId, trace_ref: summary.trace_ref ?? "", feedback: feedbackFor(traceId) }; + }, anyRecorded: async () => data.runs.length > 0, trace: (traceId) => found(run(traceId)?.trace), span: (traceId, spanId) => found(run(traceId)?.details.find((span) => span.span_id === spanId)), diff --git a/ui/litellm-dashboard/src/components/lens/traces/api.ts b/ui/litellm-dashboard/src/components/lens/traces/api.ts index a4846311028..422d2f15773 100644 --- a/ui/litellm-dashboard/src/components/lens/traces/api.ts +++ b/ui/litellm-dashboard/src/components/lens/traces/api.ts @@ -21,6 +21,10 @@ import type { TraceFindingCount, TraceFindingsRequest, TraceSignals, + FeedbackQuery, + TraceFeedback, + TraceFeedbackRequest, + TraceFeedbackSummary, TraceAgentList, TraceAgentsQuery, } from "./types"; @@ -44,6 +48,8 @@ export interface TracesApi { agents(window: TraceWindow): Promise; findings(traces: TraceFindingsRequest["traces"]): Promise; signals(traces: TraceFindingsRequest["traces"]): Promise; + feedbackSummary(traces: TraceFeedbackRequest["traces"]): Promise; + feedback(traceId: string, traceRef?: string): Promise; anyRecorded(): Promise; trace(traceId: string, traceRef?: string, cursor?: string | null): Promise; span(traceId: string, spanId: string, traceRef?: string): Promise; @@ -100,6 +106,16 @@ export function liveTracesApi(accessToken: string): TracesApi { accessToken, body: { traces } satisfies TraceFindingsRequest, }), + feedbackSummary: (traces) => + apiClient.post("/lens/feedback/summary", { + accessToken, + body: { traces } satisfies TraceFeedbackRequest, + }), + feedback: (traceId, traceRef) => + apiClient.get("/lens/feedback", { + accessToken, + query: { trace_id: traceId, trace_ref: traceRef ?? "" } satisfies FeedbackQuery, + }), anyRecorded: async () => { const page = await apiClient.get("/v1/traces", { accessToken, diff --git a/ui/litellm-dashboard/src/components/lens/traces/detail/feedback/FeedbackPanel.integration.test.tsx b/ui/litellm-dashboard/src/components/lens/traces/detail/feedback/FeedbackPanel.integration.test.tsx new file mode 100644 index 00000000000..645fe30de7f --- /dev/null +++ b/ui/litellm-dashboard/src/components/lens/traces/detail/feedback/FeedbackPanel.integration.test.tsx @@ -0,0 +1,88 @@ +import { screen, within } from "@testing-library/react"; +import { beforeEach, describe, expect, it, vi } from "vitest"; + +import { renderWithProviders, testQueryClient } from "../../../../../../tests/test-utils"; +import { type TracesApi, TracesApiContext } from "../../api"; +import type { Feedback } from "../../types"; +import { FeedbackPanel } from "./FeedbackPanel"; + +const summary = { trace_id: "trace-1", trace_ref: "REF1" }; + +const entry = (author: string, score: number, comment: string, updated_at = "2026-03-01T12:00:00Z"): Feedback => ({ + ...summary, + author, + score, + comment, + created_at: updated_at, + updated_at, +}); + +const renderPanel = (feedback: Feedback[]) => { + const api = { live: true, feedback: vi.fn(async () => ({ ...summary, feedback })) }; + renderWithProviders( + + + , + ); + return api; +}; + +describe("FeedbackPanel", () => { + beforeEach(() => testQueryClient.clear()); + + it("shows what the end user said and their score, flagged when it is low", async () => { + const api = renderPanel([entry("customer-1042", 2, "It edited the wrong file and I had to ask twice")]); + + const panel = await screen.findByRole("region", { name: "User feedback" }); + expect(panel).toHaveAttribute("data-low", "true"); + const row = within(panel).getByTestId("feedback-entry"); + expect(row).toHaveTextContent("2/10"); + expect(row).toHaveTextContent("“It edited the wrong file and I had to ask twice”"); + expect(row).toHaveTextContent("customer-1042"); + expect(api.feedback).toHaveBeenCalledWith("trace-1", "REF1"); + }); + + it("lists every user's feedback newest first with the average when several users rated the run", async () => { + renderPanel([ + entry("customer-1", 9, "Perfect", "2026-03-01T12:00:00Z"), + entry("customer-2", 3, "Too slow", "2026-03-01T12:05:00Z"), + ]); + + const panel = await screen.findByRole("region", { name: "User feedback" }); + expect( + within(panel) + .getAllByTestId("feedback-entry") + .map((row) => row.textContent), + ).toEqual([expect.stringContaining("customer-2"), expect.stringContaining("customer-1")]); + expect(panel).toHaveTextContent("6/10 avg from 2 users"); + expect(panel).toHaveAttribute("data-low", "true"); + }); + + it("does not flag a run every user scored well and offers no way to edit feedback", async () => { + renderPanel([entry("customer-1", 8, "")]); + + const panel = await screen.findByRole("region", { name: "User feedback" }); + expect(panel).not.toHaveAttribute("data-low"); + expect(within(panel).getByText("No comment")).toBeInTheDocument(); + expect(within(panel).queryByRole("button")).not.toBeInTheDocument(); + expect(within(panel).queryByRole("textbox")).not.toBeInTheDocument(); + }); + + it("stays out of the way when feedback cannot be loaded so the run still reads cleanly", async () => { + const api = { live: true, feedback: vi.fn(() => Promise.reject(new Error("ClickHouse down"))) }; + renderWithProviders( + + + , + ); + await vi.waitFor(() => expect(api.feedback).toHaveBeenCalled()); + expect(screen.queryByRole("alert")).not.toBeInTheDocument(); + expect(screen.queryByRole("region", { name: "User feedback" })).not.toBeInTheDocument(); + }); + + it("renders nothing when no end user rated the run", async () => { + const api = renderPanel([]); + await vi.waitFor(() => expect(api.feedback).toHaveBeenCalled()); + expect(screen.queryByRole("region", { name: "User feedback" })).not.toBeInTheDocument(); + }); +}); diff --git a/ui/litellm-dashboard/src/components/lens/traces/detail/feedback/FeedbackPanel.tsx b/ui/litellm-dashboard/src/components/lens/traces/detail/feedback/FeedbackPanel.tsx new file mode 100644 index 00000000000..f06591acce8 --- /dev/null +++ b/ui/litellm-dashboard/src/components/lens/traces/detail/feedback/FeedbackPanel.tsx @@ -0,0 +1,87 @@ +"use client"; + +import { useQuery } from "@tanstack/react-query"; +import { MessageSquareQuote, Star } from "lucide-react"; + +import { cn } from "@/lib/cva.config"; +import { formatActivityTimestamp } from "@/utils/activityTimestamp"; + +import { useTracesApi } from "../../api"; +import { formatScore } from "../../list/AgentTracesTable"; +import { LOW_SCORE } from "../../list/useTraceFeedback"; +import type { Feedback, TraceSummary } from "../../types"; +import { feedbackView } from "./feedback"; + +interface FeedbackPanelProps { + summary: Pick; + accessToken: string; +} + +const isLow = (score: number) => score <= LOW_SCORE; + +function Entry({ entry }: { entry: Feedback }) { + const low = isLow(entry.score); + return ( +
  • + + {entry.score} + /10 + +
    + {entry.comment ? ( +

    “{entry.comment}”

    + ) : ( +

    No comment

    + )} + + {entry.author} · {formatActivityTimestamp(entry.updated_at)} + +
    +
  • + ); +} + +/** End-user feedback on this run, shown first so a developer reads what the user said before the steps. */ +export function FeedbackPanel({ summary, accessToken }: FeedbackPanelProps) { + const api = useTracesApi(accessToken); + const traceRef = summary.trace_ref ?? ""; + const feedback = useQuery({ + queryKey: ["traceFeedbackDetail", accessToken, summary.trace_id, traceRef], + queryFn: () => api.feedback(summary.trace_id, traceRef), + retry: false, + }); + const view = feedback.data ? feedbackView(feedback.data.feedback) : null; + if (!view) return null; + const low = isLow(view.lowest); + return ( +
    +
    + + User feedback + {view.entries.length > 1 && ( + + + {formatScore(view.average)}/10 avg from {view.entries.length} users + + )} +
    +
      + {view.entries.map((entry) => ( + + ))} +
    +
    + ); +} diff --git a/ui/litellm-dashboard/src/components/lens/traces/detail/feedback/feedback.test.ts b/ui/litellm-dashboard/src/components/lens/traces/detail/feedback/feedback.test.ts new file mode 100644 index 00000000000..4af6cf5862d --- /dev/null +++ b/ui/litellm-dashboard/src/components/lens/traces/detail/feedback/feedback.test.ts @@ -0,0 +1,31 @@ +import { describe, expect, it } from "vitest"; + +import type { Feedback } from "../../types"; +import { feedbackView } from "./feedback"; + +const entry = (author: string, score: number, updated_at: string): Feedback => ({ + author, + score, + comment: "", + trace_id: "t1", + trace_ref: "", + created_at: updated_at, + updated_at, +}); + +describe("feedbackView", () => { + it("lists every end user's feedback newest first with the average and lowest score", () => { + const view = feedbackView([ + entry("alice", 2, "2026-03-01T12:00:00Z"), + entry("bob", 9, "2026-03-01T12:05:00Z"), + entry("carol", 7, "2026-03-01T12:03:00Z"), + ]); + expect(view?.entries.map((item) => item.author)).toEqual(["bob", "carol", "alice"]); + expect(view?.average).toBe(6); + expect(view?.lowest).toBe(2); + }); + + it("has nothing to show when nobody rated the run", () => { + expect(feedbackView([])).toBeNull(); + }); +}); diff --git a/ui/litellm-dashboard/src/components/lens/traces/detail/feedback/feedback.ts b/ui/litellm-dashboard/src/components/lens/traces/detail/feedback/feedback.ts new file mode 100644 index 00000000000..6a572997657 --- /dev/null +++ b/ui/litellm-dashboard/src/components/lens/traces/detail/feedback/feedback.ts @@ -0,0 +1,18 @@ +import type { Feedback } from "../../types"; + +export interface FeedbackView { + readonly entries: readonly Feedback[]; + readonly average: number; + readonly lowest: number; +} + +/** End-user feedback on one run, newest first, or null when nobody has rated it. */ +export function feedbackView(feedback: readonly Feedback[]): FeedbackView | null { + if (feedback.length === 0) return null; + const scores = feedback.map((entry) => entry.score); + return { + entries: [...feedback].sort((a, b) => Date.parse(b.updated_at) - Date.parse(a.updated_at)), + average: scores.reduce((sum, score) => sum + score, 0) / scores.length, + lowest: Math.min(...scores), + }; +} diff --git a/ui/litellm-dashboard/src/components/lens/traces/detail/run/RunView.tsx b/ui/litellm-dashboard/src/components/lens/traces/detail/run/RunView.tsx index 43b3c00048e..b325ef2fbd9 100644 --- a/ui/litellm-dashboard/src/components/lens/traces/detail/run/RunView.tsx +++ b/ui/litellm-dashboard/src/components/lens/traces/detail/run/RunView.tsx @@ -17,6 +17,7 @@ import type { Trace } from "../../types"; import { PagingBanner } from "./PagingBanner"; import { RunBody } from "./RunBody"; import { RunHeader } from "./RunHeader"; +import { FeedbackPanel } from "../feedback/FeedbackPanel"; import { useTraceSignalFlags } from "../../list/useTraceSignals"; interface RunViewProps { @@ -175,6 +176,7 @@ function LoadedRun({ onLiveChange={toggleLive} signals={signals} /> + {traceQuery.isRefetchError && (
    Could not refresh this run. Previously received steps are still shown. diff --git a/ui/litellm-dashboard/src/components/lens/traces/list/AgentTracesSection.integration.test.tsx b/ui/litellm-dashboard/src/components/lens/traces/list/AgentTracesSection.integration.test.tsx index eee113f6f0c..f0ba76367b3 100644 --- a/ui/litellm-dashboard/src/components/lens/traces/list/AgentTracesSection.integration.test.tsx +++ b/ui/litellm-dashboard/src/components/lens/traces/list/AgentTracesSection.integration.test.tsx @@ -12,7 +12,7 @@ import { filterRuns } from "./runSearch/runQuery"; import type { RelativeRangeState } from "@/components/shared/timeRange/useRelativeRange"; import { AgentTracesSection } from "./AgentTracesSection"; -import type { TraceFindingCount, TracePage, TraceSummary } from "../types"; +import type { TraceFeedbackSummary, TraceFindingCount, TracePage, TraceSummary } from "../types"; vi.mock("../../../networking", () => ({ apiClient: { get: vi.fn(), post: vi.fn() }, @@ -71,6 +71,26 @@ const renderWindowed = (timeControls?: RelativeRangeState) => , ); +type TraceKey = { trace_id: string; trace_ref?: string }; +const unrated = (traces: TraceKey[]): TraceFeedbackSummary[] => + traces.map((trace) => ({ + trace_id: trace.trace_id, + trace_ref: trace.trace_ref ?? "", + count: 0, + average: null, + lowest: null, + })); + +/** Routes the shared POST mock: findings get `findings`, the feedback summary gets `feedback`. */ +const stubPost = ( + findings: (traces: TraceKey[]) => Promise, + feedback: (traces: TraceKey[]) => Promise = async (traces) => unrated(traces), +) => + vi.mocked(apiClient.post).mockImplementation(async (path, options) => { + const traces = (options?.body as { traces: TraceKey[] }).traces; + return path === "/lens/feedback/summary" ? feedback(traces) : findings(traces); + }); + const bucketRunCounts = () => screen.getAllByTestId("timeline-bucket").map((bucket) => Number(bucket.getAttribute("data-total"))); @@ -96,10 +116,7 @@ describe("AgentTracesSection", () => { testQueryClient.clear(); vi.mocked(agentTraceListCall).mockReset(); serveService(true); - vi.mocked(apiClient.post).mockImplementation(async (_path, options) => { - const body = options?.body as { traces: { trace_id: string; trace_ref?: string }[] }; - return body.traces.map((trace) => ({ ...trace, finding_count: null })); - }); + stubPost(async (traces) => traces.map((trace) => ({ ...trace, finding_count: null }))); }); it("loads the next page only once the list scrolls near its end, then stops at the last page", async () => { @@ -316,12 +333,13 @@ describe("AgentTracesSection", () => { expect(screen.getByRole("columnheader", { name: "Findings" })).toBeInTheDocument(); expect(screen.queryByRole("columnheader", { name: "Failed" })).not.toBeInTheDocument(); expect(screen.getByRole("columnheader", { name: "Cost" })).toBeInTheDocument(); - expect(within(failed).getByText("—")).toBeInTheDocument(); + const costColumn = screen.getAllByRole("columnheader").findIndex((header) => header.textContent === "Cost"); + expect(within(failed).getAllByRole("cell")[costColumn]).toHaveTextContent("—"); }); it("distinguishes uninvestigated traces, completed clean investigations, and findings", async () => { vi.mocked(agentTraceListCall).mockResolvedValue({ data: runs, next_cursor: null }); - vi.mocked(apiClient.post).mockResolvedValue( + stubPost(async () => runs.map((run, index) => ({ trace_id: run.trace_id, trace_ref: run.trace_ref ?? "", @@ -347,7 +365,7 @@ describe("AgentTracesSection", () => { it("does not present a failed findings lookup as an uninvestigated or clean trace", async () => { vi.mocked(agentTraceListCall).mockResolvedValue({ data: runs.slice(0, 1), next_cursor: null }); - vi.mocked(apiClient.post).mockRejectedValue(new ApiError("Unavailable", 503, {})); + stubPost(() => Promise.reject(new ApiError("Unavailable", 503, {}))); renderSection(); expect(await screen.findByTitle("Could not load findings")).toHaveTextContent("Unavailable"); expect(screen.queryByTitle("No conclusive investigation for this trace")).not.toBeInTheDocument(); @@ -357,7 +375,7 @@ describe("AgentTracesSection", () => { const user = userEvent.setup(); const pending = Promise.withResolvers(); vi.mocked(agentTraceListCall).mockResolvedValue({ data: runs.slice(0, 1), next_cursor: null }); - vi.mocked(apiClient.post).mockReturnValue(pending.promise); + stubPost(() => pending.promise); renderSection(); expect(await screen.findByTestId("agent-trace-row")).toBeVisible(); await user.click(screen.getByRole("button", { name: "Columns" })); @@ -375,7 +393,7 @@ describe("AgentTracesSection", () => { renderSection(false); expect(await screen.findAllByTestId("agent-trace-row")).toHaveLength(runs.length); expect(screen.queryByRole("columnheader", { name: "Findings" })).not.toBeInTheDocument(); - expect(apiClient.post).not.toHaveBeenCalled(); + expect(apiClient.post).not.toHaveBeenCalledWith("/lens/traces/findings", expect.anything()); }); it("shows the spend returned for a run", async () => { @@ -521,6 +539,39 @@ describe("AgentTracesSection", () => { expect(lastUrl(onUrlUpdate).has("fullscreen")).toBe(false); }); + it("shows each run's end-user score and flags the whole row when a user scored it low", async () => { + vi.mocked(agentTraceListCall).mockResolvedValue({ data: runs, next_cursor: null }); + const [low, good] = runs; + const score = (trace: TraceKey, average: number, lowest: number) => ({ + trace_id: trace.trace_id, + trace_ref: trace.trace_ref ?? "", + count: 1, + average, + lowest, + }); + stubPost( + async (traces) => traces.map((trace) => ({ ...trace, finding_count: null })), + async (traces) => + traces.map((trace) => { + if (trace.trace_id === low.trace_id) return score(trace, 2, 2); + if (trace.trace_id === good.trace_id) return score(trace, 9, 9); + return unrated([trace])[0]; + }), + ); + renderSection(); + + await waitFor(() => expect(screen.getAllByTestId("feedback-score")).toHaveLength(2)); + const rows = screen.getAllByTestId("agent-trace-row"); + expect(rows).toHaveLength(runs.length); + const lowRow = rows.find((row) => within(row).queryByText("2/10")); + const goodRow = rows.find((row) => within(row).queryByText("9/10")); + expect(lowRow).toHaveAttribute("data-low-feedback", "true"); + expect(lowRow).toHaveAttribute("data-flagged", "true"); + expect(goodRow).not.toHaveAttribute("data-flagged"); + expect(rows.filter((row) => row.hasAttribute("data-low-feedback"))).toHaveLength(1); + expect(screen.queryByRole("combobox", { name: "Filter traces by feedback" })).not.toBeInTheDocument(); + }); + it("narrows the list to the zoom window named in the URL and clears it on request", async () => { vi.mocked(agentTraceListCall).mockResolvedValue(traceList as TracePage); const startMs = Date.parse(runs[0].start_time); diff --git a/ui/litellm-dashboard/src/components/lens/traces/list/AgentTracesSection.tsx b/ui/litellm-dashboard/src/components/lens/traces/list/AgentTracesSection.tsx index 97eae300106..1a83ebb9f04 100644 --- a/ui/litellm-dashboard/src/components/lens/traces/list/AgentTracesSection.tsx +++ b/ui/litellm-dashboard/src/components/lens/traces/list/AgentTracesSection.tsx @@ -13,6 +13,7 @@ import { Button } from "@/components/ui/button"; import { AgentTracesTable } from "./AgentTracesTable"; import { useTraceFindings } from "./useTraceFindings"; +import { useTraceFeedback } from "./useTraceFeedback"; import { useTraceSignals } from "./useTraceSignals"; import { useOptionalLensApi } from "../../data/LensServices"; import { lensKeys } from "../../data/queries"; @@ -125,6 +126,7 @@ export function AgentTracesSection({ [traces.traces, query, agent, status], ); const runs = useMemo(() => (zoom ? filterByWindow(filtered, zoom) : filtered), [filtered, zoom]); + const feedback = useTraceFeedback(accessToken, runs, isActive); const runRefs = useMemo(() => runs.map(traceRefOf), [runs]); const findings = useTraceFindings(accessToken, runs, isActive, canViewFindings); const signalSetup = useSignalSetup(isActive && canViewFindings !== false); @@ -231,6 +233,7 @@ export function AgentTracesSection({ ( { expect(screen.getAllByTestId("agent-trace-row").some((row) => row.hasAttribute("data-flagged"))).toBe(false); }); }); + +describe("AgentTracesTable feedback column", () => { + const template = (traceList as TracePage).data[0] as TraceSummary; + const renderFeedback = (state: TraceFeedbackState) => + renderWithProviders( + inList( + , + ), + ); + + it("shows the score and flags both the score and the row when a user scored the run low", () => { + renderFeedback({ status: "ready", summary: { count: 2, average: 5.5, lowest: LOW_SCORE } }); + const score = screen.getByTestId("feedback-score"); + expect(score).toHaveTextContent(/^5\.5\/10$/); + expect(score).toHaveAttribute("data-low", "true"); + expect(score).toHaveAttribute("title", `User feedback: 2 ratings, lowest ${LOW_SCORE}/10`); + const row = screen.getByTestId("agent-trace-row"); + expect(row).toHaveAttribute("data-flagged", "true"); + expect(row).toHaveAttribute("data-low-feedback", "true"); + }); + + it("shows a single user's whole-number score plainly and leaves a well scored row unflagged", () => { + renderFeedback({ status: "ready", summary: { count: 1, average: 9, lowest: LOW_SCORE + 1 } }); + expect(screen.getByTestId("feedback-score")).toHaveTextContent(/^9\/10$/); + expect(screen.getByTestId("feedback-score")).not.toHaveAttribute("data-low"); + expect(screen.getByTestId("agent-trace-row")).not.toHaveAttribute("data-flagged"); + }); + + it("shows a dash for a run nobody has rated", () => { + renderFeedback({ status: "ready", summary: { count: 0, average: null, lowest: null } }); + expect(screen.queryByTestId("feedback-score")).not.toBeInTheDocument(); + expect(screen.getByTitle("No feedback yet")).toHaveTextContent("—"); + }); +}); diff --git a/ui/litellm-dashboard/src/components/lens/traces/list/AgentTracesTable.tsx b/ui/litellm-dashboard/src/components/lens/traces/list/AgentTracesTable.tsx index cb39eea63e0..dcc6161a31d 100644 --- a/ui/litellm-dashboard/src/components/lens/traces/list/AgentTracesTable.tsx +++ b/ui/litellm-dashboard/src/components/lens/traces/list/AgentTracesTable.tsx @@ -1,7 +1,7 @@ "use client"; import { getCoreRowModel, useReactTable, type ColumnDef, type TableOptions } from "@tanstack/react-table"; -import { ArrowDown, ChevronRight, Plus } from "lucide-react"; +import { ArrowDown, ChevronRight, Plus, Star } from "lucide-react"; import { createContext, useContext, useEffect } from "react"; import { useInView } from "react-intersection-observer"; @@ -16,6 +16,7 @@ import { formatActivityTimestamp, formatRunTimestamp, localTimeZoneAbbreviation import { SpanIcon } from "../ui/SpanIcon"; import type { TraceFindingState } from "./useTraceFindings"; import { flaggedSignals, isFlagged, type TraceSignalState } from "./useTraceSignals"; +import { LOW_SCORE, isLowFeedback, type TraceFeedbackState } from "./useTraceFeedback"; import { SignalPills } from "../ui/SignalPills"; import { FrameworkLogo, traceFramework } from "../ui/TraceFramework"; import type { TraceSummary } from "../types"; @@ -25,6 +26,7 @@ import { fmtMs, previewText, traceAgentNames } from "../utils"; interface AgentTracesTableProps { traces: TraceSummary[]; findings: ReadonlyMap; + feedback?: ReadonlyMap; canViewFindings?: boolean; signals?: ReadonlyMap; showSignals?: boolean; @@ -89,6 +91,9 @@ const SignalsContext = createContext>(new const NO_SIGNALS: ReadonlyMap = new Map(); const SignalSetupContext = createContext<{ configured: boolean; onSetUp?: () => void }>({ configured: false }); const FLAGGED_ROW = "bg-destructive/[0.04] shadow-[inset_2px_0_0_var(--color-destructive)] hover:bg-destructive/[0.07]"; +const FeedbackContext = createContext>(new Map()); +const NO_FEEDBACK: ReadonlyMap = new Map(); +const feedbackOrEmpty = (feedback?: ReadonlyMap) => feedback ?? NO_FEEDBACK; function AgentCell({ run }: { run: TraceSummary }) { const framework = traceFramework(run); @@ -181,6 +186,32 @@ function SignalsCell({ run }: { run: TraceSummary }) { return ; } +export const formatScore = (score: number): string => (Number.isInteger(score) ? String(score) : score.toFixed(1)); + +export function FeedbackScore({ run }: { run: TraceSummary }) { + const state = useContext(FeedbackContext).get(runKey(run)); + if (!state || state.status === "pending") + return ; + if (state.status === "error") return Unavailable; + const { count, average, lowest } = state.summary; + if (count === 0 || average == null) return —; + const low = lowest != null && lowest <= LOW_SCORE; + return ( + + + {formatScore(average)}/10 + + ); +} + const RUN_COLUMNS: ColumnDef[] = [ { id: "time", @@ -263,6 +294,13 @@ const RUN_COLUMNS: ColumnDef[] = [ cell: ({ row }) => , meta: { numeric: true, className: NUM }, }, + { + id: "feedback", + size: 96, + header: "Feedback", + cell: ({ row }) => , + meta: { numeric: true, className: NUM }, + }, { id: "open", size: 32, @@ -309,12 +347,24 @@ function EmptyRuns({ rangeEmpty, onSetUpTracing }: { rangeEmpty: boolean; onSetU ); } +function rowFlags( + run: TraceSummary, + feedback: ReadonlyMap | undefined, + signals: ReadonlyMap, + showSignals: boolean, +) { + const lowFeedback = isLowFeedback(feedback?.get(runKey(run))); + const signalled = showSignals && isFlagged(signals.get(runKey(run))); + return { lowFeedback, flagged: lowFeedback || signalled }; +} + const bodyClassName = (blurred?: boolean) => cn("transition-[filter]", blurred && "blur-[1.5px]"); /** Devtool-dense runs list: one row per agent run, newest first. */ export function AgentTracesTable({ traces, findings, + feedback, canViewFindings = true, signals = NO_SIGNALS, showSignals = false, @@ -350,64 +400,67 @@ export function AgentTracesTable({ const table = useReactTable(tableOptions); return ( - - - - - - - className={bodyClassName(isPlaceholder)} - rowHeight={() => ROW_HEIGHT} - after={ - <> - {isLoading && SKELETON_ROWS.map((row) => )} - {autoContinue && } - - } - > - {(row) => { - const flagged = showSignals && isFlagged(signals.get(runKey(row.original))); - return ( - - ); - }} - - - {isLoading && ( -

    - Loading runs… -

    - )} - {error && ( -
    - - {traces.length ? "Could not load more runs" : "Could not load runs"}: {error.message} - - {onRetry && ( - + )} +
    + )} + {canContinue && traces.length === 0 && ( +
    + No loaded runs match these filters. + - )} -
    - )} - {canContinue && traces.length === 0 && ( -
    - No loaded runs match these filters. - -
    - )} - {isEmpty && } -
    -
    -
    +
    + )} + {isEmpty && } + + + + ); } diff --git a/ui/litellm-dashboard/src/components/lens/traces/list/useTraceFeedback.integration.test.tsx b/ui/litellm-dashboard/src/components/lens/traces/list/useTraceFeedback.integration.test.tsx new file mode 100644 index 00000000000..13710ecb606 --- /dev/null +++ b/ui/litellm-dashboard/src/components/lens/traces/list/useTraceFeedback.integration.test.tsx @@ -0,0 +1,35 @@ +import { screen } from "@testing-library/react"; +import { beforeEach, describe, expect, it, vi } from "vitest"; + +import { renderWithProviders, testQueryClient } from "../../../../../tests/test-utils"; +import { type TracesApi, TracesApiContext } from "../api"; +import type { TraceSummary } from "../types"; +import { useTraceFeedback } from "./useTraceFeedback"; + +const run = { trace_id: "trace-1", trace_ref: "REF1" } as TraceSummary; + +function Probe() { + const state = useTraceFeedback("sk-test", [run], true).get("REF1"); + return {state?.status ?? "missing"}; +} + +const renderWith = (feedbackSummary: () => Promise) => + renderWithProviders( + + + , + ); + +describe("useTraceFeedback", () => { + beforeEach(() => testQueryClient.clear()); + + it("reads a run's summary from the batch response", async () => { + renderWith(async () => [{ ...run, count: 1, average: 2, lowest: 2 }]); + expect(await screen.findByText("ready")).toBeInTheDocument(); + }); + + it("marks feedback unavailable instead of crashing when the proxy answers with something other than a list", async () => { + renderWith(vi.fn(async () => ({ detail: "Not Found" }))); + expect(await screen.findByText("error")).toBeInTheDocument(); + }); +}); diff --git a/ui/litellm-dashboard/src/components/lens/traces/list/useTraceFeedback.test.ts b/ui/litellm-dashboard/src/components/lens/traces/list/useTraceFeedback.test.ts new file mode 100644 index 00000000000..f4f943290a6 --- /dev/null +++ b/ui/litellm-dashboard/src/components/lens/traces/list/useTraceFeedback.test.ts @@ -0,0 +1,21 @@ +import { describe, expect, it } from "vitest"; + +import { LOW_SCORE, isLowFeedback, type TraceFeedbackState } from "./useTraceFeedback"; + +const ready = (count: number, lowest: number | null): TraceFeedbackState => ({ + status: "ready", + summary: { count, average: lowest, lowest }, +}); + +describe("isLowFeedback", () => { + it.each([ + { name: "a run an end user scored at the threshold", state: ready(1, LOW_SCORE), low: true }, + { name: "a run whose lowest score is just above the threshold", state: ready(3, LOW_SCORE + 1), low: false }, + { name: "a run nobody scored", state: ready(0, null), low: false }, + { name: "a run whose feedback is still loading", state: { status: "pending" } as const, low: false }, + { name: "a run whose feedback failed to load", state: { status: "error" } as const, low: false }, + { name: "a run with no feedback state", state: undefined, low: false }, + ])("flags $name only when it is low", ({ state, low }) => { + expect(isLowFeedback(state)).toBe(low); + }); +}); diff --git a/ui/litellm-dashboard/src/components/lens/traces/list/useTraceFeedback.ts b/ui/litellm-dashboard/src/components/lens/traces/list/useTraceFeedback.ts new file mode 100644 index 00000000000..b35542a0943 --- /dev/null +++ b/ui/litellm-dashboard/src/components/lens/traces/list/useTraceFeedback.ts @@ -0,0 +1,54 @@ +import { useQueries } from "@tanstack/react-query"; +import { chunk } from "es-toolkit"; + +import { useTracesApi } from "../api"; +import type { TraceFeedbackSummary, TraceSummary } from "../types"; + +export const LOW_SCORE = 4; + +export type TraceFeedbackState = + | { status: "ready"; summary: Pick } + | { status: "pending" } + | { status: "error" }; + +export const traceFeedbackKey = (accessToken: string) => ["traceFeedback", accessToken] as const; + +/** A run some end user scored at or below the low-score threshold. */ +export const isLowFeedback = (state: TraceFeedbackState | undefined): boolean => + state?.status === "ready" && state.summary.count > 0 && (state.summary.lowest ?? Infinity) <= LOW_SCORE; + +export function useTraceFeedback(accessToken: string, runs: TraceSummary[], isActive: boolean) { + const api = useTracesApi(accessToken); + const batches = chunk( + runs.map(({ trace_id, trace_ref }) => ({ trace_id, trace_ref: trace_ref ?? "" })), + 500, + ); + const queries = useQueries({ + queries: batches.map((traces) => ({ + queryKey: [...traceFeedbackKey(accessToken), traces], + queryFn: async () => { + const summaries: unknown = await api.feedbackSummary(traces); + if (!Array.isArray(summaries)) throw new Error("Unexpected feedback summary response"); + return summaries as TraceFeedbackSummary[]; + }, + enabled: isActive, + staleTime: 15000, + refetchInterval: isActive && api.live ? 15000 : false, + retry: false, + })), + }); + return new Map( + batches.flatMap((traces, index) => { + const query = queries[index]; + const summaries = new Map(query.data?.map((summary) => [summary.trace_ref || summary.trace_id, summary])); + return traces.map((trace): [string, TraceFeedbackState] => { + const key = trace.trace_ref || trace.trace_id; + if (query.isError) return [key, { status: "error" }]; + if (query.isPending) return [key, { status: "pending" }]; + const summary = summaries.get(key); + if (!summary) return [key, { status: "error" }]; + return [key, { status: "ready", summary }]; + }); + }), + ); +} diff --git a/ui/litellm-dashboard/src/components/lens/traces/types.ts b/ui/litellm-dashboard/src/components/lens/traces/types.ts index ecd9586e42a..04525fbb34c 100644 --- a/ui/litellm-dashboard/src/components/lens/traces/types.ts +++ b/ui/litellm-dashboard/src/components/lens/traces/types.ts @@ -18,6 +18,11 @@ export type TraceFindingsRequest = components["schemas"]["TraceFindingsRequest"] export type TraceFindingCount = components["schemas"]["TraceFindingCount"]; export type TraceSignals = components["schemas"]["TraceSignals"]; export type SignalFlag = NonNullable[number]; +export type TraceFeedback = components["schemas"]["TraceFeedback"]; +export type Feedback = components["schemas"]["Feedback"]; +export type TraceFeedbackRequest = components["schemas"]["TraceFeedbackRequest"]; +export type TraceFeedbackSummary = components["schemas"]["TraceFeedbackSummary"]; +export type FeedbackQuery = NonNullable; type ApiSpanDetail = paths["/v1/traces/{trace_id}/spans/{span_id}"]["get"]["responses"][200]["content"]["application/json"]; export type Span = Trace["spans"][number]; diff --git a/ui/litellm-dashboard/src/lib/http/schema.d.ts b/ui/litellm-dashboard/src/lib/http/schema.d.ts index bf711498307..54de1d30d3e 100644 --- a/ui/litellm-dashboard/src/lib/http/schema.d.ts +++ b/ui/litellm-dashboard/src/lib/http/schema.d.ts @@ -9088,6 +9088,42 @@ export interface paths { patch?: never; trace?: never; }; + "/lens/feedback": { + parameters: { + query?: never; + header?: never; + path?: never; + cookie?: never; + }; + /** Read Feedback */ + get: operations["read_feedback_lens_feedback_get"]; + /** Submit Feedback */ + put: operations["submit_feedback_lens_feedback_put"]; + post?: never; + /** Delete Feedback */ + delete: operations["delete_feedback_lens_feedback_delete"]; + options?: never; + head?: never; + patch?: never; + trace?: never; + }; + "/lens/feedback/summary": { + parameters: { + query?: never; + header?: never; + path?: never; + cookie?: never; + }; + get?: never; + put?: never; + /** Feedback Summary */ + post: operations["feedback_summary_lens_feedback_summary_post"]; + delete?: never; + options?: never; + head?: never; + patch?: never; + trace?: never; + }; "/lens/internal/ingestion-credentials": { parameters: { query?: never; @@ -32530,6 +32566,56 @@ export interface components { */ model: string; }; + /** Feedback */ + Feedback: { + /** Author */ + author: string; + /** Comment */ + comment: string; + /** + * Created At + * Format: date-time + */ + created_at: string; + /** Score */ + score: number; + /** Trace Id */ + trace_id: string; + /** + * Trace Ref + * @default + */ + trace_ref: string; + /** + * Updated At + * Format: date-time + */ + updated_at: string; + }; + /** FeedbackSubmission */ + FeedbackSubmission: { + /** + * Comment + * @default + */ + comment: string; + /** Score */ + score: number; + /** Session Id */ + session_id?: string | null; + /** Trace Id */ + trace_id?: string | null; + /** + * Trace Ref + * @default + */ + trace_ref: string; + /** + * User + * @default + */ + user: string; + }; /** FieldDetail */ FieldDetail: { /** Field Default Value */ @@ -47997,6 +48083,39 @@ export interface components { /** Agents */ agents: components["schemas"]["TraceAgent"][]; }; + /** TraceFeedback */ + TraceFeedback: { + /** Feedback */ + feedback: components["schemas"]["Feedback"][]; + /** Trace Id */ + trace_id: string; + /** + * Trace Ref + * @default + */ + trace_ref: string; + }; + /** TraceFeedbackRequest */ + TraceFeedbackRequest: { + /** Traces */ + traces: components["schemas"]["TraceIdentity"][]; + }; + /** TraceFeedbackSummary */ + TraceFeedbackSummary: { + /** Average */ + average: number | null; + /** Count */ + count: number; + /** Lowest */ + lowest: number | null; + /** Trace Id */ + trace_id: string; + /** + * Trace Ref + * @default + */ + trace_ref: string; + }; /** TraceFindingCount */ TraceFindingCount: { /** Finding Count */ @@ -63613,6 +63732,137 @@ export interface operations { }; }; }; + read_feedback_lens_feedback_get: { + parameters: { + query?: { + trace_id?: string | null; + session_id?: string | null; + trace_ref?: string; + }; + header?: never; + path?: never; + cookie?: never; + }; + requestBody?: never; + responses: { + /** @description Successful Response */ + 200: { + headers: { + [name: string]: unknown; + }; + content: { + "application/json": components["schemas"]["TraceFeedback"]; + }; + }; + /** @description Validation Error */ + 422: { + headers: { + [name: string]: unknown; + }; + content: { + "application/json": components["schemas"]["HTTPValidationError"]; + }; + }; + }; + }; + submit_feedback_lens_feedback_put: { + parameters: { + query?: never; + header?: never; + path?: never; + cookie?: never; + }; + requestBody: { + content: { + "application/json": components["schemas"]["FeedbackSubmission"]; + }; + }; + responses: { + /** @description Successful Response */ + 200: { + headers: { + [name: string]: unknown; + }; + content: { + "application/json": components["schemas"]["Feedback"]; + }; + }; + /** @description Validation Error */ + 422: { + headers: { + [name: string]: unknown; + }; + content: { + "application/json": components["schemas"]["HTTPValidationError"]; + }; + }; + }; + }; + delete_feedback_lens_feedback_delete: { + parameters: { + query?: { + trace_id?: string | null; + session_id?: string | null; + trace_ref?: string; + user?: string; + }; + header?: never; + path?: never; + cookie?: never; + }; + requestBody?: never; + responses: { + /** @description Successful Response */ + 204: { + headers: { + [name: string]: unknown; + }; + content?: never; + }; + /** @description Validation Error */ + 422: { + headers: { + [name: string]: unknown; + }; + content: { + "application/json": components["schemas"]["HTTPValidationError"]; + }; + }; + }; + }; + feedback_summary_lens_feedback_summary_post: { + parameters: { + query?: never; + header?: never; + path?: never; + cookie?: never; + }; + requestBody: { + content: { + "application/json": components["schemas"]["TraceFeedbackRequest"]; + }; + }; + responses: { + /** @description Successful Response */ + 200: { + headers: { + [name: string]: unknown; + }; + content: { + "application/json": components["schemas"]["TraceFeedbackSummary"][]; + }; + }; + /** @description Validation Error */ + 422: { + headers: { + [name: string]: unknown; + }; + content: { + "application/json": components["schemas"]["HTTPValidationError"]; + }; + }; + }; + }; ingestion_credentials_lens_internal_ingestion_credentials_get: { parameters: { query?: never;