From cb00dbecd7d1964b9d839000f26bf60120d7dc8e Mon Sep 17 00:00:00 2001 From: moe-berri Date: Sat, 3 Oct 2026 01:18:16 -0700 Subject: [PATCH] feat(ui): improve trace inspection and ROI estimation (#44351) * feat(lens): simplify trace inspection in the gateway drawer * feat(lens): add a conversation view for full traces * test(lens): keep normalized message fixtures type safe * refactor(lens): make conversation view read like a chat * refactor(lens): use a quiet trace view menu * refactor(lens): make trace view tabs explicit * refactor(lens): restore compact trace view switch * fix(lens): preserve complete conversation history and tool types * feat(lens): add trace full-screen and close controls * fix(lens): show forwarded answers and agent errors once * fix(lens): reset full screen when closing a trace * fix(roi): estimate linked authors and clarify model selection * fix: preserve trace errors and ROI results across partial failures * fix(roi): correct pagination variable typing * fix(ui): place loaded root failures in conversation order * fix(roi): read estimator recommendations from model catalog * revert: remove catalog-driven ROI recommendations --- litellm/proxy/_lazy_openapi_snapshot.json | 29 ++ .../roi_calculator_endpoints.py | 56 ++- litellm/proxy/roi_calculator/analytics.py | 3 +- litellm/proxy/roi_calculator/sync.py | 78 +++- litellm/types/roi_calculator.py | 6 + .../test_roi_calculator_endpoints.py | 64 +++ tests/unit/proxy/roi_calculator/test_sync.py | 385 ++++++++++++++++-- .../test_spend_management_endpoints.py | 2 +- ui/litellm-dashboard/AGENTS.md | 10 + ui/litellm-dashboard/eslint-suppressions.json | 5 - .../lens/_components/lensDemoData.test.ts | 15 + .../lens/_components/lensDemoData.ts | 10 +- .../lens/_components/lensDemoLongTrace.ts | 119 ++++++ .../ROICalculatorView.integration.test.tsx | 34 ++ .../_components/ROISettingsPanel.tsx | 31 +- .../_components/roiCalculatorData.test.ts | 43 +- .../_components/roiCalculatorData.ts | 22 + .../view_logs/TraceView/CopyButton.tsx | 6 +- .../view_logs/TraceView/DetailContent.tsx | 21 +- .../TraceView/DetailPane.integration.test.tsx | 40 +- .../view_logs/TraceView/DetailPane.tsx | 78 ++-- .../components/view_logs/TraceView/IdChip.tsx | 2 +- .../view_logs/TraceView/KeyValueRows.tsx | 22 +- .../view_logs/TraceView/MessageCard.test.tsx | 2 +- .../view_logs/TraceView/MessageCard.tsx | 162 +++----- .../view_logs/TraceView/RunDrawer.test.tsx | 64 ++- .../view_logs/TraceView/RunDrawer.tsx | 34 +- .../view_logs/TraceView/SpanIcon.tsx | 21 +- .../view_logs/TraceView/SpanTree.tsx | 253 +++++++----- .../TraceConversation.integration.test.tsx | 195 +++++++++ .../view_logs/TraceView/TraceConversation.tsx | 198 +++++++++ .../view_logs/TraceView/TraceDrawer.test.tsx | 74 +++- .../view_logs/TraceView/TraceDrawer.tsx | 189 ++++++--- .../view_logs/TraceView/conversation.test.ts | 254 ++++++++++++ .../view_logs/TraceView/conversation.ts | 256 ++++++++++++ .../view_logs/TraceView/traceUtils.test.ts | 38 ++ .../view_logs/TraceView/traceUtils.ts | 21 + ui/litellm-dashboard/src/lib/http/schema.d.ts | 12 + 38 files changed, 2411 insertions(+), 443 deletions(-) create mode 100644 ui/litellm-dashboard/src/app/(dashboard)/lens/_components/lensDemoLongTrace.ts create mode 100644 ui/litellm-dashboard/src/components/view_logs/TraceView/TraceConversation.integration.test.tsx create mode 100644 ui/litellm-dashboard/src/components/view_logs/TraceView/TraceConversation.tsx create mode 100644 ui/litellm-dashboard/src/components/view_logs/TraceView/conversation.test.ts create mode 100644 ui/litellm-dashboard/src/components/view_logs/TraceView/conversation.ts diff --git a/litellm/proxy/_lazy_openapi_snapshot.json b/litellm/proxy/_lazy_openapi_snapshot.json index 5b3736d22a9..9eaf6c7e9ed 100644 --- a/litellm/proxy/_lazy_openapi_snapshot.json +++ b/litellm/proxy/_lazy_openapi_snapshot.json @@ -49148,6 +49148,27 @@ "title": "ROIEstimateResponse", "type": "object" }, + "ROIEstimatorModel": { + "properties": { + "model_name": { + "title": "Model Name", + "type": "string" + }, + "provider_models": { + "items": { + "type": "string" + }, + "title": "Provider Models", + "type": "array" + } + }, + "required": [ + "model_name", + "provider_models" + ], + "title": "ROIEstimatorModel", + "type": "object" + }, "ROIIdentityMapResponse": { "properties": { "identity_map": { @@ -49583,6 +49604,14 @@ "title": "Estimator Model", "type": "string" }, + "estimator_models": { + "default": [], + "items": { + "$ref": "#/components/schemas/ROIEstimatorModel" + }, + "title": "Estimator Models", + "type": "array" + }, "estimator_prompt": { "title": "Estimator Prompt", "type": "string" diff --git a/litellm/proxy/management_endpoints/roi_calculator_endpoints.py b/litellm/proxy/management_endpoints/roi_calculator_endpoints.py index 728f41d7c2a..bb6d60db723 100644 --- a/litellm/proxy/management_endpoints/roi_calculator_endpoints.py +++ b/litellm/proxy/management_endpoints/roi_calculator_endpoints.py @@ -32,8 +32,10 @@ from litellm.proxy.roi_calculator.github import SourceError from litellm.proxy.roi_calculator.source import create_source from litellm.proxy.roi_calculator.sync import ( BranchSpendReader, + GatewayUserReader, SpendReader, SyncManager, + read_gateway_user_emails, read_spend, spend_prisma_client, ) @@ -43,6 +45,7 @@ from litellm.types.roi_calculator import ( DEFAULT_PROMPT, ROIBranchSpend, ROICompletionRequest, + ROIEstimatorModel, ROIIdentityMapResponse, ROIIdentityMapUpdate, ROIReport, @@ -94,11 +97,13 @@ class _RouterEstimatorModelInfo(BaseModel): model_config = ConfigDict(extra="ignore", from_attributes=True) base_model: str | None = None + mode: str | None = None class _RouterEstimatorDeployment(BaseModel): model_config = ConfigDict(extra="ignore", from_attributes=True) + model_name: str = "" litellm_params: _RouterEstimatorParams model_info: _RouterEstimatorModelInfo | None = None @@ -144,7 +149,6 @@ def get_github_transport() -> httpx.AsyncBaseTransport | None: _ROUTER_ESTIMATOR_DEPLOYMENTS: Final = TypeAdapter(tuple[_RouterEstimatorDeployment, ...]) -_MODEL_NAMES: Final = TypeAdapter(tuple[str, ...]) def _estimator_models_from_deployments(deployments: Sequence[object]) -> tuple[EstimatorModel, ...]: @@ -177,12 +181,45 @@ def _router_estimator_models(model_group: str) -> tuple[EstimatorModel, ...]: return _estimator_models_from_deployments(deployments) -def _router_models() -> tuple[str, ...]: +def _is_estimator_deployment(deployment: _RouterEstimatorDeployment) -> bool: + from litellm import model_cost + + underlying: Final = _estimator_model(deployment) + if underlying is None: + return False + model, provider = underlying + candidates: Final = (f"{provider}/{model}", model, model.split("/", 1)[-1]) + known_modes: Final = tuple( + _RouterEstimatorModelInfo.model_validate(model_cost[name]).mode for name in candidates if name in model_cost + ) + mode: Final = (deployment.model_info.mode if deployment.model_info else None) or next(iter(known_modes), None) + return mode in (None, "chat") + + +def _estimator_choices_from_deployments(deployments: Sequence[object]) -> tuple[ROIEstimatorModel, ...]: + parsed: Final = _ROUTER_ESTIMATOR_DEPLOYMENTS.validate_python(deployments) + names: Final = sorted( + frozenset(item.model_name for item in parsed if item.model_name and "*" not in item.model_name) + ) + groups: Final = tuple(tuple(item for item in parsed if item.model_name == name) for name in names) + return tuple( + ROIEstimatorModel( + model_name=group[0].model_name, + provider_models=tuple(sorted(frozenset(model[0] for item in group if (model := _estimator_model(item))))), + ) + for group in groups + if all(_is_estimator_deployment(item) for item in group) + ) + + +def _router_estimator_choices() -> tuple[ROIEstimatorModel, ...]: from litellm.proxy.proxy_server import llm_router if llm_router is None: return () - return tuple(sorted(frozenset(_MODEL_NAMES.validate_python(llm_router.get_model_names())))) + names: Final = frozenset(llm_router.get_model_names()) + choices: Final = _estimator_choices_from_deployments(llm_router.get_model_list() or ()) + return tuple(choice for choice in choices if choice.model_name in names) async def _load_stored_settings(repository: ConfigRepository) -> _StoredSettings: @@ -262,7 +299,8 @@ async def _load_report(repository: ConfigRepository, settings: ROISettings) -> R def _public_settings(settings: ROISettings) -> ROISettingsResponse: - models: Final = _router_models() + choices: Final = _router_estimator_choices() + models: Final = tuple(choice.model_name for choice in choices) return ROISettingsResponse( source_provider=settings.source_provider, gitlab_api_url=settings.gitlab_api_url, @@ -278,6 +316,7 @@ def _public_settings(settings: ROISettings) -> ROISettingsResponse: update_interval_minutes=settings.update_interval_minutes, default_prompt=DEFAULT_PROMPT, available_models=models, + estimator_models=choices, ready=bool(settings.repos and settings.estimator_model and settings.estimator_model in models), ) @@ -353,6 +392,13 @@ async def _test_estimator_access(settings: ROISettings) -> None: raise HTTPException(status_code=409, detail="The estimator key could not connect to the gateway.") from None +def _gateway_user_reader(repository: ConfigRepository) -> GatewayUserReader: + async def get_emails() -> frozenset[str]: + return await read_gateway_user_emails(spend_prisma_client(repository.prisma_client)) + + return get_emails + + def _spend_reader(repository: ConfigRepository) -> SpendReader: async def get_spend(start: date, end: date) -> tuple[ROISpendRecord, ...]: prisma_client: Final = spend_prisma_client(repository.prisma_client) @@ -540,6 +586,7 @@ async def start_roi_calculator_sync( _router_estimator_models(settings.estimator_model), SyncStore(repository.prisma_client), branch_spend_reader=_branch_spend_reader(repository, settings), + gateway_user_reader=_gateway_user_reader(repository), ): raise HTTPException(status_code=409, detail="A sync is already running.") return manager.status @@ -686,6 +733,7 @@ async def run_scheduled_sync() -> None: coordinator=store, scheduled_interval=settings.update_interval_minutes, branch_spend_reader=_branch_spend_reader(repository, settings), + gateway_user_reader=_gateway_user_reader(repository), ) diff --git a/litellm/proxy/roi_calculator/analytics.py b/litellm/proxy/roi_calculator/analytics.py index 7bf469f8936..5379ad48f18 100644 --- a/litellm/proxy/roi_calculator/analytics.py +++ b/litellm/proxy/roi_calculator/analytics.py @@ -6,6 +6,7 @@ from litellm.types.roi_calculator import ( ROIBranchAttribution, ROIBranchMetrics, ROIPersonSummary, + ROIPullEvidence, ROIPullRecord, ROIPullSummary, ROIReport, @@ -27,7 +28,7 @@ def normalize_email(value: str | None) -> str: def match_identity( - pull: ROIPullRecord, + pull: ROIPullRecord | ROIPullEvidence, observed_emails: frozenset[str], mappings: Mapping[str, str], ) -> tuple[str, str]: diff --git a/litellm/proxy/roi_calculator/sync.py b/litellm/proxy/roi_calculator/sync.py index e134c23dab1..de9a1a979f1 100644 --- a/litellm/proxy/roi_calculator/sync.py +++ b/litellm/proxy/roi_calculator/sync.py @@ -1,5 +1,5 @@ import asyncio -from collections.abc import Awaitable, Mapping, Sequence +from collections.abc import AsyncIterator, Awaitable, Mapping, Sequence from contextlib import suppress from datetime import date, datetime, timedelta, timezone from itertools import chain @@ -12,6 +12,7 @@ from pydantic import BaseModel, ConfigDict, Field, TypeAdapter from typing_extensions import ReadOnly, TypedDict, Unpack from litellm._logging import verbose_proxy_logger +from litellm.proxy.roi_calculator.analytics import match_identity, normalize_email from litellm.proxy.roi_calculator.estimator import CompletionCaller, Estimator, EstimatorModel, cache_context from litellm.proxy.roi_calculator.github import GitHubPullListItem, SourceError from litellm.proxy.roi_calculator.pull_cache import cache_key, settings_fingerprint @@ -29,6 +30,7 @@ from litellm.types.roi_calculator import ( ) PR_CONCURRENCY: Final = 3 +_GATEWAY_USER_PAGE_SIZE: Final = 1000 _ESTIMATE_ADAPTER: Final = TypeAdapter(ROIEstimate) _REPORT_ADAPTER: Final = TypeAdapter(ROIReport) _JSON_OBJECT_ADAPTER: Final = TypeAdapter(dict[str, object]) @@ -76,6 +78,8 @@ class _UserTable(Protocol): class _PrismaDatabase(Protocol): + async def query_raw(self, query: str, *args: object) -> object: ... + @property def litellm_dailyuserspend(self) -> _DailySpendTable: ... @@ -124,8 +128,6 @@ async def read_spend( start: date, end: date, ) -> tuple[ROISpendRecord, ...]: - from litellm.proxy.roi_calculator.analytics import normalize_email - database: Final = prisma_client.db daily_table: Final = database.litellm_dailyuserspend group_by: Final = TypeAdapter(list[Literal["user_id", "date"]]).validate_python(("user_id", "date")) @@ -166,6 +168,35 @@ async def read_spend( ) +async def _gateway_users(database: _PrismaDatabase) -> AsyncIterator[_UserEmail]: + cursor: str | None = None # rebind-ok: keyset pagination advances after each bounded page + while True: + users: tuple[_UserEmail, ...] = _USER_EMAILS.validate_python( + await database.query_raw( + 'SELECT "user_id", "user_email" FROM "LiteLLM_UserTable" ' + 'WHERE "user_email" IS NOT NULL AND ($1::text IS NULL OR "user_id" > $1) ' + 'ORDER BY "user_id" LIMIT $2', + cursor, + _GATEWAY_USER_PAGE_SIZE, + ) + ) + for user in users: + yield user + if len(users) < _GATEWAY_USER_PAGE_SIZE: + return + cursor = users[-1].user_id + + +async def read_gateway_user_emails(prisma_client: _SpendPrismaClient) -> frozenset[str]: + return frozenset( + [email async for user in _gateway_users(prisma_client.db) if (email := normalize_email(user.user_email))] + ) + + +class GatewayUserReader(Protocol): + def __call__(self) -> Awaitable[frozenset[str]]: ... + + class GitHubFactory(Protocol): def __call__( self, @@ -206,6 +237,26 @@ def _utc_now() -> datetime: return datetime.now(timezone.utc) +def _unlinked_estimate( + pull: ROIPullEvidence | ROIPullRecord, + gateway_emails: frozenset[str], + mappings: Mapping[str, str], +) -> ROIEstimate | None: + email, method = match_identity(pull, gateway_emails, mappings) + if email and email in gateway_emails: + return None + reason: Final = ( + "Multiple gateway users match this author." + if method == "ambiguous emails" + else "This author is not linked to a registered gateway user." + ) + return { + "status": "needs_review", + "hours": None, + "reasoning": f"Not estimated: {reason} Link the author to a gateway user and run analysis again.", + } + + async def _estimate_with_fallback( estimator: Estimator, evidence: ROIPullEvidence, @@ -322,8 +373,8 @@ def _processed_records(processed: tuple[_ProcessedPull, ...]) -> Mapping[int, RO raise SourceError( "The repository source could not provide PR metadata. No new report was published; try analysis again later." ) - if any(item.record["estimate"]["status"] == "error" for item in processed) and not any( - item.record["estimate"]["status"] == "estimated" for item in processed + if any(item.record["estimate"]["status"] == "error" for item in processed) and all( + item.record["estimate"]["status"] == "error" or item.metadata_unavailable for item in processed ): raise SourceError( "The estimator could not score any merged changes. No new report was published; " @@ -399,6 +450,8 @@ class SyncManager: coordinator: SyncCoordinator | None = None, scheduled_interval: float = 0, branch_spend_reader: BranchSpendReader | None = None, + *, + gateway_user_reader: GatewayUserReader, ) -> bool: async with self._start_lock: if not settings.repos or not settings.estimator_model: @@ -439,6 +492,7 @@ class SyncManager: coordinator, owner, branch_spend_reader, + gateway_user_reader, ) ) return True @@ -481,12 +535,14 @@ class SyncManager: coordinator: SyncCoordinator | None, owner: str, branch_spend_reader: BranchSpendReader | None, + gateway_user_reader: GatewayUserReader, ) -> None: monitor: Final = asyncio.create_task(self._heartbeat(asyncio.current_task(), coordinator, owner)) github: Final = self._github_factory(settings, github_transport) try: end: Final = self._clock().date() start: Final = end - timedelta(days=settings.backfill_days - 1) + gateway_emails: Final = await gateway_user_reader() spend: Final = await spend_reader(start, end) self._update_status(phase="repositories", stage="Reading configured repositories") repositories: Final = await _read_repositories(github, settings.repos, start, end) @@ -545,15 +601,21 @@ class SyncManager: await _cache_estimated_pull( repository, key, cached_record, cached_pull if saved is not None else None ) - self._update_estimate_progress(cached_record["estimate"]) - return _ProcessedPull(index, cached_record) + cached_estimate: Final = ( + _unlinked_estimate(cached_record, gateway_emails, settings.identity_map) + or cached_record["estimate"] + ) + self._update_estimate_progress(cached_estimate) + return _ProcessedPull(index, {**cached_record, "estimate": cached_estimate}) try: evidence: Final = await github.evidence(repo, pull) except SourceError as exc: unavailable: Final = await _unavailable_record(github, settings, repo, pull, exc) self._update_estimate_progress(unavailable["estimate"]) return _ProcessedPull(index, unavailable, metadata_unavailable=True) - estimate: Final = await _estimate_with_fallback(estimator, evidence) + estimate: Final = _unlinked_estimate( + evidence, gateway_emails, settings.identity_map + ) or await _estimate_with_fallback(estimator, evidence) evidence_item: Final = GitHubPullListItem.model_validate( MappingProxyType( { diff --git a/litellm/types/roi_calculator.py b/litellm/types/roi_calculator.py index b1c15d9a197..63a28ec71ca 100644 --- a/litellm/types/roi_calculator.py +++ b/litellm/types/roi_calculator.py @@ -134,6 +134,11 @@ class ROISettingsUpdate(BaseModel): update_interval_minutes: float | None = Field(default=None, ge=0, le=43200, allow_inf_nan=False) +class ROIEstimatorModel(BaseModel): + model_name: str + provider_models: tuple[str, ...] + + class ROISettingsResponse(BaseModel): source_provider: Literal["github", "gitlab"] = "github" gitlab_api_url: str = "https://gitlab.com/api/v4" @@ -149,6 +154,7 @@ class ROISettingsResponse(BaseModel): has_github_token: bool default_prompt: str available_models: tuple[str, ...] + estimator_models: tuple[ROIEstimatorModel, ...] = () ready: bool diff --git a/tests/unit/proxy/management_endpoints/test_roi_calculator_endpoints.py b/tests/unit/proxy/management_endpoints/test_roi_calculator_endpoints.py index 938c813f363..01d40f94d48 100644 --- a/tests/unit/proxy/management_endpoints/test_roi_calculator_endpoints.py +++ b/tests/unit/proxy/management_endpoints/test_roi_calculator_endpoints.py @@ -17,6 +17,7 @@ from litellm.proxy._types import LitellmUserRoles, UserAPIKeyAuth from litellm.proxy.auth.user_api_key_auth import user_api_key_auth from litellm.proxy.litellm_pre_call_utils import add_litellm_data_to_request from litellm.proxy.management_endpoints.roi_calculator_endpoints import ( + _estimator_choices_from_deployments, _estimator_models_from_deployments, _gateway_transport, _next_update, @@ -411,3 +412,66 @@ def test_old_source_report_is_not_returned_when_matching_new_source_identity() - assert matched.status_code == 200 assert matched.json()["report"] is None assert matched.json()["identity_map"] == {"dev.name": "dev@example.test"} + + +def test_estimator_choices_show_underlying_models_and_exclude_non_chat_routes() -> None: + deployments: Final = ( + { + "model_name": "estimator", + "litellm_params": {"model": "deployment-name"}, + "model_info": {"base_model": "gpt-6-luna", "mode": "chat"}, + }, + { + "model_name": "estimator", + "litellm_params": {"model": "second-deployment"}, + "model_info": {"base_model": "gpt-6-luna", "mode": "chat"}, + }, + { + "model_name": "embeddings", + "litellm_params": {"model": "custom-embedding"}, + "model_info": {"mode": "embedding"}, + }, + { + "model_name": "image", + "litellm_params": {"model": "custom-image"}, + "model_info": {"mode": "image_generation"}, + }, + {"model_name": "*", "litellm_params": {"model": "openai/*"}}, + {"model_name": "missing", "litellm_params": {}}, + {"model_name": "custom-chat", "litellm_params": {"model": "openai/private-model"}}, + ) + choices: Final = _estimator_choices_from_deployments(deployments) + assert tuple((choice.model_name, choice.provider_models) for choice in choices) == ( + ("custom-chat", ("openai/private-model",)), + ("estimator", ("gpt-6-luna",)), + ) + + +def test_estimator_picker_keeps_callable_aliases_and_routing_groups(monkeypatch: pytest.MonkeyPatch) -> None: + from litellm.proxy import proxy_server + from litellm.router import Router + + configured_router: Final = Router( + model_list=[ + { + "model_name": "concrete", + "litellm_params": {"model": "openai/gpt-6-luna", "api_key": "test"}, + }, + { + "model_name": "team-only", + "litellm_params": {"model": "openai/gpt-6-luna", "api_key": "test"}, + "model_info": {"team_id": "other-team", "team_public_model_name": "private-estimator"}, + }, + ], + model_group_alias={"friendly": "concrete"}, + routing_groups=[{"group_name": "balanced", "models": ["concrete"], "routing_strategy": "simple-shuffle"}], + ) + monkeypatch.setattr(proxy_server, "llm_router", configured_router) + client: Final = _client(LitellmUserRoles.PROXY_ADMIN, _ConfigRepository()) + for name in ("friendly", "balanced"): + response: Final = client.put("/roi-calculator/settings", json={"repos": ["org/repo"], "estimator_model": name}) + assert response.status_code == 200, response.text + settings: Final = response.json() + assert settings["ready"] is True + assert set(settings["available_models"]) == {"concrete", "friendly", "balanced"} + assert {"model_name": name, "provider_models": ["openai/gpt-6-luna"]} in settings["estimator_models"] diff --git a/tests/unit/proxy/roi_calculator/test_sync.py b/tests/unit/proxy/roi_calculator/test_sync.py index a0749f230fa..90b4a62fc65 100644 --- a/tests/unit/proxy/roi_calculator/test_sync.py +++ b/tests/unit/proxy/roi_calculator/test_sync.py @@ -12,7 +12,7 @@ from pydantic import TypeAdapter from litellm.proxy.roi_calculator.analytics import summarize from litellm.proxy.roi_calculator.estimator import CompletionCaller from litellm.proxy.roi_calculator.github import GitHubPullListItem -from litellm.proxy.roi_calculator.sync import SpendReader, SyncManager, read_spend +from litellm.proxy.roi_calculator.sync import SpendReader, SyncManager, read_gateway_user_emails, read_spend from litellm.types.roi_calculator import ( ROIBranchSpend, ROICompletionRequest, @@ -130,19 +130,40 @@ class _UserTable: where: Mapping[str, object], ) -> Sequence[Mapping[str, str | None]]: _assert_json_round_trip({"where": where}) + if where == {"user_email": {"not": None}}: + return ( + {"user_id": "u1", "user_email": " Alice@Example.com "}, + {"user_id": "inactive", "user_email": "inactive@example.com"}, + {"user_id": "invalid", "user_email": "not-an-email"}, + {"user_id": "private", "user_email": "123@users.noreply.github.com"}, + ) assert where == {"user_id": {"in": ["missing", "team@example.com", "u1"]}} return (MappingProxyType({"user_id": "u1", "user_email": " Alice@Example.com "}),) class _SpendDatabase: - def __init__(self) -> None: + def __init__(self, directory: tuple[Mapping[str, str], ...] = ()) -> None: self.litellm_dailyuserspend: Final = _DailySpendTable() self.litellm_usertable: Final = _UserTable() + self.directory: Final = directory or ( + {"user_id": "inactive", "user_email": "inactive@example.com"}, + {"user_id": "invalid", "user_email": "not-an-email"}, + {"user_id": "private", "user_email": "123@users.noreply.github.com"}, + {"user_id": "u1", "user_email": " Alice@Example.com "}, + ) + self.pages_read = 0 + + async def query_raw(self, query: str, *args: object) -> object: + cursor, size = args + assert cursor is None or isinstance(cursor, str) + assert isinstance(size, int) and 0 < size <= 1000 + self.pages_read += 1 + return tuple(row for row in self.directory if cursor is None or row["user_id"] > cursor)[:size] class _SpendPrismaClient: - def __init__(self) -> None: - self.db: Final = _SpendDatabase() + def __init__(self, directory: tuple[Mapping[str, str], ...] = ()) -> None: + self.db: Final = _SpendDatabase(directory) def _settings(estimator_prompt: str = "Estimate effort.") -> ROISettings: @@ -196,6 +217,10 @@ def _spend_reader() -> SpendReader: return read +async def _gateway_users() -> frozenset[str]: + return frozenset({"alice@example.com"}) + + def _completion() -> CompletionCaller: async def complete(request: ROICompletionRequest) -> object: assert request.model == "test-estimator" @@ -224,7 +249,9 @@ async def test_unchanged_estimated_pull_refreshes_identity_without_model_call() manager: Final = SyncManager(clock=_fixed_now) complete: Final = _completion() - assert await manager.start(_settings(), repository, _spend_reader(), complete, _transport()) + assert await manager.start( + _settings(), repository, _spend_reader(), complete, _transport(), gateway_user_reader=_gateway_users + ) await _wait_until_finished(manager) async def unexpected_completion(request: ROICompletionRequest) -> object: @@ -236,6 +263,7 @@ async def test_unchanged_estimated_pull_refreshes_identity_without_model_call() _spend_reader(), unexpected_completion, _transport(unexpected_details=True, profile_email="new@example.com"), + gateway_user_reader=_gateway_users, ) await _wait_until_finished(manager) @@ -305,11 +333,24 @@ async def test_gitlab_cache_refreshes_branch_attribution_when_source_access_chan async def branch_spend(start: date, end: date, repos: tuple[str, ...]) -> tuple[ROIBranchSpend, ...]: return (ROIBranchSpend(repo="gitlab.com/" + (after or "dev/fork"), branch="feature", spend=2.5, requests=3),) - assert await manager.start(settings, repository, _spend_reader(), _completion(), _gitlab_transport(before)) + assert await manager.start( + settings, + repository, + _spend_reader(), + _completion(), + _gitlab_transport(before), + gateway_user_reader=_gateway_users, + ) await _wait_until_finished(manager) assert manager.status.phase == "complete" assert await manager.start( - settings, repository, _spend_reader(), _completion(), _gitlab_transport(after), branch_spend_reader=branch_spend + settings, + repository, + _spend_reader(), + _completion(), + _gitlab_transport(after), + branch_spend_reader=branch_spend, + gateway_user_reader=_gateway_users, ) await _wait_until_finished(manager) assert manager.status.phase == "complete" @@ -321,7 +362,14 @@ async def test_gitlab_cache_refreshes_branch_attribution_when_source_access_chan async def unexpected_completion(request: ROICompletionRequest) -> object: raise AssertionError("Unchanged source metadata must reuse the estimate") - assert await manager.start(settings, repository, _spend_reader(), unexpected_completion, _gitlab_transport(after)) + assert await manager.start( + settings, + repository, + _spend_reader(), + unexpected_completion, + _gitlab_transport(after), + gateway_user_reader=_gateway_users, + ) await _wait_until_finished(manager) assert manager.status.phase == "complete" assert manager.status.reused == 1 @@ -343,6 +391,7 @@ async def test_unreadable_gitlab_details_keep_known_branch_costs() -> None: _completion(), _gitlab_transport("dev/fork", details_fail=True), branch_spend_reader=branch_spend, + gateway_user_reader=_gateway_users, ) await _wait_until_finished(manager) report: Final = TypeAdapter(ROIReport).validate_python(repository.values["roi_calculator_report"]) @@ -390,7 +439,9 @@ async def test_metadata_outage_keeps_previous_report_and_retries_on_next_run() - repository: Final = _ReportRepository() manager: Final = SyncManager(clock=_fixed_now) - assert await manager.start(_settings(), repository, _spend_reader(), _completion(), _transport()) + assert await manager.start( + _settings(), repository, _spend_reader(), _completion(), _transport(), gateway_user_reader=_gateway_users + ) await _wait_until_finished(manager) previous: Final = repository.values["roi_calculator_report"] assert await manager.start( @@ -399,6 +450,7 @@ async def test_metadata_outage_keeps_previous_report_and_retries_on_next_run() - _spend_reader(), _completion(), _transport(pull_detail_status=500), + gateway_user_reader=_gateway_users, ) await _wait_until_finished(manager) @@ -412,6 +464,7 @@ async def test_metadata_outage_keeps_previous_report_and_retries_on_next_run() - _spend_reader(), _completion(), _transport(), + gateway_user_reader=_gateway_users, ) await _wait_until_finished(manager) recovered: Final = TypeAdapter(ROIReport).validate_python(repository.values["roi_calculator_report"]) @@ -426,7 +479,9 @@ async def test_cancelling_estimation_leaves_the_previous_report_unchanged() -> N repository: Final = _ReportRepository() manager: Final = SyncManager(clock=_fixed_now) - assert await manager.start(_settings(), repository, _spend_reader(), _completion(), _transport()) + assert await manager.start( + _settings(), repository, _spend_reader(), _completion(), _transport(), gateway_user_reader=_gateway_users + ) await _wait_until_finished(manager) previous_report: Final = repository.values["roi_calculator_report"] @@ -441,6 +496,7 @@ async def test_cancelling_estimation_leaves_the_previous_report_unchanged() -> N _spend_reader(), blocked_completion, _transport(), + gateway_user_reader=_gateway_users, ) await entered_estimator.wait() @@ -453,11 +509,15 @@ async def test_cancelling_estimation_leaves_the_previous_report_unchanged() -> N async def test_immediate_cancel_allows_another_run() -> None: repository: Final = _ReportRepository() manager: Final = SyncManager(clock=_fixed_now) - assert await manager.start(_settings(), repository, _spend_reader(), _completion(), _transport()) + assert await manager.start( + _settings(), repository, _spend_reader(), _completion(), _transport(), gateway_user_reader=_gateway_users + ) assert await manager.cancel() assert manager.status.phase == "cancelled" assert manager.status.finished_at is not None - assert await manager.start(_settings(), repository, _spend_reader(), _completion(), _transport()) + assert await manager.start( + _settings(), repository, _spend_reader(), _completion(), _transport(), gateway_user_reader=_gateway_users + ) await _wait_until_finished(manager) assert manager.status.phase == "complete" @@ -466,7 +526,9 @@ async def test_immediate_cancel_allows_another_run() -> None: async def test_saved_estimates_survive_report_reset() -> None: repository: Final = _ReportRepository() manager: Final = SyncManager(clock=_fixed_now) - assert await manager.start(_settings(), repository, _spend_reader(), _completion(), _transport()) + assert await manager.start( + _settings(), repository, _spend_reader(), _completion(), _transport(), gateway_user_reader=_gateway_users + ) await _wait_until_finished(manager) repository.values = MappingProxyType( {key: value for key, value in repository.values.items() if key != "roi_calculator_report"} @@ -477,7 +539,12 @@ async def test_saved_estimates_survive_report_reset() -> None: restarted: Final = SyncManager(clock=_fixed_now) assert await restarted.start( - _settings(), repository, _spend_reader(), unexpected_completion, _transport(unexpected_details=True) + _settings(), + repository, + _spend_reader(), + unexpected_completion, + _transport(unexpected_details=True), + gateway_user_reader=_gateway_users, ) await _wait_until_finished(restarted) assert restarted.status.phase == "complete" @@ -525,16 +592,34 @@ async def test_expired_lease_can_restart_without_restarting_the_gateway() -> Non cancelled.set() assert await manager.start( - _settings(), repository, _spend_reader(), blocked_completion, _transport(), coordinator=coordinator + _settings(), + repository, + _spend_reader(), + blocked_completion, + _transport(), + coordinator=coordinator, + gateway_user_reader=_gateway_users, ) await entered.wait() assert not await manager.start( - _settings(), repository, _spend_reader(), _completion(), _transport(), coordinator=coordinator + _settings(), + repository, + _spend_reader(), + _completion(), + _transport(), + coordinator=coordinator, + gateway_user_reader=_gateway_users, ) assert coordinator.current is not None coordinator.current = coordinator.current.model_copy(update={"running": False, "phase": "error"}) assert await manager.start( - _settings(), repository, _spend_reader(), _completion(), _transport(), coordinator=coordinator + _settings(), + repository, + _spend_reader(), + _completion(), + _transport(), + coordinator=coordinator, + gateway_user_reader=_gateway_users, ) await _wait_until_finished(manager) assert cancelled.is_set() @@ -558,7 +643,14 @@ async def test_one_unreadable_pr_preserves_other_estimates_in_report() -> None: repository: Final = _ReportRepository() manager: Final = SyncManager(clock=_fixed_now) - assert await manager.start(_settings(), repository, _spend_reader(), _completion(), httpx.MockTransport(respond)) + assert await manager.start( + _settings(), + repository, + _spend_reader(), + _completion(), + httpx.MockTransport(respond), + gateway_user_reader=_gateway_users, + ) await _wait_until_finished(manager) report: Final = TypeAdapter(ROIReport).validate_python(repository.values["roi_calculator_report"]) assert tuple((pull["number"], pull["estimate"]["status"]) for pull in report["pulls"]) == ( @@ -597,7 +689,12 @@ async def test_unavailable_repository_publishes_flagged_partial_report_and_recov settings: Final = _settings().model_copy(update=MappingProxyType({"repos": ("org/repo", "org/unavailable")})) assert await manager.start( - settings, repository, _spend_reader(), _completion(), _repository_outage_transport(status) + settings, + repository, + _spend_reader(), + _completion(), + _repository_outage_transport(status), + gateway_user_reader=_gateway_users, ) await _wait_until_finished(manager) @@ -617,7 +714,12 @@ async def test_unavailable_repository_publishes_flagged_partial_report_and_recov raise AssertionError("The healthy repository's estimate must be reused after recovery") assert await manager.start( - settings, repository, _spend_reader(), unexpected_completion, _repository_outage_transport(200) + settings, + repository, + _spend_reader(), + unexpected_completion, + _repository_outage_transport(200), + gateway_user_reader=_gateway_users, ) await _wait_until_finished(manager) recovered: Final = TypeAdapter(ROIReport).validate_python(repository.values["roi_calculator_report"]) @@ -633,7 +735,14 @@ async def test_repository_outage_without_usable_pulls_preserves_previous_report( repository: Final = _ReportRepository() manager: Final = SyncManager(clock=_fixed_now) settings: Final = _settings().model_copy(update=MappingProxyType({"repos": ("org/repo", "org/unavailable")})) - assert await manager.start(settings, repository, _spend_reader(), _completion(), _repository_outage_transport(200)) + assert await manager.start( + settings, + repository, + _spend_reader(), + _completion(), + _repository_outage_transport(200), + gateway_user_reader=_gateway_users, + ) await _wait_until_finished(manager) previous: Final = repository.values["roi_calculator_report"] @@ -643,6 +752,7 @@ async def test_repository_outage_without_usable_pulls_preserves_previous_report( _spend_reader(), _completion(), _repository_outage_transport(403, all_unavailable=all_unavailable, healthy_empty=not all_unavailable), + gateway_user_reader=_gateway_users, ) await _wait_until_finished(manager) assert manager.status.phase == "error" @@ -662,7 +772,14 @@ async def test_reused_profile_preserves_email_only_when_lookup_fails(profile_sta return httpx.Response(200, content=_COMMITS_JSON.replace("alice@example.com", "")) return baseline.handle_request(request) - assert await manager.start(_settings(), repository, _spend_reader(), _completion(), httpx.MockTransport(respond)) + assert await manager.start( + _settings(), + repository, + _spend_reader(), + _completion(), + httpx.MockTransport(respond), + gateway_user_reader=_gateway_users, + ) await _wait_until_finished(manager) def refreshed(request: httpx.Request) -> httpx.Response: @@ -674,13 +791,18 @@ async def test_reused_profile_preserves_email_only_when_lookup_fails(profile_sta raise AssertionError("A reused estimate must not call the estimator") assert await manager.start( - _settings(), repository, _spend_reader(), unexpected_completion, httpx.MockTransport(refreshed) + _settings(), + repository, + _spend_reader(), + unexpected_completion, + httpx.MockTransport(refreshed), + gateway_user_reader=_gateway_users, ) await _wait_until_finished(manager) report: Final = TypeAdapter(ROIReport).validate_python(repository.values["roi_calculator_report"]) expected: Final = "" if profile_status == 200 else "alice@example.com" assert manager.status.phase == "complete" - assert manager.status.reused == 1 + assert manager.status.reused == (0 if profile_status == 200 else 1) assert report["pulls"][0]["profile_email"] == expected assert report["pulls"][0]["emails"] == ((expected,) if expected else ()) assert summarize(report, MappingProxyType({}))["metrics"]["cost_per_hour"] == (None if profile_status == 200 else 3) @@ -695,7 +817,12 @@ async def test_reused_profile_preserves_email_only_when_lookup_fails(profile_sta restarted: Final = SyncManager(clock=_fixed_now) assert await restarted.start( - _settings(), repository, _spend_reader(), unexpected_completion, httpx.MockTransport(unavailable_profile) + _settings(), + repository, + _spend_reader(), + unexpected_completion, + httpx.MockTransport(unavailable_profile), + gateway_user_reader=_gateway_users, ) await _wait_until_finished(restarted) subsequent: Final = TypeAdapter(ROIReport).validate_python(repository.values["roi_calculator_report"]) @@ -708,7 +835,9 @@ async def test_reused_profile_preserves_email_only_when_lookup_fails(profile_sta async def test_complete_estimator_outage_preserves_report_and_recovers() -> None: repository: Final = _ReportRepository() manager: Final = SyncManager(clock=_fixed_now) - assert await manager.start(_settings(), repository, _spend_reader(), _completion(), _transport()) + assert await manager.start( + _settings(), repository, _spend_reader(), _completion(), _transport(), gateway_user_reader=_gateway_users + ) await _wait_until_finished(manager) previous: Final = repository.values["roi_calculator_report"] changed: Final = _settings(estimator_prompt="Updated estimation instructions") @@ -716,13 +845,213 @@ async def test_complete_estimator_outage_preserves_report_and_recovers() -> None async def failed_completion(request: ROICompletionRequest) -> object: raise httpx.ConnectError("Estimator unavailable") - assert await manager.start(changed, repository, _spend_reader(), failed_completion, _transport()) + assert await manager.start( + changed, repository, _spend_reader(), failed_completion, _transport(), gateway_user_reader=_gateway_users + ) await _wait_until_finished(manager) assert manager.status.phase == "error" assert manager.status.error is not None and "No new report was published" in manager.status.error assert repository.values["roi_calculator_report"] == previous - assert await manager.start(changed, repository, _spend_reader(), _completion(), _transport()) + assert await manager.start( + changed, repository, _spend_reader(), _completion(), _transport(), gateway_user_reader=_gateway_users + ) await _wait_until_finished(manager) assert manager.status.phase == "complete" recovered: Final = TypeAdapter(ROIReport).validate_python(repository.values["roi_calculator_report"]) assert recovered["pulls"][0]["estimate"]["hours"] == 4 + + +class _CompletionRecorder: + def __init__(self) -> None: + self.requests: tuple[ROICompletionRequest, ...] = () + + async def __call__(self, request: ROICompletionRequest) -> object: + self.requests = (*self.requests, request) + return await _completion()(request) + + +@pytest.mark.asyncio +@pytest.mark.parametrize( + ("registered", "mapping", "expected_calls"), + ( + (frozenset(), MappingProxyType({}), 0), + (frozenset({"alice@example.com"}), MappingProxyType({}), 1), + (frozenset({"other@example.com"}), MappingProxyType({}), 0), + (frozenset({"other@example.com"}), MappingProxyType({"alice": "other@example.com"}), 1), + (frozenset({"alice@example.com"}), MappingProxyType({"alice": "outside@example.com"}), 0), + (frozenset({"alice@example.com", "profile@example.com"}), MappingProxyType({}), 0), + ), +) +async def test_only_authors_linked_to_registered_gateway_users_trigger_estimation( + registered: frozenset[str], mapping: Mapping[str, str], expected_calls: int +) -> None: + repository: Final = _ReportRepository() + manager: Final = SyncManager(clock=_fixed_now) + recorder: Final = _CompletionRecorder() + settings: Final = _settings().model_copy(update={"identity_map": mapping}) + + async def users() -> frozenset[str]: + return registered + + assert await manager.start( + settings, + repository, + _spend_reader(), + recorder, + _transport(profile_email="profile@example.com"), + gateway_user_reader=users, + ) + await _wait_until_finished(manager) + + report: Final = TypeAdapter(ROIReport).validate_python(repository.values["roi_calculator_report"]) + estimate: Final = report["pulls"][0]["estimate"] + assert manager.status.phase == "complete" + assert len(recorder.requests) == expected_calls + assert repository.pull_writes == expected_calls + assert estimate["status"] == ("estimated" if expected_calls else "needs_review") + assert estimate["hours"] == (4 if expected_calls else None) + + +@pytest.mark.asyncio +async def test_registered_author_without_spend_is_estimated() -> None: + repository: Final = _ReportRepository() + manager: Final = SyncManager(clock=_fixed_now) + recorder: Final = _CompletionRecorder() + + async def no_spend(start: date, end: date) -> tuple[ROISpendRecord, ...]: + return () + + assert await manager.start( + _settings(), repository, no_spend, recorder, _transport(), gateway_user_reader=_gateway_users + ) + await _wait_until_finished(manager) + report: Final = TypeAdapter(ROIReport).validate_python(repository.values["roi_calculator_report"]) + assert len(recorder.requests) == 1 + assert report["pulls"][0]["estimate"]["hours"] == 4 + assert report["spend"] == () + + +@pytest.mark.asyncio +async def test_unlinked_author_is_estimated_after_linking_and_cached_estimate_is_hidden_after_unlinking() -> None: + repository: Final = _ReportRepository() + manager: Final = SyncManager(clock=_fixed_now) + recorder: Final = _CompletionRecorder() + + async def users() -> frozenset[str]: + return frozenset({"member@example.com"}) + + async def run(settings: ROISettings) -> ROIReport: + assert await manager.start( + settings, repository, _spend_reader(), recorder, _transport(), gateway_user_reader=users + ) + await _wait_until_finished(manager) + assert manager.status.phase == "complete" + return TypeAdapter(ROIReport).validate_python(repository.values["roi_calculator_report"]) + + unlinked: Final = await run(_settings()) + assert unlinked["pulls"][0]["estimate"]["hours"] is None + assert len(recorder.requests) == 0 + linked_settings: Final = _settings().model_copy(update={"identity_map": {"alice": "member@example.com"}}) + linked: Final = await run(linked_settings) + assert linked["pulls"][0]["estimate"]["hours"] == 4 + assert len(recorder.requests) == 1 + unlinked_again: Final = await run(_settings()) + assert unlinked_again["pulls"][0]["estimate"]["hours"] is None + assert manager.status.reused == 0 + assert len(recorder.requests) == 1 + relinked: Final = await run(linked_settings) + assert relinked["pulls"][0]["estimate"]["hours"] == 4 + assert manager.status.reused == 1 + assert len(recorder.requests) == 1 + + +@pytest.mark.asyncio +async def test_unavailable_gateway_directory_stops_estimation_and_preserves_report() -> None: + repository: Final = _ReportRepository() + manager: Final = SyncManager(clock=_fixed_now) + recorder: Final = _CompletionRecorder() + assert await manager.start( + _settings(), repository, _spend_reader(), recorder, _transport(), gateway_user_reader=_gateway_users + ) + await _wait_until_finished(manager) + previous: Final = repository.values["roi_calculator_report"] + + async def unavailable_users() -> frozenset[str]: + raise ConnectionError("Gateway directory unavailable") + + assert await manager.start( + _settings("Changed prompt"), + repository, + _spend_reader(), + recorder, + _transport(), + gateway_user_reader=unavailable_users, + ) + await _wait_until_finished(manager) + assert manager.status.phase == "error" + assert len(recorder.requests) == 1 + assert repository.values["roi_calculator_report"] == previous + + +@pytest.mark.asyncio +async def test_gateway_directory_includes_users_without_spend_and_normalizes_emails() -> None: + assert await read_gateway_user_emails(_SpendPrismaClient()) == frozenset( + {"alice@example.com", "inactive@example.com"} + ) + + +@pytest.mark.asyncio +@pytest.mark.parametrize("size", (1000, 2501)) +async def test_gateway_directory_reads_every_page(size: int) -> None: + directory: Final = tuple( + {"user_id": f"user-{index:04d}", "user_email": f" Member-{index}@Example.com "} for index in range(size) + ) + client: Final = _SpendPrismaClient(directory) + assert await read_gateway_user_emails(client) == frozenset(f"member-{index}@example.com" for index in range(size)) + assert client.db.pages_read == size // 1000 + 1 + + +@pytest.mark.asyncio +async def test_unlinked_results_survive_when_the_only_linked_estimate_fails() -> None: + repository: Final = _ReportRepository() + manager: Final = SyncManager(clock=_fixed_now) + baseline: Final = _transport() + + def respond(request: httpx.Request) -> httpx.Response: + if request.url.path == "/repos/org/repo/pulls": + return httpx.Response( + 200, + content=_PULL_LIST_JSON[:-1] + + "," + + _PULL_LIST_JSON[1:].replace("42", "43").replace("alice", "outsider"), + ) + if request.url.path.startswith("/repos/org/repo/pulls/43"): + original: Final = baseline.handle_request(httpx.Request("GET", str(request.url).replace("/43", "/42"))) + return httpx.Response( + original.status_code, content=original.text.replace("42", "43").replace("alice", "outsider") + ) + if request.url.path == "/users/outsider": + return httpx.Response(200, json={"email": "outsider@example.com"}) + return baseline.handle_request(request) + + async def failed_completion(request: ROICompletionRequest) -> object: + raise httpx.ConnectError("Estimator unavailable") + + assert await manager.start( + _settings(), + repository, + _spend_reader(), + failed_completion, + httpx.MockTransport(respond), + gateway_user_reader=_gateway_users, + ) + await _wait_until_finished(manager) + assert manager.status.phase == "complete", manager.status.error + report: Final = TypeAdapter(ROIReport).validate_python(repository.values["roi_calculator_report"]) + assert tuple( + (pull["login"], pull["estimate"]["status"], pull["estimate"]["hours"]) for pull in report["pulls"] + ) == ( + ("alice", "error", None), + ("outsider", "needs_review", None), + ) + assert "not linked" in report["pulls"][1]["estimate"]["reasoning"] diff --git a/tests/unit/proxy/spend_tracking/test_spend_management_endpoints.py b/tests/unit/proxy/spend_tracking/test_spend_management_endpoints.py index 6aaa536a945..c27ad7ba0bf 100644 --- a/tests/unit/proxy/spend_tracking/test_spend_management_endpoints.py +++ b/tests/unit/proxy/spend_tracking/test_spend_management_endpoints.py @@ -3722,7 +3722,7 @@ class TestSpendLogsPayload: "model": "gpt-4o", "user": "", "team_id": "", - "metadata": '{"actor_agent_id": null, "target_agent_id": null, "billing_agent_id": null, "agent_execution_mode": null, "verified_human_user_id": null, "applied_guardrails": [], "attempted_fallbacks": null, "original_model_group": null, "batch_models": null, "batch_successful_requests": null, "batch_failed_requests": null, "mcp_tool_call_metadata": null, "vector_store_request_metadata": null, "routing_decision": null, "internal_call_origin": null, "guardrail_information": null, "compression_savings": null, "litellm_gateway_injected_cache": null, "router_metadata": null, "autorouter_savings_estimate": null, "autorouter_baseline_observation": null, "azure_spillover": null, "used_client_oauth_token": null, "usage_object": {"completion_tokens": 20, "prompt_tokens": 10, "total_tokens": 30, "completion_tokens_details": null, "prompt_tokens_details": null}, "model_map_information": {"model_map_key": "gpt-4o", "model_map_value": {"key": "gpt-4o", "max_tokens": 16384, "max_input_tokens": 128000, "max_output_tokens": 16384, "input_cost_per_token": 2.5e-06, "cache_creation_input_token_cost": null, "cache_read_input_token_cost": 1.25e-06, "input_cost_per_character": null, "input_cost_per_token_above_128k_tokens": null, "input_cost_per_token_above_200k_tokens": null, "input_cost_per_query": null, "input_cost_per_second": null, "input_cost_per_audio_token": null, "input_cost_per_token_batches": 1.25e-06, "output_cost_per_token_batches": 5e-06, "output_cost_per_token": 1e-05, "output_cost_per_audio_token": null, "output_cost_per_character": null, "output_cost_per_token_above_128k_tokens": null, "output_cost_per_character_above_128k_tokens": null, "output_cost_per_token_above_200k_tokens": null, "output_cost_per_second": null, "output_cost_per_reasoning_token": null, "output_cost_per_image": null, "output_vector_size": null, "litellm_provider": "openai", "mode": "chat", "supports_system_messages": true, "supports_response_schema": true, "supports_vision": true, "supports_function_calling": true, "supports_tool_choice": true, "supports_assistant_prefill": false, "supports_prompt_caching": true, "supports_audio_input": false, "supports_audio_output": false, "supports_pdf_input": false, "supports_embedding_image_input": false, "supports_native_streaming": null, "supports_web_search": true, "supports_reasoning": false, "search_context_cost_per_query": {"search_context_size_low": 0.03, "search_context_size_medium": 0.035, "search_context_size_high": 0.05}, "tpm": null, "rpm": null, "supported_openai_params": ["frequency_penalty", "logit_bias", "logprobs", "top_logprobs", "max_tokens", "max_completion_tokens", "modalities", "prediction", "n", "presence_penalty", "seed", "stop", "stream", "stream_options", "temperature", "top_p", "tools", "tool_choice", "function_call", "functions", "max_retries", "extra_headers", "parallel_tool_calls", "audio", "response_format", "user"]}}, "additional_usage_values": {"completion_tokens_details": null, "prompt_tokens_details": null}}', + "metadata": '{"actor_agent_id": null, "target_agent_id": null, "billing_agent_id": null, "agent_execution_mode": null, "verified_human_user_id": null, "applied_guardrails": [], "attempted_fallbacks": null, "original_model_group": null, "batch_models": null, "batch_successful_requests": null, "batch_failed_requests": null, "mcp_tool_call_metadata": null, "vector_store_request_metadata": null, "routing_decision": null, "internal_call_origin": null, "guardrail_information": null, "compression_savings": null, "litellm_gateway_injected_cache": null, "router_metadata": null, "autorouter_savings_estimate": null, "autorouter_baseline_observation": null, "azure_spillover": null, "used_client_oauth_token": null, "litellm_roi_estimator": false, "usage_object": {"completion_tokens": 20, "prompt_tokens": 10, "total_tokens": 30, "completion_tokens_details": null, "prompt_tokens_details": null}, "model_map_information": {"model_map_key": "gpt-4o", "model_map_value": {"key": "gpt-4o", "max_tokens": 16384, "max_input_tokens": 128000, "max_output_tokens": 16384, "input_cost_per_token": 2.5e-06, "cache_creation_input_token_cost": null, "cache_read_input_token_cost": 1.25e-06, "input_cost_per_character": null, "input_cost_per_token_above_128k_tokens": null, "input_cost_per_token_above_200k_tokens": null, "input_cost_per_query": null, "input_cost_per_second": null, "input_cost_per_audio_token": null, "input_cost_per_token_batches": 1.25e-06, "output_cost_per_token_batches": 5e-06, "output_cost_per_token": 1e-05, "output_cost_per_audio_token": null, "output_cost_per_character": null, "output_cost_per_token_above_128k_tokens": null, "output_cost_per_character_above_128k_tokens": null, "output_cost_per_token_above_200k_tokens": null, "output_cost_per_second": null, "output_cost_per_reasoning_token": null, "output_cost_per_image": null, "output_vector_size": null, "litellm_provider": "openai", "mode": "chat", "supports_system_messages": true, "supports_response_schema": true, "supports_vision": true, "supports_function_calling": true, "supports_tool_choice": true, "supports_assistant_prefill": false, "supports_prompt_caching": true, "supports_audio_input": false, "supports_audio_output": false, "supports_pdf_input": false, "supports_embedding_image_input": false, "supports_native_streaming": null, "supports_web_search": true, "supports_reasoning": false, "search_context_cost_per_query": {"search_context_size_low": 0.03, "search_context_size_medium": 0.035, "search_context_size_high": 0.05}, "tpm": null, "rpm": null, "supported_openai_params": ["frequency_penalty", "logit_bias", "logprobs", "top_logprobs", "max_tokens", "max_completion_tokens", "modalities", "prediction", "n", "presence_penalty", "seed", "stop", "stream", "stream_options", "temperature", "top_p", "tools", "tool_choice", "function_call", "functions", "max_retries", "extra_headers", "parallel_tool_calls", "audio", "response_format", "user"]}}, "additional_usage_values": {"completion_tokens_details": null, "prompt_tokens_details": null}}', "cache_key": "Cache OFF", "spend": 0.00022500000000000002, "total_tokens": 30, diff --git a/ui/litellm-dashboard/AGENTS.md b/ui/litellm-dashboard/AGENTS.md index 7b1234e1cf3..e5d876fad84 100644 --- a/ui/litellm-dashboard/AGENTS.md +++ b/ui/litellm-dashboard/AGENTS.md @@ -25,3 +25,13 @@ Rules beyond the enabled set were measured against the whole suite and left off Never run the full unit suite (`npx vitest run` with no path). It is 380 files and thousands of tests, it saturates the machine for many minutes, and CI runs it anyway. Run only the test files your change touches, plus any file whose failure your change could plausibly explain, by passing explicit paths Type tests are `*.test-d.ts` files run by the `types` vitest project (`npm run test:types`). Keep them out of the `src/app/(dashboard)/` route group. Vitest matches a tsc error back to the test file by path, the parentheses break that match, and `ignoreSourceErrors: true` then drops the error as if it came from a source file. The test still collects and still reports as passing, so a `.test-d.ts` under a parenthesized directory is green no matter what it asserts. Confirm any new one has teeth by breaking the type it guards and watching it fail + + + +# This is NOT the Next.js you know + +This version has breaking changes — APIs, conventions, and file structure may all differ from your training data. Read the relevant guide in `node_modules/next/dist/docs/` (resolved from this file's directory; in monorepos the `next` package may not be visible from the repo root) before writing any code. Heed deprecation notices. + +This block is written and re-added by `next dev` — verify at `node_modules/next/dist/server/lib/generate-agent-files.js`. Removing it from a diff only re-creates the uncommitted change; committing it with your work keeps the tree clean. + + diff --git a/ui/litellm-dashboard/eslint-suppressions.json b/ui/litellm-dashboard/eslint-suppressions.json index 2465d07129c..12b99d0ce73 100644 --- a/ui/litellm-dashboard/eslint-suppressions.json +++ b/ui/litellm-dashboard/eslint-suppressions.json @@ -2307,11 +2307,6 @@ "count": 1 } }, - "src/components/view_logs/index.tsx": { - "local/filename-pascal-case": { - "count": 1 - } - }, "src/components/view_logs/log_filter_logic.tsx": { "local/filename-pascal-case": { "count": 1 diff --git a/ui/litellm-dashboard/src/app/(dashboard)/lens/_components/lensDemoData.test.ts b/ui/litellm-dashboard/src/app/(dashboard)/lens/_components/lensDemoData.test.ts index eabae2928bd..6615b6abcba 100644 --- a/ui/litellm-dashboard/src/app/(dashboard)/lens/_components/lensDemoData.test.ts +++ b/ui/litellm-dashboard/src/app/(dashboard)/lens/_components/lensDemoData.test.ts @@ -38,6 +38,21 @@ describe("Lens demo data", () => { } }); + it("includes a long release review with unique steps, complete details and three failed checks", () => { + const run = createLensDemoData().runs[6]; + const ids = new Set(run.trace.spans.map((span) => span.span_id)); + expect(run.trace.spans).toHaveLength(362); + expect(ids.size).toBe(362); + expect(run.trace.summary.error_count).toBe(3); + expect(run.trace.summary.status).toBe("ok"); + for (const span of run.trace.spans) { + if (span.parent_span_id) expect(ids.has(span.parent_span_id)).toBe(true); + expect(run.details.find((detail) => detail.span_id === span.span_id)).toBeDefined(); + } + expect(JSON.parse(run.details.at(-1)!.input)).toHaveLength(120); + expect(run.details.at(-1)!.output).toContain("Hold the release"); + }); + it("filters time windows locally and rejects writes or unknown reads without network access", async () => { const network = vi.spyOn(globalThis, "fetch"); const now = Date.now(); diff --git a/ui/litellm-dashboard/src/app/(dashboard)/lens/_components/lensDemoData.ts b/ui/litellm-dashboard/src/app/(dashboard)/lens/_components/lensDemoData.ts index 8b4e1645c9e..8ca2f674b40 100644 --- a/ui/litellm-dashboard/src/app/(dashboard)/lens/_components/lensDemoData.ts +++ b/ui/litellm-dashboard/src/app/(dashboard)/lens/_components/lensDemoData.ts @@ -2,6 +2,7 @@ import { createApiClient } from "@/lib/http/client"; import type { LensDemo } from "@/components/lens/LensDemoContext"; import type { Trace, Span, SpanDetail } from "@/components/view_logs/TraceView/traceTypes"; import type { Lens, Finding, Job, Settings } from "./lensData"; +import { withReleaseCases } from "./lensDemoLongTrace"; type Scenario = { agent: string; @@ -61,8 +62,8 @@ const scenarios: Scenario[] = [ agent: "release_agent", question: "Review the search release", tool: "read_test_results", - result: "Search: 86 passed, 1 failed. Unicode query regression remains open.", - answer: "Hold the release. The Unicode query regression is still failing.", + result: "Search: 117 passed, 3 failed. Cases 17, 63 and 104 returned empty results.", + answer: "Hold the release. Three of 120 cases returned empty results. Review cases 17, 63 and 104 before shipping.", }, { agent: "support_agent", @@ -206,7 +207,10 @@ function makeTrace(scene: Scenario, index: number, now: number) { } export function createLensDemoData(now = Date.now()) { - const runs = scenarios.map((scene, index) => makeTrace(scene, index, now)); + const runs = scenarios.map((scene, index) => { + const run = makeTrace(scene, index, now); + return index === 6 ? withReleaseCases(run) : run; + }); const finding = ({ id, check, diff --git a/ui/litellm-dashboard/src/app/(dashboard)/lens/_components/lensDemoLongTrace.ts b/ui/litellm-dashboard/src/app/(dashboard)/lens/_components/lensDemoLongTrace.ts new file mode 100644 index 00000000000..e30cff6ee1d --- /dev/null +++ b/ui/litellm-dashboard/src/app/(dashboard)/lens/_components/lensDemoLongTrace.ts @@ -0,0 +1,119 @@ +import type { Span, SpanDetail, Trace } from "@/components/view_logs/TraceView/traceTypes"; + +export function withReleaseCases(run: { trace: Trace; details: SpanDetail[] }) { + const { trace } = run; + const root = trace.spans[0]; + const final = trace.spans.at(-1)!; + const caseCount = 120; + const caseSpans: Span[] = []; + const caseDetails: SpanDetail[] = []; + const checks = ["Unicode queries", "Empty results", "Pagination", "Ranking", "Filters", "Permissions"]; + const failedCases = new Set([17, 63, 104]); + for (let index = 1; index <= caseCount; index++) { + const failed = failedCases.has(index); + const id = (step: number) => (0x70000 + index * 10 + step).toString(16).padStart(16, "0"); + const question = `Case ${index}: ${checks[(index - 1) % checks.length]}`; + const result = failed ? "Expected matching results; received an empty result set" : "Expected results matched"; + const start = (index - 1) * 2200; + const agent: Span = { + ...root, + span_id: id(0), + parent_span_id: root.span_id, + agent: "search_case", + name: "search_case", + input_preview: question, + start_offset_ms: start, + duration_ms: 2100, + }; + const tool: Span = { + ...agent, + span_id: id(1), + parent_span_id: agent.span_id, + name: "run_search_check", + type: "tool", + start_offset_ms: start + 100, + duration_ms: 800, + status: failed ? "error" : "ok", + error: failed ? result : null, + }; + const model: Span = { + ...final, + span_id: id(2), + parent_span_id: agent.span_id, + agent: agent.agent, + name: "Review case result", + input_preview: question, + start_offset_ms: start + 950, + duration_ms: 1100, + }; + caseSpans.push(agent, tool, model); + for (const span of [agent, tool, model]) { + const detail: SpanDetail = { + span_id: span.span_id, + input: + span.type === "tool" + ? JSON.stringify({ case: index, check: question }) + : JSON.stringify([{ role: "user", content: question }]), + output: + span.type === "tool" + ? result + : JSON.stringify([ + { + role: "assistant", + content: failed ? `Hold this case for review. ${result}.` : `Case ${index} passed. ${result}.`, + }, + ]), + attributes: { "gen_ai.agent.name": "search_case", "test.case": String(index), demo: "true" }, + }; + caseDetails.push(detail); + } + } + const finalSpan = { ...final, start_offset_ms: caseCount * 2200 }; + const duration = finalSpan.start_offset_ms + finalSpan.duration_ms; + const spans = [{ ...root, duration_ms: duration }, ...caseSpans, finalSpan]; + const models = spans.filter((span) => span.type === "llm"); + const summary = { + ...trace.summary, + duration_ms: duration, + span_count: spans.length, + agent_count: 2, + agent_invocations: caseCount + 1, + agent_names: [root.name, "search_case"], + llm_calls: models.length, + tool_calls: caseCount, + error_count: failedCases.size, + input_tokens: models.reduce((sum, span) => sum + span.input_tokens, 0), + output_tokens: models.reduce((sum, span) => sum + span.output_tokens, 0), + spend: models.reduce((sum, span) => sum + (span.spend ?? 0), 0), + }; + const finalDetail = run.details.at(-1)!; + const history = caseDetails.filter((_, index) => index % 3 === 2); + return { + trace: { + summary, + spans, + agents: [ + { ...trace.agents[0], duration_ms: duration, tool_calls: 0 }, + { + name: "search_case", + parent_agent: root.name, + duration_ms: caseCount * 2100, + invocations: caseCount, + llm_calls: caseCount, + tool_calls: caseCount, + spend: summary.spend - (final.spend ?? 0), + }, + ], + }, + details: [ + run.details[0], + ...caseDetails, + { + ...finalDetail, + input: JSON.stringify( + history.map((detail) => ({ role: "user", content: JSON.parse(detail.output)[0].content })), + ), + }, + ], + }; +} diff --git a/ui/litellm-dashboard/src/app/(dashboard)/roi-calculator/_components/ROICalculatorView.integration.test.tsx b/ui/litellm-dashboard/src/app/(dashboard)/roi-calculator/_components/ROICalculatorView.integration.test.tsx index 8cd3e47f296..00bcbfbf483 100644 --- a/ui/litellm-dashboard/src/app/(dashboard)/roi-calculator/_components/ROICalculatorView.integration.test.tsx +++ b/ui/litellm-dashboard/src/app/(dashboard)/roi-calculator/_components/ROICalculatorView.integration.test.tsx @@ -1,3 +1,4 @@ +import userEvent from "@testing-library/user-event"; import { act, fireEvent, render, screen, waitFor } from "@testing-library/react"; import { beforeEach, describe, expect, it, vi } from "vitest"; @@ -283,6 +284,39 @@ describe("ROICalculatorView", () => { expect(screen.getByRole("textbox", { name: "Repository name" })).toBeEnabled(); }); + it("searches by the real model name and saves the selected gateway alias", async () => { + const user = userEvent.setup(); + const modelSettings = { + ...settings, + available_models: ["estimator", "fast-estimator"], + estimator_models: [ + { model_name: "estimator", provider_models: ["custom-model"] }, + { model_name: "fast-estimator", provider_models: ["openai/gpt-6-luna"] }, + ], + }; + vi.mocked(apiClient.get).mockImplementation((path: string) => { + if (path === "/roi-calculator/settings") return Promise.resolve(modelSettings); + if (path === "/roi-calculator/report") return Promise.resolve({ report: summary }); + return Promise.resolve(idleStatus); + }); + vi.mocked(apiClient.put).mockResolvedValue({ ...modelSettings, estimator_model: "fast-estimator" }); + render(); + await user.click(await screen.findByRole("button", { name: "Settings" })); + const search = screen.getByRole("combobox", { name: "Estimator model" }); + await user.clear(search); + await user.type(search, "Luna"); + expect(screen.queryByRole("option", { name: /custom-model/ })).not.toBeInTheDocument(); + await user.click(await screen.findByRole("option", { name: /GPT-6 Luna.*Recommended/ })); + expect(search).toHaveValue("GPT-6 Luna"); + await user.click(screen.getByRole("button", { name: "Save settings" })); + expect(apiClient.put).toHaveBeenCalledWith( + "/roi-calculator/settings", + expect.objectContaining({ + body: expect.objectContaining({ estimator_model: "fast-estimator" }), + }), + ); + }); + it("closes the old settings dialog when saving a different source", async () => { const gitlabSettings = { ...settings, diff --git a/ui/litellm-dashboard/src/app/(dashboard)/roi-calculator/_components/ROISettingsPanel.tsx b/ui/litellm-dashboard/src/app/(dashboard)/roi-calculator/_components/ROISettingsPanel.tsx index 540ac229de8..535655ed574 100644 --- a/ui/litellm-dashboard/src/app/(dashboard)/roi-calculator/_components/ROISettingsPanel.tsx +++ b/ui/litellm-dashboard/src/app/(dashboard)/roi-calculator/_components/ROISettingsPanel.tsx @@ -2,6 +2,9 @@ import React from "react"; +import { SearchSelect } from "@/components/shared/SearchSelect"; +import { estimatorModelOptions } from "./roiCalculatorData"; + import { apiClient } from "@/components/networking"; import { extractErrorMessage } from "@/utils/errorUtils"; import { Button } from "@/components/ui/button"; @@ -408,23 +411,19 @@ export default function ROISettingsPanel({ {!onboarding &&

Estimation and updates

}
- + onValueChange={(value) => setModel(value ?? "")} + placeholder="Search estimator models" + emptyText="No matching models configured on this gateway" + disabled={readOnly} + className="h-9" + /> +

+ We recommend GPT-6 Luna for estimating PR effort. Choose a model configured on your gateway. +

Advanced estimator options diff --git a/ui/litellm-dashboard/src/app/(dashboard)/roi-calculator/_components/roiCalculatorData.test.ts b/ui/litellm-dashboard/src/app/(dashboard)/roi-calculator/_components/roiCalculatorData.test.ts index cbc7a35fd0f..f292d0c8809 100644 --- a/ui/litellm-dashboard/src/app/(dashboard)/roi-calculator/_components/roiCalculatorData.test.ts +++ b/ui/litellm-dashboard/src/app/(dashboard)/roi-calculator/_components/roiCalculatorData.test.ts @@ -4,6 +4,7 @@ import { coverageLabel, effortNote, estimateLabel, + estimatorModelOptions, filterPulls, formatMoney, formatNumber, @@ -44,9 +45,17 @@ const summary = { describe("ROI calculator display helpers", () => { it("ranks only attributed PR costs, limits the overview to five and preserves report order", () => { const pulls = [2, 6, 1, 4, 3, 5].map((spend) => - pull({ number: spend, branch_cost: { status: "matched", spend, requests: 1, cost_per_hour: spend } }), + pull({ + number: spend, + branch_cost: { status: "matched", spend, requests: 1, repo: "github.com/org/repo", branch: "feature" }, + }), + ); + pulls.push( + pull({ + number: 99, + branch_cost: { status: "ambiguous", spend: 99, requests: 1, repo: "github.com/org/repo", branch: "feature" }, + }), ); - pulls.push(pull({ number: 99, branch_cost: { status: "ambiguous", spend: 99, requests: 1, cost_per_hour: null } })); pulls.push(pull({ number: 100 })); expect(highestCostPulls(pulls).map((item) => item.number)).toEqual([6, 5, 4, 3, 2]); expect(pulls.map((item) => item.number)).toEqual([2, 6, 1, 4, 3, 5, 99, 100]); @@ -109,3 +118,33 @@ it("exports precise spend, cohort eligibility and safely quoted CSV values", () expect(csv).toContain('"\'=HYPERLINK(""bad"")","alice;bob","0.0001","4","1","0","true","0.000025"'); expect(csv).toContain('"2026-09-01","2026-09-30","without_ai"'); }); + +it("recommends the real Luna model and preserves its gateway name for requests", () => { + expect( + estimatorModelOptions({ + available_models: ["general", "fast-estimator"], + estimator_models: [ + { model_name: "general", provider_models: ["anthropic/claude-haiku"] }, + { model_name: "fast-estimator", provider_models: ["openai/gpt-6-luna"] }, + ], + }), + ).toEqual([ + { + value: "fast-estimator", + label: "GPT-6 Luna", + sublabel: "Recommended · Gateway name: fast-estimator", + recommended: true, + }, + { value: "general", label: "anthropic/claude-haiku", sublabel: "Gateway name: general", recommended: false }, + ]); +}); + +it("does not invent available models or recommend an alias pointing to a different model", () => { + expect( + estimatorModelOptions({ + available_models: ["gpt-6-luna"], + estimator_models: [{ model_name: "gpt-6-luna", provider_models: ["custom-model"] }], + }), + ).toEqual([{ value: "gpt-6-luna", label: "custom-model", sublabel: "Gateway name: gpt-6-luna", recommended: false }]); + expect(estimatorModelOptions({ available_models: [], estimator_models: [] })).toEqual([]); +}); diff --git a/ui/litellm-dashboard/src/app/(dashboard)/roi-calculator/_components/roiCalculatorData.ts b/ui/litellm-dashboard/src/app/(dashboard)/roi-calculator/_components/roiCalculatorData.ts index 11a254e3908..c8ea642c39a 100644 --- a/ui/litellm-dashboard/src/app/(dashboard)/roi-calculator/_components/roiCalculatorData.ts +++ b/ui/litellm-dashboard/src/app/(dashboard)/roi-calculator/_components/roiCalculatorData.ts @@ -121,3 +121,25 @@ export const branchCostLabel = (pull: ROIPull): string => { if (pull.branch_cost.status === "unattributed") return "No tagged requests"; return formatMoney(pull.branch_cost.spend); }; + +export const estimatorModelOptions = (settings: Pick) => { + const details = new Map(settings.estimator_models?.map((model) => [model.model_name, model])); + const isLuna = (name: string) => /(?:^|\/)gpt-6-luna(?:-\d{4}-\d{2}-\d{2})?$/i.test(name); + return settings.available_models + .map((name) => { + const models = details.get(name)?.provider_models ?? []; + const recommended = models.length > 0 && models.every(isLuna); + const label = [...new Set(models.map((model) => (isLuna(model) ? "GPT-6 Luna" : model)))].join(", ") || name; + return { + value: name, + label, + sublabel: [recommended ? "Recommended" : "", label !== name ? `Gateway name: ${name}` : ""] + .filter(Boolean) + .join(" · "), + recommended, + }; + }) + .sort( + (left, right) => Number(right.recommended) - Number(left.recommended) || left.label.localeCompare(right.label), + ); +}; diff --git a/ui/litellm-dashboard/src/components/view_logs/TraceView/CopyButton.tsx b/ui/litellm-dashboard/src/components/view_logs/TraceView/CopyButton.tsx index 60dda4ce72d..fecd817f7b6 100644 --- a/ui/litellm-dashboard/src/components/view_logs/TraceView/CopyButton.tsx +++ b/ui/litellm-dashboard/src/components/view_logs/TraceView/CopyButton.tsx @@ -38,11 +38,7 @@ export function CopyButton({ @@ -78,7 +76,7 @@ function PaneHeader({ function PaneFooter({ children }: { children: React.ReactNode }) { return ( -
+
{children}
); @@ -87,7 +85,7 @@ function PaneFooter({ children }: { children: React.ReactNode }) { function Meta({ label, value }: { label: string; value: string }) { return ( - {label} + {label} {value} ); @@ -116,7 +114,7 @@ function SpanPane({ const tokens = span.input_tokens + span.output_tokens; return (