From db14931401d4ad28d3f8119f21692e4e8ed8eaae Mon Sep 17 00:00:00 2001 From: tin-berri Date: Fri, 2 Oct 2026 20:56:45 -0700 Subject: [PATCH 01/11] fix(auto-router): attribute router-day savings without spend logs or a session (#44226) Per-router auto-router savings were missing on proxies with disable_spend_logs and for requests without a session id under missing_session_id: omit. The router-day row is written by the auto-router turn, whose enqueue sat behind the spend-logs flag and whose builder dropped sessionless requests, while LiteLLM_DailyUserSpend has neither gate. That money then showed only as unattributed savings. Enqueue the turn whether or not spend logs are kept, and keep a sessionless turn with an empty session id that writes the router-day row while the session upserts skip it. The day and session rows still commit in one statement, so late baseline corrections keep their ordering. Without spend logs, write only the router-day aggregate: the turn drops its session id, so no per-session row is stored, and baseline capture is skipped, since a baseline observation can only publish once its request's spend log exists. The flag keeps its meaning of no per-request or per-session data, while the daily per-router money matches LiteLLM_DailyUserSpend. Co-authored-by: Claude Opus 5.5 --- litellm/proxy/db/autorouter_session_rollup.py | 9 +- litellm/proxy/db/db_spend_update_writer.py | 24 +++- .../spend/test_autorouter_session_rollup.py | 103 ++++++++++++++++++ .../spend/test_baseline_accounting.py | 65 +++++++++-- .../db/test_autorouter_session_rollup.py | 7 +- .../proxy/db/test_db_spend_update_writer.py | 59 ++++++++++ 6 files changed, 246 insertions(+), 21 deletions(-) diff --git a/litellm/proxy/db/autorouter_session_rollup.py b/litellm/proxy/db/autorouter_session_rollup.py index cdc949a6dca..350e4da231c 100644 --- a/litellm/proxy/db/autorouter_session_rollup.py +++ b/litellm/proxy/db/autorouter_session_rollup.py @@ -266,7 +266,8 @@ def build_autorouter_turn_transaction( the payload's own usage record through the savings owner, never handed in beside it. The baseline the turn's saved_spend was priced against travels with the turn, so the row can name the counterfactual for the money it holds even after the router is - reconfigured or removed. + reconfigured or removed. A request with no session id still owns its router-day money, + so it becomes a turn with an empty session id that writes the day row and no session row. """ if payload.get("status") != "success": return None @@ -278,9 +279,9 @@ def build_autorouter_turn_transaction( router_name: Final = routing_decision.get("router_model_name") or payload.get("model_group") api_key: Final = payload.get("api_key") or "" user_id: Final = payload.get("user") or "" - session_id: Final = payload.get("session_id") + session_id: Final = payload.get("session_id") or "" model: Final = payload.get("model") - if not (isinstance(router_name, str) and router_name and (api_key or user_id) and session_id and model): + if not (isinstance(router_name, str) and router_name and (api_key or user_id) and model): return None turn_at: Final = _turn_time_utc(str(payload.get("startTime") or "")) if turn_at is None: @@ -379,7 +380,7 @@ SELECT {_p("classifier_cost")}::float8, 1, {_TIER_DELTA}, {_BASELINE_DELTA}, {_p("savings_estimated_turns")}::int, {_p("savings_estimated_actual_spend")}::float8, {_p("savings_estimated_saved_spend")}::float8, {_ESTIMATED_BASELINE_DELTA} -WHERE {required_identity}::text <> '' +WHERE {required_identity}::text <> '' AND {_p("session_id")}::text <> '' ON CONFLICT ({user_column}api_key, session_id, router_name) DO UPDATE SET turns = t.turns + 1, total_tokens = t.total_tokens + EXCLUDED.total_tokens, diff --git a/litellm/proxy/db/db_spend_update_writer.py b/litellm/proxy/db/db_spend_update_writer.py index c64ef72ace6..26a21069c83 100644 --- a/litellm/proxy/db/db_spend_update_writer.py +++ b/litellm/proxy/db/db_spend_update_writer.py @@ -7,6 +7,7 @@ Module responsible for import asyncio import copy +import dataclasses import json import os import random @@ -550,6 +551,13 @@ class DBSpendUpdateWriter: ): return False + # The auto-router router-day rollup is an aggregate like the daily spend tables, so it is + # written whether or not per-request spend logs are kept; per-session rows are not. + await self._enqueue_autorouter_turn_transaction( + payload=payload, + prisma_client=prisma_client, + spend_logs_kept=disable_spend_logs is False, + ) if disable_spend_logs is False: await self._enqueue_tool_usage_transaction( payload=payload, @@ -557,10 +565,6 @@ class DBSpendUpdateWriter: prisma_client=prisma_client, kwargs=kwargs, ) - await self._enqueue_autorouter_turn_transaction( - payload=payload, - prisma_client=prisma_client, - ) else: verbose_proxy_logger.debug( "disable_spend_logs=True. Skipping writing spend logs to db. Other spend updates - Key/User/Team table will still occur." @@ -747,6 +751,7 @@ class DBSpendUpdateWriter: self, payload: SpendLogsPayload, prisma_client: "PrismaClient | None", + spend_logs_kept: bool = True, ) -> None: try: if prisma_client is None: @@ -787,14 +792,21 @@ class DBSpendUpdateWriter: saved_spend=savings_spend.autorouter, ) try: - if await self._enqueue_baseline_accounting(payload, metadata, transaction, prisma_client): + # A baseline observation publishes only once its spend log exists, so without spend logs + # it could never publish; the plain turn still carries this request's recorded savings. + if spend_logs_kept and await self._enqueue_baseline_accounting( + payload, metadata, transaction, prisma_client + ): return except Exception: # noqa: BLE001 # optional baseline capture must preserve the original actual-spend rollup verbose_proxy_logger.warning("Auto-router baseline observation was unavailable; actual turn retained") if transaction is None: return + # Without spend logs only the router-day aggregate is kept: an empty session id makes the + # session upserts skip the row, so no per-session record is stored. + kept: Final = transaction if spend_logs_kept else dataclasses.replace(transaction, session_id="") async with prisma_client._autorouter_turn_transactions_lock: - prisma_client.autorouter_turn_transactions.append(transaction) + prisma_client.autorouter_turn_transactions.append(kept) except Exception as e: # noqa: BLE001 # a metrics enqueue must never fail the spend write verbose_proxy_logger.debug("_enqueue_autorouter_turn_transaction error (non-blocking): %s", e) diff --git a/tests/proxy_behavior/spend/test_autorouter_session_rollup.py b/tests/proxy_behavior/spend/test_autorouter_session_rollup.py index a5c6f5962a2..d2511ba257b 100644 --- a/tests/proxy_behavior/spend/test_autorouter_session_rollup.py +++ b/tests/proxy_behavior/spend/test_autorouter_session_rollup.py @@ -20,8 +20,10 @@ from typing_extensions import ReadOnly from litellm.proxy.db.autorouter_session_rollup import ( AUTOROUTER_BENCHMARKS_SQL, UPSERT_AUTOROUTER_SESSION_SQL, + UPSERT_AUTOROUTER_USER_SESSION_SQL, AutoRouterTurnTransaction, flush_autorouter_turn_transactions, + write_autorouter_turn, ) from litellm.proxy.db.db_transaction_queue.spend_log_cleanup import SpendLogCleanup @@ -684,3 +686,104 @@ async def test_a_router_type_change_mid_session_keeps_session_shape_with_the_ses assert (rows["complexity"]["sessions"], rows["complexity"]["session_turns"], rows["complexity"]["turns"]) == (1, 2, 1) assert (rows["quality"]["sessions"], rows["quality"]["session_turns"], rows["quality"]["turns"]) == (0, 0, 1) assert rows["quality"]["spend"] == 2.0 + + +@pytest.mark.parametrize("statement", [UPSERT_AUTOROUTER_SESSION_SQL, UPSERT_AUTOROUTER_USER_SESSION_SQL]) +async def test_a_sessionless_turn_writes_its_router_day_row_and_no_session_row(db, statement: str): + key = f"k-{uuid.uuid4()}" + router = f"auto-{uuid.uuid4()}" + for offset in range(2): + await write_autorouter_turn( + db, + AutoRouterTurnTransaction( + api_key=key, + user_id="u-sessionless", + session_id="", + router_name=router, + router_type="complexity", + model="A", + turn_at=T0 + timedelta(seconds=offset), + total_tokens=10, + spend=1.0, + saved_spend=2.0, + classifier_cost=0.1, + covered=True, + cache_hit=False, + cache_ttl_seconds=None, + cache_touched=True, + savings_estimated_turns=1, + savings_estimated_actual_spend=1.0, + savings_estimated_saved_spend=2.0, + ), + statement, + ) + + (day,) = await _days(db, key, router=router) + assert (day["turns"], day["spend"], day["saved_spend"], day["classifier_cost"]) == (2, 2.0, 4.0, 0.2) + assert (day["sessions"], day["session_turns"]) == (0, 0) + for table in ("LiteLLM_AutoRouterSession", "LiteLLM_AutoRouterUserSession"): + assert await db.query_raw(f'SELECT 1 FROM "{table}" WHERE router_name = $1', router) == [] + + +async def test_router_day_money_reconciles_with_the_overall_daily_total_including_sessionless_requests(db): + from litellm.proxy.db.daily_spend_bulk_upsert import DAILY_SPEND_TABLES, build_bulk_upsert, merge_by_conflict_key + + key = f"k-{uuid.uuid4()}" + router = f"auto-{uuid.uuid4()}" + requests = (("session-1", 0.25, 1.5), ("session-1", 0.5, 2.0), ("", 0.1, 0.25)) + for offset, (session_id, spend, saved) in enumerate(requests): + await write_autorouter_turn( + db, + AutoRouterTurnTransaction( + api_key=key, + user_id="u1", + session_id=session_id, + router_name=router, + router_type="complexity", + model="A", + turn_at=T0 + timedelta(seconds=offset), + total_tokens=10, + spend=spend, + saved_spend=saved, + classifier_cost=0.0, + covered=True, + cache_hit=False, + cache_ttl_seconds=None, + cache_touched=True, + savings_estimated_turns=1, + savings_estimated_actual_spend=spend, + savings_estimated_saved_spend=saved, + ), + ) + table = DAILY_SPEND_TABLES["user"] + statement, values = build_bulk_upsert( + table, + merge_by_conflict_key( + table, + tuple( + { + "user_id": "u1", + "date": T0.date().isoformat(), + "api_key": key, + "model": "A", + "custom_llm_provider": "anthropic", + "model_group": router, + "spend": spend, + "api_requests": 1, + "successful_requests": 1, + "autorouter_savings_spend": saved, + } + for _, spend, saved in requests + ), + ), + ) + await db.execute_raw(statement, *values) + + (overall,) = await db.query_raw( + 'SELECT SUM(autorouter_savings_spend)::float8 AS saved FROM "LiteLLM_DailyUserSpend" WHERE date = $1 AND api_key = $2', + T0.date().isoformat(), + key, + ) + (row,) = await _days(db, key, router=router) + assert overall["saved"] == row["saved_spend"] == pytest.approx(3.75) + assert (row["turns"], row["spend"], row["sessions"], row["session_turns"]) == (3, pytest.approx(0.85), 1, 2) diff --git a/tests/proxy_behavior/spend/test_baseline_accounting.py b/tests/proxy_behavior/spend/test_baseline_accounting.py index 8fb82d0c80e..dbaf32d579f 100644 --- a/tests/proxy_behavior/spend/test_baseline_accounting.py +++ b/tests/proxy_behavior/spend/test_baseline_accounting.py @@ -249,17 +249,10 @@ async def test_retired_history_never_recreates_an_initial_zero(db: Prisma, recor assert after["savings_estimated_turns"] == 1 and after["savings_estimated_actual_spend"] == 0.17 -async def test_native_observation_enters_spend_pipeline_once_with_shared_daily_attribution( - db: Prisma, record: Callable[..., BaselineAccountingRecord], monkeypatch: pytest.MonkeyPatch, -) -> None: - import os - - from litellm.proxy.common_utils.user_api_key_cache import UserApiKeyCache - from litellm.proxy.db.db_spend_update_writer import DBSpendUpdateWriter +def _native_observation_payload(event: BaselineAccountingRecord) -> dict[str, object]: + """The spend payload a captured, sessioned, auto-routed anthropic_messages request produces.""" from litellm.proxy.hooks.autorouter_baseline_cache import CapturedBaselineObservation - from litellm.proxy.utils import PrismaClient, ProxyLogging - event: Final = record("routed", identical=False) capture: Final = CapturedBaselineObservation( scope=event.scope, api_key=event.api_key, session_id=event.session_id, router_name=event.router_name, baseline_model=event.baseline_model, @@ -272,7 +265,7 @@ async def test_native_observation_enters_spend_pipeline_once_with_shared_daily_a "autorouter_savings": None, "autorouter_savings_estimate": {"version": 3, "status": "unknown", "reason": "pending_projection"}, "autorouter_baseline_observation": capture.model_dump_json(), } - payload: Final = { + return { "request_id": event.observation.request_id, "api_key": event.api_key, "session_id": event.session_id, "startTime": datetime.fromtimestamp(event.observation.started_at, timezone.utc).isoformat(), "endTime": datetime.fromtimestamp(event.observation.available_at, timezone.utc).isoformat(), @@ -282,6 +275,19 @@ async def test_native_observation_enters_spend_pipeline_once_with_shared_daily_a "user": None, "team_id": "", "organization_id": "org", "agent_id": None, "end_user": "", "request_tags": '["tag","tag"]', } + + +async def test_native_observation_enters_spend_pipeline_once_with_shared_daily_attribution( + db: Prisma, record: Callable[..., BaselineAccountingRecord], monkeypatch: pytest.MonkeyPatch, +) -> None: + import os + + from litellm.proxy.common_utils.user_api_key_cache import UserApiKeyCache + from litellm.proxy.db.db_spend_update_writer import DBSpendUpdateWriter + from litellm.proxy.utils import PrismaClient, ProxyLogging + + event: Final = record("routed", identical=False) + payload: Final = _native_observation_payload(event) monkeypatch.delenv("DATABASE_URL_READ_REPLICA", raising=False) client: Final = PrismaClient(os.environ["DATABASE_URL"], ProxyLogging(UserApiKeyCache())) writer: Final = DBSpendUpdateWriter() @@ -319,3 +325,42 @@ async def test_native_observation_enters_spend_pipeline_once_with_shared_daily_a assert tag_rows[0]["spend"] == tag_rows[0]["api_requests"] == 0 finally: await client.db.disconnect() + + +async def test_without_spend_logs_a_captured_turn_keeps_only_its_router_day_row( + db: Prisma, record: Callable[..., BaselineAccountingRecord], monkeypatch: pytest.MonkeyPatch, +) -> None: + import os + + from litellm.proxy.common_utils.user_api_key_cache import UserApiKeyCache + from litellm.proxy.db.autorouter_session_rollup import flush_autorouter_turn_transactions + from litellm.proxy.db.db_spend_update_writer import DBSpendUpdateWriter + from litellm.proxy.utils import PrismaClient, ProxyLogging + + event: Final = record("unlogged", identical=False) + monkeypatch.delenv("DATABASE_URL_READ_REPLICA", raising=False) + client: Final = PrismaClient(os.environ["DATABASE_URL"], ProxyLogging(UserApiKeyCache())) + try: + await client.db.connect() + await DBSpendUpdateWriter()._enqueue_autorouter_turn_transaction( + _native_observation_payload(event), client, spend_logs_kept=False + ) + assert client.baseline_accounting_transactions == [] + (turn,) = client.autorouter_turn_transactions + await flush_autorouter_turn_transactions(client, (turn,), n_retry_times=0) + finally: + client.autorouter_turn_transactions.clear() + await client.db.disconnect() + + assert await db.query_raw( + 'SELECT 1 FROM "LiteLLM_AutoRouterBaselineObservation" WHERE request_id=$1', event.observation.request_id + ) == [] + days: Final = await db.query_raw( + 'SELECT turns, spend FROM "LiteLLM_AutoRouterDailySpend" WHERE api_key=$1 AND router_name=$2', + event.api_key, event.router_name, + ) + assert [(day["turns"], day["spend"]) for day in days] == [(1, 0.17)] + for table in ("LiteLLM_AutoRouterSession", "LiteLLM_AutoRouterUserSession"): + assert await db.query_raw( + f'SELECT 1 FROM "{table}" WHERE api_key=$1 AND router_name=$2', event.api_key, event.router_name + ) == [] diff --git a/tests/unit/proxy/db/test_autorouter_session_rollup.py b/tests/unit/proxy/db/test_autorouter_session_rollup.py index c61a489f894..835399c568f 100644 --- a/tests/unit/proxy/db/test_autorouter_session_rollup.py +++ b/tests/unit/proxy/db/test_autorouter_session_rollup.py @@ -111,7 +111,6 @@ class TestBuildTransaction: [ {"status": "failure"}, {"api_key": ""}, - {"session_id": None}, {"model": ""}, {"startTime": "not-a-time"}, ], @@ -119,6 +118,12 @@ class TestBuildTransaction: def test_incomplete_payloads_are_skipped(self, payload_overrides: dict): assert _build(payload=_payload(**payload_overrides)) is None + @pytest.mark.parametrize("session_id", [None, ""]) + def test_a_request_without_a_session_keeps_its_router_day_money(self, session_id: str | None) -> None: + transaction: Final = _build(payload=_payload(session_id=session_id)) + assert transaction is not None + assert (transaction.session_id, transaction.router_name, transaction.spend) == ("", "live-auto", 0.01) + @pytest.mark.parametrize("metadata", [{}, {"routing_decision": None}, {"routing_decision": {}}]) def test_requests_without_a_routing_decision_are_skipped(self, metadata: dict): assert _build(metadata=metadata) is None diff --git a/tests/unit/proxy/db/test_db_spend_update_writer.py b/tests/unit/proxy/db/test_db_spend_update_writer.py index 7b160c055d2..4de90d5d7f7 100644 --- a/tests/unit/proxy/db/test_db_spend_update_writer.py +++ b/tests/unit/proxy/db/test_db_spend_update_writer.py @@ -289,6 +289,65 @@ async def test_update_database_skips_tool_usage_when_spend_logs_disabled(): assert prisma.tool_usage_transactions == [] +@pytest.mark.asyncio +@pytest.mark.parametrize("disable_spend_logs", [True, False]) +@pytest.mark.parametrize("session_id", ["session-1", None]) +async def test_a_routed_request_reaches_the_auto_router_rollup_whether_or_not_spend_logs_are_kept( + disable_spend_logs: bool, session_id: str | None +) -> None: + db_writer = DBSpendUpdateWriter() + db_writer._insert_spend_log_to_db = AsyncMock() + db_writer._batch_database_updates = AsyncMock() + prisma = _tool_usage_prisma() + prisma.autorouter_turn_transactions = [] + prisma._autorouter_turn_transactions_lock = asyncio.Lock() + routed_payload: Final = { + **_minimal_spend_payload(), + "status": "success", + "api_key": "hashed-key", + "user": "u1", + "session_id": session_id, + "model": "claude-haiku-4-5", + "model_group": "smart-router", + "spend": 0.25, + "startTime": "2026-07-25T10:00:00+00:00", + "metadata": json.dumps( + { + "routing_decision": {"router_model_name": "smart-router", "router_type": "complexity"}, + "autorouter_savings": 1.5, + } + ), + } + + with ( + patch("litellm.proxy.proxy_server.disable_spend_logs", disable_spend_logs), # test-quality-ok: update_database reads this proxy_server module global at call time; no injection seam + patch("litellm.proxy.proxy_server.prisma_client", prisma), + patch("litellm.proxy.proxy_server.litellm_proxy_budget_name", "test-budget"), + patch( + "litellm.proxy.spend_tracking.spend_tracking_utils.get_logging_payload", + return_value=routed_payload, + ), + ): + await db_writer.update_database( + token="test-token", + user_id="u1", + end_user_id=None, + team_id=None, + org_id=None, + kwargs={"model": "smart-router"}, + completion_response=_tool_call_response("get_weather"), + start_time=datetime.now(timezone.utc), + end_time=datetime.now(timezone.utc), + response_cost=0.25, + ) + + (turn,) = prisma.autorouter_turn_transactions + stored_session: Final = session_id if session_id and not disable_spend_logs else "" + assert (turn.router_name, turn.router_type, turn.session_id) == ("smart-router", "complexity", stored_session) + assert (turn.spend, turn.saved_spend) == (0.25, 1.5) + assert (prisma.tool_usage_transactions == []) is disable_spend_logs + + Statement = tuple[str, tuple[object, ...]] From 6b6222578e4a4042e3ab5a591aa643640335a8ae Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Fri, 2 Oct 2026 21:01:57 -0700 Subject: [PATCH 02/11] test(integration): let run.py select cells by pytest node id (#44319) Co-authored-by: mateo-berri <277851410+mateo-berri@users.noreply.github.com> --- tests/integration/README.md | 2 +- tests/integration/run.py | 40 +++++++++++++++++++++++------- tests/unit/test_integration_run.py | 36 +++++++++++++++++++++++++++ 3 files changed, 68 insertions(+), 10 deletions(-) create mode 100644 tests/unit/test_integration_run.py diff --git a/tests/integration/README.md b/tests/integration/README.md index 2204cde11e3..79a5d9c7c03 100644 --- a/tests/integration/README.md +++ b/tests/integration/README.md @@ -14,7 +14,7 @@ Reuse the existing canned provider handlers through `_support/upstream.py`. It r The CircleCI workflow starts its own database and Redis, restricts test-phase egress to its owned services and writes JUnit plus an executed-node manifest. Missing setup, failed cleanup or a selected test with neither a passed call nor a skip fail qualification. Skipped nodes are listed under `skipped` in `execution.json`, so the skip reasons double as the open bug list. Existing GitHub Actions jobs do not own these tests -There is no per-node manifest. The runner fails only when pytest fails, when collection errors, or when a selected file collects zero tests. Older tests still carry `@pytest.mark.covers(...)` decorators; the marker stays registered so they collect, but the IDs are not checked against anything and new tests should not use it. The GitHub Actions coverage census reads the `GROUPS` literal in `run.py` and treats every `tests/integration//test_*.py` file in a scheduled group as owned by CircleCI +There is no per-node manifest. A positional argument is a file of the group or a pytest node id inside one (`path::test[param]`), so one cell of a parametrized file can run alone. The runner fails only when pytest fails, when collection errors, or when a selected file collects zero tests. Older tests still carry `@pytest.mark.covers(...)` decorators; the marker stays registered so they collect, but the IDs are not checked against anything and new tests should not use it. The GitHub Actions coverage census reads the `GROUPS` literal in `run.py` and treats every `tests/integration//test_*.py` file in a scheduled group as owned by CircleCI Provider sentinels currently use the controlled server, not live recordings. The provider shard also runs the existing strict replay controls for changed requests, exhausted interactions, leftover interactions and no provider connection. Future recorded scenarios must use that replay-only implementation; missing recordings cannot fall back to a real provider. The observation endpoint is destructive and the current selection runs serially against one owned upstream diff --git a/tests/integration/run.py b/tests/integration/run.py index 30c1352f048..c5facec0bd2 100644 --- a/tests/integration/run.py +++ b/tests/integration/run.py @@ -5,6 +5,7 @@ import json import os import subprocess import sys +from dataclasses import dataclass from pathlib import Path from types import MappingProxyType from typing import Final @@ -24,6 +25,29 @@ GROUPS: Final = MappingProxyType( ) +@dataclass(frozen=True, slots=True) +class Selection: + nodes: tuple[str, ...] + foreign: tuple[str, ...] + + +def file_of(node: str) -> str: + return node.split("::", 1)[0] + + +def select(requested: tuple[str, ...], group_files: tuple[str, ...]) -> Selection: + members: Final = frozenset(group_files) + return Selection( + nodes=requested or group_files, + foreign=tuple(sorted({node for node in requested if file_of(node) not in members})), + ) + + +def uncollected(nodes: tuple[str, ...], collected: frozenset[str]) -> tuple[str, ...]: + collected_files: Final = frozenset(file_of(node) for node in collected) + return tuple(node for node in nodes if file_of(node) not in collected_files) + + def main() -> int: parser: Final = argparse.ArgumentParser() parser.add_argument("group", choices=tuple(GROUPS)) @@ -32,7 +56,7 @@ def main() -> int: parser.add_argument("--order-seed", type=int, default=int(os.environ.get("INTEGRATION_ORDER_SEED", "0"))) parser.add_argument("--workers", type=int, default=int(os.environ.get("INTEGRATION_WORKERS", "1"))) parser.add_argument("--list", action="store_true", help="print the group's test files and exit") - parser.add_argument("files", nargs="*", help="run only these files of the group") + parser.add_argument("files", nargs="*", help="run only these files, or pytest node ids inside them, of the group") options: Final = parser.parse_intermixed_args() root: Final = Path(__file__).resolve().parents[2] group_files: Final = tuple( @@ -43,11 +67,10 @@ def main() -> int: if options.list: print("\n".join(group_files)) return 0 - foreign: Final = sorted(set(options.files) - set(group_files)) - if foreign: - parser.error(f"Not in the {options.group} group: {', '.join(foreign)}") - selected: Final = tuple(options.files) or group_files - if not selected: + selection: Final = select(tuple(options.files), group_files) + if selection.foreign: + parser.error(f"Not in the {options.group} group: {', '.join(selection.foreign)}") + if not selection.nodes: parser.error(f"No integration test files selected for {options.group}") output: Final = options.results.resolve() output.mkdir(parents=True, exist_ok=True) @@ -62,7 +85,7 @@ def main() -> int: sys.executable, "-m", "pytest", - *selected, + *selection.nodes, "-vv", "-rs", "--strict-markers", @@ -86,8 +109,7 @@ def main() -> int: if result != 0: return result evidence: Final = json.loads((output / "execution.json").read_text()) - collected_files: Final = {node.split("::", 1)[0] for node in evidence["collected"]} - empty: Final = tuple(path for path in selected if path not in collected_files) + empty: Final = uncollected(selection.nodes, frozenset(evidence["collected"])) if empty: sys.stderr.write(f"Selected integration files collected zero tests: {', '.join(empty)}\n") return 1 diff --git a/tests/unit/test_integration_run.py b/tests/unit/test_integration_run.py new file mode 100644 index 00000000000..36612525572 --- /dev/null +++ b/tests/unit/test_integration_run.py @@ -0,0 +1,36 @@ +from typing import Final + +from tests.integration.run import select, uncollected + +_GROUP: Final = ( + "tests/integration/cost_calculation/test_cost_tracking.py", + "tests/integration/cost_calculation/test_rollups.py", +) +_CELL: Final = ( + "tests/integration/cost_calculation/test_cost_tracking.py" + "::test_case_bills_expected_cost[perplexity/pplx-decider-v1-27b-decisions]" +) + + +def test_a_node_id_inside_a_group_file_is_selected_as_written() -> None: + selection: Final = select((_CELL,), _GROUP) + assert selection.nodes == (_CELL,) + assert selection.foreign == () + + +def test_a_node_id_outside_the_group_is_foreign_by_its_file() -> None: + foreign: Final = "tests/integration/providers/test_decisions_wire.py::test_key_checks_match_chat" + assert select((foreign, _CELL), _GROUP).foreign == (foreign,) + + +def test_no_request_selects_every_group_file() -> None: + assert select((), _GROUP).nodes == _GROUP + + +def test_a_node_id_whose_file_collected_tests_is_not_empty() -> None: + collected: Final = frozenset({_CELL, "tests/integration/cost_calculation/test_cost_tracking.py::test_other"}) + assert uncollected((_CELL,), collected) == () + + +def test_a_selected_file_that_collected_nothing_is_reported() -> None: + assert uncollected(_GROUP, frozenset({_CELL})) == ("tests/integration/cost_calculation/test_rollups.py",) From 54260bec887b0468eb8300f62aae9d0a7d28232c Mon Sep 17 00:00:00 2001 From: Aasif Multani Date: Sat, 3 Oct 2026 09:33:58 +0530 Subject: [PATCH 03/11] fix(bedrock): honor unsupported reasoning effort levels for OpenAI GPT models (#44183) Bedrock GPT-5.6 rejects reasoning effort minimal with 400 unsupported_value. The model map already sets supports_minimal_reasoning_effort=false for these models, but neither the Converse path nor the native Responses path read it, so the value was forwarded. Drop it under drop_params and raise UnsupportedParamsError otherwise, matching how the OpenAI GPT-5 path treats explicitly disabled effort levels Co-authored-by: Aasif-Multani <20943280+Aasif-Multani@users.noreply.github.com> --- .../bedrock/chat/converse_transformation.py | 20 +++++++++++ litellm/llms/bedrock/common_utils.py | 8 +++++ .../llms/bedrock/responses/transformation.py | 27 ++++++++++++++- .../chat/test_converse_transformation.py | 33 +++++++++++++++++++ .../test_bedrock_openai_responses.py | 30 +++++++++++++++++ 5 files changed, 117 insertions(+), 1 deletion(-) diff --git a/litellm/llms/bedrock/chat/converse_transformation.py b/litellm/llms/bedrock/chat/converse_transformation.py index 9f094521bfe..29347f6554a 100644 --- a/litellm/llms/bedrock/chat/converse_transformation.py +++ b/litellm/llms/bedrock/chat/converse_transformation.py @@ -104,6 +104,7 @@ from ..common_utils import ( BedrockModelInfo, bedrock_converse_supports_parallel_tool_use_config, bedrock_model_accepts_cache_points, + bedrock_reasoning_effort_disabled, get_anthropic_beta_from_headers, get_bedrock_tool_name, is_bedrock_application_inference_profile_arn, @@ -1135,6 +1136,25 @@ class AmazonConverseConfig(BaseConfig): "Dropping unsupported `reasoning_effort` param for Bedrock model=%s; it always reasons and rejects it.", model, ) + elif ( + param == "reasoning_effort" + and isinstance(value, str) + and self._is_openai_gpt_reasoning_model(model) + and bedrock_reasoning_effort_disabled(model=model, effort=value) + ): + if not (litellm.drop_params or drop_params): + raise litellm.utils.UnsupportedParamsError( + message=( + f"{model} does not support reasoning_effort={value}. " + "To drop unsupported params, set `litellm.drop_params = True`." + ), + status_code=400, + ) + verbose_logger.debug( + "Dropping unsupported `reasoning_effort=%s` for Bedrock model=%s.", + value, + model, + ) elif param == "reasoning_effort" and isinstance(value, str): self._handle_reasoning_effort_parameter( model=model, reasoning_effort=value, optional_params=optional_params diff --git a/litellm/llms/bedrock/common_utils.py b/litellm/llms/bedrock/common_utils.py index 2ab506771ec..12d04a08392 100644 --- a/litellm/llms/bedrock/common_utils.py +++ b/litellm/llms/bedrock/common_utils.py @@ -1017,6 +1017,14 @@ def _mantle_api_base_from_env() -> str | None: return next((base[: -len(suffix)] for suffix in _MANTLE_OPENAI_BASE_SUFFIXES if base.endswith(suffix)), base) +def bedrock_reasoning_effort_disabled(model: str, effort: str) -> bool: + from litellm.utils import is_explicitly_disabled_factory + + return is_explicitly_disabled_factory( + model=model, custom_llm_provider="bedrock_converse", key=f"supports_{effort}_reasoning_effort" + ) + + def bedrock_supports_openai_responses(model: str | None, model_cost: Mapping[str, object]) -> bool: """Whether a Bedrock model is served by bedrock-runtime's OpenAI Responses surface. diff --git a/litellm/llms/bedrock/responses/transformation.py b/litellm/llms/bedrock/responses/transformation.py index b56069c79a6..086d1211835 100644 --- a/litellm/llms/bedrock/responses/transformation.py +++ b/litellm/llms/bedrock/responses/transformation.py @@ -52,6 +52,7 @@ from litellm.llms.bedrock.base_aws_llm import BaseAWSLLM from litellm.llms.bedrock.common_utils import ( BEDROCK_CHAT_COMPLETIONS_ROUTE_PREFIX, BedrockError, + bedrock_reasoning_effort_disabled, bedrock_supports_openai_responses, ) from litellm.llms.openai.responses.transformation import OpenAIResponsesAPIConfig @@ -154,6 +155,29 @@ def inline_remote_image_urls( return items # pyright: ignore[reportReturnType] # items keep the caller's input union +def _without_disabled_reasoning_effort( + params: Mapping[str, object], model: str, drop_params: bool +) -> dict[str, object]: # mutable-ok: becomes the map_openai_params return value + reasoning: Final = params.get("reasoning") + effort: Final = reasoning.get("effort") if isinstance(reasoning, Mapping) else None + if not isinstance(reasoning, Mapping) or not isinstance(effort, str): + return dict(params) + if not bedrock_reasoning_effort_disabled(model=model, effort=effort): + return dict(params) + if not (drop_params or litellm.drop_params): + raise litellm.UnsupportedParamsError( + message=( + f"{model} does not support reasoning.effort={effort}. " + "To drop unsupported params, set `litellm.drop_params = True`." + ), + status_code=400, + ) + verbose_logger.debug("Dropping unsupported `reasoning.effort=%s` for Bedrock model=%s.", effort, model) + rest: Final = {key: value for key, value in reasoning.items() if key != "effort"} + without_reasoning: Final = {key: value for key, value in params.items() if key != "reasoning"} + return {**without_reasoning, "reasoning": rest} if rest else without_reasoning + + class BedrockOpenAIResponsesConfig(BaseAWSLLM, OpenAIResponsesAPIConfig): """Responses API config for the OpenAI models on the bedrock-runtime endpoint.""" @@ -270,7 +294,8 @@ class BedrockOpenAIResponsesConfig(BaseAWSLLM, OpenAIResponsesAPIConfig): "Bedrock Runtime Responses API: dropping unsupported parameter(s) %s that the endpoint rejects.", unsupported, ) - params: Final = {key: value for key, value in mapped.items() if key not in unsupported} + supported: Final[dict[str, object]] = {key: value for key, value in mapped.items() if key not in unsupported} + params: Final = _without_disabled_reasoning_effort(supported, model, drop_params) tools: Final = params.get("tools") if not isinstance(tools, list): return params diff --git a/tests/unit/llms/bedrock/chat/test_converse_transformation.py b/tests/unit/llms/bedrock/chat/test_converse_transformation.py index 0c0be5d868f..07c54eee395 100644 --- a/tests/unit/llms/bedrock/chat/test_converse_transformation.py +++ b/tests/unit/llms/bedrock/chat/test_converse_transformation.py @@ -520,6 +520,39 @@ def test_reasoning_effort_maps_to_reasoning_effort_for_openai_gpt5_converse(mode assert "thinking" not in additional_request_params +@pytest.mark.parametrize( + "model", + [ + "us.openai.gpt-5.6-luna", + "bedrock/converse/global.openai.gpt-5.6-terra", + "us.openai.gpt-6-astra", + ], +) +def test_openai_gpt5_converse_rejects_effort_level_disabled_in_model_map(model, local_model_cost_map): + config = AmazonConverseConfig() + assert litellm.utils.is_explicitly_disabled_factory( + model=model, custom_llm_provider="bedrock_converse", key="supports_minimal_reasoning_effort" + ) + + with pytest.raises(litellm.utils.UnsupportedParamsError, match="minimal"): + config.map_openai_params( + non_default_params={"reasoning_effort": "minimal"}, + optional_params={}, + model=model, + drop_params=False, + ) + + optional_params = config.map_openai_params( + non_default_params={"reasoning_effort": "minimal"}, + optional_params={}, + model=model, + drop_params=True, + ) + _, additional_request_params, _, _ = config._prepare_request_params(optional_params, model) + assert "reasoning" not in additional_request_params + assert "thinking" not in additional_request_params + + @pytest.mark.parametrize( "model", [ diff --git a/tests/unit/llms/bedrock/responses/test_bedrock_openai_responses.py b/tests/unit/llms/bedrock/responses/test_bedrock_openai_responses.py index c0803f6636b..6da131f38cc 100644 --- a/tests/unit/llms/bedrock/responses/test_bedrock_openai_responses.py +++ b/tests/unit/llms/bedrock/responses/test_bedrock_openai_responses.py @@ -328,6 +328,36 @@ class TestBackgroundDrop: assert not [r for r in caplog.records if "dropping unsupported parameter" in r.getMessage()] +class TestDisabledReasoningEffort: + @pytest.mark.parametrize("model", ["us.openai.gpt-5.6-luna", MODEL]) + def test_effort_level_disabled_in_model_map_is_rejected(self, model, local_model_cost_map): + with pytest.raises(litellm.UnsupportedParamsError, match="minimal"): + _cfg().map_openai_params( + response_api_optional_params={"reasoning": {"effort": "minimal"}}, model=model, drop_params=False + ) + + @pytest.mark.parametrize("model", ["us.openai.gpt-5.6-luna", MODEL]) + def test_effort_level_disabled_in_model_map_is_dropped_with_drop_params(self, model, local_model_cost_map): + params = _cfg().map_openai_params( + response_api_optional_params={"reasoning": {"effort": "minimal", "summary": "auto"}, "max_output_tokens": 64}, + model=model, + drop_params=True, + ) + assert params == {"reasoning": {"summary": "auto"}, "max_output_tokens": 64} + + def test_effort_only_reasoning_is_removed_when_dropped(self, local_model_cost_map): + params = _cfg().map_openai_params( + response_api_optional_params={"reasoning": {"effort": "minimal"}}, model=MODEL, drop_params=True + ) + assert params == {} + + def test_supported_effort_level_is_forwarded(self, local_model_cost_map): + params = _cfg().map_openai_params( + response_api_optional_params={"reasoning": {"effort": "low"}}, model=MODEL, drop_params=False + ) + assert params == {"reasoning": {"effort": "low"}} + + def _never_fetch(url: str) -> str: raise AssertionError(f"unexpected sync fetch of {url}") From 80a2f4d8a8c32271d04e9facf64b14691a3d611c Mon Sep 17 00:00:00 2001 From: ishaan-berri <155045088+ishaan-berri@users.noreply.github.com> Date: Fri, 2 Oct 2026 21:06:04 -0700 Subject: [PATCH 04/11] feat(lens): add preset watch-for checks to investigation setup (#44313) * feat(lens): add preset watch-for checks for common agent failures * feat(lens): add keyboard-driven watch-for picker with lens dot animation * feat(lens): use the watch-for picker in investigation setup * feat(lens): show preset checks by name in the criteria tab * test(lens): cover saving and editing watch-for presets * feat(lens): shorten watch-for summaries and start with three presets on * feat(lens): lay out watch-for presets as toggle tiles with a clear add-your-own button * feat(lens): open a custom check from the watch-for picker * test(lens): cover watch-for tiles and the add-your-own button * fix(lens): draw the selected tile border inside the tile so the dialog edge cannot clip it --- .../LensSetup.integration.test.tsx | 70 ++++++- .../lens/_components/LensSetup.tsx | 60 +++--- .../(dashboard)/lens/_components/LensView.tsx | 16 +- .../lens/_components/WatchPicker.tsx | 177 ++++++++++++++++++ .../(dashboard)/lens/_components/lensData.ts | 93 ++++++++- 5 files changed, 375 insertions(+), 41 deletions(-) create mode 100644 ui/litellm-dashboard/src/app/(dashboard)/lens/_components/WatchPicker.tsx diff --git a/ui/litellm-dashboard/src/app/(dashboard)/lens/_components/LensSetup.integration.test.tsx b/ui/litellm-dashboard/src/app/(dashboard)/lens/_components/LensSetup.integration.test.tsx index cdc201d8dd5..6b664c230f9 100644 --- a/ui/litellm-dashboard/src/app/(dashboard)/lens/_components/LensSetup.integration.test.tsx +++ b/ui/litellm-dashboard/src/app/(dashboard)/lens/_components/LensSetup.integration.test.tsx @@ -5,7 +5,7 @@ import { renderWithProviders } from "@/../tests/test-utils"; import { MonitoringSetup } from "./LensOverview"; import { LensSetup } from "./LensSetup"; import { apiClient } from "@/components/networking"; -import type { Settings } from "./lensData"; +import { initialWatches, watchChecks, type Settings } from "./lensData"; vi.mock("@/components/networking", () => ({ apiClient: { post: vi.fn(), get: vi.fn() } })); @@ -105,12 +105,13 @@ describe("Lens setup", () => { fireEvent.change(screen.getByRole("combobox", { name: "Metadata key 1" }), { target: { value: "swarm" } }); fireEvent.change(screen.getByRole("combobox", { name: "Metadata value 1" }), { target: { value: "research" } }); await user.click(screen.getByRole("button", { name: "Continue" })); + await user.click(screen.getByRole("button", { name: /Add your own/ })); await user.type(screen.getByRole("textbox", { name: "Check 1" }), "Find incomplete reports"); - await user.click(screen.getByRole("button", { name: "Add check" })); + await user.click(screen.getByRole("button", { name: /Add your own/ })); fireEvent.change(screen.getByRole("textbox", { name: "Check 2" }), { target: { value: "Find repeated searches\nInclude retries that add no information" }, }); - await user.click(screen.getByRole("button", { name: "Add check" })); + await user.click(screen.getByRole("button", { name: /Add your own/ })); await user.click(screen.getByRole("button", { name: "Remove check 3" })); await user.click(screen.getByRole("button", { name: "Continue" })); await waitFor(() => expect(screen.getByRole("button", { name: "Run investigation" })).toBeEnabled()); @@ -129,6 +130,7 @@ describe("Lens setup", () => { model: "analysis", monthly_budget: 100, checks: [ + ...watchChecks(initialWatches(undefined)), expect.objectContaining({ instruction: "Find incomplete reports" }), expect.objectContaining({ instruction: "Find repeated searches\nInclude retries that add no information" }), ], @@ -411,3 +413,65 @@ it("saves a discovered agent independently of the application name", async () => expect.objectContaining({ agent_name: "research_agent", service: "shared-service" }), ); }); + +describe("Watch for", () => { + const tile = (name: string) => screen.getByRole("button", { name: new RegExp(`^${name}`) }); + + it("saves exactly the presets the user toggled, by click and by number key", async () => { + const save = vi.fn().mockResolvedValue(undefined); + const user = userEvent.setup(); + renderWithProviders( + , + ); + await user.click(screen.getByRole("button", { name: "Continue" })); + await user.click(tile("unhappy")); + tile("unsolved").focus(); + await user.keyboard("6"); + await user.keyboard("{ArrowRight}{ArrowRight}"); + expect(tile("unsafe")).toHaveFocus(); + expect(tile("unhappy")).toHaveAttribute("aria-pressed", "false"); + expect(tile("looping")).toHaveAttribute("aria-pressed", "true"); + await user.click(screen.getByRole("button", { name: "Continue" })); + await waitFor(() => expect(screen.getByRole("button", { name: "Run investigation" })).toBeEnabled()); + await user.click(screen.getByRole("button", { name: "Run investigation" })); + const saved = (save.mock.calls[0][0] as Settings).checks.map((check) => check.id); + expect(saved).toEqual(["watch_unsolved", "watch_blocked", "watch_looping"]); + }); + + it("keeps an edited investigation's preset choices and custom checks apart", async () => { + const save = vi.fn().mockResolvedValue(undefined); + const user = userEvent.setup(); + const initial: Settings = { + ...settings, + checks: [{ id: "watch_invented", instruction: "old wording", enabled: true }, settings.checks[1]], + }; + renderWithProviders( + , + ); + await user.click(screen.getByRole("button", { name: "Continue" })); + expect(tile("invented")).toHaveAttribute("aria-pressed", "true"); + expect(tile("unsolved")).toHaveAttribute("aria-pressed", "false"); + expect(screen.getByRole("textbox", { name: "Check 1" })).toHaveValue("Find incomplete reports"); + await user.click(screen.getByRole("button", { name: "Continue" })); + await waitFor(() => expect(screen.getByRole("button", { name: "Save changes" })).toBeEnabled()); + await user.click(screen.getByRole("button", { name: "Save changes" })); + expect((save.mock.calls[0][0] as Settings).checks).toEqual([ + ...watchChecks(new Set(["watch_invented"])), + settings.checks[1], + ]); + }); + + it("lets a run start from presets alone and blocks it once nothing is selected", async () => { + const user = userEvent.setup(); + renderWithProviders( + , + ); + await user.click(screen.getByRole("button", { name: "Continue" })); + await user.click(screen.getByRole("button", { name: "Continue" })); + expect(await screen.findByRole("button", { name: "Run investigation" })).toBeInTheDocument(); + await user.click(screen.getByRole("button", { name: "Back" })); + for (const name of ["unsolved", "blocked", "unhappy"]) await user.click(tile(name)); + await user.click(screen.getByRole("button", { name: "Continue" })); + expect(screen.getByRole("alert")).toHaveTextContent("pick something to watch for"); + }); +}); diff --git a/ui/litellm-dashboard/src/app/(dashboard)/lens/_components/LensSetup.tsx b/ui/litellm-dashboard/src/app/(dashboard)/lens/_components/LensSetup.tsx index 32fcb2f7d43..88000eb5192 100644 --- a/ui/litellm-dashboard/src/app/(dashboard)/lens/_components/LensSetup.tsx +++ b/ui/litellm-dashboard/src/app/(dashboard)/lens/_components/LensSetup.tsx @@ -1,7 +1,7 @@ "use client"; import { useState } from "react"; -import { Plus, X } from "lucide-react"; +import { X } from "lucide-react"; import { Button } from "@/components/ui/button"; import { Input } from "@/components/ui/input"; import { Textarea } from "@/components/ui/textarea"; @@ -16,7 +16,16 @@ import { import { SearchSelect } from "@/components/shared/SearchSelect"; import { DurationInput } from "./DurationInput"; import { ActivityScope, type ActivitySelection } from "./ActivityScope"; -import { analysisModelOptions, normalizeFilters, type AnalysisModelInfo, type Settings } from "./lensData"; +import { WatchPicker } from "./WatchPicker"; +import { + analysisModelOptions, + initialWatches, + isWatch, + normalizeFilters, + watchChecks, + type AnalysisModelInfo, + type Settings, +} from "./lensData"; function validateSample(selection: ActivitySelection) { const hours = selection.lookback_hours ?? 24; @@ -77,7 +86,8 @@ export function LensSetup({ }; const [selection, setSelection] = useState(initialSelection); const [context, setContext] = useState(initial?.context ?? ""); - const [questions, setQuestions] = useState(() => (initial?.checks?.length ? initial.checks : [newCheck()])); + const [watching, setWatching] = useState>(() => initialWatches(initial?.checks)); + const [questions, setQuestions] = useState(() => (initial?.checks ?? []).filter((check) => !isWatch(check))); const [selectedModel, setModel] = useState(initial?.model ?? null); const model = selectedModel ?? defaultModel ?? ""; const [budget, setBudget] = useState(initial?.monthly_budget ?? 100); @@ -102,8 +112,8 @@ export function LensSetup({ if (step >= 2 && manualSelection && !selection.execution_ids?.length) throw new Error("Choose at least one run or turn off individual selection"); validateSample(selection); - if (step >= 1 && !context.trim() && !filledChecks.length) - throw new Error("Describe the expected behavior or what to look out for"); + const nothingToCheck = !context.trim() && !filledChecks.length && !watching.size; + if (step >= 1 && nothingToCheck) throw new Error("Describe the expected behavior or pick something to watch for"); if (filledChecks.some((check) => check.instruction.trim().length < 3)) throw new Error("Use at least three characters for each check"); }; @@ -132,7 +142,10 @@ export function LensSetup({ interval_minutes: interval, concurrency: initial?.concurrency ?? 8, filters: normalizeFilters(selection.filters ?? []), - checks: filledChecks.map((check) => ({ ...check, instruction: check.instruction.trim() })), + checks: [ + ...watchChecks(watching), + ...filledChecks.map((check) => ({ ...check, instruction: check.instruction.trim() })), + ], }; await onSave(settings); } catch (cause) { @@ -172,7 +185,7 @@ export function LensSetup({ }} > {headings[step]} @@ -239,8 +252,13 @@ export function LensSetup({ placeholder="Answer the customer's question using verified sources and explain when information is missing." /> -
- What should we look out for? + setQuestions([...questions, newCheck()])} + /> +
+ Custom checks {questions.map((check, index) => (