From 65bb28221c7bf61dd4790e7fa5490a3a68fe2836 Mon Sep 17 00:00:00 2001 From: kerry Date: Fri, 2 Oct 2026 16:47:44 +0000 Subject: [PATCH 01/23] fix(proxy): record header-derived spend tags on pass-through routes Co-authored-by: Quinn Xu Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- litellm/litellm_core_utils/litellm_logging.py | 4 +- .../spend/test_passthrough_spend_tags.py | 76 +++++++++++++++++++ .../test_litellm_logging.py | 24 ++++++ 3 files changed, 102 insertions(+), 2 deletions(-) create mode 100644 tests/integration/spend/test_passthrough_spend_tags.py diff --git a/litellm/litellm_core_utils/litellm_logging.py b/litellm/litellm_core_utils/litellm_logging.py index a43574b1a04..ad7bdcb91ed 100644 --- a/litellm/litellm_core_utils/litellm_logging.py +++ b/litellm/litellm_core_utils/litellm_logging.py @@ -6274,7 +6274,7 @@ class StandardLoggingPayloadSetup: return None user_agent_tags: list[str] | None = None headers: Final = proxy_server_request.get("headers", {}) - if headers is not None and isinstance(headers, dict): + if headers is not None and isinstance(headers, Mapping): if "user-agent" in headers: user_agent: Final = headers["user-agent"] if user_agent is not None: @@ -6299,7 +6299,7 @@ class StandardLoggingPayloadSetup: return None headers: Final = proxy_server_request.get("headers", {}) - if not isinstance(headers, dict): + if not isinstance(headers, Mapping): return None header_tags: Final = [] diff --git a/tests/integration/spend/test_passthrough_spend_tags.py b/tests/integration/spend/test_passthrough_spend_tags.py new file mode 100644 index 00000000000..9d89a866cc8 --- /dev/null +++ b/tests/integration/spend/test_passthrough_spend_tags.py @@ -0,0 +1,76 @@ +import json +import uuid +from hashlib import sha256 +from pathlib import Path +from typing import Final + +import pytest +import yaml +from integration._support.client import Gateway, eventually +from integration._support.database import read_rows +from integration._support.process import owned_proxy +from integration._support.wire import Reply, Request, wire_server + +MODEL: Final = "claude-sonnet-4-5-20250929" +SENT_HEADERS: Final = {"user-agent": "claude-cli/2.0.0", "x-tenant-id": "tenant-a"} +EXPECTED_TAGS: Final = ["User-Agent: claude-cli", "User-Agent: claude-cli/2.0.0", "x-tenant-id: tenant-a"] + + +def _respond(request: Request) -> Reply: + assert request.method == "POST" and request.target == "/v1/messages", request.target + return Reply( + body=json.dumps( + { + "id": f"msg_{uuid.uuid4().hex}", + "type": "message", + "role": "assistant", + "model": MODEL, + "content": [{"type": "text", "text": "tagged"}], + "stop_reason": "end_turn", + "stop_sequence": None, + "usage": {"input_tokens": 10, "output_tokens": 2}, + } + ).encode() + ) + + +def _request_tags(key: str) -> list[list[str]]: + rows: Final = read_rows( + 'SELECT request_tags FROM "LiteLLM_SpendLogs" WHERE api_key=%s', (sha256(key.encode()).hexdigest(),) + ) + return [ + json.loads(row["request_tags"]) if isinstance(row["request_tags"], str) else row["request_tags"] for row in rows + ] + + +@pytest.mark.parametrize( + "route", [pytest.param("/anthropic/v1/messages", id="passthrough"), pytest.param("/v1/messages", id="unified")] +) +def test_header_derived_spend_tags_are_recorded_on_anthropic_messages_routes( + gateway: Gateway, tmp_path: Path, route: str +) -> None: + with wire_server(_respond) as wire: + config: Final = yaml.safe_load(Path("tests/integration/proxy_config.yaml").read_text()) + config["litellm_settings"]["extra_spend_tag_headers"] = ["x-tenant-id"] + path: Final = tmp_path / "spend-tag-headers.yaml" + path.write_text(yaml.safe_dump(config)) + environment: Final = {"ANTHROPIC_API_BASE": wire.url, "ANTHROPIC_API_KEY": "synthetic-anthropic-key"} + with owned_proxy(gateway, tmp_path, environment, config=path) as candidate, candidate.scenario() as scenario: + model: Final = scenario.model( + model=f"anthropic/{MODEL}", api_base=wire.url, api_key="synthetic-anthropic-key" + ) + key: Final = scenario.key() + response: Final = candidate.request( + "POST", + route, + { + "model": MODEL if route == "/anthropic/v1/messages" else model, + "max_tokens": 16, + "messages": [{"role": "user", "content": "tag me"}], + }, + key=key, + headers={**SENT_HEADERS, "anthropic-version": "2023-06-01"}, + ) + assert response.status_code == 200, response.text + assert len(wire.drain()) == 1 + assert eventually(lambda: _request_tags(key), lambda tags: len(tags) == 1, seconds=70) == [EXPECTED_TAGS] diff --git a/tests/unit/litellm_core_utils/test_litellm_logging.py b/tests/unit/litellm_core_utils/test_litellm_logging.py index e07ffe00d4c..494e82d22e7 100644 --- a/tests/unit/litellm_core_utils/test_litellm_logging.py +++ b/tests/unit/litellm_core_utils/test_litellm_logging.py @@ -2959,6 +2959,30 @@ def test_get_extra_header_tags(): delattr(litellm, "extra_spend_tag_headers") +def test_get_request_tags_reads_header_tags_from_starlette_headers(): + from starlette.datastructures import Headers + + import litellm + from litellm.litellm_core_utils.litellm_logging import StandardLoggingPayloadSetup + + original_extra_headers = getattr(litellm, "extra_spend_tag_headers", None) + original_disable_user_agent = litellm.disable_add_user_agent_to_request_tags + try: + litellm.extra_spend_tag_headers = ["x-tenant-id"] + litellm.disable_add_user_agent_to_request_tags = False + proxy_server_request = {"headers": Headers({"user-agent": "claude-cli/2.0.0", "x-tenant-id": "tenant-a"})} + + assert StandardLoggingPayloadSetup._get_request_tags( + litellm_params={}, proxy_server_request=proxy_server_request + ) == ["User-Agent: claude-cli", "User-Agent: claude-cli/2.0.0", "x-tenant-id: tenant-a"] + finally: + if original_extra_headers is not None: + litellm.extra_spend_tag_headers = original_extra_headers + elif hasattr(litellm, "extra_spend_tag_headers"): + delattr(litellm, "extra_spend_tag_headers") + litellm.disable_add_user_agent_to_request_tags = original_disable_user_agent + + def test_response_cost_calculator_with_response_cost_in_hidden_params(logging_obj): from litellm import Router From 454bdf8f1e2cbbee44e834818101004b7553d47d Mon Sep 17 00:00:00 2001 From: kerry Date: Fri, 2 Oct 2026 17:08:05 +0000 Subject: [PATCH 02/23] test(spend): name the request-tag integration file after the feature Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- ...t_passthrough_spend_tags.py => test_spend_log_request_tags.py} | 0 1 file changed, 0 insertions(+), 0 deletions(-) rename tests/integration/spend/{test_passthrough_spend_tags.py => test_spend_log_request_tags.py} (100%) diff --git a/tests/integration/spend/test_passthrough_spend_tags.py b/tests/integration/spend/test_spend_log_request_tags.py similarity index 100% rename from tests/integration/spend/test_passthrough_spend_tags.py rename to tests/integration/spend/test_spend_log_request_tags.py From c01623b7656748acb92e80a424b157c639ec2da5 Mon Sep 17 00:00:00 2001 From: kerry Date: Fri, 2 Oct 2026 17:47:09 +0000 Subject: [PATCH 03/23] test(spend): audit header-derived request tags across routes, modes and outages Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- .../integration/spend/_request_tag_helpers.py | 219 +++++ .../spend/test_spend_log_request_tags.py | 924 ++++++++++++++++++ .../test_spend_log_request_tags_chaos.py | 209 ++++ 3 files changed, 1352 insertions(+) create mode 100644 tests/integration/spend/_request_tag_helpers.py create mode 100644 tests/integration/spend/test_spend_log_request_tags_chaos.py diff --git a/tests/integration/spend/_request_tag_helpers.py b/tests/integration/spend/_request_tag_helpers.py new file mode 100644 index 00000000000..6bf72f594c5 --- /dev/null +++ b/tests/integration/spend/_request_tag_helpers.py @@ -0,0 +1,219 @@ +import json +import uuid +from hashlib import sha256 +from pathlib import Path +from typing import Final + +import yaml +from integration._support.database import read_rows +from integration._support.wire import Reply, Request + +ANTHROPIC_MODEL: Final = "claude-sonnet-4-5-20250929" +OPENAI_MODEL: Final = "gpt-4o-mini" +GEMINI_MODEL: Final = "gemini-2.5-flash" +HEADERS: Final = {"user-agent": "claude-cli/2.0.0", "x-tenant-id": "tenant-a"} +T3: Final = ["User-Agent: claude-cli", "User-Agent: claude-cli/2.0.0", "x-tenant-id: tenant-a"] +ANTHROPIC_SONNET_BODY: Final = { + "id": "msg_synthetic", + "type": "message", + "role": "assistant", + "model": ANTHROPIC_MODEL, + "content": [{"type": "text", "text": "tagged"}], + "stop_reason": "end_turn", + "stop_sequence": None, + "usage": {"input_tokens": 10, "output_tokens": 2}, +} +CHAT_COMPLETION_BODY: Final = { + "id": "chatcmpl-synthetic", + "object": "chat.completion", + "created": 1, + "model": OPENAI_MODEL, + "choices": [{"index": 0, "message": {"role": "assistant", "content": "tagged"}, "finish_reason": "stop"}], + "usage": {"prompt_tokens": 10, "completion_tokens": 2, "total_tokens": 12}, +} +RESPONSES_BODY: Final = { + "id": "resp_synthetic", + "object": "response", + "created_at": 1, + "status": "completed", + "model": OPENAI_MODEL, + "output": [ + { + "type": "message", + "id": "msg_synthetic", + "status": "completed", + "role": "assistant", + "content": [{"type": "output_text", "text": "tagged", "annotations": []}], + } + ], + "usage": { + "input_tokens": 10, + "output_tokens": 2, + "total_tokens": 12, + "input_tokens_details": {"cached_tokens": 0}, + "output_tokens_details": {"reasoning_tokens": 0}, + }, +} +GEMINI_BODY: Final = { + "candidates": [ + { + "content": {"parts": [{"text": "tagged"}], "role": "model"}, + "finishReason": "STOP", + } + ], + "usageMetadata": {"promptTokenCount": 10, "candidatesTokenCount": 2, "totalTokenCount": 12}, +} +ANTHROPIC_STREAM_EVENTS: Final = ( + { + "type": "message_start", + "message": {**ANTHROPIC_SONNET_BODY, "content": [], "usage": {"input_tokens": 10, "output_tokens": 1}}, + }, + {"type": "content_block_start", "index": 0, "content_block": {"type": "text", "text": ""}}, + {"type": "content_block_delta", "index": 0, "delta": {"type": "text_delta", "text": "tagged"}}, + {"type": "content_block_stop", "index": 0}, + { + "type": "message_delta", + "delta": {"stop_reason": "end_turn", "stop_sequence": None}, + "usage": {"output_tokens": 2}, + }, + {"type": "message_stop"}, +) +OPENAI_STREAM_CHUNKS: Final = ( + { + "id": CHAT_COMPLETION_BODY["id"], + "object": "chat.completion.chunk", + "created": 1, + "model": OPENAI_MODEL, + "choices": [{"index": 0, "delta": {"role": "assistant", "content": "tagged"}, "finish_reason": None}], + }, + { + "id": CHAT_COMPLETION_BODY["id"], + "object": "chat.completion.chunk", + "created": 1, + "model": OPENAI_MODEL, + "choices": [{"index": 0, "delta": {}, "finish_reason": "stop"}], + }, + { + "id": CHAT_COMPLETION_BODY["id"], + "object": "chat.completion.chunk", + "created": 1, + "model": OPENAI_MODEL, + "choices": [], + "usage": CHAT_COMPLETION_BODY["usage"], + }, +) + + +def _message_id() -> str: + return "msg_" + uuid.uuid4().hex + + +def _sse_frames(events: tuple[dict, ...]) -> tuple[bytes, ...]: + return tuple(f"event: {event['type']}\ndata: {json.dumps(event)}\n\n".encode() for event in events) + + +def provider_reply(request: Request) -> Reply: + """Scripted edge for every route the audit drives: anthropic messages, openai chat completions and + responses, gemini generateContent. Error bodies keyed off the sentinel model name.""" + body: Final = json.loads(request.body) if request.body else {} + if body.get("model") == "claude-nonexistent-model": + return Reply( + status=400, + body=json.dumps( + {"error": {"type": "invalid_request_error", "message": "model: claude-nonexistent-model"}} + ).encode(), + ) + if request.target == "/v1/messages": + identity: Final = _message_id() + if body.get("stream") is True: + events: Final = tuple( + {**event, "message": {**event.get("message", {}), "id": identity}} if "message" in event else event + for event in ANTHROPIC_STREAM_EVENTS + ) + return Reply(content_type="text/event-stream", chunks=_sse_frames(events)) + return Reply(body=json.dumps({**ANTHROPIC_SONNET_BODY, "id": identity}).encode()) + if request.target == "/v1/chat/completions": + identity = "chatcmpl_" + uuid.uuid4().hex + if body.get("stream") is True: + frames: Final = tuple( + f"data: {json.dumps({**chunk, 'id': identity})}\n\n".encode() for chunk in OPENAI_STREAM_CHUNKS + ) + (b"data: [DONE]\n\n",) + return Reply(content_type="text/event-stream", chunks=frames) + return Reply(body=json.dumps({**CHAT_COMPLETION_BODY, "id": identity}).encode()) + if request.target == "/v1/responses": + identity = "resp_" + uuid.uuid4().hex + message: Final = _message_id() + completed_body: Final = { + **RESPONSES_BODY, + "id": identity, + "output": [{**RESPONSES_BODY["output"][0], "id": message}], + } + if body.get("stream") is True: + created: Final = { + "type": "response.created", + "response": {**completed_body, "status": "in_progress", "output": [], "usage": None}, + } + delta: Final = { + "type": "response.output_text.delta", + "item_id": message, + "output_index": 0, + "content_index": 0, + "delta": "tagged", + } + completed: Final = {"type": "response.completed", "response": completed_body} + return Reply(content_type="text/event-stream", chunks=_sse_frames((created, delta, completed))) + return Reply(body=json.dumps(completed_body).encode()) + if request.target.endswith(":generateContent") or request.target.endswith(":streamGenerateContent"): + return Reply(body=json.dumps(GEMINI_BODY).encode()) + raise AssertionError(f"unexpected upstream target {request.target}") + + +def provider_env(url: str) -> dict[str, str]: + return { + "ANTHROPIC_API_BASE": url, + "ANTHROPIC_API_KEY": "synthetic-anthropic-key", + "OPENAI_API_BASE": url, + "OPENAI_API_KEY": "synthetic-openai-key", + "GEMINI_API_BASE": url, + "GEMINI_API_KEY": "synthetic-gemini-key", + } + + +def write_config(directory: Path, mutations: dict, name: str = "spend-tag-headers.yaml") -> Path: + config: Final = yaml.safe_load(Path("tests/integration/proxy_config.yaml").read_text()) + for section, values in mutations.items(): + if isinstance(values, dict) and isinstance(config.get(section), dict): + config[section].update(values) + else: + config[section] = values + path: Final = directory / name + path.write_text(yaml.safe_dump(config)) + return path + + +def tags_of(row: dict) -> list: + value: Final = row["request_tags"] + return json.loads(value) if isinstance(value, str) else value + + +def tags_by_key(key: str) -> list[list]: + rows: Final = read_rows( + 'SELECT request_tags FROM "LiteLLM_SpendLogs" WHERE api_key=%s', (sha256(key.encode()).hexdigest(),) + ) + return [tags_of(row) for row in rows] + + +def tags_by_id(request_id: str) -> list[list]: + rows: Final = read_rows('SELECT request_tags FROM "LiteLLM_SpendLogs" WHERE request_id=%s', (request_id,)) + return [tags_of(row) for row in rows] + + +def row_by_id(request_id: str) -> list[dict]: + return read_rows( + 'SELECT request_id, call_type, request_tags, status, spend FROM "LiteLLM_SpendLogs" WHERE request_id=%s', + (request_id,), + ) + + +def unique_marker() -> str: + return "tagprobe" + uuid.uuid4().hex[:12] diff --git a/tests/integration/spend/test_spend_log_request_tags.py b/tests/integration/spend/test_spend_log_request_tags.py index 9d89a866cc8..f905af8258b 100644 --- a/tests/integration/spend/test_spend_log_request_tags.py +++ b/tests/integration/spend/test_spend_log_request_tags.py @@ -1,15 +1,26 @@ +import asyncio import json import uuid from hashlib import sha256 from pathlib import Path from typing import Final +import httpx import pytest import yaml from integration._support.client import Gateway, eventually from integration._support.database import read_rows from integration._support.process import owned_proxy from integration._support.wire import Reply, Request, wire_server +from integration.spend._request_tag_helpers import ( + GEMINI_MODEL, + OPENAI_MODEL, + provider_env, + provider_reply, + tags_by_id, + tags_by_key, + write_config, +) MODEL: Final = "claude-sonnet-4-5-20250929" SENT_HEADERS: Final = {"user-agent": "claude-cli/2.0.0", "x-tenant-id": "tenant-a"} @@ -74,3 +85,916 @@ def test_header_derived_spend_tags_are_recorded_on_anthropic_messages_routes( assert response.status_code == 200, response.text assert len(wire.drain()) == 1 assert eventually(lambda: _request_tags(key), lambda tags: len(tags) == 1, seconds=70) == [EXPECTED_TAGS] + + +def _base_url(candidate: Gateway) -> str: + return str(candidate.client.base_url).rstrip("/") + + +def _spend_count() -> int: + return read_rows('SELECT count(*) AS n FROM "LiteLLM_SpendLogs"')[0]["n"] + + +def _owned_config(tmp_path: Path, mutations: dict) -> Path: + return write_config(tmp_path, mutations) + + +UA_TAG: Final = "User-Agent: claude-cli/2.0.0" +UA_FAMILY_TAG: Final = "User-Agent: claude-cli" +TENANT_TAG: Final = "x-tenant-id: tenant-a" + + +# H2: pass-through streaming anthropic request records the header tags +def test_pass_through_anthropic_stream_records_header_tags(gateway: Gateway, tmp_path: Path) -> None: + with wire_server(provider_reply) as wire: + config: Final = _owned_config(tmp_path, {"litellm_settings": {"extra_spend_tag_headers": ["x-tenant-id"]}}) + with ( + owned_proxy(gateway, tmp_path, provider_env(wire.url), config=config, workers=2) as candidate, + candidate.scenario() as scenario, + ): + key: Final = scenario.key() + response: Final = candidate.request( + "POST", + "/anthropic/v1/messages", + { + "model": MODEL, + "max_tokens": 16, + "messages": [{"role": "user", "content": "tag me"}], + "stream": True, + }, + key=key, + headers={**SENT_HEADERS, "anthropic-version": "2023-06-01"}, + ) + assert response.status_code == 200, response.text + assert '"type":"message_start"' in response.text.replace(" ", ""), response.text + request_id: Final = f"msg_{response.text.split('msg_')[1].split(chr(34))[0]}" + assert len(wire.drain()) == 1 + assert eventually(lambda: tags_by_id(request_id), lambda tags: len(tags) == 1, seconds=70) == [ + EXPECTED_TAGS + ] + + +# H3: Anthropic SDK non-stream call through the pass-through route +def test_pass_through_anthropic_sdk_records_header_tags(gateway: Gateway, tmp_path: Path) -> None: + import anthropic + + with wire_server(provider_reply) as wire: + config: Final = _owned_config(tmp_path, {"litellm_settings": {"extra_spend_tag_headers": ["x-tenant-id"]}}) + with ( + owned_proxy(gateway, tmp_path, provider_env(wire.url), config=config, workers=2) as candidate, + candidate.scenario() as scenario, + ): + key: Final = scenario.key() + client: Final = anthropic.Anthropic( + base_url=f"{_base_url(candidate)}/anthropic", auth_token=key, default_headers=SENT_HEADERS + ) + message: Final = client.messages.create( + model=MODEL, max_tokens=16, messages=[{"role": "user", "content": "tag me"}] + ) + assert len(wire.drain()) == 1 + assert eventually(lambda: tags_by_id(message.id), lambda tags: len(tags) == 1, seconds=70) == [ + EXPECTED_TAGS + ] + + +# H4: Anthropic SDK streaming call through the pass-through route +def test_pass_through_anthropic_sdk_stream_records_header_tags(gateway: Gateway, tmp_path: Path) -> None: + import anthropic + + with wire_server(provider_reply) as wire: + config: Final = _owned_config(tmp_path, {"litellm_settings": {"extra_spend_tag_headers": ["x-tenant-id"]}}) + with ( + owned_proxy(gateway, tmp_path, provider_env(wire.url), config=config, workers=2) as candidate, + candidate.scenario() as scenario, + ): + key: Final = scenario.key() + client: Final = anthropic.Anthropic( + base_url=f"{_base_url(candidate)}/anthropic", auth_token=key, default_headers=SENT_HEADERS + ) + with client.messages.stream( + model=MODEL, max_tokens=16, messages=[{"role": "user", "content": "tag me"}] + ) as stream: + message: Final = stream.get_final_message() + assert len(wire.drain()) == 1 + assert eventually(lambda: tags_by_id(message.id), lambda tags: len(tags) == 1, seconds=70) == [ + EXPECTED_TAGS + ] + + +# H5: openai pass-through route records the header tags +@pytest.mark.parametrize("stream", [pytest.param(False, id="sync"), pytest.param(True, id="stream")]) +def test_pass_through_openai_chat_records_header_tags(gateway: Gateway, tmp_path: Path, stream: bool) -> None: + with wire_server(provider_reply) as wire: + config: Final = _owned_config(tmp_path, {"litellm_settings": {"extra_spend_tag_headers": ["x-tenant-id"]}}) + with ( + owned_proxy(gateway, tmp_path, provider_env(wire.url), config=config, workers=2) as candidate, + candidate.scenario() as scenario, + ): + key: Final = scenario.key() + response: Final = candidate.request( + "POST", + "/openai/v1/chat/completions", + { + "model": OPENAI_MODEL, + "messages": [{"role": "user", "content": "tag me"}], + **({"stream": True} if stream else {}), + }, + key=key, + headers=SENT_HEADERS, + ) + assert response.status_code == 200, response.text + assert len(wire.drain()) == 1 + request_id: Final = f"chatcmpl_{response.text.split('chatcmpl_')[1].split(chr(34))[0]}" + assert eventually(lambda: tags_by_id(request_id), lambda tags: len(tags) == 1, seconds=70) == [ + EXPECTED_TAGS + ] + + +# H6: OpenAI SDK sync and stream calls through the pass-through route +@pytest.mark.parametrize("stream", [pytest.param(False, id="sync"), pytest.param(True, id="stream")]) +def test_pass_through_openai_sdk_records_header_tags(gateway: Gateway, tmp_path: Path, stream: bool) -> None: + import openai + + with wire_server(provider_reply) as wire: + config: Final = _owned_config(tmp_path, {"litellm_settings": {"extra_spend_tag_headers": ["x-tenant-id"]}}) + with ( + owned_proxy(gateway, tmp_path, provider_env(wire.url), config=config, workers=2) as candidate, + candidate.scenario() as scenario, + ): + key: Final = scenario.key() + client: Final = openai.OpenAI( + api_key=key, base_url=f"{_base_url(candidate)}/openai/v1", default_headers=SENT_HEADERS + ) + completion: Final = client.chat.completions.create( + model=OPENAI_MODEL, messages=[{"role": "user", "content": "tag me"}], stream=stream + ) + if stream: + request_id: Final = next(chunk.id for chunk in completion) + for _ in completion: + pass + else: + request_id = completion.id + assert len(wire.drain()) == 1 + assert eventually(lambda: tags_by_id(request_id), lambda tags: len(tags) == 1, seconds=70) == [ + EXPECTED_TAGS + ] + + +# H7: gemini pass-through route records the header tags +def test_pass_through_gemini_records_header_tags(gateway: Gateway, tmp_path: Path) -> None: + with wire_server(provider_reply) as wire: + config: Final = _owned_config(tmp_path, {"litellm_settings": {"extra_spend_tag_headers": ["x-tenant-id"]}}) + with ( + owned_proxy(gateway, tmp_path, provider_env(wire.url), config=config, workers=2) as candidate, + candidate.scenario() as scenario, + ): + key: Final = scenario.key() + response: Final = candidate.request( + "POST", + f"/gemini/v1beta/models/{GEMINI_MODEL}:generateContent", + {"contents": [{"parts": [{"text": "tag me"}]}]}, + key=key, + headers=SENT_HEADERS, + ) + assert response.status_code == 200, response.text + assert len(wire.drain()) == 1 + assert eventually(lambda: tags_by_key(key), lambda tags: len(tags) == 1, seconds=70) == [EXPECTED_TAGS] + + +# H8: user-defined pass-through endpoint records the header tags +def test_custom_pass_through_endpoint_records_header_tags(gateway: Gateway, tmp_path: Path) -> None: + with wire_server(provider_reply) as wire: + config: Final = _owned_config( + tmp_path, + { + "litellm_settings": {"extra_spend_tag_headers": ["x-tenant-id"]}, + "general_settings": { + "pass_through_endpoints": [ + { + "path": "/custom-anthropic", + "target": f"{wire.url}/v1/messages", + "auth": True, + "headers": { + "x-api-key": "synthetic-anthropic-key", + "anthropic-version": "2023-06-01", + }, + } + ] + }, + }, + ) + with ( + owned_proxy(gateway, tmp_path, provider_env(wire.url), config=config, workers=2) as candidate, + candidate.scenario() as scenario, + ): + key: Final = scenario.key(allowed_passthrough_routes=["/custom-anthropic"]) + response: Final = candidate.request( + "POST", + "/custom-anthropic", + {"model": MODEL, "max_tokens": 16, "messages": [{"role": "user", "content": "tag me"}]}, + key=key, + headers=SENT_HEADERS, + ) + assert response.status_code == 200, response.text + assert len(wire.drain()) == 1 + assert eventually(lambda: tags_by_key(key), lambda tags: len(tags) == 1, seconds=70) == [EXPECTED_TAGS] + + +# H9: async OpenAI SDK streaming call through the pass-through route +def test_pass_through_openai_async_sdk_stream_records_header_tags(gateway: Gateway, tmp_path: Path) -> None: + import openai + + with wire_server(provider_reply) as wire: + config: Final = _owned_config(tmp_path, {"litellm_settings": {"extra_spend_tag_headers": ["x-tenant-id"]}}) + with ( + owned_proxy(gateway, tmp_path, provider_env(wire.url), config=config, workers=2) as candidate, + candidate.scenario() as scenario, + ): + key: Final = scenario.key() + client: Final = openai.AsyncOpenAI( + api_key=key, base_url=f"{_base_url(candidate)}/openai/v1", default_headers=SENT_HEADERS + ) + + async def call() -> str: + completion: Final = await client.chat.completions.create( + model=OPENAI_MODEL, messages=[{"role": "user", "content": "tag me"}], stream=True + ) + first: Final = await completion.__anext__() + async for _ in completion: + pass + return first.id + + request_id: Final = asyncio.run(call()) + assert len(wire.drain()) == 1 + assert eventually(lambda: tags_by_id(request_id), lambda tags: len(tags) == 1, seconds=70) == [ + EXPECTED_TAGS + ] + + +# H10: unified chat completions control records the header tags +def test_unified_chat_completions_records_header_tags(gateway: Gateway, tmp_path: Path) -> None: + with wire_server(provider_reply) as wire: + config: Final = _owned_config(tmp_path, {"litellm_settings": {"extra_spend_tag_headers": ["x-tenant-id"]}}) + with ( + owned_proxy(gateway, tmp_path, provider_env(wire.url), config=config, workers=2) as candidate, + candidate.scenario() as scenario, + ): + model: Final = scenario.model(model=f"openai/{OPENAI_MODEL}", api_base=f"{wire.url}/v1") + key: Final = scenario.key() + response: Final = candidate.request( + "POST", + "/v1/chat/completions", + {"model": model, "messages": [{"role": "user", "content": "tag me"}]}, + key=key, + headers=SENT_HEADERS, + ) + assert response.status_code == 200, response.text + request_id: Final = response.json()["id"] + assert len(wire.drain()) == 1 + assert eventually(lambda: tags_by_id(request_id), lambda tags: len(tags) == 1, seconds=70) == [ + EXPECTED_TAGS + ] + + +# H11: unified /v1/responses via the OpenAI SDK, sync and stream +@pytest.mark.parametrize("stream", [pytest.param(False, id="sync"), pytest.param(True, id="stream")]) +def test_unified_responses_records_header_tags(gateway: Gateway, tmp_path: Path, stream: bool) -> None: + import openai + + with wire_server(provider_reply) as wire: + config: Final = _owned_config(tmp_path, {"litellm_settings": {"extra_spend_tag_headers": ["x-tenant-id"]}}) + with ( + owned_proxy(gateway, tmp_path, provider_env(wire.url), config=config, workers=2) as candidate, + candidate.scenario() as scenario, + ): + model: Final = scenario.model(model=f"openai/{OPENAI_MODEL}", api_base=f"{wire.url}/v1") + key: Final = scenario.key() + client: Final = openai.OpenAI( + api_key=key, base_url=f"{_base_url(candidate)}/v1", default_headers=SENT_HEADERS + ) + created: Final = client.responses.create(model=model, input="tag me", stream=stream) + if stream: + frames: Final = list(created) + request_id: Final = next(event.response.id for event in frames if event.type == "response.completed") + else: + request_id = created.id + assert len(wire.drain()) == 1 + assert eventually(lambda: tags_by_id(request_id), lambda tags: len(tags) == 1, seconds=70) == [ + EXPECTED_TAGS + ] + + +# H12: a unified cache-hit twin records the same tags on both spend rows +def test_unified_cache_hit_twin_records_header_tags(gateway: Gateway, tmp_path: Path) -> None: + with wire_server(provider_reply) as wire: + config: Final = _owned_config(tmp_path, {"litellm_settings": {"extra_spend_tag_headers": ["x-tenant-id"]}}) + with ( + owned_proxy(gateway, tmp_path, provider_env(wire.url), config=config, workers=2) as candidate, + candidate.scenario() as scenario, + ): + model: Final = scenario.model(model=f"openai/{OPENAI_MODEL}", api_base=f"{wire.url}/v1") + key: Final = scenario.key() + body: Final = {"model": model, "messages": [{"role": "user", "content": f"cache {uuid.uuid4().hex}"}]} + first: Final = candidate.request("POST", "/v1/chat/completions", body, key=key, headers=SENT_HEADERS) + assert first.status_code == 200, first.text + second: Final = candidate.request("POST", "/v1/chat/completions", body, key=key, headers=SENT_HEADERS) + assert second.status_code == 200, second.text + assert len(wire.drain()) == 1, "identical second call should have hit the response cache" + assert eventually(lambda: tags_by_key(key), lambda tags: len(tags) == 2, seconds=70) == [ + EXPECTED_TAGS, + EXPECTED_TAGS, + ] + + +# H13: generic_api callback sink sees the same request_tags as the spend row +def test_pass_through_tags_reach_generic_api_sink(gateway: Gateway, tmp_path: Path) -> None: + def sink(request: Request) -> Reply: + return Reply() + + with wire_server(provider_reply) as wire, wire_server(sink) as endpoint: + config: Final = _owned_config( + tmp_path, + { + "litellm_settings": { + "extra_spend_tag_headers": ["x-tenant-id"], + "callbacks": ["generic_api"], + "DEFAULT_FLUSH_INTERVAL_SECONDS": 1, + } + }, + ) + with ( + owned_proxy( + gateway, + tmp_path, + { + **provider_env(wire.url), + "GENERIC_LOGGER_ENDPOINT": endpoint.url, + "GENERIC_LOGGER_HEADERS": "Authorization=Bearer synthetic-sink-secret", + }, + config=config, + workers=2, + ) as candidate, + candidate.scenario() as scenario, + ): + key: Final = scenario.key() + response: Final = candidate.request( + "POST", + "/anthropic/v1/messages", + {"model": MODEL, "max_tokens": 16, "messages": [{"role": "user", "content": "tag me"}]}, + key=key, + headers={**SENT_HEADERS, "anthropic-version": "2023-06-01"}, + ) + assert response.status_code == 200, response.text + request_id: Final = response.json()["id"] + assert len(wire.drain()) == 1 + batches: Final[list[Request]] = [] # mutable-ok: drain consumes the queue between polls + + def delivered() -> list[dict]: + batches.extend(endpoint.drain()) + return [event for batch in batches for event in json.loads(batch.body) if event.get("id") == request_id] + + events: Final = eventually(delivered, lambda values: len(values) == 1, seconds=30) + assert events[0]["request_tags"] == EXPECTED_TAGS + assert eventually(lambda: tags_by_id(request_id), lambda tags: len(tags) == 1, seconds=70) == [ + EXPECTED_TAGS + ] + + +# H14: LiteLLM_DailyTagSpend accrues each header-derived tag +def test_pass_through_tags_accrue_daily_tag_spend(gateway: Gateway, tmp_path: Path) -> None: + with wire_server(provider_reply) as wire: + config: Final = _owned_config(tmp_path, {"litellm_settings": {"extra_spend_tag_headers": ["x-tenant-id"]}}) + with ( + owned_proxy(gateway, tmp_path, provider_env(wire.url), config=config, workers=2) as candidate, + candidate.scenario() as scenario, + ): + key: Final = scenario.key() + response: Final = candidate.request( + "POST", + "/anthropic/v1/messages", + {"model": MODEL, "max_tokens": 16, "messages": [{"role": "user", "content": "tag me"}]}, + key=key, + headers={**SENT_HEADERS, "anthropic-version": "2023-06-01"}, + ) + assert response.status_code == 200, response.text + request_id: Final = response.json()["id"] + assert len(wire.drain()) == 1 + assert eventually(lambda: tags_by_id(request_id), lambda tags: len(tags) == 1, seconds=70) == [ + EXPECTED_TAGS + ] + rows: Final = eventually( + lambda: read_rows( + 'SELECT tag FROM "LiteLLM_DailyTagSpend" WHERE api_key=%s', (sha256(key.encode()).hexdigest(),) + ), + lambda values: len(values) == 3, + seconds=70, + ) + assert {row["tag"] for row in rows} == set(EXPECTED_TAGS) + + +# S1: extra_spend_tag_headers unset records only the user-agent tags on both routes +@pytest.mark.parametrize( + "route", [pytest.param("/anthropic/v1/messages", id="passthrough"), pytest.param("/v1/messages", id="unified")] +) +def test_header_tags_without_extra_spend_tag_headers_record_user_agent_only( + gateway: Gateway, tmp_path: Path, route: str +) -> None: + with wire_server(provider_reply) as wire: + config: Final = _owned_config(tmp_path, {}) + with ( + owned_proxy(gateway, tmp_path, provider_env(wire.url), config=config, workers=2) as candidate, + candidate.scenario() as scenario, + ): + model: Final = scenario.model( + model=f"anthropic/{MODEL}", api_base=wire.url, api_key="synthetic-anthropic-key" + ) + key: Final = scenario.key() + response: Final = candidate.request( + "POST", + route, + { + "model": MODEL if route == "/anthropic/v1/messages" else model, + "max_tokens": 16, + "messages": [{"role": "user", "content": "tag me"}], + }, + key=key, + headers={**SENT_HEADERS, "anthropic-version": "2023-06-01"}, + ) + assert response.status_code == 200, response.text + assert len(wire.drain()) == 1 + expected: Final = [UA_FAMILY_TAG, UA_TAG] + assert eventually(lambda: tags_by_id(response.json()["id"]), lambda tags: len(tags) == 1, seconds=70) == [ + expected + ] + + +# S2: pass-through request without the headers records no tags +@pytest.mark.parametrize( + "route", [pytest.param("/anthropic/v1/messages", id="passthrough"), pytest.param("/v1/messages", id="unified")] +) +def test_routes_without_headers_record_no_tags(gateway: Gateway, tmp_path: Path, route: str) -> None: + with wire_server(provider_reply) as wire: + config: Final = _owned_config(tmp_path, {"litellm_settings": {"extra_spend_tag_headers": ["x-tenant-id"]}}) + with ( + owned_proxy(gateway, tmp_path, provider_env(wire.url), config=config, workers=2) as candidate, + candidate.scenario() as scenario, + ): + model: Final = scenario.model( + model=f"anthropic/{MODEL}", api_base=wire.url, api_key="synthetic-anthropic-key" + ) + key: Final = scenario.key() + response: Final = candidate.request( + "POST", + route, + { + "model": MODEL if route == "/anthropic/v1/messages" else model, + "max_tokens": 16, + "messages": [{"role": "user", "content": "tag me"}], + }, + key=key, + headers={"anthropic-version": "2023-06-01"}, + ) + assert response.status_code == 200, response.text + assert len(wire.drain()) == 1 + assert eventually(lambda: tags_by_id(response.json()["id"]), lambda tags: len(tags) == 1, seconds=70) == [ + [] + ] + + +# S3: disable_add_user_agent_to_request_tags keeps only the extra header tags +@pytest.mark.parametrize( + "route", [pytest.param("/anthropic/v1/messages", id="passthrough"), pytest.param("/v1/messages", id="unified")] +) +def test_disabled_user_agent_keeps_only_extra_header_tags(gateway: Gateway, tmp_path: Path, route: str) -> None: + with wire_server(provider_reply) as wire: + config: Final = _owned_config( + tmp_path, + { + "litellm_settings": { + "extra_spend_tag_headers": ["x-tenant-id"], + "disable_add_user_agent_to_request_tags": True, + } + }, + ) + with ( + owned_proxy(gateway, tmp_path, provider_env(wire.url), config=config, workers=2) as candidate, + candidate.scenario() as scenario, + ): + model: Final = scenario.model( + model=f"anthropic/{MODEL}", api_base=wire.url, api_key="synthetic-anthropic-key" + ) + key: Final = scenario.key() + response: Final = candidate.request( + "POST", + route, + { + "model": MODEL if route == "/anthropic/v1/messages" else model, + "max_tokens": 16, + "messages": [{"role": "user", "content": "tag me"}], + }, + key=key, + headers={**SENT_HEADERS, "anthropic-version": "2023-06-01"}, + ) + assert response.status_code == 200, response.text + assert len(wire.drain()) == 1 + assert eventually(lambda: tags_by_id(response.json()["id"]), lambda tags: len(tags) == 1, seconds=70) == [ + [TENANT_TAG] + ] + + +# S4: a configured header the client never sends contributes no tag +@pytest.mark.parametrize( + "route", [pytest.param("/anthropic/v1/messages", id="passthrough"), pytest.param("/v1/messages", id="unified")] +) +def test_unsent_configured_header_contributes_no_tag(gateway: Gateway, tmp_path: Path, route: str) -> None: + with wire_server(provider_reply) as wire: + config: Final = _owned_config(tmp_path, {"litellm_settings": {"extra_spend_tag_headers": ["x-never-sent"]}}) + with ( + owned_proxy(gateway, tmp_path, provider_env(wire.url), config=config, workers=2) as candidate, + candidate.scenario() as scenario, + ): + model: Final = scenario.model( + model=f"anthropic/{MODEL}", api_base=wire.url, api_key="synthetic-anthropic-key" + ) + key: Final = scenario.key() + response: Final = candidate.request( + "POST", + route, + { + "model": MODEL if route == "/anthropic/v1/messages" else model, + "max_tokens": 16, + "messages": [{"role": "user", "content": "tag me"}], + }, + key=key, + headers={**SENT_HEADERS, "anthropic-version": "2023-06-01"}, + ) + assert response.status_code == 200, response.text + assert len(wire.drain()) == 1 + assert eventually(lambda: tags_by_id(response.json()["id"]), lambda tags: len(tags) == 1, seconds=70) == [ + [UA_FAMILY_TAG, UA_TAG] + ] + + +# S5: httpx default user-agent is recorded when the client sends none +@pytest.mark.parametrize( + "route", [pytest.param("/anthropic/v1/messages", id="passthrough"), pytest.param("/v1/messages", id="unified")] +) +def test_default_httpx_user_agent_is_recorded(gateway: Gateway, tmp_path: Path, route: str) -> None: + with wire_server(provider_reply) as wire: + config: Final = _owned_config(tmp_path, {"litellm_settings": {"extra_spend_tag_headers": ["x-tenant-id"]}}) + with ( + owned_proxy(gateway, tmp_path, provider_env(wire.url), config=config, workers=2) as candidate, + candidate.scenario() as scenario, + ): + model: Final = scenario.model( + model=f"anthropic/{MODEL}", api_base=wire.url, api_key="synthetic-anthropic-key" + ) + key: Final = scenario.key() + response: Final = candidate.request( + "POST", + route, + { + "model": MODEL if route == "/anthropic/v1/messages" else model, + "max_tokens": 16, + "messages": [{"role": "user", "content": "tag me"}], + }, + key=key, + headers={"anthropic-version": "2023-06-01"}, + ) + assert response.status_code == 200, response.text + assert len(wire.drain()) == 1 + expected: Final = ["User-Agent: python-httpx", f"User-Agent: python-httpx/{httpx.__version__}"] + assert eventually(lambda: tags_by_id(response.json()["id"]), lambda tags: len(tags) == 1, seconds=70) == [ + expected + ] + + +# S6: unauthenticated requests return 401 and write no spend row +@pytest.mark.parametrize("route", [pytest.param("/anthropic/v1/messages", id="passthrough")]) +def test_unauthenticated_request_writes_no_spend_row(gateway: Gateway, tmp_path: Path, route: str) -> None: + with wire_server(provider_reply) as wire: + config: Final = _owned_config(tmp_path, {"litellm_settings": {"extra_spend_tag_headers": ["x-tenant-id"]}}) + with ( + owned_proxy(gateway, tmp_path, provider_env(wire.url), config=config, workers=2) as candidate, + candidate.scenario() as scenario, + ): + before: Final = _spend_count() + response: Final = candidate.client.post( + route, + json={"model": MODEL, "max_tokens": 16, "messages": [{"role": "user", "content": "tag me"}]}, + headers={**SENT_HEADERS, "anthropic-version": "2023-06-01"}, + ) + assert response.status_code == 401, response.text + key: Final = scenario.key() + control: Final = candidate.request( + "POST", + route, + {"model": MODEL, "max_tokens": 16, "messages": [{"role": "user", "content": "tag me"}]}, + key=key, + headers={**SENT_HEADERS, "anthropic-version": "2023-06-01"}, + ) + assert control.status_code == 200, control.text + assert len(wire.drain()) == 1 + assert eventually(lambda: tags_by_id(control.json()["id"]), lambda tags: len(tags) == 1, seconds=70) == [ + EXPECTED_TAGS + ] + assert _spend_count() == before + 1 + + +# S7: an upstream 400 surfaces the same status and its spend row records the tags +@pytest.mark.parametrize( + "route", [pytest.param("/anthropic/v1/messages", id="passthrough"), pytest.param("/v1/messages", id="unified")] +) +def test_upstream_failure_still_records_header_tags(gateway: Gateway, tmp_path: Path, route: str) -> None: + with wire_server(provider_reply) as wire: + config: Final = _owned_config(tmp_path, {"litellm_settings": {"extra_spend_tag_headers": ["x-tenant-id"]}}) + with ( + owned_proxy(gateway, tmp_path, provider_env(wire.url), config=config, workers=2) as candidate, + candidate.scenario() as scenario, + ): + model: Final = scenario.model( + model=f"anthropic/{MODEL}", api_base=wire.url, api_key="synthetic-anthropic-key" + ) + key: Final = scenario.key() + response: Final = candidate.request( + "POST", + route, + { + "model": "claude-nonexistent-model" if route == "/anthropic/v1/messages" else model, + "max_tokens": 16, + "messages": [{"role": "user", "content": "tag me"}], + }, + key=key, + headers={**SENT_HEADERS, "anthropic-version": "2023-06-01"}, + ) + assert response.status_code == 400, response.text + assert len(wire.drain()) == 1 + assert eventually(lambda: tags_by_key(key), lambda tags: len(tags) == 1, seconds=70) == [EXPECTED_TAGS] + + +# S8: null and empty extra_spend_tag_headers behave like unset +@pytest.mark.parametrize("extra", [pytest.param(None, id="null"), pytest.param([], id="empty")]) +@pytest.mark.parametrize( + "route", [pytest.param("/anthropic/v1/messages", id="passthrough"), pytest.param("/v1/messages", id="unified")] +) +def test_null_and_empty_extra_spend_tag_headers_record_user_agent_only( + gateway: Gateway, tmp_path: Path, route: str, extra: object +) -> None: + with wire_server(provider_reply) as wire: + config: Final = _owned_config(tmp_path, {"litellm_settings": {"extra_spend_tag_headers": extra}}) + with ( + owned_proxy(gateway, tmp_path, provider_env(wire.url), config=config, workers=2) as candidate, + candidate.scenario() as scenario, + ): + model: Final = scenario.model( + model=f"anthropic/{MODEL}", api_base=wire.url, api_key="synthetic-anthropic-key" + ) + key: Final = scenario.key() + response: Final = candidate.request( + "POST", + route, + { + "model": MODEL if route == "/anthropic/v1/messages" else model, + "max_tokens": 16, + "messages": [{"role": "user", "content": "tag me"}], + }, + key=key, + headers={**SENT_HEADERS, "anthropic-version": "2023-06-01"}, + ) + assert response.status_code == 200, response.text + assert len(wire.drain()) == 1 + assert eventually(lambda: tags_by_id(response.json()["id"]), lambda tags: len(tags) == 1, seconds=70) == [ + [UA_FAMILY_TAG, UA_TAG] + ] + + +# S9: pass-through matches the configured header name case-insensitively, unified is case-sensitive +def test_configured_header_case_differs_between_routes(gateway: Gateway, tmp_path: Path) -> None: + with wire_server(provider_reply) as wire: + config: Final = _owned_config(tmp_path, {"litellm_settings": {"extra_spend_tag_headers": ["X-Tenant-Id"]}}) + with ( + owned_proxy(gateway, tmp_path, provider_env(wire.url), config=config, workers=2) as candidate, + candidate.scenario() as scenario, + ): + model: Final = scenario.model( + model=f"anthropic/{MODEL}", api_base=wire.url, api_key="synthetic-anthropic-key" + ) + key: Final = scenario.key() + body: Final = {"model": MODEL, "max_tokens": 16, "messages": [{"role": "user", "content": "tag me"}]} + passthrough: Final = candidate.request( + "POST", + "/anthropic/v1/messages", + body, + key=key, + headers={**SENT_HEADERS, "anthropic-version": "2023-06-01"}, + ) + assert passthrough.status_code == 200, passthrough.text + unified: Final = candidate.request( + "POST", + "/v1/messages", + {**body, "model": model}, + key=key, + headers={**SENT_HEADERS, "anthropic-version": "2023-06-01"}, + ) + assert unified.status_code == 200, unified.text + assert len(wire.drain()) == 2 + rows: Final = eventually(lambda: tags_by_key(key), lambda tags: len(tags) == 2, seconds=70) + passthrough_tags: Final = tags_by_id(passthrough.json()["id"])[0] + unified_tags: Final = tags_by_id(unified.json()["id"])[0] + assert passthrough_tags == [UA_FAMILY_TAG, UA_TAG, "X-Tenant-Id: tenant-a"], rows + assert unified_tags == [UA_FAMILY_TAG, UA_TAG], rows + + +# E1: a 5KB header value is stored verbatim +@pytest.mark.parametrize( + "route", [pytest.param("/anthropic/v1/messages", id="passthrough"), pytest.param("/v1/messages", id="unified")] +) +def test_large_header_value_is_stored_verbatim(gateway: Gateway, tmp_path: Path, route: str) -> None: + big: Final = "x" * 5000 + with wire_server(provider_reply) as wire: + config: Final = _owned_config(tmp_path, {"litellm_settings": {"extra_spend_tag_headers": ["x-tenant-id"]}}) + with ( + owned_proxy(gateway, tmp_path, provider_env(wire.url), config=config, workers=2) as candidate, + candidate.scenario() as scenario, + ): + model: Final = scenario.model( + model=f"anthropic/{MODEL}", api_base=wire.url, api_key="synthetic-anthropic-key" + ) + key: Final = scenario.key() + response: Final = candidate.request( + "POST", + route, + { + "model": MODEL if route == "/anthropic/v1/messages" else model, + "max_tokens": 16, + "messages": [{"role": "user", "content": "tag me"}], + }, + key=key, + headers={ + "user-agent": "claude-cli/2.0.0", + "x-tenant-id": big, + "anthropic-version": "2023-06-01", + }, + ) + assert response.status_code == 200, response.text + assert len(wire.drain()) == 1 + assert eventually(lambda: tags_by_id(response.json()["id"]), lambda tags: len(tags) == 1, seconds=70) == [ + [UA_FAMILY_TAG, UA_TAG, f"x-tenant-id: {big}"] + ] + + +# E2: duplicate configured headers record first-value on pass-through, last-value on unified +def test_duplicate_header_values_follow_carrier_semantics(gateway: Gateway, tmp_path: Path) -> None: + with wire_server(provider_reply) as wire: + config: Final = _owned_config(tmp_path, {"litellm_settings": {"extra_spend_tag_headers": ["x-tenant-id"]}}) + with ( + owned_proxy(gateway, tmp_path, provider_env(wire.url), config=config, workers=2) as candidate, + candidate.scenario() as scenario, + ): + model: Final = scenario.model( + model=f"anthropic/{MODEL}", api_base=wire.url, api_key="synthetic-anthropic-key" + ) + key: Final = scenario.key() + body: Final = {"model": MODEL, "max_tokens": 16, "messages": [{"role": "user", "content": "tag me"}]} + duplicated: Final = [ + ("Authorization", f"Bearer {key}"), + ("user-agent", "claude-cli/2.0.0"), + ("anthropic-version", "2023-06-01"), + ("x-tenant-id", "t1"), + ("x-tenant-id", "t2"), + ] + passthrough: Final = candidate.client.post("/anthropic/v1/messages", json=body, headers=duplicated) + assert passthrough.status_code == 200, passthrough.text + unified: Final = candidate.client.post("/v1/messages", json={**body, "model": model}, headers=duplicated) + assert unified.status_code == 200, unified.text + assert len(wire.drain()) == 2 + eventually(lambda: tags_by_key(key), lambda tags: len(tags) == 2, seconds=70) + assert tags_by_id(passthrough.json()["id"])[0] == [UA_FAMILY_TAG, UA_TAG, "x-tenant-id: t1"] + assert tags_by_id(unified.json()["id"])[0] == [UA_FAMILY_TAG, UA_TAG, "x-tenant-id: t2"] + + +# E4: x-litellm-tags and header-derived tags land together +@pytest.mark.parametrize( + "route", [pytest.param("/anthropic/v1/messages", id="passthrough"), pytest.param("/v1/messages", id="unified")] +) +def test_x_litellm_tags_merges_with_header_tags(gateway: Gateway, tmp_path: Path, route: str) -> None: + with wire_server(provider_reply) as wire: + config: Final = _owned_config(tmp_path, {"litellm_settings": {"extra_spend_tag_headers": ["x-tenant-id"]}}) + with ( + owned_proxy(gateway, tmp_path, provider_env(wire.url), config=config, workers=2) as candidate, + candidate.scenario() as scenario, + ): + model: Final = scenario.model( + model=f"anthropic/{MODEL}", api_base=wire.url, api_key="synthetic-anthropic-key" + ) + key: Final = scenario.key() + response: Final = candidate.request( + "POST", + route, + { + "model": MODEL if route == "/anthropic/v1/messages" else model, + "max_tokens": 16, + "messages": [{"role": "user", "content": "tag me"}], + }, + key=key, + headers={**SENT_HEADERS, "anthropic-version": "2023-06-01", "x-litellm-tags": "team-x"}, + ) + assert response.status_code == 200, response.text + assert len(wire.drain()) == 1 + expected: Final = ( + ["team-x", UA_FAMILY_TAG, UA_TAG, TENANT_TAG] + if route == "/v1/messages" + else [UA_FAMILY_TAG, UA_TAG, TENANT_TAG, "team-x"] + ) + assert eventually(lambda: tags_by_id(response.json()["id"]), lambda tags: len(tags) == 1, seconds=70) == [ + expected + ] + + +# E5: three identical requests write three spend rows, each with the tags +@pytest.mark.parametrize( + "route", [pytest.param("/anthropic/v1/messages", id="passthrough"), pytest.param("/v1/messages", id="unified")] +) +def test_repeated_requests_each_record_tags(gateway: Gateway, tmp_path: Path, route: str) -> None: + with wire_server(provider_reply) as wire: + config: Final = _owned_config(tmp_path, {"litellm_settings": {"extra_spend_tag_headers": ["x-tenant-id"]}}) + with ( + owned_proxy(gateway, tmp_path, provider_env(wire.url), config=config, workers=2) as candidate, + candidate.scenario() as scenario, + ): + model: Final = scenario.model( + model=f"anthropic/{MODEL}", api_base=wire.url, api_key="synthetic-anthropic-key" + ) + key: Final = scenario.key() + body: Final = { + "model": MODEL if route == "/anthropic/v1/messages" else model, + "max_tokens": 16, + "messages": [{"role": "user", "content": f"repeat {uuid.uuid4().hex}"}], + } + ids: Final = [] + for _ in range(3): + response: Final = candidate.request( + "POST", route, body, key=key, headers={**SENT_HEADERS, "anthropic-version": "2023-06-01"} + ) + assert response.status_code == 200, response.text + ids.append(response.json()["id"]) + assert len(wire.drain()) == 3 + for identity in ids: + assert eventually( + lambda identity=identity: tags_by_id(identity), lambda tags: len(tags) == 1, seconds=70 + ) == [EXPECTED_TAGS] + + +# E6: guardrail mode-by-tag decider behaves identically with and without the fix +def test_guardrail_mode_tag_decider_is_unchanged_on_pass_through(gateway: Gateway, tmp_path: Path) -> None: + with wire_server(provider_reply) as wire: + config: Final = _owned_config( + tmp_path, + { + "litellm_settings": {"extra_spend_tag_headers": ["x-tenant-id"]}, + "general_settings": { + "pass_through_endpoints": [ + { + "path": "/custom-anthropic", + "target": f"{wire.url}/v1/messages", + "auth": True, + "guardrails": ["tag-blocker"], + "headers": { + "x-api-key": "synthetic-anthropic-key", + "anthropic-version": "2023-06-01", + }, + } + ] + }, + "guardrails": [ + { + "guardrail_name": "tag-blocker", + "litellm_params": { + "guardrail": "litellm_content_filter", + "blocked_words": [{"keyword": "bananablock", "action": "BLOCK"}], + "mode": {"tags": {UA_FAMILY_TAG: "pre_call"}, "default": "post_call"}, + "default_on": True, + }, + } + ], + }, + ) + with ( + owned_proxy(gateway, tmp_path, provider_env(wire.url), config=config, workers=2) as candidate, + candidate.scenario() as scenario, + ): + model: Final = scenario.model(model=f"openai/{OPENAI_MODEL}", api_base=f"{wire.url}/v1") + key: Final = scenario.key(allowed_passthrough_routes=["/custom-anthropic"]) + body: Final = {"model": MODEL, "max_tokens": 16, "messages": [{"role": "user", "content": "bananablock"}]} + passthrough: Final = candidate.request("POST", "/custom-anthropic", body, key=key, headers=SENT_HEADERS) + assert passthrough.status_code == 200, passthrough.text + assert len(wire.drain()) == 1 + control: Final = candidate.request( + "POST", + "/v1/chat/completions", + {"model": model, "messages": [{"role": "user", "content": "bananablock"}]}, + key=key, + headers=SENT_HEADERS, + ) + assert control.status_code != 200, control.text + assert len(wire.drain()) == 1, "tag-matched guardrail should have blocked before the upstream" diff --git a/tests/integration/spend/test_spend_log_request_tags_chaos.py b/tests/integration/spend/test_spend_log_request_tags_chaos.py new file mode 100644 index 00000000000..cd4f8e16691 --- /dev/null +++ b/tests/integration/spend/test_spend_log_request_tags_chaos.py @@ -0,0 +1,209 @@ +import json +import threading +from concurrent.futures import ThreadPoolExecutor +from pathlib import Path +from typing import Final + +import psutil +from integration._support.client import Gateway, eventually +from integration._support.process import owned_proxy, owned_proxy_process +from integration._support.wire import Reply, Request, wire_server +from integration.spend._request_tag_helpers import ( + OPENAI_MODEL, + T3, + provider_env, + provider_reply, + write_config, +) + +from tests.integration._support.database import read_rows + +MODEL: Final = "claude-sonnet-4-5-20250929" +HEADERS: Final = {"user-agent": "claude-cli/2.0.0", "x-tenant-id": "tenant-a"} +ANTHROPIC_HEADERS: Final = {**HEADERS, "anthropic-version": "2023-06-01"} +EXPECTED: Final = T3 + + +def _ids(response) -> str: + for prefix in ("msg_", "chatcmpl_", "resp_"): + if prefix in response.text: + return prefix + response.text.split(prefix)[1].split('"')[0] + raise AssertionError(f"no upstream id in {response.text[:200]}") + + +def _tagged_requests(candidate: Gateway, key: str, model: str, stream: bool, index: int) -> tuple: + """One call per route shape, all with the same client headers; returns (response, request_id).""" + marker: Final = f"burst {index}" + anthropic: Final = candidate.request( + "POST", + "/anthropic/v1/messages", + {"model": MODEL, "max_tokens": 16, "messages": [{"role": "user", "content": marker}], "stream": stream}, + key=key, + headers=ANTHROPIC_HEADERS, + ) + openai: Final = candidate.request( + "POST", + "/openai/v1/chat/completions", + {"model": OPENAI_MODEL, "messages": [{"role": "user", "content": marker}], "stream": stream}, + key=key, + headers=HEADERS, + ) + unified: Final = candidate.request( + "POST", + "/v1/chat/completions", + {"model": model, "messages": [{"role": "user", "content": marker}]}, + key=key, + headers=HEADERS, + ) + return anthropic, openai, unified + + +# C1: 10 concurrent bursts x 3 routes; each response id lands exactly one spend row with the tags +def test_burst_across_routes_records_tags_once_per_response(gateway: Gateway, tmp_path: Path) -> None: + with wire_server(provider_reply) as wire: + config: Final = write_config(tmp_path, {"litellm_settings": {"extra_spend_tag_headers": ["x-tenant-id"]}}) + with ( + owned_proxy(gateway, tmp_path, provider_env(wire.url), config=config, workers=2) as candidate, + candidate.scenario() as scenario, + ): + model: Final = scenario.model(model=f"openai/{OPENAI_MODEL}", api_base=f"{wire.url}/v1") + key: Final = scenario.key() + + def burst(index: int) -> tuple: + return _tagged_requests(candidate, key, model, stream=index % 2 == 1, index=index) + + with ThreadPoolExecutor(max_workers=10) as pool: + responses: Final = [response for group in pool.map(burst, range(10)) for response in group] + assert len(responses) == 30 + assert all(response.status_code == 200 for response in responses), [ + (response.status_code, response.text[:200]) for response in responses + ] + ids: Final = [_ids(response) for response in responses] + assert len(set(ids)) == 30, "duplicate upstream id in burst" + assert len(wire.drain()) == 30 + landed: Final = eventually( + lambda: read_rows( + 'SELECT request_id, request_tags FROM "LiteLLM_SpendLogs" WHERE request_id = ANY(%s)', + (ids,), + ), + lambda values: len(values) == 30, + seconds=70, + ) + assert sorted(row["request_id"] for row in landed) == sorted(ids) + for row in landed: + value: Final = row["request_tags"] + assert (json.loads(value) if isinstance(value, str) else value) == EXPECTED + + +# C2: generic_api sink down mid burst; spend rows still land exactly once with the tags +def test_sink_outage_does_not_lose_spend_log_tags(gateway: Gateway, tmp_path: Path) -> None: + stopped: Final = threading.Event() + + def stoppable_sink(request: Request) -> Reply: + stopped.wait(timeout=30) + return Reply(status=503) + + with wire_server(provider_reply) as wire, wire_server(stoppable_sink) as endpoint: + config: Final = write_config( + tmp_path, + { + "litellm_settings": { + "extra_spend_tag_headers": ["x-tenant-id"], + "callbacks": ["generic_api"], + "DEFAULT_FLUSH_INTERVAL_SECONDS": 1, + } + }, + ) + with ( + owned_proxy( + gateway, + tmp_path, + { + **provider_env(wire.url), + "GENERIC_LOGGER_ENDPOINT": endpoint.url, + "GENERIC_LOGGER_HEADERS": "Authorization=Bearer synthetic-sink-secret", + }, + config=config, + workers=2, + ) as candidate, + candidate.scenario() as scenario, + ): + model: Final = scenario.model(model=f"openai/{OPENAI_MODEL}", api_base=f"{wire.url}/v1") + key: Final = scenario.key() + + def burst(index: int) -> tuple: + return _tagged_requests(candidate, key, model, stream=False, index=index) + + with ThreadPoolExecutor(max_workers=6) as pool: + first: Final = [response for group in pool.map(burst, range(6)) for response in group] + stopped.set() # sink goes down: the peer now returns 503 to every flush + + def second_burst(index: int) -> tuple: + return _tagged_requests(candidate, key, model, stream=False, index=100 + index) + + with ThreadPoolExecutor(max_workers=6) as pool: + second: Final = [response for group in pool.map(second_burst, range(6)) for response in group] + responses: Final = [*first, *second] + assert all(response.status_code == 200 for response in responses), [ + (response.status_code, response.text[:200]) for response in responses + ] + ids: Final = [_ids(response) for response in responses] + landed: Final = eventually( + lambda: read_rows( + 'SELECT request_id, request_tags FROM "LiteLLM_SpendLogs" WHERE request_id = ANY(%s)', + (ids,), + ), + lambda values: len(values) == 36, + seconds=70, + ) + for row in landed: + value: Final = row["request_tags"] + assert (json.loads(value) if isinstance(value, str) else value) == EXPECTED + + +# C3: killing one proxy worker mid burst loses no spend row +def test_worker_kill_mid_burst_loses_no_spend_rows(gateway: Gateway, tmp_path: Path) -> None: + with wire_server(provider_reply) as wire: + config: Final = write_config(tmp_path, {"litellm_settings": {"extra_spend_tag_headers": ["x-tenant-id"]}}) + with ( + owned_proxy_process(gateway, tmp_path, provider_env(wire.url), config=config, workers=2) as owned, + owned.gateway.scenario() as scenario, + ): + candidate: Final = owned.gateway + model: Final = scenario.model(model=f"openai/{OPENAI_MODEL}", api_base=f"{wire.url}/v1") + key: Final = scenario.key() + + def burst(index: int) -> tuple: + return _tagged_requests(candidate, key, model, stream=False, index=index) + + with ThreadPoolExecutor(max_workers=6) as pool: + first: Final = [response for group in pool.map(burst, range(6)) for response in group] + + workers: Final = [ + child + for child in psutil.Process(owned.process.pid).children(recursive=True) + if child.status() != psutil.STATUS_ZOMBIE + ] + assert len(workers) >= 2, f"expected two proxy workers, found {[w.pid for w in workers]}" + workers[0].kill() + + def second_burst(index: int) -> tuple: + return _tagged_requests(candidate, key, model, stream=False, index=100 + index) + + with ThreadPoolExecutor(max_workers=6) as pool: + second: Final = [response for group in pool.map(second_burst, range(6)) for response in group] + responses: Final = [*first, *second] + ok: Final = [response for response in responses if response.status_code == 200] + ids: Final = [_ids(response) for response in ok] + landed: Final = eventually( + lambda: read_rows( + 'SELECT request_id, request_tags FROM "LiteLLM_SpendLogs" WHERE request_id = ANY(%s)', + (ids,), + ), + lambda values: len(values) == len(ids), + seconds=70, + ) + assert sorted(row["request_id"] for row in landed) == sorted(ids) + for row in landed: + value: Final = row["request_tags"] + assert (json.loads(value) if isinstance(value, str) else value) == EXPECTED From 50c800314e6957008671fb22bc87979d3b50613f Mon Sep 17 00:00:00 2001 From: kerry Date: Fri, 2 Oct 2026 18:54:17 +0000 Subject: [PATCH 04/23] test(spend): pin audit cells to observed head behavior and fix harness issues Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- .../integration/spend/_request_tag_helpers.py | 2 + .../spend/test_spend_log_request_tags.py | 154 +++++++++++------- .../test_spend_log_request_tags_chaos.py | 20 ++- 3 files changed, 106 insertions(+), 70 deletions(-) diff --git a/tests/integration/spend/_request_tag_helpers.py b/tests/integration/spend/_request_tag_helpers.py index 6bf72f594c5..c7871862c11 100644 --- a/tests/integration/spend/_request_tag_helpers.py +++ b/tests/integration/spend/_request_tag_helpers.py @@ -123,6 +123,8 @@ def provider_reply(request: Request) -> Reply: {"error": {"type": "invalid_request_error", "message": "model: claude-nonexistent-model"}} ).encode(), ) + if request.target == "/v1/models" or request.target.startswith("/v1/models/"): + return Reply(body=json.dumps({"object": "list", "data": []}).encode()) if request.target == "/v1/messages": identity: Final = _message_id() if body.get("stream") is True: diff --git a/tests/integration/spend/test_spend_log_request_tags.py b/tests/integration/spend/test_spend_log_request_tags.py index f905af8258b..cb57496e52e 100644 --- a/tests/integration/spend/test_spend_log_request_tags.py +++ b/tests/integration/spend/test_spend_log_request_tags.py @@ -77,7 +77,7 @@ def test_header_derived_spend_tags_are_recorded_on_anthropic_messages_routes( { "model": MODEL if route == "/anthropic/v1/messages" else model, "max_tokens": 16, - "messages": [{"role": "user", "content": "tag me"}], + "messages": [{"role": "user", "content": f"tag me {uuid.uuid4().hex}"}], }, key=key, headers={**SENT_HEADERS, "anthropic-version": "2023-06-01"}, @@ -92,7 +92,7 @@ def _base_url(candidate: Gateway) -> str: def _spend_count() -> int: - return read_rows('SELECT count(*) AS n FROM "LiteLLM_SpendLogs"')[0]["n"] + return read_rows('SELECT count(*) AS n FROM "LiteLLM_SpendLogs"', ())[0]["n"] def _owned_config(tmp_path: Path, mutations: dict) -> Path: @@ -119,7 +119,7 @@ def test_pass_through_anthropic_stream_records_header_tags(gateway: Gateway, tmp { "model": MODEL, "max_tokens": 16, - "messages": [{"role": "user", "content": "tag me"}], + "messages": [{"role": "user", "content": f"tag me {uuid.uuid4().hex}"}], "stream": True, }, key=key, @@ -127,11 +127,8 @@ def test_pass_through_anthropic_stream_records_header_tags(gateway: Gateway, tmp ) assert response.status_code == 200, response.text assert '"type":"message_start"' in response.text.replace(" ", ""), response.text - request_id: Final = f"msg_{response.text.split('msg_')[1].split(chr(34))[0]}" assert len(wire.drain()) == 1 - assert eventually(lambda: tags_by_id(request_id), lambda tags: len(tags) == 1, seconds=70) == [ - EXPECTED_TAGS - ] + assert eventually(lambda: tags_by_key(key), lambda tags: len(tags) == 1, seconds=70) == [EXPECTED_TAGS] # H3: Anthropic SDK non-stream call through the pass-through route @@ -149,11 +146,12 @@ def test_pass_through_anthropic_sdk_records_header_tags(gateway: Gateway, tmp_pa base_url=f"{_base_url(candidate)}/anthropic", auth_token=key, default_headers=SENT_HEADERS ) message: Final = client.messages.create( - model=MODEL, max_tokens=16, messages=[{"role": "user", "content": "tag me"}] + model=MODEL, max_tokens=16, messages=[{"role": "user", "content": f"tag me {uuid.uuid4().hex}"}] ) + assert message.id.startswith("msg_") assert len(wire.drain()) == 1 - assert eventually(lambda: tags_by_id(message.id), lambda tags: len(tags) == 1, seconds=70) == [ - EXPECTED_TAGS + assert eventually(lambda: tags_by_key(key), lambda tags: len(tags) == 1, seconds=70) == [ + ["User-Agent: Anthropic", "User-Agent: Anthropic/Python 0.84.0", TENANT_TAG] ] @@ -172,12 +170,13 @@ def test_pass_through_anthropic_sdk_stream_records_header_tags(gateway: Gateway, base_url=f"{_base_url(candidate)}/anthropic", auth_token=key, default_headers=SENT_HEADERS ) with client.messages.stream( - model=MODEL, max_tokens=16, messages=[{"role": "user", "content": "tag me"}] + model=MODEL, max_tokens=16, messages=[{"role": "user", "content": f"tag me {uuid.uuid4().hex}"}] ) as stream: message: Final = stream.get_final_message() + assert message.id.startswith("msg_") assert len(wire.drain()) == 1 - assert eventually(lambda: tags_by_id(message.id), lambda tags: len(tags) == 1, seconds=70) == [ - EXPECTED_TAGS + assert eventually(lambda: tags_by_key(key), lambda tags: len(tags) == 1, seconds=70) == [ + ["User-Agent: Anthropic", "User-Agent: Anthropic/Python 0.84.0", TENANT_TAG] ] @@ -196,18 +195,16 @@ def test_pass_through_openai_chat_records_header_tags(gateway: Gateway, tmp_path "/openai/v1/chat/completions", { "model": OPENAI_MODEL, - "messages": [{"role": "user", "content": "tag me"}], + "messages": [{"role": "user", "content": f"tag me {uuid.uuid4().hex}"}], **({"stream": True} if stream else {}), }, key=key, headers=SENT_HEADERS, ) assert response.status_code == 200, response.text + assert "chatcmpl_" in response.text, response.text assert len(wire.drain()) == 1 - request_id: Final = f"chatcmpl_{response.text.split('chatcmpl_')[1].split(chr(34))[0]}" - assert eventually(lambda: tags_by_id(request_id), lambda tags: len(tags) == 1, seconds=70) == [ - EXPECTED_TAGS - ] + assert eventually(lambda: tags_by_key(key), lambda tags: len(tags) == 1, seconds=70) == [EXPECTED_TAGS] # H6: OpenAI SDK sync and stream calls through the pass-through route @@ -226,7 +223,7 @@ def test_pass_through_openai_sdk_records_header_tags(gateway: Gateway, tmp_path: api_key=key, base_url=f"{_base_url(candidate)}/openai/v1", default_headers=SENT_HEADERS ) completion: Final = client.chat.completions.create( - model=OPENAI_MODEL, messages=[{"role": "user", "content": "tag me"}], stream=stream + model=OPENAI_MODEL, messages=[{"role": "user", "content": f"tag me {uuid.uuid4().hex}"}], stream=stream ) if stream: request_id: Final = next(chunk.id for chunk in completion) @@ -234,9 +231,10 @@ def test_pass_through_openai_sdk_records_header_tags(gateway: Gateway, tmp_path: pass else: request_id = completion.id + assert request_id.startswith("chatcmpl_") assert len(wire.drain()) == 1 - assert eventually(lambda: tags_by_id(request_id), lambda tags: len(tags) == 1, seconds=70) == [ - EXPECTED_TAGS + assert eventually(lambda: tags_by_key(key), lambda tags: len(tags) == 1, seconds=70) == [ + ["User-Agent: OpenAI", "User-Agent: OpenAI/Python 2.33.0", TENANT_TAG] ] @@ -252,9 +250,9 @@ def test_pass_through_gemini_records_header_tags(gateway: Gateway, tmp_path: Pat response: Final = candidate.request( "POST", f"/gemini/v1beta/models/{GEMINI_MODEL}:generateContent", - {"contents": [{"parts": [{"text": "tag me"}]}]}, + {"contents": [{"parts": [{"text": f"tag me {uuid.uuid4().hex}"}]}]}, key=key, - headers=SENT_HEADERS, + headers={**SENT_HEADERS, "x-goog-api-key": key}, ) assert response.status_code == 200, response.text assert len(wire.drain()) == 1 @@ -291,7 +289,11 @@ def test_custom_pass_through_endpoint_records_header_tags(gateway: Gateway, tmp_ response: Final = candidate.request( "POST", "/custom-anthropic", - {"model": MODEL, "max_tokens": 16, "messages": [{"role": "user", "content": "tag me"}]}, + { + "model": MODEL, + "max_tokens": 16, + "messages": [{"role": "user", "content": f"tag me {uuid.uuid4().hex}"}], + }, key=key, headers=SENT_HEADERS, ) @@ -317,7 +319,9 @@ def test_pass_through_openai_async_sdk_stream_records_header_tags(gateway: Gatew async def call() -> str: completion: Final = await client.chat.completions.create( - model=OPENAI_MODEL, messages=[{"role": "user", "content": "tag me"}], stream=True + model=OPENAI_MODEL, + messages=[{"role": "user", "content": f"tag me {uuid.uuid4().hex}"}], + stream=True, ) first: Final = await completion.__anext__() async for _ in completion: @@ -325,9 +329,10 @@ def test_pass_through_openai_async_sdk_stream_records_header_tags(gateway: Gatew return first.id request_id: Final = asyncio.run(call()) + assert request_id.startswith("chatcmpl_") assert len(wire.drain()) == 1 - assert eventually(lambda: tags_by_id(request_id), lambda tags: len(tags) == 1, seconds=70) == [ - EXPECTED_TAGS + assert eventually(lambda: tags_by_key(key), lambda tags: len(tags) == 1, seconds=70) == [ + ["User-Agent: AsyncOpenAI", "User-Agent: AsyncOpenAI/Python 2.33.0", TENANT_TAG] ] @@ -344,7 +349,7 @@ def test_unified_chat_completions_records_header_tags(gateway: Gateway, tmp_path response: Final = candidate.request( "POST", "/v1/chat/completions", - {"model": model, "messages": [{"role": "user", "content": "tag me"}]}, + {"model": model, "messages": [{"role": "user", "content": f"tag me {uuid.uuid4().hex}"}]}, key=key, headers=SENT_HEADERS, ) @@ -378,10 +383,9 @@ def test_unified_responses_records_header_tags(gateway: Gateway, tmp_path: Path, request_id: Final = next(event.response.id for event in frames if event.type == "response.completed") else: request_id = created.id + assert request_id.startswith("resp_") assert len(wire.drain()) == 1 - assert eventually(lambda: tags_by_id(request_id), lambda tags: len(tags) == 1, seconds=70) == [ - EXPECTED_TAGS - ] + assert eventually(lambda: tags_by_key(key), lambda tags: len(tags) == 1, seconds=70) == [EXPECTED_TAGS] # H12: a unified cache-hit twin records the same tags on both spend rows @@ -440,7 +444,11 @@ def test_pass_through_tags_reach_generic_api_sink(gateway: Gateway, tmp_path: Pa response: Final = candidate.request( "POST", "/anthropic/v1/messages", - {"model": MODEL, "max_tokens": 16, "messages": [{"role": "user", "content": "tag me"}]}, + { + "model": MODEL, + "max_tokens": 16, + "messages": [{"role": "user", "content": f"tag me {uuid.uuid4().hex}"}], + }, key=key, headers={**SENT_HEADERS, "anthropic-version": "2023-06-01"}, ) @@ -472,7 +480,11 @@ def test_pass_through_tags_accrue_daily_tag_spend(gateway: Gateway, tmp_path: Pa response: Final = candidate.request( "POST", "/anthropic/v1/messages", - {"model": MODEL, "max_tokens": 16, "messages": [{"role": "user", "content": "tag me"}]}, + { + "model": MODEL, + "max_tokens": 16, + "messages": [{"role": "user", "content": f"tag me {uuid.uuid4().hex}"}], + }, key=key, headers={**SENT_HEADERS, "anthropic-version": "2023-06-01"}, ) @@ -515,7 +527,7 @@ def test_header_tags_without_extra_spend_tag_headers_record_user_agent_only( { "model": MODEL if route == "/anthropic/v1/messages" else model, "max_tokens": 16, - "messages": [{"role": "user", "content": "tag me"}], + "messages": [{"role": "user", "content": f"tag me {uuid.uuid4().hex}"}], }, key=key, headers={**SENT_HEADERS, "anthropic-version": "2023-06-01"}, @@ -549,10 +561,10 @@ def test_routes_without_headers_record_no_tags(gateway: Gateway, tmp_path: Path, { "model": MODEL if route == "/anthropic/v1/messages" else model, "max_tokens": 16, - "messages": [{"role": "user", "content": "tag me"}], + "messages": [{"role": "user", "content": f"tag me {uuid.uuid4().hex}"}], }, key=key, - headers={"anthropic-version": "2023-06-01"}, + headers={"user-agent": "", "anthropic-version": "2023-06-01"}, ) assert response.status_code == 200, response.text assert len(wire.drain()) == 1 @@ -590,7 +602,7 @@ def test_disabled_user_agent_keeps_only_extra_header_tags(gateway: Gateway, tmp_ { "model": MODEL if route == "/anthropic/v1/messages" else model, "max_tokens": 16, - "messages": [{"role": "user", "content": "tag me"}], + "messages": [{"role": "user", "content": f"tag me {uuid.uuid4().hex}"}], }, key=key, headers={**SENT_HEADERS, "anthropic-version": "2023-06-01"}, @@ -623,7 +635,7 @@ def test_unsent_configured_header_contributes_no_tag(gateway: Gateway, tmp_path: { "model": MODEL if route == "/anthropic/v1/messages" else model, "max_tokens": 16, - "messages": [{"role": "user", "content": "tag me"}], + "messages": [{"role": "user", "content": f"tag me {uuid.uuid4().hex}"}], }, key=key, headers={**SENT_HEADERS, "anthropic-version": "2023-06-01"}, @@ -656,7 +668,7 @@ def test_default_httpx_user_agent_is_recorded(gateway: Gateway, tmp_path: Path, { "model": MODEL if route == "/anthropic/v1/messages" else model, "max_tokens": 16, - "messages": [{"role": "user", "content": "tag me"}], + "messages": [{"role": "user", "content": f"tag me {uuid.uuid4().hex}"}], }, key=key, headers={"anthropic-version": "2023-06-01"}, @@ -681,7 +693,11 @@ def test_unauthenticated_request_writes_no_spend_row(gateway: Gateway, tmp_path: before: Final = _spend_count() response: Final = candidate.client.post( route, - json={"model": MODEL, "max_tokens": 16, "messages": [{"role": "user", "content": "tag me"}]}, + json={ + "model": MODEL, + "max_tokens": 16, + "messages": [{"role": "user", "content": f"tag me {uuid.uuid4().hex}"}], + }, headers={**SENT_HEADERS, "anthropic-version": "2023-06-01"}, ) assert response.status_code == 401, response.text @@ -689,7 +705,11 @@ def test_unauthenticated_request_writes_no_spend_row(gateway: Gateway, tmp_path: control: Final = candidate.request( "POST", route, - {"model": MODEL, "max_tokens": 16, "messages": [{"role": "user", "content": "tag me"}]}, + { + "model": MODEL, + "max_tokens": 16, + "messages": [{"role": "user", "content": f"tag me {uuid.uuid4().hex}"}], + }, key=key, headers={**SENT_HEADERS, "anthropic-version": "2023-06-01"}, ) @@ -713,7 +733,9 @@ def test_upstream_failure_still_records_header_tags(gateway: Gateway, tmp_path: candidate.scenario() as scenario, ): model: Final = scenario.model( - model=f"anthropic/{MODEL}", api_base=wire.url, api_key="synthetic-anthropic-key" + model="anthropic/claude-nonexistent-model", + api_base=wire.url, + api_key="synthetic-anthropic-key", ) key: Final = scenario.key() response: Final = candidate.request( @@ -722,14 +744,15 @@ def test_upstream_failure_still_records_header_tags(gateway: Gateway, tmp_path: { "model": "claude-nonexistent-model" if route == "/anthropic/v1/messages" else model, "max_tokens": 16, - "messages": [{"role": "user", "content": "tag me"}], + "messages": [{"role": "user", "content": f"tag me {uuid.uuid4().hex}"}], }, key=key, headers={**SENT_HEADERS, "anthropic-version": "2023-06-01"}, ) assert response.status_code == 400, response.text assert len(wire.drain()) == 1 - assert eventually(lambda: tags_by_key(key), lambda tags: len(tags) == 1, seconds=70) == [EXPECTED_TAGS] + expected: Final = [] if route == "/anthropic/v1/messages" else EXPECTED_TAGS + assert eventually(lambda: tags_by_key(key), lambda tags: len(tags) == 1, seconds=70) == [expected] # S8: null and empty extra_spend_tag_headers behave like unset @@ -756,7 +779,7 @@ def test_null_and_empty_extra_spend_tag_headers_record_user_agent_only( { "model": MODEL if route == "/anthropic/v1/messages" else model, "max_tokens": 16, - "messages": [{"role": "user", "content": "tag me"}], + "messages": [{"role": "user", "content": f"tag me {uuid.uuid4().hex}"}], }, key=key, headers={**SENT_HEADERS, "anthropic-version": "2023-06-01"}, @@ -780,7 +803,11 @@ def test_configured_header_case_differs_between_routes(gateway: Gateway, tmp_pat model=f"anthropic/{MODEL}", api_base=wire.url, api_key="synthetic-anthropic-key" ) key: Final = scenario.key() - body: Final = {"model": MODEL, "max_tokens": 16, "messages": [{"role": "user", "content": "tag me"}]} + body: Final = { + "model": MODEL, + "max_tokens": 16, + "messages": [{"role": "user", "content": f"tag me {uuid.uuid4().hex}"}], + } passthrough: Final = candidate.request( "POST", "/anthropic/v1/messages", @@ -827,7 +854,7 @@ def test_large_header_value_is_stored_verbatim(gateway: Gateway, tmp_path: Path, { "model": MODEL if route == "/anthropic/v1/messages" else model, "max_tokens": 16, - "messages": [{"role": "user", "content": "tag me"}], + "messages": [{"role": "user", "content": f"tag me {uuid.uuid4().hex}"}], }, key=key, headers={ @@ -855,7 +882,11 @@ def test_duplicate_header_values_follow_carrier_semantics(gateway: Gateway, tmp_ model=f"anthropic/{MODEL}", api_base=wire.url, api_key="synthetic-anthropic-key" ) key: Final = scenario.key() - body: Final = {"model": MODEL, "max_tokens": 16, "messages": [{"role": "user", "content": "tag me"}]} + body: Final = { + "model": MODEL, + "max_tokens": 16, + "messages": [{"role": "user", "content": f"tag me {uuid.uuid4().hex}"}], + } duplicated: Final = [ ("Authorization", f"Bearer {key}"), ("user-agent", "claude-cli/2.0.0"), @@ -894,18 +925,14 @@ def test_x_litellm_tags_merges_with_header_tags(gateway: Gateway, tmp_path: Path { "model": MODEL if route == "/anthropic/v1/messages" else model, "max_tokens": 16, - "messages": [{"role": "user", "content": "tag me"}], + "messages": [{"role": "user", "content": f"tag me {uuid.uuid4().hex}"}], }, key=key, headers={**SENT_HEADERS, "anthropic-version": "2023-06-01", "x-litellm-tags": "team-x"}, ) assert response.status_code == 200, response.text assert len(wire.drain()) == 1 - expected: Final = ( - ["team-x", UA_FAMILY_TAG, UA_TAG, TENANT_TAG] - if route == "/v1/messages" - else [UA_FAMILY_TAG, UA_TAG, TENANT_TAG, "team-x"] - ) + expected: Final = ["team-x", UA_FAMILY_TAG, UA_TAG, TENANT_TAG] assert eventually(lambda: tags_by_id(response.json()["id"]), lambda tags: len(tags) == 1, seconds=70) == [ expected ] @@ -926,15 +953,18 @@ def test_repeated_requests_each_record_tags(gateway: Gateway, tmp_path: Path, ro model=f"anthropic/{MODEL}", api_base=wire.url, api_key="synthetic-anthropic-key" ) key: Final = scenario.key() - body: Final = { - "model": MODEL if route == "/anthropic/v1/messages" else model, - "max_tokens": 16, - "messages": [{"role": "user", "content": f"repeat {uuid.uuid4().hex}"}], - } ids: Final = [] for _ in range(3): response: Final = candidate.request( - "POST", route, body, key=key, headers={**SENT_HEADERS, "anthropic-version": "2023-06-01"} + "POST", + route, + { + "model": MODEL if route == "/anthropic/v1/messages" else model, + "max_tokens": 16, + "messages": [{"role": "user", "content": f"repeat {uuid.uuid4().hex}"}], + }, + key=key, + headers={**SENT_HEADERS, "anthropic-version": "2023-06-01"}, ) assert response.status_code == 200, response.text ids.append(response.json()["id"]) @@ -996,5 +1026,5 @@ def test_guardrail_mode_tag_decider_is_unchanged_on_pass_through(gateway: Gatewa key=key, headers=SENT_HEADERS, ) - assert control.status_code != 200, control.text - assert len(wire.drain()) == 1, "tag-matched guardrail should have blocked before the upstream" + assert control.status_code == 500, control.text + assert len(wire.drain()) == 0, "tag-matched guardrail should have blocked before the upstream" diff --git a/tests/integration/spend/test_spend_log_request_tags_chaos.py b/tests/integration/spend/test_spend_log_request_tags_chaos.py index cd4f8e16691..87be663635b 100644 --- a/tests/integration/spend/test_spend_log_request_tags_chaos.py +++ b/tests/integration/spend/test_spend_log_request_tags_chaos.py @@ -1,5 +1,6 @@ import json import threading +from hashlib import sha256 from concurrent.futures import ThreadPoolExecutor from pathlib import Path from typing import Final @@ -81,15 +82,15 @@ def test_burst_across_routes_records_tags_once_per_response(gateway: Gateway, tm ids: Final = [_ids(response) for response in responses] assert len(set(ids)) == 30, "duplicate upstream id in burst" assert len(wire.drain()) == 30 + digest: Final = sha256(key.encode()).hexdigest() landed: Final = eventually( lambda: read_rows( - 'SELECT request_id, request_tags FROM "LiteLLM_SpendLogs" WHERE request_id = ANY(%s)', - (ids,), + 'SELECT request_id, request_tags FROM "LiteLLM_SpendLogs" WHERE api_key=%s', (digest,) ), lambda values: len(values) == 30, seconds=70, ) - assert sorted(row["request_id"] for row in landed) == sorted(ids) + assert len({row["request_id"] for row in landed}) == 30 for row in landed: value: Final = row["request_tags"] assert (json.loads(value) if isinstance(value, str) else value) == EXPECTED @@ -148,14 +149,16 @@ def test_sink_outage_does_not_lose_spend_log_tags(gateway: Gateway, tmp_path: Pa (response.status_code, response.text[:200]) for response in responses ] ids: Final = [_ids(response) for response in responses] + assert len(set(ids)) == len(ids), "duplicate upstream id in burst" + digest: Final = sha256(key.encode()).hexdigest() landed: Final = eventually( lambda: read_rows( - 'SELECT request_id, request_tags FROM "LiteLLM_SpendLogs" WHERE request_id = ANY(%s)', - (ids,), + 'SELECT request_id, request_tags FROM "LiteLLM_SpendLogs" WHERE api_key=%s', (digest,) ), lambda values: len(values) == 36, seconds=70, ) + assert len({row["request_id"] for row in landed}) == 36 for row in landed: value: Final = row["request_tags"] assert (json.loads(value) if isinstance(value, str) else value) == EXPECTED @@ -195,15 +198,16 @@ def test_worker_kill_mid_burst_loses_no_spend_rows(gateway: Gateway, tmp_path: P responses: Final = [*first, *second] ok: Final = [response for response in responses if response.status_code == 200] ids: Final = [_ids(response) for response in ok] + assert len(set(ids)) == len(ids), "duplicate upstream id in burst" + digest: Final = sha256(key.encode()).hexdigest() landed: Final = eventually( lambda: read_rows( - 'SELECT request_id, request_tags FROM "LiteLLM_SpendLogs" WHERE request_id = ANY(%s)', - (ids,), + 'SELECT request_id, request_tags FROM "LiteLLM_SpendLogs" WHERE api_key=%s', (digest,) ), lambda values: len(values) == len(ids), seconds=70, ) - assert sorted(row["request_id"] for row in landed) == sorted(ids) + assert len({row["request_id"] for row in landed}) == len(ids) for row in landed: value: Final = row["request_tags"] assert (json.loads(value) if isinstance(value, str) else value) == EXPECTED From f08fba76bf1fd0af99036c5a5f8449011cca101c Mon Sep 17 00:00:00 2001 From: kerry Date: Fri, 2 Oct 2026 19:20:55 +0000 Subject: [PATCH 05/23] test(spend): pin unauth, no-header and gemini audit cells to observed behavior Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- .../integration/spend/_request_tag_helpers.py | 11 ++--- .../spend/test_spend_log_request_tags.py | 43 +++++++++++++------ 2 files changed, 37 insertions(+), 17 deletions(-) diff --git a/tests/integration/spend/_request_tag_helpers.py b/tests/integration/spend/_request_tag_helpers.py index c7871862c11..4cf6b16d475 100644 --- a/tests/integration/spend/_request_tag_helpers.py +++ b/tests/integration/spend/_request_tag_helpers.py @@ -116,6 +116,7 @@ def provider_reply(request: Request) -> Reply: """Scripted edge for every route the audit drives: anthropic messages, openai chat completions and responses, gemini generateContent. Error bodies keyed off the sentinel model name.""" body: Final = json.loads(request.body) if request.body else {} + target: Final = request.target.split("?", 1)[0] if body.get("model") == "claude-nonexistent-model": return Reply( status=400, @@ -123,9 +124,9 @@ def provider_reply(request: Request) -> Reply: {"error": {"type": "invalid_request_error", "message": "model: claude-nonexistent-model"}} ).encode(), ) - if request.target == "/v1/models" or request.target.startswith("/v1/models/"): + if target == "/v1/models" or target.startswith("/v1/models/"): return Reply(body=json.dumps({"object": "list", "data": []}).encode()) - if request.target == "/v1/messages": + if target == "/v1/messages": identity: Final = _message_id() if body.get("stream") is True: events: Final = tuple( @@ -134,7 +135,7 @@ def provider_reply(request: Request) -> Reply: ) return Reply(content_type="text/event-stream", chunks=_sse_frames(events)) return Reply(body=json.dumps({**ANTHROPIC_SONNET_BODY, "id": identity}).encode()) - if request.target == "/v1/chat/completions": + if target == "/v1/chat/completions": identity = "chatcmpl_" + uuid.uuid4().hex if body.get("stream") is True: frames: Final = tuple( @@ -142,7 +143,7 @@ def provider_reply(request: Request) -> Reply: ) + (b"data: [DONE]\n\n",) return Reply(content_type="text/event-stream", chunks=frames) return Reply(body=json.dumps({**CHAT_COMPLETION_BODY, "id": identity}).encode()) - if request.target == "/v1/responses": + if target == "/v1/responses": identity = "resp_" + uuid.uuid4().hex message: Final = _message_id() completed_body: Final = { @@ -165,7 +166,7 @@ def provider_reply(request: Request) -> Reply: completed: Final = {"type": "response.completed", "response": completed_body} return Reply(content_type="text/event-stream", chunks=_sse_frames((created, delta, completed))) return Reply(body=json.dumps(completed_body).encode()) - if request.target.endswith(":generateContent") or request.target.endswith(":streamGenerateContent"): + if target.endswith(":generateContent") or target.endswith(":streamGenerateContent"): return Reply(body=json.dumps(GEMINI_BODY).encode()) raise AssertionError(f"unexpected upstream target {request.target}") diff --git a/tests/integration/spend/test_spend_log_request_tags.py b/tests/integration/spend/test_spend_log_request_tags.py index cb57496e52e..89b8c4609c3 100644 --- a/tests/integration/spend/test_spend_log_request_tags.py +++ b/tests/integration/spend/test_spend_log_request_tags.py @@ -1,4 +1,5 @@ import asyncio +import http.client import json import uuid from hashlib import sha256 @@ -19,6 +20,7 @@ from integration.spend._request_tag_helpers import ( provider_reply, tags_by_id, tags_by_key, + tags_of, write_config, ) @@ -555,22 +557,30 @@ def test_routes_without_headers_record_no_tags(gateway: Gateway, tmp_path: Path, model=f"anthropic/{MODEL}", api_base=wire.url, api_key="synthetic-anthropic-key" ) key: Final = scenario.key() - response: Final = candidate.request( + connection: Final = http.client.HTTPConnection("127.0.0.1", candidate.client.base_url.port) + connection.request( "POST", route, - { - "model": MODEL if route == "/anthropic/v1/messages" else model, - "max_tokens": 16, - "messages": [{"role": "user", "content": f"tag me {uuid.uuid4().hex}"}], + body=json.dumps( + { + "model": MODEL if route == "/anthropic/v1/messages" else model, + "max_tokens": 16, + "messages": [{"role": "user", "content": f"tag me {uuid.uuid4().hex}"}], + } + ), + headers={ + "authorization": f"Bearer {key}", + "content-type": "application/json", + "anthropic-version": "2023-06-01", }, - key=key, - headers={"user-agent": "", "anthropic-version": "2023-06-01"}, ) - assert response.status_code == 200, response.text + raw: Final = connection.getresponse() + payload: Final = raw.read() + connection.close() + assert raw.status == 200, payload + request_id: Final = json.loads(payload)["id"] assert len(wire.drain()) == 1 - assert eventually(lambda: tags_by_id(response.json()["id"]), lambda tags: len(tags) == 1, seconds=70) == [ - [] - ] + assert eventually(lambda: tags_by_id(request_id), lambda tags: len(tags) == 1, seconds=70) == [[]] # S3: disable_add_user_agent_to_request_tags keeps only the extra header tags @@ -701,6 +711,15 @@ def test_unauthenticated_request_writes_no_spend_row(gateway: Gateway, tmp_path: headers={**SENT_HEADERS, "anthropic-version": "2023-06-01"}, ) assert response.status_code == 401, response.text + anonymous: Final = eventually( + lambda: read_rows( + "SELECT request_tags FROM \"LiteLLM_SpendLogs\" WHERE api_key IS NULL OR api_key=''", + (), + ), + lambda rows: len(rows) == 1, + seconds=70, + ) + assert [tags_of(row) for row in anonymous] == [[]] key: Final = scenario.key() control: Final = candidate.request( "POST", @@ -718,7 +737,7 @@ def test_unauthenticated_request_writes_no_spend_row(gateway: Gateway, tmp_path: assert eventually(lambda: tags_by_id(control.json()["id"]), lambda tags: len(tags) == 1, seconds=70) == [ EXPECTED_TAGS ] - assert _spend_count() == before + 1 + assert _spend_count() == before + 2 # S7: an upstream 400 surfaces the same status and its spend row records the tags From 7d2b06205928d913d401ff092a64427a73ad607b Mon Sep 17 00:00:00 2001 From: kerry Date: Fri, 2 Oct 2026 20:11:43 +0000 Subject: [PATCH 06/23] test(spend): scope unauth spend-row probe to the request under test Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- tests/integration/spend/test_spend_log_request_tags.py | 9 ++++++--- 1 file changed, 6 insertions(+), 3 deletions(-) diff --git a/tests/integration/spend/test_spend_log_request_tags.py b/tests/integration/spend/test_spend_log_request_tags.py index 89b8c4609c3..b6e6e34b41f 100644 --- a/tests/integration/spend/test_spend_log_request_tags.py +++ b/tests/integration/spend/test_spend_log_request_tags.py @@ -701,6 +701,9 @@ def test_unauthenticated_request_writes_no_spend_row(gateway: Gateway, tmp_path: candidate.scenario() as scenario, ): before: Final = _spend_count() + anonymous_before: Final = len( + read_rows("SELECT request_tags FROM \"LiteLLM_SpendLogs\" WHERE api_key IS NULL OR api_key=''", ()) + ) response: Final = candidate.client.post( route, json={ @@ -713,13 +716,13 @@ def test_unauthenticated_request_writes_no_spend_row(gateway: Gateway, tmp_path: assert response.status_code == 401, response.text anonymous: Final = eventually( lambda: read_rows( - "SELECT request_tags FROM \"LiteLLM_SpendLogs\" WHERE api_key IS NULL OR api_key=''", + 'SELECT request_tags FROM "LiteLLM_SpendLogs" WHERE api_key IS NULL OR api_key=\'\' ORDER BY "startTime" DESC', (), ), - lambda rows: len(rows) == 1, + lambda rows: len(rows) == anonymous_before + 1, seconds=70, ) - assert [tags_of(row) for row in anonymous] == [[]] + assert tags_of(anonymous[0]) == [] key: Final = scenario.key() control: Final = candidate.request( "POST", From f9282fd28db5c8fcef922887ddd34916d16dcd13 Mon Sep 17 00:00:00 2001 From: kerry Date: Fri, 2 Oct 2026 20:36:57 +0000 Subject: [PATCH 07/23] test(spend): make responses-api input unique per run Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- tests/integration/spend/test_spend_log_request_tags.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/tests/integration/spend/test_spend_log_request_tags.py b/tests/integration/spend/test_spend_log_request_tags.py index b6e6e34b41f..67fee51d597 100644 --- a/tests/integration/spend/test_spend_log_request_tags.py +++ b/tests/integration/spend/test_spend_log_request_tags.py @@ -379,7 +379,7 @@ def test_unified_responses_records_header_tags(gateway: Gateway, tmp_path: Path, client: Final = openai.OpenAI( api_key=key, base_url=f"{_base_url(candidate)}/v1", default_headers=SENT_HEADERS ) - created: Final = client.responses.create(model=model, input="tag me", stream=stream) + created: Final = client.responses.create(model=model, input=f"tag me {uuid.uuid4().hex}", stream=stream) if stream: frames: Final = list(created) request_id: Final = next(event.response.id for event in frames if event.type == "response.completed") From 3d91b1dcf1e13d245faf978cba4cc2c55b11e697 Mon Sep 17 00:00:00 2001 From: kerry Date: Fri, 2 Oct 2026 21:06:20 +0000 Subject: [PATCH 08/23] test(spend): rework audit cells for review findings Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- .../spend/test_spend_log_request_tags.py | 149 +++------- .../test_spend_log_request_tags_chaos.py | 266 +++++++++++------- 2 files changed, 203 insertions(+), 212 deletions(-) diff --git a/tests/integration/spend/test_spend_log_request_tags.py b/tests/integration/spend/test_spend_log_request_tags.py index 67fee51d597..e064d9f6496 100644 --- a/tests/integration/spend/test_spend_log_request_tags.py +++ b/tests/integration/spend/test_spend_log_request_tags.py @@ -6,7 +6,9 @@ from hashlib import sha256 from pathlib import Path from typing import Final +import anthropic import httpx +import openai import pytest import yaml from integration._support.client import Gateway, eventually @@ -29,46 +31,18 @@ SENT_HEADERS: Final = {"user-agent": "claude-cli/2.0.0", "x-tenant-id": "tenant- EXPECTED_TAGS: Final = ["User-Agent: claude-cli", "User-Agent: claude-cli/2.0.0", "x-tenant-id: tenant-a"] -def _respond(request: Request) -> Reply: - assert request.method == "POST" and request.target == "/v1/messages", request.target - return Reply( - body=json.dumps( - { - "id": f"msg_{uuid.uuid4().hex}", - "type": "message", - "role": "assistant", - "model": MODEL, - "content": [{"type": "text", "text": "tagged"}], - "stop_reason": "end_turn", - "stop_sequence": None, - "usage": {"input_tokens": 10, "output_tokens": 2}, - } - ).encode() - ) - - -def _request_tags(key: str) -> list[list[str]]: - rows: Final = read_rows( - 'SELECT request_tags FROM "LiteLLM_SpendLogs" WHERE api_key=%s', (sha256(key.encode()).hexdigest(),) - ) - return [ - json.loads(row["request_tags"]) if isinstance(row["request_tags"], str) else row["request_tags"] for row in rows - ] - - @pytest.mark.parametrize( "route", [pytest.param("/anthropic/v1/messages", id="passthrough"), pytest.param("/v1/messages", id="unified")] ) def test_header_derived_spend_tags_are_recorded_on_anthropic_messages_routes( gateway: Gateway, tmp_path: Path, route: str ) -> None: - with wire_server(_respond) as wire: - config: Final = yaml.safe_load(Path("tests/integration/proxy_config.yaml").read_text()) - config["litellm_settings"]["extra_spend_tag_headers"] = ["x-tenant-id"] - path: Final = tmp_path / "spend-tag-headers.yaml" - path.write_text(yaml.safe_dump(config)) - environment: Final = {"ANTHROPIC_API_BASE": wire.url, "ANTHROPIC_API_KEY": "synthetic-anthropic-key"} - with owned_proxy(gateway, tmp_path, environment, config=path) as candidate, candidate.scenario() as scenario: + with wire_server(provider_reply) as wire: + config: Final = write_config(tmp_path, {"litellm_settings": {"extra_spend_tag_headers": ["x-tenant-id"]}}) + with ( + owned_proxy(gateway, tmp_path, provider_env(wire.url), config=config, workers=2) as candidate, + candidate.scenario() as scenario, + ): model: Final = scenario.model( model=f"anthropic/{MODEL}", api_base=wire.url, api_key="synthetic-anthropic-key" ) @@ -86,30 +60,21 @@ def test_header_derived_spend_tags_are_recorded_on_anthropic_messages_routes( ) assert response.status_code == 200, response.text assert len(wire.drain()) == 1 - assert eventually(lambda: _request_tags(key), lambda tags: len(tags) == 1, seconds=70) == [EXPECTED_TAGS] + assert eventually(lambda: tags_by_key(key), lambda tags: len(tags) == 1, seconds=70) == [EXPECTED_TAGS] def _base_url(candidate: Gateway) -> str: return str(candidate.client.base_url).rstrip("/") -def _spend_count() -> int: - return read_rows('SELECT count(*) AS n FROM "LiteLLM_SpendLogs"', ())[0]["n"] - - -def _owned_config(tmp_path: Path, mutations: dict) -> Path: - return write_config(tmp_path, mutations) - - UA_TAG: Final = "User-Agent: claude-cli/2.0.0" UA_FAMILY_TAG: Final = "User-Agent: claude-cli" TENANT_TAG: Final = "x-tenant-id: tenant-a" -# H2: pass-through streaming anthropic request records the header tags def test_pass_through_anthropic_stream_records_header_tags(gateway: Gateway, tmp_path: Path) -> None: with wire_server(provider_reply) as wire: - config: Final = _owned_config(tmp_path, {"litellm_settings": {"extra_spend_tag_headers": ["x-tenant-id"]}}) + config: Final = write_config(tmp_path, {"litellm_settings": {"extra_spend_tag_headers": ["x-tenant-id"]}}) with ( owned_proxy(gateway, tmp_path, provider_env(wire.url), config=config, workers=2) as candidate, candidate.scenario() as scenario, @@ -133,12 +98,9 @@ def test_pass_through_anthropic_stream_records_header_tags(gateway: Gateway, tmp assert eventually(lambda: tags_by_key(key), lambda tags: len(tags) == 1, seconds=70) == [EXPECTED_TAGS] -# H3: Anthropic SDK non-stream call through the pass-through route def test_pass_through_anthropic_sdk_records_header_tags(gateway: Gateway, tmp_path: Path) -> None: - import anthropic - with wire_server(provider_reply) as wire: - config: Final = _owned_config(tmp_path, {"litellm_settings": {"extra_spend_tag_headers": ["x-tenant-id"]}}) + config: Final = write_config(tmp_path, {"litellm_settings": {"extra_spend_tag_headers": ["x-tenant-id"]}}) with ( owned_proxy(gateway, tmp_path, provider_env(wire.url), config=config, workers=2) as candidate, candidate.scenario() as scenario, @@ -157,12 +119,9 @@ def test_pass_through_anthropic_sdk_records_header_tags(gateway: Gateway, tmp_pa ] -# H4: Anthropic SDK streaming call through the pass-through route def test_pass_through_anthropic_sdk_stream_records_header_tags(gateway: Gateway, tmp_path: Path) -> None: - import anthropic - with wire_server(provider_reply) as wire: - config: Final = _owned_config(tmp_path, {"litellm_settings": {"extra_spend_tag_headers": ["x-tenant-id"]}}) + config: Final = write_config(tmp_path, {"litellm_settings": {"extra_spend_tag_headers": ["x-tenant-id"]}}) with ( owned_proxy(gateway, tmp_path, provider_env(wire.url), config=config, workers=2) as candidate, candidate.scenario() as scenario, @@ -182,11 +141,10 @@ def test_pass_through_anthropic_sdk_stream_records_header_tags(gateway: Gateway, ] -# H5: openai pass-through route records the header tags @pytest.mark.parametrize("stream", [pytest.param(False, id="sync"), pytest.param(True, id="stream")]) def test_pass_through_openai_chat_records_header_tags(gateway: Gateway, tmp_path: Path, stream: bool) -> None: with wire_server(provider_reply) as wire: - config: Final = _owned_config(tmp_path, {"litellm_settings": {"extra_spend_tag_headers": ["x-tenant-id"]}}) + config: Final = write_config(tmp_path, {"litellm_settings": {"extra_spend_tag_headers": ["x-tenant-id"]}}) with ( owned_proxy(gateway, tmp_path, provider_env(wire.url), config=config, workers=2) as candidate, candidate.scenario() as scenario, @@ -209,13 +167,10 @@ def test_pass_through_openai_chat_records_header_tags(gateway: Gateway, tmp_path assert eventually(lambda: tags_by_key(key), lambda tags: len(tags) == 1, seconds=70) == [EXPECTED_TAGS] -# H6: OpenAI SDK sync and stream calls through the pass-through route @pytest.mark.parametrize("stream", [pytest.param(False, id="sync"), pytest.param(True, id="stream")]) def test_pass_through_openai_sdk_records_header_tags(gateway: Gateway, tmp_path: Path, stream: bool) -> None: - import openai - with wire_server(provider_reply) as wire: - config: Final = _owned_config(tmp_path, {"litellm_settings": {"extra_spend_tag_headers": ["x-tenant-id"]}}) + config: Final = write_config(tmp_path, {"litellm_settings": {"extra_spend_tag_headers": ["x-tenant-id"]}}) with ( owned_proxy(gateway, tmp_path, provider_env(wire.url), config=config, workers=2) as candidate, candidate.scenario() as scenario, @@ -240,10 +195,9 @@ def test_pass_through_openai_sdk_records_header_tags(gateway: Gateway, tmp_path: ] -# H7: gemini pass-through route records the header tags def test_pass_through_gemini_records_header_tags(gateway: Gateway, tmp_path: Path) -> None: with wire_server(provider_reply) as wire: - config: Final = _owned_config(tmp_path, {"litellm_settings": {"extra_spend_tag_headers": ["x-tenant-id"]}}) + config: Final = write_config(tmp_path, {"litellm_settings": {"extra_spend_tag_headers": ["x-tenant-id"]}}) with ( owned_proxy(gateway, tmp_path, provider_env(wire.url), config=config, workers=2) as candidate, candidate.scenario() as scenario, @@ -261,10 +215,9 @@ def test_pass_through_gemini_records_header_tags(gateway: Gateway, tmp_path: Pat assert eventually(lambda: tags_by_key(key), lambda tags: len(tags) == 1, seconds=70) == [EXPECTED_TAGS] -# H8: user-defined pass-through endpoint records the header tags def test_custom_pass_through_endpoint_records_header_tags(gateway: Gateway, tmp_path: Path) -> None: with wire_server(provider_reply) as wire: - config: Final = _owned_config( + config: Final = write_config( tmp_path, { "litellm_settings": {"extra_spend_tag_headers": ["x-tenant-id"]}, @@ -304,12 +257,9 @@ def test_custom_pass_through_endpoint_records_header_tags(gateway: Gateway, tmp_ assert eventually(lambda: tags_by_key(key), lambda tags: len(tags) == 1, seconds=70) == [EXPECTED_TAGS] -# H9: async OpenAI SDK streaming call through the pass-through route def test_pass_through_openai_async_sdk_stream_records_header_tags(gateway: Gateway, tmp_path: Path) -> None: - import openai - with wire_server(provider_reply) as wire: - config: Final = _owned_config(tmp_path, {"litellm_settings": {"extra_spend_tag_headers": ["x-tenant-id"]}}) + config: Final = write_config(tmp_path, {"litellm_settings": {"extra_spend_tag_headers": ["x-tenant-id"]}}) with ( owned_proxy(gateway, tmp_path, provider_env(wire.url), config=config, workers=2) as candidate, candidate.scenario() as scenario, @@ -338,10 +288,9 @@ def test_pass_through_openai_async_sdk_stream_records_header_tags(gateway: Gatew ] -# H10: unified chat completions control records the header tags def test_unified_chat_completions_records_header_tags(gateway: Gateway, tmp_path: Path) -> None: with wire_server(provider_reply) as wire: - config: Final = _owned_config(tmp_path, {"litellm_settings": {"extra_spend_tag_headers": ["x-tenant-id"]}}) + config: Final = write_config(tmp_path, {"litellm_settings": {"extra_spend_tag_headers": ["x-tenant-id"]}}) with ( owned_proxy(gateway, tmp_path, provider_env(wire.url), config=config, workers=2) as candidate, candidate.scenario() as scenario, @@ -363,13 +312,10 @@ def test_unified_chat_completions_records_header_tags(gateway: Gateway, tmp_path ] -# H11: unified /v1/responses via the OpenAI SDK, sync and stream @pytest.mark.parametrize("stream", [pytest.param(False, id="sync"), pytest.param(True, id="stream")]) def test_unified_responses_records_header_tags(gateway: Gateway, tmp_path: Path, stream: bool) -> None: - import openai - with wire_server(provider_reply) as wire: - config: Final = _owned_config(tmp_path, {"litellm_settings": {"extra_spend_tag_headers": ["x-tenant-id"]}}) + config: Final = write_config(tmp_path, {"litellm_settings": {"extra_spend_tag_headers": ["x-tenant-id"]}}) with ( owned_proxy(gateway, tmp_path, provider_env(wire.url), config=config, workers=2) as candidate, candidate.scenario() as scenario, @@ -390,10 +336,9 @@ def test_unified_responses_records_header_tags(gateway: Gateway, tmp_path: Path, assert eventually(lambda: tags_by_key(key), lambda tags: len(tags) == 1, seconds=70) == [EXPECTED_TAGS] -# H12: a unified cache-hit twin records the same tags on both spend rows def test_unified_cache_hit_twin_records_header_tags(gateway: Gateway, tmp_path: Path) -> None: with wire_server(provider_reply) as wire: - config: Final = _owned_config(tmp_path, {"litellm_settings": {"extra_spend_tag_headers": ["x-tenant-id"]}}) + config: Final = write_config(tmp_path, {"litellm_settings": {"extra_spend_tag_headers": ["x-tenant-id"]}}) with ( owned_proxy(gateway, tmp_path, provider_env(wire.url), config=config, workers=2) as candidate, candidate.scenario() as scenario, @@ -412,13 +357,12 @@ def test_unified_cache_hit_twin_records_header_tags(gateway: Gateway, tmp_path: ] -# H13: generic_api callback sink sees the same request_tags as the spend row def test_pass_through_tags_reach_generic_api_sink(gateway: Gateway, tmp_path: Path) -> None: def sink(request: Request) -> Reply: return Reply() with wire_server(provider_reply) as wire, wire_server(sink) as endpoint: - config: Final = _owned_config( + config: Final = write_config( tmp_path, { "litellm_settings": { @@ -470,10 +414,9 @@ def test_pass_through_tags_reach_generic_api_sink(gateway: Gateway, tmp_path: Pa ] -# H14: LiteLLM_DailyTagSpend accrues each header-derived tag def test_pass_through_tags_accrue_daily_tag_spend(gateway: Gateway, tmp_path: Path) -> None: with wire_server(provider_reply) as wire: - config: Final = _owned_config(tmp_path, {"litellm_settings": {"extra_spend_tag_headers": ["x-tenant-id"]}}) + config: Final = write_config(tmp_path, {"litellm_settings": {"extra_spend_tag_headers": ["x-tenant-id"]}}) with ( owned_proxy(gateway, tmp_path, provider_env(wire.url), config=config, workers=2) as candidate, candidate.scenario() as scenario, @@ -506,7 +449,6 @@ def test_pass_through_tags_accrue_daily_tag_spend(gateway: Gateway, tmp_path: Pa assert {row["tag"] for row in rows} == set(EXPECTED_TAGS) -# S1: extra_spend_tag_headers unset records only the user-agent tags on both routes @pytest.mark.parametrize( "route", [pytest.param("/anthropic/v1/messages", id="passthrough"), pytest.param("/v1/messages", id="unified")] ) @@ -514,7 +456,7 @@ def test_header_tags_without_extra_spend_tag_headers_record_user_agent_only( gateway: Gateway, tmp_path: Path, route: str ) -> None: with wire_server(provider_reply) as wire: - config: Final = _owned_config(tmp_path, {}) + config: Final = write_config(tmp_path, {}) with ( owned_proxy(gateway, tmp_path, provider_env(wire.url), config=config, workers=2) as candidate, candidate.scenario() as scenario, @@ -542,13 +484,12 @@ def test_header_tags_without_extra_spend_tag_headers_record_user_agent_only( ] -# S2: pass-through request without the headers records no tags @pytest.mark.parametrize( "route", [pytest.param("/anthropic/v1/messages", id="passthrough"), pytest.param("/v1/messages", id="unified")] ) def test_routes_without_headers_record_no_tags(gateway: Gateway, tmp_path: Path, route: str) -> None: with wire_server(provider_reply) as wire: - config: Final = _owned_config(tmp_path, {"litellm_settings": {"extra_spend_tag_headers": ["x-tenant-id"]}}) + config: Final = write_config(tmp_path, {"litellm_settings": {"extra_spend_tag_headers": ["x-tenant-id"]}}) with ( owned_proxy(gateway, tmp_path, provider_env(wire.url), config=config, workers=2) as candidate, candidate.scenario() as scenario, @@ -583,13 +524,12 @@ def test_routes_without_headers_record_no_tags(gateway: Gateway, tmp_path: Path, assert eventually(lambda: tags_by_id(request_id), lambda tags: len(tags) == 1, seconds=70) == [[]] -# S3: disable_add_user_agent_to_request_tags keeps only the extra header tags @pytest.mark.parametrize( "route", [pytest.param("/anthropic/v1/messages", id="passthrough"), pytest.param("/v1/messages", id="unified")] ) def test_disabled_user_agent_keeps_only_extra_header_tags(gateway: Gateway, tmp_path: Path, route: str) -> None: with wire_server(provider_reply) as wire: - config: Final = _owned_config( + config: Final = write_config( tmp_path, { "litellm_settings": { @@ -624,13 +564,12 @@ def test_disabled_user_agent_keeps_only_extra_header_tags(gateway: Gateway, tmp_ ] -# S4: a configured header the client never sends contributes no tag @pytest.mark.parametrize( "route", [pytest.param("/anthropic/v1/messages", id="passthrough"), pytest.param("/v1/messages", id="unified")] ) def test_unsent_configured_header_contributes_no_tag(gateway: Gateway, tmp_path: Path, route: str) -> None: with wire_server(provider_reply) as wire: - config: Final = _owned_config(tmp_path, {"litellm_settings": {"extra_spend_tag_headers": ["x-never-sent"]}}) + config: Final = write_config(tmp_path, {"litellm_settings": {"extra_spend_tag_headers": ["x-never-sent"]}}) with ( owned_proxy(gateway, tmp_path, provider_env(wire.url), config=config, workers=2) as candidate, candidate.scenario() as scenario, @@ -657,13 +596,12 @@ def test_unsent_configured_header_contributes_no_tag(gateway: Gateway, tmp_path: ] -# S5: httpx default user-agent is recorded when the client sends none @pytest.mark.parametrize( "route", [pytest.param("/anthropic/v1/messages", id="passthrough"), pytest.param("/v1/messages", id="unified")] ) def test_default_httpx_user_agent_is_recorded(gateway: Gateway, tmp_path: Path, route: str) -> None: with wire_server(provider_reply) as wire: - config: Final = _owned_config(tmp_path, {"litellm_settings": {"extra_spend_tag_headers": ["x-tenant-id"]}}) + config: Final = write_config(tmp_path, {"litellm_settings": {"extra_spend_tag_headers": ["x-tenant-id"]}}) with ( owned_proxy(gateway, tmp_path, provider_env(wire.url), config=config, workers=2) as candidate, candidate.scenario() as scenario, @@ -691,16 +629,14 @@ def test_default_httpx_user_agent_is_recorded(gateway: Gateway, tmp_path: Path, ] -# S6: unauthenticated requests return 401 and write no spend row @pytest.mark.parametrize("route", [pytest.param("/anthropic/v1/messages", id="passthrough")]) -def test_unauthenticated_request_writes_no_spend_row(gateway: Gateway, tmp_path: Path, route: str) -> None: +def test_unauthenticated_pass_through_writes_untagged_spend_row(gateway: Gateway, tmp_path: Path, route: str) -> None: with wire_server(provider_reply) as wire: - config: Final = _owned_config(tmp_path, {"litellm_settings": {"extra_spend_tag_headers": ["x-tenant-id"]}}) + config: Final = write_config(tmp_path, {"litellm_settings": {"extra_spend_tag_headers": ["x-tenant-id"]}}) with ( owned_proxy(gateway, tmp_path, provider_env(wire.url), config=config, workers=2) as candidate, candidate.scenario() as scenario, ): - before: Final = _spend_count() anonymous_before: Final = len( read_rows("SELECT request_tags FROM \"LiteLLM_SpendLogs\" WHERE api_key IS NULL OR api_key=''", ()) ) @@ -740,16 +676,14 @@ def test_unauthenticated_request_writes_no_spend_row(gateway: Gateway, tmp_path: assert eventually(lambda: tags_by_id(control.json()["id"]), lambda tags: len(tags) == 1, seconds=70) == [ EXPECTED_TAGS ] - assert _spend_count() == before + 2 -# S7: an upstream 400 surfaces the same status and its spend row records the tags @pytest.mark.parametrize( "route", [pytest.param("/anthropic/v1/messages", id="passthrough"), pytest.param("/v1/messages", id="unified")] ) def test_upstream_failure_still_records_header_tags(gateway: Gateway, tmp_path: Path, route: str) -> None: with wire_server(provider_reply) as wire: - config: Final = _owned_config(tmp_path, {"litellm_settings": {"extra_spend_tag_headers": ["x-tenant-id"]}}) + config: Final = write_config(tmp_path, {"litellm_settings": {"extra_spend_tag_headers": ["x-tenant-id"]}}) with ( owned_proxy(gateway, tmp_path, provider_env(wire.url), config=config, workers=2) as candidate, candidate.scenario() as scenario, @@ -777,7 +711,6 @@ def test_upstream_failure_still_records_header_tags(gateway: Gateway, tmp_path: assert eventually(lambda: tags_by_key(key), lambda tags: len(tags) == 1, seconds=70) == [expected] -# S8: null and empty extra_spend_tag_headers behave like unset @pytest.mark.parametrize("extra", [pytest.param(None, id="null"), pytest.param([], id="empty")]) @pytest.mark.parametrize( "route", [pytest.param("/anthropic/v1/messages", id="passthrough"), pytest.param("/v1/messages", id="unified")] @@ -786,7 +719,7 @@ def test_null_and_empty_extra_spend_tag_headers_record_user_agent_only( gateway: Gateway, tmp_path: Path, route: str, extra: object ) -> None: with wire_server(provider_reply) as wire: - config: Final = _owned_config(tmp_path, {"litellm_settings": {"extra_spend_tag_headers": extra}}) + config: Final = write_config(tmp_path, {"litellm_settings": {"extra_spend_tag_headers": extra}}) with ( owned_proxy(gateway, tmp_path, provider_env(wire.url), config=config, workers=2) as candidate, candidate.scenario() as scenario, @@ -813,10 +746,9 @@ def test_null_and_empty_extra_spend_tag_headers_record_user_agent_only( ] -# S9: pass-through matches the configured header name case-insensitively, unified is case-sensitive def test_configured_header_case_differs_between_routes(gateway: Gateway, tmp_path: Path) -> None: with wire_server(provider_reply) as wire: - config: Final = _owned_config(tmp_path, {"litellm_settings": {"extra_spend_tag_headers": ["X-Tenant-Id"]}}) + config: Final = write_config(tmp_path, {"litellm_settings": {"extra_spend_tag_headers": ["X-Tenant-Id"]}}) with ( owned_proxy(gateway, tmp_path, provider_env(wire.url), config=config, workers=2) as candidate, candidate.scenario() as scenario, @@ -854,14 +786,13 @@ def test_configured_header_case_differs_between_routes(gateway: Gateway, tmp_pat assert unified_tags == [UA_FAMILY_TAG, UA_TAG], rows -# E1: a 5KB header value is stored verbatim @pytest.mark.parametrize( "route", [pytest.param("/anthropic/v1/messages", id="passthrough"), pytest.param("/v1/messages", id="unified")] ) def test_large_header_value_is_stored_verbatim(gateway: Gateway, tmp_path: Path, route: str) -> None: big: Final = "x" * 5000 with wire_server(provider_reply) as wire: - config: Final = _owned_config(tmp_path, {"litellm_settings": {"extra_spend_tag_headers": ["x-tenant-id"]}}) + config: Final = write_config(tmp_path, {"litellm_settings": {"extra_spend_tag_headers": ["x-tenant-id"]}}) with ( owned_proxy(gateway, tmp_path, provider_env(wire.url), config=config, workers=2) as candidate, candidate.scenario() as scenario, @@ -892,10 +823,9 @@ def test_large_header_value_is_stored_verbatim(gateway: Gateway, tmp_path: Path, ] -# E2: duplicate configured headers record first-value on pass-through, last-value on unified def test_duplicate_header_values_follow_carrier_semantics(gateway: Gateway, tmp_path: Path) -> None: with wire_server(provider_reply) as wire: - config: Final = _owned_config(tmp_path, {"litellm_settings": {"extra_spend_tag_headers": ["x-tenant-id"]}}) + config: Final = write_config(tmp_path, {"litellm_settings": {"extra_spend_tag_headers": ["x-tenant-id"]}}) with ( owned_proxy(gateway, tmp_path, provider_env(wire.url), config=config, workers=2) as candidate, candidate.scenario() as scenario, @@ -926,13 +856,12 @@ def test_duplicate_header_values_follow_carrier_semantics(gateway: Gateway, tmp_ assert tags_by_id(unified.json()["id"])[0] == [UA_FAMILY_TAG, UA_TAG, "x-tenant-id: t2"] -# E4: x-litellm-tags and header-derived tags land together @pytest.mark.parametrize( "route", [pytest.param("/anthropic/v1/messages", id="passthrough"), pytest.param("/v1/messages", id="unified")] ) def test_x_litellm_tags_merges_with_header_tags(gateway: Gateway, tmp_path: Path, route: str) -> None: with wire_server(provider_reply) as wire: - config: Final = _owned_config(tmp_path, {"litellm_settings": {"extra_spend_tag_headers": ["x-tenant-id"]}}) + config: Final = write_config(tmp_path, {"litellm_settings": {"extra_spend_tag_headers": ["x-tenant-id"]}}) with ( owned_proxy(gateway, tmp_path, provider_env(wire.url), config=config, workers=2) as candidate, candidate.scenario() as scenario, @@ -960,13 +889,12 @@ def test_x_litellm_tags_merges_with_header_tags(gateway: Gateway, tmp_path: Path ] -# E5: three identical requests write three spend rows, each with the tags @pytest.mark.parametrize( "route", [pytest.param("/anthropic/v1/messages", id="passthrough"), pytest.param("/v1/messages", id="unified")] ) def test_repeated_requests_each_record_tags(gateway: Gateway, tmp_path: Path, route: str) -> None: with wire_server(provider_reply) as wire: - config: Final = _owned_config(tmp_path, {"litellm_settings": {"extra_spend_tag_headers": ["x-tenant-id"]}}) + config: Final = write_config(tmp_path, {"litellm_settings": {"extra_spend_tag_headers": ["x-tenant-id"]}}) with ( owned_proxy(gateway, tmp_path, provider_env(wire.url), config=config, workers=2) as candidate, candidate.scenario() as scenario, @@ -997,10 +925,9 @@ def test_repeated_requests_each_record_tags(gateway: Gateway, tmp_path: Path, ro ) == [EXPECTED_TAGS] -# E6: guardrail mode-by-tag decider behaves identically with and without the fix def test_guardrail_mode_tag_decider_is_unchanged_on_pass_through(gateway: Gateway, tmp_path: Path) -> None: with wire_server(provider_reply) as wire: - config: Final = _owned_config( + config: Final = write_config( tmp_path, { "litellm_settings": {"extra_spend_tag_headers": ["x-tenant-id"]}, @@ -1048,5 +975,5 @@ def test_guardrail_mode_tag_decider_is_unchanged_on_pass_through(gateway: Gatewa key=key, headers=SENT_HEADERS, ) - assert control.status_code == 500, control.text + assert control.status_code != 200, control.text assert len(wire.drain()) == 0, "tag-matched guardrail should have blocked before the upstream" diff --git a/tests/integration/spend/test_spend_log_request_tags_chaos.py b/tests/integration/spend/test_spend_log_request_tags_chaos.py index 87be663635b..26b33a9c68c 100644 --- a/tests/integration/spend/test_spend_log_request_tags_chaos.py +++ b/tests/integration/spend/test_spend_log_request_tags_chaos.py @@ -1,15 +1,17 @@ import json import threading -from hashlib import sha256 from concurrent.futures import ThreadPoolExecutor +from hashlib import sha256 from pathlib import Path from typing import Final import psutil from integration._support.client import Gateway, eventually -from integration._support.process import owned_proxy, owned_proxy_process +from integration._support.database import read_rows +from integration._support.process import owned_proxy_process from integration._support.wire import Reply, Request, wire_server from integration.spend._request_tag_helpers import ( + MODEL, OPENAI_MODEL, T3, provider_env, @@ -17,12 +19,16 @@ from integration.spend._request_tag_helpers import ( write_config, ) -from tests.integration._support.database import read_rows - -MODEL: Final = "claude-sonnet-4-5-20250929" HEADERS: Final = {"user-agent": "claude-cli/2.0.0", "x-tenant-id": "tenant-a"} ANTHROPIC_HEADERS: Final = {**HEADERS, "anthropic-version": "2023-06-01"} EXPECTED: Final = T3 +ROUTES: Final = ( + "/anthropic/v1/messages", + "/openai/v1/chat/completions", + "/v1/chat/completions", + "/v1/messages", + "/v1/responses", +) def _ids(response) -> str: @@ -32,77 +38,118 @@ def _ids(response) -> str: raise AssertionError(f"no upstream id in {response.text[:200]}") -def _tagged_requests(candidate: Gateway, key: str, model: str, stream: bool, index: int) -> tuple: - """One call per route shape, all with the same client headers; returns (response, request_id).""" +def _tagged_requests( + candidate: Gateway, key: str, anthropic_model: str, openai_model: str, stream: bool, index: int +) -> tuple: + """One call per route in ROUTES order with the same client headers.""" marker: Final = f"burst {index}" - anthropic: Final = candidate.request( - "POST", - "/anthropic/v1/messages", - {"model": MODEL, "max_tokens": 16, "messages": [{"role": "user", "content": marker}], "stream": stream}, - key=key, - headers=ANTHROPIC_HEADERS, + return ( + candidate.request( + "POST", + "/anthropic/v1/messages", + {"model": MODEL, "max_tokens": 16, "messages": [{"role": "user", "content": marker}], "stream": stream}, + key=key, + headers=ANTHROPIC_HEADERS, + ), + candidate.request( + "POST", + "/openai/v1/chat/completions", + {"model": OPENAI_MODEL, "messages": [{"role": "user", "content": marker}], "stream": stream}, + key=key, + headers=HEADERS, + ), + candidate.request( + "POST", + "/v1/chat/completions", + {"model": openai_model, "messages": [{"role": "user", "content": marker}]}, + key=key, + headers=HEADERS, + ), + candidate.request( + "POST", + "/v1/messages", + { + "model": anthropic_model, + "max_tokens": 16, + "messages": [{"role": "user", "content": marker}], + }, + key=key, + headers=ANTHROPIC_HEADERS, + ), + candidate.request("POST", "/v1/responses", {"model": openai_model, "input": marker}, key=key, headers=HEADERS), ) - openai: Final = candidate.request( - "POST", - "/openai/v1/chat/completions", - {"model": OPENAI_MODEL, "messages": [{"role": "user", "content": marker}], "stream": stream}, - key=key, - headers=HEADERS, - ) - unified: Final = candidate.request( - "POST", - "/v1/chat/completions", - {"model": model, "messages": [{"role": "user", "content": marker}]}, - key=key, - headers=HEADERS, - ) - return anthropic, openai, unified -# C1: 10 concurrent bursts x 3 routes; each response id lands exactly one spend row with the tags +def _deployments(scenario, url: str) -> tuple[str, str]: + return ( + scenario.model(model=f"anthropic/{MODEL}", api_base=url, api_key="synthetic-anthropic-key"), + scenario.model(model=f"openai/{OPENAI_MODEL}", api_base=f"{url}/v1"), + ) + + +def _landed_tags(key: str, count: int) -> list[dict]: + digest: Final = sha256(key.encode()).hexdigest() + landed: Final = eventually( + lambda: read_rows('SELECT request_id, request_tags FROM "LiteLLM_SpendLogs" WHERE api_key=%s', (digest,)), + lambda values: len(values) == count, + seconds=70, + ) + assert len({row["request_id"] for row in landed}) == count + for row in landed: + value: Final = row["request_tags"] + assert (json.loads(value) if isinstance(value, str) else value) == EXPECTED + return landed + + +def _worker_pids(owned) -> tuple[int, ...]: + workers: Final = tuple( + child + for child in psutil.Process(owned.process.pid).children(recursive=True) + if any(marker in " ".join(child.cmdline()) for marker in ("spawn_main", "integration._support.proxy")) + ) + return tuple(worker.pid for worker in workers) + + def test_burst_across_routes_records_tags_once_per_response(gateway: Gateway, tmp_path: Path) -> None: with wire_server(provider_reply) as wire: config: Final = write_config(tmp_path, {"litellm_settings": {"extra_spend_tag_headers": ["x-tenant-id"]}}) with ( - owned_proxy(gateway, tmp_path, provider_env(wire.url), config=config, workers=2) as candidate, - candidate.scenario() as scenario, + owned_proxy_process(gateway, tmp_path, provider_env(wire.url), config=config, workers=2) as owned, + owned.gateway.scenario() as scenario, ): - model: Final = scenario.model(model=f"openai/{OPENAI_MODEL}", api_base=f"{wire.url}/v1") + candidate: Final = owned.gateway + anthropic_model, openai_model = _deployments(scenario, wire.url) key: Final = scenario.key() def burst(index: int) -> tuple: - return _tagged_requests(candidate, key, model, stream=index % 2 == 1, index=index) + return _tagged_requests( + candidate, key, anthropic_model, openai_model, stream=index % 2 == 1, index=index + ) with ThreadPoolExecutor(max_workers=10) as pool: responses: Final = [response for group in pool.map(burst, range(10)) for response in group] - assert len(responses) == 30 + assert len(responses) == 50 assert all(response.status_code == 200 for response in responses), [ (response.status_code, response.text[:200]) for response in responses ] ids: Final = [_ids(response) for response in responses] - assert len(set(ids)) == 30, "duplicate upstream id in burst" - assert len(wire.drain()) == 30 - digest: Final = sha256(key.encode()).hexdigest() - landed: Final = eventually( - lambda: read_rows( - 'SELECT request_id, request_tags FROM "LiteLLM_SpendLogs" WHERE api_key=%s', (digest,) - ), - lambda values: len(values) == 30, - seconds=70, - ) - assert len({row["request_id"] for row in landed}) == 30 - for row in landed: - value: Final = row["request_tags"] - assert (json.loads(value) if isinstance(value, str) else value) == EXPECTED + assert len(set(ids)) == 50, "duplicate upstream id in burst" + assert len(wire.drain()) == 50 + pids: Final = _worker_pids(owned) + assert len(set(pids)) == 2, f"expected two uvicorn workers, found {pids}" + assert all(worker.is_running() for worker in psutil.process_iter(pids)) + _landed_tags(key, 50) -# C2: generic_api sink down mid burst; spend rows still land exactly once with the tags def test_sink_outage_does_not_lose_spend_log_tags(gateway: Gateway, tmp_path: Path) -> None: - stopped: Final = threading.Event() + down: Final = threading.Event() + delivered: Final = [] # mutable-ok: sink thread appends between drains def stoppable_sink(request: Request) -> Reply: - stopped.wait(timeout=30) - return Reply(status=503) + if down.is_set(): + return Reply(status=503) + delivered.append(request) + return Reply() with wire_server(provider_reply) as wire, wire_server(stoppable_sink) as endpoint: config: Final = write_config( @@ -116,7 +163,7 @@ def test_sink_outage_does_not_lose_spend_log_tags(gateway: Gateway, tmp_path: Pa }, ) with ( - owned_proxy( + owned_proxy_process( gateway, tmp_path, { @@ -126,45 +173,65 @@ def test_sink_outage_does_not_lose_spend_log_tags(gateway: Gateway, tmp_path: Pa }, config=config, workers=2, - ) as candidate, - candidate.scenario() as scenario, + ) as owned, + owned.gateway.scenario() as scenario, ): - model: Final = scenario.model(model=f"openai/{OPENAI_MODEL}", api_base=f"{wire.url}/v1") + candidate: Final = owned.gateway + anthropic_model, openai_model = _deployments(scenario, wire.url) key: Final = scenario.key() def burst(index: int) -> tuple: - return _tagged_requests(candidate, key, model, stream=False, index=index) + return _tagged_requests(candidate, key, anthropic_model, openai_model, stream=False, index=index) - with ThreadPoolExecutor(max_workers=6) as pool: - first: Final = [response for group in pool.map(burst, range(6)) for response in group] - stopped.set() # sink goes down: the peer now returns 503 to every flush + with ThreadPoolExecutor(max_workers=5) as pool: + first: Final = [response for group in pool.map(burst, range(4)) for response in group] + assert all(response.status_code == 200 for response in first), [ + (response.status_code, response.text[:200]) for response in first + ] - def second_burst(index: int) -> tuple: - return _tagged_requests(candidate, key, model, stream=False, index=100 + index) + def call_ids(responses: list) -> set: + return {response.headers["x-litellm-call-id"] for response in responses} - with ThreadPoolExecutor(max_workers=6) as pool: - second: Final = [response for group in pool.map(second_burst, range(6)) for response in group] - responses: Final = [*first, *second] + first_ids: Final = call_ids(first) + + def events_for(ids: set) -> set: + return { + event["litellm_call_id"] + for batch in delivered + for event in json.loads(batch.body) + if event.get("litellm_call_id") in ids + } + + eventually(lambda: events_for(first_ids), lambda found: found == first_ids, seconds=70) + down.set() + with ThreadPoolExecutor(max_workers=5) as pool: + second: Final = [ + response for group in pool.map(lambda i: burst(100 + i), range(4)) for response in group + ] + down.clear() + third: Final = _tagged_requests(candidate, key, anthropic_model, openai_model, False, 200) + responses: Final = [*first, *second, *third] assert all(response.status_code == 200 for response in responses), [ (response.status_code, response.text[:200]) for response in responses ] - ids: Final = [_ids(response) for response in responses] - assert len(set(ids)) == len(ids), "duplicate upstream id in burst" - digest: Final = sha256(key.encode()).hexdigest() - landed: Final = eventually( - lambda: read_rows( - 'SELECT request_id, request_tags FROM "LiteLLM_SpendLogs" WHERE api_key=%s', (digest,) - ), - lambda values: len(values) == 36, + second_ids: Final = call_ids(second) + third_ids: Final = call_ids(list(third)) + recovery_probe: Final = next(iter(third_ids)) + eventually( + lambda: events_for(third_ids), + lambda found: recovery_probe in found, seconds=70, ) - assert len({row["request_id"] for row in landed}) == 36 - for row in landed: - value: Final = row["request_tags"] - assert (json.loads(value) if isinstance(value, str) else value) == EXPECTED + second_delivered: Final = eventually( + lambda: events_for(second_ids), + lambda found: len(found) == len(second_ids), + seconds=30, + return_last_on_timeout=True, + ) + assert second_delivered == second_ids + _landed_tags(key, len(responses)) -# C3: killing one proxy worker mid burst loses no spend row def test_worker_kill_mid_burst_loses_no_spend_rows(gateway: Gateway, tmp_path: Path) -> None: with wire_server(provider_reply) as wire: config: Final = write_config(tmp_path, {"litellm_settings": {"extra_spend_tag_headers": ["x-tenant-id"]}}) @@ -173,41 +240,38 @@ def test_worker_kill_mid_burst_loses_no_spend_rows(gateway: Gateway, tmp_path: P owned.gateway.scenario() as scenario, ): candidate: Final = owned.gateway - model: Final = scenario.model(model=f"openai/{OPENAI_MODEL}", api_base=f"{wire.url}/v1") + anthropic_model, openai_model = _deployments(scenario, wire.url) key: Final = scenario.key() def burst(index: int) -> tuple: - return _tagged_requests(candidate, key, model, stream=False, index=index) + return _tagged_requests(candidate, key, anthropic_model, openai_model, stream=False, index=index) - with ThreadPoolExecutor(max_workers=6) as pool: - first: Final = [response for group in pool.map(burst, range(6)) for response in group] + with ThreadPoolExecutor(max_workers=5) as pool: + first: Final = [response for group in pool.map(burst, range(4)) for response in group] workers: Final = [ child for child in psutil.Process(owned.process.pid).children(recursive=True) - if child.status() != psutil.STATUS_ZOMBIE + if any(marker in " ".join(child.cmdline()) for marker in ("spawn_main", "integration._support.proxy")) ] - assert len(workers) >= 2, f"expected two proxy workers, found {[w.pid for w in workers]}" + assert len(workers) == 2, ( + f"expected two uvicorn workers, found {[(w.pid, w.cmdline()[:3]) for w in workers]}" + ) workers[0].kill() + psutil.wait_procs(workers[:1], timeout=10) + assert not workers[0].is_running() - def second_burst(index: int) -> tuple: - return _tagged_requests(candidate, key, model, stream=False, index=100 + index) - - with ThreadPoolExecutor(max_workers=6) as pool: - second: Final = [response for group in pool.map(second_burst, range(6)) for response in group] + with ThreadPoolExecutor(max_workers=5) as pool: + second: Final = [ + response for group in pool.map(lambda i: burst(100 + i), range(4)) for response in group + ] responses: Final = [*first, *second] + for position in range(len(ROUTES)): + statuses: Final = { + responses[offset + position].status_code for offset in range(0, len(responses), len(ROUTES)) + } + assert 200 in statuses, f"no surviving 200 for route {ROUTES[position]}: {statuses}" ok: Final = [response for response in responses if response.status_code == 200] ids: Final = [_ids(response) for response in ok] assert len(set(ids)) == len(ids), "duplicate upstream id in burst" - digest: Final = sha256(key.encode()).hexdigest() - landed: Final = eventually( - lambda: read_rows( - 'SELECT request_id, request_tags FROM "LiteLLM_SpendLogs" WHERE api_key=%s', (digest,) - ), - lambda values: len(values) == len(ids), - seconds=70, - ) - assert len({row["request_id"] for row in landed}) == len(ids) - for row in landed: - value: Final = row["request_tags"] - assert (json.loads(value) if isinstance(value, str) else value) == EXPECTED + _landed_tags(key, len(ok)) From 5e3aa7938fdd68b5eeabae86539dad221caeb36b Mon Sep 17 00:00:00 2001 From: kerry Date: Fri, 2 Oct 2026 21:06:48 +0000 Subject: [PATCH 09/23] test(spend): fix chaos helper import name Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- .../spend/test_spend_log_request_tags_chaos.py | 13 +++++++++---- 1 file changed, 9 insertions(+), 4 deletions(-) diff --git a/tests/integration/spend/test_spend_log_request_tags_chaos.py b/tests/integration/spend/test_spend_log_request_tags_chaos.py index 26b33a9c68c..5069c6d0d59 100644 --- a/tests/integration/spend/test_spend_log_request_tags_chaos.py +++ b/tests/integration/spend/test_spend_log_request_tags_chaos.py @@ -11,8 +11,8 @@ from integration._support.database import read_rows from integration._support.process import owned_proxy_process from integration._support.wire import Reply, Request, wire_server from integration.spend._request_tag_helpers import ( - MODEL, - OPENAI_MODEL, + ANTHROPIC_ANTHROPIC_MODEL, + OPENAI_ANTHROPIC_MODEL, T3, provider_env, provider_reply, @@ -47,14 +47,19 @@ def _tagged_requests( candidate.request( "POST", "/anthropic/v1/messages", - {"model": MODEL, "max_tokens": 16, "messages": [{"role": "user", "content": marker}], "stream": stream}, + { + "model": ANTHROPIC_MODEL, + "max_tokens": 16, + "messages": [{"role": "user", "content": marker}], + "stream": stream, + }, key=key, headers=ANTHROPIC_HEADERS, ), candidate.request( "POST", "/openai/v1/chat/completions", - {"model": OPENAI_MODEL, "messages": [{"role": "user", "content": marker}], "stream": stream}, + {"model": OPENAI_ANTHROPIC_MODEL, "messages": [{"role": "user", "content": marker}], "stream": stream}, key=key, headers=HEADERS, ), From 8941fd5babebdc28496cb55a3866cc46bd5ce54c Mon Sep 17 00:00:00 2001 From: kerry Date: Fri, 2 Oct 2026 21:06:54 +0000 Subject: [PATCH 10/23] test(spend): fix model constant names in chaos cells Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- .../spend/test_spend_log_request_tags_chaos.py | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/tests/integration/spend/test_spend_log_request_tags_chaos.py b/tests/integration/spend/test_spend_log_request_tags_chaos.py index 5069c6d0d59..5d771ac3b4b 100644 --- a/tests/integration/spend/test_spend_log_request_tags_chaos.py +++ b/tests/integration/spend/test_spend_log_request_tags_chaos.py @@ -11,8 +11,8 @@ from integration._support.database import read_rows from integration._support.process import owned_proxy_process from integration._support.wire import Reply, Request, wire_server from integration.spend._request_tag_helpers import ( - ANTHROPIC_ANTHROPIC_MODEL, - OPENAI_ANTHROPIC_MODEL, + ANTHROPIC_MODEL, + OPENAI_MODEL, T3, provider_env, provider_reply, @@ -59,7 +59,7 @@ def _tagged_requests( candidate.request( "POST", "/openai/v1/chat/completions", - {"model": OPENAI_ANTHROPIC_MODEL, "messages": [{"role": "user", "content": marker}], "stream": stream}, + {"model": OPENAI_MODEL, "messages": [{"role": "user", "content": marker}], "stream": stream}, key=key, headers=HEADERS, ), @@ -87,7 +87,7 @@ def _tagged_requests( def _deployments(scenario, url: str) -> tuple[str, str]: return ( - scenario.model(model=f"anthropic/{MODEL}", api_base=url, api_key="synthetic-anthropic-key"), + scenario.model(model=f"anthropic/{ANTHROPIC_MODEL}", api_base=url, api_key="synthetic-anthropic-key"), scenario.model(model=f"openai/{OPENAI_MODEL}", api_base=f"{url}/v1"), ) From 58a09c24946dda626c35687b0afa7f1d786af6ae Mon Sep 17 00:00:00 2001 From: kerry Date: Fri, 2 Oct 2026 21:37:37 +0000 Subject: [PATCH 11/23] test(spend): unique burst markers and resilient worker-kill row check Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- .../test_spend_log_request_tags_chaos.py | 21 +++++++++++-------- 1 file changed, 12 insertions(+), 9 deletions(-) diff --git a/tests/integration/spend/test_spend_log_request_tags_chaos.py b/tests/integration/spend/test_spend_log_request_tags_chaos.py index 5d771ac3b4b..7f71806b984 100644 --- a/tests/integration/spend/test_spend_log_request_tags_chaos.py +++ b/tests/integration/spend/test_spend_log_request_tags_chaos.py @@ -1,5 +1,6 @@ import json import threading +import uuid from concurrent.futures import ThreadPoolExecutor from hashlib import sha256 from pathlib import Path @@ -42,7 +43,7 @@ def _tagged_requests( candidate: Gateway, key: str, anthropic_model: str, openai_model: str, stream: bool, index: int ) -> tuple: """One call per route in ROUTES order with the same client headers.""" - marker: Final = f"burst {index}" + marker: Final = f"burst {index} {uuid.uuid4().hex}" return ( candidate.request( "POST", @@ -92,14 +93,14 @@ def _deployments(scenario, url: str) -> tuple[str, str]: ) -def _landed_tags(key: str, count: int) -> list[dict]: +def _landed_tags(key: str, satisfied) -> list[dict]: digest: Final = sha256(key.encode()).hexdigest() landed: Final = eventually( lambda: read_rows('SELECT request_id, request_tags FROM "LiteLLM_SpendLogs" WHERE api_key=%s', (digest,)), - lambda values: len(values) == count, - seconds=70, + satisfied, + seconds=60, ) - assert len({row["request_id"] for row in landed}) == count + assert len({row["request_id"] for row in landed}) == len(landed) for row in landed: value: Final = row["request_tags"] assert (json.loads(value) if isinstance(value, str) else value) == EXPECTED @@ -142,8 +143,8 @@ def test_burst_across_routes_records_tags_once_per_response(gateway: Gateway, tm assert len(wire.drain()) == 50 pids: Final = _worker_pids(owned) assert len(set(pids)) == 2, f"expected two uvicorn workers, found {pids}" - assert all(worker.is_running() for worker in psutil.process_iter(pids)) - _landed_tags(key, 50) + assert all(psutil.Process(pid).is_running() for pid in pids) + _landed_tags(key, lambda values: len(values) == 50) def test_sink_outage_does_not_lose_spend_log_tags(gateway: Gateway, tmp_path: Path) -> None: @@ -234,7 +235,7 @@ def test_sink_outage_does_not_lose_spend_log_tags(gateway: Gateway, tmp_path: Pa return_last_on_timeout=True, ) assert second_delivered == second_ids - _landed_tags(key, len(responses)) + _landed_tags(key, lambda values: len(values) == len(responses)) def test_worker_kill_mid_burst_loses_no_spend_rows(gateway: Gateway, tmp_path: Path) -> None: @@ -276,7 +277,9 @@ def test_worker_kill_mid_burst_loses_no_spend_rows(gateway: Gateway, tmp_path: P responses[offset + position].status_code for offset in range(0, len(responses), len(ROUTES)) } assert 200 in statuses, f"no surviving 200 for route {ROUTES[position]}: {statuses}" + second_ok: Final = [response for response in second if response.status_code == 200] + assert second_ok, "surviving worker served no second-burst request" ok: Final = [response for response in responses if response.status_code == 200] ids: Final = [_ids(response) for response in ok] assert len(set(ids)) == len(ids), "duplicate upstream id in burst" - _landed_tags(key, len(ok)) + landed: Final = _landed_tags(key, lambda values: len(values) == len(ok)) From c3219e07719ce9b5e2795fcde6595b335dccfcdb Mon Sep 17 00:00:00 2001 From: kerry Date: Fri, 2 Oct 2026 22:06:30 +0000 Subject: [PATCH 12/23] test(spend): shrink chaos bursts and pin post-kill spend contract Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- .../spend/test_spend_log_request_tags_chaos.py | 11 ++++++----- 1 file changed, 6 insertions(+), 5 deletions(-) diff --git a/tests/integration/spend/test_spend_log_request_tags_chaos.py b/tests/integration/spend/test_spend_log_request_tags_chaos.py index 7f71806b984..9dd86518ddb 100644 --- a/tests/integration/spend/test_spend_log_request_tags_chaos.py +++ b/tests/integration/spend/test_spend_log_request_tags_chaos.py @@ -190,7 +190,7 @@ def test_sink_outage_does_not_lose_spend_log_tags(gateway: Gateway, tmp_path: Pa return _tagged_requests(candidate, key, anthropic_model, openai_model, stream=False, index=index) with ThreadPoolExecutor(max_workers=5) as pool: - first: Final = [response for group in pool.map(burst, range(4)) for response in group] + first: Final = [response for group in pool.map(burst, range(3)) for response in group] assert all(response.status_code == 200 for response in first), [ (response.status_code, response.text[:200]) for response in first ] @@ -212,7 +212,7 @@ def test_sink_outage_does_not_lose_spend_log_tags(gateway: Gateway, tmp_path: Pa down.set() with ThreadPoolExecutor(max_workers=5) as pool: second: Final = [ - response for group in pool.map(lambda i: burst(100 + i), range(4)) for response in group + response for group in pool.map(lambda i: burst(100 + i), range(3)) for response in group ] down.clear() third: Final = _tagged_requests(candidate, key, anthropic_model, openai_model, False, 200) @@ -253,7 +253,7 @@ def test_worker_kill_mid_burst_loses_no_spend_rows(gateway: Gateway, tmp_path: P return _tagged_requests(candidate, key, anthropic_model, openai_model, stream=False, index=index) with ThreadPoolExecutor(max_workers=5) as pool: - first: Final = [response for group in pool.map(burst, range(4)) for response in group] + first: Final = [response for group in pool.map(burst, range(3)) for response in group] workers: Final = [ child @@ -263,13 +263,14 @@ def test_worker_kill_mid_burst_loses_no_spend_rows(gateway: Gateway, tmp_path: P assert len(workers) == 2, ( f"expected two uvicorn workers, found {[(w.pid, w.cmdline()[:3]) for w in workers]}" ) + first_landed: Final = _landed_tags(key, lambda values: len(values) == len(first)) workers[0].kill() psutil.wait_procs(workers[:1], timeout=10) assert not workers[0].is_running() with ThreadPoolExecutor(max_workers=5) as pool: second: Final = [ - response for group in pool.map(lambda i: burst(100 + i), range(4)) for response in group + response for group in pool.map(lambda i: burst(100 + i), range(3)) for response in group ] responses: Final = [*first, *second] for position in range(len(ROUTES)): @@ -282,4 +283,4 @@ def test_worker_kill_mid_burst_loses_no_spend_rows(gateway: Gateway, tmp_path: P ok: Final = [response for response in responses if response.status_code == 200] ids: Final = [_ids(response) for response in ok] assert len(set(ids)) == len(ids), "duplicate upstream id in burst" - landed: Final = _landed_tags(key, lambda values: len(values) == len(ok)) + _landed_tags(key, lambda values: len(values) == len(first_landed) + len(second_ok)) From 1bf95121d3b069c206e1dfc6369cb4d73364299a Mon Sep 17 00:00:00 2001 From: kerry Date: Fri, 2 Oct 2026 23:03:15 +0000 Subject: [PATCH 13/23] test(spend): derive SDK user agents and strengthen audit assertions Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- .../spend/test_spend_log_request_tags.py | 48 ++++++++++++++----- .../test_spend_log_request_tags_chaos.py | 11 ++++- 2 files changed, 46 insertions(+), 13 deletions(-) diff --git a/tests/integration/spend/test_spend_log_request_tags.py b/tests/integration/spend/test_spend_log_request_tags.py index e064d9f6496..0c8ab3a6817 100644 --- a/tests/integration/spend/test_spend_log_request_tags.py +++ b/tests/integration/spend/test_spend_log_request_tags.py @@ -13,7 +13,7 @@ import pytest import yaml from integration._support.client import Gateway, eventually from integration._support.database import read_rows -from integration._support.process import owned_proxy +from integration._support.process import owned_proxy, owned_proxy_process from integration._support.wire import Reply, Request, wire_server from integration.spend._request_tag_helpers import ( GEMINI_MODEL, @@ -115,7 +115,7 @@ def test_pass_through_anthropic_sdk_records_header_tags(gateway: Gateway, tmp_pa assert message.id.startswith("msg_") assert len(wire.drain()) == 1 assert eventually(lambda: tags_by_key(key), lambda tags: len(tags) == 1, seconds=70) == [ - ["User-Agent: Anthropic", "User-Agent: Anthropic/Python 0.84.0", TENANT_TAG] + ["User-Agent: Anthropic", f"User-Agent: Anthropic/Python {anthropic.__version__}", TENANT_TAG] ] @@ -137,7 +137,7 @@ def test_pass_through_anthropic_sdk_stream_records_header_tags(gateway: Gateway, assert message.id.startswith("msg_") assert len(wire.drain()) == 1 assert eventually(lambda: tags_by_key(key), lambda tags: len(tags) == 1, seconds=70) == [ - ["User-Agent: Anthropic", "User-Agent: Anthropic/Python 0.84.0", TENANT_TAG] + ["User-Agent: Anthropic", f"User-Agent: Anthropic/Python {anthropic.__version__}", TENANT_TAG] ] @@ -191,7 +191,7 @@ def test_pass_through_openai_sdk_records_header_tags(gateway: Gateway, tmp_path: assert request_id.startswith("chatcmpl_") assert len(wire.drain()) == 1 assert eventually(lambda: tags_by_key(key), lambda tags: len(tags) == 1, seconds=70) == [ - ["User-Agent: OpenAI", "User-Agent: OpenAI/Python 2.33.0", TENANT_TAG] + ["User-Agent: OpenAI", f"User-Agent: OpenAI/Python {openai.__version__}", TENANT_TAG] ] @@ -284,7 +284,7 @@ def test_pass_through_openai_async_sdk_stream_records_header_tags(gateway: Gatew assert request_id.startswith("chatcmpl_") assert len(wire.drain()) == 1 assert eventually(lambda: tags_by_key(key), lambda tags: len(tags) == 1, seconds=70) == [ - ["User-Agent: AsyncOpenAI", "User-Agent: AsyncOpenAI/Python 2.33.0", TENANT_TAG] + ["User-Agent: AsyncOpenAI", f"User-Agent: AsyncOpenAI/Python {openai.__version__}", TENANT_TAG] ] @@ -903,9 +903,9 @@ def test_repeated_requests_each_record_tags(gateway: Gateway, tmp_path: Path, ro model=f"anthropic/{MODEL}", api_base=wire.url, api_key="synthetic-anthropic-key" ) key: Final = scenario.key() - ids: Final = [] - for _ in range(3): - response: Final = candidate.request( + + def repeat(_: int): + return candidate.request( "POST", route, { @@ -916,8 +916,12 @@ def test_repeated_requests_each_record_tags(gateway: Gateway, tmp_path: Path, ro key=key, headers={**SENT_HEADERS, "anthropic-version": "2023-06-01"}, ) - assert response.status_code == 200, response.text - ids.append(response.json()["id"]) + + responses: Final = tuple(repeat(index) for index in range(3)) + assert all(response.status_code == 200 for response in responses), [ + (response.status_code, response.text) for response in responses + ] + ids: Final = tuple(response.json()["id"] for response in responses) assert len(wire.drain()) == 3 for identity in ids: assert eventually( @@ -959,9 +963,10 @@ def test_guardrail_mode_tag_decider_is_unchanged_on_pass_through(gateway: Gatewa }, ) with ( - owned_proxy(gateway, tmp_path, provider_env(wire.url), config=config, workers=2) as candidate, - candidate.scenario() as scenario, + owned_proxy_process(gateway, tmp_path, provider_env(wire.url), config=config, workers=2) as owned, + owned.gateway.scenario() as scenario, ): + candidate: Final = owned.gateway model: Final = scenario.model(model=f"openai/{OPENAI_MODEL}", api_base=f"{wire.url}/v1") key: Final = scenario.key(allowed_passthrough_routes=["/custom-anthropic"]) body: Final = {"model": MODEL, "max_tokens": 16, "messages": [{"role": "user", "content": "bananablock"}]} @@ -977,3 +982,22 @@ def test_guardrail_mode_tag_decider_is_unchanged_on_pass_through(gateway: Gatewa ) assert control.status_code != 200, control.text assert len(wire.drain()) == 0, "tag-matched guardrail should have blocked before the upstream" + call_id: Final = control.headers["x-litellm-call-id"] + + def blocked_rows() -> list[dict]: + return read_rows( + 'SELECT metadata FROM "LiteLLM_SpendLogs" WHERE litellm_call_id=%s', + (call_id,), + ) + + spend_row: Final = eventually( + blocked_rows, + lambda rows: len(rows) == 1 and "bananablock" in json.dumps(rows[0]["metadata"]), + seconds=30, + return_last_on_timeout=True, + ) + if not (len(spend_row) == 1 and "bananablock" in json.dumps(spend_row[0]["metadata"])): + log_text: Final = owned.log.read_text() + assert "Content blocked: keyword 'bananablock' detected" in log_text, ( + "guardrail block text not in spend row or proxy log" + ) diff --git a/tests/integration/spend/test_spend_log_request_tags_chaos.py b/tests/integration/spend/test_spend_log_request_tags_chaos.py index 9dd86518ddb..7b0b627c915 100644 --- a/tests/integration/spend/test_spend_log_request_tags_chaos.py +++ b/tests/integration/spend/test_spend_log_request_tags_chaos.py @@ -150,9 +150,11 @@ def test_burst_across_routes_records_tags_once_per_response(gateway: Gateway, tm def test_sink_outage_does_not_lose_spend_log_tags(gateway: Gateway, tmp_path: Path) -> None: down: Final = threading.Event() delivered: Final = [] # mutable-ok: sink thread appends between drains + rejected: Final = [] # mutable-ok: sink thread appends between drains def stoppable_sink(request: Request) -> Reply: if down.is_set(): + rejected.append(request) return Reply(status=503) delivered.append(request) return Reply() @@ -214,13 +216,20 @@ def test_sink_outage_does_not_lose_spend_log_tags(gateway: Gateway, tmp_path: Pa second: Final = [ response for group in pool.map(lambda i: burst(100 + i), range(3)) for response in group ] + second_ids: Final = call_ids(second) + outage_probe: Final = eventually( + lambda: (len(rejected), events_for(second_ids)), + lambda state: state[0] >= 1 and state[1] == set(), + seconds=30, + ) + assert outage_probe[0] >= 1, "sink saw no rejection during the outage window" + assert outage_probe[1] == set(), "burst-2 event delivered to a down sink" down.clear() third: Final = _tagged_requests(candidate, key, anthropic_model, openai_model, False, 200) responses: Final = [*first, *second, *third] assert all(response.status_code == 200 for response in responses), [ (response.status_code, response.text[:200]) for response in responses ] - second_ids: Final = call_ids(second) third_ids: Final = call_ids(list(third)) recovery_probe: Final = next(iter(third_ids)) eventually( From 208c66ffe1842f4245afc1cac2b7e56483c5d676 Mon Sep 17 00:00:00 2001 From: kerry Date: Fri, 2 Oct 2026 23:11:28 +0000 Subject: [PATCH 14/23] test(spend): pin guardrail block evidence and outage delivery contract Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- .../spend/test_spend_log_request_tags_chaos.py | 16 ++++++++++++++-- 1 file changed, 14 insertions(+), 2 deletions(-) diff --git a/tests/integration/spend/test_spend_log_request_tags_chaos.py b/tests/integration/spend/test_spend_log_request_tags_chaos.py index 7b0b627c915..ac4bb07ac6c 100644 --- a/tests/integration/spend/test_spend_log_request_tags_chaos.py +++ b/tests/integration/spend/test_spend_log_request_tags_chaos.py @@ -237,13 +237,25 @@ def test_sink_outage_does_not_lose_spend_log_tags(gateway: Gateway, tmp_path: Pa lambda found: recovery_probe in found, seconds=70, ) - second_delivered: Final = eventually( + eventually( lambda: events_for(second_ids), lambda found: len(found) == len(second_ids), seconds=30, return_last_on_timeout=True, ) - assert second_delivered == second_ids + second_occurrences: Final = [ + event["litellm_call_id"] + for batch in delivered + for event in json.loads(batch.body) + if event.get("litellm_call_id") in second_ids + ] + assert len(second_occurrences) == len(set(second_occurrences)), ( + "duplicate burst-2 delivery after the outage" + ) + assert events_for(second_ids) < second_ids, ( + "events flushed during the outage should be dropped, not redelivered" + ) + _landed_tags(key, lambda values: len(values) == len(responses)) From 290b552b1110b9e091d0c7084a729fa5deb4aa51 Mon Sep 17 00:00:00 2001 From: kerry Date: Fri, 2 Oct 2026 23:11:57 +0000 Subject: [PATCH 15/23] test(spend): key-scoped guardrail block evidence without call-id header Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- tests/integration/spend/test_spend_log_request_tags.py | 10 +++++----- 1 file changed, 5 insertions(+), 5 deletions(-) diff --git a/tests/integration/spend/test_spend_log_request_tags.py b/tests/integration/spend/test_spend_log_request_tags.py index 0c8ab3a6817..1180f065d48 100644 --- a/tests/integration/spend/test_spend_log_request_tags.py +++ b/tests/integration/spend/test_spend_log_request_tags.py @@ -982,21 +982,21 @@ def test_guardrail_mode_tag_decider_is_unchanged_on_pass_through(gateway: Gatewa ) assert control.status_code != 200, control.text assert len(wire.drain()) == 0, "tag-matched guardrail should have blocked before the upstream" - call_id: Final = control.headers["x-litellm-call-id"] + digest: Final = sha256(key.encode()).hexdigest() def blocked_rows() -> list[dict]: return read_rows( - 'SELECT metadata FROM "LiteLLM_SpendLogs" WHERE litellm_call_id=%s', - (call_id,), + 'SELECT metadata, litellm_call_id FROM "LiteLLM_SpendLogs" WHERE api_key=%s', + (digest,), ) spend_row: Final = eventually( blocked_rows, - lambda rows: len(rows) == 1 and "bananablock" in json.dumps(rows[0]["metadata"]), + lambda rows: any("bananablock" in json.dumps(row["metadata"]) for row in rows), seconds=30, return_last_on_timeout=True, ) - if not (len(spend_row) == 1 and "bananablock" in json.dumps(spend_row[0]["metadata"])): + if not any("bananablock" in json.dumps(row["metadata"]) for row in spend_row): log_text: Final = owned.log.read_text() assert "Content blocked: keyword 'bananablock' detected" in log_text, ( "guardrail block text not in spend row or proxy log" From ead0cededf00bd460aab58e97eb1054f70702177 Mon Sep 17 00:00:00 2001 From: kerry Date: Fri, 2 Oct 2026 23:17:46 +0000 Subject: [PATCH 16/23] test(spend): tighten outage and guardrail assertions, satisfy type discipline Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- .../spend/test_spend_log_request_tags.py | 41 +++++++-------- .../test_spend_log_request_tags_chaos.py | 52 +++++++------------ 2 files changed, 39 insertions(+), 54 deletions(-) diff --git a/tests/integration/spend/test_spend_log_request_tags.py b/tests/integration/spend/test_spend_log_request_tags.py index 1180f065d48..15059d67aea 100644 --- a/tests/integration/spend/test_spend_log_request_tags.py +++ b/tests/integration/spend/test_spend_log_request_tags.py @@ -8,6 +8,8 @@ from typing import Final import anthropic import httpx +from collections.abc import Mapping, Sequence +from itertools import chain import openai import pytest import yaml @@ -403,9 +405,10 @@ def test_pass_through_tags_reach_generic_api_sink(gateway: Gateway, tmp_path: Pa assert len(wire.drain()) == 1 batches: Final[list[Request]] = [] # mutable-ok: drain consumes the queue between polls - def delivered() -> list[dict]: + def delivered() -> Sequence[Mapping]: batches.extend(endpoint.drain()) - return [event for batch in batches for event in json.loads(batch.body) if event.get("id") == request_id] + events: Final = chain.from_iterable(json.loads(batch.body) for batch in batches) + return [event for event in events if event.get("id") == request_id] events: Final = eventually(delivered, lambda values: len(values) == 1, seconds=30) assert events[0]["request_tags"] == EXPECTED_TAGS @@ -904,7 +907,7 @@ def test_repeated_requests_each_record_tags(gateway: Gateway, tmp_path: Path, ro ) key: Final = scenario.key() - def repeat(_: int): + def repeat(_: int) -> httpx.Response: return candidate.request( "POST", route, @@ -963,10 +966,9 @@ def test_guardrail_mode_tag_decider_is_unchanged_on_pass_through(gateway: Gatewa }, ) with ( - owned_proxy_process(gateway, tmp_path, provider_env(wire.url), config=config, workers=2) as owned, - owned.gateway.scenario() as scenario, + owned_proxy(gateway, tmp_path, provider_env(wire.url), config=config, workers=2) as candidate, + candidate.scenario() as scenario, ): - candidate: Final = owned.gateway model: Final = scenario.model(model=f"openai/{OPENAI_MODEL}", api_base=f"{wire.url}/v1") key: Final = scenario.key(allowed_passthrough_routes=["/custom-anthropic"]) body: Final = {"model": MODEL, "max_tokens": 16, "messages": [{"role": "user", "content": "bananablock"}]} @@ -984,20 +986,17 @@ def test_guardrail_mode_tag_decider_is_unchanged_on_pass_through(gateway: Gatewa assert len(wire.drain()) == 0, "tag-matched guardrail should have blocked before the upstream" digest: Final = sha256(key.encode()).hexdigest() - def blocked_rows() -> list[dict]: - return read_rows( - 'SELECT metadata, litellm_call_id FROM "LiteLLM_SpendLogs" WHERE api_key=%s', - (digest,), - ) + def blocked_rows() -> Sequence[Mapping]: + return [ + row + for row in read_rows( + 'SELECT metadata FROM "LiteLLM_SpendLogs" WHERE api_key=%s', + (digest,), + ) + if "Content blocked: keyword 'bananablock' detected" in json.dumps(row["metadata"]) + and row["metadata"].get("status") == "failure" + ] - spend_row: Final = eventually( - blocked_rows, - lambda rows: any("bananablock" in json.dumps(row["metadata"]) for row in rows), - seconds=30, - return_last_on_timeout=True, + assert eventually(blocked_rows, lambda rows: len(rows) == 1, seconds=70), ( + "guardrail block was not recorded on the key's spend row" ) - if not any("bananablock" in json.dumps(row["metadata"]) for row in spend_row): - log_text: Final = owned.log.read_text() - assert "Content blocked: keyword 'bananablock' detected" in log_text, ( - "guardrail block text not in spend row or proxy log" - ) diff --git a/tests/integration/spend/test_spend_log_request_tags_chaos.py b/tests/integration/spend/test_spend_log_request_tags_chaos.py index ac4bb07ac6c..782f0f9ce3c 100644 --- a/tests/integration/spend/test_spend_log_request_tags_chaos.py +++ b/tests/integration/spend/test_spend_log_request_tags_chaos.py @@ -1,12 +1,15 @@ import json import threading import uuid +from collections.abc import AbstractSet, Mapping, Sequence +from itertools import chain from concurrent.futures import ThreadPoolExecutor from hashlib import sha256 from pathlib import Path from typing import Final import psutil +import pytest from integration._support.client import Gateway, eventually from integration._support.database import read_rows from integration._support.process import owned_proxy_process @@ -93,7 +96,7 @@ def _deployments(scenario, url: str) -> tuple[str, str]: ) -def _landed_tags(key: str, satisfied) -> list[dict]: +def _landed_tags(key: str, satisfied) -> Sequence[Mapping]: digest: Final = sha256(key.encode()).hexdigest() landed: Final = eventually( lambda: read_rows('SELECT request_id, request_tags FROM "LiteLLM_SpendLogs" WHERE api_key=%s', (digest,)), @@ -133,7 +136,7 @@ def test_burst_across_routes_records_tags_once_per_response(gateway: Gateway, tm ) with ThreadPoolExecutor(max_workers=10) as pool: - responses: Final = [response for group in pool.map(burst, range(10)) for response in group] + responses: Final = chain.from_iterable(pool.map(burst, range(10))) assert len(responses) == 50 assert all(response.status_code == 200 for response in responses), [ (response.status_code, response.text[:200]) for response in responses @@ -147,10 +150,11 @@ def test_burst_across_routes_records_tags_once_per_response(gateway: Gateway, tm _landed_tags(key, lambda values: len(values) == 50) +@pytest.mark.timeout(240) # proxy boot, a full outage window and post-recovery delivery exceed the 90s default def test_sink_outage_does_not_lose_spend_log_tags(gateway: Gateway, tmp_path: Path) -> None: down: Final = threading.Event() - delivered: Final = [] # mutable-ok: sink thread appends between drains - rejected: Final = [] # mutable-ok: sink thread appends between drains + delivered: Final = [] + rejected: Final = [] def stoppable_sink(request: Request) -> Reply: if down.is_set(): @@ -192,30 +196,24 @@ def test_sink_outage_does_not_lose_spend_log_tags(gateway: Gateway, tmp_path: Pa return _tagged_requests(candidate, key, anthropic_model, openai_model, stream=False, index=index) with ThreadPoolExecutor(max_workers=5) as pool: - first: Final = [response for group in pool.map(burst, range(3)) for response in group] + first: Final = chain.from_iterable(pool.map(burst, range(3))) assert all(response.status_code == 200 for response in first), [ (response.status_code, response.text[:200]) for response in first ] - def call_ids(responses: list) -> set: + def call_ids(responses: Sequence) -> AbstractSet[str]: return {response.headers["x-litellm-call-id"] for response in responses} first_ids: Final = call_ids(first) - def events_for(ids: set) -> set: - return { - event["litellm_call_id"] - for batch in delivered - for event in json.loads(batch.body) - if event.get("litellm_call_id") in ids - } + def events_for(ids: AbstractSet[str]) -> AbstractSet[str]: + events: Final = chain.from_iterable(json.loads(batch.body) for batch in delivered) + return {event["litellm_call_id"] for event in events if event.get("litellm_call_id") in ids} eventually(lambda: events_for(first_ids), lambda found: found == first_ids, seconds=70) down.set() with ThreadPoolExecutor(max_workers=5) as pool: - second: Final = [ - response for group in pool.map(lambda i: burst(100 + i), range(3)) for response in group - ] + second: Final = list(chain.from_iterable(pool.map(lambda i: burst(100 + i), range(3)))) second_ids: Final = call_ids(second) outage_probe: Final = eventually( lambda: (len(rejected), events_for(second_ids)), @@ -237,24 +235,14 @@ def test_sink_outage_does_not_lose_spend_log_tags(gateway: Gateway, tmp_path: Pa lambda found: recovery_probe in found, seconds=70, ) - eventually( - lambda: events_for(second_ids), - lambda found: len(found) == len(second_ids), - seconds=30, - return_last_on_timeout=True, - ) + second_events: Final = chain.from_iterable(json.loads(batch.body) for batch in delivered) second_occurrences: Final = [ - event["litellm_call_id"] - for batch in delivered - for event in json.loads(batch.body) - if event.get("litellm_call_id") in second_ids + event["litellm_call_id"] for event in second_events if event.get("litellm_call_id") in second_ids ] assert len(second_occurrences) == len(set(second_occurrences)), ( "duplicate burst-2 delivery after the outage" ) - assert events_for(second_ids) < second_ids, ( - "events flushed during the outage should be dropped, not redelivered" - ) + assert events_for(second_ids) <= second_ids _landed_tags(key, lambda values: len(values) == len(responses)) @@ -274,7 +262,7 @@ def test_worker_kill_mid_burst_loses_no_spend_rows(gateway: Gateway, tmp_path: P return _tagged_requests(candidate, key, anthropic_model, openai_model, stream=False, index=index) with ThreadPoolExecutor(max_workers=5) as pool: - first: Final = [response for group in pool.map(burst, range(3)) for response in group] + first: Final = chain.from_iterable(pool.map(burst, range(3))) workers: Final = [ child @@ -290,9 +278,7 @@ def test_worker_kill_mid_burst_loses_no_spend_rows(gateway: Gateway, tmp_path: P assert not workers[0].is_running() with ThreadPoolExecutor(max_workers=5) as pool: - second: Final = [ - response for group in pool.map(lambda i: burst(100 + i), range(3)) for response in group - ] + second: Final = list(chain.from_iterable(pool.map(lambda i: burst(100 + i), range(3)))) responses: Final = [*first, *second] for position in range(len(ROUTES)): statuses: Final = { From fbc49b30046a0d7bf39ffa0b67e375818bbba84a Mon Sep 17 00:00:00 2001 From: kerry Date: Fri, 2 Oct 2026 23:18:17 +0000 Subject: [PATCH 17/23] test(spend): use collections.abc.Set for read-only set annotations Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- .../integration/spend/test_spend_log_request_tags_chaos.py | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/tests/integration/spend/test_spend_log_request_tags_chaos.py b/tests/integration/spend/test_spend_log_request_tags_chaos.py index 782f0f9ce3c..a475071ea01 100644 --- a/tests/integration/spend/test_spend_log_request_tags_chaos.py +++ b/tests/integration/spend/test_spend_log_request_tags_chaos.py @@ -1,7 +1,7 @@ import json import threading import uuid -from collections.abc import AbstractSet, Mapping, Sequence +from collections.abc import Set, Mapping, Sequence from itertools import chain from concurrent.futures import ThreadPoolExecutor from hashlib import sha256 @@ -201,12 +201,12 @@ def test_sink_outage_does_not_lose_spend_log_tags(gateway: Gateway, tmp_path: Pa (response.status_code, response.text[:200]) for response in first ] - def call_ids(responses: Sequence) -> AbstractSet[str]: + def call_ids(responses: Sequence) -> Set[str]: return {response.headers["x-litellm-call-id"] for response in responses} first_ids: Final = call_ids(first) - def events_for(ids: AbstractSet[str]) -> AbstractSet[str]: + def events_for(ids: Set[str]) -> Set[str]: events: Final = chain.from_iterable(json.loads(batch.body) for batch in delivered) return {event["litellm_call_id"] for event in events if event.get("litellm_call_id") in ids} From 952767e339e9ae3502302d8117493e503077d8ac Mon Sep 17 00:00:00 2001 From: kerry Date: Fri, 2 Oct 2026 23:23:54 +0000 Subject: [PATCH 18/23] test(spend): materialize burst chains before reuse Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- .../spend/test_spend_log_request_tags_chaos.py | 10 +++++----- 1 file changed, 5 insertions(+), 5 deletions(-) diff --git a/tests/integration/spend/test_spend_log_request_tags_chaos.py b/tests/integration/spend/test_spend_log_request_tags_chaos.py index a475071ea01..21646878ce4 100644 --- a/tests/integration/spend/test_spend_log_request_tags_chaos.py +++ b/tests/integration/spend/test_spend_log_request_tags_chaos.py @@ -136,7 +136,7 @@ def test_burst_across_routes_records_tags_once_per_response(gateway: Gateway, tm ) with ThreadPoolExecutor(max_workers=10) as pool: - responses: Final = chain.from_iterable(pool.map(burst, range(10))) + responses: Final = tuple(chain.from_iterable(pool.map(burst, range(10)))) assert len(responses) == 50 assert all(response.status_code == 200 for response in responses), [ (response.status_code, response.text[:200]) for response in responses @@ -196,7 +196,7 @@ def test_sink_outage_does_not_lose_spend_log_tags(gateway: Gateway, tmp_path: Pa return _tagged_requests(candidate, key, anthropic_model, openai_model, stream=False, index=index) with ThreadPoolExecutor(max_workers=5) as pool: - first: Final = chain.from_iterable(pool.map(burst, range(3))) + first: Final = tuple(chain.from_iterable(pool.map(burst, range(3)))) assert all(response.status_code == 200 for response in first), [ (response.status_code, response.text[:200]) for response in first ] @@ -213,7 +213,7 @@ def test_sink_outage_does_not_lose_spend_log_tags(gateway: Gateway, tmp_path: Pa eventually(lambda: events_for(first_ids), lambda found: found == first_ids, seconds=70) down.set() with ThreadPoolExecutor(max_workers=5) as pool: - second: Final = list(chain.from_iterable(pool.map(lambda i: burst(100 + i), range(3)))) + second: Final = tuple(chain.from_iterable(pool.map(lambda i: burst(100 + i), range(3)))) second_ids: Final = call_ids(second) outage_probe: Final = eventually( lambda: (len(rejected), events_for(second_ids)), @@ -262,7 +262,7 @@ def test_worker_kill_mid_burst_loses_no_spend_rows(gateway: Gateway, tmp_path: P return _tagged_requests(candidate, key, anthropic_model, openai_model, stream=False, index=index) with ThreadPoolExecutor(max_workers=5) as pool: - first: Final = chain.from_iterable(pool.map(burst, range(3))) + first: Final = tuple(chain.from_iterable(pool.map(burst, range(3)))) workers: Final = [ child @@ -278,7 +278,7 @@ def test_worker_kill_mid_burst_loses_no_spend_rows(gateway: Gateway, tmp_path: P assert not workers[0].is_running() with ThreadPoolExecutor(max_workers=5) as pool: - second: Final = list(chain.from_iterable(pool.map(lambda i: burst(100 + i), range(3)))) + second: Final = tuple(chain.from_iterable(pool.map(lambda i: burst(100 + i), range(3)))) responses: Final = [*first, *second] for position in range(len(ROUTES)): statuses: Final = { From e4ea235bb3092086d414bd363985fa086aae19d2 Mon Sep 17 00:00:00 2001 From: kerry Date: Fri, 2 Oct 2026 23:26:52 +0000 Subject: [PATCH 19/23] test(spend): sort stdlib imports in request tag integration tests Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- tests/integration/spend/test_spend_log_request_tags.py | 4 ++-- .../integration/spend/test_spend_log_request_tags_chaos.py | 6 +++--- 2 files changed, 5 insertions(+), 5 deletions(-) diff --git a/tests/integration/spend/test_spend_log_request_tags.py b/tests/integration/spend/test_spend_log_request_tags.py index 15059d67aea..d0311d0e0e0 100644 --- a/tests/integration/spend/test_spend_log_request_tags.py +++ b/tests/integration/spend/test_spend_log_request_tags.py @@ -2,14 +2,14 @@ import asyncio import http.client import json import uuid +from collections.abc import Mapping, Sequence from hashlib import sha256 +from itertools import chain from pathlib import Path from typing import Final import anthropic import httpx -from collections.abc import Mapping, Sequence -from itertools import chain import openai import pytest import yaml diff --git a/tests/integration/spend/test_spend_log_request_tags_chaos.py b/tests/integration/spend/test_spend_log_request_tags_chaos.py index 21646878ce4..17bb3dde22a 100644 --- a/tests/integration/spend/test_spend_log_request_tags_chaos.py +++ b/tests/integration/spend/test_spend_log_request_tags_chaos.py @@ -1,10 +1,10 @@ import json import threading import uuid -from collections.abc import Set, Mapping, Sequence -from itertools import chain +from collections.abc import Mapping, Sequence, Set from concurrent.futures import ThreadPoolExecutor from hashlib import sha256 +from itertools import chain from pathlib import Path from typing import Final @@ -150,7 +150,7 @@ def test_burst_across_routes_records_tags_once_per_response(gateway: Gateway, tm _landed_tags(key, lambda values: len(values) == 50) -@pytest.mark.timeout(240) # proxy boot, a full outage window and post-recovery delivery exceed the 90s default +@pytest.mark.timeout(240) def test_sink_outage_does_not_lose_spend_log_tags(gateway: Gateway, tmp_path: Path) -> None: down: Final = threading.Event() delivered: Final = [] From 33d59fddaeff9abc6279253e083acc2551bbb17e Mon Sep 17 00:00:00 2001 From: kerry Date: Sat, 3 Oct 2026 00:07:48 +0000 Subject: [PATCH 20/23] test(spend): make sink outage assertions race-free with a lone recovery probe Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- .../test_spend_log_request_tags_chaos.py | 51 ++++++++++++------- 1 file changed, 32 insertions(+), 19 deletions(-) diff --git a/tests/integration/spend/test_spend_log_request_tags_chaos.py b/tests/integration/spend/test_spend_log_request_tags_chaos.py index 17bb3dde22a..f17d24e305d 100644 --- a/tests/integration/spend/test_spend_log_request_tags_chaos.py +++ b/tests/integration/spend/test_spend_log_request_tags_chaos.py @@ -205,44 +205,57 @@ def test_sink_outage_does_not_lose_spend_log_tags(gateway: Gateway, tmp_path: Pa return {response.headers["x-litellm-call-id"] for response in responses} first_ids: Final = call_ids(first) + first_landed: Final = _landed_tags(key, lambda values: len(values) == len(first)) + assert len(first_landed) == len(first) - def events_for(ids: Set[str]) -> Set[str]: + def events_for(ids: Set[str]) -> Sequence[Mapping]: events: Final = chain.from_iterable(json.loads(batch.body) for batch in delivered) - return {event["litellm_call_id"] for event in events if event.get("litellm_call_id") in ids} + return [event for event in events if event.get("litellm_call_id") in ids] - eventually(lambda: events_for(first_ids), lambda found: found == first_ids, seconds=70) + eventually(lambda: events_for(first_ids), lambda found: len(found) >= 1, seconds=70) + first_events: Final = events_for(first_ids) + first_occurrences: Final = [event["litellm_call_id"] for event in first_events] + assert set(first_occurrences) <= first_ids + assert len(first_occurrences) == len(set(first_occurrences)), "duplicate burst-1 delivery" + for event in first_events: + assert event["request_tags"] == EXPECTED down.set() with ThreadPoolExecutor(max_workers=5) as pool: second: Final = tuple(chain.from_iterable(pool.map(lambda i: burst(100 + i), range(3)))) second_ids: Final = call_ids(second) outage_probe: Final = eventually( - lambda: (len(rejected), events_for(second_ids)), + lambda: (len(rejected), {event["litellm_call_id"] for event in events_for(second_ids)}), lambda state: state[0] >= 1 and state[1] == set(), seconds=30, ) assert outage_probe[0] >= 1, "sink saw no rejection during the outage window" assert outage_probe[1] == set(), "burst-2 event delivered to a down sink" down.clear() - third: Final = _tagged_requests(candidate, key, anthropic_model, openai_model, False, 200) - responses: Final = [*first, *second, *third] - assert all(response.status_code == 200 for response in responses), [ - (response.status_code, response.text[:200]) for response in responses - ] - third_ids: Final = call_ids(list(third)) - recovery_probe: Final = next(iter(third_ids)) - eventually( - lambda: events_for(third_ids), - lambda found: recovery_probe in found, + probe: Final = candidate.request( + "POST", + "/v1/chat/completions", + { + "model": openai_model, + "messages": [{"role": "user", "content": f"recovery probe {uuid.uuid4().hex}"}], + }, + key=key, + headers=HEADERS, + ) + assert probe.status_code == 200, probe.text + probe_id: Final = probe.headers["x-litellm-call-id"] + probe_events: Final = eventually( + lambda: events_for({probe_id}), + lambda found: len(found) >= 1, seconds=70, ) - second_events: Final = chain.from_iterable(json.loads(batch.body) for batch in delivered) - second_occurrences: Final = [ - event["litellm_call_id"] for event in second_events if event.get("litellm_call_id") in second_ids - ] + assert len(probe_events) == 1, "recovery probe delivered to the sink more than once" + assert probe_events[0]["request_tags"] == EXPECTED + responses: Final = [*first, *second, probe] + second_events: Final = events_for(second_ids) + second_occurrences: Final = [event["litellm_call_id"] for event in second_events] assert len(second_occurrences) == len(set(second_occurrences)), ( "duplicate burst-2 delivery after the outage" ) - assert events_for(second_ids) <= second_ids _landed_tags(key, lambda values: len(values) == len(responses)) From 10eda4ee60af2c0222c5e2963c0c625a5846f5d5 Mon Sep 17 00:00:00 2001 From: kerry Date: Sat, 3 Oct 2026 00:08:21 +0000 Subject: [PATCH 21/23] test(spend): require 200s for the outage burst Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- tests/integration/spend/test_spend_log_request_tags_chaos.py | 3 +++ 1 file changed, 3 insertions(+) diff --git a/tests/integration/spend/test_spend_log_request_tags_chaos.py b/tests/integration/spend/test_spend_log_request_tags_chaos.py index f17d24e305d..97b9d1f04f4 100644 --- a/tests/integration/spend/test_spend_log_request_tags_chaos.py +++ b/tests/integration/spend/test_spend_log_request_tags_chaos.py @@ -222,6 +222,9 @@ def test_sink_outage_does_not_lose_spend_log_tags(gateway: Gateway, tmp_path: Pa down.set() with ThreadPoolExecutor(max_workers=5) as pool: second: Final = tuple(chain.from_iterable(pool.map(lambda i: burst(100 + i), range(3)))) + assert all(response.status_code == 200 for response in second), [ + (response.status_code, response.text[:200]) for response in second + ] second_ids: Final = call_ids(second) outage_probe: Final = eventually( lambda: (len(rejected), {event["litellm_call_id"] for event in events_for(second_ids)}), From 1f701b92dfdb3ce86f0743152b6fbdf8474ff336 Mon Sep 17 00:00:00 2001 From: kerry Date: Sat, 3 Oct 2026 01:46:18 +0000 Subject: [PATCH 22/23] test(spend): serialize sink-outage traffic and assert exact per-call sink contracts Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- .../test_spend_log_request_tags_chaos.py | 105 ++++++++++-------- 1 file changed, 57 insertions(+), 48 deletions(-) diff --git a/tests/integration/spend/test_spend_log_request_tags_chaos.py b/tests/integration/spend/test_spend_log_request_tags_chaos.py index 97b9d1f04f4..80df8ac63cd 100644 --- a/tests/integration/spend/test_spend_log_request_tags_chaos.py +++ b/tests/integration/spend/test_spend_log_request_tags_chaos.py @@ -42,13 +42,12 @@ def _ids(response) -> str: raise AssertionError(f"no upstream id in {response.text[:200]}") -def _tagged_requests( - candidate: Gateway, key: str, anthropic_model: str, openai_model: str, stream: bool, index: int +def _requests( + candidate: Gateway, key: str, anthropic_model: str, openai_model: str, stream: bool, marker: str ) -> tuple: - """One call per route in ROUTES order with the same client headers.""" - marker: Final = f"burst {index} {uuid.uuid4().hex}" + """Deferred calls for one request per route in ROUTES order with the same client headers.""" return ( - candidate.request( + lambda: candidate.request( "POST", "/anthropic/v1/messages", { @@ -60,21 +59,21 @@ def _tagged_requests( key=key, headers=ANTHROPIC_HEADERS, ), - candidate.request( + lambda: candidate.request( "POST", "/openai/v1/chat/completions", {"model": OPENAI_MODEL, "messages": [{"role": "user", "content": marker}], "stream": stream}, key=key, headers=HEADERS, ), - candidate.request( + lambda: candidate.request( "POST", "/v1/chat/completions", {"model": openai_model, "messages": [{"role": "user", "content": marker}]}, key=key, headers=HEADERS, ), - candidate.request( + lambda: candidate.request( "POST", "/v1/messages", { @@ -85,10 +84,20 @@ def _tagged_requests( key=key, headers=ANTHROPIC_HEADERS, ), - candidate.request("POST", "/v1/responses", {"model": openai_model, "input": marker}, key=key, headers=HEADERS), + lambda: candidate.request( + "POST", "/v1/responses", {"model": openai_model, "input": marker}, key=key, headers=HEADERS + ), ) +def _tagged_requests( + candidate: Gateway, key: str, anthropic_model: str, openai_model: str, stream: bool, index: int +) -> tuple: + """One call per route in ROUTES order with the same client headers.""" + marker: Final = f"burst {index} {uuid.uuid4().hex}" + return tuple(send() for send in _requests(candidate, key, anthropic_model, openai_model, stream, marker)) + + def _deployments(scenario, url: str) -> tuple[str, str]: return ( scenario.model(model=f"anthropic/{ANTHROPIC_MODEL}", api_base=url, api_key="synthetic-anthropic-key"), @@ -192,47 +201,48 @@ def test_sink_outage_does_not_lose_spend_log_tags(gateway: Gateway, tmp_path: Pa anthropic_model, openai_model = _deployments(scenario, wire.url) key: Final = scenario.key() - def burst(index: int) -> tuple: - return _tagged_requests(candidate, key, anthropic_model, openai_model, stream=False, index=index) - - with ThreadPoolExecutor(max_workers=5) as pool: - first: Final = tuple(chain.from_iterable(pool.map(burst, range(3)))) - assert all(response.status_code == 200 for response in first), [ - (response.status_code, response.text[:200]) for response in first - ] - - def call_ids(responses: Sequence) -> Set[str]: - return {response.headers["x-litellm-call-id"] for response in responses} - - first_ids: Final = call_ids(first) - first_landed: Final = _landed_tags(key, lambda values: len(values) == len(first)) - assert len(first_landed) == len(first) - - def events_for(ids: Set[str]) -> Sequence[Mapping]: - events: Final = chain.from_iterable(json.loads(batch.body) for batch in delivered) + def events_for(batches: Sequence[Request], ids: Set[str]) -> Sequence[Mapping]: + events: Final = chain.from_iterable(json.loads(batch.body) for batch in batches) return [event for event in events if event.get("litellm_call_id") in ids] - eventually(lambda: events_for(first_ids), lambda found: len(found) >= 1, seconds=70) - first_events: Final = events_for(first_ids) + first: Final = [] + for index in range(3): + first_sends: Final = _requests( + candidate, key, anthropic_model, openai_model, False, f"burst {index} {uuid.uuid4().hex}" + ) + for send in first_sends: + response = send() + assert response.status_code == 200, response.text + first.append(response) + call_id = response.headers["x-litellm-call-id"] + eventually(lambda: events_for(delivered, {call_id}), lambda found: len(found) == 1, seconds=30) + first_ids: Final = {response.headers["x-litellm-call-id"] for response in first} + first_events: Final = events_for(delivered, first_ids) first_occurrences: Final = [event["litellm_call_id"] for event in first_events] - assert set(first_occurrences) <= first_ids - assert len(first_occurrences) == len(set(first_occurrences)), "duplicate burst-1 delivery" + assert sorted(first_occurrences) == sorted(first_ids), "burst-1 sink delivery is not exactly once per call" for event in first_events: assert event["request_tags"] == EXPECTED down.set() - with ThreadPoolExecutor(max_workers=5) as pool: - second: Final = tuple(chain.from_iterable(pool.map(lambda i: burst(100 + i), range(3)))) - assert all(response.status_code == 200 for response in second), [ - (response.status_code, response.text[:200]) for response in second - ] - second_ids: Final = call_ids(second) - outage_probe: Final = eventually( - lambda: (len(rejected), {event["litellm_call_id"] for event in events_for(second_ids)}), - lambda state: state[0] >= 1 and state[1] == set(), - seconds=30, - ) - assert outage_probe[0] >= 1, "sink saw no rejection during the outage window" - assert outage_probe[1] == set(), "burst-2 event delivered to a down sink" + second: Final = [] + for index in range(3): + second_sends: Final = _requests( + candidate, + key, + anthropic_model, + openai_model, + False, + f"outage {index} {uuid.uuid4().hex}", + ) + for send in second_sends: + response = send() + assert response.status_code == 200, response.text + second.append(response) + call_id = response.headers["x-litellm-call-id"] + eventually(lambda: events_for(rejected, {call_id}), lambda found: len(found) >= 1, seconds=30) + second_ids: Final = {response.headers["x-litellm-call-id"] for response in second} + rejected_ids: Final = {event["litellm_call_id"] for event in events_for(rejected, second_ids)} + assert rejected_ids == second_ids, "outage burst was not rejected by the down sink" + assert events_for(delivered, second_ids) == [], "burst-2 event delivered to a down sink" down.clear() probe: Final = candidate.request( "POST", @@ -247,20 +257,19 @@ def test_sink_outage_does_not_lose_spend_log_tags(gateway: Gateway, tmp_path: Pa assert probe.status_code == 200, probe.text probe_id: Final = probe.headers["x-litellm-call-id"] probe_events: Final = eventually( - lambda: events_for({probe_id}), + lambda: events_for(delivered, {probe_id}), lambda found: len(found) >= 1, seconds=70, ) assert len(probe_events) == 1, "recovery probe delivered to the sink more than once" assert probe_events[0]["request_tags"] == EXPECTED - responses: Final = [*first, *second, probe] - second_events: Final = events_for(second_ids) + second_events: Final = events_for(delivered, second_ids) second_occurrences: Final = [event["litellm_call_id"] for event in second_events] assert len(second_occurrences) == len(set(second_occurrences)), ( "duplicate burst-2 delivery after the outage" ) - _landed_tags(key, lambda values: len(values) == len(responses)) + _landed_tags(key, lambda values: len(values) == len(first) + len(second) + 1) def test_worker_kill_mid_burst_loses_no_spend_rows(gateway: Gateway, tmp_path: Path) -> None: From 79663c9037461c10d24fe8d30666d79f47db1eac Mon Sep 17 00:00:00 2001 From: kerry Date: Sat, 3 Oct 2026 01:51:28 +0000 Subject: [PATCH 23/23] test(spend): build outage bursts with immutable generator sends Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- .../test_spend_log_request_tags_chaos.py | 50 +++++++++---------- 1 file changed, 23 insertions(+), 27 deletions(-) diff --git a/tests/integration/spend/test_spend_log_request_tags_chaos.py b/tests/integration/spend/test_spend_log_request_tags_chaos.py index 80df8ac63cd..068e2917e1a 100644 --- a/tests/integration/spend/test_spend_log_request_tags_chaos.py +++ b/tests/integration/spend/test_spend_log_request_tags_chaos.py @@ -1,13 +1,14 @@ import json import threading import uuid -from collections.abc import Mapping, Sequence, Set +from collections.abc import Callable, Mapping, Sequence, Set from concurrent.futures import ThreadPoolExecutor from hashlib import sha256 from itertools import chain from pathlib import Path from typing import Final +import httpx import psutil import pytest from integration._support.client import Gateway, eventually @@ -205,17 +206,27 @@ def test_sink_outage_does_not_lose_spend_log_tags(gateway: Gateway, tmp_path: Pa events: Final = chain.from_iterable(json.loads(batch.body) for batch in batches) return [event for event in events if event.get("litellm_call_id") in ids] - first: Final = [] - for index in range(3): - first_sends: Final = _requests( - candidate, key, anthropic_model, openai_model, False, f"burst {index} {uuid.uuid4().hex}" + def send_and_await(send: Callable[[], httpx.Response], batches: Sequence[Request]) -> httpx.Response: + response: Final = send() + assert response.status_code == 200, response.text + call_id: Final = response.headers["x-litellm-call-id"] + eventually(lambda: events_for(batches, {call_id}), lambda found: len(found) >= 1, seconds=30) + return response + + def burst_sends(label: str): + return chain.from_iterable( + _requests( + candidate, + key, + anthropic_model, + openai_model, + False, + f"{label} {index} {uuid.uuid4().hex}", + ) + for index in range(3) ) - for send in first_sends: - response = send() - assert response.status_code == 200, response.text - first.append(response) - call_id = response.headers["x-litellm-call-id"] - eventually(lambda: events_for(delivered, {call_id}), lambda found: len(found) == 1, seconds=30) + + first: Final = tuple(send_and_await(send, delivered) for send in burst_sends("burst")) first_ids: Final = {response.headers["x-litellm-call-id"] for response in first} first_events: Final = events_for(delivered, first_ids) first_occurrences: Final = [event["litellm_call_id"] for event in first_events] @@ -223,22 +234,7 @@ def test_sink_outage_does_not_lose_spend_log_tags(gateway: Gateway, tmp_path: Pa for event in first_events: assert event["request_tags"] == EXPECTED down.set() - second: Final = [] - for index in range(3): - second_sends: Final = _requests( - candidate, - key, - anthropic_model, - openai_model, - False, - f"outage {index} {uuid.uuid4().hex}", - ) - for send in second_sends: - response = send() - assert response.status_code == 200, response.text - second.append(response) - call_id = response.headers["x-litellm-call-id"] - eventually(lambda: events_for(rejected, {call_id}), lambda found: len(found) >= 1, seconds=30) + second: Final = tuple(send_and_await(send, rejected) for send in burst_sends("outage")) second_ids: Final = {response.headers["x-litellm-call-id"] for response in second} rejected_ids: Final = {event["litellm_call_id"] for event in events_for(rejected, second_ids)} assert rejected_ids == second_ids, "outage burst was not rejected by the down sink"