From 5aa40595a2b9628c17d05e3f5179f78534837f37 Mon Sep 17 00:00:00 2001 From: mubashir1osmani Date: Fri, 24 Jul 2026 12:39:51 -0700 Subject: [PATCH] test(e2e): cover vendor strategy gaps for chat contract, image edits, auth, team activity Resolves the first slice of LIT-4778 (vendor API testing strategy): image edits happy path, chat multi-turn + validation + sanitization, LLM-route auth header matrix, and /team/daily/activity structure --- .../test_chat_auth_security_e2e.py | 106 ++++++ .../coverage_registry/llm_conversational.yaml | 3 + .../llm_nonconversational.yaml | 1 + tests/e2e/coverage_registry/mgmt.yaml | 3 + tests/e2e/coverage_registry/other.yaml | 5 + tests/e2e/coverage_registry/schema.py | 4 + tests/e2e/e2e_http.py | 11 +- tests/e2e/llm_translation/endpoints_client.py | 20 + ...st_chat_completions_vendor_contract_e2e.py | 355 ++++++++++++++++++ .../llm_translation/test_image_edits_e2e.py | 54 +++ tests/e2e/models.py | 4 + .../spend_tracking/spend_e2e_client.py | 5 +- .../test_team_daily_activity_e2e.py | 82 ++++ tests/e2e/transport.py | 5 + 14 files changed, 651 insertions(+), 7 deletions(-) create mode 100644 tests/e2e/access_control/test_chat_auth_security_e2e.py create mode 100644 tests/e2e/llm_translation/test_chat_completions_vendor_contract_e2e.py create mode 100644 tests/e2e/llm_translation/test_image_edits_e2e.py create mode 100644 tests/e2e/quota_management/spend_tracking/test_team_daily_activity_e2e.py diff --git a/tests/e2e/access_control/test_chat_auth_security_e2e.py b/tests/e2e/access_control/test_chat_auth_security_e2e.py new file mode 100644 index 00000000000..89ef85385a2 --- /dev/null +++ b/tests/e2e/access_control/test_chat_auth_security_e2e.py @@ -0,0 +1,106 @@ +"""Vendor §11.1 auth security on LLM routes (LIT-4778). + +Virtual-key chat must reject missing and malformed Authorization headers before +any provider call. These cases sit next to the existing valid/invalid key check +and pin the full bearer-token failure matrix from the vendor brief. +""" + +from __future__ import annotations + +import pytest +from pydantic import BaseModel + +from e2e_config import unique_marker +from e2e_http import AuthHeaders, NoBody, StreamingResponse +from lifecycle import ResourceManager +from models import ChatBody, ChatMessage, LiteLLMParamsBody +from proxy_client import ProxyClient + +pytestmark = pytest.mark.e2e + +OPENAI_BACKEND = "openai/gpt-4o-mini" +CHAT_PATH = "/chat/completions" + + +class RawAuthorizationHeaders(BaseModel): + Authorization: str + + +def _register_model(proxy: ProxyClient, resources: ResourceManager) -> str: + model = f"e2e-auth-sec-{unique_marker()}" + model_id = proxy.create_model( + model, + LiteLLMParamsBody(model=OPENAI_BACKEND, api_key="os.environ/OPENAI_API_KEY"), + ) + resources.defer(lambda: proxy.delete_model(model_id)) + return model + + +def _chat_with_headers( + proxy: ProxyClient, headers: BaseModel, model: str +) -> StreamingResponse: + return proxy.transport.send( + CHAT_PATH, + headers=headers, + json=ChatBody( + model=model, + messages=[ChatMessage(role="user", content="should not run")], + max_tokens=8, + ), + ) + + +def _assert_auth_denied(result: StreamingResponse, context: str) -> None: + assert result.status_code in (401, 403), ( + f"{context}: expected 401/403, got {result.status_code}: {result.body[:300]}" + ) + + +class TestChatAuthSecurity: + @pytest.mark.covers("other.auth.llm_chat.missing_header_denied") + def test_missing_authorization_header_is_denied( + self, proxy: ProxyClient, resources: ResourceManager + ) -> None: + model = _register_model(proxy, resources) + result = _chat_with_headers(proxy, NoBody(), model) + _assert_auth_denied(result, "missing Authorization") + + @pytest.mark.covers("other.auth.llm_chat.invalid_bearer_denied") + def test_bearer_invalid_token_is_denied( + self, proxy: ProxyClient, resources: ResourceManager + ) -> None: + model = _register_model(proxy, resources) + result = _chat_with_headers( + proxy, AuthHeaders(authorization="Bearer invalid_token"), model + ) + _assert_auth_denied(result, "Bearer invalid_token") + + @pytest.mark.covers("other.auth.llm_chat.no_bearer_prefix_denied") + def test_token_without_bearer_prefix_is_denied( + self, proxy: ProxyClient, resources: ResourceManager + ) -> None: + model = _register_model(proxy, resources) + result = _chat_with_headers( + proxy, RawAuthorizationHeaders(Authorization="invalid_token"), model + ) + _assert_auth_denied(result, "token without Bearer prefix") + + @pytest.mark.covers("other.auth.llm_chat.empty_bearer_denied") + def test_empty_bearer_token_is_denied( + self, proxy: ProxyClient, resources: ResourceManager + ) -> None: + model = _register_model(proxy, resources) + result = _chat_with_headers( + proxy, AuthHeaders(authorization="Bearer "), model + ) + _assert_auth_denied(result, "empty Bearer token") + + @pytest.mark.covers("other.auth.llm_chat.not_bearer_scheme_denied") + def test_not_bearer_scheme_is_denied( + self, proxy: ProxyClient, resources: ResourceManager + ) -> None: + model = _register_model(proxy, resources) + result = _chat_with_headers( + proxy, RawAuthorizationHeaders(Authorization="NotBearer validtoken123"), model + ) + _assert_auth_denied(result, "NotBearer scheme") diff --git a/tests/e2e/coverage_registry/llm_conversational.yaml b/tests/e2e/coverage_registry/llm_conversational.yaml index 26280d35da0..576e3375a49 100644 --- a/tests/e2e/coverage_registry/llm_conversational.yaml +++ b/tests/e2e/coverage_registry/llm_conversational.yaml @@ -1,5 +1,8 @@ # LLM conversational endpoints (chat_completions, messages, responses). Grounded in proxy handlers + model_prices json. - {id: llm.chat_completions.openai.basic.nonstream.works, module: llm, tier: P0, subject_endpoint: chat_completions, route: openai, capability: basic, streaming: nonstream, assertions: [works], source: "proxy_server.py:8455", rationale: "Core endpoint/route/capability"} +- {id: llm.chat_completions.openai.multi_turn.nonstream.works, module: llm, tier: P0, subject_endpoint: chat_completions, route: openai, capability: multi_turn, streaming: nonstream, assertions: [works], source: "vendor testing strategy §16.2 / LIT-4778", rationale: "Multi-turn history is forwarded so turn 2 can use turn 1 answer"} +- {id: llm.chat_completions.openai.input_validation.nonstream.works, module: llm, tier: P0, subject_endpoint: chat_completions, route: openai, capability: input_validation, streaming: nonstream, assertions: [works], source: "vendor testing strategy §9.2 / LIT-4778", rationale: "Missing/invalid chat fields return client errors, not silent success"} +- {id: llm.chat_completions.openai.input_sanitization.nonstream.works, module: llm, tier: P0, subject_endpoint: chat_completions, route: openai, capability: input_sanitization, streaming: nonstream, assertions: [works], source: "vendor testing strategy §11.3 / LIT-4778", rationale: "SQL injection and XSS payloads must not 5xx the proxy"} - {id: llm.chat_completions.openai.basic.stream.works, module: llm, tier: P0, subject_endpoint: chat_completions, route: openai, capability: basic, streaming: stream, assertions: [works], source: "proxy_server.py:8455", rationale: "Core streaming"} - {id: llm.chat_completions.openai.basic.nonstream.cost_logged, module: llm, tier: P0, subject_endpoint: chat_completions, route: openai, capability: basic, streaming: nonstream, assertions: [works, cost_logged], source: "proxy_server.py:8455", rationale: "Cost logging regression catch"} - {id: llm.chat_completions.openai.tool_use.nonstream.works, module: llm, tier: P0, subject_endpoint: chat_completions, route: openai, capability: tool_use, streaming: nonstream, assertions: [works], source: "model_prices json", rationale: "OpenAI function_calling; high usage"} diff --git a/tests/e2e/coverage_registry/llm_nonconversational.yaml b/tests/e2e/coverage_registry/llm_nonconversational.yaml index 63e6fde14a3..48b27565423 100644 --- a/tests/e2e/coverage_registry/llm_nonconversational.yaml +++ b/tests/e2e/coverage_registry/llm_nonconversational.yaml @@ -36,6 +36,7 @@ - {id: llm.rerank.bedrock.basic.nonstream.works, module: llm, tier: P1, subject_endpoint: rerank, route: bedrock_converse, capability: basic, streaming: nonstream, assertions: [works], source: "llms/bedrock/rerank/handler.py", rationale: "Bedrock rerank"} - {id: llm.rerank.together_ai.basic.nonstream.works, module: llm, tier: P1, subject_endpoint: rerank, route: together_ai, capability: basic, streaming: nonstream, assertions: [works], source: "llms/together_ai/rerank/handler.py", rationale: "Together rerank"} - {id: llm.images_generations.openai.basic.nonstream.works, module: llm, tier: P1, subject_endpoint: images_generations, route: openai, capability: basic, streaming: nonstream, assertions: [works], source: "test_image_generation_e2e.py:22", rationale: "OpenAI image gen, b64/url"} +- {id: llm.images_edits.openai.basic.nonstream.works, module: llm, tier: P1, subject_endpoint: images_edits, route: openai, capability: basic, streaming: nonstream, assertions: [works], source: "test_image_edits_e2e.py", rationale: "OpenAI /v1/images/edits multipart image+prompt (vendor strategy / LIT-4778)"} - {id: llm.images_generations.azure_openai.basic.nonstream.works, module: llm, tier: P1, subject_endpoint: images_generations, route: azure_openai, capability: basic, streaming: nonstream, assertions: [works], source: "llms/azure/azure.py", rationale: "Azure DALL-E"} - {id: llm.images_generations.vertex.basic.nonstream.works, module: llm, tier: P1, subject_endpoint: images_generations, route: vertex, capability: basic, streaming: nonstream, assertions: [works], source: "vertex_ai/image_generation/image_generation_handler.py", rationale: "Vertex Imagen"} - {id: llm.images_generations.bedrock.basic.nonstream.works, module: llm, tier: P1, subject_endpoint: images_generations, route: bedrock_converse, capability: basic, streaming: nonstream, assertions: [works], source: "bedrock/image_generation/image_handler.py", rationale: "Bedrock Titan Image"} diff --git a/tests/e2e/coverage_registry/mgmt.yaml b/tests/e2e/coverage_registry/mgmt.yaml index da4652460f1..353e60db3c4 100644 --- a/tests/e2e/coverage_registry/mgmt.yaml +++ b/tests/e2e/coverage_registry/mgmt.yaml @@ -31,6 +31,9 @@ - {id: mgmt.team.delete.persists, module: mgmt, tier: P1, surface: api, assertions: [persists], source: "team_endpoints.py:1750", rationale: "Deletion prevents key access"} - {id: mgmt.team.block.persists, module: mgmt, tier: P1, surface: api, assertions: [persists], source: "team_endpoints.py", rationale: "Block suspends all members"} - {id: mgmt.team.info.happy_path, module: mgmt, tier: P1, surface: api, assertions: [happy_path], source: "team_endpoints.py:2244", rationale: "Metadata+members+budgets"} +- {id: mgmt.team.daily_activity.happy_path, module: mgmt, tier: P1, surface: api, assertions: [happy_path], source: "vendor testing strategy §9.20 / LIT-4778", rationale: "GET /team/daily/activity returns results+metadata for a valid date range"} +- {id: mgmt.team.daily_activity.missing_start_date_rejected, module: mgmt, tier: P1, surface: api, assertions: [missing_start_date_rejected], source: "vendor testing strategy §9.20 / LIT-4778", rationale: "Missing start_date on /team/daily/activity is 400"} +- {id: mgmt.team.daily_activity.missing_end_date_rejected, module: mgmt, tier: P1, surface: api, assertions: [missing_end_date_rejected], source: "vendor testing strategy §9.20 / LIT-4778", rationale: "Missing end_date on /team/daily/activity is 400"} - {id: mgmt.team.list.happy_path, module: mgmt, tier: P1, surface: api, assertions: [happy_path], source: "team_endpoints.py:3645", rationale: "Pagination/filtering"} - {id: mgmt.team.member_update.persists, module: mgmt, tier: P1, surface: api, assertions: [persists], source: "team_endpoints.py:2768", rationale: "Member budget/role updates persist"} - {id: mgmt.user.update.persists, module: mgmt, tier: P1, surface: api, assertions: [persists], source: "internal_user_endpoints.py:555", rationale: "Metadata/perm updates persist"} diff --git a/tests/e2e/coverage_registry/other.yaml b/tests/e2e/coverage_registry/other.yaml index ace4f8bcdc9..dfaffac32a0 100644 --- a/tests/e2e/coverage_registry/other.yaml +++ b/tests/e2e/coverage_registry/other.yaml @@ -2,6 +2,11 @@ # PROMOTION NOTE: the auth cluster (~14 cells) is a candidate to promote to its own module once stable. - {id: other.auth.master_key.valid_allows, module: other, tier: P0, area: auth, assertions: [valid_allows], source: "user_api_key_auth.py:1569-1588", rationale: "Master key authenticates; timing-safe compare"} - {id: other.auth.master_key.invalid_denied, module: other, tier: P0, area: auth, assertions: [invalid_denied], source: "user_api_key_auth.py:1580", rationale: "Invalid master key rejected"} +- {id: other.auth.llm_chat.missing_header_denied, module: other, tier: P0, area: auth, assertions: [missing_header_denied], source: "vendor testing strategy §11.1 / LIT-4778", rationale: "Chat with no Authorization header is 401/403"} +- {id: other.auth.llm_chat.invalid_bearer_denied, module: other, tier: P0, area: auth, assertions: [invalid_bearer_denied], source: "vendor testing strategy §11.1 / LIT-4778", rationale: "Bearer invalid_token on chat is 401/403"} +- {id: other.auth.llm_chat.no_bearer_prefix_denied, module: other, tier: P0, area: auth, assertions: [no_bearer_prefix_denied], source: "vendor testing strategy §11.1 / LIT-4778", rationale: "Token without Bearer scheme on chat is 401/403"} +- {id: other.auth.llm_chat.empty_bearer_denied, module: other, tier: P0, area: auth, assertions: [empty_bearer_denied], source: "vendor testing strategy §11.1 / LIT-4778", rationale: "Empty Bearer token on chat is 401/403"} +- {id: other.auth.llm_chat.not_bearer_scheme_denied, module: other, tier: P0, area: auth, assertions: [not_bearer_scheme_denied], source: "vendor testing strategy §11.1 / LIT-4778", rationale: "NotBearer scheme on chat is 401/403"} - {id: other.config.responses.metadata_redis_ttl_bounded, module: other, tier: P0, area: config, assertions: [ttl_bounded], source: "responses + redis cache", rationale: "Responses store+metadata must not leave TTL-unbounded Redis entries (LIT-1201)"} - {id: other.auth.jwt.valid_token_allows, module: other, tier: P0, area: auth, assertions: [valid_token_allows], source: "handle_jwt.py:77-150", rationale: "Valid JWT with correct issuer + claims grants access"} - {id: other.auth.jwt.expired_denied, module: other, tier: P0, area: auth, assertions: [expired_denied], source: "handle_jwt.py:125-135", rationale: "Expired JWT rejected even with valid signature"} diff --git a/tests/e2e/coverage_registry/schema.py b/tests/e2e/coverage_registry/schema.py index 1a6dc111e2b..4e8c3e3af92 100644 --- a/tests/e2e/coverage_registry/schema.py +++ b/tests/e2e/coverage_registry/schema.py @@ -34,6 +34,7 @@ LlmEndpoint = Literal[ "files", "rerank", "images_generations", + "images_edits", "audio_speech", "audio_transcriptions", "moderations", @@ -58,8 +59,11 @@ LlmCapability = Literal[ "assume_role", "basic", "count_tokens", + "input_sanitization", + "input_validation", "long_context_1m", "mid_conversation_system", + "multi_turn", "pdf_input", "prompt_cache_1h", "prompt_cache_5m", diff --git a/tests/e2e/e2e_http.py b/tests/e2e/e2e_http.py index 03d7b5d051a..d16e84bd754 100644 --- a/tests/e2e/e2e_http.py +++ b/tests/e2e/e2e_http.py @@ -480,14 +480,15 @@ def upload[R: BaseModel]( filename: str, content: bytes, file_content_type: str = "application/jsonl", + file_field: str = "file", params: BaseModel | None = None, response_type: type[R], timeout: float = 60.0, ) -> Result[R]: - """Multipart POST for file-bearing routes (/v1/files, /v1/audio/transcriptions). - Form fields come from `form`, the file bytes are sent as the `file` part with - `file_content_type`, and `params` carries any query routing (e.g. ?model=). - requests sets the multipart Content-Type itself.""" + """Multipart POST for file-bearing routes (/v1/files, /v1/audio/transcriptions, + /v1/images/edits). Form fields come from `form`, the file bytes are sent as the + `file_field` part with `file_content_type`, and `params` carries any query + routing (e.g. ?model=). requests sets the multipart Content-Type itself.""" dumped: dict[str, object] = form.model_dump(by_alias=True, exclude_none=True) data = {key: str(value) for key, value in dumped.items()} try: @@ -496,7 +497,7 @@ def upload[R: BaseModel]( headers=_headers(headers), params=_params(params), data=data, - files={"file": (filename, content, file_content_type)}, + files={file_field: (filename, content, file_content_type)}, timeout=timeout, ) except requests.RequestException as exc: diff --git a/tests/e2e/llm_translation/endpoints_client.py b/tests/e2e/llm_translation/endpoints_client.py index ace621d03b3..4b1d10d539f 100644 --- a/tests/e2e/llm_translation/endpoints_client.py +++ b/tests/e2e/llm_translation/endpoints_client.py @@ -110,6 +110,12 @@ class ImageRequest(BaseModel): size: str = "1024x1024" +class ImageEditForm(BaseModel): + model: str + prompt: str + n: int = 1 + + class TranscriptionForm(BaseModel): model: str response_format: str = "json" @@ -380,6 +386,20 @@ class EndpointsClient: "/v1/images/generations", key, ImageRequest(model=model, prompt=prompt) ) + def image_edit( + self, key: str, model: str, prompt: str, image: bytes, *, filename: str = "image.png" + ) -> Result[ImagesResult]: + return self.proxy.transport.upload( + "/v1/images/edits", + headers=self.proxy.transport.bearer(key), + form=ImageEditForm(model=model, prompt=prompt), + filename=filename, + content=image, + file_content_type="image/png", + file_field="image", + response_type=ImagesResult, + ) + def build_endpoints_client(proxy: ProxyClient) -> EndpointsClient: return EndpointsClient(proxy=proxy) diff --git a/tests/e2e/llm_translation/test_chat_completions_vendor_contract_e2e.py b/tests/e2e/llm_translation/test_chat_completions_vendor_contract_e2e.py new file mode 100644 index 00000000000..b22946317f2 --- /dev/null +++ b/tests/e2e/llm_translation/test_chat_completions_vendor_contract_e2e.py @@ -0,0 +1,355 @@ +"""Vendor API strategy coverage for /chat/completions (LIT-4778). + +Positive multi-turn history, input validation, boundary handling, contract shape, +and input sanitization against a live proxy and a real OpenAI-compatible model. +""" + +from __future__ import annotations + +import pytest +from pydantic import BaseModel + +from e2e_config import unique_marker +from e2e_http import AuthHeaders, StreamingResponse, require_successful_call, unwrap +from lifecycle import ResourceManager +from models import ChatBody, ChatMessage, ChatResponse, LiteLLMParamsBody +from proxy_client import ProxyClient + +pytestmark = pytest.mark.e2e + +OPENAI_BACKEND = "openai/gpt-4o-mini" +CHAT_PATH = "/chat/completions" + +SQL_INJECTION_PAYLOADS = ( + "'; DROP TABLE users; --", + "1' OR '1'='1", + "admin' --", +) +XSS_PAYLOADS = ( + "", + "", + "javascript:alert('XSS')", +) + + +class ChatMissingModelBody(BaseModel): + messages: list[ChatMessage] + + +class ChatMissingMessagesBody(BaseModel): + model: str + + +class ChatErrorBody(BaseModel): + message: str | None = None + type: str | None = None + code: str | int | None = None + + +class ChatErrorEnvelope(BaseModel): + error: ChatErrorBody | None = None + + +def _register_chat_model(proxy: ProxyClient, resources: ResourceManager) -> tuple[str, str]: + model = f"e2e-vendor-chat-{unique_marker()}" + model_id = proxy.create_model( + model, + LiteLLMParamsBody(model=OPENAI_BACKEND, api_key="os.environ/OPENAI_API_KEY"), + ) + resources.defer(lambda: proxy.delete_model(model_id)) + return model, resources.key() + + +def _chat_status( + proxy: ProxyClient, key: str, body: BaseModel, *, headers: AuthHeaders | None = None +) -> StreamingResponse: + return proxy.transport.send( + CHAT_PATH, + headers=headers if headers is not None else proxy.transport.bearer(key), + json=body, + ) + + +def _is_client_error(status: int) -> bool: + return 400 <= status < 500 + + +def _assert_not_server_error(result: StreamingResponse, context: str) -> None: + assert result.status_code not in (500, 502, 503), ( + f"{context}: proxy must not 5xx, got {result.status_code}: {result.body[:300]}" + ) + + +class TestChatCompletionsVendorContract: + @pytest.mark.covers("llm.chat_completions.openai.multi_turn.nonstream.works") + def test_multi_turn_history_is_honored( + self, proxy: ProxyClient, resources: ResourceManager + ) -> None: + model, key = _register_chat_model(proxy, resources) + turn1 = unwrap( + proxy.chat( + key, + ChatBody( + model=model, + messages=[ + ChatMessage(role="system", content="You are a helpful math tutor."), + ChatMessage(role="user", content="What is 25 + 17? Reply with only the number."), + ], + temperature=0.1, + max_completion_tokens=32, + ), + ) + ) + assert turn1.choices and turn1.choices[0].message is not None + assistant = turn1.choices[0].message.content or "" + assert "42" in assistant, f"turn1 must answer 42, got: {assistant!r}" + + turn2 = unwrap( + proxy.chat( + key, + ChatBody( + model=model, + messages=[ + ChatMessage(role="system", content="You are a helpful math tutor."), + ChatMessage(role="user", content="What is 25 + 17? Reply with only the number."), + ChatMessage(role="assistant", content=assistant), + ChatMessage( + role="user", + content="Now multiply that result by 2. Reply with only the number.", + ), + ], + temperature=0.1, + max_completion_tokens=32, + ), + ) + ) + assert turn2.choices and turn2.choices[0].message is not None + second = turn2.choices[0].message.content or "" + assert "84" in second, f"turn2 must answer 84 from history, got: {second!r}" + + @pytest.mark.covers("llm.chat_completions.openai.basic.nonstream.works") + def test_success_response_matches_chat_completion_contract( + self, proxy: ProxyClient, resources: ResourceManager + ) -> None: + model, key = _register_chat_model(proxy, resources) + result = _chat_status( + proxy, + key, + ChatBody( + model=model, + messages=[ + ChatMessage(role="user", content=f"Reply with a single word: confirmed. {unique_marker()}") + ], + max_completion_tokens=32, + temperature=0.2, + ), + ) + require_successful_call(result) + parsed = ChatResponse.model_validate_json(result.body) + assert parsed.id, f"chat completion must return id: {result.body[:300]}" + assert parsed.object in (None, "chat.completion"), ( + f"object must be chat.completion when present, got {parsed.object!r}" + ) + assert parsed.choices, f"choices must be non-empty: {result.body[:300]}" + message = parsed.choices[0].message + assert message is not None, f"choices[0].message required: {result.body[:300]}" + assert message.role in (None, "assistant"), f"unexpected role: {message.role!r}" + assert (message.content or "").strip(), f"content must be non-empty: {result.body[:300]}" + + @pytest.mark.covers("llm.chat_completions.openai.input_validation.nonstream.works") + def test_missing_model_returns_client_error( + self, proxy: ProxyClient, resources: ResourceManager + ) -> None: + _, key = _register_chat_model(proxy, resources) + result = _chat_status( + proxy, + key, + ChatMissingModelBody(messages=[ChatMessage(role="user", content="hi")]), + ) + assert _is_client_error(result.status_code), ( + f"missing model must be 4xx, got {result.status_code}: {result.body[:300]}" + ) + envelope = ChatErrorEnvelope.model_validate_json(result.body) + assert envelope.error is not None and envelope.error.message, ( + f"error body must carry error.message: {result.body[:300]}" + ) + + @pytest.mark.covers("llm.chat_completions.openai.input_validation.nonstream.works") + def test_missing_messages_returns_error( + self, proxy: ProxyClient, resources: ResourceManager + ) -> None: + model, key = _register_chat_model(proxy, resources) + result = _chat_status(proxy, key, ChatMissingMessagesBody(model=model)) + assert result.status_code in range(400, 600), ( + f"missing messages must not succeed, got {result.status_code}: {result.body[:300]}" + ) + assert result.status_code != 200 + + @pytest.mark.covers("llm.chat_completions.openai.input_validation.nonstream.works") + def test_empty_messages_returns_client_error( + self, proxy: ProxyClient, resources: ResourceManager + ) -> None: + model, key = _register_chat_model(proxy, resources) + result = _chat_status( + proxy, + key, + ChatBody(model=model, messages=[], max_completion_tokens=16), + ) + assert _is_client_error(result.status_code), ( + f"empty messages must be 4xx, got {result.status_code}: {result.body[:300]}" + ) + + @pytest.mark.covers("llm.chat_completions.openai.input_validation.nonstream.works") + def test_invalid_role_returns_client_error( + self, proxy: ProxyClient, resources: ResourceManager + ) -> None: + model, key = _register_chat_model(proxy, resources) + result = _chat_status( + proxy, + key, + ChatBody( + model=model, + messages=[ChatMessage(role="invalid_role", content="hi")], + max_completion_tokens=16, + ), + ) + assert _is_client_error(result.status_code), ( + f"invalid role must be 4xx, got {result.status_code}: {result.body[:300]}" + ) + + @pytest.mark.covers("llm.chat_completions.openai.input_validation.nonstream.works") + @pytest.mark.parametrize("temperature", [3.0, -0.1, 2.1, 100.0]) + def test_invalid_temperature_returns_client_error( + self, proxy: ProxyClient, resources: ResourceManager, temperature: float + ) -> None: + model, key = _register_chat_model(proxy, resources) + result = _chat_status( + proxy, + key, + ChatBody( + model=model, + messages=[ChatMessage(role="user", content="hi")], + temperature=temperature, + max_completion_tokens=16, + ), + ) + assert _is_client_error(result.status_code), ( + f"temperature={temperature} must be 4xx, got {result.status_code}: {result.body[:300]}" + ) + + @pytest.mark.covers("llm.chat_completions.openai.input_validation.nonstream.works") + @pytest.mark.parametrize("max_completion_tokens", [-1, 0, -100]) + def test_invalid_max_completion_tokens_returns_client_error( + self, proxy: ProxyClient, resources: ResourceManager, max_completion_tokens: int + ) -> None: + model, key = _register_chat_model(proxy, resources) + result = _chat_status( + proxy, + key, + ChatBody( + model=model, + messages=[ChatMessage(role="user", content="hi")], + max_completion_tokens=max_completion_tokens, + ), + ) + assert _is_client_error(result.status_code), ( + f"max_completion_tokens={max_completion_tokens} must be 4xx, " + f"got {result.status_code}: {result.body[:300]}" + ) + + @pytest.mark.covers("llm.chat_completions.openai.basic.nonstream.works") + @pytest.mark.parametrize("temperature", [0.0, 2.0]) + def test_temperature_boundaries_succeed( + self, proxy: ProxyClient, resources: ResourceManager, temperature: float + ) -> None: + model, key = _register_chat_model(proxy, resources) + result = _chat_status( + proxy, + key, + ChatBody( + model=model, + messages=[ + ChatMessage(role="user", content=f"Reply with ok. {unique_marker()}") + ], + temperature=temperature, + max_completion_tokens=16, + ), + ) + require_successful_call(result) + parsed = ChatResponse.model_validate_json(result.body) + assert parsed.choices, f"temperature={temperature} must return choices" + + @pytest.mark.covers("llm.chat_completions.openai.basic.nonstream.works") + def test_extremely_long_message_does_not_crash_proxy( + self, proxy: ProxyClient, resources: ResourceManager + ) -> None: + model, key = _register_chat_model(proxy, resources) + result = _chat_status( + proxy, + key, + ChatBody( + model=model, + messages=[ChatMessage(role="user", content="x" * 100_000)], + max_completion_tokens=16, + ), + ) + assert result.status_code in (200, 400, 413, 500), ( + f"long message acceptable statuses only, got {result.status_code}: {result.body[:300]}" + ) + + @pytest.mark.covers("llm.chat_completions.openai.input_sanitization.nonstream.works") + @pytest.mark.parametrize("payload", SQL_INJECTION_PAYLOADS) + def test_sql_injection_payloads_do_not_crash_proxy( + self, proxy: ProxyClient, resources: ResourceManager, payload: str + ) -> None: + model, key = _register_chat_model(proxy, resources) + result = _chat_status( + proxy, + key, + ChatBody( + model=model, + messages=[ChatMessage(role="user", content=payload)], + max_completion_tokens=32, + ), + ) + _assert_not_server_error(result, f"sql injection payload {payload!r}") + assert result.status_code in (200, 400, 401, 403, 422), ( + f"sql injection must be handled safely, got {result.status_code}: {result.body[:300]}" + ) + + @pytest.mark.covers("llm.chat_completions.openai.input_sanitization.nonstream.works") + @pytest.mark.parametrize("payload", XSS_PAYLOADS) + def test_xss_payloads_do_not_crash_or_echo_raw( + self, proxy: ProxyClient, resources: ResourceManager, payload: str + ) -> None: + model, key = _register_chat_model(proxy, resources) + result = _chat_status( + proxy, + key, + ChatBody( + model=model, + messages=[ + ChatMessage( + role="user", + content=f"Echo this exactly with no changes: {payload}", + ) + ], + max_completion_tokens=64, + temperature=0.0, + ), + ) + _assert_not_server_error(result, f"xss payload {payload!r}") + assert result.status_code in (200, 400, 401, 403, 422), ( + f"xss must be handled safely, got {result.status_code}: {result.body[:300]}" + ) + if result.status_code != 200: + return + try: + loaded = ChatResponse.model_validate_json(result.body) + except Exception: + pytest.fail(f"200 body must be JSON chat response: {result.body[:300]}") + text = loaded.model_dump_json() + if payload in text: + assert f"`{payload}`" in text or "```" in text, ( + f"200 response must not echo raw XSS unescaped: {result.body[:300]}" + ) diff --git a/tests/e2e/llm_translation/test_image_edits_e2e.py b/tests/e2e/llm_translation/test_image_edits_e2e.py new file mode 100644 index 00000000000..faad8703e74 --- /dev/null +++ b/tests/e2e/llm_translation/test_image_edits_e2e.py @@ -0,0 +1,54 @@ +"""Live e2e: POST /v1/images/edits returns an edited image. + +Registers an OpenAI image model, then sends a small PNG plus an edit prompt as a +multipart request to /v1/images/edits and asserts the response carries an image +(url or base64). /images/edits is a distinct native route from +/images/generations: it is multipart file upload with the image sent as the +`image` part, not a JSON body. The fixture image is a small generated 64x64 PNG, +so no external asset is needed. +""" + +from __future__ import annotations + +import base64 + +import pytest + +from e2e_config import unique_marker +from e2e_http import unwrap +from endpoints_client import EndpointsClient +from lifecycle import ResourceManager +from models import LiteLLMParamsBody + +pytestmark = pytest.mark.e2e + +_TEST_PNG = base64.b64decode( + "iVBORw0KGgoAAAANSUhEUgAAAEAAAABACAIAAAAlC+aJAAAAS0lEQVR42u3PMQ0AAAwDoPo3" + "3UrYvQQckD4XAQEBAQEBAQEBAQEBAQEBAQEBAQEBAQEBAQEBAQEBAQEBAQEBAQEBAQEBAQEB" + "AYHLAMpT0sIcNbcEAAAAAElFTkSuQmCC" +) + + +class TestImageEdit: + @pytest.mark.covers("llm.images_edits.openai.basic.nonstream.works") + def test_image_edit_returns_image( + self, endpoints_client: EndpointsClient, resources: ResourceManager + ) -> None: + model = f"e2e-image-edit-{unique_marker()}" + model_id = endpoints_client.create_model( + model, + LiteLLMParamsBody(model="openai/gpt-image-1", api_key="os.environ/OPENAI_API_KEY"), + ) + resources.defer(lambda: endpoints_client.delete_model(model_id)) + key = resources.key() + + edited = unwrap( + endpoints_client.image_edit( + key, model, "Add a small red circle in the center", _TEST_PNG + ) + ) + assert edited.data, f"/images/edits returned no data: {edited}" + first = edited.data[0] + assert first.b64_json or first.url, ( + f"edited image has neither b64_json nor url: {first}" + ) diff --git a/tests/e2e/models.py b/tests/e2e/models.py index d21920cf848..634a0cc9c3a 100644 --- a/tests/e2e/models.py +++ b/tests/e2e/models.py @@ -215,6 +215,8 @@ class ChatBody(BaseModel): messages: list[ChatMessage] stream: bool = False max_tokens: int | None = None + max_completion_tokens: int | None = None + temperature: float | None = None user: str | None = None metadata: ChatMetadata | None = None reasoning_effort: str | None = None @@ -292,6 +294,7 @@ class McpResponseMetadata(BaseModel): class OutMessage(BaseModel): + role: str | None = None content: str | None = None reasoning_content: str | None = None tool_calls: list[ToolCall] | None = None @@ -322,6 +325,7 @@ class Usage(BaseModel): class ChatResponse(BaseModel): id: str | None = None + object: str | None = None model: str | None = None choices: list[ChatChoice] = [] usage: Usage | None = None diff --git a/tests/e2e/quota_management/spend_tracking/spend_e2e_client.py b/tests/e2e/quota_management/spend_tracking/spend_e2e_client.py index 26860212fa3..617bb5c2ae9 100644 --- a/tests/e2e/quota_management/spend_tracking/spend_e2e_client.py +++ b/tests/e2e/quota_management/spend_tracking/spend_e2e_client.py @@ -16,6 +16,8 @@ from collections.abc import Callable from dataclasses import dataclass from datetime import datetime, timedelta, timezone +from pydantic import BaseModel + from e2e_config import unique_marker from e2e_http import ( NoBody, @@ -33,7 +35,6 @@ from models import ( ChatMessage, ChatMetadata, ChatResponse, - DateRangeParams, EmbedBody, EmbedResponse, OpenAPISchema, @@ -200,7 +201,7 @@ class SpendClient: ) ) - def probe(self, path: str, *, params: DateRangeParams) -> ProbeResult: + def probe(self, path: str, *, params: BaseModel) -> ProbeResult: return self.proxy.transport.probe(path, params=params) def openapi(self) -> OpenAPISchema: diff --git a/tests/e2e/quota_management/spend_tracking/test_team_daily_activity_e2e.py b/tests/e2e/quota_management/spend_tracking/test_team_daily_activity_e2e.py new file mode 100644 index 00000000000..086aaa74a2d --- /dev/null +++ b/tests/e2e/quota_management/spend_tracking/test_team_daily_activity_e2e.py @@ -0,0 +1,82 @@ +"""Vendor §9.20: GET /team/daily/activity structure and required query params (LIT-4778). + +The spend-route breadth probe only checks that the path responds. These cases pin +the customer-facing contract: a valid date range returns results+metadata, and +missing start/end dates are rejected. +""" + +from __future__ import annotations + +from datetime import datetime, timedelta, timezone + +import pytest +from pydantic import BaseModel + +from e2e_http import ProbeResult +from models import DateRangeParams +from spend_e2e_client import SpendClient + +pytestmark = pytest.mark.e2e + +ROUTE = "/team/daily/activity" + + +class TeamDailyActivityParams(BaseModel): + start_date: str | None = None + end_date: str | None = None + page: int = 1 + + +class TeamDailyActivityRow(BaseModel): + date: str | None = None + metrics: dict[str, object] | None = None + + +class TeamDailyActivityResponse(BaseModel): + results: list[TeamDailyActivityRow] = [] + metadata: dict[str, object] | None = None + + +def _range_days(days: int) -> DateRangeParams: + end = datetime.now(timezone.utc).date() + start = end - timedelta(days=days) + return DateRangeParams(start_date=start.isoformat(), end_date=end.isoformat()) + + +def _probe(client: SpendClient, params: BaseModel) -> ProbeResult: + return client.proxy.transport.probe(ROUTE, params=params) + + +class TestTeamDailyActivity: + @pytest.mark.covers("mgmt.team.daily_activity.happy_path") + @pytest.mark.parametrize("days", [1, 7, 30]) + def test_valid_date_range_returns_results_and_metadata( + self, client: SpendClient, days: int + ) -> None: + result = _probe(client, _range_days(days)) + assert result.status_code == 200, ( + f"{ROUTE} range={days}d must be 200, got {result.status_code}: {result.body[:600]}" + ) + parsed = TeamDailyActivityResponse.model_validate_json(result.body) + assert parsed.results is not None, f"results field required: {result.body[:600]}" + assert parsed.metadata is not None, f"metadata field required: {result.body[:600]}" + if parsed.results: + first = parsed.results[0] + assert first.date is not None, f"result row needs date: {result.body[:600]}" + assert first.metrics is not None, f"result row needs metrics: {result.body[:600]}" + + @pytest.mark.covers("mgmt.team.daily_activity.missing_start_date_rejected") + def test_missing_start_date_is_rejected(self, client: SpendClient) -> None: + end = datetime.now(timezone.utc).date().isoformat() + result = _probe(client, TeamDailyActivityParams(end_date=end, page=1)) + assert result.status_code == 400, ( + f"missing start_date must be 400, got {result.status_code}: {result.body[:600]}" + ) + + @pytest.mark.covers("mgmt.team.daily_activity.missing_end_date_rejected") + def test_missing_end_date_is_rejected(self, client: SpendClient) -> None: + start = (datetime.now(timezone.utc).date() - timedelta(days=1)).isoformat() + result = _probe(client, TeamDailyActivityParams(start_date=start, page=1)) + assert result.status_code == 400, ( + f"missing end_date must be 400, got {result.status_code}: {result.body[:600]}" + ) diff --git a/tests/e2e/transport.py b/tests/e2e/transport.py index da4252e550e..a6adf83ed1f 100644 --- a/tests/e2e/transport.py +++ b/tests/e2e/transport.py @@ -89,6 +89,7 @@ class Transport(Protocol): filename: str, content: bytes, file_content_type: str = "application/jsonl", + file_field: str = "file", params: BaseModel | None = None, response_type: type[R], ) -> Result[R]: ... @@ -242,6 +243,7 @@ class HttpTransport: filename: str, content: bytes, file_content_type: str = "application/jsonl", + file_field: str = "file", params: BaseModel | None = None, response_type: type[R], ) -> Result[R]: @@ -252,6 +254,7 @@ class HttpTransport: filename=filename, content=content, file_content_type=file_content_type, + file_field=file_field, params=params, response_type=response_type, timeout=self.request_timeout, @@ -411,6 +414,7 @@ class SplitTransport: filename: str, content: bytes, file_content_type: str = "application/jsonl", + file_field: str = "file", params: BaseModel | None = None, response_type: type[R], ) -> Result[R]: @@ -421,6 +425,7 @@ class SplitTransport: filename=filename, content=content, file_content_type=file_content_type, + file_field=file_field, params=params, response_type=response_type, )