mirror of
https://github.com/BerriAI/litellm.git
synced 2026-09-10 22:41:41 +00:00
* test(e2e): cover passthrough headers, batch assume-role, gemini, vllm, bedrock guardrails, batch rate-limit mapping Add parent-package e2e suites for the six feature gaps: pass-through header forwarding via /config/pass_through_endpoint, Bedrock batch STS assume-role, Gemini chat + files, hosted_vllm batch/files, Bedrock guardrail pre_call blocks (plus restored content-filter team opt-out), and OpenAI batch RPM 429 body mapping. Registry cells and LiteLLMParamsBody/TeamMetadata fields updated so markers collect cleanly. * test(e2e): cover LIT-4587 gaps for redis, responses, tpm cache, apply_guardrail, langfuse Adds customer-shaped live e2e for apply_guardrail, responses store+metadata TTL, TPM excluding cached tokens, redis-backed RPM, redis circuit-breaker path, Langfuse spend, Cohere chat, virtual-key auth, file content download, hosted_vllm chat, and Nova Sonic realtime. Registry cells updated for the new markers. * test(e2e): drive LIT-4587 gap suites on Anthropic to avoid Gemini quota flakes Redis RPM, circuit-breaker path, virtual-key auth, responses metadata, and Langfuse driver models now use Anthropic haiku so local runs stay green when Gemini daily quota is exhausted. * test(e2e): drop Langfuse spend suite; feature is being deprecated Remove test_langfuse_e2e.py, logging.langfuse registry cells, and the langfuse-only conftest driver/credentials fixtures. * test(e2e): fold provider/batch feature tests into their endpoint suites Keep the e2e layout endpoint- and suite-scoped instead of one file per provider or feature Move the virtual-key auth case into access_control/test_access_control_e2e.py as TestVirtualKeyAuth (replacing an incomplete stub) and drop the standalone test_virtual_key_auth_e2e.py Fold the five per-file batch suites (file content, RPM 429 mapping, Bedrock assume-role, Gemini files, hosted_vllm batch) into batches/test_batches_e2e.py. The hosted_vllm batch case is skipped for now since it needs a live vLLM server (HOSTED_VLLM_API_BASE) the e2e environment does not provision; it and the gemini-files and RPM-mapping cases reference LIT-3382 / LIT-3266 where relevant Merge the cohere, gemini and hosted_vllm chat cases into llm_translation/test_chat_completions_regression_e2e.py so /chat/completions coverage lives in one endpoint file, and repoint the coverage_registry source fields to the new homes Move the shared CacheControl / TextBlock / RichMessage request blocks into the root models.py (re-exported from endpoints_client) so quota_management can use them without a cross-suite import, which also clears the basedpyright errors in test_tpm_excludes_cached_tokens_e2e.py; type the httpbin echo body in test_passthrough_headers_e2e.py with a pydantic model to drop the Any-typed json.loads path * test(e2e): address review feedback and re-home virtual-key coverage Replace the tautological Bedrock assume-role batch id assertion (`startswith(...) or batch.id`, always true) with a managed-id shape check, since the unified target_model_names path re-encodes the id rather than returning a raw ARN Raise the batch RPM-mapping test's rpm_limit above one so the file upload can no longer consume the key's sole request unit before batch create runs; the batch create then clears the generic per-request limiter and the batch limiter is what returns the "Batch rate limit exceeded" body the assertions check Set exercised_on to [] on the pass-through header test; it drives a pass-through endpoint, not /chat/completions Move the virtual-key valid_allows / invalid_denied cells from other.yaml to mgmt.yaml as mgmt.virtual_key.* so TestVirtualKeyAuth rolls up under Management, and point its covers marker at the new ids
62 lines
2.4 KiB
Python
62 lines
2.4 KiB
Python
"""Live e2e: POST /guardrails/apply_guardrail is the customer-facing apply surface.
|
|
|
|
Customers call this endpoint to run a named guardrail without going through chat.
|
|
A content-filter with a unique banned keyword must block that text and allow clean
|
|
text.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import pytest
|
|
|
|
from e2e_config import MASTER_KEY, unique_marker
|
|
from e2e_http import Success, UnauthorizedError, UnknownApiError
|
|
from guardrails_client import GuardrailsClient
|
|
from lifecycle import ResourceManager
|
|
|
|
pytestmark = pytest.mark.e2e
|
|
|
|
|
|
class TestApplyGuardrailEndpoint:
|
|
@pytest.mark.covers(
|
|
"guardrail.litellm_content_filter.apply_endpoint.blocks",
|
|
"guardrail.litellm_content_filter.apply_endpoint.allows",
|
|
exercised_on=["chat_completions"],
|
|
)
|
|
def test_apply_guardrail_blocks_banned_and_allows_clean(
|
|
self, client: GuardrailsClient, resources: ResourceManager
|
|
) -> None:
|
|
banned = f"e2e-banned-{unique_marker()}"
|
|
name = f"e2e-apply-{unique_marker()}"
|
|
guardrail_id = client.create_content_filter_guardrail(name, banned)
|
|
resources.defer(lambda: client.delete_guardrail(guardrail_id))
|
|
|
|
blocked = client.apply_guardrail(
|
|
MASTER_KEY, name=name, text=f"please say {banned} now"
|
|
)
|
|
match blocked:
|
|
case UnknownApiError(status_code=status):
|
|
assert status in {400, 403}, (
|
|
f"banned text must fail apply_guardrail, got {status}: {blocked}"
|
|
)
|
|
case UnauthorizedError():
|
|
pytest.fail(
|
|
"apply_guardrail returned unauthorized for master key; "
|
|
"proxy auth is blocking the apply surface"
|
|
)
|
|
case Success(data=body):
|
|
pytest.fail(
|
|
f"banned text must not pass apply_guardrail; got {body}"
|
|
)
|
|
case _:
|
|
pytest.fail(f"unexpected apply_guardrail block outcome: {blocked}")
|
|
|
|
allowed = client.apply_guardrail(
|
|
MASTER_KEY, name=name, text="hello, this is clean input"
|
|
)
|
|
match allowed:
|
|
case Success(data=body):
|
|
assert body.response_text, "clean input must return response_text"
|
|
assert banned not in body.response_text
|
|
case _:
|
|
pytest.fail(f"clean input must succeed on apply_guardrail: {allowed}")
|