mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-10 03:28:53 +00:00
test(decisions): add hosted_vllm /v1/decisions and provider-400 translation cases (#45661)
Co-authored-by: kerry <kerry@berri.ai> Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
parent
4c78023db2
commit
097d018dbf
4 changed files with 122 additions and 4 deletions
|
|
@ -7,7 +7,8 @@ from pydantic import JsonValue
|
|||
@dataclass(frozen=True, slots=True, kw_only=True)
|
||||
class TranslationTestCase:
|
||||
"""One request through the proxy to a deployment in `proxy_config.yaml`: what the test sends to LiteLLM, the
|
||||
exact request the provider must receive, the fake provider's reply, and the exact response LiteLLM must return."""
|
||||
exact request the provider must receive, the fake provider's status and reply, and the exact response LiteLLM must
|
||||
return."""
|
||||
|
||||
scenario: str
|
||||
litellm_endpoint: str
|
||||
|
|
@ -15,6 +16,7 @@ class TranslationTestCase:
|
|||
expected_provider_endpoint: str
|
||||
expected_provider_headers: Mapping[str, str]
|
||||
expected_provider_request: Mapping[str, JsonValue]
|
||||
mock_provider_status_code: int = 200
|
||||
mock_provider_response: Mapping[str, JsonValue]
|
||||
expected_litellm_status_code: int = 200
|
||||
expected_litellm_response: Mapping[str, JsonValue]
|
||||
|
|
|
|||
|
|
@ -1,3 +1,5 @@
|
|||
import json
|
||||
from dataclasses import replace
|
||||
from typing import Final
|
||||
|
||||
from integration.translation.case import TranslationTestCase
|
||||
|
|
@ -65,3 +67,109 @@ QWEN3_0_6B_TEST_CASE: Final = TranslationTestCase(
|
|||
"diagnostics": {"is_urgent": {"label_mass": 0.9979320910923947, "argmax_is_label": True}},
|
||||
},
|
||||
)
|
||||
|
||||
"""OpenAI Decisions client shape in, vLLM /v1/systemone body out. Mock reply captured live on 2026-10-09 from the same
|
||||
vLLM build.
|
||||
"""
|
||||
QWEN3_0_6B_DECISIONS_TEST_CASE: Final = replace(
|
||||
QWEN3_0_6B_TEST_CASE,
|
||||
scenario="decisions",
|
||||
litellm_endpoint="/v1/decisions",
|
||||
litellm_request={
|
||||
"model": "hosted_vllm/Qwen/Qwen3-0.6B",
|
||||
"input": "The checkout page crashed twice today",
|
||||
"questions": [
|
||||
{
|
||||
"type": "choice",
|
||||
"name": "is_defect",
|
||||
"instructions": "Is this a defect?",
|
||||
"choices": [{"value": "yes"}, {"value": "no"}],
|
||||
}
|
||||
],
|
||||
"cache": {"no-cache": True},
|
||||
},
|
||||
expected_provider_request={
|
||||
"model": "Qwen/Qwen3-0.6B",
|
||||
"state": "The checkout page crashed twice today",
|
||||
"questions": {
|
||||
"is_defect": {"type": "choice", "instructions": "Is this a defect?", "criteria": {"yes": None, "no": None}}
|
||||
},
|
||||
},
|
||||
mock_provider_response={
|
||||
"id": "decision-91ce41627dbf016f",
|
||||
"object": "structured_decision",
|
||||
"created": 1791577835,
|
||||
"model": "Qwen/Qwen3-0.6B",
|
||||
"answers": {
|
||||
"is_defect": {
|
||||
"type": "choice",
|
||||
"choice": "yes",
|
||||
"probabilities": {"yes": 0.8670357477770336, "no": 0.13296425222296632},
|
||||
"confidence": 0.8669148648509282,
|
||||
}
|
||||
},
|
||||
"usage": {"input_tokens": 43, "output_tokens": 1},
|
||||
"diagnostics": {"is_defect": {"label_mass": 0.9998605790748358, "argmax_is_label": True}},
|
||||
},
|
||||
expected_litellm_response={
|
||||
"model": "Qwen/Qwen3-0.6B",
|
||||
"answers": [
|
||||
{
|
||||
"type": "choice",
|
||||
"name": "is_defect",
|
||||
"choice": "yes",
|
||||
"probabilities": [
|
||||
{"value": "yes", "probability": 0.8670357477770336},
|
||||
{"value": "no", "probability": 0.13296425222296632},
|
||||
],
|
||||
"confidence": 0.8669148648509282,
|
||||
}
|
||||
],
|
||||
"usage": {
|
||||
"input_tokens": 43,
|
||||
"input_tokens_details": {"cached_tokens": 0, "cache_write_tokens": 0},
|
||||
"output_tokens": 1,
|
||||
"output_tokens_details": {"reasoning_tokens": 0},
|
||||
"total_tokens": 44,
|
||||
},
|
||||
},
|
||||
)
|
||||
|
||||
"""vLLM only scores `choice` questions, so an OpenAI `predicate` goes out as `noul` and comes back as vLLM's 400,
|
||||
which LiteLLM passes through as a 400.
|
||||
"""
|
||||
VLLM_NOUL_REJECTED: Final = {
|
||||
"error": {
|
||||
"message": "unknown question type 'noul'; supported: ['choice']",
|
||||
"type": "BadRequestError",
|
||||
"param": None,
|
||||
"code": 400,
|
||||
}
|
||||
}
|
||||
QWEN3_0_6B_PREDICATE_REJECTED_TEST_CASE: Final = replace(
|
||||
QWEN3_0_6B_DECISIONS_TEST_CASE,
|
||||
scenario="predicate_rejected",
|
||||
litellm_request={
|
||||
"model": "hosted_vllm/Qwen/Qwen3-0.6B",
|
||||
"input": "The checkout page crashed twice today",
|
||||
"questions": [{"type": "predicate", "name": "is_bug", "instructions": "Is this a bug?"}],
|
||||
"cache": {"no-cache": True},
|
||||
},
|
||||
expected_provider_request={
|
||||
"model": "Qwen/Qwen3-0.6B",
|
||||
"state": "The checkout page crashed twice today",
|
||||
"questions": {"is_bug": {"type": "noul", "instructions": "Is this a bug?"}},
|
||||
},
|
||||
mock_provider_status_code=400,
|
||||
mock_provider_response=VLLM_NOUL_REJECTED,
|
||||
expected_litellm_status_code=400,
|
||||
expected_litellm_response={
|
||||
"error": {
|
||||
"message": f"litellm.BadRequestError: Hosted_vllmException - {json.dumps(VLLM_NOUL_REJECTED)}\n\n"
|
||||
"LiteLLM: model group 'hosted_vllm/Qwen/Qwen3-0.6B' failed with the error above. No fallback was attempted.",
|
||||
"type": "invalid_request_error",
|
||||
"param": None,
|
||||
"code": "400",
|
||||
}
|
||||
},
|
||||
)
|
||||
|
|
|
|||
|
|
@ -2,10 +2,18 @@ import pytest
|
|||
from integration._support.client import Gateway
|
||||
from integration._support.provider import SharedProvider
|
||||
from integration.translation.case import TranslationTestCase
|
||||
from integration.translation.decisions.bases.hosted_vllm import QWEN3_0_6B_TEST_CASE
|
||||
from integration.translation.decisions.bases.hosted_vllm import (
|
||||
QWEN3_0_6B_DECISIONS_TEST_CASE,
|
||||
QWEN3_0_6B_PREDICATE_REJECTED_TEST_CASE,
|
||||
QWEN3_0_6B_TEST_CASE,
|
||||
)
|
||||
from integration.translation.runner import assert_translation
|
||||
|
||||
|
||||
@pytest.mark.parametrize("case", [QWEN3_0_6B_TEST_CASE], ids=lambda case: case.id)
|
||||
@pytest.mark.parametrize(
|
||||
"case",
|
||||
[QWEN3_0_6B_TEST_CASE, QWEN3_0_6B_DECISIONS_TEST_CASE, QWEN3_0_6B_PREDICATE_REJECTED_TEST_CASE],
|
||||
ids=lambda case: case.id,
|
||||
)
|
||||
def test_decisions_basic_hosted_vllm(case: TranslationTestCase, gateway: Gateway, provider: SharedProvider) -> None:
|
||||
assert_translation(case, gateway, provider)
|
||||
|
|
|
|||
|
|
@ -10,7 +10,7 @@ TRANSPORT_HEADERS: Final = frozenset({"host", "accept", "accept-encoding", "conn
|
|||
|
||||
|
||||
def assert_translation(case: TranslationTestCase, gateway: Gateway, provider: SharedProvider) -> None:
|
||||
provider.expect(Reply(body=json.dumps(case.mock_provider_response).encode()))
|
||||
provider.expect(Reply(status=case.mock_provider_status_code, body=json.dumps(case.mock_provider_response).encode()))
|
||||
response: Final = gateway.request("POST", case.litellm_endpoint, case.litellm_request)
|
||||
received: Final = provider.received()
|
||||
assert [(request.method, request.target) for request in received] == [("POST", case.expected_provider_endpoint)]
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue