test(decisions): add hosted_vllm /v1/decisions and provider-400 translation cases (#45661)

Co-authored-by: kerry <kerry@berri.ai>
Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
devin-ai-integration[bot] 2026-10-09 20:42:58 +00:00 • committed by GitHub
parent 4c78023db2
commit 097d018dbf
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
4 changed files with 122 additions and 4 deletions

View file

@ -7,7 +7,8 @@ from pydantic import JsonValue
@dataclass(frozen=True, slots=True, kw_only=True)
class TranslationTestCase:
"""One request through the proxy to a deployment in `proxy_config.yaml`: what the test sends to LiteLLM, the
exact request the provider must receive, the fake provider's reply, and the exact response LiteLLM must return."""
exact request the provider must receive, the fake provider's status and reply, and the exact response LiteLLM must
return."""
scenario: str
litellm_endpoint: str
@ -15,6 +16,7 @@ class TranslationTestCase:
expected_provider_endpoint: str
expected_provider_headers: Mapping[str, str]
expected_provider_request: Mapping[str, JsonValue]
mock_provider_status_code: int = 200
mock_provider_response: Mapping[str, JsonValue]
expected_litellm_status_code: int = 200
expected_litellm_response: Mapping[str, JsonValue]

View file

@ -1,3 +1,5 @@
import json
from dataclasses import replace
from typing import Final
from integration.translation.case import TranslationTestCase
@ -65,3 +67,109 @@ QWEN3_0_6B_TEST_CASE: Final = TranslationTestCase(
"diagnostics": {"is_urgent": {"label_mass": 0.9979320910923947, "argmax_is_label": True}},
},
)
"""OpenAI Decisions client shape in, vLLM /v1/systemone body out. Mock reply captured live on 2026-10-09 from the same
vLLM build.
"""
QWEN3_0_6B_DECISIONS_TEST_CASE: Final = replace(
QWEN3_0_6B_TEST_CASE,
scenario="decisions",
litellm_endpoint="/v1/decisions",
litellm_request={
"model": "hosted_vllm/Qwen/Qwen3-0.6B",
"input": "The checkout page crashed twice today",
"questions": [
{
"type": "choice",
"name": "is_defect",
"instructions": "Is this a defect?",
"choices": [{"value": "yes"}, {"value": "no"}],
}
],
"cache": {"no-cache": True},
},
expected_provider_request={
"model": "Qwen/Qwen3-0.6B",
"state": "The checkout page crashed twice today",
"questions": {
"is_defect": {"type": "choice", "instructions": "Is this a defect?", "criteria": {"yes": None, "no": None}}
},
},
mock_provider_response={
"id": "decision-91ce41627dbf016f",
"object": "structured_decision",
"created": 1791577835,
"model": "Qwen/Qwen3-0.6B",
"answers": {
"is_defect": {
"type": "choice",
"choice": "yes",
"probabilities": {"yes": 0.8670357477770336, "no": 0.13296425222296632},
"confidence": 0.8669148648509282,
}
},
"usage": {"input_tokens": 43, "output_tokens": 1},
"diagnostics": {"is_defect": {"label_mass": 0.9998605790748358, "argmax_is_label": True}},
},
expected_litellm_response={
"model": "Qwen/Qwen3-0.6B",
"answers": [
{
"type": "choice",
"name": "is_defect",
"choice": "yes",
"probabilities": [
{"value": "yes", "probability": 0.8670357477770336},
{"value": "no", "probability": 0.13296425222296632},
],
"confidence": 0.8669148648509282,
}
],
"usage": {
"input_tokens": 43,
"input_tokens_details": {"cached_tokens": 0, "cache_write_tokens": 0},
"output_tokens": 1,
"output_tokens_details": {"reasoning_tokens": 0},
"total_tokens": 44,
},
},
)
"""vLLM only scores `choice` questions, so an OpenAI `predicate` goes out as `noul` and comes back as vLLM's 400,
which LiteLLM passes through as a 400.
"""
VLLM_NOUL_REJECTED: Final = {
"error": {
"message": "unknown question type 'noul'; supported: ['choice']",
"type": "BadRequestError",
"param": None,
"code": 400,
}
}
QWEN3_0_6B_PREDICATE_REJECTED_TEST_CASE: Final = replace(
QWEN3_0_6B_DECISIONS_TEST_CASE,
scenario="predicate_rejected",
litellm_request={
"model": "hosted_vllm/Qwen/Qwen3-0.6B",
"input": "The checkout page crashed twice today",
"questions": [{"type": "predicate", "name": "is_bug", "instructions": "Is this a bug?"}],
"cache": {"no-cache": True},
},
expected_provider_request={
"model": "Qwen/Qwen3-0.6B",
"state": "The checkout page crashed twice today",
"questions": {"is_bug": {"type": "noul", "instructions": "Is this a bug?"}},
},
mock_provider_status_code=400,
mock_provider_response=VLLM_NOUL_REJECTED,
expected_litellm_status_code=400,
expected_litellm_response={
"error": {
"message": f"litellm.BadRequestError: Hosted_vllmException - {json.dumps(VLLM_NOUL_REJECTED)}\n\n"
"LiteLLM: model group 'hosted_vllm/Qwen/Qwen3-0.6B' failed with the error above. No fallback was attempted.",
"type": "invalid_request_error",
"param": None,
"code": "400",
}
},
)

View file

@ -2,10 +2,18 @@ import pytest
from integration._support.client import Gateway
from integration._support.provider import SharedProvider
from integration.translation.case import TranslationTestCase
from integration.translation.decisions.bases.hosted_vllm import QWEN3_0_6B_TEST_CASE
from integration.translation.decisions.bases.hosted_vllm import (
QWEN3_0_6B_DECISIONS_TEST_CASE,
QWEN3_0_6B_PREDICATE_REJECTED_TEST_CASE,
QWEN3_0_6B_TEST_CASE,
)
from integration.translation.runner import assert_translation
@pytest.mark.parametrize("case", [QWEN3_0_6B_TEST_CASE], ids=lambda case: case.id)
@pytest.mark.parametrize(
"case",
[QWEN3_0_6B_TEST_CASE, QWEN3_0_6B_DECISIONS_TEST_CASE, QWEN3_0_6B_PREDICATE_REJECTED_TEST_CASE],
ids=lambda case: case.id,
)
def test_decisions_basic_hosted_vllm(case: TranslationTestCase, gateway: Gateway, provider: SharedProvider) -> None:
assert_translation(case, gateway, provider)

View file

@ -10,7 +10,7 @@ TRANSPORT_HEADERS: Final = frozenset({"host", "accept", "accept-encoding", "conn
def assert_translation(case: TranslationTestCase, gateway: Gateway, provider: SharedProvider) -> None:
provider.expect(Reply(body=json.dumps(case.mock_provider_response).encode()))
provider.expect(Reply(status=case.mock_provider_status_code, body=json.dumps(case.mock_provider_response).encode()))
response: Final = gateway.request("POST", case.litellm_endpoint, case.litellm_request)
received: Final = provider.received()
assert [(request.method, request.target) for request in received] == [("POST", case.expected_provider_endpoint)]