fix(decisions)!: refuse safety_identifier on providers that cannot take it unless drop_params drops it (#44955)

* feat(decisions): add the OpenAI Decisions spec types and the System One translation

Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>

* refactor(decisions): share one DecisionsModel config and require model and usage on responses

Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>

* refactor(decisions): rename the shared pydantic parent to DecisionsObjectBase

Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>

* refactor(decisions): dispatch /v1/decisions through provider configs and the shared HTTP handler

Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>

* fix(decisions): map OpenRouter connection failures to APIConnectionError and drop explanatory docstrings

Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>

* feat(decisions): switch /v1/decisions to the OpenAI Decisions schema

Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>

* test(decisions): build the usage the OpenAI response schema now requires

Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>

* fix(lens): ask signal questions through the OpenAI Decisions schema

Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>

* feat(decisions): refuse safety_identifier for System One providers unless drop_params

* feat(decisions): refuse safety_identifier on providers that cannot take it unless drop_params drops it

* test(decisions): cover hosted_vllm in the safety_identifier and per-provider wire tests

* fix(decisions): check a System One request's safety_identifier against the provider too

* fix(decisions): drop a non-string safety_identifier under drop_params

A malformed safety_identifier now follows the drop_params convention in both body shapes: it answers 400 without drop_params and is dropped before the provider call with it. A string identifier on OpenAI stays on the wire either way.

* fix(decisions): let /v1/decisions drop a non-string safety_identifier under drop_params

The route checked the whole body before routing, so a non-string
safety_identifier answered 400 even when the deployment or the body set
drop_params. The route now leaves that field to the Decisions call, which
knows every drop_params source

* refactor(decisions): return the unsupported safety_identifier as a value and raise it in _prepare_call

---------

Co-authored-by: kerry <kerry@berri.ai>
Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
Co-authored-by: mateo-berri <277851410+mateo-berri@users.noreply.github.com>
This commit is contained in:
devin-ai-integration[bot] 2026-10-09 19:14:48 -07:00 • committed by GitHub
parent d6db8e8744
commit 546402c98c
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
18 changed files with 1289 additions and 19 deletions

View file

@ -1,13 +1,13 @@
from collections.abc import Mapping, Sequence
from dataclasses import dataclass, field
from dataclasses import dataclass, field, replace
from typing import Final, TypeAlias
import httpx
from pydantic import TypeAdapter, ValidationError
from pydantic import ConfigDict, TypeAdapter, ValidationError
from typing_extensions import assert_never
import litellm
from litellm.litellm_core_utils.core_helpers import RESPONSE_COST_HEADER
from litellm.litellm_core_utils.core_helpers import RESPONSE_COST_HEADER, normalize_drop_params
from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj
from litellm.llms.base_llm.chat.transformation import BaseLLMException
from litellm.llms.base_llm.decisions.transformation import (
@ -40,6 +40,9 @@ DecisionsRequestFormat: TypeAlias = DecisionsRequestBody | OpenAIDecisionRequest
_SYSTEMONE_REQUEST_ADAPTER: Final[TypeAdapter[DecisionsRequestBody]] = TypeAdapter(DecisionsRequestBody)
_OPENAI_REQUEST_ADAPTER: Final[TypeAdapter[OpenAIDecisionRequestBody]] = TypeAdapter(OpenAIDecisionRequestBody)
_SAFETY_IDENTIFIER_ADAPTER: Final[TypeAdapter[str | None]] = TypeAdapter(
str | None, config=ConfigDict(title="safety_identifier")
)
_HANDLER: Final = BaseLLMHTTPHandler()
@ -96,16 +99,46 @@ def _validate_request(
)
def _ir_request(request: DecisionsRequestFormat) -> DecisionsIRRequest:
def _ir_request(request: DecisionsRequestFormat, safety_identifier: str | None) -> DecisionsIRRequest:
match request:
case DecisionsRequestBody():
return systemone_request_to_ir(request)
return replace(systemone_request_to_ir(request), safety_identifier=safety_identifier)
case OpenAIDecisionRequestBody():
return openai_request_to_ir(request)
case _:
assert_never(request)
def _drops_params(kwargs: Mapping[str, object]) -> bool:
return litellm.drop_params is True or normalize_drop_params(kwargs.get("drop_params")) is True
def _request_safety_identifier(safety_identifier: object, kwargs: Mapping[str, object]) -> str | None:
if isinstance(safety_identifier, str) or not _drops_params(kwargs):
return _SAFETY_IDENTIFIER_ADAPTER.validate_python(safety_identifier)
return None
@dataclass(frozen=True, slots=True)
class _UnsupportedSafetyIdentifier:
pass
def _provider_ir_request(
request: DecisionsRequestFormat,
*,
safety_identifier: str | None,
provider_config: BaseDecisionsConfig,
kwargs: Mapping[str, object],
) -> DecisionsIRRequest | _UnsupportedSafetyIdentifier:
ir_request: Final = _ir_request(request, safety_identifier)
if ir_request.safety_identifier is None or provider_config.supports_safety_identifier:
return ir_request
if _drops_params(kwargs):
return replace(ir_request, safety_identifier=None)
return _UnsupportedSafetyIdentifier()
def _prepare_call(
*,
model: str,
@ -144,8 +177,12 @@ def _prepare_call(
except ValueError as error:
raise litellm.BadRequestError(message=str(error), model=model, llm_provider=provider) from error
try:
request_safety_identifier: Final = _request_safety_identifier(safety_identifier, kwargs)
request: Final = _validate_request(
state=state, questions=questions, decision_input=decision_input, safety_identifier=safety_identifier
state=state,
questions=questions,
decision_input=decision_input,
safety_identifier=request_safety_identifier,
)
except ValidationError as error:
raise litellm.BadRequestError(
@ -169,7 +206,22 @@ def _prepare_call(
llm_provider=provider,
)
ir_request: Final = _ir_request(request)
ir_request: Final = _provider_ir_request(
request,
safety_identifier=request_safety_identifier,
provider_config=provider_config,
kwargs=kwargs,
)
if isinstance(ir_request, _UnsupportedSafetyIdentifier):
raise litellm.UnsupportedParamsError(
message=(
f"{provider} does not support parameters: ['safety_identifier'], for model={model}. "
"To drop these, set `litellm.drop_params=True` or for proxy:\n\n"
"`litellm_settings:\n drop_params: true`\n"
),
model=model,
llm_provider=provider,
)
body: Final = provider_config.transform_decisions_request(
model=canonical_model, request=ir_request, custom_llm_provider=provider
)

View file

@ -301,6 +301,7 @@ class BaseDecisionsConfig(ABC):
api_key_env: tuple[str, ...] = ()
api_base_env: tuple[str, ...] = ()
api_key_required: bool = True
supports_safety_identifier: bool = False
health_check_questions: Mapping[str, Mapping[str, object]] = MappingProxyType(
{"reachable": MappingProxyType({"type": "noul", "instructions": "Is the service reachable?"})}
)

View file

@ -292,6 +292,7 @@ def ir_to_openai_response(
class OpenAIDecisionsConfig(BaseDecisionsConfig):
path = "/v1/decisions"
supports_safety_identifier = True
def get_default_api_base(self) -> str | None:
return "https://api.openai.com"

View file

@ -1,4 +1,5 @@
from collections.abc import Mapping
from types import MappingProxyType
from typing import Annotated, Final
from fastapi import APIRouter, Depends, Request, Response
@ -38,6 +39,10 @@ async def _invalid_request(
)
def _fields_checked_before_routing(data: Mapping[str, object]) -> Mapping[str, object]:
return MappingProxyType({key: value for key, value in data.items() if key != "safety_identifier"})
async def _request_data(request: Request, user_api_key_dict: UserAPIKeyAuth) -> dict[str, object]:
body: Final = await request.body()
try:
@ -75,7 +80,7 @@ async def _process_decisions(
data: Final = await _request_data(request, user_api_key_dict)
try:
body_adapter.validate_python(data)
body_adapter.validate_python(_fields_checked_before_routing(data))
except ValidationError as error:
raise await _invalid_request(raw_data=data, error=error, user_api_key_dict=user_api_key_dict)
general_settings: Final = _GENERAL_SETTINGS_ADAPTER.validate_python(proxy_general_settings)

View file

@ -208,7 +208,6 @@ def test_an_openai_format_request_at_v1_decisions_reaches_the_serving_endpoint_a
"model": deployment,
"input": _OPENAI_INPUT,
"questions": [_OPENAI_TIER_QUESTION],
"safety_identifier": "end-user-1",
},
)
assert response.status_code == 200, response.text

View file

@ -0,0 +1,586 @@
import math
import uuid
from collections.abc import Callable, Sequence
from dataclasses import dataclass
from pathlib import Path
from typing import Final
import httpx
import pytest
import yaml
from integration._support.client import Gateway, Scenario, eventually, object_value, string_value
from integration._support.database import read_rows
from integration._support.process import owned_proxy_process
from integration._support.upstream import ScenarioHandle, delete_scenario, register_scenario
from integration.cost_calculation.cost_tracking_case import JsonResponse
from pydantic import JsonValue, TypeAdapter
import litellm
_API_KEY: Final = "synthetic-decisions-key"
_SAFETY_IDENTIFIER: Final = "end-user-7"
_DROPPING_DEPLOYMENT: Final = "decisions-under-litellm-settings-drop-params"
_JSON_OBJECT: Final = TypeAdapter(dict[str, JsonValue])
_INPUT: Final = "Ticket (billing): The export job hangs at 99%"
_FOLLOW_UP: Final = "Customer: still stuck after retrying"
_QUESTIONS: Final[list[JsonValue]] = [
{"type": "predicate", "name": "defect", "instructions": "Is this a defect?"},
{
"type": "choice",
"name": "severity",
"instructions": "How severe is it?",
"choices": [{"value": "low", "description": "cosmetic"}, {"value": "high", "description": "blocks users"}],
},
{
"type": "score",
"name": "confidence",
"instructions": "How sure are you?",
"levels": [{"label": "unsure"}, {"label": "sure"}],
},
]
_SYSTEM_ONE_QUESTIONS: Final[dict[str, JsonValue]] = {
"defect": {"type": "noul", "instructions": "Is this a defect?"},
"severity": {
"type": "choice",
"instructions": "How severe is it?",
"criteria": {"low": "cosmetic", "high": "blocks users"},
},
"confidence": {"type": "score", "instructions": "How sure are you?", "criteria": ["unsure", "sure"]},
}
_SYSTEM_ONE_ANSWERS: Final[dict[str, JsonValue]] = {
"defect": {"type": "noul", "noul": 0.93},
"severity": {"type": "choice", "choice": "high", "confidence": 0.8, "probabilities": {"low": 0.2, "high": 0.8}},
"confidence": {
"type": "score",
"score": 1.0,
"confidence": 0.7,
"legend": {"0": "unsure", "1": "sure"},
"probabilities": {"0": 0.3, "1": 0.7},
},
}
_ANSWERS: Final[list[JsonValue]] = [
{"type": "predicate", "name": "defect", "probability": 0.93},
{
"type": "choice",
"name": "severity",
"choice": "high",
"probabilities": [{"value": "low", "probability": 0.2}, {"value": "high", "probability": 0.8}],
"confidence": 0.8,
},
{
"type": "score",
"name": "confidence",
"score": 1.0,
"probabilities": [
{"value": 0, "label": "unsure", "probability": 0.3},
{"value": 1, "label": "sure", "probability": 0.7},
],
"confidence": 0.7,
},
]
_INPUT_TOKENS: Final = 367
_OUTPUT_TOKENS: Final = 3
_SYSTEM_ONE_USAGE: Final[dict[str, JsonValue]] = {"input_tokens": _INPUT_TOKENS, "output_tokens": _OUTPUT_TOKENS}
_USAGE: Final[dict[str, JsonValue]] = {
"input_tokens": _INPUT_TOKENS,
"input_tokens_details": {"cached_tokens": 0, "cache_write_tokens": 0},
"output_tokens": _OUTPUT_TOKENS,
"output_tokens_details": {"reasoning_tokens": 0},
"total_tokens": _INPUT_TOKENS + _OUTPUT_TOKENS,
}
_SDK_QUESTIONS: Final = TypeAdapter(list[dict[str, object]]).validate_python(_QUESTIONS)
_SPEND_QUERY: Final = (
"SELECT spend, status, call_type, model_group, custom_llm_provider, api_base, prompt_tokens, completion_tokens, "
'request_tags FROM "LiteLLM_SpendLogs" WHERE request_id = %s'
)
@dataclass(frozen=True, slots=True)
class _Provider:
name: str
model: str
path: str
body_model: str
api_key: str | None
wraps_result: bool
cost_map_key: str | None
speaks_openai: bool = False
def upstream_body(self) -> dict[str, JsonValue]:
if self.speaks_openai:
return {"model": self.body_model, "input": _INPUT, "questions": _QUESTIONS}
return {"model": self.body_model, "state": _INPUT, "questions": _SYSTEM_ONE_QUESTIONS}
def upstream_reply(self) -> dict[str, JsonValue]:
if self.speaks_openai:
return {"model": self.body_model, "answers": _ANSWERS, "usage": _USAGE}
answer: Final[dict[str, JsonValue]] = {
"model": self.body_model,
"answers": _SYSTEM_ONE_ANSWERS,
"usage": _SYSTEM_ONE_USAGE,
}
return {"result": answer, "success": True} if self.wraps_result else answer
def litellm_response(self) -> dict[str, JsonValue]:
return {"model": self.body_model, "answers": _ANSWERS, "usage": _USAGE}
_PROVIDERS: Final = (
_Provider(
"perplexity",
"perplexity/pplx-decider-v1-27b",
"/v1/decisions",
"pplx-decider-v1-27b",
_API_KEY,
False,
"perplexity/pplx-decider-v1-27b",
),
_Provider("typesafe", "typesafe/jev-1.13.0", "/v1/systemone", "jev-1.13.0", _API_KEY, False, "typesafe/jev-1.13.0"),
_Provider(
"openrouter",
"openrouter/typesafe/jev-1.13",
"/alpha/decisions",
"typesafe/jev-1.13",
_API_KEY,
False,
"openrouter/typesafe/jev-1.13",
),
_Provider(
"strands_decider", "strands_decider/systemone-decider", "/v1/systemone", "systemone-decider", None, False, None
),
_Provider(
"cloudflare",
"cloudflare/clef",
"/ai/run/@cf/cloudflare/clef",
"clef",
_API_KEY,
True,
"cloudflare/@cf/cloudflare/clef",
),
_Provider("hosted_vllm", "hosted_vllm/Qwen/Qwen3-0.6B", "/v1/systemone", "Qwen/Qwen3-0.6B", None, False, None),
_Provider(
"databricks",
"databricks/databricks-openjev-qwen35-4b",
"/databricks-openjev-qwen35-4b/invocations",
"databricks-openjev-qwen35-4b",
_API_KEY,
False,
None,
),
_Provider(
"azure_ai", "azure_ai/decision-1", "/providers/microsoft/v1/systemone", "decision-1", _API_KEY, False, None
),
_Provider("openai", "openai/gpt-6-luna", "/v1/decisions", "gpt-6-luna", _API_KEY, False, "gpt-6-luna", True),
)
_SYSTEM_ONE_PROVIDERS: Final = tuple(provider for provider in _PROVIDERS if not provider.speaks_openai)
_PERPLEXITY: Final = _PROVIDERS[0]
_TYPESAFE: Final = _PROVIDERS[1]
_OPENAI: Final = next(provider for provider in _PROVIDERS if provider.speaks_openai)
_PREDICATE: Final[dict[str, JsonValue]] = {"type": "predicate", "name": "q", "instructions": "Is it?"}
_INVALID_BODIES: Final[tuple[tuple[str, dict[str, JsonValue]], ...]] = (
("missing questions", {"input": _INPUT}),
("missing input", {"questions": _QUESTIONS}),
("numeric input", {"input": 5, "questions": _QUESTIONS}),
("assistant message input", {"input": [{"role": "assistant", "content": "hi"}], "questions": _QUESTIONS}),
("empty questions", {"input": _INPUT, "questions": []}),
("questions as a map", {"input": _INPUT, "questions": {"q": _PREDICATE}}),
("predicate without instructions", {"input": _INPUT, "questions": [{"type": "predicate", "name": "q"}]}),
(
"choice with one choice",
{"input": _INPUT, "questions": [{**_PREDICATE, "type": "choice", "choices": [{"value": "only"}]}]},
),
(
"score with one level",
{"input": _INPUT, "questions": [{**_PREDICATE, "type": "score", "levels": [{"label": "only"}]}]},
),
("unknown question type", {"input": _INPUT, "questions": [{**_PREDICATE, "type": "ranking"}]}),
("numeric safety_identifier", {"input": _INPUT, "questions": _QUESTIONS, "safety_identifier": 7}),
("list safety_identifier", {"input": _INPUT, "questions": _QUESTIONS, "safety_identifier": [_SAFETY_IDENTIFIER]}),
)
_IMAGE_INPUT: Final[list[JsonValue]] = [
{
"role": "user",
"content": [
{"type": "input_text", "text": _INPUT},
{"type": "input_image", "image_url": "data:image/png;base64,iVBORw0KGgo=", "detail": "auto"},
],
}
]
_OPENAI_IMAGE_MESSAGES: Final[list[JsonValue]] = [
{
"role": "user",
"type": "message",
"content": [
{"type": "input_text", "text": _INPUT},
{"type": "input_image", "image_url": "data:image/png;base64,iVBORw0KGgo=", "detail": "auto"},
],
}
]
_MESSAGE_LIST_INPUT: Final[list[JsonValue]] = [
{"role": "user", "content": _INPUT},
{"role": "user", "content": [{"type": "input_text", "text": _FOLLOW_UP}]},
]
def _response_cost(response: httpx.Response) -> float:
return float(response.headers["x-litellm-response-cost"]) if "x-litellm-response-cost" in response.headers else 0.0
def _provider_id(provider: _Provider) -> str:
return provider.name
def _number(value: JsonValue) -> float:
assert isinstance(value, (int, float)) and not isinstance(value, bool), value
return float(value)
def _expected_spend(cost_map_key: str | None) -> float:
if cost_map_key is None:
return 0.0
cost_map: Final = _JSON_OBJECT.validate_json(Path("model_prices_and_context_window.json").read_bytes())
prices: Final = object_value(cost_map[cost_map_key])
return _INPUT_TOKENS * _number(prices["input_cost_per_token"]) + _OUTPUT_TOKENS * _number(
prices["output_cost_per_token"]
)
def _register(scenario: Scenario, body: dict[str, JsonValue], *, status: int = 200) -> ScenarioHandle:
handle: Final = register_scenario(
f"decisions-{uuid.uuid4().hex[:12]}", JsonResponse(content_type="application/json", body=body, status=status)
)
scenario.cleanups.callback(delete_scenario, handle)
return handle
def _deployment(scenario: Scenario, handle: ScenarioHandle, provider: _Provider, *, drop_params: bool = False) -> str:
if drop_params:
return scenario.model(
model=provider.model, api_base=handle.api_base(), api_key=provider.api_key, drop_params=True
)
return scenario.model(model=provider.model, api_base=handle.api_base(), api_key=provider.api_key)
def _decide(gateway: Gateway, model: str, *, route: str = "/v1/decisions", **extra: JsonValue) -> httpx.Response:
return gateway.request("POST", route, {"model": model, "input": _INPUT, "questions": _QUESTIONS, **extra})
def _decide_in_system_one_format(gateway: Gateway, model: str, **extra: JsonValue) -> httpx.Response:
return gateway.request(
"POST", "/v1/systemone", {"model": model, "state": _INPUT, "questions": _SYSTEM_ONE_QUESTIONS, **extra}
)
def _observed_requests(gateway: Gateway) -> tuple[dict[str, JsonValue], ...]:
with httpx.Client(base_url=gateway.upstream_url, timeout=5, trust_env=False) as upstream:
observations: Final = _JSON_OBJECT.validate_json(upstream.get("/__observations").content)["requests"]
assert isinstance(observations, list), observations
return tuple(map(object_value, observations))
def _calls_to(requests: Sequence[dict[str, JsonValue]], handle: ScenarioHandle) -> list[dict[str, JsonValue]]:
return [request for request in requests if string_value(request["path"]).startswith(f"/{handle.scenario_id}/")]
def _upstream_calls(gateway: Gateway, handle: ScenarioHandle) -> list[dict[str, JsonValue]]:
return _calls_to(_observed_requests(gateway), handle)
def _spend_row(call_id: str) -> dict[str, JsonValue]:
rows: Final = eventually(lambda: read_rows(_SPEND_QUERY, (call_id,)), lambda found: len(found) == 1, seconds=70)
return rows[0]
def _assert_refused_as_invalid(gateway: Gateway, model: str, label: str, body: dict[str, JsonValue]) -> None:
response: Final = gateway.request("POST", "/v1/decisions", {"model": model, **body})
assert response.status_code == 400, (label, response.text)
assert "Invalid Decisions request" in response.text, (label, response.text)
def _assert_refused_for_safety_identifier(response: httpx.Response) -> None:
assert response.status_code == 400, response.text
assert "safety_identifier" in response.text and "drop_params" in response.text, response.text
def _drop_params_config(directory: Path, api_base: str) -> Path:
base: Final = _JSON_OBJECT.validate_python(yaml.safe_load(Path("tests/integration/proxy_config.yaml").read_text()))
config: Final = {
**base,
"litellm_settings": {**object_value(base["litellm_settings"]), "drop_params": True},
"model_list": [
{
"model_name": _DROPPING_DEPLOYMENT,
"litellm_params": {"model": _PERPLEXITY.model, "api_base": api_base, "api_key": _API_KEY},
}
],
}
path: Final = directory / "decisions-drop-params.yaml"
path.write_text(yaml.safe_dump(config))
return path
@pytest.mark.parametrize("provider", _PROVIDERS, ids=_provider_id)
def test_each_provider_gets_its_own_path_key_and_body_and_is_billed_from_the_cost_map(
gateway: Gateway, provider: _Provider
) -> None:
expected_spend: Final = _expected_spend(provider.cost_map_key)
with gateway.scenario() as scenario:
handle: Final = _register(scenario, provider.upstream_reply())
model: Final = _deployment(scenario, handle, provider)
response: Final = _decide(gateway, model)
assert response.status_code == 200, response.text
assert response.json() == provider.litellm_response()
assert response.headers["x-litellm-model-group"] == model
assert math.isclose(_response_cost(response), expected_spend, rel_tol=1e-9)
(call,) = _upstream_calls(gateway, handle)
assert call["path"] == f"/{handle.scenario_id}{provider.path}"
assert call["authorization"] == (f"Bearer {provider.api_key}" if provider.api_key else "")
assert call["body"] == provider.upstream_body()
row: Final = _spend_row(response.headers["x-litellm-call-id"])
assert (
row["status"],
row["call_type"],
row["custom_llm_provider"],
row["model_group"],
row["api_base"],
row["prompt_tokens"],
row["completion_tokens"],
) == (
"success",
"adecisions",
provider.name,
model,
f"{handle.api_base()}{provider.path}",
_INPUT_TOKENS,
_OUTPUT_TOKENS,
)
assert math.isclose(_number(row["spend"]), expected_spend, rel_tol=1e-9), row
def test_the_unversioned_alias_serves_the_same_request(gateway: Gateway) -> None:
with gateway.scenario() as scenario:
handle: Final = _register(scenario, _PERPLEXITY.upstream_reply())
model: Final = _deployment(scenario, handle, _PERPLEXITY)
response: Final = _decide(gateway, model, route="/decisions")
assert response.status_code == 200, response.text
assert response.json() == _PERPLEXITY.litellm_response()
(call,) = _upstream_calls(gateway, handle)
assert call["body"] == _PERPLEXITY.upstream_body()
def test_repeated_identical_requests_each_reach_the_upstream_and_are_each_billed(gateway: Gateway) -> None:
with gateway.scenario() as scenario:
handle: Final = _register(scenario, _PERPLEXITY.upstream_reply())
model: Final = _deployment(scenario, handle, _PERPLEXITY)
responses: Final = tuple(_decide(gateway, model) for _ in range(2))
assert [response.status_code for response in responses] == [200, 200], [r.text for r in responses]
call_ids: Final = tuple(response.headers["x-litellm-call-id"] for response in responses)
assert len(set(call_ids)) == 2, call_ids
assert len(_upstream_calls(gateway, handle)) == 2
for call_id in call_ids:
assert _spend_row(call_id)["status"] == "success"
async def test_sdk_sync_and_async_clients_send_the_same_request(gateway: Gateway) -> None:
with gateway.scenario() as scenario:
handle: Final = _register(scenario, _TYPESAFE.upstream_reply())
synchronous: Final = litellm.decisions(
model=_TYPESAFE.model, input=_INPUT, questions=_SDK_QUESTIONS, api_base=handle.api_base(), api_key=_API_KEY
)
asynchronous: Final = await litellm.adecisions(
model=_TYPESAFE.model, input=_INPUT, questions=_SDK_QUESTIONS, api_base=handle.api_base(), api_key=_API_KEY
)
for response in (synchronous, asynchronous):
assert response.model_dump(mode="json") == _TYPESAFE.litellm_response()
calls: Final = _upstream_calls(gateway, handle)
assert len(calls) == 2, calls
for call in calls:
assert call["path"] == f"/{handle.scenario_id}{_TYPESAFE.path}"
assert call["authorization"] == f"Bearer {_API_KEY}"
assert call["body"] == _TYPESAFE.upstream_body()
def test_gateway_only_fields_stay_at_the_gateway_and_tags_reach_the_spend_log(gateway: Gateway) -> None:
tag: Final = f"decisions-openai-format-{uuid.uuid4().hex[:8]}"
with gateway.scenario() as scenario:
handle: Final = _register(scenario, _PERPLEXITY.upstream_reply())
model: Final = _deployment(scenario, handle, _PERPLEXITY)
response: Final = _decide(
gateway, model, user="auditor", num_retries=0, temperature=0.2, metadata={"tags": [tag]}
)
assert response.status_code == 200, response.text
(call,) = _upstream_calls(gateway, handle)
assert call["body"] == _PERPLEXITY.upstream_body()
row: Final = _spend_row(response.headers["x-litellm-call-id"])
tags: Final = row["request_tags"]
assert isinstance(tags, list) and tag in tags, row
def test_invalid_bodies_are_refused_at_the_gateway_without_an_upstream_call(gateway: Gateway) -> None:
with gateway.scenario() as scenario:
handle: Final = _register(scenario, _PERPLEXITY.upstream_reply())
model: Final = _deployment(scenario, handle, _PERPLEXITY)
for label, body in _INVALID_BODIES:
_assert_refused_as_invalid(gateway, model, label, body)
assert _upstream_calls(gateway, handle) == []
@pytest.mark.parametrize("provider", _SYSTEM_ONE_PROVIDERS, ids=_provider_id)
def test_system_one_providers_refuse_images_at_the_gateway_without_an_upstream_call(
gateway: Gateway, provider: _Provider
) -> None:
with gateway.scenario() as scenario:
handle: Final = _register(scenario, provider.upstream_reply())
model: Final = _deployment(scenario, handle, provider)
response: Final = _decide(gateway, model, input=_IMAGE_INPUT)
assert response.status_code == 400, response.text
assert "input_image" in response.text
assert _upstream_calls(gateway, handle) == []
def test_openai_forwards_image_input_in_its_own_message_shape(gateway: Gateway) -> None:
with gateway.scenario() as scenario:
handle: Final = _register(scenario, _OPENAI.upstream_reply())
model: Final = _deployment(scenario, handle, _OPENAI)
response: Final = _decide(gateway, model, input=_IMAGE_INPUT)
assert response.status_code == 200, response.text
(call,) = _upstream_calls(gateway, handle)
assert call["body"] == {"model": _OPENAI.body_model, "input": _OPENAI_IMAGE_MESSAGES, "questions": _QUESTIONS}
def test_a_message_list_input_reaches_a_system_one_provider_as_its_flattened_text(gateway: Gateway) -> None:
with gateway.scenario() as scenario:
handle: Final = _register(scenario, _PERPLEXITY.upstream_reply())
model: Final = _deployment(scenario, handle, _PERPLEXITY)
response: Final = _decide(gateway, model, input=_MESSAGE_LIST_INPUT)
assert response.status_code == 200, response.text
assert response.json() == _PERPLEXITY.litellm_response()
(call,) = _upstream_calls(gateway, handle)
assert call["body"] == {
"model": _PERPLEXITY.body_model,
"state": f"{_INPUT}\n\n{_FOLLOW_UP}",
"questions": _SYSTEM_ONE_QUESTIONS,
}
row: Final = _spend_row(response.headers["x-litellm-call-id"])
assert (row["status"], row["call_type"], row["prompt_tokens"], row["completion_tokens"]) == (
"success",
"adecisions",
_INPUT_TOKENS,
_OUTPUT_TOKENS,
)
@pytest.mark.parametrize("provider", _SYSTEM_ONE_PROVIDERS, ids=_provider_id)
def test_safety_identifier_is_refused_by_system_one_providers_unless_the_deployment_drops_params(
gateway: Gateway, provider: _Provider
) -> None:
with gateway.scenario() as scenario:
handle: Final = _register(scenario, provider.upstream_reply())
strict: Final = _deployment(scenario, handle, provider)
refused: Final = _decide(gateway, strict, safety_identifier=_SAFETY_IDENTIFIER)
_assert_refused_for_safety_identifier(refused)
assert _upstream_calls(gateway, handle) == []
dropping: Final = _deployment(scenario, handle, provider, drop_params=True)
accepted: Final = _decide(gateway, dropping, safety_identifier=_SAFETY_IDENTIFIER)
assert accepted.status_code == 200, accepted.text
assert accepted.json() == provider.litellm_response()
(call,) = _upstream_calls(gateway, handle)
assert call["body"] == provider.upstream_body()
row: Final = _spend_row(accepted.headers["x-litellm-call-id"])
assert (row["status"], row["call_type"], row["model_group"]) == ("success", "adecisions", dropping)
@pytest.mark.parametrize("value", ("", "x" * 5000), ids=("empty", "5kb"))
def test_every_string_safety_identifier_is_refused_or_dropped_like_the_usual_one(gateway: Gateway, value: str) -> None:
with gateway.scenario() as scenario:
handle: Final = _register(scenario, _PERPLEXITY.upstream_reply())
strict: Final = _deployment(scenario, handle, _PERPLEXITY)
_assert_refused_for_safety_identifier(_decide(gateway, strict, safety_identifier=value))
dropping: Final = _deployment(scenario, handle, _PERPLEXITY, drop_params=True)
accepted: Final = _decide(gateway, dropping, safety_identifier=value)
assert accepted.status_code == 200, accepted.text
(call,) = _upstream_calls(gateway, handle)
assert call["body"] == _PERPLEXITY.upstream_body()
def test_a_request_body_drop_params_drops_the_safety_identifier_like_chat(gateway: Gateway) -> None:
with gateway.scenario() as scenario:
handle: Final = _register(scenario, _PERPLEXITY.upstream_reply())
strict: Final = _deployment(scenario, handle, _PERPLEXITY)
response: Final = _decide(gateway, strict, safety_identifier=_SAFETY_IDENTIFIER, drop_params=True)
assert response.status_code == 200, response.text
assert response.json() == _PERPLEXITY.litellm_response()
(call,) = _upstream_calls(gateway, handle)
assert call["body"] == _PERPLEXITY.upstream_body()
def test_openai_keeps_the_safety_identifier_on_the_wire_without_drop_params(gateway: Gateway) -> None:
with gateway.scenario() as scenario:
handle: Final = _register(scenario, _OPENAI.upstream_reply())
model: Final = _deployment(scenario, handle, _OPENAI)
response: Final = _decide(gateway, model, safety_identifier=_SAFETY_IDENTIFIER)
assert response.status_code == 200, response.text
assert response.json() == _OPENAI.litellm_response()
(call,) = _upstream_calls(gateway, handle)
assert call["body"] == {**_OPENAI.upstream_body(), "safety_identifier": _SAFETY_IDENTIFIER}
def test_a_system_one_format_safety_identifier_is_refused_unless_the_deployment_drops_params(gateway: Gateway) -> None:
with gateway.scenario() as scenario:
handle: Final = _register(scenario, _PERPLEXITY.upstream_reply())
strict: Final = _deployment(scenario, handle, _PERPLEXITY)
refused: Final = _decide_in_system_one_format(gateway, strict, safety_identifier=_SAFETY_IDENTIFIER)
_assert_refused_for_safety_identifier(refused)
assert _upstream_calls(gateway, handle) == []
dropping: Final = _deployment(scenario, handle, _PERPLEXITY, drop_params=True)
accepted: Final = _decide_in_system_one_format(gateway, dropping, safety_identifier=_SAFETY_IDENTIFIER)
assert accepted.status_code == 200, accepted.text
assert accepted.json()["answers"] == _SYSTEM_ONE_ANSWERS
(call,) = _upstream_calls(gateway, handle)
assert call["body"] == _PERPLEXITY.upstream_body()
def test_openai_keeps_a_system_one_format_safety_identifier_on_the_wire(gateway: Gateway) -> None:
with gateway.scenario() as scenario:
handle: Final = _register(scenario, _OPENAI.upstream_reply())
model: Final = _deployment(scenario, handle, _OPENAI)
response: Final = _decide_in_system_one_format(gateway, model, safety_identifier=_SAFETY_IDENTIFIER)
assert response.status_code == 200, response.text
(call,) = _upstream_calls(gateway, handle)
assert call["body"] == {**_OPENAI.upstream_body(), "safety_identifier": _SAFETY_IDENTIFIER}
@pytest.mark.parametrize("decide", (_decide, _decide_in_system_one_format), ids=("openai_format", "system_one_format"))
@pytest.mark.parametrize("value", (7, [_SAFETY_IDENTIFIER]), ids=("numeric", "list"))
def test_a_non_string_safety_identifier_is_refused_as_invalid_unless_the_deployment_drops_params(
gateway: Gateway, value: JsonValue, decide: Callable[..., httpx.Response]
) -> None:
with gateway.scenario() as scenario:
handle: Final = _register(scenario, _OPENAI.upstream_reply())
strict: Final = _deployment(scenario, handle, _OPENAI)
refused: Final = decide(gateway, strict, safety_identifier=value)
assert refused.status_code == 400, refused.text
assert "Invalid Decisions request" in refused.text, refused.text
assert _upstream_calls(gateway, handle) == []
dropping: Final = _deployment(scenario, handle, _OPENAI, drop_params=True)
accepted: Final = decide(gateway, dropping, safety_identifier=value)
assert accepted.status_code == 200, accepted.text
(call,) = _upstream_calls(gateway, handle)
assert call["body"] == _OPENAI.upstream_body()
def test_litellm_settings_drop_params_drops_the_safety_identifier_for_a_strict_deployment(
gateway: Gateway, tmp_path: Path
) -> None:
with gateway.scenario() as scenario:
handle: Final = _register(scenario, _PERPLEXITY.upstream_reply())
config: Final = _drop_params_config(tmp_path, handle.api_base())
with owned_proxy_process(gateway, tmp_path, {}, config=config) as owned:
response: Final = _decide(owned.gateway, _DROPPING_DEPLOYMENT, safety_identifier=_SAFETY_IDENTIFIER)
assert response.status_code == 200, response.text
assert response.json() == _PERPLEXITY.litellm_response()
(call,) = _upstream_calls(gateway, handle)
assert call["body"] == _PERPLEXITY.upstream_body()

View file

@ -6,6 +6,103 @@ from integration.translation.case import TranslationTestCase
"""
CLEF_TEST_CASE: Final = TranslationTestCase(
scenario="basic",
litellm_endpoint="/v1/decisions",
litellm_request={
"model": "cloudflare/@cf/cloudflare/clef",
"input": "Ticket (billing): The export job hangs at 99% and never finishes",
"questions": [
{"type": "predicate", "name": "defect", "instructions": "Is this a defect?"},
{
"type": "choice",
"name": "severity",
"instructions": "How severe is it?",
"choices": [
{"value": "low", "description": "cosmetic"},
{"value": "high", "description": "blocks users"},
],
},
{
"type": "score",
"name": "confidence",
"instructions": "How sure are you?",
"levels": [{"label": "unsure"}, {"label": "sure"}],
},
],
"cache": {"no-cache": True},
},
expected_provider_endpoint="/ai/run/@cf/cloudflare/clef",
expected_provider_headers={"authorization": "Bearer synthetic-cloudflare-key", "content-type": "application/json"},
expected_provider_request={
"model": "clef",
"state": "Ticket (billing): The export job hangs at 99% and never finishes",
"questions": {
"defect": {"type": "noul", "instructions": "Is this a defect?"},
"severity": {
"type": "choice",
"instructions": "How severe is it?",
"criteria": {"low": "cosmetic", "high": "blocks users"},
},
"confidence": {"type": "score", "instructions": "How sure are you?", "criteria": ["unsure", "sure"]},
},
},
mock_provider_response={
"result": {
"model": "clef",
"answers": {
"defect": {"type": "noul", "noul": 0.9345},
"severity": {
"type": "choice",
"choice": "high",
"confidence": 0.8067,
"probabilities": {"low": 0.0509, "high": 0.9491},
},
"confidence": {
"type": "score",
"score": 0.9036,
"confidence": 0.6515,
"legend": {"0": "unsure", "1": "sure"},
"probabilities": {"0": 0.0964, "1": 0.9036},
},
},
"usage": {"input_tokens": 290, "output_tokens": 0},
},
"success": True,
"errors": [],
"messages": [],
},
expected_litellm_response={
"model": "clef",
"answers": [
{"type": "predicate", "name": "defect", "probability": 0.9345},
{
"type": "choice",
"name": "severity",
"choice": "high",
"probabilities": [{"value": "low", "probability": 0.0509}, {"value": "high", "probability": 0.9491}],
"confidence": 0.8067,
},
{
"type": "score",
"name": "confidence",
"score": 0.9036,
"probabilities": [
{"value": 0, "label": "unsure", "probability": 0.0964},
{"value": 1, "label": "sure", "probability": 0.9036},
],
"confidence": 0.6515,
},
],
"usage": {
"input_tokens": 290,
"input_tokens_details": {"cached_tokens": 0, "cache_write_tokens": 0},
"output_tokens": 0,
"output_tokens_details": {"reasoning_tokens": 0},
"total_tokens": 290,
},
},
)
CLEF_SYSTEMONE_TEST_CASE: Final = TranslationTestCase(
scenario="systemone",
litellm_endpoint="/v1/systemone",
litellm_request={
"model": "cloudflare/@cf/cloudflare/clef",

View file

@ -6,6 +6,100 @@ from integration.translation.case import TranslationTestCase
"""
TYPESAFE_JEV_1_13_TEST_CASE: Final = TranslationTestCase(
scenario="basic",
litellm_endpoint="/v1/decisions",
litellm_request={
"model": "openrouter/typesafe/jev-1.13",
"input": "Ticket (billing): The export job hangs at 99% and never finishes",
"questions": [
{"type": "predicate", "name": "defect", "instructions": "Is this a defect?"},
{
"type": "choice",
"name": "severity",
"instructions": "How severe is it?",
"choices": [
{"value": "low", "description": "cosmetic"},
{"value": "high", "description": "blocks users"},
],
},
{
"type": "score",
"name": "confidence",
"instructions": "How sure are you?",
"levels": [{"label": "unsure"}, {"label": "sure"}],
},
],
"cache": {"no-cache": True},
},
expected_provider_endpoint="/alpha/decisions",
expected_provider_headers={"authorization": "Bearer synthetic-openrouter-key", "content-type": "application/json"},
expected_provider_request={
"model": "typesafe/jev-1.13",
"state": "Ticket (billing): The export job hangs at 99% and never finishes",
"questions": {
"defect": {"type": "noul", "instructions": "Is this a defect?"},
"severity": {
"type": "choice",
"instructions": "How severe is it?",
"criteria": {"low": "cosmetic", "high": "blocks users"},
},
"confidence": {"type": "score", "instructions": "How sure are you?", "criteria": ["unsure", "sure"]},
},
},
mock_provider_response={
"model": "typesafe/jev-1.13-20260917",
"answers": {
"defect": {"type": "noul", "noul": 0.81},
"severity": {
"type": "choice",
"choice": "high",
"confidence": 0.99,
"probabilities": {"low": 0.01, "high": 0.99},
},
"confidence": {
"type": "score",
"score": 0.5,
"confidence": 0,
"legend": {"0": "unsure", "1": "sure"},
"probabilities": {"0": 0.5, "1": 0.5},
},
},
"usage": {"input_tokens": 377, "output_tokens": 62, "cost": 1.5834e-05},
"id": "gen-dec-1791323839-GA15kY0nt34oiJ7srfki",
"provider": "TypeSafe",
},
expected_litellm_response={
"model": "typesafe/jev-1.13-20260917",
"answers": [
{"type": "predicate", "name": "defect", "probability": 0.81},
{
"type": "choice",
"name": "severity",
"choice": "high",
"probabilities": [{"value": "low", "probability": 0.01}, {"value": "high", "probability": 0.99}],
"confidence": 0.99,
},
{
"type": "score",
"name": "confidence",
"score": 0.5,
"probabilities": [
{"value": 0, "label": "unsure", "probability": 0.5},
{"value": 1, "label": "sure", "probability": 0.5},
],
"confidence": 0,
},
],
"usage": {
"input_tokens": 377,
"input_tokens_details": {"cached_tokens": 0, "cache_write_tokens": 0},
"output_tokens": 62,
"output_tokens_details": {"reasoning_tokens": 0},
"total_tokens": 439,
},
},
)
TYPESAFE_JEV_1_13_SYSTEMONE_TEST_CASE: Final = TranslationTestCase(
scenario="systemone",
litellm_endpoint="/v1/systemone",
litellm_request={
"model": "openrouter/typesafe/jev-1.13",

View file

@ -6,6 +6,101 @@ from integration.translation.case import TranslationTestCase
"""
PPLX_DECIDER_V1_27B_TEST_CASE: Final = TranslationTestCase(
scenario="basic",
litellm_endpoint="/v1/decisions",
litellm_request={
"model": "perplexity/pplx-decider-v1-27b",
"input": "Ticket (billing): The export job hangs at 99% and never finishes",
"questions": [
{"type": "predicate", "name": "defect", "instructions": "Is this a defect?"},
{
"type": "choice",
"name": "severity",
"instructions": "How severe is it?",
"choices": [
{"value": "low", "description": "cosmetic"},
{"value": "high", "description": "blocks users"},
],
},
{
"type": "score",
"name": "confidence",
"instructions": "How sure are you?",
"levels": [{"label": "unsure"}, {"label": "sure"}],
},
],
"cache": {"no-cache": True},
},
expected_provider_endpoint="/v1/decisions",
expected_provider_headers={"authorization": "Bearer synthetic-perplexity-key", "content-type": "application/json"},
expected_provider_request={
"model": "pplx-decider-v1-27b",
"state": "Ticket (billing): The export job hangs at 99% and never finishes",
"questions": {
"defect": {"type": "noul", "instructions": "Is this a defect?"},
"severity": {
"type": "choice",
"instructions": "How severe is it?",
"criteria": {"low": "cosmetic", "high": "blocks users"},
},
"confidence": {"type": "score", "instructions": "How sure are you?", "criteria": ["unsure", "sure"]},
},
},
mock_provider_response={
"model": "pplx-decider-v1-27b",
"answers": {
"defect": {"type": "noul", "noul": 0.9989100737587077},
"severity": {
"type": "choice",
"choice": "high",
"confidence": 0.9964631215356778,
"probabilities": {"low": 0.0017684392321610232, "high": 0.9982315607678389},
},
"confidence": {
"type": "score",
"score": 0.07367392327139817,
"confidence": 0.8526521534572037,
"legend": {"0": "unsure", "1": "sure"},
"probabilities": {"0": 0.9263260767286018, "1": 0.07367392327139817},
},
},
"usage": {"input_tokens": 318, "output_tokens": 3},
},
expected_litellm_response={
"model": "pplx-decider-v1-27b",
"answers": [
{"type": "predicate", "name": "defect", "probability": 0.9989100737587077},
{
"type": "choice",
"name": "severity",
"choice": "high",
"probabilities": [
{"value": "low", "probability": 0.0017684392321610232},
{"value": "high", "probability": 0.9982315607678389},
],
"confidence": 0.9964631215356778,
},
{
"type": "score",
"name": "confidence",
"score": 0.07367392327139817,
"probabilities": [
{"value": 0, "label": "unsure", "probability": 0.9263260767286018},
{"value": 1, "label": "sure", "probability": 0.07367392327139817},
],
"confidence": 0.8526521534572037,
},
],
"usage": {
"input_tokens": 318,
"input_tokens_details": {"cached_tokens": 0, "cache_write_tokens": 0},
"output_tokens": 3,
"output_tokens_details": {"reasoning_tokens": 0},
"total_tokens": 321,
},
},
)
PPLX_DECIDER_V1_27B_SYSTEMONE_TEST_CASE: Final = TranslationTestCase(
scenario="systemone",
litellm_endpoint="/v1/systemone",
litellm_request={
"model": "perplexity/pplx-decider-v1-27b",

View file

@ -6,6 +6,40 @@ from integration.translation.case import TranslationTestCase
"""
STRANDS_DECIDER_2B_HOBSON_V19_TEST_CASE: Final = TranslationTestCase(
scenario="basic",
litellm_endpoint="/v1/decisions",
litellm_request={
"model": "strands_decider/strands-decider-2B-hobson-v19",
"input": "Help! My payouts have been failing for 3 days!",
"questions": [{"type": "predicate", "name": "is_urgent", "instructions": "Does this convey urgency?"}],
"cache": {"no-cache": True},
},
expected_provider_endpoint="/v1/systemone",
expected_provider_headers={"content-type": "application/json"},
expected_provider_request={
"model": "strands-decider-2B-hobson-v19",
"state": "Help! My payouts have been failing for 3 days!",
"questions": {"is_urgent": {"type": "noul", "instructions": "Does this convey urgency?"}},
},
mock_provider_response={
"model": "strands-decider-2B-hobson-v19",
"answers": {"is_urgent": {"type": "noul", "noul": 0.8277}},
"usage": {"input_tokens": 86, "output_tokens": 1},
"latency_ms": 140.03,
},
expected_litellm_response={
"model": "strands-decider-2B-hobson-v19",
"answers": [{"type": "predicate", "name": "is_urgent", "probability": 0.8277}],
"usage": {
"input_tokens": 86,
"input_tokens_details": {"cached_tokens": 0, "cache_write_tokens": 0},
"output_tokens": 1,
"output_tokens_details": {"reasoning_tokens": 0},
"total_tokens": 87,
},
},
)
STRANDS_DECIDER_2B_HOBSON_V19_SYSTEMONE_TEST_CASE: Final = TranslationTestCase(
scenario="systemone",
litellm_endpoint="/v1/systemone",
litellm_request={
"model": "strands_decider/strands-decider-2B-hobson-v19",

View file

@ -6,6 +6,98 @@ from integration.translation.case import TranslationTestCase
"""
JEV_1_13_0_TEST_CASE: Final = TranslationTestCase(
scenario="basic",
litellm_endpoint="/v1/decisions",
litellm_request={
"model": "typesafe/jev-1.13.0",
"input": "Ticket (billing): The export job hangs at 99% and never finishes",
"questions": [
{"type": "predicate", "name": "defect", "instructions": "Is this a defect?"},
{
"type": "choice",
"name": "severity",
"instructions": "How severe is it?",
"choices": [
{"value": "low", "description": "cosmetic"},
{"value": "high", "description": "blocks users"},
],
},
{
"type": "score",
"name": "confidence",
"instructions": "How sure are you?",
"levels": [{"label": "unsure"}, {"label": "sure"}],
},
],
"cache": {"no-cache": True},
},
expected_provider_endpoint="/v1/systemone",
expected_provider_headers={"authorization": "Bearer synthetic-typesafe-key", "content-type": "application/json"},
expected_provider_request={
"model": "jev-1.13.0",
"state": "Ticket (billing): The export job hangs at 99% and never finishes",
"questions": {
"defect": {"type": "noul", "instructions": "Is this a defect?"},
"severity": {
"type": "choice",
"instructions": "How severe is it?",
"criteria": {"low": "cosmetic", "high": "blocks users"},
},
"confidence": {"type": "score", "instructions": "How sure are you?", "criteria": ["unsure", "sure"]},
},
},
mock_provider_response={
"model": "jev-1.13.0",
"answers": {
"defect": {"type": "noul", "noul": 0.78},
"severity": {
"type": "choice",
"choice": "high",
"confidence": 0.99,
"probabilities": {"low": 0.01, "high": 0.99},
},
"confidence": {
"type": "score",
"score": 0.51,
"confidence": 0.03,
"legend": {"0": "unsure", "1": "sure"},
"probabilities": {"0": 0.49, "1": 0.51},
},
},
"usage": {"input_tokens": 377, "output_tokens": 62},
},
expected_litellm_response={
"model": "jev-1.13.0",
"answers": [
{"type": "predicate", "name": "defect", "probability": 0.78},
{
"type": "choice",
"name": "severity",
"choice": "high",
"probabilities": [{"value": "low", "probability": 0.01}, {"value": "high", "probability": 0.99}],
"confidence": 0.99,
},
{
"type": "score",
"name": "confidence",
"score": 0.51,
"probabilities": [
{"value": 0, "label": "unsure", "probability": 0.49},
{"value": 1, "label": "sure", "probability": 0.51},
],
"confidence": 0.03,
},
],
"usage": {
"input_tokens": 377,
"input_tokens_details": {"cached_tokens": 0, "cache_write_tokens": 0},
"output_tokens": 62,
"output_tokens_details": {"reasoning_tokens": 0},
"total_tokens": 439,
},
},
)
JEV_1_13_0_SYSTEMONE_TEST_CASE: Final = TranslationTestCase(
scenario="systemone",
litellm_endpoint="/v1/systemone",
litellm_request={
"model": "typesafe/jev-1.13.0",

View file

@ -2,10 +2,10 @@ import pytest
from integration._support.client import Gateway
from integration._support.provider import SharedProvider
from integration.translation.case import TranslationTestCase
from integration.translation.decisions.bases.cloudflare import CLEF_TEST_CASE
from integration.translation.decisions.bases.cloudflare import CLEF_TEST_CASE, CLEF_SYSTEMONE_TEST_CASE
from integration.translation.runner import assert_translation
@pytest.mark.parametrize("case", [CLEF_TEST_CASE], ids=lambda case: case.id)
@pytest.mark.parametrize("case", [CLEF_TEST_CASE, CLEF_SYSTEMONE_TEST_CASE], ids=lambda case: case.id)
def test_decisions_basic_cloudflare(case: TranslationTestCase, gateway: Gateway, provider: SharedProvider) -> None:
assert_translation(case, gateway, provider)

View file

@ -2,10 +2,15 @@ import pytest
from integration._support.client import Gateway
from integration._support.provider import SharedProvider
from integration.translation.case import TranslationTestCase
from integration.translation.decisions.bases.openrouter import TYPESAFE_JEV_1_13_TEST_CASE
from integration.translation.decisions.bases.openrouter import (
TYPESAFE_JEV_1_13_TEST_CASE,
TYPESAFE_JEV_1_13_SYSTEMONE_TEST_CASE,
)
from integration.translation.runner import assert_translation
@pytest.mark.parametrize("case", [TYPESAFE_JEV_1_13_TEST_CASE], ids=lambda case: case.id)
@pytest.mark.parametrize(
"case", [TYPESAFE_JEV_1_13_TEST_CASE, TYPESAFE_JEV_1_13_SYSTEMONE_TEST_CASE], ids=lambda case: case.id
)
def test_decisions_basic_openrouter(case: TranslationTestCase, gateway: Gateway, provider: SharedProvider) -> None:
assert_translation(case, gateway, provider)

View file

@ -2,10 +2,15 @@ import pytest
from integration._support.client import Gateway
from integration._support.provider import SharedProvider
from integration.translation.case import TranslationTestCase
from integration.translation.decisions.bases.perplexity import PPLX_DECIDER_V1_27B_TEST_CASE
from integration.translation.decisions.bases.perplexity import (
PPLX_DECIDER_V1_27B_TEST_CASE,
PPLX_DECIDER_V1_27B_SYSTEMONE_TEST_CASE,
)
from integration.translation.runner import assert_translation
@pytest.mark.parametrize("case", [PPLX_DECIDER_V1_27B_TEST_CASE], ids=lambda case: case.id)
@pytest.mark.parametrize(
"case", [PPLX_DECIDER_V1_27B_TEST_CASE, PPLX_DECIDER_V1_27B_SYSTEMONE_TEST_CASE], ids=lambda case: case.id
)
def test_decisions_basic_perplexity(case: TranslationTestCase, gateway: Gateway, provider: SharedProvider) -> None:
assert_translation(case, gateway, provider)

View file

@ -2,10 +2,17 @@ import pytest
from integration._support.client import Gateway
from integration._support.provider import SharedProvider
from integration.translation.case import TranslationTestCase
from integration.translation.decisions.bases.strands_decider import STRANDS_DECIDER_2B_HOBSON_V19_TEST_CASE
from integration.translation.decisions.bases.strands_decider import (
STRANDS_DECIDER_2B_HOBSON_V19_TEST_CASE,
STRANDS_DECIDER_2B_HOBSON_V19_SYSTEMONE_TEST_CASE,
)
from integration.translation.runner import assert_translation
@pytest.mark.parametrize("case", [STRANDS_DECIDER_2B_HOBSON_V19_TEST_CASE], ids=lambda case: case.id)
@pytest.mark.parametrize(
"case",
[STRANDS_DECIDER_2B_HOBSON_V19_TEST_CASE, STRANDS_DECIDER_2B_HOBSON_V19_SYSTEMONE_TEST_CASE],
ids=lambda case: case.id,
)
def test_decisions_basic_strands_decider(case: TranslationTestCase, gateway: Gateway, provider: SharedProvider) -> None:
assert_translation(case, gateway, provider)

View file

@ -2,10 +2,10 @@ import pytest
from integration._support.client import Gateway
from integration._support.provider import SharedProvider
from integration.translation.case import TranslationTestCase
from integration.translation.decisions.bases.typesafe import JEV_1_13_0_TEST_CASE
from integration.translation.decisions.bases.typesafe import JEV_1_13_0_TEST_CASE, JEV_1_13_0_SYSTEMONE_TEST_CASE
from integration.translation.runner import assert_translation
@pytest.mark.parametrize("case", [JEV_1_13_0_TEST_CASE], ids=lambda case: case.id)
@pytest.mark.parametrize("case", [JEV_1_13_0_TEST_CASE, JEV_1_13_0_SYSTEMONE_TEST_CASE], ids=lambda case: case.id)
def test_decisions_basic_typesafe(case: TranslationTestCase, gateway: Gateway, provider: SharedProvider) -> None:
assert_translation(case, gateway, provider)

View file

@ -1262,6 +1262,141 @@ async def test_openrouter_decisions_uses_provider_reported_cost_without_cost_map
assert get_response_cost_from_hidden_params(response.hidden_params) == cost
_SAFETY_IDENTIFIER: Final = "end-user-7"
_PREDICATE_QUESTIONS: Final[tuple[Mapping[str, object], ...]] = (
{"type": "predicate", "name": "is_defect", "instructions": "Is this a defect?"},
)
_PREDICATE_REQUESTS: Final[tuple[Mapping[str, object], ...]] = (
MappingProxyType({"input": "review", "questions": _PREDICATE_QUESTIONS}),
MappingProxyType(
{"state": "review", "questions": {"is_defect": {"type": "noul", "instructions": "Is this a defect?"}}}
),
)
_PREDICATE_REQUEST_IDS: Final = ("input_format", "state_format")
@pytest.mark.parametrize("request_kwargs", _PREDICATE_REQUESTS, ids=_PREDICATE_REQUEST_IDS)
def test_safety_identifier_is_refused_before_http_when_the_provider_cannot_take_it(
request_kwargs: Mapping[str, object], respx_mock: respx.MockRouter, monkeypatch: pytest.MonkeyPatch
) -> None:
monkeypatch.setattr(litellm, "drop_params", False)
route: Final = respx_mock.post("https://api.perplexity.ai/v1/decisions").respond(json=_RESPONSE)
with pytest.raises(litellm.UnsupportedParamsError, match=r"safety_identifier.*drop_params") as caught:
litellm.decisions(
model="perplexity/pplx-decider-v1-27b",
safety_identifier=_SAFETY_IDENTIFIER,
api_key="caller-key",
**request_kwargs,
)
assert caught.value.status_code == 400
assert not route.called
@pytest.mark.asyncio
@pytest.mark.parametrize("scope", ("call", "global"))
async def test_safety_identifier_is_dropped_from_the_wire_under_drop_params(
scope: str, respx_mock: respx.MockRouter, monkeypatch: pytest.MonkeyPatch
) -> None:
monkeypatch.setattr(litellm, "drop_params", scope == "global")
route: Final = respx_mock.post("https://api.perplexity.ai/v1/decisions").respond(json=_RESPONSE)
response: Final = await litellm.adecisions(
model="perplexity/pplx-decider-v1-27b",
input="review",
questions=_PREDICATE_QUESTIONS,
safety_identifier=_SAFETY_IDENTIFIER,
api_key="caller-key",
**({"drop_params": True} if scope == "call" else {}),
)
assert route.called
assert json.loads(respx_mock.calls[0].request.content) == {
"model": "pplx-decider-v1-27b",
"state": "review",
"questions": {"is_defect": {"type": "noul", "instructions": "Is this a defect?"}},
}
assert isinstance(response, OpenAIDecisionResponse)
assert [answer.model_dump(mode="json") for answer in response.answers] == [
{"type": "predicate", "name": "is_defect", "probability": 0.9}
]
@pytest.mark.asyncio
@pytest.mark.parametrize("request_kwargs", _PREDICATE_REQUESTS, ids=_PREDICATE_REQUEST_IDS)
@pytest.mark.parametrize("drop_params", (False, True), ids=("strict", "drop_params"))
async def test_safety_identifier_reaches_openai_whether_or_not_params_are_dropped(
drop_params: bool,
request_kwargs: Mapping[str, object],
respx_mock: respx.MockRouter,
monkeypatch: pytest.MonkeyPatch,
) -> None:
monkeypatch.setattr(litellm, "drop_params", drop_params)
monkeypatch.setattr(litellm, "api_base", None)
monkeypatch.delenv("OPENAI_BASE_URL", raising=False)
monkeypatch.delenv("OPENAI_API_BASE", raising=False)
route: Final = respx_mock.post("https://api.openai.com/v1/decisions").respond(json=_OPENAI_RESPONSE)
await litellm.adecisions(
model="openai/gpt-6-luna",
safety_identifier=_SAFETY_IDENTIFIER,
api_key="caller-key",
**request_kwargs,
)
assert route.called
assert json.loads(respx_mock.calls[0].request.content) == {
"model": "gpt-6-luna",
"input": "review",
"questions": [{"type": "predicate", "name": "is_defect", "instructions": "Is this a defect?"}],
"safety_identifier": _SAFETY_IDENTIFIER,
}
@pytest.mark.asyncio
@pytest.mark.parametrize("request_kwargs", _PREDICATE_REQUESTS, ids=_PREDICATE_REQUEST_IDS)
async def test_non_string_safety_identifier_is_rejected_before_http(
request_kwargs: Mapping[str, object], respx_mock: respx.MockRouter, monkeypatch: pytest.MonkeyPatch
) -> None:
monkeypatch.setattr(litellm, "drop_params", False)
route: Final = respx_mock.post("https://api.openai.com/v1/decisions").respond(json=_OPENAI_RESPONSE)
invalid_safety_identifier: Final[Mapping[str, object]] = MappingProxyType({"safety_identifier": 7})
with pytest.raises(litellm.BadRequestError, match=r"(?s)Invalid Decisions request.*safety_identifier"):
await litellm.adecisions(
model="openai/gpt-6-luna", api_key="caller-key", **request_kwargs, **invalid_safety_identifier
)
assert not route.called
@pytest.mark.asyncio
@pytest.mark.parametrize("request_kwargs", _PREDICATE_REQUESTS, ids=_PREDICATE_REQUEST_IDS)
async def test_non_string_safety_identifier_is_dropped_from_the_wire_under_drop_params(
request_kwargs: Mapping[str, object], respx_mock: respx.MockRouter, monkeypatch: pytest.MonkeyPatch
) -> None:
monkeypatch.setattr(litellm, "drop_params", False)
monkeypatch.setattr(litellm, "api_base", None)
monkeypatch.delenv("OPENAI_BASE_URL", raising=False)
monkeypatch.delenv("OPENAI_API_BASE", raising=False)
route: Final = respx_mock.post("https://api.openai.com/v1/decisions").respond(json=_OPENAI_RESPONSE)
dropped_safety_identifier: Final[Mapping[str, object]] = MappingProxyType(
{"safety_identifier": 7, "drop_params": True}
)
await litellm.adecisions(
model="openai/gpt-6-luna", api_key="caller-key", **request_kwargs, **dropped_safety_identifier
)
assert route.called
assert json.loads(respx_mock.calls[0].request.content) == {
"model": "gpt-6-luna",
"input": "review",
"questions": [{"type": "predicate", "name": "is_defect", "instructions": "Is this a defect?"}],
}
@pytest.mark.asyncio
@pytest.mark.parametrize(
"api_base",

View file

@ -575,6 +575,68 @@ def test_openai_format_decisions_reach_an_openai_deployment_unchanged_including_
assert float(response.headers["x-litellm-response-cost"]) == pytest.approx(expected_cost)
@pytest.mark.parametrize("safety_identifier", (7, ["end-user-1"]), ids=("numeric", "list"))
@pytest.mark.parametrize("deployment_drops_params", (False, True), ids=("strict", "drop_params"))
def test_a_non_string_safety_identifier_is_refused_unless_the_deployment_drops_params(
client: TestClient,
monkeypatch: pytest.MonkeyPatch,
respx_mock: respx.MockRouter,
safety_identifier: object,
deployment_drops_params: bool,
) -> None:
monkeypatch.setattr(litellm, "drop_params", False)
monkeypatch.setattr(litellm, "api_base", None)
monkeypatch.delenv("OPENAI_BASE_URL", raising=False)
monkeypatch.delenv("OPENAI_API_BASE", raising=False)
monkeypatch.setattr(
litellm.proxy.proxy_server,
"llm_router",
litellm.Router(
model_list=[
{
"model_name": "decider",
"litellm_params": {
"model": "openai/gpt-6-luna",
"api_key": "k",
"drop_params": deployment_drops_params,
},
}
]
),
)
upstream: Final = respx_mock.post("https://api.openai.com/v1/decisions").respond(
json={
"model": "gpt-6-luna",
"answers": _OPENAI_FORMAT_ANSWERS[:1],
"usage": {
"input_tokens": _INPUT_TOKENS,
"output_tokens": _OUTPUT_TOKENS,
"total_tokens": _INPUT_TOKENS + _OUTPUT_TOKENS,
},
}
)
request_body: Final = {
"model": "decider",
"input": "The package arrived with a broken screen.",
"questions": _OPENAI_FORMAT_REQUEST["questions"][:1],
"safety_identifier": safety_identifier,
}
response: Final = client.post("/v1/decisions", json=request_body)
if not deployment_drops_params:
assert response.status_code == 400, response.text
assert "safety_identifier" in response.json()["error"]["message"]
assert not upstream.called
return
assert response.status_code == 200, response.text
assert json.loads(upstream.calls[0].request.content) == {
"model": "gpt-6-luna",
"input": "The package arrived with a broken screen.",
"questions": _OPENAI_FORMAT_REQUEST["questions"][:1],
}
def _decisions_feature() -> LazyFeature:
return next(feature for feature in LAZY_FEATURES if feature.name == "decisions")