From 546402c98c7ecea5f32d263fadd3c95bd9b934ff Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Fri, 9 Oct 2026 19:14:48 -0700 Subject: [PATCH] fix(decisions)!: refuse safety_identifier on providers that cannot take it unless drop_params drops it (#44955) * feat(decisions): add the OpenAI Decisions spec types and the System One translation Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * refactor(decisions): share one DecisionsModel config and require model and usage on responses Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * refactor(decisions): rename the shared pydantic parent to DecisionsObjectBase Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * refactor(decisions): dispatch /v1/decisions through provider configs and the shared HTTP handler Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(decisions): map OpenRouter connection failures to APIConnectionError and drop explanatory docstrings Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * feat(decisions): switch /v1/decisions to the OpenAI Decisions schema Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(decisions): build the usage the OpenAI response schema now requires Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(lens): ask signal questions through the OpenAI Decisions schema Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * feat(decisions): refuse safety_identifier for System One providers unless drop_params * feat(decisions): refuse safety_identifier on providers that cannot take it unless drop_params drops it * test(decisions): cover hosted_vllm in the safety_identifier and per-provider wire tests * fix(decisions): check a System One request's safety_identifier against the provider too * fix(decisions): drop a non-string safety_identifier under drop_params A malformed safety_identifier now follows the drop_params convention in both body shapes: it answers 400 without drop_params and is dropped before the provider call with it. A string identifier on OpenAI stays on the wire either way. * fix(decisions): let /v1/decisions drop a non-string safety_identifier under drop_params The route checked the whole body before routing, so a non-string safety_identifier answered 400 even when the deployment or the body set drop_params. The route now leaves that field to the Decisions call, which knows every drop_params source * refactor(decisions): return the unsupported safety_identifier as a value and raise it in _prepare_call --------- Co-authored-by: kerry Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> Co-authored-by: mateo-berri <277851410+mateo-berri@users.noreply.github.com> --- litellm/decisions/main.py | 66 +- .../llms/base_llm/decisions/transformation.py | 1 + .../llms/openai/decisions/transformation.py | 1 + .../proxy/decisions_endpoints/endpoints.py | 7 +- .../test_databricks_decisions_wire.py | 1 - .../test_decisions_openai_format_wire.py | 586 ++++++++++++++++++ .../translation/decisions/bases/cloudflare.py | 97 +++ .../translation/decisions/bases/openrouter.py | 94 +++ .../translation/decisions/bases/perplexity.py | 95 +++ .../decisions/bases/strands_decider.py | 34 + .../translation/decisions/bases/typesafe.py | 92 +++ .../basic/test_decisions_basic_cloudflare.py | 4 +- .../basic/test_decisions_basic_openrouter.py | 9 +- .../basic/test_decisions_basic_perplexity.py | 9 +- .../test_decisions_basic_strands_decider.py | 11 +- .../basic/test_decisions_basic_typesafe.py | 4 +- tests/unit/decisions/test_main.py | 135 ++++ .../decisions_endpoints/test_endpoints.py | 62 ++ 18 files changed, 1289 insertions(+), 19 deletions(-) create mode 100644 tests/integration/providers/test_decisions_openai_format_wire.py diff --git a/litellm/decisions/main.py b/litellm/decisions/main.py index 4a4af2b580e..032cad00454 100644 --- a/litellm/decisions/main.py +++ b/litellm/decisions/main.py @@ -1,13 +1,13 @@ from collections.abc import Mapping, Sequence -from dataclasses import dataclass, field +from dataclasses import dataclass, field, replace from typing import Final, TypeAlias import httpx -from pydantic import TypeAdapter, ValidationError +from pydantic import ConfigDict, TypeAdapter, ValidationError from typing_extensions import assert_never import litellm -from litellm.litellm_core_utils.core_helpers import RESPONSE_COST_HEADER +from litellm.litellm_core_utils.core_helpers import RESPONSE_COST_HEADER, normalize_drop_params from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj from litellm.llms.base_llm.chat.transformation import BaseLLMException from litellm.llms.base_llm.decisions.transformation import ( @@ -40,6 +40,9 @@ DecisionsRequestFormat: TypeAlias = DecisionsRequestBody | OpenAIDecisionRequest _SYSTEMONE_REQUEST_ADAPTER: Final[TypeAdapter[DecisionsRequestBody]] = TypeAdapter(DecisionsRequestBody) _OPENAI_REQUEST_ADAPTER: Final[TypeAdapter[OpenAIDecisionRequestBody]] = TypeAdapter(OpenAIDecisionRequestBody) +_SAFETY_IDENTIFIER_ADAPTER: Final[TypeAdapter[str | None]] = TypeAdapter( + str | None, config=ConfigDict(title="safety_identifier") +) _HANDLER: Final = BaseLLMHTTPHandler() @@ -96,16 +99,46 @@ def _validate_request( ) -def _ir_request(request: DecisionsRequestFormat) -> DecisionsIRRequest: +def _ir_request(request: DecisionsRequestFormat, safety_identifier: str | None) -> DecisionsIRRequest: match request: case DecisionsRequestBody(): - return systemone_request_to_ir(request) + return replace(systemone_request_to_ir(request), safety_identifier=safety_identifier) case OpenAIDecisionRequestBody(): return openai_request_to_ir(request) case _: assert_never(request) +def _drops_params(kwargs: Mapping[str, object]) -> bool: + return litellm.drop_params is True or normalize_drop_params(kwargs.get("drop_params")) is True + + +def _request_safety_identifier(safety_identifier: object, kwargs: Mapping[str, object]) -> str | None: + if isinstance(safety_identifier, str) or not _drops_params(kwargs): + return _SAFETY_IDENTIFIER_ADAPTER.validate_python(safety_identifier) + return None + + +@dataclass(frozen=True, slots=True) +class _UnsupportedSafetyIdentifier: + pass + + +def _provider_ir_request( + request: DecisionsRequestFormat, + *, + safety_identifier: str | None, + provider_config: BaseDecisionsConfig, + kwargs: Mapping[str, object], +) -> DecisionsIRRequest | _UnsupportedSafetyIdentifier: + ir_request: Final = _ir_request(request, safety_identifier) + if ir_request.safety_identifier is None or provider_config.supports_safety_identifier: + return ir_request + if _drops_params(kwargs): + return replace(ir_request, safety_identifier=None) + return _UnsupportedSafetyIdentifier() + + def _prepare_call( *, model: str, @@ -144,8 +177,12 @@ def _prepare_call( except ValueError as error: raise litellm.BadRequestError(message=str(error), model=model, llm_provider=provider) from error try: + request_safety_identifier: Final = _request_safety_identifier(safety_identifier, kwargs) request: Final = _validate_request( - state=state, questions=questions, decision_input=decision_input, safety_identifier=safety_identifier + state=state, + questions=questions, + decision_input=decision_input, + safety_identifier=request_safety_identifier, ) except ValidationError as error: raise litellm.BadRequestError( @@ -169,7 +206,22 @@ def _prepare_call( llm_provider=provider, ) - ir_request: Final = _ir_request(request) + ir_request: Final = _provider_ir_request( + request, + safety_identifier=request_safety_identifier, + provider_config=provider_config, + kwargs=kwargs, + ) + if isinstance(ir_request, _UnsupportedSafetyIdentifier): + raise litellm.UnsupportedParamsError( + message=( + f"{provider} does not support parameters: ['safety_identifier'], for model={model}. " + "To drop these, set `litellm.drop_params=True` or for proxy:\n\n" + "`litellm_settings:\n drop_params: true`\n" + ), + model=model, + llm_provider=provider, + ) body: Final = provider_config.transform_decisions_request( model=canonical_model, request=ir_request, custom_llm_provider=provider ) diff --git a/litellm/llms/base_llm/decisions/transformation.py b/litellm/llms/base_llm/decisions/transformation.py index c57c643983a..baeeb3f0449 100644 --- a/litellm/llms/base_llm/decisions/transformation.py +++ b/litellm/llms/base_llm/decisions/transformation.py @@ -301,6 +301,7 @@ class BaseDecisionsConfig(ABC): api_key_env: tuple[str, ...] = () api_base_env: tuple[str, ...] = () api_key_required: bool = True + supports_safety_identifier: bool = False health_check_questions: Mapping[str, Mapping[str, object]] = MappingProxyType( {"reachable": MappingProxyType({"type": "noul", "instructions": "Is the service reachable?"})} ) diff --git a/litellm/llms/openai/decisions/transformation.py b/litellm/llms/openai/decisions/transformation.py index 0afbe1a434e..317e5abeeb5 100644 --- a/litellm/llms/openai/decisions/transformation.py +++ b/litellm/llms/openai/decisions/transformation.py @@ -292,6 +292,7 @@ def ir_to_openai_response( class OpenAIDecisionsConfig(BaseDecisionsConfig): path = "/v1/decisions" + supports_safety_identifier = True def get_default_api_base(self) -> str | None: return "https://api.openai.com" diff --git a/litellm/proxy/decisions_endpoints/endpoints.py b/litellm/proxy/decisions_endpoints/endpoints.py index f2891c184bf..4f4209208ac 100644 --- a/litellm/proxy/decisions_endpoints/endpoints.py +++ b/litellm/proxy/decisions_endpoints/endpoints.py @@ -1,4 +1,5 @@ from collections.abc import Mapping +from types import MappingProxyType from typing import Annotated, Final from fastapi import APIRouter, Depends, Request, Response @@ -38,6 +39,10 @@ async def _invalid_request( ) +def _fields_checked_before_routing(data: Mapping[str, object]) -> Mapping[str, object]: + return MappingProxyType({key: value for key, value in data.items() if key != "safety_identifier"}) + + async def _request_data(request: Request, user_api_key_dict: UserAPIKeyAuth) -> dict[str, object]: body: Final = await request.body() try: @@ -75,7 +80,7 @@ async def _process_decisions( data: Final = await _request_data(request, user_api_key_dict) try: - body_adapter.validate_python(data) + body_adapter.validate_python(_fields_checked_before_routing(data)) except ValidationError as error: raise await _invalid_request(raw_data=data, error=error, user_api_key_dict=user_api_key_dict) general_settings: Final = _GENERAL_SETTINGS_ADAPTER.validate_python(proxy_general_settings) diff --git a/tests/integration/providers/test_databricks_decisions_wire.py b/tests/integration/providers/test_databricks_decisions_wire.py index 5c3b7d1b1a1..93dc2ea04fa 100644 --- a/tests/integration/providers/test_databricks_decisions_wire.py +++ b/tests/integration/providers/test_databricks_decisions_wire.py @@ -208,7 +208,6 @@ def test_an_openai_format_request_at_v1_decisions_reaches_the_serving_endpoint_a "model": deployment, "input": _OPENAI_INPUT, "questions": [_OPENAI_TIER_QUESTION], - "safety_identifier": "end-user-1", }, ) assert response.status_code == 200, response.text diff --git a/tests/integration/providers/test_decisions_openai_format_wire.py b/tests/integration/providers/test_decisions_openai_format_wire.py new file mode 100644 index 00000000000..f0e7163e264 --- /dev/null +++ b/tests/integration/providers/test_decisions_openai_format_wire.py @@ -0,0 +1,586 @@ +import math +import uuid +from collections.abc import Callable, Sequence +from dataclasses import dataclass +from pathlib import Path +from typing import Final + +import httpx +import pytest +import yaml +from integration._support.client import Gateway, Scenario, eventually, object_value, string_value +from integration._support.database import read_rows +from integration._support.process import owned_proxy_process +from integration._support.upstream import ScenarioHandle, delete_scenario, register_scenario +from integration.cost_calculation.cost_tracking_case import JsonResponse +from pydantic import JsonValue, TypeAdapter + +import litellm + +_API_KEY: Final = "synthetic-decisions-key" +_SAFETY_IDENTIFIER: Final = "end-user-7" +_DROPPING_DEPLOYMENT: Final = "decisions-under-litellm-settings-drop-params" +_JSON_OBJECT: Final = TypeAdapter(dict[str, JsonValue]) +_INPUT: Final = "Ticket (billing): The export job hangs at 99%" +_FOLLOW_UP: Final = "Customer: still stuck after retrying" +_QUESTIONS: Final[list[JsonValue]] = [ + {"type": "predicate", "name": "defect", "instructions": "Is this a defect?"}, + { + "type": "choice", + "name": "severity", + "instructions": "How severe is it?", + "choices": [{"value": "low", "description": "cosmetic"}, {"value": "high", "description": "blocks users"}], + }, + { + "type": "score", + "name": "confidence", + "instructions": "How sure are you?", + "levels": [{"label": "unsure"}, {"label": "sure"}], + }, +] +_SYSTEM_ONE_QUESTIONS: Final[dict[str, JsonValue]] = { + "defect": {"type": "noul", "instructions": "Is this a defect?"}, + "severity": { + "type": "choice", + "instructions": "How severe is it?", + "criteria": {"low": "cosmetic", "high": "blocks users"}, + }, + "confidence": {"type": "score", "instructions": "How sure are you?", "criteria": ["unsure", "sure"]}, +} +_SYSTEM_ONE_ANSWERS: Final[dict[str, JsonValue]] = { + "defect": {"type": "noul", "noul": 0.93}, + "severity": {"type": "choice", "choice": "high", "confidence": 0.8, "probabilities": {"low": 0.2, "high": 0.8}}, + "confidence": { + "type": "score", + "score": 1.0, + "confidence": 0.7, + "legend": {"0": "unsure", "1": "sure"}, + "probabilities": {"0": 0.3, "1": 0.7}, + }, +} +_ANSWERS: Final[list[JsonValue]] = [ + {"type": "predicate", "name": "defect", "probability": 0.93}, + { + "type": "choice", + "name": "severity", + "choice": "high", + "probabilities": [{"value": "low", "probability": 0.2}, {"value": "high", "probability": 0.8}], + "confidence": 0.8, + }, + { + "type": "score", + "name": "confidence", + "score": 1.0, + "probabilities": [ + {"value": 0, "label": "unsure", "probability": 0.3}, + {"value": 1, "label": "sure", "probability": 0.7}, + ], + "confidence": 0.7, + }, +] +_INPUT_TOKENS: Final = 367 +_OUTPUT_TOKENS: Final = 3 +_SYSTEM_ONE_USAGE: Final[dict[str, JsonValue]] = {"input_tokens": _INPUT_TOKENS, "output_tokens": _OUTPUT_TOKENS} +_USAGE: Final[dict[str, JsonValue]] = { + "input_tokens": _INPUT_TOKENS, + "input_tokens_details": {"cached_tokens": 0, "cache_write_tokens": 0}, + "output_tokens": _OUTPUT_TOKENS, + "output_tokens_details": {"reasoning_tokens": 0}, + "total_tokens": _INPUT_TOKENS + _OUTPUT_TOKENS, +} +_SDK_QUESTIONS: Final = TypeAdapter(list[dict[str, object]]).validate_python(_QUESTIONS) +_SPEND_QUERY: Final = ( + "SELECT spend, status, call_type, model_group, custom_llm_provider, api_base, prompt_tokens, completion_tokens, " + 'request_tags FROM "LiteLLM_SpendLogs" WHERE request_id = %s' +) + + +@dataclass(frozen=True, slots=True) +class _Provider: + name: str + model: str + path: str + body_model: str + api_key: str | None + wraps_result: bool + cost_map_key: str | None + speaks_openai: bool = False + + def upstream_body(self) -> dict[str, JsonValue]: + if self.speaks_openai: + return {"model": self.body_model, "input": _INPUT, "questions": _QUESTIONS} + return {"model": self.body_model, "state": _INPUT, "questions": _SYSTEM_ONE_QUESTIONS} + + def upstream_reply(self) -> dict[str, JsonValue]: + if self.speaks_openai: + return {"model": self.body_model, "answers": _ANSWERS, "usage": _USAGE} + answer: Final[dict[str, JsonValue]] = { + "model": self.body_model, + "answers": _SYSTEM_ONE_ANSWERS, + "usage": _SYSTEM_ONE_USAGE, + } + return {"result": answer, "success": True} if self.wraps_result else answer + + def litellm_response(self) -> dict[str, JsonValue]: + return {"model": self.body_model, "answers": _ANSWERS, "usage": _USAGE} + + +_PROVIDERS: Final = ( + _Provider( + "perplexity", + "perplexity/pplx-decider-v1-27b", + "/v1/decisions", + "pplx-decider-v1-27b", + _API_KEY, + False, + "perplexity/pplx-decider-v1-27b", + ), + _Provider("typesafe", "typesafe/jev-1.13.0", "/v1/systemone", "jev-1.13.0", _API_KEY, False, "typesafe/jev-1.13.0"), + _Provider( + "openrouter", + "openrouter/typesafe/jev-1.13", + "/alpha/decisions", + "typesafe/jev-1.13", + _API_KEY, + False, + "openrouter/typesafe/jev-1.13", + ), + _Provider( + "strands_decider", "strands_decider/systemone-decider", "/v1/systemone", "systemone-decider", None, False, None + ), + _Provider( + "cloudflare", + "cloudflare/clef", + "/ai/run/@cf/cloudflare/clef", + "clef", + _API_KEY, + True, + "cloudflare/@cf/cloudflare/clef", + ), + _Provider("hosted_vllm", "hosted_vllm/Qwen/Qwen3-0.6B", "/v1/systemone", "Qwen/Qwen3-0.6B", None, False, None), + _Provider( + "databricks", + "databricks/databricks-openjev-qwen35-4b", + "/databricks-openjev-qwen35-4b/invocations", + "databricks-openjev-qwen35-4b", + _API_KEY, + False, + None, + ), + _Provider( + "azure_ai", "azure_ai/decision-1", "/providers/microsoft/v1/systemone", "decision-1", _API_KEY, False, None + ), + _Provider("openai", "openai/gpt-6-luna", "/v1/decisions", "gpt-6-luna", _API_KEY, False, "gpt-6-luna", True), +) +_SYSTEM_ONE_PROVIDERS: Final = tuple(provider for provider in _PROVIDERS if not provider.speaks_openai) +_PERPLEXITY: Final = _PROVIDERS[0] +_TYPESAFE: Final = _PROVIDERS[1] +_OPENAI: Final = next(provider for provider in _PROVIDERS if provider.speaks_openai) +_PREDICATE: Final[dict[str, JsonValue]] = {"type": "predicate", "name": "q", "instructions": "Is it?"} +_INVALID_BODIES: Final[tuple[tuple[str, dict[str, JsonValue]], ...]] = ( + ("missing questions", {"input": _INPUT}), + ("missing input", {"questions": _QUESTIONS}), + ("numeric input", {"input": 5, "questions": _QUESTIONS}), + ("assistant message input", {"input": [{"role": "assistant", "content": "hi"}], "questions": _QUESTIONS}), + ("empty questions", {"input": _INPUT, "questions": []}), + ("questions as a map", {"input": _INPUT, "questions": {"q": _PREDICATE}}), + ("predicate without instructions", {"input": _INPUT, "questions": [{"type": "predicate", "name": "q"}]}), + ( + "choice with one choice", + {"input": _INPUT, "questions": [{**_PREDICATE, "type": "choice", "choices": [{"value": "only"}]}]}, + ), + ( + "score with one level", + {"input": _INPUT, "questions": [{**_PREDICATE, "type": "score", "levels": [{"label": "only"}]}]}, + ), + ("unknown question type", {"input": _INPUT, "questions": [{**_PREDICATE, "type": "ranking"}]}), + ("numeric safety_identifier", {"input": _INPUT, "questions": _QUESTIONS, "safety_identifier": 7}), + ("list safety_identifier", {"input": _INPUT, "questions": _QUESTIONS, "safety_identifier": [_SAFETY_IDENTIFIER]}), +) +_IMAGE_INPUT: Final[list[JsonValue]] = [ + { + "role": "user", + "content": [ + {"type": "input_text", "text": _INPUT}, + {"type": "input_image", "image_url": "data:image/png;base64,iVBORw0KGgo=", "detail": "auto"}, + ], + } +] +_OPENAI_IMAGE_MESSAGES: Final[list[JsonValue]] = [ + { + "role": "user", + "type": "message", + "content": [ + {"type": "input_text", "text": _INPUT}, + {"type": "input_image", "image_url": "data:image/png;base64,iVBORw0KGgo=", "detail": "auto"}, + ], + } +] +_MESSAGE_LIST_INPUT: Final[list[JsonValue]] = [ + {"role": "user", "content": _INPUT}, + {"role": "user", "content": [{"type": "input_text", "text": _FOLLOW_UP}]}, +] + + +def _response_cost(response: httpx.Response) -> float: + return float(response.headers["x-litellm-response-cost"]) if "x-litellm-response-cost" in response.headers else 0.0 + + +def _provider_id(provider: _Provider) -> str: + return provider.name + + +def _number(value: JsonValue) -> float: + assert isinstance(value, (int, float)) and not isinstance(value, bool), value + return float(value) + + +def _expected_spend(cost_map_key: str | None) -> float: + if cost_map_key is None: + return 0.0 + cost_map: Final = _JSON_OBJECT.validate_json(Path("model_prices_and_context_window.json").read_bytes()) + prices: Final = object_value(cost_map[cost_map_key]) + return _INPUT_TOKENS * _number(prices["input_cost_per_token"]) + _OUTPUT_TOKENS * _number( + prices["output_cost_per_token"] + ) + + +def _register(scenario: Scenario, body: dict[str, JsonValue], *, status: int = 200) -> ScenarioHandle: + handle: Final = register_scenario( + f"decisions-{uuid.uuid4().hex[:12]}", JsonResponse(content_type="application/json", body=body, status=status) + ) + scenario.cleanups.callback(delete_scenario, handle) + return handle + + +def _deployment(scenario: Scenario, handle: ScenarioHandle, provider: _Provider, *, drop_params: bool = False) -> str: + if drop_params: + return scenario.model( + model=provider.model, api_base=handle.api_base(), api_key=provider.api_key, drop_params=True + ) + return scenario.model(model=provider.model, api_base=handle.api_base(), api_key=provider.api_key) + + +def _decide(gateway: Gateway, model: str, *, route: str = "/v1/decisions", **extra: JsonValue) -> httpx.Response: + return gateway.request("POST", route, {"model": model, "input": _INPUT, "questions": _QUESTIONS, **extra}) + + +def _decide_in_system_one_format(gateway: Gateway, model: str, **extra: JsonValue) -> httpx.Response: + return gateway.request( + "POST", "/v1/systemone", {"model": model, "state": _INPUT, "questions": _SYSTEM_ONE_QUESTIONS, **extra} + ) + + +def _observed_requests(gateway: Gateway) -> tuple[dict[str, JsonValue], ...]: + with httpx.Client(base_url=gateway.upstream_url, timeout=5, trust_env=False) as upstream: + observations: Final = _JSON_OBJECT.validate_json(upstream.get("/__observations").content)["requests"] + assert isinstance(observations, list), observations + return tuple(map(object_value, observations)) + + +def _calls_to(requests: Sequence[dict[str, JsonValue]], handle: ScenarioHandle) -> list[dict[str, JsonValue]]: + return [request for request in requests if string_value(request["path"]).startswith(f"/{handle.scenario_id}/")] + + +def _upstream_calls(gateway: Gateway, handle: ScenarioHandle) -> list[dict[str, JsonValue]]: + return _calls_to(_observed_requests(gateway), handle) + + +def _spend_row(call_id: str) -> dict[str, JsonValue]: + rows: Final = eventually(lambda: read_rows(_SPEND_QUERY, (call_id,)), lambda found: len(found) == 1, seconds=70) + return rows[0] + + +def _assert_refused_as_invalid(gateway: Gateway, model: str, label: str, body: dict[str, JsonValue]) -> None: + response: Final = gateway.request("POST", "/v1/decisions", {"model": model, **body}) + assert response.status_code == 400, (label, response.text) + assert "Invalid Decisions request" in response.text, (label, response.text) + + +def _assert_refused_for_safety_identifier(response: httpx.Response) -> None: + assert response.status_code == 400, response.text + assert "safety_identifier" in response.text and "drop_params" in response.text, response.text + + +def _drop_params_config(directory: Path, api_base: str) -> Path: + base: Final = _JSON_OBJECT.validate_python(yaml.safe_load(Path("tests/integration/proxy_config.yaml").read_text())) + config: Final = { + **base, + "litellm_settings": {**object_value(base["litellm_settings"]), "drop_params": True}, + "model_list": [ + { + "model_name": _DROPPING_DEPLOYMENT, + "litellm_params": {"model": _PERPLEXITY.model, "api_base": api_base, "api_key": _API_KEY}, + } + ], + } + path: Final = directory / "decisions-drop-params.yaml" + path.write_text(yaml.safe_dump(config)) + return path + + +@pytest.mark.parametrize("provider", _PROVIDERS, ids=_provider_id) +def test_each_provider_gets_its_own_path_key_and_body_and_is_billed_from_the_cost_map( + gateway: Gateway, provider: _Provider +) -> None: + expected_spend: Final = _expected_spend(provider.cost_map_key) + with gateway.scenario() as scenario: + handle: Final = _register(scenario, provider.upstream_reply()) + model: Final = _deployment(scenario, handle, provider) + response: Final = _decide(gateway, model) + assert response.status_code == 200, response.text + assert response.json() == provider.litellm_response() + assert response.headers["x-litellm-model-group"] == model + assert math.isclose(_response_cost(response), expected_spend, rel_tol=1e-9) + (call,) = _upstream_calls(gateway, handle) + assert call["path"] == f"/{handle.scenario_id}{provider.path}" + assert call["authorization"] == (f"Bearer {provider.api_key}" if provider.api_key else "") + assert call["body"] == provider.upstream_body() + row: Final = _spend_row(response.headers["x-litellm-call-id"]) + assert ( + row["status"], + row["call_type"], + row["custom_llm_provider"], + row["model_group"], + row["api_base"], + row["prompt_tokens"], + row["completion_tokens"], + ) == ( + "success", + "adecisions", + provider.name, + model, + f"{handle.api_base()}{provider.path}", + _INPUT_TOKENS, + _OUTPUT_TOKENS, + ) + assert math.isclose(_number(row["spend"]), expected_spend, rel_tol=1e-9), row + + +def test_the_unversioned_alias_serves_the_same_request(gateway: Gateway) -> None: + with gateway.scenario() as scenario: + handle: Final = _register(scenario, _PERPLEXITY.upstream_reply()) + model: Final = _deployment(scenario, handle, _PERPLEXITY) + response: Final = _decide(gateway, model, route="/decisions") + assert response.status_code == 200, response.text + assert response.json() == _PERPLEXITY.litellm_response() + (call,) = _upstream_calls(gateway, handle) + assert call["body"] == _PERPLEXITY.upstream_body() + + +def test_repeated_identical_requests_each_reach_the_upstream_and_are_each_billed(gateway: Gateway) -> None: + with gateway.scenario() as scenario: + handle: Final = _register(scenario, _PERPLEXITY.upstream_reply()) + model: Final = _deployment(scenario, handle, _PERPLEXITY) + responses: Final = tuple(_decide(gateway, model) for _ in range(2)) + assert [response.status_code for response in responses] == [200, 200], [r.text for r in responses] + call_ids: Final = tuple(response.headers["x-litellm-call-id"] for response in responses) + assert len(set(call_ids)) == 2, call_ids + assert len(_upstream_calls(gateway, handle)) == 2 + for call_id in call_ids: + assert _spend_row(call_id)["status"] == "success" + + +async def test_sdk_sync_and_async_clients_send_the_same_request(gateway: Gateway) -> None: + with gateway.scenario() as scenario: + handle: Final = _register(scenario, _TYPESAFE.upstream_reply()) + synchronous: Final = litellm.decisions( + model=_TYPESAFE.model, input=_INPUT, questions=_SDK_QUESTIONS, api_base=handle.api_base(), api_key=_API_KEY + ) + asynchronous: Final = await litellm.adecisions( + model=_TYPESAFE.model, input=_INPUT, questions=_SDK_QUESTIONS, api_base=handle.api_base(), api_key=_API_KEY + ) + for response in (synchronous, asynchronous): + assert response.model_dump(mode="json") == _TYPESAFE.litellm_response() + calls: Final = _upstream_calls(gateway, handle) + assert len(calls) == 2, calls + for call in calls: + assert call["path"] == f"/{handle.scenario_id}{_TYPESAFE.path}" + assert call["authorization"] == f"Bearer {_API_KEY}" + assert call["body"] == _TYPESAFE.upstream_body() + + +def test_gateway_only_fields_stay_at_the_gateway_and_tags_reach_the_spend_log(gateway: Gateway) -> None: + tag: Final = f"decisions-openai-format-{uuid.uuid4().hex[:8]}" + with gateway.scenario() as scenario: + handle: Final = _register(scenario, _PERPLEXITY.upstream_reply()) + model: Final = _deployment(scenario, handle, _PERPLEXITY) + response: Final = _decide( + gateway, model, user="auditor", num_retries=0, temperature=0.2, metadata={"tags": [tag]} + ) + assert response.status_code == 200, response.text + (call,) = _upstream_calls(gateway, handle) + assert call["body"] == _PERPLEXITY.upstream_body() + row: Final = _spend_row(response.headers["x-litellm-call-id"]) + tags: Final = row["request_tags"] + assert isinstance(tags, list) and tag in tags, row + + +def test_invalid_bodies_are_refused_at_the_gateway_without_an_upstream_call(gateway: Gateway) -> None: + with gateway.scenario() as scenario: + handle: Final = _register(scenario, _PERPLEXITY.upstream_reply()) + model: Final = _deployment(scenario, handle, _PERPLEXITY) + for label, body in _INVALID_BODIES: + _assert_refused_as_invalid(gateway, model, label, body) + assert _upstream_calls(gateway, handle) == [] + + +@pytest.mark.parametrize("provider", _SYSTEM_ONE_PROVIDERS, ids=_provider_id) +def test_system_one_providers_refuse_images_at_the_gateway_without_an_upstream_call( + gateway: Gateway, provider: _Provider +) -> None: + with gateway.scenario() as scenario: + handle: Final = _register(scenario, provider.upstream_reply()) + model: Final = _deployment(scenario, handle, provider) + response: Final = _decide(gateway, model, input=_IMAGE_INPUT) + assert response.status_code == 400, response.text + assert "input_image" in response.text + assert _upstream_calls(gateway, handle) == [] + + +def test_openai_forwards_image_input_in_its_own_message_shape(gateway: Gateway) -> None: + with gateway.scenario() as scenario: + handle: Final = _register(scenario, _OPENAI.upstream_reply()) + model: Final = _deployment(scenario, handle, _OPENAI) + response: Final = _decide(gateway, model, input=_IMAGE_INPUT) + assert response.status_code == 200, response.text + (call,) = _upstream_calls(gateway, handle) + assert call["body"] == {"model": _OPENAI.body_model, "input": _OPENAI_IMAGE_MESSAGES, "questions": _QUESTIONS} + + +def test_a_message_list_input_reaches_a_system_one_provider_as_its_flattened_text(gateway: Gateway) -> None: + with gateway.scenario() as scenario: + handle: Final = _register(scenario, _PERPLEXITY.upstream_reply()) + model: Final = _deployment(scenario, handle, _PERPLEXITY) + response: Final = _decide(gateway, model, input=_MESSAGE_LIST_INPUT) + assert response.status_code == 200, response.text + assert response.json() == _PERPLEXITY.litellm_response() + (call,) = _upstream_calls(gateway, handle) + assert call["body"] == { + "model": _PERPLEXITY.body_model, + "state": f"{_INPUT}\n\n{_FOLLOW_UP}", + "questions": _SYSTEM_ONE_QUESTIONS, + } + row: Final = _spend_row(response.headers["x-litellm-call-id"]) + assert (row["status"], row["call_type"], row["prompt_tokens"], row["completion_tokens"]) == ( + "success", + "adecisions", + _INPUT_TOKENS, + _OUTPUT_TOKENS, + ) + + +@pytest.mark.parametrize("provider", _SYSTEM_ONE_PROVIDERS, ids=_provider_id) +def test_safety_identifier_is_refused_by_system_one_providers_unless_the_deployment_drops_params( + gateway: Gateway, provider: _Provider +) -> None: + with gateway.scenario() as scenario: + handle: Final = _register(scenario, provider.upstream_reply()) + strict: Final = _deployment(scenario, handle, provider) + refused: Final = _decide(gateway, strict, safety_identifier=_SAFETY_IDENTIFIER) + _assert_refused_for_safety_identifier(refused) + assert _upstream_calls(gateway, handle) == [] + + dropping: Final = _deployment(scenario, handle, provider, drop_params=True) + accepted: Final = _decide(gateway, dropping, safety_identifier=_SAFETY_IDENTIFIER) + assert accepted.status_code == 200, accepted.text + assert accepted.json() == provider.litellm_response() + (call,) = _upstream_calls(gateway, handle) + assert call["body"] == provider.upstream_body() + row: Final = _spend_row(accepted.headers["x-litellm-call-id"]) + assert (row["status"], row["call_type"], row["model_group"]) == ("success", "adecisions", dropping) + + +@pytest.mark.parametrize("value", ("", "x" * 5000), ids=("empty", "5kb")) +def test_every_string_safety_identifier_is_refused_or_dropped_like_the_usual_one(gateway: Gateway, value: str) -> None: + with gateway.scenario() as scenario: + handle: Final = _register(scenario, _PERPLEXITY.upstream_reply()) + strict: Final = _deployment(scenario, handle, _PERPLEXITY) + _assert_refused_for_safety_identifier(_decide(gateway, strict, safety_identifier=value)) + dropping: Final = _deployment(scenario, handle, _PERPLEXITY, drop_params=True) + accepted: Final = _decide(gateway, dropping, safety_identifier=value) + assert accepted.status_code == 200, accepted.text + (call,) = _upstream_calls(gateway, handle) + assert call["body"] == _PERPLEXITY.upstream_body() + + +def test_a_request_body_drop_params_drops_the_safety_identifier_like_chat(gateway: Gateway) -> None: + with gateway.scenario() as scenario: + handle: Final = _register(scenario, _PERPLEXITY.upstream_reply()) + strict: Final = _deployment(scenario, handle, _PERPLEXITY) + response: Final = _decide(gateway, strict, safety_identifier=_SAFETY_IDENTIFIER, drop_params=True) + assert response.status_code == 200, response.text + assert response.json() == _PERPLEXITY.litellm_response() + (call,) = _upstream_calls(gateway, handle) + assert call["body"] == _PERPLEXITY.upstream_body() + + +def test_openai_keeps_the_safety_identifier_on_the_wire_without_drop_params(gateway: Gateway) -> None: + with gateway.scenario() as scenario: + handle: Final = _register(scenario, _OPENAI.upstream_reply()) + model: Final = _deployment(scenario, handle, _OPENAI) + response: Final = _decide(gateway, model, safety_identifier=_SAFETY_IDENTIFIER) + assert response.status_code == 200, response.text + assert response.json() == _OPENAI.litellm_response() + (call,) = _upstream_calls(gateway, handle) + assert call["body"] == {**_OPENAI.upstream_body(), "safety_identifier": _SAFETY_IDENTIFIER} + + +def test_a_system_one_format_safety_identifier_is_refused_unless_the_deployment_drops_params(gateway: Gateway) -> None: + with gateway.scenario() as scenario: + handle: Final = _register(scenario, _PERPLEXITY.upstream_reply()) + strict: Final = _deployment(scenario, handle, _PERPLEXITY) + refused: Final = _decide_in_system_one_format(gateway, strict, safety_identifier=_SAFETY_IDENTIFIER) + _assert_refused_for_safety_identifier(refused) + assert _upstream_calls(gateway, handle) == [] + + dropping: Final = _deployment(scenario, handle, _PERPLEXITY, drop_params=True) + accepted: Final = _decide_in_system_one_format(gateway, dropping, safety_identifier=_SAFETY_IDENTIFIER) + assert accepted.status_code == 200, accepted.text + assert accepted.json()["answers"] == _SYSTEM_ONE_ANSWERS + (call,) = _upstream_calls(gateway, handle) + assert call["body"] == _PERPLEXITY.upstream_body() + + +def test_openai_keeps_a_system_one_format_safety_identifier_on_the_wire(gateway: Gateway) -> None: + with gateway.scenario() as scenario: + handle: Final = _register(scenario, _OPENAI.upstream_reply()) + model: Final = _deployment(scenario, handle, _OPENAI) + response: Final = _decide_in_system_one_format(gateway, model, safety_identifier=_SAFETY_IDENTIFIER) + assert response.status_code == 200, response.text + (call,) = _upstream_calls(gateway, handle) + assert call["body"] == {**_OPENAI.upstream_body(), "safety_identifier": _SAFETY_IDENTIFIER} + + +@pytest.mark.parametrize("decide", (_decide, _decide_in_system_one_format), ids=("openai_format", "system_one_format")) +@pytest.mark.parametrize("value", (7, [_SAFETY_IDENTIFIER]), ids=("numeric", "list")) +def test_a_non_string_safety_identifier_is_refused_as_invalid_unless_the_deployment_drops_params( + gateway: Gateway, value: JsonValue, decide: Callable[..., httpx.Response] +) -> None: + with gateway.scenario() as scenario: + handle: Final = _register(scenario, _OPENAI.upstream_reply()) + strict: Final = _deployment(scenario, handle, _OPENAI) + refused: Final = decide(gateway, strict, safety_identifier=value) + assert refused.status_code == 400, refused.text + assert "Invalid Decisions request" in refused.text, refused.text + assert _upstream_calls(gateway, handle) == [] + + dropping: Final = _deployment(scenario, handle, _OPENAI, drop_params=True) + accepted: Final = decide(gateway, dropping, safety_identifier=value) + assert accepted.status_code == 200, accepted.text + (call,) = _upstream_calls(gateway, handle) + assert call["body"] == _OPENAI.upstream_body() + + +def test_litellm_settings_drop_params_drops_the_safety_identifier_for_a_strict_deployment( + gateway: Gateway, tmp_path: Path +) -> None: + with gateway.scenario() as scenario: + handle: Final = _register(scenario, _PERPLEXITY.upstream_reply()) + config: Final = _drop_params_config(tmp_path, handle.api_base()) + with owned_proxy_process(gateway, tmp_path, {}, config=config) as owned: + response: Final = _decide(owned.gateway, _DROPPING_DEPLOYMENT, safety_identifier=_SAFETY_IDENTIFIER) + assert response.status_code == 200, response.text + assert response.json() == _PERPLEXITY.litellm_response() + (call,) = _upstream_calls(gateway, handle) + assert call["body"] == _PERPLEXITY.upstream_body() diff --git a/tests/integration/translation/decisions/bases/cloudflare.py b/tests/integration/translation/decisions/bases/cloudflare.py index ae86a2ccd87..3efc96eeb87 100644 --- a/tests/integration/translation/decisions/bases/cloudflare.py +++ b/tests/integration/translation/decisions/bases/cloudflare.py @@ -6,6 +6,103 @@ from integration.translation.case import TranslationTestCase """ CLEF_TEST_CASE: Final = TranslationTestCase( scenario="basic", + litellm_endpoint="/v1/decisions", + litellm_request={ + "model": "cloudflare/@cf/cloudflare/clef", + "input": "Ticket (billing): The export job hangs at 99% and never finishes", + "questions": [ + {"type": "predicate", "name": "defect", "instructions": "Is this a defect?"}, + { + "type": "choice", + "name": "severity", + "instructions": "How severe is it?", + "choices": [ + {"value": "low", "description": "cosmetic"}, + {"value": "high", "description": "blocks users"}, + ], + }, + { + "type": "score", + "name": "confidence", + "instructions": "How sure are you?", + "levels": [{"label": "unsure"}, {"label": "sure"}], + }, + ], + "cache": {"no-cache": True}, + }, + expected_provider_endpoint="/ai/run/@cf/cloudflare/clef", + expected_provider_headers={"authorization": "Bearer synthetic-cloudflare-key", "content-type": "application/json"}, + expected_provider_request={ + "model": "clef", + "state": "Ticket (billing): The export job hangs at 99% and never finishes", + "questions": { + "defect": {"type": "noul", "instructions": "Is this a defect?"}, + "severity": { + "type": "choice", + "instructions": "How severe is it?", + "criteria": {"low": "cosmetic", "high": "blocks users"}, + }, + "confidence": {"type": "score", "instructions": "How sure are you?", "criteria": ["unsure", "sure"]}, + }, + }, + mock_provider_response={ + "result": { + "model": "clef", + "answers": { + "defect": {"type": "noul", "noul": 0.9345}, + "severity": { + "type": "choice", + "choice": "high", + "confidence": 0.8067, + "probabilities": {"low": 0.0509, "high": 0.9491}, + }, + "confidence": { + "type": "score", + "score": 0.9036, + "confidence": 0.6515, + "legend": {"0": "unsure", "1": "sure"}, + "probabilities": {"0": 0.0964, "1": 0.9036}, + }, + }, + "usage": {"input_tokens": 290, "output_tokens": 0}, + }, + "success": True, + "errors": [], + "messages": [], + }, + expected_litellm_response={ + "model": "clef", + "answers": [ + {"type": "predicate", "name": "defect", "probability": 0.9345}, + { + "type": "choice", + "name": "severity", + "choice": "high", + "probabilities": [{"value": "low", "probability": 0.0509}, {"value": "high", "probability": 0.9491}], + "confidence": 0.8067, + }, + { + "type": "score", + "name": "confidence", + "score": 0.9036, + "probabilities": [ + {"value": 0, "label": "unsure", "probability": 0.0964}, + {"value": 1, "label": "sure", "probability": 0.9036}, + ], + "confidence": 0.6515, + }, + ], + "usage": { + "input_tokens": 290, + "input_tokens_details": {"cached_tokens": 0, "cache_write_tokens": 0}, + "output_tokens": 0, + "output_tokens_details": {"reasoning_tokens": 0}, + "total_tokens": 290, + }, + }, +) +CLEF_SYSTEMONE_TEST_CASE: Final = TranslationTestCase( + scenario="systemone", litellm_endpoint="/v1/systemone", litellm_request={ "model": "cloudflare/@cf/cloudflare/clef", diff --git a/tests/integration/translation/decisions/bases/openrouter.py b/tests/integration/translation/decisions/bases/openrouter.py index b65acff6b06..ed39bc1a55b 100644 --- a/tests/integration/translation/decisions/bases/openrouter.py +++ b/tests/integration/translation/decisions/bases/openrouter.py @@ -6,6 +6,100 @@ from integration.translation.case import TranslationTestCase """ TYPESAFE_JEV_1_13_TEST_CASE: Final = TranslationTestCase( scenario="basic", + litellm_endpoint="/v1/decisions", + litellm_request={ + "model": "openrouter/typesafe/jev-1.13", + "input": "Ticket (billing): The export job hangs at 99% and never finishes", + "questions": [ + {"type": "predicate", "name": "defect", "instructions": "Is this a defect?"}, + { + "type": "choice", + "name": "severity", + "instructions": "How severe is it?", + "choices": [ + {"value": "low", "description": "cosmetic"}, + {"value": "high", "description": "blocks users"}, + ], + }, + { + "type": "score", + "name": "confidence", + "instructions": "How sure are you?", + "levels": [{"label": "unsure"}, {"label": "sure"}], + }, + ], + "cache": {"no-cache": True}, + }, + expected_provider_endpoint="/alpha/decisions", + expected_provider_headers={"authorization": "Bearer synthetic-openrouter-key", "content-type": "application/json"}, + expected_provider_request={ + "model": "typesafe/jev-1.13", + "state": "Ticket (billing): The export job hangs at 99% and never finishes", + "questions": { + "defect": {"type": "noul", "instructions": "Is this a defect?"}, + "severity": { + "type": "choice", + "instructions": "How severe is it?", + "criteria": {"low": "cosmetic", "high": "blocks users"}, + }, + "confidence": {"type": "score", "instructions": "How sure are you?", "criteria": ["unsure", "sure"]}, + }, + }, + mock_provider_response={ + "model": "typesafe/jev-1.13-20260917", + "answers": { + "defect": {"type": "noul", "noul": 0.81}, + "severity": { + "type": "choice", + "choice": "high", + "confidence": 0.99, + "probabilities": {"low": 0.01, "high": 0.99}, + }, + "confidence": { + "type": "score", + "score": 0.5, + "confidence": 0, + "legend": {"0": "unsure", "1": "sure"}, + "probabilities": {"0": 0.5, "1": 0.5}, + }, + }, + "usage": {"input_tokens": 377, "output_tokens": 62, "cost": 1.5834e-05}, + "id": "gen-dec-1791323839-GA15kY0nt34oiJ7srfki", + "provider": "TypeSafe", + }, + expected_litellm_response={ + "model": "typesafe/jev-1.13-20260917", + "answers": [ + {"type": "predicate", "name": "defect", "probability": 0.81}, + { + "type": "choice", + "name": "severity", + "choice": "high", + "probabilities": [{"value": "low", "probability": 0.01}, {"value": "high", "probability": 0.99}], + "confidence": 0.99, + }, + { + "type": "score", + "name": "confidence", + "score": 0.5, + "probabilities": [ + {"value": 0, "label": "unsure", "probability": 0.5}, + {"value": 1, "label": "sure", "probability": 0.5}, + ], + "confidence": 0, + }, + ], + "usage": { + "input_tokens": 377, + "input_tokens_details": {"cached_tokens": 0, "cache_write_tokens": 0}, + "output_tokens": 62, + "output_tokens_details": {"reasoning_tokens": 0}, + "total_tokens": 439, + }, + }, +) +TYPESAFE_JEV_1_13_SYSTEMONE_TEST_CASE: Final = TranslationTestCase( + scenario="systemone", litellm_endpoint="/v1/systemone", litellm_request={ "model": "openrouter/typesafe/jev-1.13", diff --git a/tests/integration/translation/decisions/bases/perplexity.py b/tests/integration/translation/decisions/bases/perplexity.py index 02b1de51f24..9e29ceab8cc 100644 --- a/tests/integration/translation/decisions/bases/perplexity.py +++ b/tests/integration/translation/decisions/bases/perplexity.py @@ -6,6 +6,101 @@ from integration.translation.case import TranslationTestCase """ PPLX_DECIDER_V1_27B_TEST_CASE: Final = TranslationTestCase( scenario="basic", + litellm_endpoint="/v1/decisions", + litellm_request={ + "model": "perplexity/pplx-decider-v1-27b", + "input": "Ticket (billing): The export job hangs at 99% and never finishes", + "questions": [ + {"type": "predicate", "name": "defect", "instructions": "Is this a defect?"}, + { + "type": "choice", + "name": "severity", + "instructions": "How severe is it?", + "choices": [ + {"value": "low", "description": "cosmetic"}, + {"value": "high", "description": "blocks users"}, + ], + }, + { + "type": "score", + "name": "confidence", + "instructions": "How sure are you?", + "levels": [{"label": "unsure"}, {"label": "sure"}], + }, + ], + "cache": {"no-cache": True}, + }, + expected_provider_endpoint="/v1/decisions", + expected_provider_headers={"authorization": "Bearer synthetic-perplexity-key", "content-type": "application/json"}, + expected_provider_request={ + "model": "pplx-decider-v1-27b", + "state": "Ticket (billing): The export job hangs at 99% and never finishes", + "questions": { + "defect": {"type": "noul", "instructions": "Is this a defect?"}, + "severity": { + "type": "choice", + "instructions": "How severe is it?", + "criteria": {"low": "cosmetic", "high": "blocks users"}, + }, + "confidence": {"type": "score", "instructions": "How sure are you?", "criteria": ["unsure", "sure"]}, + }, + }, + mock_provider_response={ + "model": "pplx-decider-v1-27b", + "answers": { + "defect": {"type": "noul", "noul": 0.9989100737587077}, + "severity": { + "type": "choice", + "choice": "high", + "confidence": 0.9964631215356778, + "probabilities": {"low": 0.0017684392321610232, "high": 0.9982315607678389}, + }, + "confidence": { + "type": "score", + "score": 0.07367392327139817, + "confidence": 0.8526521534572037, + "legend": {"0": "unsure", "1": "sure"}, + "probabilities": {"0": 0.9263260767286018, "1": 0.07367392327139817}, + }, + }, + "usage": {"input_tokens": 318, "output_tokens": 3}, + }, + expected_litellm_response={ + "model": "pplx-decider-v1-27b", + "answers": [ + {"type": "predicate", "name": "defect", "probability": 0.9989100737587077}, + { + "type": "choice", + "name": "severity", + "choice": "high", + "probabilities": [ + {"value": "low", "probability": 0.0017684392321610232}, + {"value": "high", "probability": 0.9982315607678389}, + ], + "confidence": 0.9964631215356778, + }, + { + "type": "score", + "name": "confidence", + "score": 0.07367392327139817, + "probabilities": [ + {"value": 0, "label": "unsure", "probability": 0.9263260767286018}, + {"value": 1, "label": "sure", "probability": 0.07367392327139817}, + ], + "confidence": 0.8526521534572037, + }, + ], + "usage": { + "input_tokens": 318, + "input_tokens_details": {"cached_tokens": 0, "cache_write_tokens": 0}, + "output_tokens": 3, + "output_tokens_details": {"reasoning_tokens": 0}, + "total_tokens": 321, + }, + }, +) +PPLX_DECIDER_V1_27B_SYSTEMONE_TEST_CASE: Final = TranslationTestCase( + scenario="systemone", litellm_endpoint="/v1/systemone", litellm_request={ "model": "perplexity/pplx-decider-v1-27b", diff --git a/tests/integration/translation/decisions/bases/strands_decider.py b/tests/integration/translation/decisions/bases/strands_decider.py index 8a1a51f8c3e..2950b090737 100644 --- a/tests/integration/translation/decisions/bases/strands_decider.py +++ b/tests/integration/translation/decisions/bases/strands_decider.py @@ -6,6 +6,40 @@ from integration.translation.case import TranslationTestCase """ STRANDS_DECIDER_2B_HOBSON_V19_TEST_CASE: Final = TranslationTestCase( scenario="basic", + litellm_endpoint="/v1/decisions", + litellm_request={ + "model": "strands_decider/strands-decider-2B-hobson-v19", + "input": "Help! My payouts have been failing for 3 days!", + "questions": [{"type": "predicate", "name": "is_urgent", "instructions": "Does this convey urgency?"}], + "cache": {"no-cache": True}, + }, + expected_provider_endpoint="/v1/systemone", + expected_provider_headers={"content-type": "application/json"}, + expected_provider_request={ + "model": "strands-decider-2B-hobson-v19", + "state": "Help! My payouts have been failing for 3 days!", + "questions": {"is_urgent": {"type": "noul", "instructions": "Does this convey urgency?"}}, + }, + mock_provider_response={ + "model": "strands-decider-2B-hobson-v19", + "answers": {"is_urgent": {"type": "noul", "noul": 0.8277}}, + "usage": {"input_tokens": 86, "output_tokens": 1}, + "latency_ms": 140.03, + }, + expected_litellm_response={ + "model": "strands-decider-2B-hobson-v19", + "answers": [{"type": "predicate", "name": "is_urgent", "probability": 0.8277}], + "usage": { + "input_tokens": 86, + "input_tokens_details": {"cached_tokens": 0, "cache_write_tokens": 0}, + "output_tokens": 1, + "output_tokens_details": {"reasoning_tokens": 0}, + "total_tokens": 87, + }, + }, +) +STRANDS_DECIDER_2B_HOBSON_V19_SYSTEMONE_TEST_CASE: Final = TranslationTestCase( + scenario="systemone", litellm_endpoint="/v1/systemone", litellm_request={ "model": "strands_decider/strands-decider-2B-hobson-v19", diff --git a/tests/integration/translation/decisions/bases/typesafe.py b/tests/integration/translation/decisions/bases/typesafe.py index 5da5b05cb09..7d47ce6b082 100644 --- a/tests/integration/translation/decisions/bases/typesafe.py +++ b/tests/integration/translation/decisions/bases/typesafe.py @@ -6,6 +6,98 @@ from integration.translation.case import TranslationTestCase """ JEV_1_13_0_TEST_CASE: Final = TranslationTestCase( scenario="basic", + litellm_endpoint="/v1/decisions", + litellm_request={ + "model": "typesafe/jev-1.13.0", + "input": "Ticket (billing): The export job hangs at 99% and never finishes", + "questions": [ + {"type": "predicate", "name": "defect", "instructions": "Is this a defect?"}, + { + "type": "choice", + "name": "severity", + "instructions": "How severe is it?", + "choices": [ + {"value": "low", "description": "cosmetic"}, + {"value": "high", "description": "blocks users"}, + ], + }, + { + "type": "score", + "name": "confidence", + "instructions": "How sure are you?", + "levels": [{"label": "unsure"}, {"label": "sure"}], + }, + ], + "cache": {"no-cache": True}, + }, + expected_provider_endpoint="/v1/systemone", + expected_provider_headers={"authorization": "Bearer synthetic-typesafe-key", "content-type": "application/json"}, + expected_provider_request={ + "model": "jev-1.13.0", + "state": "Ticket (billing): The export job hangs at 99% and never finishes", + "questions": { + "defect": {"type": "noul", "instructions": "Is this a defect?"}, + "severity": { + "type": "choice", + "instructions": "How severe is it?", + "criteria": {"low": "cosmetic", "high": "blocks users"}, + }, + "confidence": {"type": "score", "instructions": "How sure are you?", "criteria": ["unsure", "sure"]}, + }, + }, + mock_provider_response={ + "model": "jev-1.13.0", + "answers": { + "defect": {"type": "noul", "noul": 0.78}, + "severity": { + "type": "choice", + "choice": "high", + "confidence": 0.99, + "probabilities": {"low": 0.01, "high": 0.99}, + }, + "confidence": { + "type": "score", + "score": 0.51, + "confidence": 0.03, + "legend": {"0": "unsure", "1": "sure"}, + "probabilities": {"0": 0.49, "1": 0.51}, + }, + }, + "usage": {"input_tokens": 377, "output_tokens": 62}, + }, + expected_litellm_response={ + "model": "jev-1.13.0", + "answers": [ + {"type": "predicate", "name": "defect", "probability": 0.78}, + { + "type": "choice", + "name": "severity", + "choice": "high", + "probabilities": [{"value": "low", "probability": 0.01}, {"value": "high", "probability": 0.99}], + "confidence": 0.99, + }, + { + "type": "score", + "name": "confidence", + "score": 0.51, + "probabilities": [ + {"value": 0, "label": "unsure", "probability": 0.49}, + {"value": 1, "label": "sure", "probability": 0.51}, + ], + "confidence": 0.03, + }, + ], + "usage": { + "input_tokens": 377, + "input_tokens_details": {"cached_tokens": 0, "cache_write_tokens": 0}, + "output_tokens": 62, + "output_tokens_details": {"reasoning_tokens": 0}, + "total_tokens": 439, + }, + }, +) +JEV_1_13_0_SYSTEMONE_TEST_CASE: Final = TranslationTestCase( + scenario="systemone", litellm_endpoint="/v1/systemone", litellm_request={ "model": "typesafe/jev-1.13.0", diff --git a/tests/integration/translation/decisions/basic/test_decisions_basic_cloudflare.py b/tests/integration/translation/decisions/basic/test_decisions_basic_cloudflare.py index 79fc0574b3b..21e0538897d 100644 --- a/tests/integration/translation/decisions/basic/test_decisions_basic_cloudflare.py +++ b/tests/integration/translation/decisions/basic/test_decisions_basic_cloudflare.py @@ -2,10 +2,10 @@ import pytest from integration._support.client import Gateway from integration._support.provider import SharedProvider from integration.translation.case import TranslationTestCase -from integration.translation.decisions.bases.cloudflare import CLEF_TEST_CASE +from integration.translation.decisions.bases.cloudflare import CLEF_TEST_CASE, CLEF_SYSTEMONE_TEST_CASE from integration.translation.runner import assert_translation -@pytest.mark.parametrize("case", [CLEF_TEST_CASE], ids=lambda case: case.id) +@pytest.mark.parametrize("case", [CLEF_TEST_CASE, CLEF_SYSTEMONE_TEST_CASE], ids=lambda case: case.id) def test_decisions_basic_cloudflare(case: TranslationTestCase, gateway: Gateway, provider: SharedProvider) -> None: assert_translation(case, gateway, provider) diff --git a/tests/integration/translation/decisions/basic/test_decisions_basic_openrouter.py b/tests/integration/translation/decisions/basic/test_decisions_basic_openrouter.py index a74cb23f6b4..973526d7272 100644 --- a/tests/integration/translation/decisions/basic/test_decisions_basic_openrouter.py +++ b/tests/integration/translation/decisions/basic/test_decisions_basic_openrouter.py @@ -2,10 +2,15 @@ import pytest from integration._support.client import Gateway from integration._support.provider import SharedProvider from integration.translation.case import TranslationTestCase -from integration.translation.decisions.bases.openrouter import TYPESAFE_JEV_1_13_TEST_CASE +from integration.translation.decisions.bases.openrouter import ( + TYPESAFE_JEV_1_13_TEST_CASE, + TYPESAFE_JEV_1_13_SYSTEMONE_TEST_CASE, +) from integration.translation.runner import assert_translation -@pytest.mark.parametrize("case", [TYPESAFE_JEV_1_13_TEST_CASE], ids=lambda case: case.id) +@pytest.mark.parametrize( + "case", [TYPESAFE_JEV_1_13_TEST_CASE, TYPESAFE_JEV_1_13_SYSTEMONE_TEST_CASE], ids=lambda case: case.id +) def test_decisions_basic_openrouter(case: TranslationTestCase, gateway: Gateway, provider: SharedProvider) -> None: assert_translation(case, gateway, provider) diff --git a/tests/integration/translation/decisions/basic/test_decisions_basic_perplexity.py b/tests/integration/translation/decisions/basic/test_decisions_basic_perplexity.py index e341d24ba53..d9662ec4b9f 100644 --- a/tests/integration/translation/decisions/basic/test_decisions_basic_perplexity.py +++ b/tests/integration/translation/decisions/basic/test_decisions_basic_perplexity.py @@ -2,10 +2,15 @@ import pytest from integration._support.client import Gateway from integration._support.provider import SharedProvider from integration.translation.case import TranslationTestCase -from integration.translation.decisions.bases.perplexity import PPLX_DECIDER_V1_27B_TEST_CASE +from integration.translation.decisions.bases.perplexity import ( + PPLX_DECIDER_V1_27B_TEST_CASE, + PPLX_DECIDER_V1_27B_SYSTEMONE_TEST_CASE, +) from integration.translation.runner import assert_translation -@pytest.mark.parametrize("case", [PPLX_DECIDER_V1_27B_TEST_CASE], ids=lambda case: case.id) +@pytest.mark.parametrize( + "case", [PPLX_DECIDER_V1_27B_TEST_CASE, PPLX_DECIDER_V1_27B_SYSTEMONE_TEST_CASE], ids=lambda case: case.id +) def test_decisions_basic_perplexity(case: TranslationTestCase, gateway: Gateway, provider: SharedProvider) -> None: assert_translation(case, gateway, provider) diff --git a/tests/integration/translation/decisions/basic/test_decisions_basic_strands_decider.py b/tests/integration/translation/decisions/basic/test_decisions_basic_strands_decider.py index cae7ca70c78..2d2cd9caaf5 100644 --- a/tests/integration/translation/decisions/basic/test_decisions_basic_strands_decider.py +++ b/tests/integration/translation/decisions/basic/test_decisions_basic_strands_decider.py @@ -2,10 +2,17 @@ import pytest from integration._support.client import Gateway from integration._support.provider import SharedProvider from integration.translation.case import TranslationTestCase -from integration.translation.decisions.bases.strands_decider import STRANDS_DECIDER_2B_HOBSON_V19_TEST_CASE +from integration.translation.decisions.bases.strands_decider import ( + STRANDS_DECIDER_2B_HOBSON_V19_TEST_CASE, + STRANDS_DECIDER_2B_HOBSON_V19_SYSTEMONE_TEST_CASE, +) from integration.translation.runner import assert_translation -@pytest.mark.parametrize("case", [STRANDS_DECIDER_2B_HOBSON_V19_TEST_CASE], ids=lambda case: case.id) +@pytest.mark.parametrize( + "case", + [STRANDS_DECIDER_2B_HOBSON_V19_TEST_CASE, STRANDS_DECIDER_2B_HOBSON_V19_SYSTEMONE_TEST_CASE], + ids=lambda case: case.id, +) def test_decisions_basic_strands_decider(case: TranslationTestCase, gateway: Gateway, provider: SharedProvider) -> None: assert_translation(case, gateway, provider) diff --git a/tests/integration/translation/decisions/basic/test_decisions_basic_typesafe.py b/tests/integration/translation/decisions/basic/test_decisions_basic_typesafe.py index 302636e6fb4..c1111cc6ba9 100644 --- a/tests/integration/translation/decisions/basic/test_decisions_basic_typesafe.py +++ b/tests/integration/translation/decisions/basic/test_decisions_basic_typesafe.py @@ -2,10 +2,10 @@ import pytest from integration._support.client import Gateway from integration._support.provider import SharedProvider from integration.translation.case import TranslationTestCase -from integration.translation.decisions.bases.typesafe import JEV_1_13_0_TEST_CASE +from integration.translation.decisions.bases.typesafe import JEV_1_13_0_TEST_CASE, JEV_1_13_0_SYSTEMONE_TEST_CASE from integration.translation.runner import assert_translation -@pytest.mark.parametrize("case", [JEV_1_13_0_TEST_CASE], ids=lambda case: case.id) +@pytest.mark.parametrize("case", [JEV_1_13_0_TEST_CASE, JEV_1_13_0_SYSTEMONE_TEST_CASE], ids=lambda case: case.id) def test_decisions_basic_typesafe(case: TranslationTestCase, gateway: Gateway, provider: SharedProvider) -> None: assert_translation(case, gateway, provider) diff --git a/tests/unit/decisions/test_main.py b/tests/unit/decisions/test_main.py index 8f05a492119..c1cac2d386f 100644 --- a/tests/unit/decisions/test_main.py +++ b/tests/unit/decisions/test_main.py @@ -1262,6 +1262,141 @@ async def test_openrouter_decisions_uses_provider_reported_cost_without_cost_map assert get_response_cost_from_hidden_params(response.hidden_params) == cost +_SAFETY_IDENTIFIER: Final = "end-user-7" +_PREDICATE_QUESTIONS: Final[tuple[Mapping[str, object], ...]] = ( + {"type": "predicate", "name": "is_defect", "instructions": "Is this a defect?"}, +) +_PREDICATE_REQUESTS: Final[tuple[Mapping[str, object], ...]] = ( + MappingProxyType({"input": "review", "questions": _PREDICATE_QUESTIONS}), + MappingProxyType( + {"state": "review", "questions": {"is_defect": {"type": "noul", "instructions": "Is this a defect?"}}} + ), +) +_PREDICATE_REQUEST_IDS: Final = ("input_format", "state_format") + + +@pytest.mark.parametrize("request_kwargs", _PREDICATE_REQUESTS, ids=_PREDICATE_REQUEST_IDS) +def test_safety_identifier_is_refused_before_http_when_the_provider_cannot_take_it( + request_kwargs: Mapping[str, object], respx_mock: respx.MockRouter, monkeypatch: pytest.MonkeyPatch +) -> None: + monkeypatch.setattr(litellm, "drop_params", False) + route: Final = respx_mock.post("https://api.perplexity.ai/v1/decisions").respond(json=_RESPONSE) + + with pytest.raises(litellm.UnsupportedParamsError, match=r"safety_identifier.*drop_params") as caught: + litellm.decisions( + model="perplexity/pplx-decider-v1-27b", + safety_identifier=_SAFETY_IDENTIFIER, + api_key="caller-key", + **request_kwargs, + ) + + assert caught.value.status_code == 400 + assert not route.called + + +@pytest.mark.asyncio +@pytest.mark.parametrize("scope", ("call", "global")) +async def test_safety_identifier_is_dropped_from_the_wire_under_drop_params( + scope: str, respx_mock: respx.MockRouter, monkeypatch: pytest.MonkeyPatch +) -> None: + monkeypatch.setattr(litellm, "drop_params", scope == "global") + route: Final = respx_mock.post("https://api.perplexity.ai/v1/decisions").respond(json=_RESPONSE) + + response: Final = await litellm.adecisions( + model="perplexity/pplx-decider-v1-27b", + input="review", + questions=_PREDICATE_QUESTIONS, + safety_identifier=_SAFETY_IDENTIFIER, + api_key="caller-key", + **({"drop_params": True} if scope == "call" else {}), + ) + + assert route.called + assert json.loads(respx_mock.calls[0].request.content) == { + "model": "pplx-decider-v1-27b", + "state": "review", + "questions": {"is_defect": {"type": "noul", "instructions": "Is this a defect?"}}, + } + assert isinstance(response, OpenAIDecisionResponse) + assert [answer.model_dump(mode="json") for answer in response.answers] == [ + {"type": "predicate", "name": "is_defect", "probability": 0.9} + ] + + +@pytest.mark.asyncio +@pytest.mark.parametrize("request_kwargs", _PREDICATE_REQUESTS, ids=_PREDICATE_REQUEST_IDS) +@pytest.mark.parametrize("drop_params", (False, True), ids=("strict", "drop_params")) +async def test_safety_identifier_reaches_openai_whether_or_not_params_are_dropped( + drop_params: bool, + request_kwargs: Mapping[str, object], + respx_mock: respx.MockRouter, + monkeypatch: pytest.MonkeyPatch, +) -> None: + monkeypatch.setattr(litellm, "drop_params", drop_params) + monkeypatch.setattr(litellm, "api_base", None) + monkeypatch.delenv("OPENAI_BASE_URL", raising=False) + monkeypatch.delenv("OPENAI_API_BASE", raising=False) + route: Final = respx_mock.post("https://api.openai.com/v1/decisions").respond(json=_OPENAI_RESPONSE) + + await litellm.adecisions( + model="openai/gpt-6-luna", + safety_identifier=_SAFETY_IDENTIFIER, + api_key="caller-key", + **request_kwargs, + ) + + assert route.called + assert json.loads(respx_mock.calls[0].request.content) == { + "model": "gpt-6-luna", + "input": "review", + "questions": [{"type": "predicate", "name": "is_defect", "instructions": "Is this a defect?"}], + "safety_identifier": _SAFETY_IDENTIFIER, + } + + +@pytest.mark.asyncio +@pytest.mark.parametrize("request_kwargs", _PREDICATE_REQUESTS, ids=_PREDICATE_REQUEST_IDS) +async def test_non_string_safety_identifier_is_rejected_before_http( + request_kwargs: Mapping[str, object], respx_mock: respx.MockRouter, monkeypatch: pytest.MonkeyPatch +) -> None: + monkeypatch.setattr(litellm, "drop_params", False) + route: Final = respx_mock.post("https://api.openai.com/v1/decisions").respond(json=_OPENAI_RESPONSE) + invalid_safety_identifier: Final[Mapping[str, object]] = MappingProxyType({"safety_identifier": 7}) + + with pytest.raises(litellm.BadRequestError, match=r"(?s)Invalid Decisions request.*safety_identifier"): + await litellm.adecisions( + model="openai/gpt-6-luna", api_key="caller-key", **request_kwargs, **invalid_safety_identifier + ) + + assert not route.called + + +@pytest.mark.asyncio +@pytest.mark.parametrize("request_kwargs", _PREDICATE_REQUESTS, ids=_PREDICATE_REQUEST_IDS) +async def test_non_string_safety_identifier_is_dropped_from_the_wire_under_drop_params( + request_kwargs: Mapping[str, object], respx_mock: respx.MockRouter, monkeypatch: pytest.MonkeyPatch +) -> None: + monkeypatch.setattr(litellm, "drop_params", False) + monkeypatch.setattr(litellm, "api_base", None) + monkeypatch.delenv("OPENAI_BASE_URL", raising=False) + monkeypatch.delenv("OPENAI_API_BASE", raising=False) + route: Final = respx_mock.post("https://api.openai.com/v1/decisions").respond(json=_OPENAI_RESPONSE) + dropped_safety_identifier: Final[Mapping[str, object]] = MappingProxyType( + {"safety_identifier": 7, "drop_params": True} + ) + + await litellm.adecisions( + model="openai/gpt-6-luna", api_key="caller-key", **request_kwargs, **dropped_safety_identifier + ) + + assert route.called + assert json.loads(respx_mock.calls[0].request.content) == { + "model": "gpt-6-luna", + "input": "review", + "questions": [{"type": "predicate", "name": "is_defect", "instructions": "Is this a defect?"}], + } + + @pytest.mark.asyncio @pytest.mark.parametrize( "api_base", diff --git a/tests/unit/proxy/decisions_endpoints/test_endpoints.py b/tests/unit/proxy/decisions_endpoints/test_endpoints.py index 6ce15636f1e..ed78fc7063a 100644 --- a/tests/unit/proxy/decisions_endpoints/test_endpoints.py +++ b/tests/unit/proxy/decisions_endpoints/test_endpoints.py @@ -575,6 +575,68 @@ def test_openai_format_decisions_reach_an_openai_deployment_unchanged_including_ assert float(response.headers["x-litellm-response-cost"]) == pytest.approx(expected_cost) +@pytest.mark.parametrize("safety_identifier", (7, ["end-user-1"]), ids=("numeric", "list")) +@pytest.mark.parametrize("deployment_drops_params", (False, True), ids=("strict", "drop_params")) +def test_a_non_string_safety_identifier_is_refused_unless_the_deployment_drops_params( + client: TestClient, + monkeypatch: pytest.MonkeyPatch, + respx_mock: respx.MockRouter, + safety_identifier: object, + deployment_drops_params: bool, +) -> None: + monkeypatch.setattr(litellm, "drop_params", False) + monkeypatch.setattr(litellm, "api_base", None) + monkeypatch.delenv("OPENAI_BASE_URL", raising=False) + monkeypatch.delenv("OPENAI_API_BASE", raising=False) + monkeypatch.setattr( + litellm.proxy.proxy_server, + "llm_router", + litellm.Router( + model_list=[ + { + "model_name": "decider", + "litellm_params": { + "model": "openai/gpt-6-luna", + "api_key": "k", + "drop_params": deployment_drops_params, + }, + } + ] + ), + ) + upstream: Final = respx_mock.post("https://api.openai.com/v1/decisions").respond( + json={ + "model": "gpt-6-luna", + "answers": _OPENAI_FORMAT_ANSWERS[:1], + "usage": { + "input_tokens": _INPUT_TOKENS, + "output_tokens": _OUTPUT_TOKENS, + "total_tokens": _INPUT_TOKENS + _OUTPUT_TOKENS, + }, + } + ) + request_body: Final = { + "model": "decider", + "input": "The package arrived with a broken screen.", + "questions": _OPENAI_FORMAT_REQUEST["questions"][:1], + "safety_identifier": safety_identifier, + } + + response: Final = client.post("/v1/decisions", json=request_body) + + if not deployment_drops_params: + assert response.status_code == 400, response.text + assert "safety_identifier" in response.json()["error"]["message"] + assert not upstream.called + return + assert response.status_code == 200, response.text + assert json.loads(upstream.calls[0].request.content) == { + "model": "gpt-6-luna", + "input": "The package arrived with a broken screen.", + "questions": _OPENAI_FORMAT_REQUEST["questions"][:1], + } + + def _decisions_feature() -> LazyFeature: return next(feature for feature in LAZY_FEATURES if feature.name == "decisions")