mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-10 03:28:53 +00:00
fix(decisions)!: refuse safety_identifier on providers that cannot take it unless drop_params drops it (#44955)
* feat(decisions): add the OpenAI Decisions spec types and the System One translation Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * refactor(decisions): share one DecisionsModel config and require model and usage on responses Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * refactor(decisions): rename the shared pydantic parent to DecisionsObjectBase Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * refactor(decisions): dispatch /v1/decisions through provider configs and the shared HTTP handler Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(decisions): map OpenRouter connection failures to APIConnectionError and drop explanatory docstrings Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * feat(decisions): switch /v1/decisions to the OpenAI Decisions schema Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(decisions): build the usage the OpenAI response schema now requires Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(lens): ask signal questions through the OpenAI Decisions schema Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * feat(decisions): refuse safety_identifier for System One providers unless drop_params * feat(decisions): refuse safety_identifier on providers that cannot take it unless drop_params drops it * test(decisions): cover hosted_vllm in the safety_identifier and per-provider wire tests * fix(decisions): check a System One request's safety_identifier against the provider too * fix(decisions): drop a non-string safety_identifier under drop_params A malformed safety_identifier now follows the drop_params convention in both body shapes: it answers 400 without drop_params and is dropped before the provider call with it. A string identifier on OpenAI stays on the wire either way. * fix(decisions): let /v1/decisions drop a non-string safety_identifier under drop_params The route checked the whole body before routing, so a non-string safety_identifier answered 400 even when the deployment or the body set drop_params. The route now leaves that field to the Decisions call, which knows every drop_params source * refactor(decisions): return the unsupported safety_identifier as a value and raise it in _prepare_call --------- Co-authored-by: kerry <kerry@berri.ai> Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> Co-authored-by: mateo-berri <277851410+mateo-berri@users.noreply.github.com>
This commit is contained in:
parent
d6db8e8744
commit
546402c98c
18 changed files with 1289 additions and 19 deletions
|
|
@ -1,13 +1,13 @@
|
|||
from collections.abc import Mapping, Sequence
|
||||
from dataclasses import dataclass, field
|
||||
from dataclasses import dataclass, field, replace
|
||||
from typing import Final, TypeAlias
|
||||
|
||||
import httpx
|
||||
from pydantic import TypeAdapter, ValidationError
|
||||
from pydantic import ConfigDict, TypeAdapter, ValidationError
|
||||
from typing_extensions import assert_never
|
||||
|
||||
import litellm
|
||||
from litellm.litellm_core_utils.core_helpers import RESPONSE_COST_HEADER
|
||||
from litellm.litellm_core_utils.core_helpers import RESPONSE_COST_HEADER, normalize_drop_params
|
||||
from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj
|
||||
from litellm.llms.base_llm.chat.transformation import BaseLLMException
|
||||
from litellm.llms.base_llm.decisions.transformation import (
|
||||
|
|
@ -40,6 +40,9 @@ DecisionsRequestFormat: TypeAlias = DecisionsRequestBody | OpenAIDecisionRequest
|
|||
|
||||
_SYSTEMONE_REQUEST_ADAPTER: Final[TypeAdapter[DecisionsRequestBody]] = TypeAdapter(DecisionsRequestBody)
|
||||
_OPENAI_REQUEST_ADAPTER: Final[TypeAdapter[OpenAIDecisionRequestBody]] = TypeAdapter(OpenAIDecisionRequestBody)
|
||||
_SAFETY_IDENTIFIER_ADAPTER: Final[TypeAdapter[str | None]] = TypeAdapter(
|
||||
str | None, config=ConfigDict(title="safety_identifier")
|
||||
)
|
||||
_HANDLER: Final = BaseLLMHTTPHandler()
|
||||
|
||||
|
||||
|
|
@ -96,16 +99,46 @@ def _validate_request(
|
|||
)
|
||||
|
||||
|
||||
def _ir_request(request: DecisionsRequestFormat) -> DecisionsIRRequest:
|
||||
def _ir_request(request: DecisionsRequestFormat, safety_identifier: str | None) -> DecisionsIRRequest:
|
||||
match request:
|
||||
case DecisionsRequestBody():
|
||||
return systemone_request_to_ir(request)
|
||||
return replace(systemone_request_to_ir(request), safety_identifier=safety_identifier)
|
||||
case OpenAIDecisionRequestBody():
|
||||
return openai_request_to_ir(request)
|
||||
case _:
|
||||
assert_never(request)
|
||||
|
||||
|
||||
def _drops_params(kwargs: Mapping[str, object]) -> bool:
|
||||
return litellm.drop_params is True or normalize_drop_params(kwargs.get("drop_params")) is True
|
||||
|
||||
|
||||
def _request_safety_identifier(safety_identifier: object, kwargs: Mapping[str, object]) -> str | None:
|
||||
if isinstance(safety_identifier, str) or not _drops_params(kwargs):
|
||||
return _SAFETY_IDENTIFIER_ADAPTER.validate_python(safety_identifier)
|
||||
return None
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class _UnsupportedSafetyIdentifier:
|
||||
pass
|
||||
|
||||
|
||||
def _provider_ir_request(
|
||||
request: DecisionsRequestFormat,
|
||||
*,
|
||||
safety_identifier: str | None,
|
||||
provider_config: BaseDecisionsConfig,
|
||||
kwargs: Mapping[str, object],
|
||||
) -> DecisionsIRRequest | _UnsupportedSafetyIdentifier:
|
||||
ir_request: Final = _ir_request(request, safety_identifier)
|
||||
if ir_request.safety_identifier is None or provider_config.supports_safety_identifier:
|
||||
return ir_request
|
||||
if _drops_params(kwargs):
|
||||
return replace(ir_request, safety_identifier=None)
|
||||
return _UnsupportedSafetyIdentifier()
|
||||
|
||||
|
||||
def _prepare_call(
|
||||
*,
|
||||
model: str,
|
||||
|
|
@ -144,8 +177,12 @@ def _prepare_call(
|
|||
except ValueError as error:
|
||||
raise litellm.BadRequestError(message=str(error), model=model, llm_provider=provider) from error
|
||||
try:
|
||||
request_safety_identifier: Final = _request_safety_identifier(safety_identifier, kwargs)
|
||||
request: Final = _validate_request(
|
||||
state=state, questions=questions, decision_input=decision_input, safety_identifier=safety_identifier
|
||||
state=state,
|
||||
questions=questions,
|
||||
decision_input=decision_input,
|
||||
safety_identifier=request_safety_identifier,
|
||||
)
|
||||
except ValidationError as error:
|
||||
raise litellm.BadRequestError(
|
||||
|
|
@ -169,7 +206,22 @@ def _prepare_call(
|
|||
llm_provider=provider,
|
||||
)
|
||||
|
||||
ir_request: Final = _ir_request(request)
|
||||
ir_request: Final = _provider_ir_request(
|
||||
request,
|
||||
safety_identifier=request_safety_identifier,
|
||||
provider_config=provider_config,
|
||||
kwargs=kwargs,
|
||||
)
|
||||
if isinstance(ir_request, _UnsupportedSafetyIdentifier):
|
||||
raise litellm.UnsupportedParamsError(
|
||||
message=(
|
||||
f"{provider} does not support parameters: ['safety_identifier'], for model={model}. "
|
||||
"To drop these, set `litellm.drop_params=True` or for proxy:\n\n"
|
||||
"`litellm_settings:\n drop_params: true`\n"
|
||||
),
|
||||
model=model,
|
||||
llm_provider=provider,
|
||||
)
|
||||
body: Final = provider_config.transform_decisions_request(
|
||||
model=canonical_model, request=ir_request, custom_llm_provider=provider
|
||||
)
|
||||
|
|
|
|||
|
|
@ -301,6 +301,7 @@ class BaseDecisionsConfig(ABC):
|
|||
api_key_env: tuple[str, ...] = ()
|
||||
api_base_env: tuple[str, ...] = ()
|
||||
api_key_required: bool = True
|
||||
supports_safety_identifier: bool = False
|
||||
health_check_questions: Mapping[str, Mapping[str, object]] = MappingProxyType(
|
||||
{"reachable": MappingProxyType({"type": "noul", "instructions": "Is the service reachable?"})}
|
||||
)
|
||||
|
|
|
|||
|
|
@ -292,6 +292,7 @@ def ir_to_openai_response(
|
|||
|
||||
class OpenAIDecisionsConfig(BaseDecisionsConfig):
|
||||
path = "/v1/decisions"
|
||||
supports_safety_identifier = True
|
||||
|
||||
def get_default_api_base(self) -> str | None:
|
||||
return "https://api.openai.com"
|
||||
|
|
|
|||
|
|
@ -1,4 +1,5 @@
|
|||
from collections.abc import Mapping
|
||||
from types import MappingProxyType
|
||||
from typing import Annotated, Final
|
||||
|
||||
from fastapi import APIRouter, Depends, Request, Response
|
||||
|
|
@ -38,6 +39,10 @@ async def _invalid_request(
|
|||
)
|
||||
|
||||
|
||||
def _fields_checked_before_routing(data: Mapping[str, object]) -> Mapping[str, object]:
|
||||
return MappingProxyType({key: value for key, value in data.items() if key != "safety_identifier"})
|
||||
|
||||
|
||||
async def _request_data(request: Request, user_api_key_dict: UserAPIKeyAuth) -> dict[str, object]:
|
||||
body: Final = await request.body()
|
||||
try:
|
||||
|
|
@ -75,7 +80,7 @@ async def _process_decisions(
|
|||
|
||||
data: Final = await _request_data(request, user_api_key_dict)
|
||||
try:
|
||||
body_adapter.validate_python(data)
|
||||
body_adapter.validate_python(_fields_checked_before_routing(data))
|
||||
except ValidationError as error:
|
||||
raise await _invalid_request(raw_data=data, error=error, user_api_key_dict=user_api_key_dict)
|
||||
general_settings: Final = _GENERAL_SETTINGS_ADAPTER.validate_python(proxy_general_settings)
|
||||
|
|
|
|||
|
|
@ -208,7 +208,6 @@ def test_an_openai_format_request_at_v1_decisions_reaches_the_serving_endpoint_a
|
|||
"model": deployment,
|
||||
"input": _OPENAI_INPUT,
|
||||
"questions": [_OPENAI_TIER_QUESTION],
|
||||
"safety_identifier": "end-user-1",
|
||||
},
|
||||
)
|
||||
assert response.status_code == 200, response.text
|
||||
|
|
|
|||
586
tests/integration/providers/test_decisions_openai_format_wire.py
Normal file
586
tests/integration/providers/test_decisions_openai_format_wire.py
Normal file
|
|
@ -0,0 +1,586 @@
|
|||
import math
|
||||
import uuid
|
||||
from collections.abc import Callable, Sequence
|
||||
from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
from typing import Final
|
||||
|
||||
import httpx
|
||||
import pytest
|
||||
import yaml
|
||||
from integration._support.client import Gateway, Scenario, eventually, object_value, string_value
|
||||
from integration._support.database import read_rows
|
||||
from integration._support.process import owned_proxy_process
|
||||
from integration._support.upstream import ScenarioHandle, delete_scenario, register_scenario
|
||||
from integration.cost_calculation.cost_tracking_case import JsonResponse
|
||||
from pydantic import JsonValue, TypeAdapter
|
||||
|
||||
import litellm
|
||||
|
||||
_API_KEY: Final = "synthetic-decisions-key"
|
||||
_SAFETY_IDENTIFIER: Final = "end-user-7"
|
||||
_DROPPING_DEPLOYMENT: Final = "decisions-under-litellm-settings-drop-params"
|
||||
_JSON_OBJECT: Final = TypeAdapter(dict[str, JsonValue])
|
||||
_INPUT: Final = "Ticket (billing): The export job hangs at 99%"
|
||||
_FOLLOW_UP: Final = "Customer: still stuck after retrying"
|
||||
_QUESTIONS: Final[list[JsonValue]] = [
|
||||
{"type": "predicate", "name": "defect", "instructions": "Is this a defect?"},
|
||||
{
|
||||
"type": "choice",
|
||||
"name": "severity",
|
||||
"instructions": "How severe is it?",
|
||||
"choices": [{"value": "low", "description": "cosmetic"}, {"value": "high", "description": "blocks users"}],
|
||||
},
|
||||
{
|
||||
"type": "score",
|
||||
"name": "confidence",
|
||||
"instructions": "How sure are you?",
|
||||
"levels": [{"label": "unsure"}, {"label": "sure"}],
|
||||
},
|
||||
]
|
||||
_SYSTEM_ONE_QUESTIONS: Final[dict[str, JsonValue]] = {
|
||||
"defect": {"type": "noul", "instructions": "Is this a defect?"},
|
||||
"severity": {
|
||||
"type": "choice",
|
||||
"instructions": "How severe is it?",
|
||||
"criteria": {"low": "cosmetic", "high": "blocks users"},
|
||||
},
|
||||
"confidence": {"type": "score", "instructions": "How sure are you?", "criteria": ["unsure", "sure"]},
|
||||
}
|
||||
_SYSTEM_ONE_ANSWERS: Final[dict[str, JsonValue]] = {
|
||||
"defect": {"type": "noul", "noul": 0.93},
|
||||
"severity": {"type": "choice", "choice": "high", "confidence": 0.8, "probabilities": {"low": 0.2, "high": 0.8}},
|
||||
"confidence": {
|
||||
"type": "score",
|
||||
"score": 1.0,
|
||||
"confidence": 0.7,
|
||||
"legend": {"0": "unsure", "1": "sure"},
|
||||
"probabilities": {"0": 0.3, "1": 0.7},
|
||||
},
|
||||
}
|
||||
_ANSWERS: Final[list[JsonValue]] = [
|
||||
{"type": "predicate", "name": "defect", "probability": 0.93},
|
||||
{
|
||||
"type": "choice",
|
||||
"name": "severity",
|
||||
"choice": "high",
|
||||
"probabilities": [{"value": "low", "probability": 0.2}, {"value": "high", "probability": 0.8}],
|
||||
"confidence": 0.8,
|
||||
},
|
||||
{
|
||||
"type": "score",
|
||||
"name": "confidence",
|
||||
"score": 1.0,
|
||||
"probabilities": [
|
||||
{"value": 0, "label": "unsure", "probability": 0.3},
|
||||
{"value": 1, "label": "sure", "probability": 0.7},
|
||||
],
|
||||
"confidence": 0.7,
|
||||
},
|
||||
]
|
||||
_INPUT_TOKENS: Final = 367
|
||||
_OUTPUT_TOKENS: Final = 3
|
||||
_SYSTEM_ONE_USAGE: Final[dict[str, JsonValue]] = {"input_tokens": _INPUT_TOKENS, "output_tokens": _OUTPUT_TOKENS}
|
||||
_USAGE: Final[dict[str, JsonValue]] = {
|
||||
"input_tokens": _INPUT_TOKENS,
|
||||
"input_tokens_details": {"cached_tokens": 0, "cache_write_tokens": 0},
|
||||
"output_tokens": _OUTPUT_TOKENS,
|
||||
"output_tokens_details": {"reasoning_tokens": 0},
|
||||
"total_tokens": _INPUT_TOKENS + _OUTPUT_TOKENS,
|
||||
}
|
||||
_SDK_QUESTIONS: Final = TypeAdapter(list[dict[str, object]]).validate_python(_QUESTIONS)
|
||||
_SPEND_QUERY: Final = (
|
||||
"SELECT spend, status, call_type, model_group, custom_llm_provider, api_base, prompt_tokens, completion_tokens, "
|
||||
'request_tags FROM "LiteLLM_SpendLogs" WHERE request_id = %s'
|
||||
)
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class _Provider:
|
||||
name: str
|
||||
model: str
|
||||
path: str
|
||||
body_model: str
|
||||
api_key: str | None
|
||||
wraps_result: bool
|
||||
cost_map_key: str | None
|
||||
speaks_openai: bool = False
|
||||
|
||||
def upstream_body(self) -> dict[str, JsonValue]:
|
||||
if self.speaks_openai:
|
||||
return {"model": self.body_model, "input": _INPUT, "questions": _QUESTIONS}
|
||||
return {"model": self.body_model, "state": _INPUT, "questions": _SYSTEM_ONE_QUESTIONS}
|
||||
|
||||
def upstream_reply(self) -> dict[str, JsonValue]:
|
||||
if self.speaks_openai:
|
||||
return {"model": self.body_model, "answers": _ANSWERS, "usage": _USAGE}
|
||||
answer: Final[dict[str, JsonValue]] = {
|
||||
"model": self.body_model,
|
||||
"answers": _SYSTEM_ONE_ANSWERS,
|
||||
"usage": _SYSTEM_ONE_USAGE,
|
||||
}
|
||||
return {"result": answer, "success": True} if self.wraps_result else answer
|
||||
|
||||
def litellm_response(self) -> dict[str, JsonValue]:
|
||||
return {"model": self.body_model, "answers": _ANSWERS, "usage": _USAGE}
|
||||
|
||||
|
||||
_PROVIDERS: Final = (
|
||||
_Provider(
|
||||
"perplexity",
|
||||
"perplexity/pplx-decider-v1-27b",
|
||||
"/v1/decisions",
|
||||
"pplx-decider-v1-27b",
|
||||
_API_KEY,
|
||||
False,
|
||||
"perplexity/pplx-decider-v1-27b",
|
||||
),
|
||||
_Provider("typesafe", "typesafe/jev-1.13.0", "/v1/systemone", "jev-1.13.0", _API_KEY, False, "typesafe/jev-1.13.0"),
|
||||
_Provider(
|
||||
"openrouter",
|
||||
"openrouter/typesafe/jev-1.13",
|
||||
"/alpha/decisions",
|
||||
"typesafe/jev-1.13",
|
||||
_API_KEY,
|
||||
False,
|
||||
"openrouter/typesafe/jev-1.13",
|
||||
),
|
||||
_Provider(
|
||||
"strands_decider", "strands_decider/systemone-decider", "/v1/systemone", "systemone-decider", None, False, None
|
||||
),
|
||||
_Provider(
|
||||
"cloudflare",
|
||||
"cloudflare/clef",
|
||||
"/ai/run/@cf/cloudflare/clef",
|
||||
"clef",
|
||||
_API_KEY,
|
||||
True,
|
||||
"cloudflare/@cf/cloudflare/clef",
|
||||
),
|
||||
_Provider("hosted_vllm", "hosted_vllm/Qwen/Qwen3-0.6B", "/v1/systemone", "Qwen/Qwen3-0.6B", None, False, None),
|
||||
_Provider(
|
||||
"databricks",
|
||||
"databricks/databricks-openjev-qwen35-4b",
|
||||
"/databricks-openjev-qwen35-4b/invocations",
|
||||
"databricks-openjev-qwen35-4b",
|
||||
_API_KEY,
|
||||
False,
|
||||
None,
|
||||
),
|
||||
_Provider(
|
||||
"azure_ai", "azure_ai/decision-1", "/providers/microsoft/v1/systemone", "decision-1", _API_KEY, False, None
|
||||
),
|
||||
_Provider("openai", "openai/gpt-6-luna", "/v1/decisions", "gpt-6-luna", _API_KEY, False, "gpt-6-luna", True),
|
||||
)
|
||||
_SYSTEM_ONE_PROVIDERS: Final = tuple(provider for provider in _PROVIDERS if not provider.speaks_openai)
|
||||
_PERPLEXITY: Final = _PROVIDERS[0]
|
||||
_TYPESAFE: Final = _PROVIDERS[1]
|
||||
_OPENAI: Final = next(provider for provider in _PROVIDERS if provider.speaks_openai)
|
||||
_PREDICATE: Final[dict[str, JsonValue]] = {"type": "predicate", "name": "q", "instructions": "Is it?"}
|
||||
_INVALID_BODIES: Final[tuple[tuple[str, dict[str, JsonValue]], ...]] = (
|
||||
("missing questions", {"input": _INPUT}),
|
||||
("missing input", {"questions": _QUESTIONS}),
|
||||
("numeric input", {"input": 5, "questions": _QUESTIONS}),
|
||||
("assistant message input", {"input": [{"role": "assistant", "content": "hi"}], "questions": _QUESTIONS}),
|
||||
("empty questions", {"input": _INPUT, "questions": []}),
|
||||
("questions as a map", {"input": _INPUT, "questions": {"q": _PREDICATE}}),
|
||||
("predicate without instructions", {"input": _INPUT, "questions": [{"type": "predicate", "name": "q"}]}),
|
||||
(
|
||||
"choice with one choice",
|
||||
{"input": _INPUT, "questions": [{**_PREDICATE, "type": "choice", "choices": [{"value": "only"}]}]},
|
||||
),
|
||||
(
|
||||
"score with one level",
|
||||
{"input": _INPUT, "questions": [{**_PREDICATE, "type": "score", "levels": [{"label": "only"}]}]},
|
||||
),
|
||||
("unknown question type", {"input": _INPUT, "questions": [{**_PREDICATE, "type": "ranking"}]}),
|
||||
("numeric safety_identifier", {"input": _INPUT, "questions": _QUESTIONS, "safety_identifier": 7}),
|
||||
("list safety_identifier", {"input": _INPUT, "questions": _QUESTIONS, "safety_identifier": [_SAFETY_IDENTIFIER]}),
|
||||
)
|
||||
_IMAGE_INPUT: Final[list[JsonValue]] = [
|
||||
{
|
||||
"role": "user",
|
||||
"content": [
|
||||
{"type": "input_text", "text": _INPUT},
|
||||
{"type": "input_image", "image_url": "data:image/png;base64,iVBORw0KGgo=", "detail": "auto"},
|
||||
],
|
||||
}
|
||||
]
|
||||
_OPENAI_IMAGE_MESSAGES: Final[list[JsonValue]] = [
|
||||
{
|
||||
"role": "user",
|
||||
"type": "message",
|
||||
"content": [
|
||||
{"type": "input_text", "text": _INPUT},
|
||||
{"type": "input_image", "image_url": "data:image/png;base64,iVBORw0KGgo=", "detail": "auto"},
|
||||
],
|
||||
}
|
||||
]
|
||||
_MESSAGE_LIST_INPUT: Final[list[JsonValue]] = [
|
||||
{"role": "user", "content": _INPUT},
|
||||
{"role": "user", "content": [{"type": "input_text", "text": _FOLLOW_UP}]},
|
||||
]
|
||||
|
||||
|
||||
def _response_cost(response: httpx.Response) -> float:
|
||||
return float(response.headers["x-litellm-response-cost"]) if "x-litellm-response-cost" in response.headers else 0.0
|
||||
|
||||
|
||||
def _provider_id(provider: _Provider) -> str:
|
||||
return provider.name
|
||||
|
||||
|
||||
def _number(value: JsonValue) -> float:
|
||||
assert isinstance(value, (int, float)) and not isinstance(value, bool), value
|
||||
return float(value)
|
||||
|
||||
|
||||
def _expected_spend(cost_map_key: str | None) -> float:
|
||||
if cost_map_key is None:
|
||||
return 0.0
|
||||
cost_map: Final = _JSON_OBJECT.validate_json(Path("model_prices_and_context_window.json").read_bytes())
|
||||
prices: Final = object_value(cost_map[cost_map_key])
|
||||
return _INPUT_TOKENS * _number(prices["input_cost_per_token"]) + _OUTPUT_TOKENS * _number(
|
||||
prices["output_cost_per_token"]
|
||||
)
|
||||
|
||||
|
||||
def _register(scenario: Scenario, body: dict[str, JsonValue], *, status: int = 200) -> ScenarioHandle:
|
||||
handle: Final = register_scenario(
|
||||
f"decisions-{uuid.uuid4().hex[:12]}", JsonResponse(content_type="application/json", body=body, status=status)
|
||||
)
|
||||
scenario.cleanups.callback(delete_scenario, handle)
|
||||
return handle
|
||||
|
||||
|
||||
def _deployment(scenario: Scenario, handle: ScenarioHandle, provider: _Provider, *, drop_params: bool = False) -> str:
|
||||
if drop_params:
|
||||
return scenario.model(
|
||||
model=provider.model, api_base=handle.api_base(), api_key=provider.api_key, drop_params=True
|
||||
)
|
||||
return scenario.model(model=provider.model, api_base=handle.api_base(), api_key=provider.api_key)
|
||||
|
||||
|
||||
def _decide(gateway: Gateway, model: str, *, route: str = "/v1/decisions", **extra: JsonValue) -> httpx.Response:
|
||||
return gateway.request("POST", route, {"model": model, "input": _INPUT, "questions": _QUESTIONS, **extra})
|
||||
|
||||
|
||||
def _decide_in_system_one_format(gateway: Gateway, model: str, **extra: JsonValue) -> httpx.Response:
|
||||
return gateway.request(
|
||||
"POST", "/v1/systemone", {"model": model, "state": _INPUT, "questions": _SYSTEM_ONE_QUESTIONS, **extra}
|
||||
)
|
||||
|
||||
|
||||
def _observed_requests(gateway: Gateway) -> tuple[dict[str, JsonValue], ...]:
|
||||
with httpx.Client(base_url=gateway.upstream_url, timeout=5, trust_env=False) as upstream:
|
||||
observations: Final = _JSON_OBJECT.validate_json(upstream.get("/__observations").content)["requests"]
|
||||
assert isinstance(observations, list), observations
|
||||
return tuple(map(object_value, observations))
|
||||
|
||||
|
||||
def _calls_to(requests: Sequence[dict[str, JsonValue]], handle: ScenarioHandle) -> list[dict[str, JsonValue]]:
|
||||
return [request for request in requests if string_value(request["path"]).startswith(f"/{handle.scenario_id}/")]
|
||||
|
||||
|
||||
def _upstream_calls(gateway: Gateway, handle: ScenarioHandle) -> list[dict[str, JsonValue]]:
|
||||
return _calls_to(_observed_requests(gateway), handle)
|
||||
|
||||
|
||||
def _spend_row(call_id: str) -> dict[str, JsonValue]:
|
||||
rows: Final = eventually(lambda: read_rows(_SPEND_QUERY, (call_id,)), lambda found: len(found) == 1, seconds=70)
|
||||
return rows[0]
|
||||
|
||||
|
||||
def _assert_refused_as_invalid(gateway: Gateway, model: str, label: str, body: dict[str, JsonValue]) -> None:
|
||||
response: Final = gateway.request("POST", "/v1/decisions", {"model": model, **body})
|
||||
assert response.status_code == 400, (label, response.text)
|
||||
assert "Invalid Decisions request" in response.text, (label, response.text)
|
||||
|
||||
|
||||
def _assert_refused_for_safety_identifier(response: httpx.Response) -> None:
|
||||
assert response.status_code == 400, response.text
|
||||
assert "safety_identifier" in response.text and "drop_params" in response.text, response.text
|
||||
|
||||
|
||||
def _drop_params_config(directory: Path, api_base: str) -> Path:
|
||||
base: Final = _JSON_OBJECT.validate_python(yaml.safe_load(Path("tests/integration/proxy_config.yaml").read_text()))
|
||||
config: Final = {
|
||||
**base,
|
||||
"litellm_settings": {**object_value(base["litellm_settings"]), "drop_params": True},
|
||||
"model_list": [
|
||||
{
|
||||
"model_name": _DROPPING_DEPLOYMENT,
|
||||
"litellm_params": {"model": _PERPLEXITY.model, "api_base": api_base, "api_key": _API_KEY},
|
||||
}
|
||||
],
|
||||
}
|
||||
path: Final = directory / "decisions-drop-params.yaml"
|
||||
path.write_text(yaml.safe_dump(config))
|
||||
return path
|
||||
|
||||
|
||||
@pytest.mark.parametrize("provider", _PROVIDERS, ids=_provider_id)
|
||||
def test_each_provider_gets_its_own_path_key_and_body_and_is_billed_from_the_cost_map(
|
||||
gateway: Gateway, provider: _Provider
|
||||
) -> None:
|
||||
expected_spend: Final = _expected_spend(provider.cost_map_key)
|
||||
with gateway.scenario() as scenario:
|
||||
handle: Final = _register(scenario, provider.upstream_reply())
|
||||
model: Final = _deployment(scenario, handle, provider)
|
||||
response: Final = _decide(gateway, model)
|
||||
assert response.status_code == 200, response.text
|
||||
assert response.json() == provider.litellm_response()
|
||||
assert response.headers["x-litellm-model-group"] == model
|
||||
assert math.isclose(_response_cost(response), expected_spend, rel_tol=1e-9)
|
||||
(call,) = _upstream_calls(gateway, handle)
|
||||
assert call["path"] == f"/{handle.scenario_id}{provider.path}"
|
||||
assert call["authorization"] == (f"Bearer {provider.api_key}" if provider.api_key else "")
|
||||
assert call["body"] == provider.upstream_body()
|
||||
row: Final = _spend_row(response.headers["x-litellm-call-id"])
|
||||
assert (
|
||||
row["status"],
|
||||
row["call_type"],
|
||||
row["custom_llm_provider"],
|
||||
row["model_group"],
|
||||
row["api_base"],
|
||||
row["prompt_tokens"],
|
||||
row["completion_tokens"],
|
||||
) == (
|
||||
"success",
|
||||
"adecisions",
|
||||
provider.name,
|
||||
model,
|
||||
f"{handle.api_base()}{provider.path}",
|
||||
_INPUT_TOKENS,
|
||||
_OUTPUT_TOKENS,
|
||||
)
|
||||
assert math.isclose(_number(row["spend"]), expected_spend, rel_tol=1e-9), row
|
||||
|
||||
|
||||
def test_the_unversioned_alias_serves_the_same_request(gateway: Gateway) -> None:
|
||||
with gateway.scenario() as scenario:
|
||||
handle: Final = _register(scenario, _PERPLEXITY.upstream_reply())
|
||||
model: Final = _deployment(scenario, handle, _PERPLEXITY)
|
||||
response: Final = _decide(gateway, model, route="/decisions")
|
||||
assert response.status_code == 200, response.text
|
||||
assert response.json() == _PERPLEXITY.litellm_response()
|
||||
(call,) = _upstream_calls(gateway, handle)
|
||||
assert call["body"] == _PERPLEXITY.upstream_body()
|
||||
|
||||
|
||||
def test_repeated_identical_requests_each_reach_the_upstream_and_are_each_billed(gateway: Gateway) -> None:
|
||||
with gateway.scenario() as scenario:
|
||||
handle: Final = _register(scenario, _PERPLEXITY.upstream_reply())
|
||||
model: Final = _deployment(scenario, handle, _PERPLEXITY)
|
||||
responses: Final = tuple(_decide(gateway, model) for _ in range(2))
|
||||
assert [response.status_code for response in responses] == [200, 200], [r.text for r in responses]
|
||||
call_ids: Final = tuple(response.headers["x-litellm-call-id"] for response in responses)
|
||||
assert len(set(call_ids)) == 2, call_ids
|
||||
assert len(_upstream_calls(gateway, handle)) == 2
|
||||
for call_id in call_ids:
|
||||
assert _spend_row(call_id)["status"] == "success"
|
||||
|
||||
|
||||
async def test_sdk_sync_and_async_clients_send_the_same_request(gateway: Gateway) -> None:
|
||||
with gateway.scenario() as scenario:
|
||||
handle: Final = _register(scenario, _TYPESAFE.upstream_reply())
|
||||
synchronous: Final = litellm.decisions(
|
||||
model=_TYPESAFE.model, input=_INPUT, questions=_SDK_QUESTIONS, api_base=handle.api_base(), api_key=_API_KEY
|
||||
)
|
||||
asynchronous: Final = await litellm.adecisions(
|
||||
model=_TYPESAFE.model, input=_INPUT, questions=_SDK_QUESTIONS, api_base=handle.api_base(), api_key=_API_KEY
|
||||
)
|
||||
for response in (synchronous, asynchronous):
|
||||
assert response.model_dump(mode="json") == _TYPESAFE.litellm_response()
|
||||
calls: Final = _upstream_calls(gateway, handle)
|
||||
assert len(calls) == 2, calls
|
||||
for call in calls:
|
||||
assert call["path"] == f"/{handle.scenario_id}{_TYPESAFE.path}"
|
||||
assert call["authorization"] == f"Bearer {_API_KEY}"
|
||||
assert call["body"] == _TYPESAFE.upstream_body()
|
||||
|
||||
|
||||
def test_gateway_only_fields_stay_at_the_gateway_and_tags_reach_the_spend_log(gateway: Gateway) -> None:
|
||||
tag: Final = f"decisions-openai-format-{uuid.uuid4().hex[:8]}"
|
||||
with gateway.scenario() as scenario:
|
||||
handle: Final = _register(scenario, _PERPLEXITY.upstream_reply())
|
||||
model: Final = _deployment(scenario, handle, _PERPLEXITY)
|
||||
response: Final = _decide(
|
||||
gateway, model, user="auditor", num_retries=0, temperature=0.2, metadata={"tags": [tag]}
|
||||
)
|
||||
assert response.status_code == 200, response.text
|
||||
(call,) = _upstream_calls(gateway, handle)
|
||||
assert call["body"] == _PERPLEXITY.upstream_body()
|
||||
row: Final = _spend_row(response.headers["x-litellm-call-id"])
|
||||
tags: Final = row["request_tags"]
|
||||
assert isinstance(tags, list) and tag in tags, row
|
||||
|
||||
|
||||
def test_invalid_bodies_are_refused_at_the_gateway_without_an_upstream_call(gateway: Gateway) -> None:
|
||||
with gateway.scenario() as scenario:
|
||||
handle: Final = _register(scenario, _PERPLEXITY.upstream_reply())
|
||||
model: Final = _deployment(scenario, handle, _PERPLEXITY)
|
||||
for label, body in _INVALID_BODIES:
|
||||
_assert_refused_as_invalid(gateway, model, label, body)
|
||||
assert _upstream_calls(gateway, handle) == []
|
||||
|
||||
|
||||
@pytest.mark.parametrize("provider", _SYSTEM_ONE_PROVIDERS, ids=_provider_id)
|
||||
def test_system_one_providers_refuse_images_at_the_gateway_without_an_upstream_call(
|
||||
gateway: Gateway, provider: _Provider
|
||||
) -> None:
|
||||
with gateway.scenario() as scenario:
|
||||
handle: Final = _register(scenario, provider.upstream_reply())
|
||||
model: Final = _deployment(scenario, handle, provider)
|
||||
response: Final = _decide(gateway, model, input=_IMAGE_INPUT)
|
||||
assert response.status_code == 400, response.text
|
||||
assert "input_image" in response.text
|
||||
assert _upstream_calls(gateway, handle) == []
|
||||
|
||||
|
||||
def test_openai_forwards_image_input_in_its_own_message_shape(gateway: Gateway) -> None:
|
||||
with gateway.scenario() as scenario:
|
||||
handle: Final = _register(scenario, _OPENAI.upstream_reply())
|
||||
model: Final = _deployment(scenario, handle, _OPENAI)
|
||||
response: Final = _decide(gateway, model, input=_IMAGE_INPUT)
|
||||
assert response.status_code == 200, response.text
|
||||
(call,) = _upstream_calls(gateway, handle)
|
||||
assert call["body"] == {"model": _OPENAI.body_model, "input": _OPENAI_IMAGE_MESSAGES, "questions": _QUESTIONS}
|
||||
|
||||
|
||||
def test_a_message_list_input_reaches_a_system_one_provider_as_its_flattened_text(gateway: Gateway) -> None:
|
||||
with gateway.scenario() as scenario:
|
||||
handle: Final = _register(scenario, _PERPLEXITY.upstream_reply())
|
||||
model: Final = _deployment(scenario, handle, _PERPLEXITY)
|
||||
response: Final = _decide(gateway, model, input=_MESSAGE_LIST_INPUT)
|
||||
assert response.status_code == 200, response.text
|
||||
assert response.json() == _PERPLEXITY.litellm_response()
|
||||
(call,) = _upstream_calls(gateway, handle)
|
||||
assert call["body"] == {
|
||||
"model": _PERPLEXITY.body_model,
|
||||
"state": f"{_INPUT}\n\n{_FOLLOW_UP}",
|
||||
"questions": _SYSTEM_ONE_QUESTIONS,
|
||||
}
|
||||
row: Final = _spend_row(response.headers["x-litellm-call-id"])
|
||||
assert (row["status"], row["call_type"], row["prompt_tokens"], row["completion_tokens"]) == (
|
||||
"success",
|
||||
"adecisions",
|
||||
_INPUT_TOKENS,
|
||||
_OUTPUT_TOKENS,
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("provider", _SYSTEM_ONE_PROVIDERS, ids=_provider_id)
|
||||
def test_safety_identifier_is_refused_by_system_one_providers_unless_the_deployment_drops_params(
|
||||
gateway: Gateway, provider: _Provider
|
||||
) -> None:
|
||||
with gateway.scenario() as scenario:
|
||||
handle: Final = _register(scenario, provider.upstream_reply())
|
||||
strict: Final = _deployment(scenario, handle, provider)
|
||||
refused: Final = _decide(gateway, strict, safety_identifier=_SAFETY_IDENTIFIER)
|
||||
_assert_refused_for_safety_identifier(refused)
|
||||
assert _upstream_calls(gateway, handle) == []
|
||||
|
||||
dropping: Final = _deployment(scenario, handle, provider, drop_params=True)
|
||||
accepted: Final = _decide(gateway, dropping, safety_identifier=_SAFETY_IDENTIFIER)
|
||||
assert accepted.status_code == 200, accepted.text
|
||||
assert accepted.json() == provider.litellm_response()
|
||||
(call,) = _upstream_calls(gateway, handle)
|
||||
assert call["body"] == provider.upstream_body()
|
||||
row: Final = _spend_row(accepted.headers["x-litellm-call-id"])
|
||||
assert (row["status"], row["call_type"], row["model_group"]) == ("success", "adecisions", dropping)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("value", ("", "x" * 5000), ids=("empty", "5kb"))
|
||||
def test_every_string_safety_identifier_is_refused_or_dropped_like_the_usual_one(gateway: Gateway, value: str) -> None:
|
||||
with gateway.scenario() as scenario:
|
||||
handle: Final = _register(scenario, _PERPLEXITY.upstream_reply())
|
||||
strict: Final = _deployment(scenario, handle, _PERPLEXITY)
|
||||
_assert_refused_for_safety_identifier(_decide(gateway, strict, safety_identifier=value))
|
||||
dropping: Final = _deployment(scenario, handle, _PERPLEXITY, drop_params=True)
|
||||
accepted: Final = _decide(gateway, dropping, safety_identifier=value)
|
||||
assert accepted.status_code == 200, accepted.text
|
||||
(call,) = _upstream_calls(gateway, handle)
|
||||
assert call["body"] == _PERPLEXITY.upstream_body()
|
||||
|
||||
|
||||
def test_a_request_body_drop_params_drops_the_safety_identifier_like_chat(gateway: Gateway) -> None:
|
||||
with gateway.scenario() as scenario:
|
||||
handle: Final = _register(scenario, _PERPLEXITY.upstream_reply())
|
||||
strict: Final = _deployment(scenario, handle, _PERPLEXITY)
|
||||
response: Final = _decide(gateway, strict, safety_identifier=_SAFETY_IDENTIFIER, drop_params=True)
|
||||
assert response.status_code == 200, response.text
|
||||
assert response.json() == _PERPLEXITY.litellm_response()
|
||||
(call,) = _upstream_calls(gateway, handle)
|
||||
assert call["body"] == _PERPLEXITY.upstream_body()
|
||||
|
||||
|
||||
def test_openai_keeps_the_safety_identifier_on_the_wire_without_drop_params(gateway: Gateway) -> None:
|
||||
with gateway.scenario() as scenario:
|
||||
handle: Final = _register(scenario, _OPENAI.upstream_reply())
|
||||
model: Final = _deployment(scenario, handle, _OPENAI)
|
||||
response: Final = _decide(gateway, model, safety_identifier=_SAFETY_IDENTIFIER)
|
||||
assert response.status_code == 200, response.text
|
||||
assert response.json() == _OPENAI.litellm_response()
|
||||
(call,) = _upstream_calls(gateway, handle)
|
||||
assert call["body"] == {**_OPENAI.upstream_body(), "safety_identifier": _SAFETY_IDENTIFIER}
|
||||
|
||||
|
||||
def test_a_system_one_format_safety_identifier_is_refused_unless_the_deployment_drops_params(gateway: Gateway) -> None:
|
||||
with gateway.scenario() as scenario:
|
||||
handle: Final = _register(scenario, _PERPLEXITY.upstream_reply())
|
||||
strict: Final = _deployment(scenario, handle, _PERPLEXITY)
|
||||
refused: Final = _decide_in_system_one_format(gateway, strict, safety_identifier=_SAFETY_IDENTIFIER)
|
||||
_assert_refused_for_safety_identifier(refused)
|
||||
assert _upstream_calls(gateway, handle) == []
|
||||
|
||||
dropping: Final = _deployment(scenario, handle, _PERPLEXITY, drop_params=True)
|
||||
accepted: Final = _decide_in_system_one_format(gateway, dropping, safety_identifier=_SAFETY_IDENTIFIER)
|
||||
assert accepted.status_code == 200, accepted.text
|
||||
assert accepted.json()["answers"] == _SYSTEM_ONE_ANSWERS
|
||||
(call,) = _upstream_calls(gateway, handle)
|
||||
assert call["body"] == _PERPLEXITY.upstream_body()
|
||||
|
||||
|
||||
def test_openai_keeps_a_system_one_format_safety_identifier_on_the_wire(gateway: Gateway) -> None:
|
||||
with gateway.scenario() as scenario:
|
||||
handle: Final = _register(scenario, _OPENAI.upstream_reply())
|
||||
model: Final = _deployment(scenario, handle, _OPENAI)
|
||||
response: Final = _decide_in_system_one_format(gateway, model, safety_identifier=_SAFETY_IDENTIFIER)
|
||||
assert response.status_code == 200, response.text
|
||||
(call,) = _upstream_calls(gateway, handle)
|
||||
assert call["body"] == {**_OPENAI.upstream_body(), "safety_identifier": _SAFETY_IDENTIFIER}
|
||||
|
||||
|
||||
@pytest.mark.parametrize("decide", (_decide, _decide_in_system_one_format), ids=("openai_format", "system_one_format"))
|
||||
@pytest.mark.parametrize("value", (7, [_SAFETY_IDENTIFIER]), ids=("numeric", "list"))
|
||||
def test_a_non_string_safety_identifier_is_refused_as_invalid_unless_the_deployment_drops_params(
|
||||
gateway: Gateway, value: JsonValue, decide: Callable[..., httpx.Response]
|
||||
) -> None:
|
||||
with gateway.scenario() as scenario:
|
||||
handle: Final = _register(scenario, _OPENAI.upstream_reply())
|
||||
strict: Final = _deployment(scenario, handle, _OPENAI)
|
||||
refused: Final = decide(gateway, strict, safety_identifier=value)
|
||||
assert refused.status_code == 400, refused.text
|
||||
assert "Invalid Decisions request" in refused.text, refused.text
|
||||
assert _upstream_calls(gateway, handle) == []
|
||||
|
||||
dropping: Final = _deployment(scenario, handle, _OPENAI, drop_params=True)
|
||||
accepted: Final = decide(gateway, dropping, safety_identifier=value)
|
||||
assert accepted.status_code == 200, accepted.text
|
||||
(call,) = _upstream_calls(gateway, handle)
|
||||
assert call["body"] == _OPENAI.upstream_body()
|
||||
|
||||
|
||||
def test_litellm_settings_drop_params_drops_the_safety_identifier_for_a_strict_deployment(
|
||||
gateway: Gateway, tmp_path: Path
|
||||
) -> None:
|
||||
with gateway.scenario() as scenario:
|
||||
handle: Final = _register(scenario, _PERPLEXITY.upstream_reply())
|
||||
config: Final = _drop_params_config(tmp_path, handle.api_base())
|
||||
with owned_proxy_process(gateway, tmp_path, {}, config=config) as owned:
|
||||
response: Final = _decide(owned.gateway, _DROPPING_DEPLOYMENT, safety_identifier=_SAFETY_IDENTIFIER)
|
||||
assert response.status_code == 200, response.text
|
||||
assert response.json() == _PERPLEXITY.litellm_response()
|
||||
(call,) = _upstream_calls(gateway, handle)
|
||||
assert call["body"] == _PERPLEXITY.upstream_body()
|
||||
|
|
@ -6,6 +6,103 @@ from integration.translation.case import TranslationTestCase
|
|||
"""
|
||||
CLEF_TEST_CASE: Final = TranslationTestCase(
|
||||
scenario="basic",
|
||||
litellm_endpoint="/v1/decisions",
|
||||
litellm_request={
|
||||
"model": "cloudflare/@cf/cloudflare/clef",
|
||||
"input": "Ticket (billing): The export job hangs at 99% and never finishes",
|
||||
"questions": [
|
||||
{"type": "predicate", "name": "defect", "instructions": "Is this a defect?"},
|
||||
{
|
||||
"type": "choice",
|
||||
"name": "severity",
|
||||
"instructions": "How severe is it?",
|
||||
"choices": [
|
||||
{"value": "low", "description": "cosmetic"},
|
||||
{"value": "high", "description": "blocks users"},
|
||||
],
|
||||
},
|
||||
{
|
||||
"type": "score",
|
||||
"name": "confidence",
|
||||
"instructions": "How sure are you?",
|
||||
"levels": [{"label": "unsure"}, {"label": "sure"}],
|
||||
},
|
||||
],
|
||||
"cache": {"no-cache": True},
|
||||
},
|
||||
expected_provider_endpoint="/ai/run/@cf/cloudflare/clef",
|
||||
expected_provider_headers={"authorization": "Bearer synthetic-cloudflare-key", "content-type": "application/json"},
|
||||
expected_provider_request={
|
||||
"model": "clef",
|
||||
"state": "Ticket (billing): The export job hangs at 99% and never finishes",
|
||||
"questions": {
|
||||
"defect": {"type": "noul", "instructions": "Is this a defect?"},
|
||||
"severity": {
|
||||
"type": "choice",
|
||||
"instructions": "How severe is it?",
|
||||
"criteria": {"low": "cosmetic", "high": "blocks users"},
|
||||
},
|
||||
"confidence": {"type": "score", "instructions": "How sure are you?", "criteria": ["unsure", "sure"]},
|
||||
},
|
||||
},
|
||||
mock_provider_response={
|
||||
"result": {
|
||||
"model": "clef",
|
||||
"answers": {
|
||||
"defect": {"type": "noul", "noul": 0.9345},
|
||||
"severity": {
|
||||
"type": "choice",
|
||||
"choice": "high",
|
||||
"confidence": 0.8067,
|
||||
"probabilities": {"low": 0.0509, "high": 0.9491},
|
||||
},
|
||||
"confidence": {
|
||||
"type": "score",
|
||||
"score": 0.9036,
|
||||
"confidence": 0.6515,
|
||||
"legend": {"0": "unsure", "1": "sure"},
|
||||
"probabilities": {"0": 0.0964, "1": 0.9036},
|
||||
},
|
||||
},
|
||||
"usage": {"input_tokens": 290, "output_tokens": 0},
|
||||
},
|
||||
"success": True,
|
||||
"errors": [],
|
||||
"messages": [],
|
||||
},
|
||||
expected_litellm_response={
|
||||
"model": "clef",
|
||||
"answers": [
|
||||
{"type": "predicate", "name": "defect", "probability": 0.9345},
|
||||
{
|
||||
"type": "choice",
|
||||
"name": "severity",
|
||||
"choice": "high",
|
||||
"probabilities": [{"value": "low", "probability": 0.0509}, {"value": "high", "probability": 0.9491}],
|
||||
"confidence": 0.8067,
|
||||
},
|
||||
{
|
||||
"type": "score",
|
||||
"name": "confidence",
|
||||
"score": 0.9036,
|
||||
"probabilities": [
|
||||
{"value": 0, "label": "unsure", "probability": 0.0964},
|
||||
{"value": 1, "label": "sure", "probability": 0.9036},
|
||||
],
|
||||
"confidence": 0.6515,
|
||||
},
|
||||
],
|
||||
"usage": {
|
||||
"input_tokens": 290,
|
||||
"input_tokens_details": {"cached_tokens": 0, "cache_write_tokens": 0},
|
||||
"output_tokens": 0,
|
||||
"output_tokens_details": {"reasoning_tokens": 0},
|
||||
"total_tokens": 290,
|
||||
},
|
||||
},
|
||||
)
|
||||
CLEF_SYSTEMONE_TEST_CASE: Final = TranslationTestCase(
|
||||
scenario="systemone",
|
||||
litellm_endpoint="/v1/systemone",
|
||||
litellm_request={
|
||||
"model": "cloudflare/@cf/cloudflare/clef",
|
||||
|
|
|
|||
|
|
@ -6,6 +6,100 @@ from integration.translation.case import TranslationTestCase
|
|||
"""
|
||||
TYPESAFE_JEV_1_13_TEST_CASE: Final = TranslationTestCase(
|
||||
scenario="basic",
|
||||
litellm_endpoint="/v1/decisions",
|
||||
litellm_request={
|
||||
"model": "openrouter/typesafe/jev-1.13",
|
||||
"input": "Ticket (billing): The export job hangs at 99% and never finishes",
|
||||
"questions": [
|
||||
{"type": "predicate", "name": "defect", "instructions": "Is this a defect?"},
|
||||
{
|
||||
"type": "choice",
|
||||
"name": "severity",
|
||||
"instructions": "How severe is it?",
|
||||
"choices": [
|
||||
{"value": "low", "description": "cosmetic"},
|
||||
{"value": "high", "description": "blocks users"},
|
||||
],
|
||||
},
|
||||
{
|
||||
"type": "score",
|
||||
"name": "confidence",
|
||||
"instructions": "How sure are you?",
|
||||
"levels": [{"label": "unsure"}, {"label": "sure"}],
|
||||
},
|
||||
],
|
||||
"cache": {"no-cache": True},
|
||||
},
|
||||
expected_provider_endpoint="/alpha/decisions",
|
||||
expected_provider_headers={"authorization": "Bearer synthetic-openrouter-key", "content-type": "application/json"},
|
||||
expected_provider_request={
|
||||
"model": "typesafe/jev-1.13",
|
||||
"state": "Ticket (billing): The export job hangs at 99% and never finishes",
|
||||
"questions": {
|
||||
"defect": {"type": "noul", "instructions": "Is this a defect?"},
|
||||
"severity": {
|
||||
"type": "choice",
|
||||
"instructions": "How severe is it?",
|
||||
"criteria": {"low": "cosmetic", "high": "blocks users"},
|
||||
},
|
||||
"confidence": {"type": "score", "instructions": "How sure are you?", "criteria": ["unsure", "sure"]},
|
||||
},
|
||||
},
|
||||
mock_provider_response={
|
||||
"model": "typesafe/jev-1.13-20260917",
|
||||
"answers": {
|
||||
"defect": {"type": "noul", "noul": 0.81},
|
||||
"severity": {
|
||||
"type": "choice",
|
||||
"choice": "high",
|
||||
"confidence": 0.99,
|
||||
"probabilities": {"low": 0.01, "high": 0.99},
|
||||
},
|
||||
"confidence": {
|
||||
"type": "score",
|
||||
"score": 0.5,
|
||||
"confidence": 0,
|
||||
"legend": {"0": "unsure", "1": "sure"},
|
||||
"probabilities": {"0": 0.5, "1": 0.5},
|
||||
},
|
||||
},
|
||||
"usage": {"input_tokens": 377, "output_tokens": 62, "cost": 1.5834e-05},
|
||||
"id": "gen-dec-1791323839-GA15kY0nt34oiJ7srfki",
|
||||
"provider": "TypeSafe",
|
||||
},
|
||||
expected_litellm_response={
|
||||
"model": "typesafe/jev-1.13-20260917",
|
||||
"answers": [
|
||||
{"type": "predicate", "name": "defect", "probability": 0.81},
|
||||
{
|
||||
"type": "choice",
|
||||
"name": "severity",
|
||||
"choice": "high",
|
||||
"probabilities": [{"value": "low", "probability": 0.01}, {"value": "high", "probability": 0.99}],
|
||||
"confidence": 0.99,
|
||||
},
|
||||
{
|
||||
"type": "score",
|
||||
"name": "confidence",
|
||||
"score": 0.5,
|
||||
"probabilities": [
|
||||
{"value": 0, "label": "unsure", "probability": 0.5},
|
||||
{"value": 1, "label": "sure", "probability": 0.5},
|
||||
],
|
||||
"confidence": 0,
|
||||
},
|
||||
],
|
||||
"usage": {
|
||||
"input_tokens": 377,
|
||||
"input_tokens_details": {"cached_tokens": 0, "cache_write_tokens": 0},
|
||||
"output_tokens": 62,
|
||||
"output_tokens_details": {"reasoning_tokens": 0},
|
||||
"total_tokens": 439,
|
||||
},
|
||||
},
|
||||
)
|
||||
TYPESAFE_JEV_1_13_SYSTEMONE_TEST_CASE: Final = TranslationTestCase(
|
||||
scenario="systemone",
|
||||
litellm_endpoint="/v1/systemone",
|
||||
litellm_request={
|
||||
"model": "openrouter/typesafe/jev-1.13",
|
||||
|
|
|
|||
|
|
@ -6,6 +6,101 @@ from integration.translation.case import TranslationTestCase
|
|||
"""
|
||||
PPLX_DECIDER_V1_27B_TEST_CASE: Final = TranslationTestCase(
|
||||
scenario="basic",
|
||||
litellm_endpoint="/v1/decisions",
|
||||
litellm_request={
|
||||
"model": "perplexity/pplx-decider-v1-27b",
|
||||
"input": "Ticket (billing): The export job hangs at 99% and never finishes",
|
||||
"questions": [
|
||||
{"type": "predicate", "name": "defect", "instructions": "Is this a defect?"},
|
||||
{
|
||||
"type": "choice",
|
||||
"name": "severity",
|
||||
"instructions": "How severe is it?",
|
||||
"choices": [
|
||||
{"value": "low", "description": "cosmetic"},
|
||||
{"value": "high", "description": "blocks users"},
|
||||
],
|
||||
},
|
||||
{
|
||||
"type": "score",
|
||||
"name": "confidence",
|
||||
"instructions": "How sure are you?",
|
||||
"levels": [{"label": "unsure"}, {"label": "sure"}],
|
||||
},
|
||||
],
|
||||
"cache": {"no-cache": True},
|
||||
},
|
||||
expected_provider_endpoint="/v1/decisions",
|
||||
expected_provider_headers={"authorization": "Bearer synthetic-perplexity-key", "content-type": "application/json"},
|
||||
expected_provider_request={
|
||||
"model": "pplx-decider-v1-27b",
|
||||
"state": "Ticket (billing): The export job hangs at 99% and never finishes",
|
||||
"questions": {
|
||||
"defect": {"type": "noul", "instructions": "Is this a defect?"},
|
||||
"severity": {
|
||||
"type": "choice",
|
||||
"instructions": "How severe is it?",
|
||||
"criteria": {"low": "cosmetic", "high": "blocks users"},
|
||||
},
|
||||
"confidence": {"type": "score", "instructions": "How sure are you?", "criteria": ["unsure", "sure"]},
|
||||
},
|
||||
},
|
||||
mock_provider_response={
|
||||
"model": "pplx-decider-v1-27b",
|
||||
"answers": {
|
||||
"defect": {"type": "noul", "noul": 0.9989100737587077},
|
||||
"severity": {
|
||||
"type": "choice",
|
||||
"choice": "high",
|
||||
"confidence": 0.9964631215356778,
|
||||
"probabilities": {"low": 0.0017684392321610232, "high": 0.9982315607678389},
|
||||
},
|
||||
"confidence": {
|
||||
"type": "score",
|
||||
"score": 0.07367392327139817,
|
||||
"confidence": 0.8526521534572037,
|
||||
"legend": {"0": "unsure", "1": "sure"},
|
||||
"probabilities": {"0": 0.9263260767286018, "1": 0.07367392327139817},
|
||||
},
|
||||
},
|
||||
"usage": {"input_tokens": 318, "output_tokens": 3},
|
||||
},
|
||||
expected_litellm_response={
|
||||
"model": "pplx-decider-v1-27b",
|
||||
"answers": [
|
||||
{"type": "predicate", "name": "defect", "probability": 0.9989100737587077},
|
||||
{
|
||||
"type": "choice",
|
||||
"name": "severity",
|
||||
"choice": "high",
|
||||
"probabilities": [
|
||||
{"value": "low", "probability": 0.0017684392321610232},
|
||||
{"value": "high", "probability": 0.9982315607678389},
|
||||
],
|
||||
"confidence": 0.9964631215356778,
|
||||
},
|
||||
{
|
||||
"type": "score",
|
||||
"name": "confidence",
|
||||
"score": 0.07367392327139817,
|
||||
"probabilities": [
|
||||
{"value": 0, "label": "unsure", "probability": 0.9263260767286018},
|
||||
{"value": 1, "label": "sure", "probability": 0.07367392327139817},
|
||||
],
|
||||
"confidence": 0.8526521534572037,
|
||||
},
|
||||
],
|
||||
"usage": {
|
||||
"input_tokens": 318,
|
||||
"input_tokens_details": {"cached_tokens": 0, "cache_write_tokens": 0},
|
||||
"output_tokens": 3,
|
||||
"output_tokens_details": {"reasoning_tokens": 0},
|
||||
"total_tokens": 321,
|
||||
},
|
||||
},
|
||||
)
|
||||
PPLX_DECIDER_V1_27B_SYSTEMONE_TEST_CASE: Final = TranslationTestCase(
|
||||
scenario="systemone",
|
||||
litellm_endpoint="/v1/systemone",
|
||||
litellm_request={
|
||||
"model": "perplexity/pplx-decider-v1-27b",
|
||||
|
|
|
|||
|
|
@ -6,6 +6,40 @@ from integration.translation.case import TranslationTestCase
|
|||
"""
|
||||
STRANDS_DECIDER_2B_HOBSON_V19_TEST_CASE: Final = TranslationTestCase(
|
||||
scenario="basic",
|
||||
litellm_endpoint="/v1/decisions",
|
||||
litellm_request={
|
||||
"model": "strands_decider/strands-decider-2B-hobson-v19",
|
||||
"input": "Help! My payouts have been failing for 3 days!",
|
||||
"questions": [{"type": "predicate", "name": "is_urgent", "instructions": "Does this convey urgency?"}],
|
||||
"cache": {"no-cache": True},
|
||||
},
|
||||
expected_provider_endpoint="/v1/systemone",
|
||||
expected_provider_headers={"content-type": "application/json"},
|
||||
expected_provider_request={
|
||||
"model": "strands-decider-2B-hobson-v19",
|
||||
"state": "Help! My payouts have been failing for 3 days!",
|
||||
"questions": {"is_urgent": {"type": "noul", "instructions": "Does this convey urgency?"}},
|
||||
},
|
||||
mock_provider_response={
|
||||
"model": "strands-decider-2B-hobson-v19",
|
||||
"answers": {"is_urgent": {"type": "noul", "noul": 0.8277}},
|
||||
"usage": {"input_tokens": 86, "output_tokens": 1},
|
||||
"latency_ms": 140.03,
|
||||
},
|
||||
expected_litellm_response={
|
||||
"model": "strands-decider-2B-hobson-v19",
|
||||
"answers": [{"type": "predicate", "name": "is_urgent", "probability": 0.8277}],
|
||||
"usage": {
|
||||
"input_tokens": 86,
|
||||
"input_tokens_details": {"cached_tokens": 0, "cache_write_tokens": 0},
|
||||
"output_tokens": 1,
|
||||
"output_tokens_details": {"reasoning_tokens": 0},
|
||||
"total_tokens": 87,
|
||||
},
|
||||
},
|
||||
)
|
||||
STRANDS_DECIDER_2B_HOBSON_V19_SYSTEMONE_TEST_CASE: Final = TranslationTestCase(
|
||||
scenario="systemone",
|
||||
litellm_endpoint="/v1/systemone",
|
||||
litellm_request={
|
||||
"model": "strands_decider/strands-decider-2B-hobson-v19",
|
||||
|
|
|
|||
|
|
@ -6,6 +6,98 @@ from integration.translation.case import TranslationTestCase
|
|||
"""
|
||||
JEV_1_13_0_TEST_CASE: Final = TranslationTestCase(
|
||||
scenario="basic",
|
||||
litellm_endpoint="/v1/decisions",
|
||||
litellm_request={
|
||||
"model": "typesafe/jev-1.13.0",
|
||||
"input": "Ticket (billing): The export job hangs at 99% and never finishes",
|
||||
"questions": [
|
||||
{"type": "predicate", "name": "defect", "instructions": "Is this a defect?"},
|
||||
{
|
||||
"type": "choice",
|
||||
"name": "severity",
|
||||
"instructions": "How severe is it?",
|
||||
"choices": [
|
||||
{"value": "low", "description": "cosmetic"},
|
||||
{"value": "high", "description": "blocks users"},
|
||||
],
|
||||
},
|
||||
{
|
||||
"type": "score",
|
||||
"name": "confidence",
|
||||
"instructions": "How sure are you?",
|
||||
"levels": [{"label": "unsure"}, {"label": "sure"}],
|
||||
},
|
||||
],
|
||||
"cache": {"no-cache": True},
|
||||
},
|
||||
expected_provider_endpoint="/v1/systemone",
|
||||
expected_provider_headers={"authorization": "Bearer synthetic-typesafe-key", "content-type": "application/json"},
|
||||
expected_provider_request={
|
||||
"model": "jev-1.13.0",
|
||||
"state": "Ticket (billing): The export job hangs at 99% and never finishes",
|
||||
"questions": {
|
||||
"defect": {"type": "noul", "instructions": "Is this a defect?"},
|
||||
"severity": {
|
||||
"type": "choice",
|
||||
"instructions": "How severe is it?",
|
||||
"criteria": {"low": "cosmetic", "high": "blocks users"},
|
||||
},
|
||||
"confidence": {"type": "score", "instructions": "How sure are you?", "criteria": ["unsure", "sure"]},
|
||||
},
|
||||
},
|
||||
mock_provider_response={
|
||||
"model": "jev-1.13.0",
|
||||
"answers": {
|
||||
"defect": {"type": "noul", "noul": 0.78},
|
||||
"severity": {
|
||||
"type": "choice",
|
||||
"choice": "high",
|
||||
"confidence": 0.99,
|
||||
"probabilities": {"low": 0.01, "high": 0.99},
|
||||
},
|
||||
"confidence": {
|
||||
"type": "score",
|
||||
"score": 0.51,
|
||||
"confidence": 0.03,
|
||||
"legend": {"0": "unsure", "1": "sure"},
|
||||
"probabilities": {"0": 0.49, "1": 0.51},
|
||||
},
|
||||
},
|
||||
"usage": {"input_tokens": 377, "output_tokens": 62},
|
||||
},
|
||||
expected_litellm_response={
|
||||
"model": "jev-1.13.0",
|
||||
"answers": [
|
||||
{"type": "predicate", "name": "defect", "probability": 0.78},
|
||||
{
|
||||
"type": "choice",
|
||||
"name": "severity",
|
||||
"choice": "high",
|
||||
"probabilities": [{"value": "low", "probability": 0.01}, {"value": "high", "probability": 0.99}],
|
||||
"confidence": 0.99,
|
||||
},
|
||||
{
|
||||
"type": "score",
|
||||
"name": "confidence",
|
||||
"score": 0.51,
|
||||
"probabilities": [
|
||||
{"value": 0, "label": "unsure", "probability": 0.49},
|
||||
{"value": 1, "label": "sure", "probability": 0.51},
|
||||
],
|
||||
"confidence": 0.03,
|
||||
},
|
||||
],
|
||||
"usage": {
|
||||
"input_tokens": 377,
|
||||
"input_tokens_details": {"cached_tokens": 0, "cache_write_tokens": 0},
|
||||
"output_tokens": 62,
|
||||
"output_tokens_details": {"reasoning_tokens": 0},
|
||||
"total_tokens": 439,
|
||||
},
|
||||
},
|
||||
)
|
||||
JEV_1_13_0_SYSTEMONE_TEST_CASE: Final = TranslationTestCase(
|
||||
scenario="systemone",
|
||||
litellm_endpoint="/v1/systemone",
|
||||
litellm_request={
|
||||
"model": "typesafe/jev-1.13.0",
|
||||
|
|
|
|||
|
|
@ -2,10 +2,10 @@ import pytest
|
|||
from integration._support.client import Gateway
|
||||
from integration._support.provider import SharedProvider
|
||||
from integration.translation.case import TranslationTestCase
|
||||
from integration.translation.decisions.bases.cloudflare import CLEF_TEST_CASE
|
||||
from integration.translation.decisions.bases.cloudflare import CLEF_TEST_CASE, CLEF_SYSTEMONE_TEST_CASE
|
||||
from integration.translation.runner import assert_translation
|
||||
|
||||
|
||||
@pytest.mark.parametrize("case", [CLEF_TEST_CASE], ids=lambda case: case.id)
|
||||
@pytest.mark.parametrize("case", [CLEF_TEST_CASE, CLEF_SYSTEMONE_TEST_CASE], ids=lambda case: case.id)
|
||||
def test_decisions_basic_cloudflare(case: TranslationTestCase, gateway: Gateway, provider: SharedProvider) -> None:
|
||||
assert_translation(case, gateway, provider)
|
||||
|
|
|
|||
|
|
@ -2,10 +2,15 @@ import pytest
|
|||
from integration._support.client import Gateway
|
||||
from integration._support.provider import SharedProvider
|
||||
from integration.translation.case import TranslationTestCase
|
||||
from integration.translation.decisions.bases.openrouter import TYPESAFE_JEV_1_13_TEST_CASE
|
||||
from integration.translation.decisions.bases.openrouter import (
|
||||
TYPESAFE_JEV_1_13_TEST_CASE,
|
||||
TYPESAFE_JEV_1_13_SYSTEMONE_TEST_CASE,
|
||||
)
|
||||
from integration.translation.runner import assert_translation
|
||||
|
||||
|
||||
@pytest.mark.parametrize("case", [TYPESAFE_JEV_1_13_TEST_CASE], ids=lambda case: case.id)
|
||||
@pytest.mark.parametrize(
|
||||
"case", [TYPESAFE_JEV_1_13_TEST_CASE, TYPESAFE_JEV_1_13_SYSTEMONE_TEST_CASE], ids=lambda case: case.id
|
||||
)
|
||||
def test_decisions_basic_openrouter(case: TranslationTestCase, gateway: Gateway, provider: SharedProvider) -> None:
|
||||
assert_translation(case, gateway, provider)
|
||||
|
|
|
|||
|
|
@ -2,10 +2,15 @@ import pytest
|
|||
from integration._support.client import Gateway
|
||||
from integration._support.provider import SharedProvider
|
||||
from integration.translation.case import TranslationTestCase
|
||||
from integration.translation.decisions.bases.perplexity import PPLX_DECIDER_V1_27B_TEST_CASE
|
||||
from integration.translation.decisions.bases.perplexity import (
|
||||
PPLX_DECIDER_V1_27B_TEST_CASE,
|
||||
PPLX_DECIDER_V1_27B_SYSTEMONE_TEST_CASE,
|
||||
)
|
||||
from integration.translation.runner import assert_translation
|
||||
|
||||
|
||||
@pytest.mark.parametrize("case", [PPLX_DECIDER_V1_27B_TEST_CASE], ids=lambda case: case.id)
|
||||
@pytest.mark.parametrize(
|
||||
"case", [PPLX_DECIDER_V1_27B_TEST_CASE, PPLX_DECIDER_V1_27B_SYSTEMONE_TEST_CASE], ids=lambda case: case.id
|
||||
)
|
||||
def test_decisions_basic_perplexity(case: TranslationTestCase, gateway: Gateway, provider: SharedProvider) -> None:
|
||||
assert_translation(case, gateway, provider)
|
||||
|
|
|
|||
|
|
@ -2,10 +2,17 @@ import pytest
|
|||
from integration._support.client import Gateway
|
||||
from integration._support.provider import SharedProvider
|
||||
from integration.translation.case import TranslationTestCase
|
||||
from integration.translation.decisions.bases.strands_decider import STRANDS_DECIDER_2B_HOBSON_V19_TEST_CASE
|
||||
from integration.translation.decisions.bases.strands_decider import (
|
||||
STRANDS_DECIDER_2B_HOBSON_V19_TEST_CASE,
|
||||
STRANDS_DECIDER_2B_HOBSON_V19_SYSTEMONE_TEST_CASE,
|
||||
)
|
||||
from integration.translation.runner import assert_translation
|
||||
|
||||
|
||||
@pytest.mark.parametrize("case", [STRANDS_DECIDER_2B_HOBSON_V19_TEST_CASE], ids=lambda case: case.id)
|
||||
@pytest.mark.parametrize(
|
||||
"case",
|
||||
[STRANDS_DECIDER_2B_HOBSON_V19_TEST_CASE, STRANDS_DECIDER_2B_HOBSON_V19_SYSTEMONE_TEST_CASE],
|
||||
ids=lambda case: case.id,
|
||||
)
|
||||
def test_decisions_basic_strands_decider(case: TranslationTestCase, gateway: Gateway, provider: SharedProvider) -> None:
|
||||
assert_translation(case, gateway, provider)
|
||||
|
|
|
|||
|
|
@ -2,10 +2,10 @@ import pytest
|
|||
from integration._support.client import Gateway
|
||||
from integration._support.provider import SharedProvider
|
||||
from integration.translation.case import TranslationTestCase
|
||||
from integration.translation.decisions.bases.typesafe import JEV_1_13_0_TEST_CASE
|
||||
from integration.translation.decisions.bases.typesafe import JEV_1_13_0_TEST_CASE, JEV_1_13_0_SYSTEMONE_TEST_CASE
|
||||
from integration.translation.runner import assert_translation
|
||||
|
||||
|
||||
@pytest.mark.parametrize("case", [JEV_1_13_0_TEST_CASE], ids=lambda case: case.id)
|
||||
@pytest.mark.parametrize("case", [JEV_1_13_0_TEST_CASE, JEV_1_13_0_SYSTEMONE_TEST_CASE], ids=lambda case: case.id)
|
||||
def test_decisions_basic_typesafe(case: TranslationTestCase, gateway: Gateway, provider: SharedProvider) -> None:
|
||||
assert_translation(case, gateway, provider)
|
||||
|
|
|
|||
|
|
@ -1262,6 +1262,141 @@ async def test_openrouter_decisions_uses_provider_reported_cost_without_cost_map
|
|||
assert get_response_cost_from_hidden_params(response.hidden_params) == cost
|
||||
|
||||
|
||||
_SAFETY_IDENTIFIER: Final = "end-user-7"
|
||||
_PREDICATE_QUESTIONS: Final[tuple[Mapping[str, object], ...]] = (
|
||||
{"type": "predicate", "name": "is_defect", "instructions": "Is this a defect?"},
|
||||
)
|
||||
_PREDICATE_REQUESTS: Final[tuple[Mapping[str, object], ...]] = (
|
||||
MappingProxyType({"input": "review", "questions": _PREDICATE_QUESTIONS}),
|
||||
MappingProxyType(
|
||||
{"state": "review", "questions": {"is_defect": {"type": "noul", "instructions": "Is this a defect?"}}}
|
||||
),
|
||||
)
|
||||
_PREDICATE_REQUEST_IDS: Final = ("input_format", "state_format")
|
||||
|
||||
|
||||
@pytest.mark.parametrize("request_kwargs", _PREDICATE_REQUESTS, ids=_PREDICATE_REQUEST_IDS)
|
||||
def test_safety_identifier_is_refused_before_http_when_the_provider_cannot_take_it(
|
||||
request_kwargs: Mapping[str, object], respx_mock: respx.MockRouter, monkeypatch: pytest.MonkeyPatch
|
||||
) -> None:
|
||||
monkeypatch.setattr(litellm, "drop_params", False)
|
||||
route: Final = respx_mock.post("https://api.perplexity.ai/v1/decisions").respond(json=_RESPONSE)
|
||||
|
||||
with pytest.raises(litellm.UnsupportedParamsError, match=r"safety_identifier.*drop_params") as caught:
|
||||
litellm.decisions(
|
||||
model="perplexity/pplx-decider-v1-27b",
|
||||
safety_identifier=_SAFETY_IDENTIFIER,
|
||||
api_key="caller-key",
|
||||
**request_kwargs,
|
||||
)
|
||||
|
||||
assert caught.value.status_code == 400
|
||||
assert not route.called
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
@pytest.mark.parametrize("scope", ("call", "global"))
|
||||
async def test_safety_identifier_is_dropped_from_the_wire_under_drop_params(
|
||||
scope: str, respx_mock: respx.MockRouter, monkeypatch: pytest.MonkeyPatch
|
||||
) -> None:
|
||||
monkeypatch.setattr(litellm, "drop_params", scope == "global")
|
||||
route: Final = respx_mock.post("https://api.perplexity.ai/v1/decisions").respond(json=_RESPONSE)
|
||||
|
||||
response: Final = await litellm.adecisions(
|
||||
model="perplexity/pplx-decider-v1-27b",
|
||||
input="review",
|
||||
questions=_PREDICATE_QUESTIONS,
|
||||
safety_identifier=_SAFETY_IDENTIFIER,
|
||||
api_key="caller-key",
|
||||
**({"drop_params": True} if scope == "call" else {}),
|
||||
)
|
||||
|
||||
assert route.called
|
||||
assert json.loads(respx_mock.calls[0].request.content) == {
|
||||
"model": "pplx-decider-v1-27b",
|
||||
"state": "review",
|
||||
"questions": {"is_defect": {"type": "noul", "instructions": "Is this a defect?"}},
|
||||
}
|
||||
assert isinstance(response, OpenAIDecisionResponse)
|
||||
assert [answer.model_dump(mode="json") for answer in response.answers] == [
|
||||
{"type": "predicate", "name": "is_defect", "probability": 0.9}
|
||||
]
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
@pytest.mark.parametrize("request_kwargs", _PREDICATE_REQUESTS, ids=_PREDICATE_REQUEST_IDS)
|
||||
@pytest.mark.parametrize("drop_params", (False, True), ids=("strict", "drop_params"))
|
||||
async def test_safety_identifier_reaches_openai_whether_or_not_params_are_dropped(
|
||||
drop_params: bool,
|
||||
request_kwargs: Mapping[str, object],
|
||||
respx_mock: respx.MockRouter,
|
||||
monkeypatch: pytest.MonkeyPatch,
|
||||
) -> None:
|
||||
monkeypatch.setattr(litellm, "drop_params", drop_params)
|
||||
monkeypatch.setattr(litellm, "api_base", None)
|
||||
monkeypatch.delenv("OPENAI_BASE_URL", raising=False)
|
||||
monkeypatch.delenv("OPENAI_API_BASE", raising=False)
|
||||
route: Final = respx_mock.post("https://api.openai.com/v1/decisions").respond(json=_OPENAI_RESPONSE)
|
||||
|
||||
await litellm.adecisions(
|
||||
model="openai/gpt-6-luna",
|
||||
safety_identifier=_SAFETY_IDENTIFIER,
|
||||
api_key="caller-key",
|
||||
**request_kwargs,
|
||||
)
|
||||
|
||||
assert route.called
|
||||
assert json.loads(respx_mock.calls[0].request.content) == {
|
||||
"model": "gpt-6-luna",
|
||||
"input": "review",
|
||||
"questions": [{"type": "predicate", "name": "is_defect", "instructions": "Is this a defect?"}],
|
||||
"safety_identifier": _SAFETY_IDENTIFIER,
|
||||
}
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
@pytest.mark.parametrize("request_kwargs", _PREDICATE_REQUESTS, ids=_PREDICATE_REQUEST_IDS)
|
||||
async def test_non_string_safety_identifier_is_rejected_before_http(
|
||||
request_kwargs: Mapping[str, object], respx_mock: respx.MockRouter, monkeypatch: pytest.MonkeyPatch
|
||||
) -> None:
|
||||
monkeypatch.setattr(litellm, "drop_params", False)
|
||||
route: Final = respx_mock.post("https://api.openai.com/v1/decisions").respond(json=_OPENAI_RESPONSE)
|
||||
invalid_safety_identifier: Final[Mapping[str, object]] = MappingProxyType({"safety_identifier": 7})
|
||||
|
||||
with pytest.raises(litellm.BadRequestError, match=r"(?s)Invalid Decisions request.*safety_identifier"):
|
||||
await litellm.adecisions(
|
||||
model="openai/gpt-6-luna", api_key="caller-key", **request_kwargs, **invalid_safety_identifier
|
||||
)
|
||||
|
||||
assert not route.called
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
@pytest.mark.parametrize("request_kwargs", _PREDICATE_REQUESTS, ids=_PREDICATE_REQUEST_IDS)
|
||||
async def test_non_string_safety_identifier_is_dropped_from_the_wire_under_drop_params(
|
||||
request_kwargs: Mapping[str, object], respx_mock: respx.MockRouter, monkeypatch: pytest.MonkeyPatch
|
||||
) -> None:
|
||||
monkeypatch.setattr(litellm, "drop_params", False)
|
||||
monkeypatch.setattr(litellm, "api_base", None)
|
||||
monkeypatch.delenv("OPENAI_BASE_URL", raising=False)
|
||||
monkeypatch.delenv("OPENAI_API_BASE", raising=False)
|
||||
route: Final = respx_mock.post("https://api.openai.com/v1/decisions").respond(json=_OPENAI_RESPONSE)
|
||||
dropped_safety_identifier: Final[Mapping[str, object]] = MappingProxyType(
|
||||
{"safety_identifier": 7, "drop_params": True}
|
||||
)
|
||||
|
||||
await litellm.adecisions(
|
||||
model="openai/gpt-6-luna", api_key="caller-key", **request_kwargs, **dropped_safety_identifier
|
||||
)
|
||||
|
||||
assert route.called
|
||||
assert json.loads(respx_mock.calls[0].request.content) == {
|
||||
"model": "gpt-6-luna",
|
||||
"input": "review",
|
||||
"questions": [{"type": "predicate", "name": "is_defect", "instructions": "Is this a defect?"}],
|
||||
}
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
@pytest.mark.parametrize(
|
||||
"api_base",
|
||||
|
|
|
|||
|
|
@ -575,6 +575,68 @@ def test_openai_format_decisions_reach_an_openai_deployment_unchanged_including_
|
|||
assert float(response.headers["x-litellm-response-cost"]) == pytest.approx(expected_cost)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("safety_identifier", (7, ["end-user-1"]), ids=("numeric", "list"))
|
||||
@pytest.mark.parametrize("deployment_drops_params", (False, True), ids=("strict", "drop_params"))
|
||||
def test_a_non_string_safety_identifier_is_refused_unless_the_deployment_drops_params(
|
||||
client: TestClient,
|
||||
monkeypatch: pytest.MonkeyPatch,
|
||||
respx_mock: respx.MockRouter,
|
||||
safety_identifier: object,
|
||||
deployment_drops_params: bool,
|
||||
) -> None:
|
||||
monkeypatch.setattr(litellm, "drop_params", False)
|
||||
monkeypatch.setattr(litellm, "api_base", None)
|
||||
monkeypatch.delenv("OPENAI_BASE_URL", raising=False)
|
||||
monkeypatch.delenv("OPENAI_API_BASE", raising=False)
|
||||
monkeypatch.setattr(
|
||||
litellm.proxy.proxy_server,
|
||||
"llm_router",
|
||||
litellm.Router(
|
||||
model_list=[
|
||||
{
|
||||
"model_name": "decider",
|
||||
"litellm_params": {
|
||||
"model": "openai/gpt-6-luna",
|
||||
"api_key": "k",
|
||||
"drop_params": deployment_drops_params,
|
||||
},
|
||||
}
|
||||
]
|
||||
),
|
||||
)
|
||||
upstream: Final = respx_mock.post("https://api.openai.com/v1/decisions").respond(
|
||||
json={
|
||||
"model": "gpt-6-luna",
|
||||
"answers": _OPENAI_FORMAT_ANSWERS[:1],
|
||||
"usage": {
|
||||
"input_tokens": _INPUT_TOKENS,
|
||||
"output_tokens": _OUTPUT_TOKENS,
|
||||
"total_tokens": _INPUT_TOKENS + _OUTPUT_TOKENS,
|
||||
},
|
||||
}
|
||||
)
|
||||
request_body: Final = {
|
||||
"model": "decider",
|
||||
"input": "The package arrived with a broken screen.",
|
||||
"questions": _OPENAI_FORMAT_REQUEST["questions"][:1],
|
||||
"safety_identifier": safety_identifier,
|
||||
}
|
||||
|
||||
response: Final = client.post("/v1/decisions", json=request_body)
|
||||
|
||||
if not deployment_drops_params:
|
||||
assert response.status_code == 400, response.text
|
||||
assert "safety_identifier" in response.json()["error"]["message"]
|
||||
assert not upstream.called
|
||||
return
|
||||
assert response.status_code == 200, response.text
|
||||
assert json.loads(upstream.calls[0].request.content) == {
|
||||
"model": "gpt-6-luna",
|
||||
"input": "The package arrived with a broken screen.",
|
||||
"questions": _OPENAI_FORMAT_REQUEST["questions"][:1],
|
||||
}
|
||||
|
||||
|
||||
def _decisions_feature() -> LazyFeature:
|
||||
return next(feature for feature in LAZY_FEATURES if feature.name == "decisions")
|
||||
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue