mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-08 03:08:45 +00:00
feat(decisions): add the OpenAI Decisions spec types and the System One translation (#45129)
* feat(decisions): add the OpenAI Decisions spec types and the System One translation Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * refactor(decisions): share one DecisionsModel config and require model and usage on responses Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * refactor(decisions): rename the shared pydantic parent to DecisionsObjectBase Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(decisions): match the SDK on strict choice values and drop the invented min_length bounds Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * refactor(decisions): compose DecisionsRequest from model and body and read the translator top down Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * refactor(decisions): use the plural Decisions prefix only for the request and response envelopes Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(decisions): import assert_never from typing_extensions for Python 3.10 Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --------- Co-authored-by: kerry <kerry@berri.ai> Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
parent
037378ece8
commit
79f62db620
5 changed files with 821 additions and 0 deletions
247
litellm/llms/base_llm/decisions/systemone.py
Normal file
247
litellm/llms/base_llm/decisions/systemone.py
Normal file
|
|
@ -0,0 +1,247 @@
|
|||
"""The Jev / System One wire shape and its translation to and from the OpenAI Decisions shape.
|
||||
|
||||
System One (TypeSafe, Perplexity, OpenRouter, Cloudflare Clef, Strands Decider) takes
|
||||
{"model", "state", "questions": {name: question}} and answers with {"model", "answers": {name: answer}, "usage"}.
|
||||
Predicates are `noul` questions, choice options are a `criteria` map, score levels are a `criteria` list.
|
||||
"""
|
||||
|
||||
import itertools
|
||||
from collections.abc import Mapping, Sequence
|
||||
from typing import Final, Literal, TypeAlias
|
||||
|
||||
from pydantic import ConfigDict, TypeAdapter
|
||||
from typing_extensions import assert_never
|
||||
|
||||
from litellm.llms.base_llm.chat.transformation import BaseLLMException
|
||||
from litellm.types.llms.base import LiteLLMPydanticObjectBase
|
||||
from litellm.types.openai_decisions import (
|
||||
ChoiceAnswer,
|
||||
ChoiceProbability,
|
||||
ChoiceQuestion,
|
||||
DecisionAnswer,
|
||||
DecisionChoice,
|
||||
DecisionInput,
|
||||
DecisionInputMessage,
|
||||
DecisionInputPart,
|
||||
DecisionInputTokensDetails,
|
||||
DecisionOutputTokensDetails,
|
||||
DecisionQuestion,
|
||||
DecisionsRequest,
|
||||
DecisionsRequestBody,
|
||||
DecisionsResponse,
|
||||
DecisionUsage,
|
||||
PredicateAnswer,
|
||||
PredicateQuestion,
|
||||
ScoreAnswer,
|
||||
ScoreProbability,
|
||||
ScoreQuestion,
|
||||
)
|
||||
|
||||
|
||||
class SystemOneObjectBase(LiteLLMPydanticObjectBase):
|
||||
model_config = ConfigDict(extra="allow", frozen=True)
|
||||
|
||||
|
||||
class SystemOneNoulAnswer(SystemOneObjectBase):
|
||||
type: Literal["noul"]
|
||||
noul: float
|
||||
|
||||
|
||||
class SystemOneChoiceAnswer(SystemOneObjectBase):
|
||||
type: Literal["choice"]
|
||||
choice: str
|
||||
confidence: float
|
||||
probabilities: Mapping[str, float]
|
||||
|
||||
|
||||
class SystemOneScoreAnswer(SystemOneObjectBase):
|
||||
type: Literal["score"]
|
||||
score: float
|
||||
confidence: float
|
||||
probabilities: Mapping[str, float]
|
||||
|
||||
|
||||
SystemOneAnswer: TypeAlias = SystemOneNoulAnswer | SystemOneChoiceAnswer | SystemOneScoreAnswer
|
||||
|
||||
|
||||
class SystemOneUsage(SystemOneObjectBase):
|
||||
input_tokens: int = 0
|
||||
output_tokens: int = 0
|
||||
|
||||
|
||||
class SystemOneResponse(SystemOneObjectBase):
|
||||
model: str | None = None
|
||||
answers: Mapping[str, SystemOneAnswer]
|
||||
usage: SystemOneUsage | None = None
|
||||
|
||||
|
||||
SYSTEM_ONE_RESPONSE_ADAPTER: Final[TypeAdapter[SystemOneResponse]] = TypeAdapter(SystemOneResponse)
|
||||
|
||||
|
||||
def _unsupported(what: str, custom_llm_provider: str) -> BaseLLMException:
|
||||
return BaseLLMException(
|
||||
status_code=400,
|
||||
message=f"Decisions provider '{custom_llm_provider}' does not support {what}",
|
||||
)
|
||||
|
||||
|
||||
def to_system_one_request(model: str, body: DecisionsRequestBody, custom_llm_provider: str) -> dict[str, object]:
|
||||
keys: Final = question_keys(body.questions, custom_llm_provider)
|
||||
return {
|
||||
"model": model,
|
||||
"state": _state(body.input, custom_llm_provider),
|
||||
"questions": {
|
||||
key: _question(question, custom_llm_provider) for key, question in zip(keys, body.questions, strict=True)
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
def question_keys(questions: Sequence[DecisionQuestion], custom_llm_provider: str) -> tuple[str, ...]:
|
||||
"""System One keys questions and answers by name, so unnamed questions get a positional key."""
|
||||
names: Final = tuple(question.name for question in questions if question.name is not None)
|
||||
if len(set(names)) != len(names):
|
||||
raise BaseLLMException(
|
||||
status_code=400,
|
||||
message=f"Decisions provider '{custom_llm_provider}' requires a unique name per question",
|
||||
)
|
||||
taken: Final = frozenset(names)
|
||||
return tuple(
|
||||
question.name if question.name is not None else _positional_key(index, taken)
|
||||
for index, question in enumerate(questions)
|
||||
)
|
||||
|
||||
|
||||
def _positional_key(index: int, taken: frozenset[str]) -> str:
|
||||
candidates: Final = (f"{'_' * depth}q{index}" for depth in itertools.count())
|
||||
return next(key for key in candidates if key not in taken)
|
||||
|
||||
|
||||
def _state(input_value: DecisionInput, custom_llm_provider: str) -> str:
|
||||
if isinstance(input_value, str):
|
||||
return input_value
|
||||
return "\n".join(_message_text(message, custom_llm_provider) for message in input_value)
|
||||
|
||||
|
||||
def _message_text(message: DecisionInputMessage, custom_llm_provider: str) -> str:
|
||||
if isinstance(message.content, str):
|
||||
return message.content
|
||||
return "\n".join(_part_text(part, custom_llm_provider) for part in message.content)
|
||||
|
||||
|
||||
def _part_text(part: DecisionInputPart, custom_llm_provider: str) -> str:
|
||||
if part.type != "input_text":
|
||||
raise _unsupported("input_image parts", custom_llm_provider)
|
||||
return part.text
|
||||
|
||||
|
||||
def _question(question: DecisionQuestion, custom_llm_provider: str) -> dict[str, object]:
|
||||
match question:
|
||||
case PredicateQuestion():
|
||||
return {"type": "noul", "instructions": question.instructions}
|
||||
case ChoiceQuestion():
|
||||
return {
|
||||
"type": "choice",
|
||||
"instructions": question.instructions,
|
||||
"criteria": _choice_criteria(question, custom_llm_provider),
|
||||
}
|
||||
case ScoreQuestion():
|
||||
return {
|
||||
"type": "score",
|
||||
"instructions": question.instructions,
|
||||
"criteria": [_level_text(level.label, level.description) for level in question.levels],
|
||||
}
|
||||
case _:
|
||||
assert_never(question)
|
||||
|
||||
|
||||
def _choice_criteria(question: ChoiceQuestion, custom_llm_provider: str) -> dict[str, str | None]:
|
||||
values: Final = tuple(_choice_key(choice, custom_llm_provider) for choice in question.choices)
|
||||
if len(set(values)) != len(values):
|
||||
raise _unsupported("repeated choice values", custom_llm_provider)
|
||||
return {value: choice.description for value, choice in zip(values, question.choices, strict=True)}
|
||||
|
||||
|
||||
def _choice_key(choice: DecisionChoice, custom_llm_provider: str) -> str:
|
||||
if not isinstance(choice.value, str):
|
||||
raise _unsupported("boolean choice values", custom_llm_provider)
|
||||
return choice.value
|
||||
|
||||
|
||||
def _level_text(label: str, description: str | None) -> str:
|
||||
return description if description is not None else label
|
||||
|
||||
|
||||
def to_decisions_response(
|
||||
system_one: SystemOneResponse,
|
||||
request: DecisionsRequest,
|
||||
custom_llm_provider: str,
|
||||
) -> DecisionsResponse:
|
||||
keys: Final = question_keys(request.body.questions, custom_llm_provider)
|
||||
return DecisionsResponse(
|
||||
model=system_one.model if system_one.model is not None else request.model,
|
||||
answers=[
|
||||
_answer(key, question, system_one.answers, custom_llm_provider)
|
||||
for key, question in zip(keys, request.body.questions, strict=True)
|
||||
],
|
||||
usage=_usage(system_one.usage),
|
||||
)
|
||||
|
||||
|
||||
def _answer(
|
||||
key: str,
|
||||
question: DecisionQuestion,
|
||||
answers: Mapping[str, SystemOneAnswer],
|
||||
custom_llm_provider: str,
|
||||
) -> DecisionAnswer:
|
||||
match question, answers.get(key):
|
||||
case PredicateQuestion(), SystemOneNoulAnswer() as answer:
|
||||
return PredicateAnswer(type="predicate", name=question.name, probability=answer.noul)
|
||||
case ChoiceQuestion(), SystemOneChoiceAnswer() as answer:
|
||||
return ChoiceAnswer(
|
||||
type="choice",
|
||||
name=question.name,
|
||||
choice=answer.choice,
|
||||
probabilities=_choice_probabilities(question, answer),
|
||||
confidence=answer.confidence,
|
||||
)
|
||||
case ScoreQuestion(), SystemOneScoreAnswer() as answer:
|
||||
return ScoreAnswer(
|
||||
type="score",
|
||||
name=question.name,
|
||||
score=answer.score,
|
||||
probabilities=_score_probabilities(question, answer),
|
||||
confidence=answer.confidence,
|
||||
)
|
||||
case _:
|
||||
raise BaseLLMException(
|
||||
status_code=500,
|
||||
message=(
|
||||
f"Decisions provider '{custom_llm_provider}' returned no {question.type} answer "
|
||||
f"for question '{key}'"
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
def _choice_probabilities(question: ChoiceQuestion, answer: SystemOneChoiceAnswer) -> list[ChoiceProbability]:
|
||||
return [
|
||||
ChoiceProbability(value=choice.value, probability=answer.probabilities.get(str(choice.value), 0.0))
|
||||
for choice in question.choices
|
||||
]
|
||||
|
||||
|
||||
def _score_probabilities(question: ScoreQuestion, answer: SystemOneScoreAnswer) -> list[ScoreProbability]:
|
||||
return [
|
||||
ScoreProbability(value=index, label=level.label, probability=answer.probabilities.get(str(index), 0.0))
|
||||
for index, level in enumerate(question.levels)
|
||||
]
|
||||
|
||||
|
||||
def _usage(usage: SystemOneUsage | None) -> DecisionUsage:
|
||||
counted: Final = usage if usage is not None else SystemOneUsage()
|
||||
return DecisionUsage(
|
||||
input_tokens=counted.input_tokens,
|
||||
input_tokens_details=DecisionInputTokensDetails(cached_tokens=0, cache_write_tokens=0),
|
||||
output_tokens=counted.output_tokens,
|
||||
output_tokens_details=DecisionOutputTokensDetails(reasoning_tokens=0),
|
||||
total_tokens=counted.input_tokens + counted.output_tokens,
|
||||
)
|
||||
162
litellm/types/openai_decisions.py
Normal file
162
litellm/types/openai_decisions.py
Normal file
|
|
@ -0,0 +1,162 @@
|
|||
from collections.abc import Mapping, Sequence
|
||||
from dataclasses import dataclass
|
||||
from typing import Annotated, Literal, TypeAlias
|
||||
|
||||
from pydantic import ConfigDict, Field, PrivateAttr, StrictBool, StrictStr
|
||||
|
||||
from litellm.types.llms.base import LiteLLMPydanticObjectBase
|
||||
|
||||
ChoiceValue: TypeAlias = StrictStr | StrictBool
|
||||
|
||||
|
||||
class DecisionsObjectBase(LiteLLMPydanticObjectBase):
|
||||
model_config = ConfigDict(extra="allow", frozen=True)
|
||||
|
||||
|
||||
class DecisionInputText(DecisionsObjectBase):
|
||||
type: Literal["input_text"]
|
||||
text: str
|
||||
|
||||
|
||||
class DecisionInputImage(DecisionsObjectBase):
|
||||
type: Literal["input_image"]
|
||||
image_url: str
|
||||
detail: Literal["low", "high", "auto", "original"] | None = None
|
||||
|
||||
|
||||
DecisionInputPart: TypeAlias = Annotated[DecisionInputText | DecisionInputImage, Field(discriminator="type")]
|
||||
|
||||
|
||||
class DecisionInputMessage(DecisionsObjectBase):
|
||||
role: Literal["user"]
|
||||
content: str | Sequence[DecisionInputPart]
|
||||
type: Literal["message"] | None = None
|
||||
|
||||
|
||||
DecisionInput: TypeAlias = str | Sequence[DecisionInputMessage]
|
||||
|
||||
|
||||
class DecisionChoice(DecisionsObjectBase):
|
||||
value: ChoiceValue
|
||||
description: str | None = None
|
||||
|
||||
|
||||
class DecisionLevel(DecisionsObjectBase):
|
||||
label: str
|
||||
description: str | None = None
|
||||
|
||||
|
||||
class PredicateQuestion(DecisionsObjectBase):
|
||||
type: Literal["predicate"]
|
||||
instructions: str
|
||||
name: str | None = None
|
||||
|
||||
|
||||
class ChoiceQuestion(DecisionsObjectBase):
|
||||
type: Literal["choice"]
|
||||
instructions: str
|
||||
choices: Sequence[DecisionChoice]
|
||||
name: str | None = None
|
||||
|
||||
|
||||
class ScoreQuestion(DecisionsObjectBase):
|
||||
type: Literal["score"]
|
||||
instructions: str
|
||||
levels: Sequence[DecisionLevel]
|
||||
name: str | None = None
|
||||
|
||||
|
||||
DecisionQuestion: TypeAlias = Annotated[
|
||||
PredicateQuestion | ChoiceQuestion | ScoreQuestion,
|
||||
Field(discriminator="type"),
|
||||
]
|
||||
|
||||
DecisionQuestions: TypeAlias = Sequence[DecisionQuestion]
|
||||
|
||||
|
||||
class DecisionsRequestBody(DecisionsObjectBase):
|
||||
input: DecisionInput
|
||||
questions: DecisionQuestions
|
||||
safety_identifier: str | None = None
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class DecisionsRequest:
|
||||
model: str
|
||||
body: DecisionsRequestBody
|
||||
|
||||
|
||||
class PredicateAnswer(DecisionsObjectBase):
|
||||
type: Literal["predicate"]
|
||||
name: str | None = None
|
||||
probability: float
|
||||
|
||||
|
||||
class ChoiceProbability(DecisionsObjectBase):
|
||||
value: ChoiceValue
|
||||
probability: float
|
||||
|
||||
|
||||
class ChoiceAnswer(DecisionsObjectBase):
|
||||
type: Literal["choice"]
|
||||
name: str | None = None
|
||||
choice: ChoiceValue
|
||||
probabilities: Sequence[ChoiceProbability]
|
||||
confidence: float
|
||||
|
||||
|
||||
class ScoreProbability(DecisionsObjectBase):
|
||||
value: int
|
||||
label: str
|
||||
probability: float
|
||||
|
||||
|
||||
class ScoreAnswer(DecisionsObjectBase):
|
||||
type: Literal["score"]
|
||||
name: str | None = None
|
||||
score: float
|
||||
probabilities: Sequence[ScoreProbability]
|
||||
confidence: float
|
||||
|
||||
|
||||
class RefusalAnswer(DecisionsObjectBase):
|
||||
type: Literal["refusal"]
|
||||
name: str | None = None
|
||||
|
||||
|
||||
DecisionAnswer: TypeAlias = Annotated[
|
||||
PredicateAnswer | ChoiceAnswer | ScoreAnswer | RefusalAnswer,
|
||||
Field(discriminator="type"),
|
||||
]
|
||||
|
||||
|
||||
class DecisionInputTokensDetails(DecisionsObjectBase):
|
||||
cached_tokens: int
|
||||
cache_write_tokens: int
|
||||
|
||||
|
||||
class DecisionOutputTokensDetails(DecisionsObjectBase):
|
||||
reasoning_tokens: int
|
||||
|
||||
|
||||
class DecisionUsage(DecisionsObjectBase):
|
||||
input_tokens: int
|
||||
input_tokens_details: DecisionInputTokensDetails
|
||||
output_tokens: int
|
||||
output_tokens_details: DecisionOutputTokensDetails
|
||||
total_tokens: int
|
||||
|
||||
|
||||
class DecisionsResponse(DecisionsObjectBase):
|
||||
model: str
|
||||
answers: Sequence[DecisionAnswer]
|
||||
usage: DecisionUsage
|
||||
|
||||
_hidden_params: dict[str, object] = PrivateAttr(default_factory=dict)
|
||||
|
||||
@property
|
||||
def hidden_params(self) -> dict[str, object]: # mutable-ok: API requires mutation
|
||||
return self._hidden_params
|
||||
|
||||
def set_hidden_params(self, params: Mapping[str, object]) -> None:
|
||||
self._hidden_params.update(params)
|
||||
0
tests/unit/llms/base_llm/decisions/__init__.py
Normal file
0
tests/unit/llms/base_llm/decisions/__init__.py
Normal file
262
tests/unit/llms/base_llm/decisions/test_systemone.py
Normal file
262
tests/unit/llms/base_llm/decisions/test_systemone.py
Normal file
|
|
@ -0,0 +1,262 @@
|
|||
from __future__ import annotations
|
||||
|
||||
from collections.abc import Mapping, Sequence
|
||||
from typing import Final
|
||||
|
||||
import pytest
|
||||
from pydantic import TypeAdapter
|
||||
|
||||
from litellm.llms.base_llm.chat.transformation import BaseLLMException
|
||||
from litellm.llms.base_llm.decisions.systemone import (
|
||||
SYSTEM_ONE_RESPONSE_ADAPTER,
|
||||
question_keys,
|
||||
to_decisions_response,
|
||||
to_system_one_request,
|
||||
)
|
||||
from litellm.types.openai_decisions import (
|
||||
ChoiceAnswer,
|
||||
DecisionsRequest,
|
||||
DecisionsRequestBody,
|
||||
PredicateAnswer,
|
||||
ScoreAnswer,
|
||||
)
|
||||
|
||||
_INPUT: Final = "The export job hangs at 99% and never finishes"
|
||||
_QUESTIONS: Final[Sequence[Mapping[str, object]]] = (
|
||||
{"type": "predicate", "name": "is_defect", "instructions": "Is this a defect?"},
|
||||
{
|
||||
"type": "choice",
|
||||
"name": "sentiment",
|
||||
"instructions": "How does the customer feel?",
|
||||
"choices": [{"value": "positive"}, {"value": "negative", "description": "unhappy"}],
|
||||
},
|
||||
{
|
||||
"type": "score",
|
||||
"name": "severity",
|
||||
"instructions": "How severe is it?",
|
||||
"levels": [{"label": "none"}, {"label": "low"}, {"label": "high", "description": "blocks users"}],
|
||||
},
|
||||
)
|
||||
_SYSTEM_ONE_QUESTIONS: Final[Mapping[str, object]] = {
|
||||
"is_defect": {"type": "noul", "instructions": "Is this a defect?"},
|
||||
"sentiment": {
|
||||
"type": "choice",
|
||||
"instructions": "How does the customer feel?",
|
||||
"criteria": {"positive": None, "negative": "unhappy"},
|
||||
},
|
||||
"severity": {"type": "score", "instructions": "How severe is it?", "criteria": ["none", "low", "blocks users"]},
|
||||
}
|
||||
_SYSTEM_ONE_RESPONSE: Final[Mapping[str, object]] = {
|
||||
"model": "jev-1.13",
|
||||
"answers": {
|
||||
"is_defect": {"type": "noul", "noul": 0.9},
|
||||
"sentiment": {
|
||||
"type": "choice",
|
||||
"choice": "positive",
|
||||
"confidence": 0.8,
|
||||
"probabilities": {"positive": 0.8, "negative": 0.2},
|
||||
},
|
||||
"severity": {
|
||||
"type": "score",
|
||||
"score": 1,
|
||||
"confidence": 0.7,
|
||||
"legend": {"0": "none", "1": "low", "2": "high"},
|
||||
"probabilities": {"0": 0.1, "1": 0.8, "2": 0.1},
|
||||
},
|
||||
},
|
||||
"usage": {"input_tokens": 367, "output_tokens": 3},
|
||||
}
|
||||
_EXPECTED_ANSWERS: Final[Sequence[Mapping[str, object]]] = (
|
||||
{"type": "predicate", "name": "is_defect", "probability": 0.9},
|
||||
{
|
||||
"type": "choice",
|
||||
"name": "sentiment",
|
||||
"choice": "positive",
|
||||
"probabilities": [{"value": "positive", "probability": 0.8}, {"value": "negative", "probability": 0.2}],
|
||||
"confidence": 0.8,
|
||||
},
|
||||
{
|
||||
"type": "score",
|
||||
"name": "severity",
|
||||
"score": 1.0,
|
||||
"probabilities": [
|
||||
{"value": 0, "label": "none", "probability": 0.1},
|
||||
{"value": 1, "label": "low", "probability": 0.8},
|
||||
{"value": 2, "label": "high", "probability": 0.1},
|
||||
],
|
||||
"confidence": 0.7,
|
||||
},
|
||||
)
|
||||
_EXPECTED_USAGE: Final[Mapping[str, object]] = {
|
||||
"input_tokens": 367,
|
||||
"input_tokens_details": {"cached_tokens": 0, "cache_write_tokens": 0},
|
||||
"output_tokens": 3,
|
||||
"output_tokens_details": {"reasoning_tokens": 0},
|
||||
"total_tokens": 370,
|
||||
}
|
||||
_BODY_ADAPTER: Final[TypeAdapter[DecisionsRequestBody]] = TypeAdapter(DecisionsRequestBody)
|
||||
|
||||
|
||||
def _body(
|
||||
input_value: object = _INPUT,
|
||||
questions: Sequence[Mapping[str, object]] = _QUESTIONS,
|
||||
) -> DecisionsRequestBody:
|
||||
return _BODY_ADAPTER.validate_python({"input": input_value, "questions": questions})
|
||||
|
||||
|
||||
def _request(
|
||||
input_value: object = _INPUT,
|
||||
questions: Sequence[Mapping[str, object]] = _QUESTIONS,
|
||||
model: str = "jev-1.13",
|
||||
) -> DecisionsRequest:
|
||||
return DecisionsRequest(model=model, body=_body(input_value, questions))
|
||||
|
||||
|
||||
def _predicate(name: str | None = "is_defect") -> tuple[Mapping[str, object]]:
|
||||
return ({"type": "predicate", "name": name, "instructions": "Is this a defect?"},)
|
||||
|
||||
|
||||
def test_openai_request_becomes_the_system_one_body() -> None:
|
||||
assert to_system_one_request("jev-1.13", _body(), "typesafe") == {
|
||||
"model": "jev-1.13",
|
||||
"state": _INPUT,
|
||||
"questions": _SYSTEM_ONE_QUESTIONS,
|
||||
}
|
||||
|
||||
|
||||
def test_system_one_answers_become_openai_answers_in_question_order() -> None:
|
||||
response: Final = to_decisions_response(
|
||||
SYSTEM_ONE_RESPONSE_ADAPTER.validate_python(_SYSTEM_ONE_RESPONSE), _request(), "typesafe"
|
||||
)
|
||||
|
||||
assert response.model_dump(mode="json") == {
|
||||
"model": "jev-1.13",
|
||||
"answers": list(_EXPECTED_ANSWERS),
|
||||
"usage": _EXPECTED_USAGE,
|
||||
}
|
||||
assert isinstance(response.answers[0], PredicateAnswer)
|
||||
assert isinstance(response.answers[1], ChoiceAnswer)
|
||||
assert isinstance(response.answers[2], ScoreAnswer)
|
||||
|
||||
|
||||
def test_user_messages_are_joined_into_one_system_one_state() -> None:
|
||||
messages: Final = [
|
||||
{"role": "user", "content": "first"},
|
||||
{
|
||||
"role": "user",
|
||||
"content": [{"type": "input_text", "text": "second"}, {"type": "input_text", "text": "third"}],
|
||||
},
|
||||
]
|
||||
|
||||
body: Final = to_system_one_request("jev-1.13", _body(messages, _predicate()), "typesafe")
|
||||
|
||||
assert body["state"] == "first\nsecond\nthird"
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("label", "input_value", "questions"),
|
||||
(
|
||||
(
|
||||
"input_image",
|
||||
[{"role": "user", "content": [{"type": "input_image", "image_url": "data:image/png;base64,AA=="}]}],
|
||||
_predicate(),
|
||||
),
|
||||
("boolean choice", _INPUT, [{"type": "choice", "instructions": "Refund?", "choices": [{"value": True}]}]),
|
||||
("unique name", _INPUT, [*_predicate(), *_predicate()]),
|
||||
(
|
||||
"repeated choice",
|
||||
_INPUT,
|
||||
[{"type": "choice", "instructions": "Refund?", "choices": [{"value": "yes"}, {"value": "yes"}]}],
|
||||
),
|
||||
),
|
||||
)
|
||||
def test_what_system_one_cannot_express_is_a_400(
|
||||
label: str,
|
||||
input_value: object,
|
||||
questions: Sequence[Mapping[str, object]],
|
||||
) -> None:
|
||||
with pytest.raises(BaseLLMException, match=label) as error:
|
||||
to_system_one_request("jev-1.13", _body(input_value, questions), "perplexity")
|
||||
|
||||
assert error.value.status_code == 400
|
||||
assert "perplexity" in error.value.message
|
||||
|
||||
|
||||
def test_unnamed_questions_get_positional_keys_that_never_shadow_a_supplied_name() -> None:
|
||||
body: Final = _body(questions=[*_predicate(None), *_predicate("q0"), *_predicate(None)])
|
||||
|
||||
assert question_keys(body.questions, "typesafe") == ("_q0", "q0", "q2")
|
||||
noul: Final = {"type": "noul", "instructions": "Is this a defect?"}
|
||||
assert to_system_one_request("jev-1.13", body, "typesafe")["questions"] == {"_q0": noul, "q0": noul, "q2": noul}
|
||||
|
||||
|
||||
def test_positional_answers_come_back_in_question_order_without_a_name() -> None:
|
||||
request: Final = _request(questions=[*_predicate(None), *_predicate("q0")])
|
||||
system_one: Final = SYSTEM_ONE_RESPONSE_ADAPTER.validate_python(
|
||||
{"answers": {"_q0": {"type": "noul", "noul": 0.25}, "q0": {"type": "noul", "noul": 0.75}}}
|
||||
)
|
||||
|
||||
response: Final = to_decisions_response(system_one, request, "typesafe")
|
||||
|
||||
assert [answer.model_dump(mode="json") for answer in response.answers] == [
|
||||
{"type": "predicate", "name": None, "probability": 0.25},
|
||||
{"type": "predicate", "name": "q0", "probability": 0.75},
|
||||
]
|
||||
assert response.usage.model_dump(mode="json") == {
|
||||
**_EXPECTED_USAGE,
|
||||
"input_tokens": 0,
|
||||
"output_tokens": 0,
|
||||
"total_tokens": 0,
|
||||
}
|
||||
|
||||
|
||||
def test_a_reply_without_a_model_reports_the_requested_model() -> None:
|
||||
system_one: Final = SYSTEM_ONE_RESPONSE_ADAPTER.validate_python(
|
||||
{k: v for k, v in _SYSTEM_ONE_RESPONSE.items() if k != "model"}
|
||||
)
|
||||
|
||||
response: Final = to_decisions_response(system_one, _request(model="typesafe/jev-1.13.0"), "typesafe")
|
||||
|
||||
assert response.model == "typesafe/jev-1.13.0"
|
||||
|
||||
|
||||
def test_a_choice_the_provider_left_out_of_probabilities_is_reported_at_zero() -> None:
|
||||
system_one: Final = SYSTEM_ONE_RESPONSE_ADAPTER.validate_python(
|
||||
{
|
||||
"answers": {
|
||||
"sentiment": {
|
||||
"type": "choice",
|
||||
"choice": "positive",
|
||||
"confidence": 1.0,
|
||||
"probabilities": {"positive": 1.0},
|
||||
}
|
||||
}
|
||||
}
|
||||
)
|
||||
request: Final = _request(questions=_QUESTIONS[1:2])
|
||||
|
||||
response: Final = to_decisions_response(system_one, request, "typesafe")
|
||||
|
||||
assert response.answers[0].model_dump(mode="json") == {
|
||||
"type": "choice",
|
||||
"name": "sentiment",
|
||||
"choice": "positive",
|
||||
"probabilities": [{"value": "positive", "probability": 1.0}, {"value": "negative", "probability": 0.0}],
|
||||
"confidence": 1.0,
|
||||
}
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"answers",
|
||||
(
|
||||
{},
|
||||
{"is_defect": {"type": "choice", "choice": "yes", "confidence": 1.0, "probabilities": {"yes": 1.0}}},
|
||||
),
|
||||
)
|
||||
def test_a_reply_without_a_matching_answer_is_a_server_error(answers: Mapping[str, object]) -> None:
|
||||
system_one: Final = SYSTEM_ONE_RESPONSE_ADAPTER.validate_python({"answers": answers})
|
||||
|
||||
with pytest.raises(BaseLLMException, match="no predicate answer for question 'is_defect'") as error:
|
||||
to_decisions_response(system_one, _request(questions=_predicate()), "typesafe")
|
||||
|
||||
assert error.value.status_code == 500
|
||||
150
tests/unit/types/test_openai_decisions.py
Normal file
150
tests/unit/types/test_openai_decisions.py
Normal file
|
|
@ -0,0 +1,150 @@
|
|||
from __future__ import annotations
|
||||
|
||||
from collections.abc import Mapping
|
||||
from typing import Final
|
||||
|
||||
import pytest
|
||||
from pydantic import TypeAdapter, ValidationError
|
||||
|
||||
from litellm.types.openai_decisions import (
|
||||
ChoiceAnswer,
|
||||
DecisionsRequestBody,
|
||||
DecisionsResponse,
|
||||
RefusalAnswer,
|
||||
ScoreQuestion,
|
||||
)
|
||||
|
||||
_REQUEST: Final[Mapping[str, object]] = {
|
||||
"input": [
|
||||
{
|
||||
"role": "user",
|
||||
"content": [
|
||||
{"type": "input_text", "text": "Is this receipt a valid business expense?"},
|
||||
{"type": "input_image", "image_url": "https://example.com/receipt.png", "detail": "high"},
|
||||
],
|
||||
}
|
||||
],
|
||||
"questions": [
|
||||
{"type": "predicate", "name": "is_expense", "instructions": "Is this a business expense?"},
|
||||
{
|
||||
"type": "choice",
|
||||
"name": "approve",
|
||||
"instructions": "Should this be approved?",
|
||||
"choices": [{"value": True, "description": "approve"}, {"value": False, "description": "reject"}],
|
||||
},
|
||||
{
|
||||
"type": "score",
|
||||
"name": "risk",
|
||||
"instructions": "How risky is this expense?",
|
||||
"levels": [{"label": "low"}, {"label": "high", "description": "needs a manager"}],
|
||||
},
|
||||
],
|
||||
"safety_identifier": "user-123",
|
||||
}
|
||||
_RESPONSE: Final[Mapping[str, object]] = {
|
||||
"model": "gpt-6-luna",
|
||||
"answers": [
|
||||
{"type": "predicate", "name": "is_expense", "probability": 0.92},
|
||||
{
|
||||
"type": "choice",
|
||||
"name": "approve",
|
||||
"choice": True,
|
||||
"probabilities": [{"value": True, "probability": 0.7}, {"value": False, "probability": 0.3}],
|
||||
"confidence": 0.7,
|
||||
},
|
||||
{"type": "refusal", "name": "risk"},
|
||||
],
|
||||
"usage": {
|
||||
"input_tokens": 120,
|
||||
"input_tokens_details": {"cached_tokens": 100, "cache_write_tokens": 0},
|
||||
"output_tokens": 12,
|
||||
"output_tokens_details": {"reasoning_tokens": 4},
|
||||
"total_tokens": 132,
|
||||
},
|
||||
}
|
||||
_REQUEST_ADAPTER: Final[TypeAdapter[DecisionsRequestBody]] = TypeAdapter(DecisionsRequestBody)
|
||||
_RESPONSE_ADAPTER: Final[TypeAdapter[DecisionsResponse]] = TypeAdapter(DecisionsResponse)
|
||||
|
||||
|
||||
def test_the_documented_request_round_trips_with_its_boolean_choices_and_image_part() -> None:
|
||||
request: Final = _REQUEST_ADAPTER.validate_python(_REQUEST)
|
||||
|
||||
assert request.model_dump(mode="json", exclude_none=True) == _REQUEST
|
||||
assert isinstance(request.questions[2], ScoreQuestion)
|
||||
|
||||
|
||||
def test_the_documented_response_keeps_answer_order_refusals_and_token_details() -> None:
|
||||
response: Final = _RESPONSE_ADAPTER.validate_python(_RESPONSE)
|
||||
|
||||
assert response.model_dump(mode="json") == _RESPONSE
|
||||
assert isinstance(response.answers[1], ChoiceAnswer)
|
||||
assert isinstance(response.answers[2], RefusalAnswer)
|
||||
|
||||
|
||||
_OFF_SPEC: Final[tuple[tuple[str, object], ...]] = (
|
||||
("questions", [{"type": "noul", "name": "q", "instructions": "x"}]),
|
||||
("questions", [{"type": "choice", "name": "q", "instructions": "x", "choices": [{"value": 1}]}]),
|
||||
("questions", [{"type": "score", "name": "q", "levels": [{"label": "low"}]}]),
|
||||
("input", {"state": "not an OpenAI input"}),
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(("field", "value"), _OFF_SPEC)
|
||||
def test_requests_off_the_spec_are_rejected(field: str, value: object) -> None:
|
||||
with pytest.raises(ValidationError):
|
||||
_REQUEST_ADAPTER.validate_python({**_REQUEST, field: value})
|
||||
|
||||
|
||||
_EMPTY_COLLECTIONS: Final[tuple[tuple[str, list[object]], ...]] = (
|
||||
("questions", []),
|
||||
("questions", [{"type": "choice", "name": "q", "instructions": "x", "choices": []}]),
|
||||
("questions", [{"type": "score", "name": "q", "instructions": "x", "levels": []}]),
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(("field", "value"), _EMPTY_COLLECTIONS)
|
||||
def test_empty_collections_are_left_for_the_provider_to_judge(field: str, value: list[object]) -> None:
|
||||
request: Final = _REQUEST_ADAPTER.validate_python({**_REQUEST, field: value})
|
||||
|
||||
assert request.model_dump(mode="json", exclude_none=True)[field] == value
|
||||
|
||||
|
||||
def test_choice_values_keep_their_type_so_a_string_true_and_a_boolean_true_stay_distinct() -> None:
|
||||
answer: Final = {
|
||||
"type": "choice",
|
||||
"name": "approve",
|
||||
"choice": "true",
|
||||
"probabilities": [{"value": "true", "probability": 0.6}, {"value": True, "probability": 0.4}],
|
||||
"confidence": 0.6,
|
||||
}
|
||||
|
||||
response: Final = _RESPONSE_ADAPTER.validate_python({**_RESPONSE, "answers": [answer]})
|
||||
|
||||
assert response.model_dump(mode="json")["answers"] == [answer]
|
||||
with pytest.raises(ValidationError):
|
||||
_RESPONSE_ADAPTER.validate_python({**_RESPONSE, "answers": [{**answer, "choice": 1}]})
|
||||
|
||||
|
||||
_OFF_SPEC_RESPONSE: Final[tuple[str, ...]] = ("model", "usage")
|
||||
|
||||
|
||||
@pytest.mark.parametrize("field", _OFF_SPEC_RESPONSE)
|
||||
def test_responses_missing_a_required_field_are_rejected(field: str) -> None:
|
||||
with pytest.raises(ValidationError):
|
||||
_RESPONSE_ADAPTER.validate_python({k: v for k, v in _RESPONSE.items() if k != field})
|
||||
|
||||
|
||||
def test_usage_without_token_details_is_rejected() -> None:
|
||||
usage: Final = {"input_tokens": 120, "output_tokens": 12, "total_tokens": 132}
|
||||
|
||||
with pytest.raises(ValidationError):
|
||||
_RESPONSE_ADAPTER.validate_python({**_RESPONSE, "usage": usage})
|
||||
|
||||
|
||||
def test_hidden_params_live_outside_the_wire_body() -> None:
|
||||
response: Final = _RESPONSE_ADAPTER.validate_python(_RESPONSE)
|
||||
|
||||
response.set_hidden_params({"custom_llm_provider": "openai"})
|
||||
|
||||
assert response.hidden_params == {"custom_llm_provider": "openai"}
|
||||
assert "_hidden_params" not in response.model_dump(mode="json")
|
||||
Loading…
Add table
Reference in a new issue