diff --git a/litellm/llms/base_llm/decisions/systemone.py b/litellm/llms/base_llm/decisions/systemone.py new file mode 100644 index 00000000000..43be31b0c30 --- /dev/null +++ b/litellm/llms/base_llm/decisions/systemone.py @@ -0,0 +1,247 @@ +"""The Jev / System One wire shape and its translation to and from the OpenAI Decisions shape. + +System One (TypeSafe, Perplexity, OpenRouter, Cloudflare Clef, Strands Decider) takes +{"model", "state", "questions": {name: question}} and answers with {"model", "answers": {name: answer}, "usage"}. +Predicates are `noul` questions, choice options are a `criteria` map, score levels are a `criteria` list. +""" + +import itertools +from collections.abc import Mapping, Sequence +from typing import Final, Literal, TypeAlias + +from pydantic import ConfigDict, TypeAdapter +from typing_extensions import assert_never + +from litellm.llms.base_llm.chat.transformation import BaseLLMException +from litellm.types.llms.base import LiteLLMPydanticObjectBase +from litellm.types.openai_decisions import ( + ChoiceAnswer, + ChoiceProbability, + ChoiceQuestion, + DecisionAnswer, + DecisionChoice, + DecisionInput, + DecisionInputMessage, + DecisionInputPart, + DecisionInputTokensDetails, + DecisionOutputTokensDetails, + DecisionQuestion, + DecisionsRequest, + DecisionsRequestBody, + DecisionsResponse, + DecisionUsage, + PredicateAnswer, + PredicateQuestion, + ScoreAnswer, + ScoreProbability, + ScoreQuestion, +) + + +class SystemOneObjectBase(LiteLLMPydanticObjectBase): + model_config = ConfigDict(extra="allow", frozen=True) + + +class SystemOneNoulAnswer(SystemOneObjectBase): + type: Literal["noul"] + noul: float + + +class SystemOneChoiceAnswer(SystemOneObjectBase): + type: Literal["choice"] + choice: str + confidence: float + probabilities: Mapping[str, float] + + +class SystemOneScoreAnswer(SystemOneObjectBase): + type: Literal["score"] + score: float + confidence: float + probabilities: Mapping[str, float] + + +SystemOneAnswer: TypeAlias = SystemOneNoulAnswer | SystemOneChoiceAnswer | SystemOneScoreAnswer + + +class SystemOneUsage(SystemOneObjectBase): + input_tokens: int = 0 + output_tokens: int = 0 + + +class SystemOneResponse(SystemOneObjectBase): + model: str | None = None + answers: Mapping[str, SystemOneAnswer] + usage: SystemOneUsage | None = None + + +SYSTEM_ONE_RESPONSE_ADAPTER: Final[TypeAdapter[SystemOneResponse]] = TypeAdapter(SystemOneResponse) + + +def _unsupported(what: str, custom_llm_provider: str) -> BaseLLMException: + return BaseLLMException( + status_code=400, + message=f"Decisions provider '{custom_llm_provider}' does not support {what}", + ) + + +def to_system_one_request(model: str, body: DecisionsRequestBody, custom_llm_provider: str) -> dict[str, object]: + keys: Final = question_keys(body.questions, custom_llm_provider) + return { + "model": model, + "state": _state(body.input, custom_llm_provider), + "questions": { + key: _question(question, custom_llm_provider) for key, question in zip(keys, body.questions, strict=True) + }, + } + + +def question_keys(questions: Sequence[DecisionQuestion], custom_llm_provider: str) -> tuple[str, ...]: + """System One keys questions and answers by name, so unnamed questions get a positional key.""" + names: Final = tuple(question.name for question in questions if question.name is not None) + if len(set(names)) != len(names): + raise BaseLLMException( + status_code=400, + message=f"Decisions provider '{custom_llm_provider}' requires a unique name per question", + ) + taken: Final = frozenset(names) + return tuple( + question.name if question.name is not None else _positional_key(index, taken) + for index, question in enumerate(questions) + ) + + +def _positional_key(index: int, taken: frozenset[str]) -> str: + candidates: Final = (f"{'_' * depth}q{index}" for depth in itertools.count()) + return next(key for key in candidates if key not in taken) + + +def _state(input_value: DecisionInput, custom_llm_provider: str) -> str: + if isinstance(input_value, str): + return input_value + return "\n".join(_message_text(message, custom_llm_provider) for message in input_value) + + +def _message_text(message: DecisionInputMessage, custom_llm_provider: str) -> str: + if isinstance(message.content, str): + return message.content + return "\n".join(_part_text(part, custom_llm_provider) for part in message.content) + + +def _part_text(part: DecisionInputPart, custom_llm_provider: str) -> str: + if part.type != "input_text": + raise _unsupported("input_image parts", custom_llm_provider) + return part.text + + +def _question(question: DecisionQuestion, custom_llm_provider: str) -> dict[str, object]: + match question: + case PredicateQuestion(): + return {"type": "noul", "instructions": question.instructions} + case ChoiceQuestion(): + return { + "type": "choice", + "instructions": question.instructions, + "criteria": _choice_criteria(question, custom_llm_provider), + } + case ScoreQuestion(): + return { + "type": "score", + "instructions": question.instructions, + "criteria": [_level_text(level.label, level.description) for level in question.levels], + } + case _: + assert_never(question) + + +def _choice_criteria(question: ChoiceQuestion, custom_llm_provider: str) -> dict[str, str | None]: + values: Final = tuple(_choice_key(choice, custom_llm_provider) for choice in question.choices) + if len(set(values)) != len(values): + raise _unsupported("repeated choice values", custom_llm_provider) + return {value: choice.description for value, choice in zip(values, question.choices, strict=True)} + + +def _choice_key(choice: DecisionChoice, custom_llm_provider: str) -> str: + if not isinstance(choice.value, str): + raise _unsupported("boolean choice values", custom_llm_provider) + return choice.value + + +def _level_text(label: str, description: str | None) -> str: + return description if description is not None else label + + +def to_decisions_response( + system_one: SystemOneResponse, + request: DecisionsRequest, + custom_llm_provider: str, +) -> DecisionsResponse: + keys: Final = question_keys(request.body.questions, custom_llm_provider) + return DecisionsResponse( + model=system_one.model if system_one.model is not None else request.model, + answers=[ + _answer(key, question, system_one.answers, custom_llm_provider) + for key, question in zip(keys, request.body.questions, strict=True) + ], + usage=_usage(system_one.usage), + ) + + +def _answer( + key: str, + question: DecisionQuestion, + answers: Mapping[str, SystemOneAnswer], + custom_llm_provider: str, +) -> DecisionAnswer: + match question, answers.get(key): + case PredicateQuestion(), SystemOneNoulAnswer() as answer: + return PredicateAnswer(type="predicate", name=question.name, probability=answer.noul) + case ChoiceQuestion(), SystemOneChoiceAnswer() as answer: + return ChoiceAnswer( + type="choice", + name=question.name, + choice=answer.choice, + probabilities=_choice_probabilities(question, answer), + confidence=answer.confidence, + ) + case ScoreQuestion(), SystemOneScoreAnswer() as answer: + return ScoreAnswer( + type="score", + name=question.name, + score=answer.score, + probabilities=_score_probabilities(question, answer), + confidence=answer.confidence, + ) + case _: + raise BaseLLMException( + status_code=500, + message=( + f"Decisions provider '{custom_llm_provider}' returned no {question.type} answer " + f"for question '{key}'" + ), + ) + + +def _choice_probabilities(question: ChoiceQuestion, answer: SystemOneChoiceAnswer) -> list[ChoiceProbability]: + return [ + ChoiceProbability(value=choice.value, probability=answer.probabilities.get(str(choice.value), 0.0)) + for choice in question.choices + ] + + +def _score_probabilities(question: ScoreQuestion, answer: SystemOneScoreAnswer) -> list[ScoreProbability]: + return [ + ScoreProbability(value=index, label=level.label, probability=answer.probabilities.get(str(index), 0.0)) + for index, level in enumerate(question.levels) + ] + + +def _usage(usage: SystemOneUsage | None) -> DecisionUsage: + counted: Final = usage if usage is not None else SystemOneUsage() + return DecisionUsage( + input_tokens=counted.input_tokens, + input_tokens_details=DecisionInputTokensDetails(cached_tokens=0, cache_write_tokens=0), + output_tokens=counted.output_tokens, + output_tokens_details=DecisionOutputTokensDetails(reasoning_tokens=0), + total_tokens=counted.input_tokens + counted.output_tokens, + ) diff --git a/litellm/types/openai_decisions.py b/litellm/types/openai_decisions.py new file mode 100644 index 00000000000..074c0a803f3 --- /dev/null +++ b/litellm/types/openai_decisions.py @@ -0,0 +1,162 @@ +from collections.abc import Mapping, Sequence +from dataclasses import dataclass +from typing import Annotated, Literal, TypeAlias + +from pydantic import ConfigDict, Field, PrivateAttr, StrictBool, StrictStr + +from litellm.types.llms.base import LiteLLMPydanticObjectBase + +ChoiceValue: TypeAlias = StrictStr | StrictBool + + +class DecisionsObjectBase(LiteLLMPydanticObjectBase): + model_config = ConfigDict(extra="allow", frozen=True) + + +class DecisionInputText(DecisionsObjectBase): + type: Literal["input_text"] + text: str + + +class DecisionInputImage(DecisionsObjectBase): + type: Literal["input_image"] + image_url: str + detail: Literal["low", "high", "auto", "original"] | None = None + + +DecisionInputPart: TypeAlias = Annotated[DecisionInputText | DecisionInputImage, Field(discriminator="type")] + + +class DecisionInputMessage(DecisionsObjectBase): + role: Literal["user"] + content: str | Sequence[DecisionInputPart] + type: Literal["message"] | None = None + + +DecisionInput: TypeAlias = str | Sequence[DecisionInputMessage] + + +class DecisionChoice(DecisionsObjectBase): + value: ChoiceValue + description: str | None = None + + +class DecisionLevel(DecisionsObjectBase): + label: str + description: str | None = None + + +class PredicateQuestion(DecisionsObjectBase): + type: Literal["predicate"] + instructions: str + name: str | None = None + + +class ChoiceQuestion(DecisionsObjectBase): + type: Literal["choice"] + instructions: str + choices: Sequence[DecisionChoice] + name: str | None = None + + +class ScoreQuestion(DecisionsObjectBase): + type: Literal["score"] + instructions: str + levels: Sequence[DecisionLevel] + name: str | None = None + + +DecisionQuestion: TypeAlias = Annotated[ + PredicateQuestion | ChoiceQuestion | ScoreQuestion, + Field(discriminator="type"), +] + +DecisionQuestions: TypeAlias = Sequence[DecisionQuestion] + + +class DecisionsRequestBody(DecisionsObjectBase): + input: DecisionInput + questions: DecisionQuestions + safety_identifier: str | None = None + + +@dataclass(frozen=True, slots=True) +class DecisionsRequest: + model: str + body: DecisionsRequestBody + + +class PredicateAnswer(DecisionsObjectBase): + type: Literal["predicate"] + name: str | None = None + probability: float + + +class ChoiceProbability(DecisionsObjectBase): + value: ChoiceValue + probability: float + + +class ChoiceAnswer(DecisionsObjectBase): + type: Literal["choice"] + name: str | None = None + choice: ChoiceValue + probabilities: Sequence[ChoiceProbability] + confidence: float + + +class ScoreProbability(DecisionsObjectBase): + value: int + label: str + probability: float + + +class ScoreAnswer(DecisionsObjectBase): + type: Literal["score"] + name: str | None = None + score: float + probabilities: Sequence[ScoreProbability] + confidence: float + + +class RefusalAnswer(DecisionsObjectBase): + type: Literal["refusal"] + name: str | None = None + + +DecisionAnswer: TypeAlias = Annotated[ + PredicateAnswer | ChoiceAnswer | ScoreAnswer | RefusalAnswer, + Field(discriminator="type"), +] + + +class DecisionInputTokensDetails(DecisionsObjectBase): + cached_tokens: int + cache_write_tokens: int + + +class DecisionOutputTokensDetails(DecisionsObjectBase): + reasoning_tokens: int + + +class DecisionUsage(DecisionsObjectBase): + input_tokens: int + input_tokens_details: DecisionInputTokensDetails + output_tokens: int + output_tokens_details: DecisionOutputTokensDetails + total_tokens: int + + +class DecisionsResponse(DecisionsObjectBase): + model: str + answers: Sequence[DecisionAnswer] + usage: DecisionUsage + + _hidden_params: dict[str, object] = PrivateAttr(default_factory=dict) + + @property + def hidden_params(self) -> dict[str, object]: # mutable-ok: API requires mutation + return self._hidden_params + + def set_hidden_params(self, params: Mapping[str, object]) -> None: + self._hidden_params.update(params) diff --git a/tests/unit/llms/base_llm/decisions/__init__.py b/tests/unit/llms/base_llm/decisions/__init__.py new file mode 100644 index 00000000000..e69de29bb2d diff --git a/tests/unit/llms/base_llm/decisions/test_systemone.py b/tests/unit/llms/base_llm/decisions/test_systemone.py new file mode 100644 index 00000000000..42a1c41b3d3 --- /dev/null +++ b/tests/unit/llms/base_llm/decisions/test_systemone.py @@ -0,0 +1,262 @@ +from __future__ import annotations + +from collections.abc import Mapping, Sequence +from typing import Final + +import pytest +from pydantic import TypeAdapter + +from litellm.llms.base_llm.chat.transformation import BaseLLMException +from litellm.llms.base_llm.decisions.systemone import ( + SYSTEM_ONE_RESPONSE_ADAPTER, + question_keys, + to_decisions_response, + to_system_one_request, +) +from litellm.types.openai_decisions import ( + ChoiceAnswer, + DecisionsRequest, + DecisionsRequestBody, + PredicateAnswer, + ScoreAnswer, +) + +_INPUT: Final = "The export job hangs at 99% and never finishes" +_QUESTIONS: Final[Sequence[Mapping[str, object]]] = ( + {"type": "predicate", "name": "is_defect", "instructions": "Is this a defect?"}, + { + "type": "choice", + "name": "sentiment", + "instructions": "How does the customer feel?", + "choices": [{"value": "positive"}, {"value": "negative", "description": "unhappy"}], + }, + { + "type": "score", + "name": "severity", + "instructions": "How severe is it?", + "levels": [{"label": "none"}, {"label": "low"}, {"label": "high", "description": "blocks users"}], + }, +) +_SYSTEM_ONE_QUESTIONS: Final[Mapping[str, object]] = { + "is_defect": {"type": "noul", "instructions": "Is this a defect?"}, + "sentiment": { + "type": "choice", + "instructions": "How does the customer feel?", + "criteria": {"positive": None, "negative": "unhappy"}, + }, + "severity": {"type": "score", "instructions": "How severe is it?", "criteria": ["none", "low", "blocks users"]}, +} +_SYSTEM_ONE_RESPONSE: Final[Mapping[str, object]] = { + "model": "jev-1.13", + "answers": { + "is_defect": {"type": "noul", "noul": 0.9}, + "sentiment": { + "type": "choice", + "choice": "positive", + "confidence": 0.8, + "probabilities": {"positive": 0.8, "negative": 0.2}, + }, + "severity": { + "type": "score", + "score": 1, + "confidence": 0.7, + "legend": {"0": "none", "1": "low", "2": "high"}, + "probabilities": {"0": 0.1, "1": 0.8, "2": 0.1}, + }, + }, + "usage": {"input_tokens": 367, "output_tokens": 3}, +} +_EXPECTED_ANSWERS: Final[Sequence[Mapping[str, object]]] = ( + {"type": "predicate", "name": "is_defect", "probability": 0.9}, + { + "type": "choice", + "name": "sentiment", + "choice": "positive", + "probabilities": [{"value": "positive", "probability": 0.8}, {"value": "negative", "probability": 0.2}], + "confidence": 0.8, + }, + { + "type": "score", + "name": "severity", + "score": 1.0, + "probabilities": [ + {"value": 0, "label": "none", "probability": 0.1}, + {"value": 1, "label": "low", "probability": 0.8}, + {"value": 2, "label": "high", "probability": 0.1}, + ], + "confidence": 0.7, + }, +) +_EXPECTED_USAGE: Final[Mapping[str, object]] = { + "input_tokens": 367, + "input_tokens_details": {"cached_tokens": 0, "cache_write_tokens": 0}, + "output_tokens": 3, + "output_tokens_details": {"reasoning_tokens": 0}, + "total_tokens": 370, +} +_BODY_ADAPTER: Final[TypeAdapter[DecisionsRequestBody]] = TypeAdapter(DecisionsRequestBody) + + +def _body( + input_value: object = _INPUT, + questions: Sequence[Mapping[str, object]] = _QUESTIONS, +) -> DecisionsRequestBody: + return _BODY_ADAPTER.validate_python({"input": input_value, "questions": questions}) + + +def _request( + input_value: object = _INPUT, + questions: Sequence[Mapping[str, object]] = _QUESTIONS, + model: str = "jev-1.13", +) -> DecisionsRequest: + return DecisionsRequest(model=model, body=_body(input_value, questions)) + + +def _predicate(name: str | None = "is_defect") -> tuple[Mapping[str, object]]: + return ({"type": "predicate", "name": name, "instructions": "Is this a defect?"},) + + +def test_openai_request_becomes_the_system_one_body() -> None: + assert to_system_one_request("jev-1.13", _body(), "typesafe") == { + "model": "jev-1.13", + "state": _INPUT, + "questions": _SYSTEM_ONE_QUESTIONS, + } + + +def test_system_one_answers_become_openai_answers_in_question_order() -> None: + response: Final = to_decisions_response( + SYSTEM_ONE_RESPONSE_ADAPTER.validate_python(_SYSTEM_ONE_RESPONSE), _request(), "typesafe" + ) + + assert response.model_dump(mode="json") == { + "model": "jev-1.13", + "answers": list(_EXPECTED_ANSWERS), + "usage": _EXPECTED_USAGE, + } + assert isinstance(response.answers[0], PredicateAnswer) + assert isinstance(response.answers[1], ChoiceAnswer) + assert isinstance(response.answers[2], ScoreAnswer) + + +def test_user_messages_are_joined_into_one_system_one_state() -> None: + messages: Final = [ + {"role": "user", "content": "first"}, + { + "role": "user", + "content": [{"type": "input_text", "text": "second"}, {"type": "input_text", "text": "third"}], + }, + ] + + body: Final = to_system_one_request("jev-1.13", _body(messages, _predicate()), "typesafe") + + assert body["state"] == "first\nsecond\nthird" + + +@pytest.mark.parametrize( + ("label", "input_value", "questions"), + ( + ( + "input_image", + [{"role": "user", "content": [{"type": "input_image", "image_url": "data:image/png;base64,AA=="}]}], + _predicate(), + ), + ("boolean choice", _INPUT, [{"type": "choice", "instructions": "Refund?", "choices": [{"value": True}]}]), + ("unique name", _INPUT, [*_predicate(), *_predicate()]), + ( + "repeated choice", + _INPUT, + [{"type": "choice", "instructions": "Refund?", "choices": [{"value": "yes"}, {"value": "yes"}]}], + ), + ), +) +def test_what_system_one_cannot_express_is_a_400( + label: str, + input_value: object, + questions: Sequence[Mapping[str, object]], +) -> None: + with pytest.raises(BaseLLMException, match=label) as error: + to_system_one_request("jev-1.13", _body(input_value, questions), "perplexity") + + assert error.value.status_code == 400 + assert "perplexity" in error.value.message + + +def test_unnamed_questions_get_positional_keys_that_never_shadow_a_supplied_name() -> None: + body: Final = _body(questions=[*_predicate(None), *_predicate("q0"), *_predicate(None)]) + + assert question_keys(body.questions, "typesafe") == ("_q0", "q0", "q2") + noul: Final = {"type": "noul", "instructions": "Is this a defect?"} + assert to_system_one_request("jev-1.13", body, "typesafe")["questions"] == {"_q0": noul, "q0": noul, "q2": noul} + + +def test_positional_answers_come_back_in_question_order_without_a_name() -> None: + request: Final = _request(questions=[*_predicate(None), *_predicate("q0")]) + system_one: Final = SYSTEM_ONE_RESPONSE_ADAPTER.validate_python( + {"answers": {"_q0": {"type": "noul", "noul": 0.25}, "q0": {"type": "noul", "noul": 0.75}}} + ) + + response: Final = to_decisions_response(system_one, request, "typesafe") + + assert [answer.model_dump(mode="json") for answer in response.answers] == [ + {"type": "predicate", "name": None, "probability": 0.25}, + {"type": "predicate", "name": "q0", "probability": 0.75}, + ] + assert response.usage.model_dump(mode="json") == { + **_EXPECTED_USAGE, + "input_tokens": 0, + "output_tokens": 0, + "total_tokens": 0, + } + + +def test_a_reply_without_a_model_reports_the_requested_model() -> None: + system_one: Final = SYSTEM_ONE_RESPONSE_ADAPTER.validate_python( + {k: v for k, v in _SYSTEM_ONE_RESPONSE.items() if k != "model"} + ) + + response: Final = to_decisions_response(system_one, _request(model="typesafe/jev-1.13.0"), "typesafe") + + assert response.model == "typesafe/jev-1.13.0" + + +def test_a_choice_the_provider_left_out_of_probabilities_is_reported_at_zero() -> None: + system_one: Final = SYSTEM_ONE_RESPONSE_ADAPTER.validate_python( + { + "answers": { + "sentiment": { + "type": "choice", + "choice": "positive", + "confidence": 1.0, + "probabilities": {"positive": 1.0}, + } + } + } + ) + request: Final = _request(questions=_QUESTIONS[1:2]) + + response: Final = to_decisions_response(system_one, request, "typesafe") + + assert response.answers[0].model_dump(mode="json") == { + "type": "choice", + "name": "sentiment", + "choice": "positive", + "probabilities": [{"value": "positive", "probability": 1.0}, {"value": "negative", "probability": 0.0}], + "confidence": 1.0, + } + + +@pytest.mark.parametrize( + "answers", + ( + {}, + {"is_defect": {"type": "choice", "choice": "yes", "confidence": 1.0, "probabilities": {"yes": 1.0}}}, + ), +) +def test_a_reply_without_a_matching_answer_is_a_server_error(answers: Mapping[str, object]) -> None: + system_one: Final = SYSTEM_ONE_RESPONSE_ADAPTER.validate_python({"answers": answers}) + + with pytest.raises(BaseLLMException, match="no predicate answer for question 'is_defect'") as error: + to_decisions_response(system_one, _request(questions=_predicate()), "typesafe") + + assert error.value.status_code == 500 diff --git a/tests/unit/types/test_openai_decisions.py b/tests/unit/types/test_openai_decisions.py new file mode 100644 index 00000000000..d9387d644d5 --- /dev/null +++ b/tests/unit/types/test_openai_decisions.py @@ -0,0 +1,150 @@ +from __future__ import annotations + +from collections.abc import Mapping +from typing import Final + +import pytest +from pydantic import TypeAdapter, ValidationError + +from litellm.types.openai_decisions import ( + ChoiceAnswer, + DecisionsRequestBody, + DecisionsResponse, + RefusalAnswer, + ScoreQuestion, +) + +_REQUEST: Final[Mapping[str, object]] = { + "input": [ + { + "role": "user", + "content": [ + {"type": "input_text", "text": "Is this receipt a valid business expense?"}, + {"type": "input_image", "image_url": "https://example.com/receipt.png", "detail": "high"}, + ], + } + ], + "questions": [ + {"type": "predicate", "name": "is_expense", "instructions": "Is this a business expense?"}, + { + "type": "choice", + "name": "approve", + "instructions": "Should this be approved?", + "choices": [{"value": True, "description": "approve"}, {"value": False, "description": "reject"}], + }, + { + "type": "score", + "name": "risk", + "instructions": "How risky is this expense?", + "levels": [{"label": "low"}, {"label": "high", "description": "needs a manager"}], + }, + ], + "safety_identifier": "user-123", +} +_RESPONSE: Final[Mapping[str, object]] = { + "model": "gpt-6-luna", + "answers": [ + {"type": "predicate", "name": "is_expense", "probability": 0.92}, + { + "type": "choice", + "name": "approve", + "choice": True, + "probabilities": [{"value": True, "probability": 0.7}, {"value": False, "probability": 0.3}], + "confidence": 0.7, + }, + {"type": "refusal", "name": "risk"}, + ], + "usage": { + "input_tokens": 120, + "input_tokens_details": {"cached_tokens": 100, "cache_write_tokens": 0}, + "output_tokens": 12, + "output_tokens_details": {"reasoning_tokens": 4}, + "total_tokens": 132, + }, +} +_REQUEST_ADAPTER: Final[TypeAdapter[DecisionsRequestBody]] = TypeAdapter(DecisionsRequestBody) +_RESPONSE_ADAPTER: Final[TypeAdapter[DecisionsResponse]] = TypeAdapter(DecisionsResponse) + + +def test_the_documented_request_round_trips_with_its_boolean_choices_and_image_part() -> None: + request: Final = _REQUEST_ADAPTER.validate_python(_REQUEST) + + assert request.model_dump(mode="json", exclude_none=True) == _REQUEST + assert isinstance(request.questions[2], ScoreQuestion) + + +def test_the_documented_response_keeps_answer_order_refusals_and_token_details() -> None: + response: Final = _RESPONSE_ADAPTER.validate_python(_RESPONSE) + + assert response.model_dump(mode="json") == _RESPONSE + assert isinstance(response.answers[1], ChoiceAnswer) + assert isinstance(response.answers[2], RefusalAnswer) + + +_OFF_SPEC: Final[tuple[tuple[str, object], ...]] = ( + ("questions", [{"type": "noul", "name": "q", "instructions": "x"}]), + ("questions", [{"type": "choice", "name": "q", "instructions": "x", "choices": [{"value": 1}]}]), + ("questions", [{"type": "score", "name": "q", "levels": [{"label": "low"}]}]), + ("input", {"state": "not an OpenAI input"}), +) + + +@pytest.mark.parametrize(("field", "value"), _OFF_SPEC) +def test_requests_off_the_spec_are_rejected(field: str, value: object) -> None: + with pytest.raises(ValidationError): + _REQUEST_ADAPTER.validate_python({**_REQUEST, field: value}) + + +_EMPTY_COLLECTIONS: Final[tuple[tuple[str, list[object]], ...]] = ( + ("questions", []), + ("questions", [{"type": "choice", "name": "q", "instructions": "x", "choices": []}]), + ("questions", [{"type": "score", "name": "q", "instructions": "x", "levels": []}]), +) + + +@pytest.mark.parametrize(("field", "value"), _EMPTY_COLLECTIONS) +def test_empty_collections_are_left_for_the_provider_to_judge(field: str, value: list[object]) -> None: + request: Final = _REQUEST_ADAPTER.validate_python({**_REQUEST, field: value}) + + assert request.model_dump(mode="json", exclude_none=True)[field] == value + + +def test_choice_values_keep_their_type_so_a_string_true_and_a_boolean_true_stay_distinct() -> None: + answer: Final = { + "type": "choice", + "name": "approve", + "choice": "true", + "probabilities": [{"value": "true", "probability": 0.6}, {"value": True, "probability": 0.4}], + "confidence": 0.6, + } + + response: Final = _RESPONSE_ADAPTER.validate_python({**_RESPONSE, "answers": [answer]}) + + assert response.model_dump(mode="json")["answers"] == [answer] + with pytest.raises(ValidationError): + _RESPONSE_ADAPTER.validate_python({**_RESPONSE, "answers": [{**answer, "choice": 1}]}) + + +_OFF_SPEC_RESPONSE: Final[tuple[str, ...]] = ("model", "usage") + + +@pytest.mark.parametrize("field", _OFF_SPEC_RESPONSE) +def test_responses_missing_a_required_field_are_rejected(field: str) -> None: + with pytest.raises(ValidationError): + _RESPONSE_ADAPTER.validate_python({k: v for k, v in _RESPONSE.items() if k != field}) + + +def test_usage_without_token_details_is_rejected() -> None: + usage: Final = {"input_tokens": 120, "output_tokens": 12, "total_tokens": 132} + + with pytest.raises(ValidationError): + _RESPONSE_ADAPTER.validate_python({**_RESPONSE, "usage": usage}) + + +def test_hidden_params_live_outside_the_wire_body() -> None: + response: Final = _RESPONSE_ADAPTER.validate_python(_RESPONSE) + + response.set_hidden_params({"custom_llm_provider": "openai"}) + + assert response.hidden_params == {"custom_llm_provider": "openai"} + assert "_hidden_params" not in response.model_dump(mode="json")