diff --git a/litellm/__init__.py b/litellm/__init__.py index 4cc1399ee9e..3f93166ecc8 100644 --- a/litellm/__init__.py +++ b/litellm/__init__.py @@ -1682,6 +1682,9 @@ if TYPE_CHECKING: from .llms.strands_decider.decisions.transformation import ( StrandsDeciderDecisionsConfig as StrandsDeciderDecisionsConfig, ) + from .llms.openai.decisions.transformation import ( + OpenAIDecisionsConfig as OpenAIDecisionsConfig, + ) from .llms.nvidia_nim.rerank.transformation import ( NvidiaNimRerankConfig as NvidiaNimRerankConfig, ) diff --git a/litellm/_lazy_imports_registry.py b/litellm/_lazy_imports_registry.py index f95679684b8..abefab95e9f 100644 --- a/litellm/_lazy_imports_registry.py +++ b/litellm/_lazy_imports_registry.py @@ -161,6 +161,7 @@ LLM_CONFIG_NAMES: Final = ( "OpenRouterDecisionsConfig", "CloudflareDecisionsConfig", "StrandsDeciderDecisionsConfig", + "OpenAIDecisionsConfig", "NvidiaNimRerankConfig", "NvidiaNimRankingConfig", "VertexAIRerankConfig", @@ -720,6 +721,7 @@ _LLM_CONFIGS_IMPORT_MAP: Final = { ".llms.strands_decider.decisions.transformation", "StrandsDeciderDecisionsConfig", ), + "OpenAIDecisionsConfig": (".llms.openai.decisions.transformation", "OpenAIDecisionsConfig"), "NvidiaNimRerankConfig": ( ".llms.nvidia_nim.rerank.transformation", "NvidiaNimRerankConfig", diff --git a/litellm/cost_calculator.py b/litellm/cost_calculator.py index 4d5f882ea10..da420ab0578 100644 --- a/litellm/cost_calculator.py +++ b/litellm/cost_calculator.py @@ -9,7 +9,7 @@ from typing import TYPE_CHECKING, Any, Final, Literal, cast from httpx import Response from pydantic import BaseModel -from typing_extensions import ReadOnly, TypedDict +from typing_extensions import ReadOnly, TypedDict, assert_never import litellm import litellm._logging @@ -100,7 +100,7 @@ from litellm.llms.vertex_ai.cost_calculator import cost_router as google_cost_ro from litellm.llms.xai.cost_calculator import cost_per_token as xai_cost_per_token from litellm.responses.utils import ResponseAPILoggingUtils from litellm.types.agents import LiteLLMSendMessageResponse -from litellm.types.decisions import DecisionsResponse, DecisionsUsage +from litellm.types.decisions import DecisionsResponse, DecisionsUsage, OpenAIDecisionResponse, OpenAIDecisionUsage from litellm.types.llms.base import CachedTokensDetails, LiteLLMBaseModel from litellm.types.llms.openai import ( HttpxBinaryResponseContent, @@ -1034,6 +1034,8 @@ def get_usage_object( ), ) + if isinstance(completion_response, (DecisionsResponse, OpenAIDecisionResponse)): + return None if completion_response.usage is None else _decisions_usage(completion_response.usage) if usage_obj is None: return None if isinstance(usage_obj, Usage): @@ -1064,12 +1066,37 @@ def get_usage_object( return None +def _decisions_prompt_tokens_details(usage: DecisionsUsage | OpenAIDecisionUsage) -> PromptTokensDetailsWrapper: + match usage: + case DecisionsUsage(): + return PromptTokensDetailsWrapper( + cached_tokens=usage.cached_tokens, cache_write_tokens=usage.cache_write_tokens + ) + case OpenAIDecisionUsage(): + return PromptTokensDetailsWrapper( + cached_tokens=usage.input_tokens_details.cached_tokens, + cache_write_tokens=usage.input_tokens_details.cache_write_tokens, + ) + case _: + assert_never(usage) + + +def _decisions_usage(usage: DecisionsUsage | OpenAIDecisionUsage) -> Usage: + return Usage( + prompt_tokens=usage.input_tokens, + completion_tokens=usage.output_tokens, + total_tokens=usage.input_tokens + usage.output_tokens, + prompt_tokens_details=_decisions_prompt_tokens_details(usage), + ) + + def _is_known_usage_objects(usage_obj): """Returns True if the usage obj is a known Usage type""" return ( isinstance(usage_obj, litellm.Usage) or isinstance(usage_obj, ResponseAPIUsage) or isinstance(usage_obj, DecisionsUsage) + or isinstance(usage_obj, OpenAIDecisionUsage) or TranscriptionUsageObjectTransformation.is_transcription_usage_object(usage_obj) ) @@ -1481,11 +1508,8 @@ def completion_cost( "usage", litellm.Usage(**_usage_for_dump.model_dump()), ) - if isinstance(usage_obj, DecisionsUsage): - _usage = { - "prompt_tokens": usage_obj.input_tokens, - "completion_tokens": usage_obj.output_tokens, - } + if isinstance(usage_obj, (DecisionsUsage, OpenAIDecisionUsage)): + _usage = _decisions_usage(usage_obj).model_dump() elif usage_obj is None: _usage = {} elif isinstance(usage_obj, BaseModel): @@ -1977,7 +2001,8 @@ def response_cost_calculator( | OpenAIModerationResponse | Response | SearchResponse - | DecisionsResponse, + | DecisionsResponse + | OpenAIDecisionResponse, model: str, custom_llm_provider: str | None, call_type: Literal[ diff --git a/litellm/decisions/main.py b/litellm/decisions/main.py index ecb2669db9c..25ec501fd48 100644 --- a/litellm/decisions/main.py +++ b/litellm/decisions/main.py @@ -1,34 +1,56 @@ -from collections.abc import Mapping +from collections.abc import Mapping, Sequence from dataclasses import dataclass, field -from typing import Final +from typing import Final, TypeAlias import httpx from pydantic import TypeAdapter, ValidationError +from typing_extensions import assert_never import litellm from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj from litellm.llms.base_llm.chat.transformation import BaseLLMException -from litellm.llms.base_llm.decisions.transformation import BaseDecisionsConfig +from litellm.llms.base_llm.decisions.transformation import ( + BaseDecisionsConfig, + ir_to_systemone_response, + systemone_request_to_ir, +) from litellm.llms.custom_httpx.llm_http_handler import BaseLLMHTTPHandler +from litellm.llms.openai.decisions.transformation import ir_to_openai_response, openai_request_to_ir from litellm.types.decisions import ( DecisionQuestion, + DecisionsIRRequest, + DecisionsIRResponse, DecisionsJSON, - DecisionsRequest, + DecisionsRequestBody, DecisionsResponse, + OpenAIDecisionInput, + OpenAIDecisionQuestion, + OpenAIDecisionRequestBody, + OpenAIDecisionResponse, + UnsupportedDecisionsRequest, ) from litellm.types.utils import LlmProviders from litellm.utils import ProviderConfigManager, client -_DECISIONS_REQUEST_ADAPTER: Final[TypeAdapter[DecisionsRequest]] = TypeAdapter(DecisionsRequest) +DecisionsQuestions: TypeAlias = ( + Mapping[str, DecisionQuestion | Mapping[str, object]] | Sequence[OpenAIDecisionQuestion | Mapping[str, object]] +) +DecisionsRequestFormat: TypeAlias = DecisionsRequestBody | OpenAIDecisionRequestBody + +_SYSTEMONE_REQUEST_ADAPTER: Final[TypeAdapter[DecisionsRequestBody]] = TypeAdapter(DecisionsRequestBody) +_OPENAI_REQUEST_ADAPTER: Final[TypeAdapter[OpenAIDecisionRequestBody]] = TypeAdapter(OpenAIDecisionRequestBody) _HANDLER: Final = BaseLLMHTTPHandler() @dataclass(frozen=True, slots=True, repr=False) class _DecisionsCall: model: str + requested_model: str custom_llm_provider: str provider_config: BaseDecisionsConfig - request: DecisionsRequest + request: DecisionsRequestFormat + ir_request: DecisionsIRRequest + body: Mapping[str, object] = field(repr=False) api_base: str api_key: str | None = field(repr=False) logging_obj: LiteLLMLoggingObj | None @@ -59,11 +81,37 @@ def _provider_config(model: str, custom_llm_provider: str) -> BaseDecisionsConfi return provider_config +def _validate_request( + *, + state: DecisionsJSON | None, + questions: DecisionsQuestions | None, + decision_input: OpenAIDecisionInput | None, + safety_identifier: str | None, +) -> DecisionsRequestFormat: + if decision_input is None: + return _SYSTEMONE_REQUEST_ADAPTER.validate_python({"state": state, "questions": questions}) + return _OPENAI_REQUEST_ADAPTER.validate_python( + {"input": decision_input, "questions": questions, "safety_identifier": safety_identifier} + ) + + +def _ir_request(request: DecisionsRequestFormat) -> DecisionsIRRequest: + match request: + case DecisionsRequestBody(): + return systemone_request_to_ir(request) + case OpenAIDecisionRequestBody(): + return openai_request_to_ir(request) + case _: + assert_never(request) + + def _prepare_call( *, model: str, - state: DecisionsJSON, - questions: Mapping[str, DecisionQuestion | Mapping[str, object]], + state: DecisionsJSON | None, + questions: DecisionsQuestions | None, + decision_input: OpenAIDecisionInput | None, + safety_identifier: str | None, api_key: str | None, api_base: str | None, timeout: float | httpx.Timeout | None, @@ -85,9 +133,15 @@ def _prepare_call( model=model, llm_provider=provider, ) + if state is not None and decision_input is not None: + raise litellm.BadRequestError( + message="Pass either state (System One format) or input (OpenAI format) to the Decisions API, not both", + model=model, + llm_provider=provider, + ) try: - request: Final = _DECISIONS_REQUEST_ADAPTER.validate_python( - {"model": canonical_model, "state": state, "questions": questions} + request: Final = _validate_request( + state=state, questions=questions, decision_input=decision_input, safety_identifier=safety_identifier ) except ValidationError as error: raise litellm.BadRequestError( @@ -111,6 +165,17 @@ def _prepare_call( llm_provider=provider, ) + ir_request: Final = _ir_request(request) + body: Final = provider_config.transform_decisions_request( + model=canonical_model, request=ir_request, custom_llm_provider=provider + ) + if isinstance(body, UnsupportedDecisionsRequest): + raise litellm.BadRequestError( + message=f"Decisions provider '{provider}' cannot serve this request: {body.reason}", + model=model, + llm_provider=provider, + ) + logging_obj: Final = kwargs.get("litellm_logging_obj") if isinstance(logging_obj, LiteLLMLoggingObj): logging_obj.update_from_kwargs( @@ -124,9 +189,12 @@ def _prepare_call( ) return _DecisionsCall( model=canonical_model, + requested_model=model, custom_llm_provider=provider, provider_config=provider_config, request=request, + ir_request=ir_request, + body=body, api_base=resolved_api_base, api_key=resolved_api_key, logging_obj=logging_obj if isinstance(logging_obj, LiteLLMLoggingObj) else None, @@ -135,6 +203,30 @@ def _prepare_call( ) +def _format_response(response: DecisionsIRResponse, call: _DecisionsCall) -> DecisionsResponse | OpenAIDecisionResponse: + formatted: Final = _formatted_response(response, call) + formatted.set_hidden_params( + { + "model": f"{call.custom_llm_provider}/{call.model}", + "custom_llm_provider": call.custom_llm_provider, + "provider_response_model": f"{call.custom_llm_provider}/{call.model}", + } + ) + return formatted + + +def _formatted_response( + response: DecisionsIRResponse, call: _DecisionsCall +) -> DecisionsResponse | OpenAIDecisionResponse: + match call.request: + case DecisionsRequestBody(): + return ir_to_systemone_response(response, call.ir_request) + case OpenAIDecisionRequestBody(): + return ir_to_openai_response(response, call.ir_request, call.requested_model) + case _: + assert_never(call.request) + + def _map_upstream_exception(error: Exception, call: _DecisionsCall) -> Exception: if isinstance(error, BaseLLMException) and error.status_code_is_synthesized: provider_label: Final = f"{call.custom_llm_provider[0].upper()}{call.custom_llm_provider[1:]}Exception" @@ -153,19 +245,23 @@ def _map_upstream_exception(error: Exception, call: _DecisionsCall) -> Exception @client async def adecisions( model: str, - state: DecisionsJSON, - questions: Mapping[str, DecisionQuestion | Mapping[str, object]], + state: DecisionsJSON | None = None, + questions: DecisionsQuestions | None = None, api_key: str | None = None, api_base: str | None = None, timeout: float | httpx.Timeout | None = None, custom_llm_provider: str | None = None, extra_headers: Mapping[str, str] | None = None, + input: OpenAIDecisionInput | None = None, + safety_identifier: str | None = None, **kwargs: object, -) -> DecisionsResponse: +) -> DecisionsResponse | OpenAIDecisionResponse: call: Final = _prepare_call( model=model, state=state, questions=questions, + decision_input=input, + safety_identifier=safety_identifier, api_key=api_key, api_base=api_base, timeout=timeout, @@ -174,12 +270,13 @@ async def adecisions( kwargs=kwargs, ) try: - return await _HANDLER.adecisions( + response: Final = await _HANDLER.adecisions( model=call.model, custom_llm_provider=call.custom_llm_provider, logging_obj=call.logging_obj, provider_config=call.provider_config, - request=call.request, + request=call.ir_request, + body=call.body, api_base=call.api_base, api_key=call.api_key, headers=call.headers, @@ -187,24 +284,29 @@ async def adecisions( ) except Exception as error: raise _map_upstream_exception(error, call) from error + return _format_response(response, call) @client def decisions( model: str, - state: DecisionsJSON, - questions: Mapping[str, DecisionQuestion | Mapping[str, object]], + state: DecisionsJSON | None = None, + questions: DecisionsQuestions | None = None, api_key: str | None = None, api_base: str | None = None, timeout: float | httpx.Timeout | None = None, custom_llm_provider: str | None = None, extra_headers: Mapping[str, str] | None = None, + input: OpenAIDecisionInput | None = None, + safety_identifier: str | None = None, **kwargs: object, -) -> DecisionsResponse: +) -> DecisionsResponse | OpenAIDecisionResponse: call: Final = _prepare_call( model=model, state=state, questions=questions, + decision_input=input, + safety_identifier=safety_identifier, api_key=api_key, api_base=api_base, timeout=timeout, @@ -213,12 +315,13 @@ def decisions( kwargs=kwargs, ) try: - return _HANDLER.decisions( + response: Final = _HANDLER.decisions( model=call.model, custom_llm_provider=call.custom_llm_provider, logging_obj=call.logging_obj, provider_config=call.provider_config, - request=call.request, + request=call.ir_request, + body=call.body, api_base=call.api_base, api_key=call.api_key, headers=call.headers, @@ -226,6 +329,7 @@ def decisions( ) except Exception as error: raise _map_upstream_exception(error, call) from error + return _format_response(response, call) __all__ = ["adecisions", "decisions"] diff --git a/litellm/decisions/openai_transformation.py b/litellm/decisions/openai_transformation.py deleted file mode 100644 index af2c838b9ab..00000000000 --- a/litellm/decisions/openai_transformation.py +++ /dev/null @@ -1,127 +0,0 @@ -from collections.abc import Mapping, Sequence -from typing import Final - -from typing_extensions import assert_never - -from litellm.types.decisions import ( - ChoiceAnswer, - DecisionAnswer, - DecisionsResponse, - DecisionsUsage, - NoulAnswer, - OpenAIChoiceAnswer, - OpenAIChoiceProbability, - OpenAIChoiceQuestion, - OpenAIDecisionAnswer, - OpenAIDecisionInputMessage, - OpenAIDecisionQuestion, - OpenAIDecisionRequestBody, - OpenAIDecisionResponse, - OpenAIDecisionUsage, - OpenAIPredicateAnswer, - OpenAIPredicateQuestion, - OpenAIRefusalAnswer, - OpenAIScoreAnswer, - OpenAIScoreLevel, - OpenAIScoreProbability, - OpenAIScoreQuestion, - ScoreAnswer, - systemone_choice_key, -) - -_OPENAI_ONLY_FIELDS: Final = frozenset({"input", "questions", "safety_identifier"}) - - -def _message_text(message: OpenAIDecisionInputMessage) -> str: - if isinstance(message.content, str): - return message.content - return "\n\n".join(part.text for part in message.content) - - -def _state(decision_input: str | Sequence[OpenAIDecisionInputMessage]) -> str: - if isinstance(decision_input, str): - return decision_input - return "\n\n".join(_message_text(message) for message in decision_input) - - -def _level_criterion(level: OpenAIScoreLevel) -> str: - return level.label if level.description is None else f"{level.label}: {level.description}" - - -def _systemone_question(question: OpenAIDecisionQuestion) -> Mapping[str, object]: - match question: - case OpenAIPredicateQuestion(): - return {"type": "noul", "instructions": question.instructions} - case OpenAIChoiceQuestion(): - return { - "type": "choice", - "instructions": question.instructions, - "criteria": {systemone_choice_key(option.value): option.description for option in question.choices}, - } - case OpenAIScoreQuestion(): - return { - "type": "score", - "instructions": question.instructions, - "criteria": [_level_criterion(level) for level in question.levels], - } - case _: - assert_never(question) - - -def to_systemone_request(request_data: Mapping[str, object], body: OpenAIDecisionRequestBody) -> Mapping[str, object]: - return { - **{key: value for key, value in request_data.items() if key not in _OPENAI_ONLY_FIELDS}, - "state": _state(body.input), - "questions": {str(index): _systemone_question(question) for index, question in enumerate(body.questions)}, - } - - -def _openai_answer(question: OpenAIDecisionQuestion, answer: DecisionAnswer | None) -> OpenAIDecisionAnswer: - match question, answer: - case OpenAIPredicateQuestion(), NoulAnswer(): - return OpenAIPredicateAnswer(name=question.name, probability=answer.noul) - case OpenAIChoiceQuestion(), ChoiceAnswer(): - typed_values: Final = {systemone_choice_key(option.value): option.value for option in question.choices} - return OpenAIChoiceAnswer( - name=question.name, - choice=typed_values.get(answer.choice, answer.choice), - probabilities=tuple( - OpenAIChoiceProbability( - value=option.value, - probability=answer.probabilities.get(systemone_choice_key(option.value), 0.0), - ) - for option in question.choices - ), - confidence=answer.confidence, - ) - case OpenAIScoreQuestion(), ScoreAnswer(): - return OpenAIScoreAnswer( - name=question.name, - score=answer.score, - probabilities=tuple( - OpenAIScoreProbability( - value=index, label=level.label, probability=answer.probabilities.get(str(index), 0.0) - ) - for index, level in enumerate(question.levels) - ), - confidence=answer.confidence, - ) - case _: - return OpenAIRefusalAnswer(name=question.name) - - -def to_openai_response( - response: DecisionsResponse, questions: Sequence[OpenAIDecisionQuestion], requested_model: str -) -> OpenAIDecisionResponse: - usage: Final = response.usage or DecisionsUsage() - return OpenAIDecisionResponse( - model=response.model or requested_model, - answers=tuple( - _openai_answer(question, response.answers.get(str(index))) for index, question in enumerate(questions) - ), - usage=OpenAIDecisionUsage( - input_tokens=usage.input_tokens, - output_tokens=usage.output_tokens, - total_tokens=usage.input_tokens + usage.output_tokens, - ), - ) diff --git a/litellm/litellm_core_utils/litellm_logging.py b/litellm/litellm_core_utils/litellm_logging.py index eb11bd9c11c..715e492b159 100644 --- a/litellm/litellm_core_utils/litellm_logging.py +++ b/litellm/litellm_core_utils/litellm_logging.py @@ -120,7 +120,7 @@ from litellm.llms.base_llm.search.transformation import SearchResponse from litellm.responses.utils import ResponseAPILoggingUtils from litellm.types.agents import LiteLLMSendMessageResponse from litellm.types.containers.main import ContainerObject -from litellm.types.decisions import DecisionsResponse +from litellm.types.decisions import DecisionsResponse, OpenAIDecisionResponse from litellm.types.integrations.s3_v2 import S3PartitionGranularity from litellm.types.interactions import ( InteractionsAPIResponse, @@ -2650,6 +2650,7 @@ class Logging(LiteLLMLoggingBaseClass): or isinstance(logging_result, OCRResponse) # OCR or isinstance(logging_result, SearchResponse) # Search API or isinstance(logging_result, DecisionsResponse) + or isinstance(logging_result, OpenAIDecisionResponse) or ( isinstance(logging_result, InteractionsAPIResponse) and logging_result.usage is not None diff --git a/litellm/llms/base_llm/decisions/transformation.py b/litellm/llms/base_llm/decisions/transformation.py index 36739555cea..a8d14ca59a3 100644 --- a/litellm/llms/base_llm/decisions/transformation.py +++ b/litellm/llms/base_llm/decisions/transformation.py @@ -1,17 +1,299 @@ +import json from abc import ABC -from collections.abc import Mapping +from collections.abc import Mapping, Sequence +from types import MappingProxyType from typing import Final import httpx from pydantic import TypeAdapter, ValidationError +from typing_extensions import assert_never from litellm.llms.base_llm.chat.transformation import BaseLLMException from litellm.secret_managers.main import get_secret_str -from litellm.types.decisions import DecisionsRequest, DecisionsResponse +from litellm.types.decisions import ( + ChoiceAnswer, + ChoiceQuestion, + DecisionAnswer, + DecisionQuestion, + DecisionsIRAnswer, + DecisionsIRChoiceAnswer, + DecisionsIRChoiceOption, + DecisionsIRChoiceProbability, + DecisionsIRChoiceQuestion, + DecisionsIRMessages, + DecisionsIRPredicateAnswer, + DecisionsIRPredicateQuestion, + DecisionsIRQuestion, + DecisionsIRRefusal, + DecisionsIRRequest, + DecisionsIRResponse, + DecisionsIRScoreAnswer, + DecisionsIRScoreLevel, + DecisionsIRScoreProbability, + DecisionsIRScoreQuestion, + DecisionsIRState, + DecisionsIRUsage, + DecisionsJSON, + DecisionsRequestBody, + DecisionsResponse, + DecisionsUsage, + NoulAnswer, + NoulQuestion, + OpenAIDecisionInputImage, + OpenAIDecisionInputMessage, + OpenAIDecisionInputText, + ScoreAnswer, + ScoreQuestion, + UnsupportedDecisionsRequest, + systemone_choice_key, +) -PAYLOAD_ADAPTER: Final[TypeAdapter[object]] = TypeAdapter(object) -_RESPONSE_ADAPTER: Final[TypeAdapter[DecisionsResponse]] = TypeAdapter(DecisionsResponse) +_PAYLOAD_ADAPTER: Final[TypeAdapter[object]] = TypeAdapter(object) +_SYSTEMONE_RESPONSE_ADAPTER: Final[TypeAdapter[DecisionsResponse]] = TypeAdapter(DecisionsResponse) _RESERVED_HEADERS: Final[frozenset[str]] = frozenset({"authorization", "content-type"}) +_TEXT_ONLY: Final = UnsupportedDecisionsRequest( + reason="input_image content parts are not supported because System One providers accept text input only" +) + + +def decisions_text(value: DecisionsJSON) -> str: + return value if isinstance(value, str) else json.dumps(value) + + +def systemone_keys(questions: Sequence[DecisionsIRQuestion]) -> tuple[str, ...]: + names: Final = tuple(question.name for question in questions if question.name is not None) + if len(frozenset(names)) == len(questions): + return names + return tuple(str(index) for index in range(len(questions))) + + +def _ir_question(name: str, question: DecisionQuestion) -> DecisionsIRQuestion: + extra: Final = MappingProxyType(question.model_extra or {}) + match question: + case NoulQuestion(): + return DecisionsIRPredicateQuestion( + name=name, instructions=question.instructions, criteria=question.criteria, extra=extra + ) + case ChoiceQuestion(): + return DecisionsIRChoiceQuestion( + name=name, + instructions=question.instructions, + choices=tuple( + DecisionsIRChoiceOption(value=value, description=description) + for value, description in question.criteria.items() + ), + extra=extra, + ) + case ScoreQuestion(): + return DecisionsIRScoreQuestion( + name=name, + instructions=question.instructions, + levels=tuple( + DecisionsIRScoreLevel(label=criterion, description=None) for criterion in question.criteria + ), + extra=extra, + ) + case _: + assert_never(question) + + +def systemone_request_to_ir(request: DecisionsRequestBody) -> DecisionsIRRequest: + return DecisionsIRRequest( + input=DecisionsIRState(state=request.state), + questions=tuple(_ir_question(name, question) for name, question in request.questions.items()), + ) + + +def _has_image(message: OpenAIDecisionInputMessage) -> bool: + return not isinstance(message.content, str) and any( + isinstance(part, OpenAIDecisionInputImage) for part in message.content + ) + + +def _message_text(message: OpenAIDecisionInputMessage) -> str: + if isinstance(message.content, str): + return message.content + return "\n\n".join(part.text for part in message.content if isinstance(part, OpenAIDecisionInputText)) + + +def _systemone_state( + decision_input: DecisionsIRState | DecisionsIRMessages, +) -> DecisionsJSON | UnsupportedDecisionsRequest: + match decision_input: + case DecisionsIRState(): + return decision_input.state + case DecisionsIRMessages(): + if any(_has_image(message) for message in decision_input.messages): + return _TEXT_ONLY + return "\n\n".join(_message_text(message) for message in decision_input.messages) + case _: + assert_never(decision_input) + + +def _optional(key: str, value: object) -> Mapping[str, object]: + return {} if value is None else {key: value} + + +def _level_criterion(level: DecisionsIRScoreLevel) -> DecisionsJSON: + return level.label if level.description is None else f"{decisions_text(level.label)}: {level.description}" + + +def _systemone_question(question: DecisionsIRQuestion) -> Mapping[str, object]: + match question: + case DecisionsIRPredicateQuestion(): + return { + "type": "noul", + **_optional("instructions", question.instructions), + **_optional("criteria", question.criteria), + **question.extra, + } + case DecisionsIRChoiceQuestion(): + return { + "type": "choice", + **_optional("instructions", question.instructions), + "criteria": {systemone_choice_key(option.value): option.description for option in question.choices}, + **question.extra, + } + case DecisionsIRScoreQuestion(): + return { + "type": "score", + **_optional("instructions", question.instructions), + "criteria": [_level_criterion(level) for level in question.levels], + **question.extra, + } + case _: + assert_never(question) + + +def ir_to_systemone_request( + model: str, request: DecisionsIRRequest +) -> Mapping[str, object] | UnsupportedDecisionsRequest: + state: Final = _systemone_state(request.input) + if isinstance(state, UnsupportedDecisionsRequest): + return state + keyed_questions: Final = zip(systemone_keys(request.questions), request.questions, strict=True) + return { + "model": model, + "state": state, + "questions": {key: _systemone_question(question) for key, question in keyed_questions}, + } + + +def _ir_choice_answer(question: DecisionsIRChoiceQuestion, answer: ChoiceAnswer) -> DecisionsIRChoiceAnswer: + typed_values: Final = {systemone_choice_key(option.value): option.value for option in question.choices} + return DecisionsIRChoiceAnswer( + choice=typed_values.get(answer.choice, answer.choice), + confidence=answer.confidence, + probabilities=tuple( + DecisionsIRChoiceProbability(value=typed_values.get(key, key), probability=probability) + for key, probability in answer.probabilities.items() + ), + extra=MappingProxyType(answer.model_extra or {}), + ) + + +def _ir_score_answer(question: DecisionsIRScoreQuestion, answer: ScoreAnswer) -> DecisionsIRScoreAnswer: + return DecisionsIRScoreAnswer( + score=answer.score, + confidence=answer.confidence, + probabilities=tuple( + DecisionsIRScoreProbability( + value=index, + label=answer.legend.get(str(index), level.label), + probability=answer.probabilities.get(str(index), 0.0), + ) + for index, level in enumerate(question.levels) + ), + extra=MappingProxyType(answer.model_extra or {}), + ) + + +def _ir_answer(question: DecisionsIRQuestion, answer: DecisionAnswer | None) -> DecisionsIRAnswer: + match question, answer: + case DecisionsIRPredicateQuestion(), NoulAnswer(): + return DecisionsIRPredicateAnswer(probability=answer.noul, extra=MappingProxyType(answer.model_extra or {})) + case DecisionsIRChoiceQuestion(), ChoiceAnswer(): + return _ir_choice_answer(question, answer) + case DecisionsIRScoreQuestion(), ScoreAnswer(): + return _ir_score_answer(question, answer) + case _: + return DecisionsIRRefusal() + + +def systemone_response_to_ir(response: DecisionsResponse, request: DecisionsIRRequest) -> DecisionsIRResponse: + usage: Final = response.usage or DecisionsUsage() + keyed_questions: Final = zip(systemone_keys(request.questions), request.questions, strict=True) + return DecisionsIRResponse( + model=response.model, + answers=tuple(_ir_answer(question, response.answers.get(key)) for key, question in keyed_questions), + usage=DecisionsIRUsage( + input_tokens=usage.input_tokens, + output_tokens=usage.output_tokens, + cached_tokens=usage.cached_tokens, + cache_write_tokens=usage.cache_write_tokens, + extra=MappingProxyType(usage.model_extra or {}), + ), + extra=MappingProxyType(response.model_extra or {}), + ) + + +def parse_systemone_response(payload: object, request: DecisionsIRRequest) -> DecisionsIRResponse: + return systemone_response_to_ir(_SYSTEMONE_RESPONSE_ADAPTER.validate_python(payload), request) + + +def _systemone_answer(answer: DecisionsIRAnswer) -> DecisionAnswer | None: + match answer: + case DecisionsIRPredicateAnswer(): + return NoulAnswer.model_validate({**answer.extra, "type": "noul", "noul": answer.probability}) + case DecisionsIRChoiceAnswer(): + return ChoiceAnswer.model_validate( + { + **answer.extra, + "type": "choice", + "choice": systemone_choice_key(answer.choice), + "confidence": answer.confidence, + "probabilities": { + systemone_choice_key(item.value): item.probability for item in answer.probabilities + }, + } + ) + case DecisionsIRScoreAnswer(): + return ScoreAnswer.model_validate( + { + **answer.extra, + "type": "score", + "score": answer.score, + "confidence": answer.confidence, + "legend": {str(item.value): item.label for item in answer.probabilities}, + "probabilities": {str(item.value): item.probability for item in answer.probabilities}, + } + ) + case DecisionsIRRefusal(): + return None + case _: + assert_never(answer) + + +def ir_to_systemone_response(response: DecisionsIRResponse, request: DecisionsIRRequest) -> DecisionsResponse: + keyed_answers: Final = zip( + systemone_keys(request.questions), (_systemone_answer(answer) for answer in response.answers), strict=True + ) + return DecisionsResponse.model_validate( + { + **response.extra, + "model": response.model, + "answers": {key: answer for key, answer in keyed_answers if answer is not None}, + "usage": DecisionsUsage.model_validate( + { + **response.usage.extra, + "input_tokens": response.usage.input_tokens, + "output_tokens": response.usage.output_tokens, + "cached_tokens": response.usage.cached_tokens, + "cache_write_tokens": response.usage.cache_write_tokens, + } + ), + } + ) class BaseDecisionsConfig(ABC): @@ -55,48 +337,32 @@ class BaseDecisionsConfig(ABC): def transform_decisions_request( self, model: str, - request: DecisionsRequest, + request: DecisionsIRRequest, custom_llm_provider: str, - ) -> dict[str, object]: - return { - "model": self.request_model(model), - "state": request.state, - "questions": { - name: question.model_dump(mode="json", exclude_none=True) - for name, question in request.questions.items() - }, - } + ) -> Mapping[str, object] | UnsupportedDecisionsRequest: + return ir_to_systemone_request(self.request_model(model), request) def unwrap_response(self, payload: object) -> object: return payload + def parse_response(self, payload: object, request: DecisionsIRRequest) -> DecisionsIRResponse: + return parse_systemone_response(self.unwrap_response(payload), request) + def transform_decisions_response( self, model: str, custom_llm_provider: str, raw_response: httpx.Response, - request: DecisionsRequest, - ) -> DecisionsResponse: - payload: Final[object] = PAYLOAD_ADAPTER.validate_json(raw_response.content) + request: DecisionsIRRequest, + ) -> DecisionsIRResponse: + payload: Final[object] = _PAYLOAD_ADAPTER.validate_json(raw_response.content) try: - response: Final = _RESPONSE_ADAPTER.validate_python(self.unwrap_response(payload)) + return self.parse_response(payload, request) except ValidationError as error: raise BaseLLMException( status_code=500, message=f"Decisions provider '{custom_llm_provider}' returned an unexpected response: {error}", ) from error - self.set_hidden_params(response, model, custom_llm_provider) - return response - - @staticmethod - def set_hidden_params(response: DecisionsResponse, model: str, custom_llm_provider: str) -> None: - response.set_hidden_params( - { - "model": f"{custom_llm_provider}/{model}", - "custom_llm_provider": custom_llm_provider, - "provider_response_model": f"{custom_llm_provider}/{model}", - } - ) def get_error_class( self, diff --git a/litellm/llms/custom_httpx/llm_http_handler.py b/litellm/llms/custom_httpx/llm_http_handler.py index de271cb09ba..124f45e9bf8 100644 --- a/litellm/llms/custom_httpx/llm_http_handler.py +++ b/litellm/llms/custom_httpx/llm_http_handler.py @@ -123,7 +123,7 @@ from litellm.types.containers.main import ( ContainerObject, DeleteContainerResult, ) -from litellm.types.decisions import DecisionsRequest, DecisionsResponse +from litellm.types.decisions import DecisionsIRRequest, DecisionsIRResponse from litellm.types.files import StreamingMediaUploadConfig, TwoStepFileUploadConfig from litellm.types.integrations.custom_logger import ( NON_CODE_INTERPRETER_INTERCEPTION_INTERNAL_PREFIXES, @@ -1486,19 +1486,16 @@ class BaseLLMHTTPHandler: def _prepare_decisions_request( self, model: str, - custom_llm_provider: str, logging_obj: LiteLLMLoggingObj | None, provider_config: BaseDecisionsConfig, - request: DecisionsRequest, + body: Mapping[str, object], api_base: str, api_key: str | None, headers: Mapping[str, str], ) -> tuple[str, dict[str, str], dict[str, object]]: outbound_headers: Final = provider_config.validate_environment(headers=headers, model=model, api_key=api_key) url: Final = provider_config.get_complete_url(api_base=api_base, model=model) - data: Final = provider_config.transform_decisions_request( - model=model, request=request, custom_llm_provider=custom_llm_provider - ) + data: Final = dict(body) if logging_obj is not None: logging_obj.pre_call( input=data, @@ -1514,19 +1511,19 @@ class BaseLLMHTTPHandler: custom_llm_provider: str, logging_obj: LiteLLMLoggingObj | None, provider_config: BaseDecisionsConfig, - request: DecisionsRequest, + request: DecisionsIRRequest, + body: Mapping[str, object], api_base: str, api_key: str | None, headers: Mapping[str, str], timeout: float | httpx.Timeout | None, client: HTTPHandler | None = None, - ) -> DecisionsResponse: + ) -> DecisionsIRResponse: url, outbound_headers, data = self._prepare_decisions_request( model=model, - custom_llm_provider=custom_llm_provider, logging_obj=logging_obj, provider_config=provider_config, - request=request, + body=body, api_base=api_base, api_key=api_key, headers=headers, @@ -1548,19 +1545,19 @@ class BaseLLMHTTPHandler: custom_llm_provider: str, logging_obj: LiteLLMLoggingObj | None, provider_config: BaseDecisionsConfig, - request: DecisionsRequest, + request: DecisionsIRRequest, + body: Mapping[str, object], api_base: str, api_key: str | None, headers: Mapping[str, str], timeout: float | httpx.Timeout | None, client: AsyncHTTPHandler | None = None, - ) -> DecisionsResponse: + ) -> DecisionsIRResponse: url, outbound_headers, data = self._prepare_decisions_request( model=model, - custom_llm_provider=custom_llm_provider, logging_obj=logging_obj, provider_config=provider_config, - request=request, + body=body, api_base=api_base, api_key=api_key, headers=headers, diff --git a/litellm/llms/openai/decisions/transformation.py b/litellm/llms/openai/decisions/transformation.py new file mode 100644 index 00000000000..0afbe1a434e --- /dev/null +++ b/litellm/llms/openai/decisions/transformation.py @@ -0,0 +1,320 @@ +from collections.abc import Mapping, Sequence +from typing import Final + +from pydantic import TypeAdapter +from typing_extensions import assert_never + +import litellm +from litellm.llms.base_llm.decisions.transformation import BaseDecisionsConfig, decisions_text +from litellm.secret_managers.main import get_secret_str +from litellm.types.decisions import ( + DecisionsIRAnswer, + DecisionsIRChoiceAnswer, + DecisionsIRChoiceOption, + DecisionsIRChoiceProbability, + DecisionsIRChoiceQuestion, + DecisionsIRMessages, + DecisionsIRPredicateAnswer, + DecisionsIRPredicateQuestion, + DecisionsIRQuestion, + DecisionsIRRefusal, + DecisionsIRRequest, + DecisionsIRResponse, + DecisionsIRScoreAnswer, + DecisionsIRScoreLevel, + DecisionsIRScoreProbability, + DecisionsIRScoreQuestion, + DecisionsIRState, + DecisionsIRUsage, + DecisionsJSON, + OpenAIChoiceAnswer, + OpenAIChoiceProbability, + OpenAIChoiceQuestion, + OpenAIDecisionAnswer, + OpenAIDecisionInput, + OpenAIDecisionInputTokensDetails, + OpenAIDecisionOutputTokensDetails, + OpenAIDecisionQuestion, + OpenAIDecisionRequestBody, + OpenAIDecisionResponse, + OpenAIDecisionUsage, + OpenAIPredicateAnswer, + OpenAIPredicateQuestion, + OpenAIRefusalAnswer, + OpenAIScoreAnswer, + OpenAIScoreProbability, + OpenAIScoreQuestion, + UnsupportedDecisionsRequest, + systemone_choice_key, +) + +_OPENAI_RESPONSE_ADAPTER: Final[TypeAdapter[OpenAIDecisionResponse]] = TypeAdapter(OpenAIDecisionResponse) +_SINGLE_OPTION: Final = UnsupportedDecisionsRequest( + reason="OpenAI needs at least 2 choices or levels on every choice or score question" +) + + +def _ir_input(decision_input: OpenAIDecisionInput) -> DecisionsIRState | DecisionsIRMessages: + if isinstance(decision_input, str): + return DecisionsIRState(state=decision_input) + return DecisionsIRMessages(messages=tuple(decision_input)) + + +def _ir_question(question: OpenAIDecisionQuestion) -> DecisionsIRQuestion: + match question: + case OpenAIPredicateQuestion(): + return DecisionsIRPredicateQuestion(name=question.name, instructions=question.instructions) + case OpenAIChoiceQuestion(): + return DecisionsIRChoiceQuestion( + name=question.name, + instructions=question.instructions, + choices=tuple( + DecisionsIRChoiceOption(value=option.value, description=option.description) + for option in question.choices + ), + ) + case OpenAIScoreQuestion(): + return DecisionsIRScoreQuestion( + name=question.name, + instructions=question.instructions, + levels=tuple( + DecisionsIRScoreLevel(label=level.label, description=level.description) for level in question.levels + ), + ) + case _: + assert_never(question) + + +def openai_request_to_ir(request: OpenAIDecisionRequestBody) -> DecisionsIRRequest: + return DecisionsIRRequest( + input=_ir_input(request.input), + questions=tuple(_ir_question(question) for question in request.questions), + safety_identifier=request.safety_identifier, + ) + + +def _text_field(key: str, value: DecisionsJSON | None) -> Mapping[str, str]: + return {} if value is None else {key: decisions_text(value)} + + +def _instructions(value: DecisionsJSON | None, default: str) -> str: + return default if value is None else decisions_text(value) + + +def _predicate_instructions(question: DecisionsIRPredicateQuestion) -> str: + instructions: Final = () if question.instructions is None else (decisions_text(question.instructions),) + criteria: Final = tuple( + f"Answer {answer} when: {decisions_text(rule)}" + for answer, rule in (question.criteria or {}).items() + if rule is not None + ) + return "\n\n".join((*instructions, *criteria)) + + +def _openai_question(question: DecisionsIRQuestion) -> Mapping[str, object]: + match question: + case DecisionsIRPredicateQuestion(): + return { + "type": "predicate", + **_text_field("name", question.name), + "instructions": _predicate_instructions(question) or "Is this true of the input?", + } + case DecisionsIRChoiceQuestion(): + return { + "type": "choice", + **_text_field("name", question.name), + "instructions": _instructions(question.instructions, "Which choice best fits the input?"), + "choices": [ + {"value": option.value, **_text_field("description", option.description)} + for option in question.choices + ], + } + case DecisionsIRScoreQuestion(): + return { + "type": "score", + **_text_field("name", question.name), + "instructions": _instructions(question.instructions, "Which level best fits the input?"), + "levels": [ + {"label": decisions_text(level.label), **_text_field("description", level.description)} + for level in question.levels + ], + } + case _: + assert_never(question) + + +def _openai_input(decision_input: DecisionsIRState | DecisionsIRMessages) -> str | Sequence[Mapping[str, object]]: + match decision_input: + case DecisionsIRState(): + return decisions_text(decision_input.state) + case DecisionsIRMessages(): + return [message.model_dump(mode="json", exclude_none=True) for message in decision_input.messages] + case _: + assert_never(decision_input) + + +def _has_one_option(question: DecisionsIRQuestion) -> bool: + match question: + case DecisionsIRChoiceQuestion(): + return len(question.choices) < 2 + case DecisionsIRScoreQuestion(): + return len(question.levels) < 2 + case _: + return False + + +def ir_to_openai_request(model: str, request: DecisionsIRRequest) -> Mapping[str, object] | UnsupportedDecisionsRequest: + if any(_has_one_option(question) for question in request.questions): + return _SINGLE_OPTION + return { + "model": model, + "input": _openai_input(request.input), + "questions": [_openai_question(question) for question in request.questions], + **_text_field("safety_identifier", request.safety_identifier), + } + + +def _level_label(question: DecisionsIRScoreQuestion, probability: OpenAIScoreProbability) -> DecisionsJSON: + if 0 <= probability.value < len(question.levels): + return question.levels[probability.value].label + return probability.label + + +def _ir_answer(question: DecisionsIRQuestion, answer: OpenAIDecisionAnswer | None) -> DecisionsIRAnswer: + match question, answer: + case DecisionsIRPredicateQuestion(), OpenAIPredicateAnswer(): + return DecisionsIRPredicateAnswer(probability=answer.probability) + case DecisionsIRChoiceQuestion(), OpenAIChoiceAnswer(): + return DecisionsIRChoiceAnswer( + choice=answer.choice, + confidence=answer.confidence, + probabilities=tuple( + DecisionsIRChoiceProbability(value=item.value, probability=item.probability) + for item in answer.probabilities + ), + ) + case DecisionsIRScoreQuestion(), OpenAIScoreAnswer(): + return DecisionsIRScoreAnswer( + score=answer.score, + confidence=answer.confidence, + probabilities=tuple( + DecisionsIRScoreProbability( + value=item.value, label=_level_label(question, item), probability=item.probability + ) + for item in answer.probabilities + ), + ) + case _: + return DecisionsIRRefusal() + + +def openai_response_to_ir(response: OpenAIDecisionResponse, request: DecisionsIRRequest) -> DecisionsIRResponse: + answers: Final = response.answers + return DecisionsIRResponse( + model=response.model, + answers=tuple( + _ir_answer(question, answers[index] if index < len(answers) else None) + for index, question in enumerate(request.questions) + ), + usage=DecisionsIRUsage( + input_tokens=response.usage.input_tokens, + output_tokens=response.usage.output_tokens, + cached_tokens=response.usage.input_tokens_details.cached_tokens, + cache_write_tokens=response.usage.input_tokens_details.cache_write_tokens, + reasoning_tokens=response.usage.output_tokens_details.reasoning_tokens, + ), + ) + + +def _openai_choice_answer(question: DecisionsIRChoiceQuestion, answer: DecisionsIRChoiceAnswer) -> OpenAIChoiceAnswer: + probabilities: Final = {systemone_choice_key(item.value): item.probability for item in answer.probabilities} + return OpenAIChoiceAnswer( + name=question.name, + choice=answer.choice, + probabilities=tuple( + OpenAIChoiceProbability( + value=option.value, probability=probabilities.get(systemone_choice_key(option.value), 0.0) + ) + for option in question.choices + ), + confidence=answer.confidence, + ) + + +def _openai_score_answer(question: DecisionsIRScoreQuestion, answer: DecisionsIRScoreAnswer) -> OpenAIScoreAnswer: + probabilities: Final = {item.value: item.probability for item in answer.probabilities} + return OpenAIScoreAnswer( + name=question.name, + score=answer.score, + probabilities=tuple( + OpenAIScoreProbability( + value=index, label=decisions_text(level.label), probability=probabilities.get(index, 0.0) + ) + for index, level in enumerate(question.levels) + ), + confidence=answer.confidence, + ) + + +def _openai_answer(question: DecisionsIRQuestion, answer: DecisionsIRAnswer) -> OpenAIDecisionAnswer: + match question, answer: + case DecisionsIRPredicateQuestion(), DecisionsIRPredicateAnswer(): + return OpenAIPredicateAnswer(name=question.name, probability=answer.probability) + case DecisionsIRChoiceQuestion(), DecisionsIRChoiceAnswer(): + return _openai_choice_answer(question, answer) + case DecisionsIRScoreQuestion(), DecisionsIRScoreAnswer(): + return _openai_score_answer(question, answer) + case _: + return OpenAIRefusalAnswer(name=question.name) + + +def ir_to_openai_response( + response: DecisionsIRResponse, request: DecisionsIRRequest, requested_model: str +) -> OpenAIDecisionResponse: + return OpenAIDecisionResponse( + model=response.model or requested_model, + answers=tuple( + _openai_answer(question, answer) + for question, answer in zip(request.questions, response.answers, strict=True) + ), + usage=OpenAIDecisionUsage( + input_tokens=response.usage.input_tokens, + input_tokens_details=OpenAIDecisionInputTokensDetails( + cached_tokens=response.usage.cached_tokens, + cache_write_tokens=response.usage.cache_write_tokens, + ), + output_tokens=response.usage.output_tokens, + output_tokens_details=OpenAIDecisionOutputTokensDetails(reasoning_tokens=response.usage.reasoning_tokens), + total_tokens=response.usage.input_tokens + response.usage.output_tokens, + ), + ) + + +class OpenAIDecisionsConfig(BaseDecisionsConfig): + path = "/v1/decisions" + + def get_default_api_base(self) -> str | None: + return "https://api.openai.com" + + def resolve_api_base(self, api_base: str | None) -> str | None: + return ( + api_base + or litellm.api_base + or get_secret_str("OPENAI_BASE_URL") + or get_secret_str("OPENAI_API_BASE") + or self.get_default_api_base() + ) + + def resolve_api_key(self, api_key: str | None) -> str | None: + return api_key or litellm.api_key or litellm.openai_key or get_secret_str("OPENAI_API_KEY") + + def transform_decisions_request( + self, + model: str, + request: DecisionsIRRequest, + custom_llm_provider: str, + ) -> Mapping[str, object] | UnsupportedDecisionsRequest: + return ir_to_openai_request(self.request_model(model), request) + + def parse_response(self, payload: object, request: DecisionsIRRequest) -> DecisionsIRResponse: + return openai_response_to_ir(_OPENAI_RESPONSE_ADAPTER.validate_python(payload), request) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 5453985dadb..f34606ce127 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -34527,7 +34527,8 @@ "supported_endpoints": [ "/v1/chat/completions", "/v1/batch", - "/v1/responses" + "/v1/responses", + "/v1/decisions" ], "supported_modalities": [ "text", diff --git a/litellm/proxy/decisions_endpoints/endpoints.py b/litellm/proxy/decisions_endpoints/endpoints.py index 265dbbb2957..f2891c184bf 100644 --- a/litellm/proxy/decisions_endpoints/endpoints.py +++ b/litellm/proxy/decisions_endpoints/endpoints.py @@ -5,15 +5,10 @@ from fastapi import APIRouter, Depends, Request, Response from fastapi.responses import ORJSONResponse # pyright: ignore[reportDeprecated] # required endpoint contract from pydantic import TypeAdapter, ValidationError -from litellm.decisions.openai_transformation import to_openai_response, to_systemone_request from litellm.exceptions import BadRequestError from litellm.proxy.auth.user_api_key_auth import UserAPIKeyAuth, user_api_key_auth -from litellm.proxy.common_request_processing import ( - ProxyBaseLLMRequestProcessing, - attach_guardrail_information, - include_guardrail_response_requested, -) -from litellm.types.decisions import DecisionsRequestBody, DecisionsResponse, OpenAIDecisionRequestBody +from litellm.proxy.common_request_processing import ProxyBaseLLMRequestProcessing +from litellm.types.decisions import DecisionsRequestBody, OpenAIDecisionRequestBody router: Final = APIRouter() _REQUEST_DATA_ADAPTER: Final[TypeAdapter[dict[str, object]]] = TypeAdapter(dict[str, object]) @@ -21,7 +16,6 @@ _DECISIONS_REQUEST_BODY_ADAPTER: Final[TypeAdapter[DecisionsRequestBody]] = Type _OPENAI_DECISION_REQUEST_BODY_ADAPTER: Final[TypeAdapter[OpenAIDecisionRequestBody]] = TypeAdapter( OpenAIDecisionRequestBody ) -_DECISIONS_RESPONSE_ADAPTER: Final[TypeAdapter[DecisionsResponse]] = TypeAdapter(DecisionsResponse) _GENERAL_SETTINGS_ADAPTER: Final[TypeAdapter[dict[str, object]]] = TypeAdapter(dict[str, object]) _OPTIONAL_STRING_ADAPTER: Final[TypeAdapter[str | None]] = TypeAdapter(str | None) _OPTIONAL_FLOAT_ADAPTER: Final[TypeAdapter[float | None]] = TypeAdapter(float | None) @@ -52,12 +46,11 @@ async def _request_data(request: Request, user_api_key_dict: UserAPIKeyAuth) -> raise await _invalid_request(raw_data={}, error=error, user_api_key_dict=user_api_key_dict) -async def _process_systemone( +async def _process_decisions( request: Request, fastapi_response: Response, user_api_key_dict: UserAPIKeyAuth, - raw_data: Mapping[str, object], - openai_body: OpenAIDecisionRequestBody | None, + body_adapter: TypeAdapter[DecisionsRequestBody] | TypeAdapter[OpenAIDecisionRequestBody], ) -> object: from litellm.proxy.proxy_server import ( general_settings as proxy_general_settings, @@ -80,15 +73,18 @@ async def _process_systemone( user_temperature as proxy_user_temperature, ) - data: Final = dict(raw_data if openai_body is None else to_systemone_request(raw_data, openai_body)) + data: Final = await _request_data(request, user_api_key_dict) + try: + body_adapter.validate_python(data) + except ValidationError as error: + raise await _invalid_request(raw_data=data, error=error, user_api_key_dict=user_api_key_dict) general_settings: Final = _GENERAL_SETTINGS_ADAPTER.validate_python(proxy_general_settings) user_api_base: Final = _OPTIONAL_STRING_ADAPTER.validate_python(proxy_user_api_base) user_model: Final = _OPTIONAL_STRING_ADAPTER.validate_python(proxy_user_model) user_temperature: Final = _OPTIONAL_FLOAT_ADAPTER.validate_python(proxy_user_temperature) processor: Final = ProxyBaseLLMRequestProcessing(data=data) try: - _DECISIONS_REQUEST_BODY_ADAPTER.validate_python(data) - result: Final[object] = await processor.base_process_llm_request( + return await processor.base_process_llm_request( request=request, fastapi_response=fastapi_response, user_api_key_dict=user_api_key_dict, @@ -106,21 +102,6 @@ async def _process_systemone( user_api_base=user_api_base, version=version, ) - if openai_body is None or isinstance(result, Response): - return result - openai_response: Final = to_openai_response( - _DECISIONS_RESPONSE_ADAPTER.validate_python(result), - openai_body.questions, - str(data.get("model", "")), - ) - request_data: Final = _REQUEST_DATA_ADAPTER.validate_python( - processor.data # pyright: ignore[reportUnknownMemberType] # ProxyBaseLLMRequestProcessing.data is a bare dict - ) - if include_guardrail_response_requested(request_data): - return attach_guardrail_information(response=openai_response, request_data=request_data) - return openai_response - except ValidationError as error: - raise await _invalid_request(raw_data=data, error=error, user_api_key_dict=user_api_key_dict) except Exception as error: raise await processor.handle_llm_api_exception( e=error, @@ -147,12 +128,11 @@ async def systemone( fastapi_response: Response, user_api_key_dict: Annotated[UserAPIKeyAuth, Depends(user_api_key_auth)], ): - return await _process_systemone( + return await _process_decisions( request=request, fastapi_response=fastapi_response, user_api_key_dict=user_api_key_dict, - raw_data=await _request_data(request, user_api_key_dict), - openai_body=None, + body_adapter=_DECISIONS_REQUEST_BODY_ADAPTER, ) @@ -173,15 +153,9 @@ async def decisions( fastapi_response: Response, user_api_key_dict: Annotated[UserAPIKeyAuth, Depends(user_api_key_auth)], ): - raw_data: Final = await _request_data(request, user_api_key_dict) - try: - openai_body: Final = _OPENAI_DECISION_REQUEST_BODY_ADAPTER.validate_python(raw_data) - except ValidationError as error: - raise await _invalid_request(raw_data=raw_data, error=error, user_api_key_dict=user_api_key_dict) - return await _process_systemone( + return await _process_decisions( request=request, fastapi_response=fastapi_response, user_api_key_dict=user_api_key_dict, - raw_data=raw_data, - openai_body=openai_body, + body_adapter=_OPENAI_DECISION_REQUEST_BODY_ADAPTER, ) diff --git a/litellm/types/decisions.py b/litellm/types/decisions.py index b543b3b32df..b24690dbd9d 100644 --- a/litellm/types/decisions.py +++ b/litellm/types/decisions.py @@ -1,4 +1,6 @@ from collections.abc import Mapping, Sequence +from dataclasses import dataclass, field +from types import MappingProxyType from typing import Annotated, Final, Literal, TypeAlias from pydantic import ConfigDict, Field, PrivateAttr, model_validator, with_config @@ -110,17 +112,13 @@ DecisionAnswer: TypeAlias = Annotated[ class DecisionsUsage(LiteLLMPydanticObjectBase): input_tokens: int = 0 output_tokens: int = 0 + cached_tokens: Annotated[int, Field(exclude=True)] = 0 + cache_write_tokens: Annotated[int, Field(exclude=True)] = 0 model_config = ConfigDict(extra="allow", frozen=True) -class DecisionsResponse(LiteLLMPydanticObjectBase): - model: str | None = None - answers: Mapping[str, DecisionAnswer] - usage: DecisionsUsage | None = None - - model_config = ConfigDict(extra="allow", frozen=True) - +class _HiddenParamsResponse(LiteLLMPydanticObjectBase): _hidden_params: dict[str, object] = PrivateAttr(default_factory=dict) @property @@ -131,6 +129,14 @@ class DecisionsResponse(LiteLLMPydanticObjectBase): self._hidden_params.update(params) +class DecisionsResponse(_HiddenParamsResponse): + model: str | None = None + answers: Mapping[str, DecisionAnswer] + usage: DecisionsUsage | None = None + + model_config = ConfigDict(extra="allow", frozen=True) + + class OpenAIDecisionInputText(LiteLLMPydanticObjectBase): type: Literal["input_text"] text: str @@ -138,14 +144,31 @@ class OpenAIDecisionInputText(LiteLLMPydanticObjectBase): model_config = ConfigDict(extra="forbid", frozen=True) +class OpenAIDecisionInputImage(LiteLLMPydanticObjectBase): + type: Literal["input_image"] + image_url: str + detail: str | None = None + + model_config = ConfigDict(extra="forbid", frozen=True) + + +OpenAIDecisionContentPart: TypeAlias = Annotated[ + OpenAIDecisionInputText | OpenAIDecisionInputImage, + Field(discriminator="type"), +] + + class OpenAIDecisionInputMessage(LiteLLMPydanticObjectBase): role: Literal["user"] = "user" type: Literal["message"] = "message" - content: str | Sequence[OpenAIDecisionInputText] + content: str | Sequence[OpenAIDecisionContentPart] model_config = ConfigDict(extra="forbid", frozen=True) +OpenAIDecisionInput: TypeAlias = str | Sequence[OpenAIDecisionInputMessage] + + class OpenAIPredicateQuestion(LiteLLMPydanticObjectBase): type: Literal["predicate"] name: str | None = None @@ -206,7 +229,7 @@ OpenAIDecisionQuestion: TypeAlias = Annotated[ class OpenAIDecisionRequestBody(LiteLLMPydanticObjectBase): - input: str | Sequence[OpenAIDecisionInputMessage] + input: OpenAIDecisionInput questions: Annotated[Sequence[OpenAIDecisionQuestion], Field(min_length=1, max_length=MAX_DECISION_QUESTIONS)] safety_identifier: str | None = None @@ -263,7 +286,10 @@ class OpenAIRefusalAnswer(LiteLLMPydanticObjectBase): model_config = ConfigDict(frozen=True) -OpenAIDecisionAnswer: TypeAlias = OpenAIPredicateAnswer | OpenAIChoiceAnswer | OpenAIScoreAnswer | OpenAIRefusalAnswer +OpenAIDecisionAnswer: TypeAlias = Annotated[ + OpenAIPredicateAnswer | OpenAIChoiceAnswer | OpenAIScoreAnswer | OpenAIRefusalAnswer, + Field(discriminator="type"), +] class OpenAIDecisionInputTokensDetails(LiteLLMPydanticObjectBase): @@ -289,9 +315,136 @@ class OpenAIDecisionUsage(LiteLLMPydanticObjectBase): model_config = ConfigDict(frozen=True) -class OpenAIDecisionResponse(LiteLLMPydanticObjectBase): +class OpenAIDecisionResponse(_HiddenParamsResponse): model: str answers: tuple[OpenAIDecisionAnswer, ...] usage: OpenAIDecisionUsage model_config = ConfigDict(extra="allow", frozen=True) + + +_NO_EXTRA: Final[Mapping[str, object]] = MappingProxyType({}) + + +@dataclass(frozen=True, slots=True) +class DecisionsIRState: + state: DecisionsJSON + + +@dataclass(frozen=True, slots=True) +class DecisionsIRMessages: + messages: tuple[OpenAIDecisionInputMessage, ...] + + +@dataclass(frozen=True, slots=True) +class DecisionsIRPredicateQuestion: + name: str | None + instructions: DecisionsJSON | None + criteria: NoulCriteria | None = None + extra: Mapping[str, object] = field(default_factory=lambda: _NO_EXTRA) + + +@dataclass(frozen=True, slots=True) +class DecisionsIRChoiceOption: + value: str | bool + description: DecisionsJSON | None + + +@dataclass(frozen=True, slots=True) +class DecisionsIRChoiceQuestion: + name: str | None + instructions: DecisionsJSON | None + choices: tuple[DecisionsIRChoiceOption, ...] + extra: Mapping[str, object] = field(default_factory=lambda: _NO_EXTRA) + + +@dataclass(frozen=True, slots=True) +class DecisionsIRScoreLevel: + label: DecisionsJSON + description: str | None + + +@dataclass(frozen=True, slots=True) +class DecisionsIRScoreQuestion: + name: str | None + instructions: DecisionsJSON | None + levels: tuple[DecisionsIRScoreLevel, ...] + extra: Mapping[str, object] = field(default_factory=lambda: _NO_EXTRA) + + +DecisionsIRQuestion: TypeAlias = DecisionsIRPredicateQuestion | DecisionsIRChoiceQuestion | DecisionsIRScoreQuestion + + +@dataclass(frozen=True, slots=True) +class DecisionsIRRequest: + input: DecisionsIRState | DecisionsIRMessages + questions: tuple[DecisionsIRQuestion, ...] + safety_identifier: str | None = None + + +@dataclass(frozen=True, slots=True) +class DecisionsIRPredicateAnswer: + probability: float + extra: Mapping[str, object] = field(default_factory=lambda: _NO_EXTRA) + + +@dataclass(frozen=True, slots=True) +class DecisionsIRChoiceProbability: + value: str | bool + probability: float + + +@dataclass(frozen=True, slots=True) +class DecisionsIRChoiceAnswer: + choice: str | bool + confidence: float + probabilities: tuple[DecisionsIRChoiceProbability, ...] + extra: Mapping[str, object] = field(default_factory=lambda: _NO_EXTRA) + + +@dataclass(frozen=True, slots=True) +class DecisionsIRScoreProbability: + value: int + label: DecisionsJSON + probability: float + + +@dataclass(frozen=True, slots=True) +class DecisionsIRScoreAnswer: + score: float + confidence: float + probabilities: tuple[DecisionsIRScoreProbability, ...] + extra: Mapping[str, object] = field(default_factory=lambda: _NO_EXTRA) + + +@dataclass(frozen=True, slots=True) +class DecisionsIRRefusal: + pass + + +DecisionsIRAnswer: TypeAlias = ( + DecisionsIRPredicateAnswer | DecisionsIRChoiceAnswer | DecisionsIRScoreAnswer | DecisionsIRRefusal +) + + +@dataclass(frozen=True, slots=True) +class DecisionsIRUsage: + input_tokens: int = 0 + output_tokens: int = 0 + cached_tokens: int = 0 + cache_write_tokens: int = 0 + reasoning_tokens: int = 0 + extra: Mapping[str, object] = field(default_factory=lambda: _NO_EXTRA) + + +@dataclass(frozen=True, slots=True) +class DecisionsIRResponse: + model: str | None + answers: tuple[DecisionsIRAnswer, ...] + usage: DecisionsIRUsage + extra: Mapping[str, object] = field(default_factory=lambda: _NO_EXTRA) + + +@dataclass(frozen=True, slots=True) +class UnsupportedDecisionsRequest: + reason: str diff --git a/litellm/utils.py b/litellm/utils.py index 7207c6bacaa..89544cf31e2 100644 --- a/litellm/utils.py +++ b/litellm/utils.py @@ -1219,8 +1219,8 @@ def function_setup( else search_query ) elif call_type in (CallTypes.decisions.value, CallTypes.adecisions.value): - decisions_state: Final = args[1] if len(args) > 1 else kwargs.get("state", "") - messages = decisions_state if isinstance(decisions_state, str) else json.dumps(decisions_state) + decisions_state: Final = args[1] if len(args) > 1 else kwargs.get("state") or kwargs.get("input") or "" + messages = decisions_state if isinstance(decisions_state, str) else json.dumps(decisions_state, default=str) elif call_type in (CallTypes.image_edit.value, CallTypes.aimage_edit.value): messages = args[1] if len(args) > 1 else kwargs.get("prompt") elif call_type in (CallTypes.ocr.value, CallTypes.aocr.value): @@ -8974,6 +8974,8 @@ class ProviderConfigManager: return litellm.CloudflareDecisionsConfig() if provider == LlmProviders.STRANDS_DECIDER: return litellm.StrandsDeciderDecisionsConfig() + if provider == LlmProviders.OPENAI: + return litellm.OpenAIDecisionsConfig() return None @staticmethod diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 5453985dadb..f34606ce127 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -34527,7 +34527,8 @@ "supported_endpoints": [ "/v1/chat/completions", "/v1/batch", - "/v1/responses" + "/v1/responses", + "/v1/decisions" ], "supported_modalities": [ "text", diff --git a/tests/unit/decisions/test_main.py b/tests/unit/decisions/test_main.py index e5964f972b9..820451ed943 100644 --- a/tests/unit/decisions/test_main.py +++ b/tests/unit/decisions/test_main.py @@ -18,6 +18,7 @@ from litellm.types.decisions import ( DecisionsResponse, DecisionsUsage, NoulAnswer, + OpenAIDecisionResponse, ScoreAnswer, ) @@ -30,6 +31,8 @@ _QUESTIONS: Final[Mapping[str, object]] = MappingProxyType( ) _INPUT_TOKENS: Final[int] = 367 _OUTPUT_TOKENS: Final[int] = 3 +_CACHED_TOKENS: Final[int] = 256 +_CACHE_WRITE_TOKENS: Final[int] = 64 _RESPONSE: Final[Mapping[str, object]] = { "model": "jev-1.13", "answers": { @@ -80,6 +83,21 @@ _PROVIDERS: Final[tuple[tuple[str, str, str, str], ...]] = ( ), ) +_OPENAI_RESPONSE: Final[Mapping[str, object]] = { + "model": "gpt-6-luna", + "answers": [ + {"type": "predicate", "name": "is_defect", "probability": 0.9}, + {"type": "refusal", "name": "sentiment"}, + ], + "usage": { + "input_tokens": _INPUT_TOKENS, + "input_tokens_details": {"cached_tokens": _CACHED_TOKENS, "cache_write_tokens": _CACHE_WRITE_TOKENS}, + "output_tokens": _OUTPUT_TOKENS, + "output_tokens_details": {"reasoning_tokens": 0}, + "total_tokens": _INPUT_TOKENS + _OUTPUT_TOKENS, + }, +} + class _RecordingLogger(CustomLogger): def __init__(self) -> None: @@ -265,6 +283,46 @@ def test_decisions_cost_uses_litellm_token_pricing() -> None: assert cost == pytest.approx(expected_cost) +@pytest.mark.parametrize( + "response", + ( + DecisionsResponse( + model="jev-latest", + answers={}, + usage=DecisionsUsage( + input_tokens=_INPUT_TOKENS, + output_tokens=_OUTPUT_TOKENS, + cached_tokens=_CACHED_TOKENS, + cache_write_tokens=_CACHE_WRITE_TOKENS, + ), + ), + OpenAIDecisionResponse.model_validate(_OPENAI_RESPONSE), + ), + ids=("systemone", "openai"), +) +def test_custom_token_pricing_bills_cached_decisions_input_tokens_once( + response: DecisionsResponse | OpenAIDecisionResponse, +) -> None: + cost: Final = litellm.completion_cost( + completion_response=response, + model="gpt-6-luna", + custom_llm_provider="openai", + custom_cost_per_token={ + "input_cost_per_token": 1.0, + "output_cost_per_token": 2.0, + "cache_read_input_token_cost": 0.1, + "cache_creation_input_token_cost": 1.25, + }, + ) + + assert cost == pytest.approx( + (_INPUT_TOKENS - _CACHED_TOKENS - _CACHE_WRITE_TOKENS) * 1.0 + + _CACHED_TOKENS * 0.1 + + _CACHE_WRITE_TOKENS * 1.25 + + _OUTPUT_TOKENS * 2.0 + ) + + def test_decisions_response_hidden_params_getter_preserves_mutable_identity() -> None: response: Final = DecisionsResponse(model="decider", answers={}, usage=None) @@ -487,7 +545,7 @@ async def test_cloudflare_clef_resolves_model_and_response_envelope( "state": "review", "questions": {"is_defect": {"type": "noul", "instructions": "Is this a defect?"}}, } - assert response.answers == DecisionsResponse.model_validate(_RESPONSE).answers + assert response.answers == {"is_defect": NoulAnswer(type="noul", noul=0.9)} assert response._hidden_params["model"] == "cloudflare/@cf/cloudflare/clef" @@ -673,3 +731,194 @@ async def test_strands_decider_provider_resolution_and_router_dispatch( assert provider_resolution[:2] == ("strands-decider-2B-hobson-v19", "strands_decider") assert route.called assert response.model == _STRANDS_RESPONSE["model"] + + +@pytest.mark.asyncio +@pytest.mark.parametrize( + ("api_base", "url"), + ( + (None, "https://api.openai.com/v1/decisions"), + ("https://gateway.example/v1", "https://gateway.example/v1/decisions"), + ), +) +async def test_openai_decisions_translate_systemone_to_the_openai_wire_contract_and_back( + api_base: str | None, + url: str, + monkeypatch: pytest.MonkeyPatch, + respx_mock: respx.MockRouter, +) -> None: + monkeypatch.setattr(litellm, "api_base", None) + monkeypatch.delenv("OPENAI_BASE_URL", raising=False) + monkeypatch.delenv("OPENAI_API_BASE", raising=False) + route: Final = respx_mock.post(url).respond(json=_OPENAI_RESPONSE) + + response: Final = await litellm.adecisions( + model="openai/gpt-6-luna", + state="The package arrived broken.", + questions={ + "is_defect": {"type": "noul", "instructions": "Is this a defect?"}, + "sentiment": { + "type": "choice", + "instructions": "How does the customer feel?", + "criteria": {"positive": None, "negative": "unhappy"}, + }, + }, + api_key="caller-key", + api_base=api_base, + ) + + assert route.called + request: Final = respx_mock.calls[0].request + assert request.headers["authorization"] == "Bearer caller-key" + assert json.loads(request.content) == { + "model": "gpt-6-luna", + "input": "The package arrived broken.", + "questions": [ + {"type": "predicate", "name": "is_defect", "instructions": "Is this a defect?"}, + { + "type": "choice", + "name": "sentiment", + "instructions": "How does the customer feel?", + "choices": [{"value": "positive"}, {"value": "negative", "description": "unhappy"}], + }, + ], + } + assert response.answers == {"is_defect": NoulAnswer(type="noul", noul=0.9)} + assert response.hidden_params["custom_llm_provider"] == "openai" + luna_cost: Final = litellm.model_cost["gpt-6-luna"] + expected_cost: Final = ( + (_INPUT_TOKENS - _CACHED_TOKENS - _CACHE_WRITE_TOKENS) * float(luna_cost["input_cost_per_token"]) + + _CACHED_TOKENS * float(luna_cost["cache_read_input_token_cost"]) + + _CACHE_WRITE_TOKENS * float(luna_cost["cache_creation_input_token_cost"]) + + _OUTPUT_TOKENS * float(luna_cost["output_cost_per_token"]) + ) + assert expected_cost > 0 + assert litellm.completion_cost(completion_response=response) == pytest.approx(expected_cost) + + +@pytest.mark.parametrize( + ("settings", "env", "url", "authorization"), + ( + ({"openai_key": "sdk-key"}, {}, "https://api.openai.com/v1/decisions", "Bearer sdk-key"), + ( + {"api_key": "global-key", "openai_key": "sdk-key"}, + {"OPENAI_API_KEY": "env-key"}, + "https://api.openai.com/v1/decisions", + "Bearer global-key", + ), + ( + {}, + {"OPENAI_API_KEY": "env-key", "OPENAI_API_BASE": "https://legacy.example/v1"}, + "https://legacy.example/v1/decisions", + "Bearer env-key", + ), + ( + {"api_base": "https://sdk.example"}, + {"OPENAI_API_KEY": "env-key", "OPENAI_BASE_URL": "https://env.example"}, + "https://sdk.example/v1/decisions", + "Bearer env-key", + ), + ), + ids=("openai_key", "api_key_before_env", "openai_api_base_env", "api_base_before_env"), +) +def test_openai_decisions_use_the_same_settings_as_other_openai_calls( + settings: Mapping[str, str], + env: Mapping[str, str], + url: str, + authorization: str, + monkeypatch: pytest.MonkeyPatch, + respx_mock: respx.MockRouter, +) -> None: + for name in ("api_key", "openai_key", "api_base"): + monkeypatch.setattr(litellm, name, settings.get(name)) + for name in ("OPENAI_API_KEY", "OPENAI_BASE_URL", "OPENAI_API_BASE"): + monkeypatch.delenv(name, raising=False) + for name, value in env.items(): + monkeypatch.setenv(name, value) + route: Final = respx_mock.post(url).respond(json=_OPENAI_RESPONSE) + + litellm.decisions( + model="openai/gpt-6-luna", + state="review", + questions={"is_defect": {"type": "noul", "instructions": "Is this a defect?"}}, + ) + + assert route.call_count == 1 + assert route.calls[0].request.headers["authorization"] == authorization + + +@pytest.mark.asyncio +async def test_openai_format_calls_to_a_systemone_provider_get_openai_format_answers( + respx_mock: respx.MockRouter, +) -> None: + route: Final = respx_mock.post("https://api.typesafe.ai/v1/systemone").respond(json=_RESPONSE) + + response: Final = await litellm.adecisions( + model="typesafe/jev-1.13", + input="review", + questions=[ + {"type": "predicate", "name": "is_defect", "instructions": "Is this a defect?"}, + { + "type": "choice", + "name": "sentiment", + "instructions": "Tone?", + "choices": [{"value": "positive"}, {"value": "negative"}], + }, + ], + api_key="caller-key", + ) + + assert route.called + assert tuple(json.loads(respx_mock.calls[0].request.content)["questions"]) == ("is_defect", "sentiment") + assert isinstance(response, OpenAIDecisionResponse) + assert [answer.model_dump(mode="json") for answer in response.answers] == [ + {"type": "predicate", "name": "is_defect", "probability": 0.9}, + { + "type": "choice", + "name": "sentiment", + "choice": "positive", + "probabilities": [{"value": "positive", "probability": 0.8}, {"value": "negative", "probability": 0.2}], + "confidence": 0.8, + }, + ] + + +@pytest.mark.asyncio +@pytest.mark.parametrize( + ("model", "request_kwargs", "message"), + ( + ( + "typesafe/jev-1.13", + { + "input": [ + { + "role": "user", + "content": [{"type": "input_image", "image_url": "data:image/png;base64,AA=="}], + } + ], + "questions": [{"type": "predicate", "instructions": "Is this a defect?"}], + }, + "cannot serve this request", + ), + ( + "openai/gpt-6-luna", + { + "state": "review", + "input": "review", + "questions": [{"type": "predicate", "instructions": "Is this a defect?"}], + }, + "not both", + ), + ), + ids=("image_to_systemone_provider", "state_and_input"), +) +async def test_requests_a_provider_cannot_serve_are_rejected_before_http( + respx_mock: respx.MockRouter, + model: str, + request_kwargs: Mapping[str, object], + message: str, +) -> None: + with pytest.raises(litellm.BadRequestError, match=message): + await litellm.adecisions(model=model, api_key="caller-key", **request_kwargs) + + assert len(respx_mock.calls) == 0 diff --git a/tests/unit/decisions/test_openai_transformation.py b/tests/unit/decisions/test_openai_transformation.py deleted file mode 100644 index a3cc1583d40..00000000000 --- a/tests/unit/decisions/test_openai_transformation.py +++ /dev/null @@ -1,32 +0,0 @@ -from collections.abc import Mapping -from typing import Final - -import pytest -from pydantic import TypeAdapter, ValidationError - -from litellm.decisions.openai_transformation import to_systemone_request -from litellm.types.decisions import MAX_DECISION_QUESTIONS, DecisionsRequestBody, OpenAIDecisionRequestBody - -_OPENAI_BODY: Final[TypeAdapter[OpenAIDecisionRequestBody]] = TypeAdapter(OpenAIDecisionRequestBody) -_SYSTEMONE_BODY: Final[TypeAdapter[DecisionsRequestBody]] = TypeAdapter(DecisionsRequestBody) - - -def _openai_request(question_count: int) -> Mapping[str, object]: - return { - "model": "decider", - "input": "The package arrived with a broken screen.", - "questions": [{"type": "predicate", "instructions": f"Question {index}?"} for index in range(question_count)], - } - - -def test_the_largest_openai_request_accepted_translates_to_a_valid_systemone_request() -> None: - raw: Final = _openai_request(MAX_DECISION_QUESTIONS) - - translated: Final = _SYSTEMONE_BODY.validate_python(to_systemone_request(raw, _OPENAI_BODY.validate_python(raw))) - - assert len(translated.questions) == MAX_DECISION_QUESTIONS - - -def test_an_openai_request_with_more_questions_than_systemone_takes_is_rejected_before_translation() -> None: - with pytest.raises(ValidationError, match="questions"): - _OPENAI_BODY.validate_python(_openai_request(MAX_DECISION_QUESTIONS + 1)) diff --git a/tests/unit/llms/base_llm/decisions/test_base_decisions_transformation.py b/tests/unit/llms/base_llm/decisions/test_base_decisions_transformation.py new file mode 100644 index 00000000000..a452a621d2e --- /dev/null +++ b/tests/unit/llms/base_llm/decisions/test_base_decisions_transformation.py @@ -0,0 +1,180 @@ +from collections.abc import Mapping +from typing import Final + +import pytest +from pydantic import TypeAdapter + +from litellm.llms.base_llm.decisions.transformation import ( + ir_to_systemone_request, + ir_to_systemone_response, + parse_systemone_response, + systemone_request_to_ir, +) +from litellm.llms.openai.decisions.transformation import openai_request_to_ir +from litellm.types.decisions import ( + MAX_DECISION_QUESTIONS, + DecisionsIRRefusal, + DecisionsIRRequest, + DecisionsRequestBody, + OpenAIDecisionRequestBody, + UnsupportedDecisionsRequest, +) + +_SYSTEMONE_BODY: Final[TypeAdapter[DecisionsRequestBody]] = TypeAdapter(DecisionsRequestBody) +_OPENAI_BODY: Final[TypeAdapter[OpenAIDecisionRequestBody]] = TypeAdapter(OpenAIDecisionRequestBody) + +_SYSTEMONE_REQUEST: Final[Mapping[str, object]] = { + "state": {"ticket": 1234, "text": "Screen cracked"}, + "questions": { + "damaged": { + "type": "noul", + "instructions": "Is the item damaged?", + "criteria": {"true": "Visible damage", "false": None}, + "provider_field": "dropped", + }, + "rubric_only": {"type": "noul", "criteria": {"true": {"signal": "refund"}}}, + "action": { + "type": "choice", + "instructions": {"policy": "refund-v2"}, + "criteria": {"refund": "Within 30 days", "escalate": None}, + }, + "severity": {"type": "score", "criteria": ["minor", {"label": "major"}]}, + }, +} + + +def _openai_ir(raw: Mapping[str, object]) -> DecisionsIRRequest: + return openai_request_to_ir(_OPENAI_BODY.validate_python(raw)) + + +def test_a_systemone_request_reaches_a_systemone_provider_unchanged() -> None: + ir: Final = systemone_request_to_ir(_SYSTEMONE_BODY.validate_python(_SYSTEMONE_REQUEST)) + + assert ir_to_systemone_request("jev-latest", ir) == {"model": "jev-latest", **_SYSTEMONE_REQUEST} + + +@pytest.mark.parametrize( + ("names", "keys"), + ( + (("damaged", "action"), ("damaged", "action")), + (("damaged", None), ("0", "1")), + (("damaged", "damaged"), ("0", "1")), + ), + ids=("all_named", "one_unnamed", "duplicate_names"), +) +def test_openai_questions_are_keyed_by_name_only_when_every_name_is_unique( + names: tuple[str | None, str | None], keys: tuple[str, str] +) -> None: + questions: Final = [ + {"type": "predicate", "instructions": f"Question {index}?", **({} if name is None else {"name": name})} + for index, name in enumerate(names) + ] + + body: Final = ir_to_systemone_request("jev-latest", _openai_ir({"input": "review", "questions": questions})) + + assert tuple(_SYSTEMONE_BODY.validate_python(body).questions) == keys + + +def test_openai_messages_become_systemone_state_text() -> None: + ir: Final = _openai_ir( + { + "input": [ + { + "role": "user", + "content": [ + {"type": "input_text", "text": "The package arrived broken."}, + {"type": "input_text", "text": "I want a refund."}, + ], + }, + {"role": "user", "content": "Order 1234."}, + ], + "questions": [{"type": "predicate", "instructions": "Is this a defect?"}], + } + ) + + body: Final = ir_to_systemone_request("jev-latest", ir) + + assert isinstance(body, Mapping) + assert body["state"] == "The package arrived broken.\n\nI want a refund.\n\nOrder 1234." + + +def test_image_input_is_unsupported_by_systemone_providers() -> None: + ir: Final = _openai_ir( + { + "input": [ + { + "role": "user", + "content": [ + {"type": "input_text", "text": "Is the screen cracked?"}, + {"type": "input_image", "image_url": "data:image/png;base64,AA=="}, + ], + } + ], + "questions": [{"type": "predicate", "instructions": "Is this a defect?"}], + } + ) + + assert isinstance(ir_to_systemone_request("jev-latest", ir), UnsupportedDecisionsRequest) + + +def test_the_largest_openai_request_accepted_translates_to_a_valid_systemone_request() -> None: + ir: Final = _openai_ir( + { + "input": "review", + "questions": [ + {"type": "predicate", "instructions": f"Question {index}?"} for index in range(MAX_DECISION_QUESTIONS) + ], + } + ) + + translated: Final = _SYSTEMONE_BODY.validate_python(ir_to_systemone_request("jev-latest", ir)) + + assert len(translated.questions) == MAX_DECISION_QUESTIONS + + +def test_a_systemone_response_reaches_the_caller_unchanged_with_provider_extras() -> None: + payload: Final = { + "model": "jev-latest", + "answers": { + "damaged": {"type": "noul", "noul": 0.95, "rationale": "crack visible"}, + "action": { + "type": "choice", + "choice": "refund", + "confidence": 0.8, + "probabilities": {"refund": 0.9, "escalate": 0.1}, + "calibrated": True, + }, + "severity": { + "type": "score", + "score": 0.4, + "confidence": 0.6, + "legend": {"0": "minor", "1": {"label": "major"}}, + "probabilities": {"0": 0.6, "1": 0.4}, + "raw_logits": [0.1, 0.2], + }, + }, + "usage": {"input_tokens": 383, "output_tokens": 2, "cost": 0.25}, + "latency_ms": 3722.17, + } + ir: Final = systemone_request_to_ir(_SYSTEMONE_BODY.validate_python(_SYSTEMONE_REQUEST)) + + response: Final = ir_to_systemone_response(parse_systemone_response(payload, ir), ir) + + assert response.model_dump(mode="json") == payload + + +def test_missing_or_mismatched_systemone_answers_are_refusals_left_out_of_the_response() -> None: + ir: Final = systemone_request_to_ir(_SYSTEMONE_BODY.validate_python(_SYSTEMONE_REQUEST)) + payload: Final = { + "model": "jev-latest", + "answers": { + "damaged": {"type": "noul", "noul": 0.95}, + "action": {"type": "noul", "noul": 0.5}, + }, + "usage": {"input_tokens": 10, "output_tokens": 1}, + } + + parsed: Final = parse_systemone_response(payload, ir) + + assert parsed.answers[1:] == (DecisionsIRRefusal(), DecisionsIRRefusal(), DecisionsIRRefusal()) + assert tuple(ir_to_systemone_response(parsed, ir).answers) == ("damaged",) diff --git a/tests/unit/llms/openai/decisions/__init__.py b/tests/unit/llms/openai/decisions/__init__.py new file mode 100644 index 00000000000..e69de29bb2d diff --git a/tests/unit/llms/openai/decisions/test_openai_decisions_transformation.py b/tests/unit/llms/openai/decisions/test_openai_decisions_transformation.py new file mode 100644 index 00000000000..e73c5b92ab6 --- /dev/null +++ b/tests/unit/llms/openai/decisions/test_openai_decisions_transformation.py @@ -0,0 +1,256 @@ +from collections.abc import Mapping +from typing import Final + +import pytest +from pydantic import TypeAdapter + +from litellm.llms.base_llm.decisions.transformation import ir_to_systemone_response, systemone_request_to_ir +from litellm.llms.openai.decisions.transformation import ( + OpenAIDecisionsConfig, + ir_to_openai_request, + ir_to_openai_response, + openai_request_to_ir, +) +from litellm.types.decisions import ( + DecisionsIRRequest, + DecisionsRequestBody, + OpenAIDecisionRequestBody, + UnsupportedDecisionsRequest, +) + +_SYSTEMONE_BODY: Final[TypeAdapter[DecisionsRequestBody]] = TypeAdapter(DecisionsRequestBody) +_OPENAI_BODY: Final[TypeAdapter[OpenAIDecisionRequestBody]] = TypeAdapter(OpenAIDecisionRequestBody) + +_OPENAI_REQUEST: Final[Mapping[str, object]] = { + "input": [ + { + "type": "message", + "role": "user", + "content": [ + {"type": "input_text", "text": "The screen is cracked."}, + {"type": "input_image", "image_url": "data:image/png;base64,AA==", "detail": "high"}, + ], + }, + {"type": "message", "role": "user", "content": "Order 1234."}, + ], + "questions": [ + {"type": "predicate", "name": "damaged", "instructions": "Is the item damaged?"}, + { + "type": "choice", + "instructions": "Should we refund?", + "choices": [{"value": True, "description": "Refund now"}, {"value": "escalate"}], + }, + { + "type": "score", + "name": "severity", + "instructions": "How severe is it?", + "levels": [{"label": "minor"}, {"label": "major", "description": "Product unusable"}], + }, + {"type": "predicate", "name": "fraud", "instructions": "Is this fraud?"}, + ], + "safety_identifier": "end-user-1", +} +_PREDICATE_ANSWER: Final[Mapping[str, object]] = {"type": "predicate", "name": "damaged", "probability": 0.95} +_CHOICE_ANSWER: Final[Mapping[str, object]] = { + "type": "choice", + "name": None, + "choice": True, + "probabilities": [{"value": True, "probability": 0.9}, {"value": "escalate", "probability": 0.1}], + "confidence": 0.8, +} +_REFUSAL_ANSWER: Final[Mapping[str, object]] = {"type": "refusal", "name": "fraud"} +_USAGE: Final[Mapping[str, object]] = { + "input_tokens": 383, + "input_tokens_details": {"cached_tokens": 256, "cache_write_tokens": 64}, + "output_tokens": 2, + "output_tokens_details": {"reasoning_tokens": 1}, + "total_tokens": 385, +} +_OPENAI_RESPONSE: Final[Mapping[str, object]] = { + "model": "gpt-6-luna", + "answers": [ + _PREDICATE_ANSWER, + _CHOICE_ANSWER, + { + "type": "score", + "name": "severity", + "score": 0.7, + "probabilities": [ + {"value": 0, "label": "minor", "probability": 0.3}, + {"value": 1, "label": "major", "probability": 0.7}, + ], + "confidence": 0.6, + }, + _REFUSAL_ANSWER, + ], + "usage": _USAGE, +} + + +def _openai_ir(raw: Mapping[str, object]) -> DecisionsIRRequest: + return openai_request_to_ir(_OPENAI_BODY.validate_python(raw)) + + +def test_an_openai_request_reaches_openai_unchanged() -> None: + assert ir_to_openai_request("gpt-6-luna", _openai_ir(_OPENAI_REQUEST)) == { + "model": "gpt-6-luna", + **_OPENAI_REQUEST, + } + + +def test_an_openai_response_reaches_the_caller_unchanged() -> None: + ir: Final = _openai_ir(_OPENAI_REQUEST) + + parsed: Final = OpenAIDecisionsConfig().parse_response(_OPENAI_RESPONSE, ir) + + assert ir_to_openai_response(parsed, ir, "requested").model_dump(mode="json") == _OPENAI_RESPONSE + + +def test_answers_openai_did_not_return_are_refusals() -> None: + ir: Final = _openai_ir(_OPENAI_REQUEST) + payload: Final = {**_OPENAI_RESPONSE, "answers": [_PREDICATE_ANSWER]} + + response: Final = ir_to_openai_response(OpenAIDecisionsConfig().parse_response(payload, ir), ir, "requested") + + assert [answer.type for answer in response.answers] == ["predicate", "refusal", "refusal", "refusal"] + assert [answer.name for answer in response.answers] == ["damaged", None, "severity", "fraud"] + + +def test_a_systemone_request_becomes_an_openai_request_with_questions_named_by_their_keys() -> None: + request: Final = _SYSTEMONE_BODY.validate_python( + { + "state": {"ticket": 1234, "text": "Screen cracked"}, + "questions": { + "damaged": { + "type": "noul", + "instructions": "Is the item damaged?", + "criteria": {"true": "Visible damage", "false": None}, + "provider_field": "dropped", + }, + "rubric_only": {"type": "noul", "criteria": {"true": {"signal": "refund"}}}, + "action": { + "type": "choice", + "instructions": {"policy": "refund-v2"}, + "criteria": {"refund": "Within 30 days", "escalate": None}, + }, + "severity": {"type": "score", "criteria": ["minor", {"label": "major"}]}, + }, + } + ) + + assert ir_to_openai_request("gpt-6-luna", systemone_request_to_ir(request)) == { + "model": "gpt-6-luna", + "input": '{"ticket": 1234, "text": "Screen cracked"}', + "questions": [ + { + "type": "predicate", + "name": "damaged", + "instructions": "Is the item damaged?\n\nAnswer true when: Visible damage", + }, + {"type": "predicate", "name": "rubric_only", "instructions": 'Answer true when: {"signal": "refund"}'}, + { + "type": "choice", + "name": "action", + "instructions": '{"policy": "refund-v2"}', + "choices": [{"value": "refund", "description": "Within 30 days"}, {"value": "escalate"}], + }, + { + "type": "score", + "name": "severity", + "instructions": "Which level best fits the input?", + "levels": [{"label": "minor"}, {"label": '{"label": "major"}'}], + }, + ], + } + + +def test_systemone_questions_without_instructions_become_valid_openai_questions() -> None: + request: Final = _SYSTEMONE_BODY.validate_python( + { + "state": "Screen cracked", + "questions": { + "damaged": {"type": "noul", "criteria": {"false": None}}, + "action": {"type": "choice", "criteria": {"refund": None, "escalate": None}}, + "severity": {"type": "score", "criteria": ["minor", "major"]}, + }, + } + ) + + body: Final = ir_to_openai_request("gpt-6-luna", systemone_request_to_ir(request)) + + assert all(question.instructions for question in _OPENAI_BODY.validate_python(body).questions) + + +@pytest.mark.parametrize( + "question", + ({"type": "choice", "criteria": {"refund": None}}, {"type": "score", "criteria": ["minor"]}), + ids=("choice", "score"), +) +def test_a_systemone_question_with_one_option_is_unsupported_by_openai(question: Mapping[str, object]) -> None: + request: Final = _SYSTEMONE_BODY.validate_python( + { + "state": "Screen cracked", + "questions": {"damaged": {"type": "noul", "instructions": "Damaged?"}, "q": question}, + } + ) + + assert isinstance(ir_to_openai_request("gpt-6-luna", systemone_request_to_ir(request)), UnsupportedDecisionsRequest) + + +def test_an_openai_response_becomes_systemone_answers_with_the_callers_score_labels() -> None: + ir: Final = systemone_request_to_ir( + _SYSTEMONE_BODY.validate_python( + { + "state": "Screen cracked", + "questions": { + "damaged": {"type": "noul", "instructions": "Damaged?"}, + "action": {"type": "choice", "criteria": {"true": None, "escalate": None}}, + "severity": {"type": "score", "criteria": ["minor", {"label": "major"}]}, + "fraud": {"type": "noul", "instructions": "Fraud?"}, + }, + } + ) + ) + payload: Final = { + **_OPENAI_RESPONSE, + "answers": [ + _PREDICATE_ANSWER, + _CHOICE_ANSWER, + { + "type": "score", + "name": "severity", + "score": 0.7, + "probabilities": [ + {"value": 0, "label": "minor", "probability": 0.3}, + {"value": 1, "label": '{"label": "major"}', "probability": 0.7}, + ], + "confidence": 0.6, + }, + _REFUSAL_ANSWER, + ], + } + + response: Final = ir_to_systemone_response(OpenAIDecisionsConfig().parse_response(payload, ir), ir) + + assert response.model_dump(mode="json") == { + "model": "gpt-6-luna", + "answers": { + "damaged": {"type": "noul", "noul": 0.95}, + "action": { + "type": "choice", + "choice": "true", + "confidence": 0.8, + "probabilities": {"true": 0.9, "escalate": 0.1}, + }, + "severity": { + "type": "score", + "score": 0.7, + "confidence": 0.6, + "legend": {"0": "minor", "1": {"label": "major"}}, + "probabilities": {"0": 0.3, "1": 0.7}, + }, + }, + "usage": {"input_tokens": 383, "output_tokens": 2}, + } + assert response.usage is not None + assert (response.usage.cached_tokens, response.usage.cache_write_tokens) == (256, 64) diff --git a/tests/unit/proxy/decisions_endpoints/test_endpoints.py b/tests/unit/proxy/decisions_endpoints/test_endpoints.py index b00f2be092c..ccfa12e10d6 100644 --- a/tests/unit/proxy/decisions_endpoints/test_endpoints.py +++ b/tests/unit/proxy/decisions_endpoints/test_endpoints.py @@ -320,6 +320,28 @@ _SYSTEMONE_ANSWERS_FOR_OPENAI_REQUEST: Final[Mapping[str, object]] = { "usage": {"input_tokens": _INPUT_TOKENS, "output_tokens": _OUTPUT_TOKENS}, } +_OPENAI_FORMAT_ANSWERS: Final = [ + {"type": "predicate", "name": "damaged", "probability": 0.95}, + { + "type": "choice", + "name": None, + "choice": True, + "probabilities": [{"value": True, "probability": 0.9}, {"value": "escalate", "probability": 0.1}], + "confidence": 0.8, + }, + { + "type": "score", + "name": "severity", + "score": 0.7, + "probabilities": [ + {"value": 0, "label": "minor", "probability": 0.3}, + {"value": 1, "label": "major", "probability": 0.7}, + ], + "confidence": 0.6, + }, + {"type": "refusal", "name": "fraud"}, +] + @pytest.mark.parametrize("endpoint", ("/v1/decisions", "/decisions")) def test_openai_format_decisions_translate_through_systemone( @@ -354,27 +376,7 @@ def test_openai_format_decisions_translate_through_systemone( } body: Final = response.json() assert body["model"] == _SYSTEMONE_ANSWERS_FOR_OPENAI_REQUEST["model"] - assert body["answers"] == [ - {"type": "predicate", "name": "damaged", "probability": 0.95}, - { - "type": "choice", - "name": None, - "choice": True, - "probabilities": [{"value": True, "probability": 0.9}, {"value": "escalate", "probability": 0.1}], - "confidence": 0.8, - }, - { - "type": "score", - "name": "severity", - "score": 0.7, - "probabilities": [ - {"value": 0, "label": "minor", "probability": 0.3}, - {"value": 1, "label": "major", "probability": 0.7}, - ], - "confidence": 0.6, - }, - {"type": "refusal", "name": "fraud"}, - ] + assert body["answers"] == _OPENAI_FORMAT_ANSWERS assert body["usage"] == { "input_tokens": _INPUT_TOKENS, "input_tokens_details": {"cached_tokens": 0, "cache_write_tokens": 0}, @@ -462,6 +464,72 @@ def test_a_body_that_is_not_json_is_a_client_error( assert not upstream.called +def test_openai_format_decisions_reach_an_openai_deployment_unchanged_including_images( + client: TestClient, + monkeypatch: pytest.MonkeyPatch, + respx_mock: respx.MockRouter, +) -> None: + monkeypatch.setattr(litellm, "api_base", None) + monkeypatch.delenv("OPENAI_BASE_URL", raising=False) + monkeypatch.delenv("OPENAI_API_BASE", raising=False) + monkeypatch.setattr( + litellm.proxy.proxy_server, + "llm_router", + litellm.Router( + model_list=[{"model_name": "decider", "litellm_params": {"model": "openai/gpt-6-luna", "api_key": "k"}}] + ), + ) + image_message: Final = { + "type": "message", + "role": "user", + "content": [{"type": "input_image", "image_url": "data:image/png;base64,AA==", "detail": "low"}], + } + request_body: Final = {**_OPENAI_FORMAT_REQUEST, "input": [*_OPENAI_FORMAT_REQUEST["input"], image_message]} + cached_tokens: Final = 128 + upstream_usage: Final = { + "input_tokens": _INPUT_TOKENS, + "input_tokens_details": {"cached_tokens": cached_tokens, "cache_write_tokens": 0}, + "output_tokens": _OUTPUT_TOKENS, + "output_tokens_details": {"reasoning_tokens": 0}, + "total_tokens": _INPUT_TOKENS + _OUTPUT_TOKENS, + } + upstream: Final = respx_mock.post("https://api.openai.com/v1/decisions").respond( + json={"model": "gpt-6-luna", "answers": _OPENAI_FORMAT_ANSWERS, "usage": upstream_usage} + ) + + response: Final = client.post("/v1/decisions", json=request_body) + + assert response.status_code == 200, response.text + assert json.loads(upstream.calls[0].request.content) == { + "model": "gpt-6-luna", + "input": [ + { + "type": "message", + "role": "user", + "content": [ + {"type": "input_text", "text": "The package arrived with a broken screen."}, + {"type": "input_text", "text": "I want a refund."}, + ], + }, + {"type": "message", "role": "user", "content": "Order 1234."}, + image_message, + ], + "questions": _OPENAI_FORMAT_REQUEST["questions"], + "safety_identifier": "end-user-1", + } + body: Final = response.json() + assert body["answers"] == _OPENAI_FORMAT_ANSWERS + assert body["usage"] == upstream_usage + luna_cost: Final = litellm.model_cost["gpt-6-luna"] + expected_cost: Final = ( + (_INPUT_TOKENS - cached_tokens) * float(luna_cost["input_cost_per_token"]) + + cached_tokens * float(luna_cost["cache_read_input_token_cost"]) + + _OUTPUT_TOKENS * float(luna_cost["output_cost_per_token"]) + ) + assert expected_cost > 0 + assert float(response.headers["x-litellm-response-cost"]) == pytest.approx(expected_cost) + + def _decisions_feature() -> LazyFeature: return next(feature for feature in LAZY_FEATURES if feature.name == "decisions") diff --git a/tests/unit/test_utils.py b/tests/unit/test_utils.py index ac5dc6e7c45..7551a310977 100644 --- a/tests/unit/test_utils.py +++ b/tests/unit/test_utils.py @@ -1064,6 +1064,7 @@ def test_aaamodel_prices_and_context_window_json_is_valid(): "/vertex_ai/live", "/v1/listen", "/v1/systemone", + "/v1/decisions", "/v1beta/interactions", ], },