mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-09 03:18:44 +00:00
feat(decisions): add OpenAI as a Decisions provider behind a shared decisions format (#45214)
* feat(decisions): serve System One format at /v1/systemone and OpenAI format at /v1/decisions The System One request format moves to /v1/systemone and /systemone. /v1/decisions and /decisions now accept the OpenAI Decisions API format, translate it into a System One request, route it through the same pipeline, and translate the answers back. * fix(decisions): import assert_never from typing_extensions for Python 3.10 * fix(decisions): cap OpenAI-format questions at the System One limit /v1/decisions accepted up to 200 questions, but the System One request it translates to takes at most 128, so 129 to 200 questions failed with an internal validation error. Both now share MAX_DECISION_QUESTIONS. * fix(decisions): return 400 for bodies that are not JSON * test(decisions): send System One bodies to /v1/systemone in integration tests * feat(decisions): add OpenAI as a Decisions provider System One requests to openai/ models are translated to OpenAI's Decisions shape on the way out and OpenAI's answers are translated back, so both /v1/systemone and /v1/decisions can route to gpt-6-luna. * refactor(decisions): move the OpenAI wire translation into llms/openai * feat(decisions): route both request formats through a shared decisions IR System One and OpenAI-format bodies now convert to one internal representation, and each provider translates it to its own wire format. OpenAI deployments receive the caller's messages, images, names, typed choice values, level descriptions and safety_identifier unchanged. System One providers return a 400 for image input. Cached and cache-write tokens are priced on both response shapes, and provider response extras survive the round trip. * fix(decisions): keep provider fields inside System One answers * fix(decisions): load on Python 3.10 and 3.11 and bill cached tokens once under custom pricing Decisions IR dataclasses used a mappingproxy default, which is only hashable from Python 3.12, so import litellm failed on 3.10 and 3.11. Decisions usage now reaches cost_per_token as a Usage with prompt_tokens_details, so custom pricing no longer adds cached and cache-write tokens on top of input_tokens that already include them * fix(decisions): honor litellm OpenAI key and base settings for the openai provider OpenAI decisions now resolve the key from litellm.api_key, litellm.openai_key, then OPENAI_API_KEY, and the base from litellm.api_base, OPENAI_BASE_URL, then OPENAI_API_BASE, matching other OpenAI calls. Providers own their configured key and base lookup through the endpoint config. * test(decisions): clear every OpenAI base setting in the proxy OpenAI deployment test * fix(decisions): send OpenAI instructions for System One questions without them and reject single-option questions * fix(decisions): return guardrail_information on OpenAI-format decisions when requested * refactor(decisions): validate proxy request data before reading guardrail settings
This commit is contained in:
parent
8fc31b1e4e
commit
33d908e0ae
21 changed files with 1753 additions and 309 deletions
|
|
@ -1682,6 +1682,9 @@ if TYPE_CHECKING:
|
|||
from .llms.strands_decider.decisions.transformation import (
|
||||
StrandsDeciderDecisionsConfig as StrandsDeciderDecisionsConfig,
|
||||
)
|
||||
from .llms.openai.decisions.transformation import (
|
||||
OpenAIDecisionsConfig as OpenAIDecisionsConfig,
|
||||
)
|
||||
from .llms.nvidia_nim.rerank.transformation import (
|
||||
NvidiaNimRerankConfig as NvidiaNimRerankConfig,
|
||||
)
|
||||
|
|
|
|||
|
|
@ -161,6 +161,7 @@ LLM_CONFIG_NAMES: Final = (
|
|||
"OpenRouterDecisionsConfig",
|
||||
"CloudflareDecisionsConfig",
|
||||
"StrandsDeciderDecisionsConfig",
|
||||
"OpenAIDecisionsConfig",
|
||||
"NvidiaNimRerankConfig",
|
||||
"NvidiaNimRankingConfig",
|
||||
"VertexAIRerankConfig",
|
||||
|
|
@ -720,6 +721,7 @@ _LLM_CONFIGS_IMPORT_MAP: Final = {
|
|||
".llms.strands_decider.decisions.transformation",
|
||||
"StrandsDeciderDecisionsConfig",
|
||||
),
|
||||
"OpenAIDecisionsConfig": (".llms.openai.decisions.transformation", "OpenAIDecisionsConfig"),
|
||||
"NvidiaNimRerankConfig": (
|
||||
".llms.nvidia_nim.rerank.transformation",
|
||||
"NvidiaNimRerankConfig",
|
||||
|
|
|
|||
|
|
@ -9,7 +9,7 @@ from typing import TYPE_CHECKING, Any, Final, Literal, cast
|
|||
|
||||
from httpx import Response
|
||||
from pydantic import BaseModel
|
||||
from typing_extensions import ReadOnly, TypedDict
|
||||
from typing_extensions import ReadOnly, TypedDict, assert_never
|
||||
|
||||
import litellm
|
||||
import litellm._logging
|
||||
|
|
@ -100,7 +100,7 @@ from litellm.llms.vertex_ai.cost_calculator import cost_router as google_cost_ro
|
|||
from litellm.llms.xai.cost_calculator import cost_per_token as xai_cost_per_token
|
||||
from litellm.responses.utils import ResponseAPILoggingUtils
|
||||
from litellm.types.agents import LiteLLMSendMessageResponse
|
||||
from litellm.types.decisions import DecisionsResponse, DecisionsUsage
|
||||
from litellm.types.decisions import DecisionsResponse, DecisionsUsage, OpenAIDecisionResponse, OpenAIDecisionUsage
|
||||
from litellm.types.llms.base import CachedTokensDetails, LiteLLMBaseModel
|
||||
from litellm.types.llms.openai import (
|
||||
HttpxBinaryResponseContent,
|
||||
|
|
@ -1034,6 +1034,8 @@ def get_usage_object(
|
|||
),
|
||||
)
|
||||
|
||||
if isinstance(completion_response, (DecisionsResponse, OpenAIDecisionResponse)):
|
||||
return None if completion_response.usage is None else _decisions_usage(completion_response.usage)
|
||||
if usage_obj is None:
|
||||
return None
|
||||
if isinstance(usage_obj, Usage):
|
||||
|
|
@ -1064,12 +1066,37 @@ def get_usage_object(
|
|||
return None
|
||||
|
||||
|
||||
def _decisions_prompt_tokens_details(usage: DecisionsUsage | OpenAIDecisionUsage) -> PromptTokensDetailsWrapper:
|
||||
match usage:
|
||||
case DecisionsUsage():
|
||||
return PromptTokensDetailsWrapper(
|
||||
cached_tokens=usage.cached_tokens, cache_write_tokens=usage.cache_write_tokens
|
||||
)
|
||||
case OpenAIDecisionUsage():
|
||||
return PromptTokensDetailsWrapper(
|
||||
cached_tokens=usage.input_tokens_details.cached_tokens,
|
||||
cache_write_tokens=usage.input_tokens_details.cache_write_tokens,
|
||||
)
|
||||
case _:
|
||||
assert_never(usage)
|
||||
|
||||
|
||||
def _decisions_usage(usage: DecisionsUsage | OpenAIDecisionUsage) -> Usage:
|
||||
return Usage(
|
||||
prompt_tokens=usage.input_tokens,
|
||||
completion_tokens=usage.output_tokens,
|
||||
total_tokens=usage.input_tokens + usage.output_tokens,
|
||||
prompt_tokens_details=_decisions_prompt_tokens_details(usage),
|
||||
)
|
||||
|
||||
|
||||
def _is_known_usage_objects(usage_obj):
|
||||
"""Returns True if the usage obj is a known Usage type"""
|
||||
return (
|
||||
isinstance(usage_obj, litellm.Usage)
|
||||
or isinstance(usage_obj, ResponseAPIUsage)
|
||||
or isinstance(usage_obj, DecisionsUsage)
|
||||
or isinstance(usage_obj, OpenAIDecisionUsage)
|
||||
or TranscriptionUsageObjectTransformation.is_transcription_usage_object(usage_obj)
|
||||
)
|
||||
|
||||
|
|
@ -1481,11 +1508,8 @@ def completion_cost(
|
|||
"usage",
|
||||
litellm.Usage(**_usage_for_dump.model_dump()),
|
||||
)
|
||||
if isinstance(usage_obj, DecisionsUsage):
|
||||
_usage = {
|
||||
"prompt_tokens": usage_obj.input_tokens,
|
||||
"completion_tokens": usage_obj.output_tokens,
|
||||
}
|
||||
if isinstance(usage_obj, (DecisionsUsage, OpenAIDecisionUsage)):
|
||||
_usage = _decisions_usage(usage_obj).model_dump()
|
||||
elif usage_obj is None:
|
||||
_usage = {}
|
||||
elif isinstance(usage_obj, BaseModel):
|
||||
|
|
@ -1977,7 +2001,8 @@ def response_cost_calculator(
|
|||
| OpenAIModerationResponse
|
||||
| Response
|
||||
| SearchResponse
|
||||
| DecisionsResponse,
|
||||
| DecisionsResponse
|
||||
| OpenAIDecisionResponse,
|
||||
model: str,
|
||||
custom_llm_provider: str | None,
|
||||
call_type: Literal[
|
||||
|
|
|
|||
|
|
@ -1,34 +1,56 @@
|
|||
from collections.abc import Mapping
|
||||
from collections.abc import Mapping, Sequence
|
||||
from dataclasses import dataclass, field
|
||||
from typing import Final
|
||||
from typing import Final, TypeAlias
|
||||
|
||||
import httpx
|
||||
from pydantic import TypeAdapter, ValidationError
|
||||
from typing_extensions import assert_never
|
||||
|
||||
import litellm
|
||||
from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj
|
||||
from litellm.llms.base_llm.chat.transformation import BaseLLMException
|
||||
from litellm.llms.base_llm.decisions.transformation import BaseDecisionsConfig
|
||||
from litellm.llms.base_llm.decisions.transformation import (
|
||||
BaseDecisionsConfig,
|
||||
ir_to_systemone_response,
|
||||
systemone_request_to_ir,
|
||||
)
|
||||
from litellm.llms.custom_httpx.llm_http_handler import BaseLLMHTTPHandler
|
||||
from litellm.llms.openai.decisions.transformation import ir_to_openai_response, openai_request_to_ir
|
||||
from litellm.types.decisions import (
|
||||
DecisionQuestion,
|
||||
DecisionsIRRequest,
|
||||
DecisionsIRResponse,
|
||||
DecisionsJSON,
|
||||
DecisionsRequest,
|
||||
DecisionsRequestBody,
|
||||
DecisionsResponse,
|
||||
OpenAIDecisionInput,
|
||||
OpenAIDecisionQuestion,
|
||||
OpenAIDecisionRequestBody,
|
||||
OpenAIDecisionResponse,
|
||||
UnsupportedDecisionsRequest,
|
||||
)
|
||||
from litellm.types.utils import LlmProviders
|
||||
from litellm.utils import ProviderConfigManager, client
|
||||
|
||||
_DECISIONS_REQUEST_ADAPTER: Final[TypeAdapter[DecisionsRequest]] = TypeAdapter(DecisionsRequest)
|
||||
DecisionsQuestions: TypeAlias = (
|
||||
Mapping[str, DecisionQuestion | Mapping[str, object]] | Sequence[OpenAIDecisionQuestion | Mapping[str, object]]
|
||||
)
|
||||
DecisionsRequestFormat: TypeAlias = DecisionsRequestBody | OpenAIDecisionRequestBody
|
||||
|
||||
_SYSTEMONE_REQUEST_ADAPTER: Final[TypeAdapter[DecisionsRequestBody]] = TypeAdapter(DecisionsRequestBody)
|
||||
_OPENAI_REQUEST_ADAPTER: Final[TypeAdapter[OpenAIDecisionRequestBody]] = TypeAdapter(OpenAIDecisionRequestBody)
|
||||
_HANDLER: Final = BaseLLMHTTPHandler()
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True, repr=False)
|
||||
class _DecisionsCall:
|
||||
model: str
|
||||
requested_model: str
|
||||
custom_llm_provider: str
|
||||
provider_config: BaseDecisionsConfig
|
||||
request: DecisionsRequest
|
||||
request: DecisionsRequestFormat
|
||||
ir_request: DecisionsIRRequest
|
||||
body: Mapping[str, object] = field(repr=False)
|
||||
api_base: str
|
||||
api_key: str | None = field(repr=False)
|
||||
logging_obj: LiteLLMLoggingObj | None
|
||||
|
|
@ -59,11 +81,37 @@ def _provider_config(model: str, custom_llm_provider: str) -> BaseDecisionsConfi
|
|||
return provider_config
|
||||
|
||||
|
||||
def _validate_request(
|
||||
*,
|
||||
state: DecisionsJSON | None,
|
||||
questions: DecisionsQuestions | None,
|
||||
decision_input: OpenAIDecisionInput | None,
|
||||
safety_identifier: str | None,
|
||||
) -> DecisionsRequestFormat:
|
||||
if decision_input is None:
|
||||
return _SYSTEMONE_REQUEST_ADAPTER.validate_python({"state": state, "questions": questions})
|
||||
return _OPENAI_REQUEST_ADAPTER.validate_python(
|
||||
{"input": decision_input, "questions": questions, "safety_identifier": safety_identifier}
|
||||
)
|
||||
|
||||
|
||||
def _ir_request(request: DecisionsRequestFormat) -> DecisionsIRRequest:
|
||||
match request:
|
||||
case DecisionsRequestBody():
|
||||
return systemone_request_to_ir(request)
|
||||
case OpenAIDecisionRequestBody():
|
||||
return openai_request_to_ir(request)
|
||||
case _:
|
||||
assert_never(request)
|
||||
|
||||
|
||||
def _prepare_call(
|
||||
*,
|
||||
model: str,
|
||||
state: DecisionsJSON,
|
||||
questions: Mapping[str, DecisionQuestion | Mapping[str, object]],
|
||||
state: DecisionsJSON | None,
|
||||
questions: DecisionsQuestions | None,
|
||||
decision_input: OpenAIDecisionInput | None,
|
||||
safety_identifier: str | None,
|
||||
api_key: str | None,
|
||||
api_base: str | None,
|
||||
timeout: float | httpx.Timeout | None,
|
||||
|
|
@ -85,9 +133,15 @@ def _prepare_call(
|
|||
model=model,
|
||||
llm_provider=provider,
|
||||
)
|
||||
if state is not None and decision_input is not None:
|
||||
raise litellm.BadRequestError(
|
||||
message="Pass either state (System One format) or input (OpenAI format) to the Decisions API, not both",
|
||||
model=model,
|
||||
llm_provider=provider,
|
||||
)
|
||||
try:
|
||||
request: Final = _DECISIONS_REQUEST_ADAPTER.validate_python(
|
||||
{"model": canonical_model, "state": state, "questions": questions}
|
||||
request: Final = _validate_request(
|
||||
state=state, questions=questions, decision_input=decision_input, safety_identifier=safety_identifier
|
||||
)
|
||||
except ValidationError as error:
|
||||
raise litellm.BadRequestError(
|
||||
|
|
@ -111,6 +165,17 @@ def _prepare_call(
|
|||
llm_provider=provider,
|
||||
)
|
||||
|
||||
ir_request: Final = _ir_request(request)
|
||||
body: Final = provider_config.transform_decisions_request(
|
||||
model=canonical_model, request=ir_request, custom_llm_provider=provider
|
||||
)
|
||||
if isinstance(body, UnsupportedDecisionsRequest):
|
||||
raise litellm.BadRequestError(
|
||||
message=f"Decisions provider '{provider}' cannot serve this request: {body.reason}",
|
||||
model=model,
|
||||
llm_provider=provider,
|
||||
)
|
||||
|
||||
logging_obj: Final = kwargs.get("litellm_logging_obj")
|
||||
if isinstance(logging_obj, LiteLLMLoggingObj):
|
||||
logging_obj.update_from_kwargs(
|
||||
|
|
@ -124,9 +189,12 @@ def _prepare_call(
|
|||
)
|
||||
return _DecisionsCall(
|
||||
model=canonical_model,
|
||||
requested_model=model,
|
||||
custom_llm_provider=provider,
|
||||
provider_config=provider_config,
|
||||
request=request,
|
||||
ir_request=ir_request,
|
||||
body=body,
|
||||
api_base=resolved_api_base,
|
||||
api_key=resolved_api_key,
|
||||
logging_obj=logging_obj if isinstance(logging_obj, LiteLLMLoggingObj) else None,
|
||||
|
|
@ -135,6 +203,30 @@ def _prepare_call(
|
|||
)
|
||||
|
||||
|
||||
def _format_response(response: DecisionsIRResponse, call: _DecisionsCall) -> DecisionsResponse | OpenAIDecisionResponse:
|
||||
formatted: Final = _formatted_response(response, call)
|
||||
formatted.set_hidden_params(
|
||||
{
|
||||
"model": f"{call.custom_llm_provider}/{call.model}",
|
||||
"custom_llm_provider": call.custom_llm_provider,
|
||||
"provider_response_model": f"{call.custom_llm_provider}/{call.model}",
|
||||
}
|
||||
)
|
||||
return formatted
|
||||
|
||||
|
||||
def _formatted_response(
|
||||
response: DecisionsIRResponse, call: _DecisionsCall
|
||||
) -> DecisionsResponse | OpenAIDecisionResponse:
|
||||
match call.request:
|
||||
case DecisionsRequestBody():
|
||||
return ir_to_systemone_response(response, call.ir_request)
|
||||
case OpenAIDecisionRequestBody():
|
||||
return ir_to_openai_response(response, call.ir_request, call.requested_model)
|
||||
case _:
|
||||
assert_never(call.request)
|
||||
|
||||
|
||||
def _map_upstream_exception(error: Exception, call: _DecisionsCall) -> Exception:
|
||||
if isinstance(error, BaseLLMException) and error.status_code_is_synthesized:
|
||||
provider_label: Final = f"{call.custom_llm_provider[0].upper()}{call.custom_llm_provider[1:]}Exception"
|
||||
|
|
@ -153,19 +245,23 @@ def _map_upstream_exception(error: Exception, call: _DecisionsCall) -> Exception
|
|||
@client
|
||||
async def adecisions(
|
||||
model: str,
|
||||
state: DecisionsJSON,
|
||||
questions: Mapping[str, DecisionQuestion | Mapping[str, object]],
|
||||
state: DecisionsJSON | None = None,
|
||||
questions: DecisionsQuestions | None = None,
|
||||
api_key: str | None = None,
|
||||
api_base: str | None = None,
|
||||
timeout: float | httpx.Timeout | None = None,
|
||||
custom_llm_provider: str | None = None,
|
||||
extra_headers: Mapping[str, str] | None = None,
|
||||
input: OpenAIDecisionInput | None = None,
|
||||
safety_identifier: str | None = None,
|
||||
**kwargs: object,
|
||||
) -> DecisionsResponse:
|
||||
) -> DecisionsResponse | OpenAIDecisionResponse:
|
||||
call: Final = _prepare_call(
|
||||
model=model,
|
||||
state=state,
|
||||
questions=questions,
|
||||
decision_input=input,
|
||||
safety_identifier=safety_identifier,
|
||||
api_key=api_key,
|
||||
api_base=api_base,
|
||||
timeout=timeout,
|
||||
|
|
@ -174,12 +270,13 @@ async def adecisions(
|
|||
kwargs=kwargs,
|
||||
)
|
||||
try:
|
||||
return await _HANDLER.adecisions(
|
||||
response: Final = await _HANDLER.adecisions(
|
||||
model=call.model,
|
||||
custom_llm_provider=call.custom_llm_provider,
|
||||
logging_obj=call.logging_obj,
|
||||
provider_config=call.provider_config,
|
||||
request=call.request,
|
||||
request=call.ir_request,
|
||||
body=call.body,
|
||||
api_base=call.api_base,
|
||||
api_key=call.api_key,
|
||||
headers=call.headers,
|
||||
|
|
@ -187,24 +284,29 @@ async def adecisions(
|
|||
)
|
||||
except Exception as error:
|
||||
raise _map_upstream_exception(error, call) from error
|
||||
return _format_response(response, call)
|
||||
|
||||
|
||||
@client
|
||||
def decisions(
|
||||
model: str,
|
||||
state: DecisionsJSON,
|
||||
questions: Mapping[str, DecisionQuestion | Mapping[str, object]],
|
||||
state: DecisionsJSON | None = None,
|
||||
questions: DecisionsQuestions | None = None,
|
||||
api_key: str | None = None,
|
||||
api_base: str | None = None,
|
||||
timeout: float | httpx.Timeout | None = None,
|
||||
custom_llm_provider: str | None = None,
|
||||
extra_headers: Mapping[str, str] | None = None,
|
||||
input: OpenAIDecisionInput | None = None,
|
||||
safety_identifier: str | None = None,
|
||||
**kwargs: object,
|
||||
) -> DecisionsResponse:
|
||||
) -> DecisionsResponse | OpenAIDecisionResponse:
|
||||
call: Final = _prepare_call(
|
||||
model=model,
|
||||
state=state,
|
||||
questions=questions,
|
||||
decision_input=input,
|
||||
safety_identifier=safety_identifier,
|
||||
api_key=api_key,
|
||||
api_base=api_base,
|
||||
timeout=timeout,
|
||||
|
|
@ -213,12 +315,13 @@ def decisions(
|
|||
kwargs=kwargs,
|
||||
)
|
||||
try:
|
||||
return _HANDLER.decisions(
|
||||
response: Final = _HANDLER.decisions(
|
||||
model=call.model,
|
||||
custom_llm_provider=call.custom_llm_provider,
|
||||
logging_obj=call.logging_obj,
|
||||
provider_config=call.provider_config,
|
||||
request=call.request,
|
||||
request=call.ir_request,
|
||||
body=call.body,
|
||||
api_base=call.api_base,
|
||||
api_key=call.api_key,
|
||||
headers=call.headers,
|
||||
|
|
@ -226,6 +329,7 @@ def decisions(
|
|||
)
|
||||
except Exception as error:
|
||||
raise _map_upstream_exception(error, call) from error
|
||||
return _format_response(response, call)
|
||||
|
||||
|
||||
__all__ = ["adecisions", "decisions"]
|
||||
|
|
|
|||
|
|
@ -1,127 +0,0 @@
|
|||
from collections.abc import Mapping, Sequence
|
||||
from typing import Final
|
||||
|
||||
from typing_extensions import assert_never
|
||||
|
||||
from litellm.types.decisions import (
|
||||
ChoiceAnswer,
|
||||
DecisionAnswer,
|
||||
DecisionsResponse,
|
||||
DecisionsUsage,
|
||||
NoulAnswer,
|
||||
OpenAIChoiceAnswer,
|
||||
OpenAIChoiceProbability,
|
||||
OpenAIChoiceQuestion,
|
||||
OpenAIDecisionAnswer,
|
||||
OpenAIDecisionInputMessage,
|
||||
OpenAIDecisionQuestion,
|
||||
OpenAIDecisionRequestBody,
|
||||
OpenAIDecisionResponse,
|
||||
OpenAIDecisionUsage,
|
||||
OpenAIPredicateAnswer,
|
||||
OpenAIPredicateQuestion,
|
||||
OpenAIRefusalAnswer,
|
||||
OpenAIScoreAnswer,
|
||||
OpenAIScoreLevel,
|
||||
OpenAIScoreProbability,
|
||||
OpenAIScoreQuestion,
|
||||
ScoreAnswer,
|
||||
systemone_choice_key,
|
||||
)
|
||||
|
||||
_OPENAI_ONLY_FIELDS: Final = frozenset({"input", "questions", "safety_identifier"})
|
||||
|
||||
|
||||
def _message_text(message: OpenAIDecisionInputMessage) -> str:
|
||||
if isinstance(message.content, str):
|
||||
return message.content
|
||||
return "\n\n".join(part.text for part in message.content)
|
||||
|
||||
|
||||
def _state(decision_input: str | Sequence[OpenAIDecisionInputMessage]) -> str:
|
||||
if isinstance(decision_input, str):
|
||||
return decision_input
|
||||
return "\n\n".join(_message_text(message) for message in decision_input)
|
||||
|
||||
|
||||
def _level_criterion(level: OpenAIScoreLevel) -> str:
|
||||
return level.label if level.description is None else f"{level.label}: {level.description}"
|
||||
|
||||
|
||||
def _systemone_question(question: OpenAIDecisionQuestion) -> Mapping[str, object]:
|
||||
match question:
|
||||
case OpenAIPredicateQuestion():
|
||||
return {"type": "noul", "instructions": question.instructions}
|
||||
case OpenAIChoiceQuestion():
|
||||
return {
|
||||
"type": "choice",
|
||||
"instructions": question.instructions,
|
||||
"criteria": {systemone_choice_key(option.value): option.description for option in question.choices},
|
||||
}
|
||||
case OpenAIScoreQuestion():
|
||||
return {
|
||||
"type": "score",
|
||||
"instructions": question.instructions,
|
||||
"criteria": [_level_criterion(level) for level in question.levels],
|
||||
}
|
||||
case _:
|
||||
assert_never(question)
|
||||
|
||||
|
||||
def to_systemone_request(request_data: Mapping[str, object], body: OpenAIDecisionRequestBody) -> Mapping[str, object]:
|
||||
return {
|
||||
**{key: value for key, value in request_data.items() if key not in _OPENAI_ONLY_FIELDS},
|
||||
"state": _state(body.input),
|
||||
"questions": {str(index): _systemone_question(question) for index, question in enumerate(body.questions)},
|
||||
}
|
||||
|
||||
|
||||
def _openai_answer(question: OpenAIDecisionQuestion, answer: DecisionAnswer | None) -> OpenAIDecisionAnswer:
|
||||
match question, answer:
|
||||
case OpenAIPredicateQuestion(), NoulAnswer():
|
||||
return OpenAIPredicateAnswer(name=question.name, probability=answer.noul)
|
||||
case OpenAIChoiceQuestion(), ChoiceAnswer():
|
||||
typed_values: Final = {systemone_choice_key(option.value): option.value for option in question.choices}
|
||||
return OpenAIChoiceAnswer(
|
||||
name=question.name,
|
||||
choice=typed_values.get(answer.choice, answer.choice),
|
||||
probabilities=tuple(
|
||||
OpenAIChoiceProbability(
|
||||
value=option.value,
|
||||
probability=answer.probabilities.get(systemone_choice_key(option.value), 0.0),
|
||||
)
|
||||
for option in question.choices
|
||||
),
|
||||
confidence=answer.confidence,
|
||||
)
|
||||
case OpenAIScoreQuestion(), ScoreAnswer():
|
||||
return OpenAIScoreAnswer(
|
||||
name=question.name,
|
||||
score=answer.score,
|
||||
probabilities=tuple(
|
||||
OpenAIScoreProbability(
|
||||
value=index, label=level.label, probability=answer.probabilities.get(str(index), 0.0)
|
||||
)
|
||||
for index, level in enumerate(question.levels)
|
||||
),
|
||||
confidence=answer.confidence,
|
||||
)
|
||||
case _:
|
||||
return OpenAIRefusalAnswer(name=question.name)
|
||||
|
||||
|
||||
def to_openai_response(
|
||||
response: DecisionsResponse, questions: Sequence[OpenAIDecisionQuestion], requested_model: str
|
||||
) -> OpenAIDecisionResponse:
|
||||
usage: Final = response.usage or DecisionsUsage()
|
||||
return OpenAIDecisionResponse(
|
||||
model=response.model or requested_model,
|
||||
answers=tuple(
|
||||
_openai_answer(question, response.answers.get(str(index))) for index, question in enumerate(questions)
|
||||
),
|
||||
usage=OpenAIDecisionUsage(
|
||||
input_tokens=usage.input_tokens,
|
||||
output_tokens=usage.output_tokens,
|
||||
total_tokens=usage.input_tokens + usage.output_tokens,
|
||||
),
|
||||
)
|
||||
|
|
@ -120,7 +120,7 @@ from litellm.llms.base_llm.search.transformation import SearchResponse
|
|||
from litellm.responses.utils import ResponseAPILoggingUtils
|
||||
from litellm.types.agents import LiteLLMSendMessageResponse
|
||||
from litellm.types.containers.main import ContainerObject
|
||||
from litellm.types.decisions import DecisionsResponse
|
||||
from litellm.types.decisions import DecisionsResponse, OpenAIDecisionResponse
|
||||
from litellm.types.integrations.s3_v2 import S3PartitionGranularity
|
||||
from litellm.types.interactions import (
|
||||
InteractionsAPIResponse,
|
||||
|
|
@ -2650,6 +2650,7 @@ class Logging(LiteLLMLoggingBaseClass):
|
|||
or isinstance(logging_result, OCRResponse) # OCR
|
||||
or isinstance(logging_result, SearchResponse) # Search API
|
||||
or isinstance(logging_result, DecisionsResponse)
|
||||
or isinstance(logging_result, OpenAIDecisionResponse)
|
||||
or (
|
||||
isinstance(logging_result, InteractionsAPIResponse)
|
||||
and logging_result.usage is not None
|
||||
|
|
|
|||
|
|
@ -1,17 +1,299 @@
|
|||
import json
|
||||
from abc import ABC
|
||||
from collections.abc import Mapping
|
||||
from collections.abc import Mapping, Sequence
|
||||
from types import MappingProxyType
|
||||
from typing import Final
|
||||
|
||||
import httpx
|
||||
from pydantic import TypeAdapter, ValidationError
|
||||
from typing_extensions import assert_never
|
||||
|
||||
from litellm.llms.base_llm.chat.transformation import BaseLLMException
|
||||
from litellm.secret_managers.main import get_secret_str
|
||||
from litellm.types.decisions import DecisionsRequest, DecisionsResponse
|
||||
from litellm.types.decisions import (
|
||||
ChoiceAnswer,
|
||||
ChoiceQuestion,
|
||||
DecisionAnswer,
|
||||
DecisionQuestion,
|
||||
DecisionsIRAnswer,
|
||||
DecisionsIRChoiceAnswer,
|
||||
DecisionsIRChoiceOption,
|
||||
DecisionsIRChoiceProbability,
|
||||
DecisionsIRChoiceQuestion,
|
||||
DecisionsIRMessages,
|
||||
DecisionsIRPredicateAnswer,
|
||||
DecisionsIRPredicateQuestion,
|
||||
DecisionsIRQuestion,
|
||||
DecisionsIRRefusal,
|
||||
DecisionsIRRequest,
|
||||
DecisionsIRResponse,
|
||||
DecisionsIRScoreAnswer,
|
||||
DecisionsIRScoreLevel,
|
||||
DecisionsIRScoreProbability,
|
||||
DecisionsIRScoreQuestion,
|
||||
DecisionsIRState,
|
||||
DecisionsIRUsage,
|
||||
DecisionsJSON,
|
||||
DecisionsRequestBody,
|
||||
DecisionsResponse,
|
||||
DecisionsUsage,
|
||||
NoulAnswer,
|
||||
NoulQuestion,
|
||||
OpenAIDecisionInputImage,
|
||||
OpenAIDecisionInputMessage,
|
||||
OpenAIDecisionInputText,
|
||||
ScoreAnswer,
|
||||
ScoreQuestion,
|
||||
UnsupportedDecisionsRequest,
|
||||
systemone_choice_key,
|
||||
)
|
||||
|
||||
PAYLOAD_ADAPTER: Final[TypeAdapter[object]] = TypeAdapter(object)
|
||||
_RESPONSE_ADAPTER: Final[TypeAdapter[DecisionsResponse]] = TypeAdapter(DecisionsResponse)
|
||||
_PAYLOAD_ADAPTER: Final[TypeAdapter[object]] = TypeAdapter(object)
|
||||
_SYSTEMONE_RESPONSE_ADAPTER: Final[TypeAdapter[DecisionsResponse]] = TypeAdapter(DecisionsResponse)
|
||||
_RESERVED_HEADERS: Final[frozenset[str]] = frozenset({"authorization", "content-type"})
|
||||
_TEXT_ONLY: Final = UnsupportedDecisionsRequest(
|
||||
reason="input_image content parts are not supported because System One providers accept text input only"
|
||||
)
|
||||
|
||||
|
||||
def decisions_text(value: DecisionsJSON) -> str:
|
||||
return value if isinstance(value, str) else json.dumps(value)
|
||||
|
||||
|
||||
def systemone_keys(questions: Sequence[DecisionsIRQuestion]) -> tuple[str, ...]:
|
||||
names: Final = tuple(question.name for question in questions if question.name is not None)
|
||||
if len(frozenset(names)) == len(questions):
|
||||
return names
|
||||
return tuple(str(index) for index in range(len(questions)))
|
||||
|
||||
|
||||
def _ir_question(name: str, question: DecisionQuestion) -> DecisionsIRQuestion:
|
||||
extra: Final = MappingProxyType(question.model_extra or {})
|
||||
match question:
|
||||
case NoulQuestion():
|
||||
return DecisionsIRPredicateQuestion(
|
||||
name=name, instructions=question.instructions, criteria=question.criteria, extra=extra
|
||||
)
|
||||
case ChoiceQuestion():
|
||||
return DecisionsIRChoiceQuestion(
|
||||
name=name,
|
||||
instructions=question.instructions,
|
||||
choices=tuple(
|
||||
DecisionsIRChoiceOption(value=value, description=description)
|
||||
for value, description in question.criteria.items()
|
||||
),
|
||||
extra=extra,
|
||||
)
|
||||
case ScoreQuestion():
|
||||
return DecisionsIRScoreQuestion(
|
||||
name=name,
|
||||
instructions=question.instructions,
|
||||
levels=tuple(
|
||||
DecisionsIRScoreLevel(label=criterion, description=None) for criterion in question.criteria
|
||||
),
|
||||
extra=extra,
|
||||
)
|
||||
case _:
|
||||
assert_never(question)
|
||||
|
||||
|
||||
def systemone_request_to_ir(request: DecisionsRequestBody) -> DecisionsIRRequest:
|
||||
return DecisionsIRRequest(
|
||||
input=DecisionsIRState(state=request.state),
|
||||
questions=tuple(_ir_question(name, question) for name, question in request.questions.items()),
|
||||
)
|
||||
|
||||
|
||||
def _has_image(message: OpenAIDecisionInputMessage) -> bool:
|
||||
return not isinstance(message.content, str) and any(
|
||||
isinstance(part, OpenAIDecisionInputImage) for part in message.content
|
||||
)
|
||||
|
||||
|
||||
def _message_text(message: OpenAIDecisionInputMessage) -> str:
|
||||
if isinstance(message.content, str):
|
||||
return message.content
|
||||
return "\n\n".join(part.text for part in message.content if isinstance(part, OpenAIDecisionInputText))
|
||||
|
||||
|
||||
def _systemone_state(
|
||||
decision_input: DecisionsIRState | DecisionsIRMessages,
|
||||
) -> DecisionsJSON | UnsupportedDecisionsRequest:
|
||||
match decision_input:
|
||||
case DecisionsIRState():
|
||||
return decision_input.state
|
||||
case DecisionsIRMessages():
|
||||
if any(_has_image(message) for message in decision_input.messages):
|
||||
return _TEXT_ONLY
|
||||
return "\n\n".join(_message_text(message) for message in decision_input.messages)
|
||||
case _:
|
||||
assert_never(decision_input)
|
||||
|
||||
|
||||
def _optional(key: str, value: object) -> Mapping[str, object]:
|
||||
return {} if value is None else {key: value}
|
||||
|
||||
|
||||
def _level_criterion(level: DecisionsIRScoreLevel) -> DecisionsJSON:
|
||||
return level.label if level.description is None else f"{decisions_text(level.label)}: {level.description}"
|
||||
|
||||
|
||||
def _systemone_question(question: DecisionsIRQuestion) -> Mapping[str, object]:
|
||||
match question:
|
||||
case DecisionsIRPredicateQuestion():
|
||||
return {
|
||||
"type": "noul",
|
||||
**_optional("instructions", question.instructions),
|
||||
**_optional("criteria", question.criteria),
|
||||
**question.extra,
|
||||
}
|
||||
case DecisionsIRChoiceQuestion():
|
||||
return {
|
||||
"type": "choice",
|
||||
**_optional("instructions", question.instructions),
|
||||
"criteria": {systemone_choice_key(option.value): option.description for option in question.choices},
|
||||
**question.extra,
|
||||
}
|
||||
case DecisionsIRScoreQuestion():
|
||||
return {
|
||||
"type": "score",
|
||||
**_optional("instructions", question.instructions),
|
||||
"criteria": [_level_criterion(level) for level in question.levels],
|
||||
**question.extra,
|
||||
}
|
||||
case _:
|
||||
assert_never(question)
|
||||
|
||||
|
||||
def ir_to_systemone_request(
|
||||
model: str, request: DecisionsIRRequest
|
||||
) -> Mapping[str, object] | UnsupportedDecisionsRequest:
|
||||
state: Final = _systemone_state(request.input)
|
||||
if isinstance(state, UnsupportedDecisionsRequest):
|
||||
return state
|
||||
keyed_questions: Final = zip(systemone_keys(request.questions), request.questions, strict=True)
|
||||
return {
|
||||
"model": model,
|
||||
"state": state,
|
||||
"questions": {key: _systemone_question(question) for key, question in keyed_questions},
|
||||
}
|
||||
|
||||
|
||||
def _ir_choice_answer(question: DecisionsIRChoiceQuestion, answer: ChoiceAnswer) -> DecisionsIRChoiceAnswer:
|
||||
typed_values: Final = {systemone_choice_key(option.value): option.value for option in question.choices}
|
||||
return DecisionsIRChoiceAnswer(
|
||||
choice=typed_values.get(answer.choice, answer.choice),
|
||||
confidence=answer.confidence,
|
||||
probabilities=tuple(
|
||||
DecisionsIRChoiceProbability(value=typed_values.get(key, key), probability=probability)
|
||||
for key, probability in answer.probabilities.items()
|
||||
),
|
||||
extra=MappingProxyType(answer.model_extra or {}),
|
||||
)
|
||||
|
||||
|
||||
def _ir_score_answer(question: DecisionsIRScoreQuestion, answer: ScoreAnswer) -> DecisionsIRScoreAnswer:
|
||||
return DecisionsIRScoreAnswer(
|
||||
score=answer.score,
|
||||
confidence=answer.confidence,
|
||||
probabilities=tuple(
|
||||
DecisionsIRScoreProbability(
|
||||
value=index,
|
||||
label=answer.legend.get(str(index), level.label),
|
||||
probability=answer.probabilities.get(str(index), 0.0),
|
||||
)
|
||||
for index, level in enumerate(question.levels)
|
||||
),
|
||||
extra=MappingProxyType(answer.model_extra or {}),
|
||||
)
|
||||
|
||||
|
||||
def _ir_answer(question: DecisionsIRQuestion, answer: DecisionAnswer | None) -> DecisionsIRAnswer:
|
||||
match question, answer:
|
||||
case DecisionsIRPredicateQuestion(), NoulAnswer():
|
||||
return DecisionsIRPredicateAnswer(probability=answer.noul, extra=MappingProxyType(answer.model_extra or {}))
|
||||
case DecisionsIRChoiceQuestion(), ChoiceAnswer():
|
||||
return _ir_choice_answer(question, answer)
|
||||
case DecisionsIRScoreQuestion(), ScoreAnswer():
|
||||
return _ir_score_answer(question, answer)
|
||||
case _:
|
||||
return DecisionsIRRefusal()
|
||||
|
||||
|
||||
def systemone_response_to_ir(response: DecisionsResponse, request: DecisionsIRRequest) -> DecisionsIRResponse:
|
||||
usage: Final = response.usage or DecisionsUsage()
|
||||
keyed_questions: Final = zip(systemone_keys(request.questions), request.questions, strict=True)
|
||||
return DecisionsIRResponse(
|
||||
model=response.model,
|
||||
answers=tuple(_ir_answer(question, response.answers.get(key)) for key, question in keyed_questions),
|
||||
usage=DecisionsIRUsage(
|
||||
input_tokens=usage.input_tokens,
|
||||
output_tokens=usage.output_tokens,
|
||||
cached_tokens=usage.cached_tokens,
|
||||
cache_write_tokens=usage.cache_write_tokens,
|
||||
extra=MappingProxyType(usage.model_extra or {}),
|
||||
),
|
||||
extra=MappingProxyType(response.model_extra or {}),
|
||||
)
|
||||
|
||||
|
||||
def parse_systemone_response(payload: object, request: DecisionsIRRequest) -> DecisionsIRResponse:
|
||||
return systemone_response_to_ir(_SYSTEMONE_RESPONSE_ADAPTER.validate_python(payload), request)
|
||||
|
||||
|
||||
def _systemone_answer(answer: DecisionsIRAnswer) -> DecisionAnswer | None:
|
||||
match answer:
|
||||
case DecisionsIRPredicateAnswer():
|
||||
return NoulAnswer.model_validate({**answer.extra, "type": "noul", "noul": answer.probability})
|
||||
case DecisionsIRChoiceAnswer():
|
||||
return ChoiceAnswer.model_validate(
|
||||
{
|
||||
**answer.extra,
|
||||
"type": "choice",
|
||||
"choice": systemone_choice_key(answer.choice),
|
||||
"confidence": answer.confidence,
|
||||
"probabilities": {
|
||||
systemone_choice_key(item.value): item.probability for item in answer.probabilities
|
||||
},
|
||||
}
|
||||
)
|
||||
case DecisionsIRScoreAnswer():
|
||||
return ScoreAnswer.model_validate(
|
||||
{
|
||||
**answer.extra,
|
||||
"type": "score",
|
||||
"score": answer.score,
|
||||
"confidence": answer.confidence,
|
||||
"legend": {str(item.value): item.label for item in answer.probabilities},
|
||||
"probabilities": {str(item.value): item.probability for item in answer.probabilities},
|
||||
}
|
||||
)
|
||||
case DecisionsIRRefusal():
|
||||
return None
|
||||
case _:
|
||||
assert_never(answer)
|
||||
|
||||
|
||||
def ir_to_systemone_response(response: DecisionsIRResponse, request: DecisionsIRRequest) -> DecisionsResponse:
|
||||
keyed_answers: Final = zip(
|
||||
systemone_keys(request.questions), (_systemone_answer(answer) for answer in response.answers), strict=True
|
||||
)
|
||||
return DecisionsResponse.model_validate(
|
||||
{
|
||||
**response.extra,
|
||||
"model": response.model,
|
||||
"answers": {key: answer for key, answer in keyed_answers if answer is not None},
|
||||
"usage": DecisionsUsage.model_validate(
|
||||
{
|
||||
**response.usage.extra,
|
||||
"input_tokens": response.usage.input_tokens,
|
||||
"output_tokens": response.usage.output_tokens,
|
||||
"cached_tokens": response.usage.cached_tokens,
|
||||
"cache_write_tokens": response.usage.cache_write_tokens,
|
||||
}
|
||||
),
|
||||
}
|
||||
)
|
||||
|
||||
|
||||
class BaseDecisionsConfig(ABC):
|
||||
|
|
@ -55,48 +337,32 @@ class BaseDecisionsConfig(ABC):
|
|||
def transform_decisions_request(
|
||||
self,
|
||||
model: str,
|
||||
request: DecisionsRequest,
|
||||
request: DecisionsIRRequest,
|
||||
custom_llm_provider: str,
|
||||
) -> dict[str, object]:
|
||||
return {
|
||||
"model": self.request_model(model),
|
||||
"state": request.state,
|
||||
"questions": {
|
||||
name: question.model_dump(mode="json", exclude_none=True)
|
||||
for name, question in request.questions.items()
|
||||
},
|
||||
}
|
||||
) -> Mapping[str, object] | UnsupportedDecisionsRequest:
|
||||
return ir_to_systemone_request(self.request_model(model), request)
|
||||
|
||||
def unwrap_response(self, payload: object) -> object:
|
||||
return payload
|
||||
|
||||
def parse_response(self, payload: object, request: DecisionsIRRequest) -> DecisionsIRResponse:
|
||||
return parse_systemone_response(self.unwrap_response(payload), request)
|
||||
|
||||
def transform_decisions_response(
|
||||
self,
|
||||
model: str,
|
||||
custom_llm_provider: str,
|
||||
raw_response: httpx.Response,
|
||||
request: DecisionsRequest,
|
||||
) -> DecisionsResponse:
|
||||
payload: Final[object] = PAYLOAD_ADAPTER.validate_json(raw_response.content)
|
||||
request: DecisionsIRRequest,
|
||||
) -> DecisionsIRResponse:
|
||||
payload: Final[object] = _PAYLOAD_ADAPTER.validate_json(raw_response.content)
|
||||
try:
|
||||
response: Final = _RESPONSE_ADAPTER.validate_python(self.unwrap_response(payload))
|
||||
return self.parse_response(payload, request)
|
||||
except ValidationError as error:
|
||||
raise BaseLLMException(
|
||||
status_code=500,
|
||||
message=f"Decisions provider '{custom_llm_provider}' returned an unexpected response: {error}",
|
||||
) from error
|
||||
self.set_hidden_params(response, model, custom_llm_provider)
|
||||
return response
|
||||
|
||||
@staticmethod
|
||||
def set_hidden_params(response: DecisionsResponse, model: str, custom_llm_provider: str) -> None:
|
||||
response.set_hidden_params(
|
||||
{
|
||||
"model": f"{custom_llm_provider}/{model}",
|
||||
"custom_llm_provider": custom_llm_provider,
|
||||
"provider_response_model": f"{custom_llm_provider}/{model}",
|
||||
}
|
||||
)
|
||||
|
||||
def get_error_class(
|
||||
self,
|
||||
|
|
|
|||
|
|
@ -123,7 +123,7 @@ from litellm.types.containers.main import (
|
|||
ContainerObject,
|
||||
DeleteContainerResult,
|
||||
)
|
||||
from litellm.types.decisions import DecisionsRequest, DecisionsResponse
|
||||
from litellm.types.decisions import DecisionsIRRequest, DecisionsIRResponse
|
||||
from litellm.types.files import StreamingMediaUploadConfig, TwoStepFileUploadConfig
|
||||
from litellm.types.integrations.custom_logger import (
|
||||
NON_CODE_INTERPRETER_INTERCEPTION_INTERNAL_PREFIXES,
|
||||
|
|
@ -1486,19 +1486,16 @@ class BaseLLMHTTPHandler:
|
|||
def _prepare_decisions_request(
|
||||
self,
|
||||
model: str,
|
||||
custom_llm_provider: str,
|
||||
logging_obj: LiteLLMLoggingObj | None,
|
||||
provider_config: BaseDecisionsConfig,
|
||||
request: DecisionsRequest,
|
||||
body: Mapping[str, object],
|
||||
api_base: str,
|
||||
api_key: str | None,
|
||||
headers: Mapping[str, str],
|
||||
) -> tuple[str, dict[str, str], dict[str, object]]:
|
||||
outbound_headers: Final = provider_config.validate_environment(headers=headers, model=model, api_key=api_key)
|
||||
url: Final = provider_config.get_complete_url(api_base=api_base, model=model)
|
||||
data: Final = provider_config.transform_decisions_request(
|
||||
model=model, request=request, custom_llm_provider=custom_llm_provider
|
||||
)
|
||||
data: Final = dict(body)
|
||||
if logging_obj is not None:
|
||||
logging_obj.pre_call(
|
||||
input=data,
|
||||
|
|
@ -1514,19 +1511,19 @@ class BaseLLMHTTPHandler:
|
|||
custom_llm_provider: str,
|
||||
logging_obj: LiteLLMLoggingObj | None,
|
||||
provider_config: BaseDecisionsConfig,
|
||||
request: DecisionsRequest,
|
||||
request: DecisionsIRRequest,
|
||||
body: Mapping[str, object],
|
||||
api_base: str,
|
||||
api_key: str | None,
|
||||
headers: Mapping[str, str],
|
||||
timeout: float | httpx.Timeout | None,
|
||||
client: HTTPHandler | None = None,
|
||||
) -> DecisionsResponse:
|
||||
) -> DecisionsIRResponse:
|
||||
url, outbound_headers, data = self._prepare_decisions_request(
|
||||
model=model,
|
||||
custom_llm_provider=custom_llm_provider,
|
||||
logging_obj=logging_obj,
|
||||
provider_config=provider_config,
|
||||
request=request,
|
||||
body=body,
|
||||
api_base=api_base,
|
||||
api_key=api_key,
|
||||
headers=headers,
|
||||
|
|
@ -1548,19 +1545,19 @@ class BaseLLMHTTPHandler:
|
|||
custom_llm_provider: str,
|
||||
logging_obj: LiteLLMLoggingObj | None,
|
||||
provider_config: BaseDecisionsConfig,
|
||||
request: DecisionsRequest,
|
||||
request: DecisionsIRRequest,
|
||||
body: Mapping[str, object],
|
||||
api_base: str,
|
||||
api_key: str | None,
|
||||
headers: Mapping[str, str],
|
||||
timeout: float | httpx.Timeout | None,
|
||||
client: AsyncHTTPHandler | None = None,
|
||||
) -> DecisionsResponse:
|
||||
) -> DecisionsIRResponse:
|
||||
url, outbound_headers, data = self._prepare_decisions_request(
|
||||
model=model,
|
||||
custom_llm_provider=custom_llm_provider,
|
||||
logging_obj=logging_obj,
|
||||
provider_config=provider_config,
|
||||
request=request,
|
||||
body=body,
|
||||
api_base=api_base,
|
||||
api_key=api_key,
|
||||
headers=headers,
|
||||
|
|
|
|||
320
litellm/llms/openai/decisions/transformation.py
Normal file
320
litellm/llms/openai/decisions/transformation.py
Normal file
|
|
@ -0,0 +1,320 @@
|
|||
from collections.abc import Mapping, Sequence
|
||||
from typing import Final
|
||||
|
||||
from pydantic import TypeAdapter
|
||||
from typing_extensions import assert_never
|
||||
|
||||
import litellm
|
||||
from litellm.llms.base_llm.decisions.transformation import BaseDecisionsConfig, decisions_text
|
||||
from litellm.secret_managers.main import get_secret_str
|
||||
from litellm.types.decisions import (
|
||||
DecisionsIRAnswer,
|
||||
DecisionsIRChoiceAnswer,
|
||||
DecisionsIRChoiceOption,
|
||||
DecisionsIRChoiceProbability,
|
||||
DecisionsIRChoiceQuestion,
|
||||
DecisionsIRMessages,
|
||||
DecisionsIRPredicateAnswer,
|
||||
DecisionsIRPredicateQuestion,
|
||||
DecisionsIRQuestion,
|
||||
DecisionsIRRefusal,
|
||||
DecisionsIRRequest,
|
||||
DecisionsIRResponse,
|
||||
DecisionsIRScoreAnswer,
|
||||
DecisionsIRScoreLevel,
|
||||
DecisionsIRScoreProbability,
|
||||
DecisionsIRScoreQuestion,
|
||||
DecisionsIRState,
|
||||
DecisionsIRUsage,
|
||||
DecisionsJSON,
|
||||
OpenAIChoiceAnswer,
|
||||
OpenAIChoiceProbability,
|
||||
OpenAIChoiceQuestion,
|
||||
OpenAIDecisionAnswer,
|
||||
OpenAIDecisionInput,
|
||||
OpenAIDecisionInputTokensDetails,
|
||||
OpenAIDecisionOutputTokensDetails,
|
||||
OpenAIDecisionQuestion,
|
||||
OpenAIDecisionRequestBody,
|
||||
OpenAIDecisionResponse,
|
||||
OpenAIDecisionUsage,
|
||||
OpenAIPredicateAnswer,
|
||||
OpenAIPredicateQuestion,
|
||||
OpenAIRefusalAnswer,
|
||||
OpenAIScoreAnswer,
|
||||
OpenAIScoreProbability,
|
||||
OpenAIScoreQuestion,
|
||||
UnsupportedDecisionsRequest,
|
||||
systemone_choice_key,
|
||||
)
|
||||
|
||||
_OPENAI_RESPONSE_ADAPTER: Final[TypeAdapter[OpenAIDecisionResponse]] = TypeAdapter(OpenAIDecisionResponse)
|
||||
_SINGLE_OPTION: Final = UnsupportedDecisionsRequest(
|
||||
reason="OpenAI needs at least 2 choices or levels on every choice or score question"
|
||||
)
|
||||
|
||||
|
||||
def _ir_input(decision_input: OpenAIDecisionInput) -> DecisionsIRState | DecisionsIRMessages:
|
||||
if isinstance(decision_input, str):
|
||||
return DecisionsIRState(state=decision_input)
|
||||
return DecisionsIRMessages(messages=tuple(decision_input))
|
||||
|
||||
|
||||
def _ir_question(question: OpenAIDecisionQuestion) -> DecisionsIRQuestion:
|
||||
match question:
|
||||
case OpenAIPredicateQuestion():
|
||||
return DecisionsIRPredicateQuestion(name=question.name, instructions=question.instructions)
|
||||
case OpenAIChoiceQuestion():
|
||||
return DecisionsIRChoiceQuestion(
|
||||
name=question.name,
|
||||
instructions=question.instructions,
|
||||
choices=tuple(
|
||||
DecisionsIRChoiceOption(value=option.value, description=option.description)
|
||||
for option in question.choices
|
||||
),
|
||||
)
|
||||
case OpenAIScoreQuestion():
|
||||
return DecisionsIRScoreQuestion(
|
||||
name=question.name,
|
||||
instructions=question.instructions,
|
||||
levels=tuple(
|
||||
DecisionsIRScoreLevel(label=level.label, description=level.description) for level in question.levels
|
||||
),
|
||||
)
|
||||
case _:
|
||||
assert_never(question)
|
||||
|
||||
|
||||
def openai_request_to_ir(request: OpenAIDecisionRequestBody) -> DecisionsIRRequest:
|
||||
return DecisionsIRRequest(
|
||||
input=_ir_input(request.input),
|
||||
questions=tuple(_ir_question(question) for question in request.questions),
|
||||
safety_identifier=request.safety_identifier,
|
||||
)
|
||||
|
||||
|
||||
def _text_field(key: str, value: DecisionsJSON | None) -> Mapping[str, str]:
|
||||
return {} if value is None else {key: decisions_text(value)}
|
||||
|
||||
|
||||
def _instructions(value: DecisionsJSON | None, default: str) -> str:
|
||||
return default if value is None else decisions_text(value)
|
||||
|
||||
|
||||
def _predicate_instructions(question: DecisionsIRPredicateQuestion) -> str:
|
||||
instructions: Final = () if question.instructions is None else (decisions_text(question.instructions),)
|
||||
criteria: Final = tuple(
|
||||
f"Answer {answer} when: {decisions_text(rule)}"
|
||||
for answer, rule in (question.criteria or {}).items()
|
||||
if rule is not None
|
||||
)
|
||||
return "\n\n".join((*instructions, *criteria))
|
||||
|
||||
|
||||
def _openai_question(question: DecisionsIRQuestion) -> Mapping[str, object]:
|
||||
match question:
|
||||
case DecisionsIRPredicateQuestion():
|
||||
return {
|
||||
"type": "predicate",
|
||||
**_text_field("name", question.name),
|
||||
"instructions": _predicate_instructions(question) or "Is this true of the input?",
|
||||
}
|
||||
case DecisionsIRChoiceQuestion():
|
||||
return {
|
||||
"type": "choice",
|
||||
**_text_field("name", question.name),
|
||||
"instructions": _instructions(question.instructions, "Which choice best fits the input?"),
|
||||
"choices": [
|
||||
{"value": option.value, **_text_field("description", option.description)}
|
||||
for option in question.choices
|
||||
],
|
||||
}
|
||||
case DecisionsIRScoreQuestion():
|
||||
return {
|
||||
"type": "score",
|
||||
**_text_field("name", question.name),
|
||||
"instructions": _instructions(question.instructions, "Which level best fits the input?"),
|
||||
"levels": [
|
||||
{"label": decisions_text(level.label), **_text_field("description", level.description)}
|
||||
for level in question.levels
|
||||
],
|
||||
}
|
||||
case _:
|
||||
assert_never(question)
|
||||
|
||||
|
||||
def _openai_input(decision_input: DecisionsIRState | DecisionsIRMessages) -> str | Sequence[Mapping[str, object]]:
|
||||
match decision_input:
|
||||
case DecisionsIRState():
|
||||
return decisions_text(decision_input.state)
|
||||
case DecisionsIRMessages():
|
||||
return [message.model_dump(mode="json", exclude_none=True) for message in decision_input.messages]
|
||||
case _:
|
||||
assert_never(decision_input)
|
||||
|
||||
|
||||
def _has_one_option(question: DecisionsIRQuestion) -> bool:
|
||||
match question:
|
||||
case DecisionsIRChoiceQuestion():
|
||||
return len(question.choices) < 2
|
||||
case DecisionsIRScoreQuestion():
|
||||
return len(question.levels) < 2
|
||||
case _:
|
||||
return False
|
||||
|
||||
|
||||
def ir_to_openai_request(model: str, request: DecisionsIRRequest) -> Mapping[str, object] | UnsupportedDecisionsRequest:
|
||||
if any(_has_one_option(question) for question in request.questions):
|
||||
return _SINGLE_OPTION
|
||||
return {
|
||||
"model": model,
|
||||
"input": _openai_input(request.input),
|
||||
"questions": [_openai_question(question) for question in request.questions],
|
||||
**_text_field("safety_identifier", request.safety_identifier),
|
||||
}
|
||||
|
||||
|
||||
def _level_label(question: DecisionsIRScoreQuestion, probability: OpenAIScoreProbability) -> DecisionsJSON:
|
||||
if 0 <= probability.value < len(question.levels):
|
||||
return question.levels[probability.value].label
|
||||
return probability.label
|
||||
|
||||
|
||||
def _ir_answer(question: DecisionsIRQuestion, answer: OpenAIDecisionAnswer | None) -> DecisionsIRAnswer:
|
||||
match question, answer:
|
||||
case DecisionsIRPredicateQuestion(), OpenAIPredicateAnswer():
|
||||
return DecisionsIRPredicateAnswer(probability=answer.probability)
|
||||
case DecisionsIRChoiceQuestion(), OpenAIChoiceAnswer():
|
||||
return DecisionsIRChoiceAnswer(
|
||||
choice=answer.choice,
|
||||
confidence=answer.confidence,
|
||||
probabilities=tuple(
|
||||
DecisionsIRChoiceProbability(value=item.value, probability=item.probability)
|
||||
for item in answer.probabilities
|
||||
),
|
||||
)
|
||||
case DecisionsIRScoreQuestion(), OpenAIScoreAnswer():
|
||||
return DecisionsIRScoreAnswer(
|
||||
score=answer.score,
|
||||
confidence=answer.confidence,
|
||||
probabilities=tuple(
|
||||
DecisionsIRScoreProbability(
|
||||
value=item.value, label=_level_label(question, item), probability=item.probability
|
||||
)
|
||||
for item in answer.probabilities
|
||||
),
|
||||
)
|
||||
case _:
|
||||
return DecisionsIRRefusal()
|
||||
|
||||
|
||||
def openai_response_to_ir(response: OpenAIDecisionResponse, request: DecisionsIRRequest) -> DecisionsIRResponse:
|
||||
answers: Final = response.answers
|
||||
return DecisionsIRResponse(
|
||||
model=response.model,
|
||||
answers=tuple(
|
||||
_ir_answer(question, answers[index] if index < len(answers) else None)
|
||||
for index, question in enumerate(request.questions)
|
||||
),
|
||||
usage=DecisionsIRUsage(
|
||||
input_tokens=response.usage.input_tokens,
|
||||
output_tokens=response.usage.output_tokens,
|
||||
cached_tokens=response.usage.input_tokens_details.cached_tokens,
|
||||
cache_write_tokens=response.usage.input_tokens_details.cache_write_tokens,
|
||||
reasoning_tokens=response.usage.output_tokens_details.reasoning_tokens,
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
def _openai_choice_answer(question: DecisionsIRChoiceQuestion, answer: DecisionsIRChoiceAnswer) -> OpenAIChoiceAnswer:
|
||||
probabilities: Final = {systemone_choice_key(item.value): item.probability for item in answer.probabilities}
|
||||
return OpenAIChoiceAnswer(
|
||||
name=question.name,
|
||||
choice=answer.choice,
|
||||
probabilities=tuple(
|
||||
OpenAIChoiceProbability(
|
||||
value=option.value, probability=probabilities.get(systemone_choice_key(option.value), 0.0)
|
||||
)
|
||||
for option in question.choices
|
||||
),
|
||||
confidence=answer.confidence,
|
||||
)
|
||||
|
||||
|
||||
def _openai_score_answer(question: DecisionsIRScoreQuestion, answer: DecisionsIRScoreAnswer) -> OpenAIScoreAnswer:
|
||||
probabilities: Final = {item.value: item.probability for item in answer.probabilities}
|
||||
return OpenAIScoreAnswer(
|
||||
name=question.name,
|
||||
score=answer.score,
|
||||
probabilities=tuple(
|
||||
OpenAIScoreProbability(
|
||||
value=index, label=decisions_text(level.label), probability=probabilities.get(index, 0.0)
|
||||
)
|
||||
for index, level in enumerate(question.levels)
|
||||
),
|
||||
confidence=answer.confidence,
|
||||
)
|
||||
|
||||
|
||||
def _openai_answer(question: DecisionsIRQuestion, answer: DecisionsIRAnswer) -> OpenAIDecisionAnswer:
|
||||
match question, answer:
|
||||
case DecisionsIRPredicateQuestion(), DecisionsIRPredicateAnswer():
|
||||
return OpenAIPredicateAnswer(name=question.name, probability=answer.probability)
|
||||
case DecisionsIRChoiceQuestion(), DecisionsIRChoiceAnswer():
|
||||
return _openai_choice_answer(question, answer)
|
||||
case DecisionsIRScoreQuestion(), DecisionsIRScoreAnswer():
|
||||
return _openai_score_answer(question, answer)
|
||||
case _:
|
||||
return OpenAIRefusalAnswer(name=question.name)
|
||||
|
||||
|
||||
def ir_to_openai_response(
|
||||
response: DecisionsIRResponse, request: DecisionsIRRequest, requested_model: str
|
||||
) -> OpenAIDecisionResponse:
|
||||
return OpenAIDecisionResponse(
|
||||
model=response.model or requested_model,
|
||||
answers=tuple(
|
||||
_openai_answer(question, answer)
|
||||
for question, answer in zip(request.questions, response.answers, strict=True)
|
||||
),
|
||||
usage=OpenAIDecisionUsage(
|
||||
input_tokens=response.usage.input_tokens,
|
||||
input_tokens_details=OpenAIDecisionInputTokensDetails(
|
||||
cached_tokens=response.usage.cached_tokens,
|
||||
cache_write_tokens=response.usage.cache_write_tokens,
|
||||
),
|
||||
output_tokens=response.usage.output_tokens,
|
||||
output_tokens_details=OpenAIDecisionOutputTokensDetails(reasoning_tokens=response.usage.reasoning_tokens),
|
||||
total_tokens=response.usage.input_tokens + response.usage.output_tokens,
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
class OpenAIDecisionsConfig(BaseDecisionsConfig):
|
||||
path = "/v1/decisions"
|
||||
|
||||
def get_default_api_base(self) -> str | None:
|
||||
return "https://api.openai.com"
|
||||
|
||||
def resolve_api_base(self, api_base: str | None) -> str | None:
|
||||
return (
|
||||
api_base
|
||||
or litellm.api_base
|
||||
or get_secret_str("OPENAI_BASE_URL")
|
||||
or get_secret_str("OPENAI_API_BASE")
|
||||
or self.get_default_api_base()
|
||||
)
|
||||
|
||||
def resolve_api_key(self, api_key: str | None) -> str | None:
|
||||
return api_key or litellm.api_key or litellm.openai_key or get_secret_str("OPENAI_API_KEY")
|
||||
|
||||
def transform_decisions_request(
|
||||
self,
|
||||
model: str,
|
||||
request: DecisionsIRRequest,
|
||||
custom_llm_provider: str,
|
||||
) -> Mapping[str, object] | UnsupportedDecisionsRequest:
|
||||
return ir_to_openai_request(self.request_model(model), request)
|
||||
|
||||
def parse_response(self, payload: object, request: DecisionsIRRequest) -> DecisionsIRResponse:
|
||||
return openai_response_to_ir(_OPENAI_RESPONSE_ADAPTER.validate_python(payload), request)
|
||||
|
|
@ -34527,7 +34527,8 @@
|
|||
"supported_endpoints": [
|
||||
"/v1/chat/completions",
|
||||
"/v1/batch",
|
||||
"/v1/responses"
|
||||
"/v1/responses",
|
||||
"/v1/decisions"
|
||||
],
|
||||
"supported_modalities": [
|
||||
"text",
|
||||
|
|
|
|||
|
|
@ -5,15 +5,10 @@ from fastapi import APIRouter, Depends, Request, Response
|
|||
from fastapi.responses import ORJSONResponse # pyright: ignore[reportDeprecated] # required endpoint contract
|
||||
from pydantic import TypeAdapter, ValidationError
|
||||
|
||||
from litellm.decisions.openai_transformation import to_openai_response, to_systemone_request
|
||||
from litellm.exceptions import BadRequestError
|
||||
from litellm.proxy.auth.user_api_key_auth import UserAPIKeyAuth, user_api_key_auth
|
||||
from litellm.proxy.common_request_processing import (
|
||||
ProxyBaseLLMRequestProcessing,
|
||||
attach_guardrail_information,
|
||||
include_guardrail_response_requested,
|
||||
)
|
||||
from litellm.types.decisions import DecisionsRequestBody, DecisionsResponse, OpenAIDecisionRequestBody
|
||||
from litellm.proxy.common_request_processing import ProxyBaseLLMRequestProcessing
|
||||
from litellm.types.decisions import DecisionsRequestBody, OpenAIDecisionRequestBody
|
||||
|
||||
router: Final = APIRouter()
|
||||
_REQUEST_DATA_ADAPTER: Final[TypeAdapter[dict[str, object]]] = TypeAdapter(dict[str, object])
|
||||
|
|
@ -21,7 +16,6 @@ _DECISIONS_REQUEST_BODY_ADAPTER: Final[TypeAdapter[DecisionsRequestBody]] = Type
|
|||
_OPENAI_DECISION_REQUEST_BODY_ADAPTER: Final[TypeAdapter[OpenAIDecisionRequestBody]] = TypeAdapter(
|
||||
OpenAIDecisionRequestBody
|
||||
)
|
||||
_DECISIONS_RESPONSE_ADAPTER: Final[TypeAdapter[DecisionsResponse]] = TypeAdapter(DecisionsResponse)
|
||||
_GENERAL_SETTINGS_ADAPTER: Final[TypeAdapter[dict[str, object]]] = TypeAdapter(dict[str, object])
|
||||
_OPTIONAL_STRING_ADAPTER: Final[TypeAdapter[str | None]] = TypeAdapter(str | None)
|
||||
_OPTIONAL_FLOAT_ADAPTER: Final[TypeAdapter[float | None]] = TypeAdapter(float | None)
|
||||
|
|
@ -52,12 +46,11 @@ async def _request_data(request: Request, user_api_key_dict: UserAPIKeyAuth) ->
|
|||
raise await _invalid_request(raw_data={}, error=error, user_api_key_dict=user_api_key_dict)
|
||||
|
||||
|
||||
async def _process_systemone(
|
||||
async def _process_decisions(
|
||||
request: Request,
|
||||
fastapi_response: Response,
|
||||
user_api_key_dict: UserAPIKeyAuth,
|
||||
raw_data: Mapping[str, object],
|
||||
openai_body: OpenAIDecisionRequestBody | None,
|
||||
body_adapter: TypeAdapter[DecisionsRequestBody] | TypeAdapter[OpenAIDecisionRequestBody],
|
||||
) -> object:
|
||||
from litellm.proxy.proxy_server import (
|
||||
general_settings as proxy_general_settings,
|
||||
|
|
@ -80,15 +73,18 @@ async def _process_systemone(
|
|||
user_temperature as proxy_user_temperature,
|
||||
)
|
||||
|
||||
data: Final = dict(raw_data if openai_body is None else to_systemone_request(raw_data, openai_body))
|
||||
data: Final = await _request_data(request, user_api_key_dict)
|
||||
try:
|
||||
body_adapter.validate_python(data)
|
||||
except ValidationError as error:
|
||||
raise await _invalid_request(raw_data=data, error=error, user_api_key_dict=user_api_key_dict)
|
||||
general_settings: Final = _GENERAL_SETTINGS_ADAPTER.validate_python(proxy_general_settings)
|
||||
user_api_base: Final = _OPTIONAL_STRING_ADAPTER.validate_python(proxy_user_api_base)
|
||||
user_model: Final = _OPTIONAL_STRING_ADAPTER.validate_python(proxy_user_model)
|
||||
user_temperature: Final = _OPTIONAL_FLOAT_ADAPTER.validate_python(proxy_user_temperature)
|
||||
processor: Final = ProxyBaseLLMRequestProcessing(data=data)
|
||||
try:
|
||||
_DECISIONS_REQUEST_BODY_ADAPTER.validate_python(data)
|
||||
result: Final[object] = await processor.base_process_llm_request(
|
||||
return await processor.base_process_llm_request(
|
||||
request=request,
|
||||
fastapi_response=fastapi_response,
|
||||
user_api_key_dict=user_api_key_dict,
|
||||
|
|
@ -106,21 +102,6 @@ async def _process_systemone(
|
|||
user_api_base=user_api_base,
|
||||
version=version,
|
||||
)
|
||||
if openai_body is None or isinstance(result, Response):
|
||||
return result
|
||||
openai_response: Final = to_openai_response(
|
||||
_DECISIONS_RESPONSE_ADAPTER.validate_python(result),
|
||||
openai_body.questions,
|
||||
str(data.get("model", "")),
|
||||
)
|
||||
request_data: Final = _REQUEST_DATA_ADAPTER.validate_python(
|
||||
processor.data # pyright: ignore[reportUnknownMemberType] # ProxyBaseLLMRequestProcessing.data is a bare dict
|
||||
)
|
||||
if include_guardrail_response_requested(request_data):
|
||||
return attach_guardrail_information(response=openai_response, request_data=request_data)
|
||||
return openai_response
|
||||
except ValidationError as error:
|
||||
raise await _invalid_request(raw_data=data, error=error, user_api_key_dict=user_api_key_dict)
|
||||
except Exception as error:
|
||||
raise await processor.handle_llm_api_exception(
|
||||
e=error,
|
||||
|
|
@ -147,12 +128,11 @@ async def systemone(
|
|||
fastapi_response: Response,
|
||||
user_api_key_dict: Annotated[UserAPIKeyAuth, Depends(user_api_key_auth)],
|
||||
):
|
||||
return await _process_systemone(
|
||||
return await _process_decisions(
|
||||
request=request,
|
||||
fastapi_response=fastapi_response,
|
||||
user_api_key_dict=user_api_key_dict,
|
||||
raw_data=await _request_data(request, user_api_key_dict),
|
||||
openai_body=None,
|
||||
body_adapter=_DECISIONS_REQUEST_BODY_ADAPTER,
|
||||
)
|
||||
|
||||
|
||||
|
|
@ -173,15 +153,9 @@ async def decisions(
|
|||
fastapi_response: Response,
|
||||
user_api_key_dict: Annotated[UserAPIKeyAuth, Depends(user_api_key_auth)],
|
||||
):
|
||||
raw_data: Final = await _request_data(request, user_api_key_dict)
|
||||
try:
|
||||
openai_body: Final = _OPENAI_DECISION_REQUEST_BODY_ADAPTER.validate_python(raw_data)
|
||||
except ValidationError as error:
|
||||
raise await _invalid_request(raw_data=raw_data, error=error, user_api_key_dict=user_api_key_dict)
|
||||
return await _process_systemone(
|
||||
return await _process_decisions(
|
||||
request=request,
|
||||
fastapi_response=fastapi_response,
|
||||
user_api_key_dict=user_api_key_dict,
|
||||
raw_data=raw_data,
|
||||
openai_body=openai_body,
|
||||
body_adapter=_OPENAI_DECISION_REQUEST_BODY_ADAPTER,
|
||||
)
|
||||
|
|
|
|||
|
|
@ -1,4 +1,6 @@
|
|||
from collections.abc import Mapping, Sequence
|
||||
from dataclasses import dataclass, field
|
||||
from types import MappingProxyType
|
||||
from typing import Annotated, Final, Literal, TypeAlias
|
||||
|
||||
from pydantic import ConfigDict, Field, PrivateAttr, model_validator, with_config
|
||||
|
|
@ -110,17 +112,13 @@ DecisionAnswer: TypeAlias = Annotated[
|
|||
class DecisionsUsage(LiteLLMPydanticObjectBase):
|
||||
input_tokens: int = 0
|
||||
output_tokens: int = 0
|
||||
cached_tokens: Annotated[int, Field(exclude=True)] = 0
|
||||
cache_write_tokens: Annotated[int, Field(exclude=True)] = 0
|
||||
|
||||
model_config = ConfigDict(extra="allow", frozen=True)
|
||||
|
||||
|
||||
class DecisionsResponse(LiteLLMPydanticObjectBase):
|
||||
model: str | None = None
|
||||
answers: Mapping[str, DecisionAnswer]
|
||||
usage: DecisionsUsage | None = None
|
||||
|
||||
model_config = ConfigDict(extra="allow", frozen=True)
|
||||
|
||||
class _HiddenParamsResponse(LiteLLMPydanticObjectBase):
|
||||
_hidden_params: dict[str, object] = PrivateAttr(default_factory=dict)
|
||||
|
||||
@property
|
||||
|
|
@ -131,6 +129,14 @@ class DecisionsResponse(LiteLLMPydanticObjectBase):
|
|||
self._hidden_params.update(params)
|
||||
|
||||
|
||||
class DecisionsResponse(_HiddenParamsResponse):
|
||||
model: str | None = None
|
||||
answers: Mapping[str, DecisionAnswer]
|
||||
usage: DecisionsUsage | None = None
|
||||
|
||||
model_config = ConfigDict(extra="allow", frozen=True)
|
||||
|
||||
|
||||
class OpenAIDecisionInputText(LiteLLMPydanticObjectBase):
|
||||
type: Literal["input_text"]
|
||||
text: str
|
||||
|
|
@ -138,14 +144,31 @@ class OpenAIDecisionInputText(LiteLLMPydanticObjectBase):
|
|||
model_config = ConfigDict(extra="forbid", frozen=True)
|
||||
|
||||
|
||||
class OpenAIDecisionInputImage(LiteLLMPydanticObjectBase):
|
||||
type: Literal["input_image"]
|
||||
image_url: str
|
||||
detail: str | None = None
|
||||
|
||||
model_config = ConfigDict(extra="forbid", frozen=True)
|
||||
|
||||
|
||||
OpenAIDecisionContentPart: TypeAlias = Annotated[
|
||||
OpenAIDecisionInputText | OpenAIDecisionInputImage,
|
||||
Field(discriminator="type"),
|
||||
]
|
||||
|
||||
|
||||
class OpenAIDecisionInputMessage(LiteLLMPydanticObjectBase):
|
||||
role: Literal["user"] = "user"
|
||||
type: Literal["message"] = "message"
|
||||
content: str | Sequence[OpenAIDecisionInputText]
|
||||
content: str | Sequence[OpenAIDecisionContentPart]
|
||||
|
||||
model_config = ConfigDict(extra="forbid", frozen=True)
|
||||
|
||||
|
||||
OpenAIDecisionInput: TypeAlias = str | Sequence[OpenAIDecisionInputMessage]
|
||||
|
||||
|
||||
class OpenAIPredicateQuestion(LiteLLMPydanticObjectBase):
|
||||
type: Literal["predicate"]
|
||||
name: str | None = None
|
||||
|
|
@ -206,7 +229,7 @@ OpenAIDecisionQuestion: TypeAlias = Annotated[
|
|||
|
||||
|
||||
class OpenAIDecisionRequestBody(LiteLLMPydanticObjectBase):
|
||||
input: str | Sequence[OpenAIDecisionInputMessage]
|
||||
input: OpenAIDecisionInput
|
||||
questions: Annotated[Sequence[OpenAIDecisionQuestion], Field(min_length=1, max_length=MAX_DECISION_QUESTIONS)]
|
||||
safety_identifier: str | None = None
|
||||
|
||||
|
|
@ -263,7 +286,10 @@ class OpenAIRefusalAnswer(LiteLLMPydanticObjectBase):
|
|||
model_config = ConfigDict(frozen=True)
|
||||
|
||||
|
||||
OpenAIDecisionAnswer: TypeAlias = OpenAIPredicateAnswer | OpenAIChoiceAnswer | OpenAIScoreAnswer | OpenAIRefusalAnswer
|
||||
OpenAIDecisionAnswer: TypeAlias = Annotated[
|
||||
OpenAIPredicateAnswer | OpenAIChoiceAnswer | OpenAIScoreAnswer | OpenAIRefusalAnswer,
|
||||
Field(discriminator="type"),
|
||||
]
|
||||
|
||||
|
||||
class OpenAIDecisionInputTokensDetails(LiteLLMPydanticObjectBase):
|
||||
|
|
@ -289,9 +315,136 @@ class OpenAIDecisionUsage(LiteLLMPydanticObjectBase):
|
|||
model_config = ConfigDict(frozen=True)
|
||||
|
||||
|
||||
class OpenAIDecisionResponse(LiteLLMPydanticObjectBase):
|
||||
class OpenAIDecisionResponse(_HiddenParamsResponse):
|
||||
model: str
|
||||
answers: tuple[OpenAIDecisionAnswer, ...]
|
||||
usage: OpenAIDecisionUsage
|
||||
|
||||
model_config = ConfigDict(extra="allow", frozen=True)
|
||||
|
||||
|
||||
_NO_EXTRA: Final[Mapping[str, object]] = MappingProxyType({})
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class DecisionsIRState:
|
||||
state: DecisionsJSON
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class DecisionsIRMessages:
|
||||
messages: tuple[OpenAIDecisionInputMessage, ...]
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class DecisionsIRPredicateQuestion:
|
||||
name: str | None
|
||||
instructions: DecisionsJSON | None
|
||||
criteria: NoulCriteria | None = None
|
||||
extra: Mapping[str, object] = field(default_factory=lambda: _NO_EXTRA)
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class DecisionsIRChoiceOption:
|
||||
value: str | bool
|
||||
description: DecisionsJSON | None
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class DecisionsIRChoiceQuestion:
|
||||
name: str | None
|
||||
instructions: DecisionsJSON | None
|
||||
choices: tuple[DecisionsIRChoiceOption, ...]
|
||||
extra: Mapping[str, object] = field(default_factory=lambda: _NO_EXTRA)
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class DecisionsIRScoreLevel:
|
||||
label: DecisionsJSON
|
||||
description: str | None
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class DecisionsIRScoreQuestion:
|
||||
name: str | None
|
||||
instructions: DecisionsJSON | None
|
||||
levels: tuple[DecisionsIRScoreLevel, ...]
|
||||
extra: Mapping[str, object] = field(default_factory=lambda: _NO_EXTRA)
|
||||
|
||||
|
||||
DecisionsIRQuestion: TypeAlias = DecisionsIRPredicateQuestion | DecisionsIRChoiceQuestion | DecisionsIRScoreQuestion
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class DecisionsIRRequest:
|
||||
input: DecisionsIRState | DecisionsIRMessages
|
||||
questions: tuple[DecisionsIRQuestion, ...]
|
||||
safety_identifier: str | None = None
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class DecisionsIRPredicateAnswer:
|
||||
probability: float
|
||||
extra: Mapping[str, object] = field(default_factory=lambda: _NO_EXTRA)
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class DecisionsIRChoiceProbability:
|
||||
value: str | bool
|
||||
probability: float
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class DecisionsIRChoiceAnswer:
|
||||
choice: str | bool
|
||||
confidence: float
|
||||
probabilities: tuple[DecisionsIRChoiceProbability, ...]
|
||||
extra: Mapping[str, object] = field(default_factory=lambda: _NO_EXTRA)
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class DecisionsIRScoreProbability:
|
||||
value: int
|
||||
label: DecisionsJSON
|
||||
probability: float
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class DecisionsIRScoreAnswer:
|
||||
score: float
|
||||
confidence: float
|
||||
probabilities: tuple[DecisionsIRScoreProbability, ...]
|
||||
extra: Mapping[str, object] = field(default_factory=lambda: _NO_EXTRA)
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class DecisionsIRRefusal:
|
||||
pass
|
||||
|
||||
|
||||
DecisionsIRAnswer: TypeAlias = (
|
||||
DecisionsIRPredicateAnswer | DecisionsIRChoiceAnswer | DecisionsIRScoreAnswer | DecisionsIRRefusal
|
||||
)
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class DecisionsIRUsage:
|
||||
input_tokens: int = 0
|
||||
output_tokens: int = 0
|
||||
cached_tokens: int = 0
|
||||
cache_write_tokens: int = 0
|
||||
reasoning_tokens: int = 0
|
||||
extra: Mapping[str, object] = field(default_factory=lambda: _NO_EXTRA)
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class DecisionsIRResponse:
|
||||
model: str | None
|
||||
answers: tuple[DecisionsIRAnswer, ...]
|
||||
usage: DecisionsIRUsage
|
||||
extra: Mapping[str, object] = field(default_factory=lambda: _NO_EXTRA)
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class UnsupportedDecisionsRequest:
|
||||
reason: str
|
||||
|
|
|
|||
|
|
@ -1219,8 +1219,8 @@ def function_setup(
|
|||
else search_query
|
||||
)
|
||||
elif call_type in (CallTypes.decisions.value, CallTypes.adecisions.value):
|
||||
decisions_state: Final = args[1] if len(args) > 1 else kwargs.get("state", "")
|
||||
messages = decisions_state if isinstance(decisions_state, str) else json.dumps(decisions_state)
|
||||
decisions_state: Final = args[1] if len(args) > 1 else kwargs.get("state") or kwargs.get("input") or ""
|
||||
messages = decisions_state if isinstance(decisions_state, str) else json.dumps(decisions_state, default=str)
|
||||
elif call_type in (CallTypes.image_edit.value, CallTypes.aimage_edit.value):
|
||||
messages = args[1] if len(args) > 1 else kwargs.get("prompt")
|
||||
elif call_type in (CallTypes.ocr.value, CallTypes.aocr.value):
|
||||
|
|
@ -8974,6 +8974,8 @@ class ProviderConfigManager:
|
|||
return litellm.CloudflareDecisionsConfig()
|
||||
if provider == LlmProviders.STRANDS_DECIDER:
|
||||
return litellm.StrandsDeciderDecisionsConfig()
|
||||
if provider == LlmProviders.OPENAI:
|
||||
return litellm.OpenAIDecisionsConfig()
|
||||
return None
|
||||
|
||||
@staticmethod
|
||||
|
|
|
|||
|
|
@ -34527,7 +34527,8 @@
|
|||
"supported_endpoints": [
|
||||
"/v1/chat/completions",
|
||||
"/v1/batch",
|
||||
"/v1/responses"
|
||||
"/v1/responses",
|
||||
"/v1/decisions"
|
||||
],
|
||||
"supported_modalities": [
|
||||
"text",
|
||||
|
|
|
|||
|
|
@ -18,6 +18,7 @@ from litellm.types.decisions import (
|
|||
DecisionsResponse,
|
||||
DecisionsUsage,
|
||||
NoulAnswer,
|
||||
OpenAIDecisionResponse,
|
||||
ScoreAnswer,
|
||||
)
|
||||
|
||||
|
|
@ -30,6 +31,8 @@ _QUESTIONS: Final[Mapping[str, object]] = MappingProxyType(
|
|||
)
|
||||
_INPUT_TOKENS: Final[int] = 367
|
||||
_OUTPUT_TOKENS: Final[int] = 3
|
||||
_CACHED_TOKENS: Final[int] = 256
|
||||
_CACHE_WRITE_TOKENS: Final[int] = 64
|
||||
_RESPONSE: Final[Mapping[str, object]] = {
|
||||
"model": "jev-1.13",
|
||||
"answers": {
|
||||
|
|
@ -80,6 +83,21 @@ _PROVIDERS: Final[tuple[tuple[str, str, str, str], ...]] = (
|
|||
),
|
||||
)
|
||||
|
||||
_OPENAI_RESPONSE: Final[Mapping[str, object]] = {
|
||||
"model": "gpt-6-luna",
|
||||
"answers": [
|
||||
{"type": "predicate", "name": "is_defect", "probability": 0.9},
|
||||
{"type": "refusal", "name": "sentiment"},
|
||||
],
|
||||
"usage": {
|
||||
"input_tokens": _INPUT_TOKENS,
|
||||
"input_tokens_details": {"cached_tokens": _CACHED_TOKENS, "cache_write_tokens": _CACHE_WRITE_TOKENS},
|
||||
"output_tokens": _OUTPUT_TOKENS,
|
||||
"output_tokens_details": {"reasoning_tokens": 0},
|
||||
"total_tokens": _INPUT_TOKENS + _OUTPUT_TOKENS,
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
class _RecordingLogger(CustomLogger):
|
||||
def __init__(self) -> None:
|
||||
|
|
@ -265,6 +283,46 @@ def test_decisions_cost_uses_litellm_token_pricing() -> None:
|
|||
assert cost == pytest.approx(expected_cost)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"response",
|
||||
(
|
||||
DecisionsResponse(
|
||||
model="jev-latest",
|
||||
answers={},
|
||||
usage=DecisionsUsage(
|
||||
input_tokens=_INPUT_TOKENS,
|
||||
output_tokens=_OUTPUT_TOKENS,
|
||||
cached_tokens=_CACHED_TOKENS,
|
||||
cache_write_tokens=_CACHE_WRITE_TOKENS,
|
||||
),
|
||||
),
|
||||
OpenAIDecisionResponse.model_validate(_OPENAI_RESPONSE),
|
||||
),
|
||||
ids=("systemone", "openai"),
|
||||
)
|
||||
def test_custom_token_pricing_bills_cached_decisions_input_tokens_once(
|
||||
response: DecisionsResponse | OpenAIDecisionResponse,
|
||||
) -> None:
|
||||
cost: Final = litellm.completion_cost(
|
||||
completion_response=response,
|
||||
model="gpt-6-luna",
|
||||
custom_llm_provider="openai",
|
||||
custom_cost_per_token={
|
||||
"input_cost_per_token": 1.0,
|
||||
"output_cost_per_token": 2.0,
|
||||
"cache_read_input_token_cost": 0.1,
|
||||
"cache_creation_input_token_cost": 1.25,
|
||||
},
|
||||
)
|
||||
|
||||
assert cost == pytest.approx(
|
||||
(_INPUT_TOKENS - _CACHED_TOKENS - _CACHE_WRITE_TOKENS) * 1.0
|
||||
+ _CACHED_TOKENS * 0.1
|
||||
+ _CACHE_WRITE_TOKENS * 1.25
|
||||
+ _OUTPUT_TOKENS * 2.0
|
||||
)
|
||||
|
||||
|
||||
def test_decisions_response_hidden_params_getter_preserves_mutable_identity() -> None:
|
||||
response: Final = DecisionsResponse(model="decider", answers={}, usage=None)
|
||||
|
||||
|
|
@ -487,7 +545,7 @@ async def test_cloudflare_clef_resolves_model_and_response_envelope(
|
|||
"state": "review",
|
||||
"questions": {"is_defect": {"type": "noul", "instructions": "Is this a defect?"}},
|
||||
}
|
||||
assert response.answers == DecisionsResponse.model_validate(_RESPONSE).answers
|
||||
assert response.answers == {"is_defect": NoulAnswer(type="noul", noul=0.9)}
|
||||
assert response._hidden_params["model"] == "cloudflare/@cf/cloudflare/clef"
|
||||
|
||||
|
||||
|
|
@ -673,3 +731,194 @@ async def test_strands_decider_provider_resolution_and_router_dispatch(
|
|||
assert provider_resolution[:2] == ("strands-decider-2B-hobson-v19", "strands_decider")
|
||||
assert route.called
|
||||
assert response.model == _STRANDS_RESPONSE["model"]
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
@pytest.mark.parametrize(
|
||||
("api_base", "url"),
|
||||
(
|
||||
(None, "https://api.openai.com/v1/decisions"),
|
||||
("https://gateway.example/v1", "https://gateway.example/v1/decisions"),
|
||||
),
|
||||
)
|
||||
async def test_openai_decisions_translate_systemone_to_the_openai_wire_contract_and_back(
|
||||
api_base: str | None,
|
||||
url: str,
|
||||
monkeypatch: pytest.MonkeyPatch,
|
||||
respx_mock: respx.MockRouter,
|
||||
) -> None:
|
||||
monkeypatch.setattr(litellm, "api_base", None)
|
||||
monkeypatch.delenv("OPENAI_BASE_URL", raising=False)
|
||||
monkeypatch.delenv("OPENAI_API_BASE", raising=False)
|
||||
route: Final = respx_mock.post(url).respond(json=_OPENAI_RESPONSE)
|
||||
|
||||
response: Final = await litellm.adecisions(
|
||||
model="openai/gpt-6-luna",
|
||||
state="The package arrived broken.",
|
||||
questions={
|
||||
"is_defect": {"type": "noul", "instructions": "Is this a defect?"},
|
||||
"sentiment": {
|
||||
"type": "choice",
|
||||
"instructions": "How does the customer feel?",
|
||||
"criteria": {"positive": None, "negative": "unhappy"},
|
||||
},
|
||||
},
|
||||
api_key="caller-key",
|
||||
api_base=api_base,
|
||||
)
|
||||
|
||||
assert route.called
|
||||
request: Final = respx_mock.calls[0].request
|
||||
assert request.headers["authorization"] == "Bearer caller-key"
|
||||
assert json.loads(request.content) == {
|
||||
"model": "gpt-6-luna",
|
||||
"input": "The package arrived broken.",
|
||||
"questions": [
|
||||
{"type": "predicate", "name": "is_defect", "instructions": "Is this a defect?"},
|
||||
{
|
||||
"type": "choice",
|
||||
"name": "sentiment",
|
||||
"instructions": "How does the customer feel?",
|
||||
"choices": [{"value": "positive"}, {"value": "negative", "description": "unhappy"}],
|
||||
},
|
||||
],
|
||||
}
|
||||
assert response.answers == {"is_defect": NoulAnswer(type="noul", noul=0.9)}
|
||||
assert response.hidden_params["custom_llm_provider"] == "openai"
|
||||
luna_cost: Final = litellm.model_cost["gpt-6-luna"]
|
||||
expected_cost: Final = (
|
||||
(_INPUT_TOKENS - _CACHED_TOKENS - _CACHE_WRITE_TOKENS) * float(luna_cost["input_cost_per_token"])
|
||||
+ _CACHED_TOKENS * float(luna_cost["cache_read_input_token_cost"])
|
||||
+ _CACHE_WRITE_TOKENS * float(luna_cost["cache_creation_input_token_cost"])
|
||||
+ _OUTPUT_TOKENS * float(luna_cost["output_cost_per_token"])
|
||||
)
|
||||
assert expected_cost > 0
|
||||
assert litellm.completion_cost(completion_response=response) == pytest.approx(expected_cost)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("settings", "env", "url", "authorization"),
|
||||
(
|
||||
({"openai_key": "sdk-key"}, {}, "https://api.openai.com/v1/decisions", "Bearer sdk-key"),
|
||||
(
|
||||
{"api_key": "global-key", "openai_key": "sdk-key"},
|
||||
{"OPENAI_API_KEY": "env-key"},
|
||||
"https://api.openai.com/v1/decisions",
|
||||
"Bearer global-key",
|
||||
),
|
||||
(
|
||||
{},
|
||||
{"OPENAI_API_KEY": "env-key", "OPENAI_API_BASE": "https://legacy.example/v1"},
|
||||
"https://legacy.example/v1/decisions",
|
||||
"Bearer env-key",
|
||||
),
|
||||
(
|
||||
{"api_base": "https://sdk.example"},
|
||||
{"OPENAI_API_KEY": "env-key", "OPENAI_BASE_URL": "https://env.example"},
|
||||
"https://sdk.example/v1/decisions",
|
||||
"Bearer env-key",
|
||||
),
|
||||
),
|
||||
ids=("openai_key", "api_key_before_env", "openai_api_base_env", "api_base_before_env"),
|
||||
)
|
||||
def test_openai_decisions_use_the_same_settings_as_other_openai_calls(
|
||||
settings: Mapping[str, str],
|
||||
env: Mapping[str, str],
|
||||
url: str,
|
||||
authorization: str,
|
||||
monkeypatch: pytest.MonkeyPatch,
|
||||
respx_mock: respx.MockRouter,
|
||||
) -> None:
|
||||
for name in ("api_key", "openai_key", "api_base"):
|
||||
monkeypatch.setattr(litellm, name, settings.get(name))
|
||||
for name in ("OPENAI_API_KEY", "OPENAI_BASE_URL", "OPENAI_API_BASE"):
|
||||
monkeypatch.delenv(name, raising=False)
|
||||
for name, value in env.items():
|
||||
monkeypatch.setenv(name, value)
|
||||
route: Final = respx_mock.post(url).respond(json=_OPENAI_RESPONSE)
|
||||
|
||||
litellm.decisions(
|
||||
model="openai/gpt-6-luna",
|
||||
state="review",
|
||||
questions={"is_defect": {"type": "noul", "instructions": "Is this a defect?"}},
|
||||
)
|
||||
|
||||
assert route.call_count == 1
|
||||
assert route.calls[0].request.headers["authorization"] == authorization
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_openai_format_calls_to_a_systemone_provider_get_openai_format_answers(
|
||||
respx_mock: respx.MockRouter,
|
||||
) -> None:
|
||||
route: Final = respx_mock.post("https://api.typesafe.ai/v1/systemone").respond(json=_RESPONSE)
|
||||
|
||||
response: Final = await litellm.adecisions(
|
||||
model="typesafe/jev-1.13",
|
||||
input="review",
|
||||
questions=[
|
||||
{"type": "predicate", "name": "is_defect", "instructions": "Is this a defect?"},
|
||||
{
|
||||
"type": "choice",
|
||||
"name": "sentiment",
|
||||
"instructions": "Tone?",
|
||||
"choices": [{"value": "positive"}, {"value": "negative"}],
|
||||
},
|
||||
],
|
||||
api_key="caller-key",
|
||||
)
|
||||
|
||||
assert route.called
|
||||
assert tuple(json.loads(respx_mock.calls[0].request.content)["questions"]) == ("is_defect", "sentiment")
|
||||
assert isinstance(response, OpenAIDecisionResponse)
|
||||
assert [answer.model_dump(mode="json") for answer in response.answers] == [
|
||||
{"type": "predicate", "name": "is_defect", "probability": 0.9},
|
||||
{
|
||||
"type": "choice",
|
||||
"name": "sentiment",
|
||||
"choice": "positive",
|
||||
"probabilities": [{"value": "positive", "probability": 0.8}, {"value": "negative", "probability": 0.2}],
|
||||
"confidence": 0.8,
|
||||
},
|
||||
]
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
@pytest.mark.parametrize(
|
||||
("model", "request_kwargs", "message"),
|
||||
(
|
||||
(
|
||||
"typesafe/jev-1.13",
|
||||
{
|
||||
"input": [
|
||||
{
|
||||
"role": "user",
|
||||
"content": [{"type": "input_image", "image_url": "data:image/png;base64,AA=="}],
|
||||
}
|
||||
],
|
||||
"questions": [{"type": "predicate", "instructions": "Is this a defect?"}],
|
||||
},
|
||||
"cannot serve this request",
|
||||
),
|
||||
(
|
||||
"openai/gpt-6-luna",
|
||||
{
|
||||
"state": "review",
|
||||
"input": "review",
|
||||
"questions": [{"type": "predicate", "instructions": "Is this a defect?"}],
|
||||
},
|
||||
"not both",
|
||||
),
|
||||
),
|
||||
ids=("image_to_systemone_provider", "state_and_input"),
|
||||
)
|
||||
async def test_requests_a_provider_cannot_serve_are_rejected_before_http(
|
||||
respx_mock: respx.MockRouter,
|
||||
model: str,
|
||||
request_kwargs: Mapping[str, object],
|
||||
message: str,
|
||||
) -> None:
|
||||
with pytest.raises(litellm.BadRequestError, match=message):
|
||||
await litellm.adecisions(model=model, api_key="caller-key", **request_kwargs)
|
||||
|
||||
assert len(respx_mock.calls) == 0
|
||||
|
|
|
|||
|
|
@ -1,32 +0,0 @@
|
|||
from collections.abc import Mapping
|
||||
from typing import Final
|
||||
|
||||
import pytest
|
||||
from pydantic import TypeAdapter, ValidationError
|
||||
|
||||
from litellm.decisions.openai_transformation import to_systemone_request
|
||||
from litellm.types.decisions import MAX_DECISION_QUESTIONS, DecisionsRequestBody, OpenAIDecisionRequestBody
|
||||
|
||||
_OPENAI_BODY: Final[TypeAdapter[OpenAIDecisionRequestBody]] = TypeAdapter(OpenAIDecisionRequestBody)
|
||||
_SYSTEMONE_BODY: Final[TypeAdapter[DecisionsRequestBody]] = TypeAdapter(DecisionsRequestBody)
|
||||
|
||||
|
||||
def _openai_request(question_count: int) -> Mapping[str, object]:
|
||||
return {
|
||||
"model": "decider",
|
||||
"input": "The package arrived with a broken screen.",
|
||||
"questions": [{"type": "predicate", "instructions": f"Question {index}?"} for index in range(question_count)],
|
||||
}
|
||||
|
||||
|
||||
def test_the_largest_openai_request_accepted_translates_to_a_valid_systemone_request() -> None:
|
||||
raw: Final = _openai_request(MAX_DECISION_QUESTIONS)
|
||||
|
||||
translated: Final = _SYSTEMONE_BODY.validate_python(to_systemone_request(raw, _OPENAI_BODY.validate_python(raw)))
|
||||
|
||||
assert len(translated.questions) == MAX_DECISION_QUESTIONS
|
||||
|
||||
|
||||
def test_an_openai_request_with_more_questions_than_systemone_takes_is_rejected_before_translation() -> None:
|
||||
with pytest.raises(ValidationError, match="questions"):
|
||||
_OPENAI_BODY.validate_python(_openai_request(MAX_DECISION_QUESTIONS + 1))
|
||||
|
|
@ -0,0 +1,180 @@
|
|||
from collections.abc import Mapping
|
||||
from typing import Final
|
||||
|
||||
import pytest
|
||||
from pydantic import TypeAdapter
|
||||
|
||||
from litellm.llms.base_llm.decisions.transformation import (
|
||||
ir_to_systemone_request,
|
||||
ir_to_systemone_response,
|
||||
parse_systemone_response,
|
||||
systemone_request_to_ir,
|
||||
)
|
||||
from litellm.llms.openai.decisions.transformation import openai_request_to_ir
|
||||
from litellm.types.decisions import (
|
||||
MAX_DECISION_QUESTIONS,
|
||||
DecisionsIRRefusal,
|
||||
DecisionsIRRequest,
|
||||
DecisionsRequestBody,
|
||||
OpenAIDecisionRequestBody,
|
||||
UnsupportedDecisionsRequest,
|
||||
)
|
||||
|
||||
_SYSTEMONE_BODY: Final[TypeAdapter[DecisionsRequestBody]] = TypeAdapter(DecisionsRequestBody)
|
||||
_OPENAI_BODY: Final[TypeAdapter[OpenAIDecisionRequestBody]] = TypeAdapter(OpenAIDecisionRequestBody)
|
||||
|
||||
_SYSTEMONE_REQUEST: Final[Mapping[str, object]] = {
|
||||
"state": {"ticket": 1234, "text": "Screen cracked"},
|
||||
"questions": {
|
||||
"damaged": {
|
||||
"type": "noul",
|
||||
"instructions": "Is the item damaged?",
|
||||
"criteria": {"true": "Visible damage", "false": None},
|
||||
"provider_field": "dropped",
|
||||
},
|
||||
"rubric_only": {"type": "noul", "criteria": {"true": {"signal": "refund"}}},
|
||||
"action": {
|
||||
"type": "choice",
|
||||
"instructions": {"policy": "refund-v2"},
|
||||
"criteria": {"refund": "Within 30 days", "escalate": None},
|
||||
},
|
||||
"severity": {"type": "score", "criteria": ["minor", {"label": "major"}]},
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
def _openai_ir(raw: Mapping[str, object]) -> DecisionsIRRequest:
|
||||
return openai_request_to_ir(_OPENAI_BODY.validate_python(raw))
|
||||
|
||||
|
||||
def test_a_systemone_request_reaches_a_systemone_provider_unchanged() -> None:
|
||||
ir: Final = systemone_request_to_ir(_SYSTEMONE_BODY.validate_python(_SYSTEMONE_REQUEST))
|
||||
|
||||
assert ir_to_systemone_request("jev-latest", ir) == {"model": "jev-latest", **_SYSTEMONE_REQUEST}
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("names", "keys"),
|
||||
(
|
||||
(("damaged", "action"), ("damaged", "action")),
|
||||
(("damaged", None), ("0", "1")),
|
||||
(("damaged", "damaged"), ("0", "1")),
|
||||
),
|
||||
ids=("all_named", "one_unnamed", "duplicate_names"),
|
||||
)
|
||||
def test_openai_questions_are_keyed_by_name_only_when_every_name_is_unique(
|
||||
names: tuple[str | None, str | None], keys: tuple[str, str]
|
||||
) -> None:
|
||||
questions: Final = [
|
||||
{"type": "predicate", "instructions": f"Question {index}?", **({} if name is None else {"name": name})}
|
||||
for index, name in enumerate(names)
|
||||
]
|
||||
|
||||
body: Final = ir_to_systemone_request("jev-latest", _openai_ir({"input": "review", "questions": questions}))
|
||||
|
||||
assert tuple(_SYSTEMONE_BODY.validate_python(body).questions) == keys
|
||||
|
||||
|
||||
def test_openai_messages_become_systemone_state_text() -> None:
|
||||
ir: Final = _openai_ir(
|
||||
{
|
||||
"input": [
|
||||
{
|
||||
"role": "user",
|
||||
"content": [
|
||||
{"type": "input_text", "text": "The package arrived broken."},
|
||||
{"type": "input_text", "text": "I want a refund."},
|
||||
],
|
||||
},
|
||||
{"role": "user", "content": "Order 1234."},
|
||||
],
|
||||
"questions": [{"type": "predicate", "instructions": "Is this a defect?"}],
|
||||
}
|
||||
)
|
||||
|
||||
body: Final = ir_to_systemone_request("jev-latest", ir)
|
||||
|
||||
assert isinstance(body, Mapping)
|
||||
assert body["state"] == "The package arrived broken.\n\nI want a refund.\n\nOrder 1234."
|
||||
|
||||
|
||||
def test_image_input_is_unsupported_by_systemone_providers() -> None:
|
||||
ir: Final = _openai_ir(
|
||||
{
|
||||
"input": [
|
||||
{
|
||||
"role": "user",
|
||||
"content": [
|
||||
{"type": "input_text", "text": "Is the screen cracked?"},
|
||||
{"type": "input_image", "image_url": "data:image/png;base64,AA=="},
|
||||
],
|
||||
}
|
||||
],
|
||||
"questions": [{"type": "predicate", "instructions": "Is this a defect?"}],
|
||||
}
|
||||
)
|
||||
|
||||
assert isinstance(ir_to_systemone_request("jev-latest", ir), UnsupportedDecisionsRequest)
|
||||
|
||||
|
||||
def test_the_largest_openai_request_accepted_translates_to_a_valid_systemone_request() -> None:
|
||||
ir: Final = _openai_ir(
|
||||
{
|
||||
"input": "review",
|
||||
"questions": [
|
||||
{"type": "predicate", "instructions": f"Question {index}?"} for index in range(MAX_DECISION_QUESTIONS)
|
||||
],
|
||||
}
|
||||
)
|
||||
|
||||
translated: Final = _SYSTEMONE_BODY.validate_python(ir_to_systemone_request("jev-latest", ir))
|
||||
|
||||
assert len(translated.questions) == MAX_DECISION_QUESTIONS
|
||||
|
||||
|
||||
def test_a_systemone_response_reaches_the_caller_unchanged_with_provider_extras() -> None:
|
||||
payload: Final = {
|
||||
"model": "jev-latest",
|
||||
"answers": {
|
||||
"damaged": {"type": "noul", "noul": 0.95, "rationale": "crack visible"},
|
||||
"action": {
|
||||
"type": "choice",
|
||||
"choice": "refund",
|
||||
"confidence": 0.8,
|
||||
"probabilities": {"refund": 0.9, "escalate": 0.1},
|
||||
"calibrated": True,
|
||||
},
|
||||
"severity": {
|
||||
"type": "score",
|
||||
"score": 0.4,
|
||||
"confidence": 0.6,
|
||||
"legend": {"0": "minor", "1": {"label": "major"}},
|
||||
"probabilities": {"0": 0.6, "1": 0.4},
|
||||
"raw_logits": [0.1, 0.2],
|
||||
},
|
||||
},
|
||||
"usage": {"input_tokens": 383, "output_tokens": 2, "cost": 0.25},
|
||||
"latency_ms": 3722.17,
|
||||
}
|
||||
ir: Final = systemone_request_to_ir(_SYSTEMONE_BODY.validate_python(_SYSTEMONE_REQUEST))
|
||||
|
||||
response: Final = ir_to_systemone_response(parse_systemone_response(payload, ir), ir)
|
||||
|
||||
assert response.model_dump(mode="json") == payload
|
||||
|
||||
|
||||
def test_missing_or_mismatched_systemone_answers_are_refusals_left_out_of_the_response() -> None:
|
||||
ir: Final = systemone_request_to_ir(_SYSTEMONE_BODY.validate_python(_SYSTEMONE_REQUEST))
|
||||
payload: Final = {
|
||||
"model": "jev-latest",
|
||||
"answers": {
|
||||
"damaged": {"type": "noul", "noul": 0.95},
|
||||
"action": {"type": "noul", "noul": 0.5},
|
||||
},
|
||||
"usage": {"input_tokens": 10, "output_tokens": 1},
|
||||
}
|
||||
|
||||
parsed: Final = parse_systemone_response(payload, ir)
|
||||
|
||||
assert parsed.answers[1:] == (DecisionsIRRefusal(), DecisionsIRRefusal(), DecisionsIRRefusal())
|
||||
assert tuple(ir_to_systemone_response(parsed, ir).answers) == ("damaged",)
|
||||
0
tests/unit/llms/openai/decisions/__init__.py
Normal file
0
tests/unit/llms/openai/decisions/__init__.py
Normal file
|
|
@ -0,0 +1,256 @@
|
|||
from collections.abc import Mapping
|
||||
from typing import Final
|
||||
|
||||
import pytest
|
||||
from pydantic import TypeAdapter
|
||||
|
||||
from litellm.llms.base_llm.decisions.transformation import ir_to_systemone_response, systemone_request_to_ir
|
||||
from litellm.llms.openai.decisions.transformation import (
|
||||
OpenAIDecisionsConfig,
|
||||
ir_to_openai_request,
|
||||
ir_to_openai_response,
|
||||
openai_request_to_ir,
|
||||
)
|
||||
from litellm.types.decisions import (
|
||||
DecisionsIRRequest,
|
||||
DecisionsRequestBody,
|
||||
OpenAIDecisionRequestBody,
|
||||
UnsupportedDecisionsRequest,
|
||||
)
|
||||
|
||||
_SYSTEMONE_BODY: Final[TypeAdapter[DecisionsRequestBody]] = TypeAdapter(DecisionsRequestBody)
|
||||
_OPENAI_BODY: Final[TypeAdapter[OpenAIDecisionRequestBody]] = TypeAdapter(OpenAIDecisionRequestBody)
|
||||
|
||||
_OPENAI_REQUEST: Final[Mapping[str, object]] = {
|
||||
"input": [
|
||||
{
|
||||
"type": "message",
|
||||
"role": "user",
|
||||
"content": [
|
||||
{"type": "input_text", "text": "The screen is cracked."},
|
||||
{"type": "input_image", "image_url": "data:image/png;base64,AA==", "detail": "high"},
|
||||
],
|
||||
},
|
||||
{"type": "message", "role": "user", "content": "Order 1234."},
|
||||
],
|
||||
"questions": [
|
||||
{"type": "predicate", "name": "damaged", "instructions": "Is the item damaged?"},
|
||||
{
|
||||
"type": "choice",
|
||||
"instructions": "Should we refund?",
|
||||
"choices": [{"value": True, "description": "Refund now"}, {"value": "escalate"}],
|
||||
},
|
||||
{
|
||||
"type": "score",
|
||||
"name": "severity",
|
||||
"instructions": "How severe is it?",
|
||||
"levels": [{"label": "minor"}, {"label": "major", "description": "Product unusable"}],
|
||||
},
|
||||
{"type": "predicate", "name": "fraud", "instructions": "Is this fraud?"},
|
||||
],
|
||||
"safety_identifier": "end-user-1",
|
||||
}
|
||||
_PREDICATE_ANSWER: Final[Mapping[str, object]] = {"type": "predicate", "name": "damaged", "probability": 0.95}
|
||||
_CHOICE_ANSWER: Final[Mapping[str, object]] = {
|
||||
"type": "choice",
|
||||
"name": None,
|
||||
"choice": True,
|
||||
"probabilities": [{"value": True, "probability": 0.9}, {"value": "escalate", "probability": 0.1}],
|
||||
"confidence": 0.8,
|
||||
}
|
||||
_REFUSAL_ANSWER: Final[Mapping[str, object]] = {"type": "refusal", "name": "fraud"}
|
||||
_USAGE: Final[Mapping[str, object]] = {
|
||||
"input_tokens": 383,
|
||||
"input_tokens_details": {"cached_tokens": 256, "cache_write_tokens": 64},
|
||||
"output_tokens": 2,
|
||||
"output_tokens_details": {"reasoning_tokens": 1},
|
||||
"total_tokens": 385,
|
||||
}
|
||||
_OPENAI_RESPONSE: Final[Mapping[str, object]] = {
|
||||
"model": "gpt-6-luna",
|
||||
"answers": [
|
||||
_PREDICATE_ANSWER,
|
||||
_CHOICE_ANSWER,
|
||||
{
|
||||
"type": "score",
|
||||
"name": "severity",
|
||||
"score": 0.7,
|
||||
"probabilities": [
|
||||
{"value": 0, "label": "minor", "probability": 0.3},
|
||||
{"value": 1, "label": "major", "probability": 0.7},
|
||||
],
|
||||
"confidence": 0.6,
|
||||
},
|
||||
_REFUSAL_ANSWER,
|
||||
],
|
||||
"usage": _USAGE,
|
||||
}
|
||||
|
||||
|
||||
def _openai_ir(raw: Mapping[str, object]) -> DecisionsIRRequest:
|
||||
return openai_request_to_ir(_OPENAI_BODY.validate_python(raw))
|
||||
|
||||
|
||||
def test_an_openai_request_reaches_openai_unchanged() -> None:
|
||||
assert ir_to_openai_request("gpt-6-luna", _openai_ir(_OPENAI_REQUEST)) == {
|
||||
"model": "gpt-6-luna",
|
||||
**_OPENAI_REQUEST,
|
||||
}
|
||||
|
||||
|
||||
def test_an_openai_response_reaches_the_caller_unchanged() -> None:
|
||||
ir: Final = _openai_ir(_OPENAI_REQUEST)
|
||||
|
||||
parsed: Final = OpenAIDecisionsConfig().parse_response(_OPENAI_RESPONSE, ir)
|
||||
|
||||
assert ir_to_openai_response(parsed, ir, "requested").model_dump(mode="json") == _OPENAI_RESPONSE
|
||||
|
||||
|
||||
def test_answers_openai_did_not_return_are_refusals() -> None:
|
||||
ir: Final = _openai_ir(_OPENAI_REQUEST)
|
||||
payload: Final = {**_OPENAI_RESPONSE, "answers": [_PREDICATE_ANSWER]}
|
||||
|
||||
response: Final = ir_to_openai_response(OpenAIDecisionsConfig().parse_response(payload, ir), ir, "requested")
|
||||
|
||||
assert [answer.type for answer in response.answers] == ["predicate", "refusal", "refusal", "refusal"]
|
||||
assert [answer.name for answer in response.answers] == ["damaged", None, "severity", "fraud"]
|
||||
|
||||
|
||||
def test_a_systemone_request_becomes_an_openai_request_with_questions_named_by_their_keys() -> None:
|
||||
request: Final = _SYSTEMONE_BODY.validate_python(
|
||||
{
|
||||
"state": {"ticket": 1234, "text": "Screen cracked"},
|
||||
"questions": {
|
||||
"damaged": {
|
||||
"type": "noul",
|
||||
"instructions": "Is the item damaged?",
|
||||
"criteria": {"true": "Visible damage", "false": None},
|
||||
"provider_field": "dropped",
|
||||
},
|
||||
"rubric_only": {"type": "noul", "criteria": {"true": {"signal": "refund"}}},
|
||||
"action": {
|
||||
"type": "choice",
|
||||
"instructions": {"policy": "refund-v2"},
|
||||
"criteria": {"refund": "Within 30 days", "escalate": None},
|
||||
},
|
||||
"severity": {"type": "score", "criteria": ["minor", {"label": "major"}]},
|
||||
},
|
||||
}
|
||||
)
|
||||
|
||||
assert ir_to_openai_request("gpt-6-luna", systemone_request_to_ir(request)) == {
|
||||
"model": "gpt-6-luna",
|
||||
"input": '{"ticket": 1234, "text": "Screen cracked"}',
|
||||
"questions": [
|
||||
{
|
||||
"type": "predicate",
|
||||
"name": "damaged",
|
||||
"instructions": "Is the item damaged?\n\nAnswer true when: Visible damage",
|
||||
},
|
||||
{"type": "predicate", "name": "rubric_only", "instructions": 'Answer true when: {"signal": "refund"}'},
|
||||
{
|
||||
"type": "choice",
|
||||
"name": "action",
|
||||
"instructions": '{"policy": "refund-v2"}',
|
||||
"choices": [{"value": "refund", "description": "Within 30 days"}, {"value": "escalate"}],
|
||||
},
|
||||
{
|
||||
"type": "score",
|
||||
"name": "severity",
|
||||
"instructions": "Which level best fits the input?",
|
||||
"levels": [{"label": "minor"}, {"label": '{"label": "major"}'}],
|
||||
},
|
||||
],
|
||||
}
|
||||
|
||||
|
||||
def test_systemone_questions_without_instructions_become_valid_openai_questions() -> None:
|
||||
request: Final = _SYSTEMONE_BODY.validate_python(
|
||||
{
|
||||
"state": "Screen cracked",
|
||||
"questions": {
|
||||
"damaged": {"type": "noul", "criteria": {"false": None}},
|
||||
"action": {"type": "choice", "criteria": {"refund": None, "escalate": None}},
|
||||
"severity": {"type": "score", "criteria": ["minor", "major"]},
|
||||
},
|
||||
}
|
||||
)
|
||||
|
||||
body: Final = ir_to_openai_request("gpt-6-luna", systemone_request_to_ir(request))
|
||||
|
||||
assert all(question.instructions for question in _OPENAI_BODY.validate_python(body).questions)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"question",
|
||||
({"type": "choice", "criteria": {"refund": None}}, {"type": "score", "criteria": ["minor"]}),
|
||||
ids=("choice", "score"),
|
||||
)
|
||||
def test_a_systemone_question_with_one_option_is_unsupported_by_openai(question: Mapping[str, object]) -> None:
|
||||
request: Final = _SYSTEMONE_BODY.validate_python(
|
||||
{
|
||||
"state": "Screen cracked",
|
||||
"questions": {"damaged": {"type": "noul", "instructions": "Damaged?"}, "q": question},
|
||||
}
|
||||
)
|
||||
|
||||
assert isinstance(ir_to_openai_request("gpt-6-luna", systemone_request_to_ir(request)), UnsupportedDecisionsRequest)
|
||||
|
||||
|
||||
def test_an_openai_response_becomes_systemone_answers_with_the_callers_score_labels() -> None:
|
||||
ir: Final = systemone_request_to_ir(
|
||||
_SYSTEMONE_BODY.validate_python(
|
||||
{
|
||||
"state": "Screen cracked",
|
||||
"questions": {
|
||||
"damaged": {"type": "noul", "instructions": "Damaged?"},
|
||||
"action": {"type": "choice", "criteria": {"true": None, "escalate": None}},
|
||||
"severity": {"type": "score", "criteria": ["minor", {"label": "major"}]},
|
||||
"fraud": {"type": "noul", "instructions": "Fraud?"},
|
||||
},
|
||||
}
|
||||
)
|
||||
)
|
||||
payload: Final = {
|
||||
**_OPENAI_RESPONSE,
|
||||
"answers": [
|
||||
_PREDICATE_ANSWER,
|
||||
_CHOICE_ANSWER,
|
||||
{
|
||||
"type": "score",
|
||||
"name": "severity",
|
||||
"score": 0.7,
|
||||
"probabilities": [
|
||||
{"value": 0, "label": "minor", "probability": 0.3},
|
||||
{"value": 1, "label": '{"label": "major"}', "probability": 0.7},
|
||||
],
|
||||
"confidence": 0.6,
|
||||
},
|
||||
_REFUSAL_ANSWER,
|
||||
],
|
||||
}
|
||||
|
||||
response: Final = ir_to_systemone_response(OpenAIDecisionsConfig().parse_response(payload, ir), ir)
|
||||
|
||||
assert response.model_dump(mode="json") == {
|
||||
"model": "gpt-6-luna",
|
||||
"answers": {
|
||||
"damaged": {"type": "noul", "noul": 0.95},
|
||||
"action": {
|
||||
"type": "choice",
|
||||
"choice": "true",
|
||||
"confidence": 0.8,
|
||||
"probabilities": {"true": 0.9, "escalate": 0.1},
|
||||
},
|
||||
"severity": {
|
||||
"type": "score",
|
||||
"score": 0.7,
|
||||
"confidence": 0.6,
|
||||
"legend": {"0": "minor", "1": {"label": "major"}},
|
||||
"probabilities": {"0": 0.3, "1": 0.7},
|
||||
},
|
||||
},
|
||||
"usage": {"input_tokens": 383, "output_tokens": 2},
|
||||
}
|
||||
assert response.usage is not None
|
||||
assert (response.usage.cached_tokens, response.usage.cache_write_tokens) == (256, 64)
|
||||
|
|
@ -320,6 +320,28 @@ _SYSTEMONE_ANSWERS_FOR_OPENAI_REQUEST: Final[Mapping[str, object]] = {
|
|||
"usage": {"input_tokens": _INPUT_TOKENS, "output_tokens": _OUTPUT_TOKENS},
|
||||
}
|
||||
|
||||
_OPENAI_FORMAT_ANSWERS: Final = [
|
||||
{"type": "predicate", "name": "damaged", "probability": 0.95},
|
||||
{
|
||||
"type": "choice",
|
||||
"name": None,
|
||||
"choice": True,
|
||||
"probabilities": [{"value": True, "probability": 0.9}, {"value": "escalate", "probability": 0.1}],
|
||||
"confidence": 0.8,
|
||||
},
|
||||
{
|
||||
"type": "score",
|
||||
"name": "severity",
|
||||
"score": 0.7,
|
||||
"probabilities": [
|
||||
{"value": 0, "label": "minor", "probability": 0.3},
|
||||
{"value": 1, "label": "major", "probability": 0.7},
|
||||
],
|
||||
"confidence": 0.6,
|
||||
},
|
||||
{"type": "refusal", "name": "fraud"},
|
||||
]
|
||||
|
||||
|
||||
@pytest.mark.parametrize("endpoint", ("/v1/decisions", "/decisions"))
|
||||
def test_openai_format_decisions_translate_through_systemone(
|
||||
|
|
@ -354,27 +376,7 @@ def test_openai_format_decisions_translate_through_systemone(
|
|||
}
|
||||
body: Final = response.json()
|
||||
assert body["model"] == _SYSTEMONE_ANSWERS_FOR_OPENAI_REQUEST["model"]
|
||||
assert body["answers"] == [
|
||||
{"type": "predicate", "name": "damaged", "probability": 0.95},
|
||||
{
|
||||
"type": "choice",
|
||||
"name": None,
|
||||
"choice": True,
|
||||
"probabilities": [{"value": True, "probability": 0.9}, {"value": "escalate", "probability": 0.1}],
|
||||
"confidence": 0.8,
|
||||
},
|
||||
{
|
||||
"type": "score",
|
||||
"name": "severity",
|
||||
"score": 0.7,
|
||||
"probabilities": [
|
||||
{"value": 0, "label": "minor", "probability": 0.3},
|
||||
{"value": 1, "label": "major", "probability": 0.7},
|
||||
],
|
||||
"confidence": 0.6,
|
||||
},
|
||||
{"type": "refusal", "name": "fraud"},
|
||||
]
|
||||
assert body["answers"] == _OPENAI_FORMAT_ANSWERS
|
||||
assert body["usage"] == {
|
||||
"input_tokens": _INPUT_TOKENS,
|
||||
"input_tokens_details": {"cached_tokens": 0, "cache_write_tokens": 0},
|
||||
|
|
@ -462,6 +464,72 @@ def test_a_body_that_is_not_json_is_a_client_error(
|
|||
assert not upstream.called
|
||||
|
||||
|
||||
def test_openai_format_decisions_reach_an_openai_deployment_unchanged_including_images(
|
||||
client: TestClient,
|
||||
monkeypatch: pytest.MonkeyPatch,
|
||||
respx_mock: respx.MockRouter,
|
||||
) -> None:
|
||||
monkeypatch.setattr(litellm, "api_base", None)
|
||||
monkeypatch.delenv("OPENAI_BASE_URL", raising=False)
|
||||
monkeypatch.delenv("OPENAI_API_BASE", raising=False)
|
||||
monkeypatch.setattr(
|
||||
litellm.proxy.proxy_server,
|
||||
"llm_router",
|
||||
litellm.Router(
|
||||
model_list=[{"model_name": "decider", "litellm_params": {"model": "openai/gpt-6-luna", "api_key": "k"}}]
|
||||
),
|
||||
)
|
||||
image_message: Final = {
|
||||
"type": "message",
|
||||
"role": "user",
|
||||
"content": [{"type": "input_image", "image_url": "data:image/png;base64,AA==", "detail": "low"}],
|
||||
}
|
||||
request_body: Final = {**_OPENAI_FORMAT_REQUEST, "input": [*_OPENAI_FORMAT_REQUEST["input"], image_message]}
|
||||
cached_tokens: Final = 128
|
||||
upstream_usage: Final = {
|
||||
"input_tokens": _INPUT_TOKENS,
|
||||
"input_tokens_details": {"cached_tokens": cached_tokens, "cache_write_tokens": 0},
|
||||
"output_tokens": _OUTPUT_TOKENS,
|
||||
"output_tokens_details": {"reasoning_tokens": 0},
|
||||
"total_tokens": _INPUT_TOKENS + _OUTPUT_TOKENS,
|
||||
}
|
||||
upstream: Final = respx_mock.post("https://api.openai.com/v1/decisions").respond(
|
||||
json={"model": "gpt-6-luna", "answers": _OPENAI_FORMAT_ANSWERS, "usage": upstream_usage}
|
||||
)
|
||||
|
||||
response: Final = client.post("/v1/decisions", json=request_body)
|
||||
|
||||
assert response.status_code == 200, response.text
|
||||
assert json.loads(upstream.calls[0].request.content) == {
|
||||
"model": "gpt-6-luna",
|
||||
"input": [
|
||||
{
|
||||
"type": "message",
|
||||
"role": "user",
|
||||
"content": [
|
||||
{"type": "input_text", "text": "The package arrived with a broken screen."},
|
||||
{"type": "input_text", "text": "I want a refund."},
|
||||
],
|
||||
},
|
||||
{"type": "message", "role": "user", "content": "Order 1234."},
|
||||
image_message,
|
||||
],
|
||||
"questions": _OPENAI_FORMAT_REQUEST["questions"],
|
||||
"safety_identifier": "end-user-1",
|
||||
}
|
||||
body: Final = response.json()
|
||||
assert body["answers"] == _OPENAI_FORMAT_ANSWERS
|
||||
assert body["usage"] == upstream_usage
|
||||
luna_cost: Final = litellm.model_cost["gpt-6-luna"]
|
||||
expected_cost: Final = (
|
||||
(_INPUT_TOKENS - cached_tokens) * float(luna_cost["input_cost_per_token"])
|
||||
+ cached_tokens * float(luna_cost["cache_read_input_token_cost"])
|
||||
+ _OUTPUT_TOKENS * float(luna_cost["output_cost_per_token"])
|
||||
)
|
||||
assert expected_cost > 0
|
||||
assert float(response.headers["x-litellm-response-cost"]) == pytest.approx(expected_cost)
|
||||
|
||||
|
||||
def _decisions_feature() -> LazyFeature:
|
||||
return next(feature for feature in LAZY_FEATURES if feature.name == "decisions")
|
||||
|
||||
|
|
|
|||
|
|
@ -1064,6 +1064,7 @@ def test_aaamodel_prices_and_context_window_json_is_valid():
|
|||
"/vertex_ai/live",
|
||||
"/v1/listen",
|
||||
"/v1/systemone",
|
||||
"/v1/decisions",
|
||||
"/v1beta/interactions",
|
||||
],
|
||||
},
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue