mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-11 03:38:38 +00:00
* feat(decisions): add unified /v1/decisions endpoint for Jev-compatible providers Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(decisions): register typesafe as a provider so Jev deployments load Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * refactor(decisions): move provider endpoints under llms and validate proxy bodies Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * feat(decisions): add Cloudflare Clef and Strands Decider backends Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(decisions): register decisions routes for managed agents and gateway Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(decisions): use raw regex for cloudflare missing account match Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(decisions): avoid cast in Cloudflare response unwrapping Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(decisions): default model, evaluation health probe, short Cloudflare names The proxy validates only state and questions, so a request without a model falls through to the configured default model like every other route. Health checks probe evaluation-mode deployments through the Decisions API instead of failing with an unsupported mode, and cloudflare/clef and cloudflare/clef-flash get cost-map rows so the short names resolve a mode and a price. The registry no longer claims typed decisions for a provider with no backend. * fix(decisions): let health_check_params override the evaluation probe Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(integration): audit the decisions endpoint across providers, limits, health and chaos Adds the /v1/decisions audit cells: one wire contract per provider (path, key, body and cost-map billing), the gateway-only fields and tags, the sad paths (invalid bodies, unknown model, key checks, api_base in the body, upstream 401/429/500, a 200 without answers, an unreachable upstream), the two evaluation-mode health probes, and three chaos cells (a mixed-failure burst over both routes, a worker SIGKILL mid-burst, an upstream outage and restart on the same port). The PR's cost case read the upstream observations through the gateway, which answers 404 for that path; it now reads them from the upstream URL. The owned proxy harness takes extra CLI arguments, and its graceful stop waits as long as a worker boot may take, since a worker still starting honors SIGTERM only once it is up and the 30 second wait forced a cleanup under load. * fix(decisions): send env API keys to a configured api_base Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * feat(decisions): add zero-cost evaluation cost-map entry for Strands Decider Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(decisions): register the routes through the lazy feature registry The Decisions router was included at import, ahead of the config and DB pass-through endpoints, so a pass-through configured at /v1/decisions was skipped and answered 400 as an unknown Decisions provider. The routes now register through LAZY_FEATURES, which splices them in after every eager route, so a pass-through at /v1/decisions keeps its route while /decisions still serves natively. The lazy OpenAPI snapshot carries the two paths so the schema shows them before the first call. The audit cells add the env-key egress to a configured api_base, the client api_base opt-in shared with chat, the pass-through precedence on an owned proxy, and the Strands evaluation health check resolved from the cost map. The integration config exports the Perplexity env key the first cell needs. * fix(decisions): keep the Cloudflare api_base message in its transformation and read the audit upstream once per cell * fix(proxy): let a config pass-through beat a lazily registered route in eager mode With LITELLM_DISABLE_LAZY_ROUTES set the decisions routes are registered at startup, so SafeRouteAdder treated a config pass-through at exactly /v1/decisions as already registered and dropped it. In lazy mode a pass-through created through the API after the first native call was skipped the same way. Routes a lazy feature owns no longer count as registered, and a route added at one of their paths is placed ahead of them, the precedence lazy mode gives a config pass-through when the feature has not loaded yet. --------- Co-authored-by: mateo <mateo@berri.ai> Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> Co-authored-by: mateo-berri <277851410+mateo-berri@users.noreply.github.com>
592 lines
21 KiB
Python
592 lines
21 KiB
Python
from __future__ import annotations
|
|
|
|
import asyncio
|
|
import json
|
|
from collections.abc import Mapping
|
|
from types import MappingProxyType
|
|
from typing import Final
|
|
|
|
import pytest
|
|
import respx
|
|
|
|
import litellm
|
|
from litellm.integrations.custom_logger import CustomLogger
|
|
from litellm.litellm_core_utils.logging_worker import GLOBAL_LOGGING_WORKER
|
|
from litellm.types.decisions import (
|
|
ChoiceAnswer,
|
|
DecisionsResponse,
|
|
DecisionsUsage,
|
|
NoulAnswer,
|
|
ScoreAnswer,
|
|
)
|
|
|
|
_QUESTIONS: Final[Mapping[str, object]] = MappingProxyType(
|
|
{
|
|
"is_defect": {"type": "noul", "instructions": "Is this a defect?", "provider_field": "kept"},
|
|
"sentiment": {"type": "choice", "criteria": {"positive": None, "negative": "unhappy"}},
|
|
"severity": {"type": "score", "criteria": ["none", "low", "high"]},
|
|
}
|
|
)
|
|
_INPUT_TOKENS: Final[int] = 367
|
|
_OUTPUT_TOKENS: Final[int] = 3
|
|
_RESPONSE: Final[Mapping[str, object]] = {
|
|
"model": "jev-1.13",
|
|
"answers": {
|
|
"is_defect": {"type": "noul", "noul": 0.9},
|
|
"sentiment": {
|
|
"type": "choice",
|
|
"choice": "positive",
|
|
"confidence": 0.8,
|
|
"probabilities": {"positive": 0.8, "negative": 0.2},
|
|
},
|
|
"severity": {
|
|
"type": "score",
|
|
"score": 1,
|
|
"confidence": 0.7,
|
|
"legend": {"0": "none", "1": "low", "2": "high"},
|
|
"probabilities": {"0": 0.1, "1": 0.8, "2": 0.1},
|
|
},
|
|
},
|
|
"usage": {"input_tokens": _INPUT_TOKENS, "output_tokens": _OUTPUT_TOKENS},
|
|
}
|
|
_STRANDS_RESPONSE: Final[Mapping[str, object]] = {
|
|
"model": "strands-decider-2B-hobson-v19",
|
|
"answers": {
|
|
"severity": {
|
|
"type": "score",
|
|
"score": 1,
|
|
"confidence": 0.7,
|
|
"legend": {"0": "none", "1": "low", "2": "high"},
|
|
"probabilities": {"0": 0.1, "1": 0.8, "2": 0.1},
|
|
}
|
|
},
|
|
"usage": {"input_tokens": 216, "output_tokens": 3},
|
|
"latency_ms": 3722.17,
|
|
}
|
|
_PROVIDERS: Final[tuple[tuple[str, str, str, str], ...]] = (
|
|
(
|
|
"perplexity",
|
|
"perplexity/pplx-decider-v1-27b",
|
|
"https://api.perplexity.ai/v1/decisions",
|
|
"pplx-decider-v1-27b",
|
|
),
|
|
("typesafe", "typesafe/jev-1.13", "https://api.typesafe.ai/v1/systemone", "jev-1.13"),
|
|
(
|
|
"openrouter",
|
|
"openrouter/typesafe/jev-1.13",
|
|
"https://openrouter.ai/api/alpha/decisions",
|
|
"typesafe/jev-1.13",
|
|
),
|
|
)
|
|
|
|
|
|
class _RecordingLogger(CustomLogger):
|
|
def __init__(self) -> None:
|
|
super().__init__()
|
|
self.standard_logging_object: Mapping[str, object] | None = None
|
|
|
|
async def async_log_success_event(
|
|
self,
|
|
kwargs: Mapping[str, object],
|
|
response_obj: object,
|
|
start_time: object,
|
|
end_time: object,
|
|
) -> None:
|
|
standard_logging_object: Final = kwargs.get("standard_logging_object")
|
|
if isinstance(standard_logging_object, dict):
|
|
self.standard_logging_object = standard_logging_object
|
|
|
|
|
|
async def _drain_logging_worker() -> None:
|
|
await asyncio.sleep(0)
|
|
GLOBAL_LOGGING_WORKER.start()
|
|
await asyncio.wait_for(GLOBAL_LOGGING_WORKER.flush(), timeout=10.0)
|
|
|
|
|
|
@pytest.fixture(autouse=True)
|
|
def _httpx_transport(monkeypatch: pytest.MonkeyPatch) -> None:
|
|
monkeypatch.setattr(litellm, "disable_aiohttp_transport", True)
|
|
litellm.in_memory_llm_clients_cache.flush_cache()
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
@pytest.mark.parametrize(("provider", "model", "url", "upstream_model"), _PROVIDERS)
|
|
async def test_adecisions_sends_the_provider_wire_contract(
|
|
provider: str,
|
|
model: str,
|
|
url: str,
|
|
upstream_model: str,
|
|
respx_mock: respx.MockRouter,
|
|
) -> None:
|
|
route: Final = respx_mock.post(url).respond(json=_RESPONSE)
|
|
|
|
response: Final = await litellm.adecisions(
|
|
model=model,
|
|
state={"source": "unit-test"},
|
|
questions=_QUESTIONS,
|
|
api_key="caller-key",
|
|
extra_headers={
|
|
"x-request-tag": "decisions-test",
|
|
"AUTHORIZATION": "attacker-key",
|
|
"Content-Type": "text/plain",
|
|
},
|
|
internal_kwarg="must-not-leak",
|
|
)
|
|
|
|
assert route.called
|
|
assert len(respx_mock.calls) == 1
|
|
request: Final = respx_mock.calls[0].request
|
|
assert request.headers["authorization"] == "Bearer caller-key"
|
|
assert request.headers["content-type"] == "application/json"
|
|
assert request.headers["x-request-tag"] == "decisions-test"
|
|
assert json.loads(request.content) == {
|
|
"model": upstream_model,
|
|
"state": {"source": "unit-test"},
|
|
"questions": {
|
|
"is_defect": {
|
|
"type": "noul",
|
|
"instructions": "Is this a defect?",
|
|
"provider_field": "kept",
|
|
},
|
|
"sentiment": {"type": "choice", "criteria": {"positive": None, "negative": "unhappy"}},
|
|
"severity": {"type": "score", "criteria": ["none", "low", "high"]},
|
|
},
|
|
}
|
|
assert isinstance(response.answers["is_defect"], NoulAnswer)
|
|
assert isinstance(response.answers["sentiment"], ChoiceAnswer)
|
|
assert isinstance(response.answers["severity"], ScoreAnswer)
|
|
assert response._hidden_params["custom_llm_provider"] == provider
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_router_dispatches_typesafe_decisions_without_api_base(
|
|
monkeypatch: pytest.MonkeyPatch,
|
|
respx_mock: respx.MockRouter,
|
|
) -> None:
|
|
monkeypatch.delenv("TYPESAFE_API_KEY", raising=False)
|
|
monkeypatch.delenv("TYPESAFE_API_BASE", raising=False)
|
|
provider_resolution: Final = litellm.get_llm_provider("typesafe/jev-latest")
|
|
|
|
assert provider_resolution[:2] == ("jev-latest", "typesafe")
|
|
|
|
router: Final = litellm.Router(
|
|
model_list=[
|
|
{
|
|
"model_name": "jev",
|
|
"litellm_params": {
|
|
"model": "typesafe/jev-latest",
|
|
"api_key": "k",
|
|
},
|
|
}
|
|
]
|
|
)
|
|
upstream: Final = respx_mock.post("https://api.typesafe.ai/v1/systemone").respond(json=_RESPONSE)
|
|
|
|
response: Final = await router.adecisions(
|
|
model="jev",
|
|
state="router-test",
|
|
questions={
|
|
"sentiment": {
|
|
"type": "choice",
|
|
"criteria": {"positive": None, "negative": "unhappy"},
|
|
}
|
|
},
|
|
)
|
|
|
|
assert upstream.called
|
|
assert len(respx_mock.calls) == 1
|
|
assert json.loads(respx_mock.calls[0].request.content) == {
|
|
"model": "jev-latest",
|
|
"state": "router-test",
|
|
"questions": {
|
|
"sentiment": {
|
|
"type": "choice",
|
|
"criteria": {"positive": None, "negative": "unhappy"},
|
|
}
|
|
},
|
|
}
|
|
assert respx_mock.calls[0].request.headers["authorization"] == "Bearer k"
|
|
assert isinstance(response.answers["sentiment"], ChoiceAnswer)
|
|
assert response.answers["sentiment"].choice == "positive"
|
|
|
|
|
|
def test_decisions_uses_the_same_wire_contract_for_sync_calls(respx_mock: respx.MockRouter) -> None:
|
|
route: Final = respx_mock.post("https://api.perplexity.ai/v1/decisions").respond(json=_RESPONSE)
|
|
|
|
response: Final = litellm.decisions(
|
|
model="perplexity/pplx-decider-v1-27b",
|
|
state="review",
|
|
questions={"is_defect": {"type": "noul", "instructions": "Is this a defect?"}},
|
|
api_key="caller-key",
|
|
)
|
|
|
|
assert route.called
|
|
assert response.model == "jev-1.13"
|
|
|
|
|
|
def test_openrouter_response_keeps_provider_fields(respx_mock: respx.MockRouter) -> None:
|
|
payload: Final = {
|
|
**_RESPONSE,
|
|
"id": "decision-1",
|
|
"provider": "typesafe",
|
|
"usage": {**_RESPONSE["usage"], "cost": 0.25},
|
|
}
|
|
respx_mock.post("https://openrouter.ai/api/alpha/decisions").respond(json=payload)
|
|
|
|
response: Final = litellm.decisions(
|
|
model="openrouter/typesafe/jev-1.13",
|
|
state="review",
|
|
questions=_QUESTIONS,
|
|
api_key="caller-key",
|
|
)
|
|
|
|
assert response.model_extra["id"] == "decision-1"
|
|
assert response.model_extra["provider"] == "typesafe"
|
|
assert response.usage is not None
|
|
assert response.usage.model_extra["cost"] == 0.25
|
|
|
|
|
|
def test_decisions_cost_uses_litellm_token_pricing() -> None:
|
|
response: Final = DecisionsResponse(
|
|
model="pplx-decider-v1-27b",
|
|
answers={},
|
|
usage=DecisionsUsage(input_tokens=_INPUT_TOKENS, output_tokens=_OUTPUT_TOKENS),
|
|
)
|
|
response._hidden_params = {
|
|
"model": "perplexity/pplx-decider-v1-27b",
|
|
"custom_llm_provider": "perplexity",
|
|
}
|
|
|
|
cost: Final = litellm.completion_cost(completion_response=response)
|
|
perplexity_cost: Final = litellm.model_cost["perplexity/pplx-decider-v1-27b"]
|
|
expected_cost: Final = _INPUT_TOKENS * float(perplexity_cost["input_cost_per_token"]) + _OUTPUT_TOKENS * float(
|
|
perplexity_cost["output_cost_per_token"]
|
|
)
|
|
|
|
assert expected_cost > 0
|
|
assert cost == pytest.approx(expected_cost)
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_decisions_cost_is_in_standard_logging_object(respx_mock: respx.MockRouter) -> None:
|
|
respx_mock.post("https://api.perplexity.ai/v1/decisions").respond(json=_RESPONSE)
|
|
recording_logger: Final = _RecordingLogger()
|
|
original_callbacks: Final = litellm.callbacks
|
|
litellm.callbacks = [recording_logger]
|
|
|
|
try:
|
|
await litellm.adecisions(
|
|
model="perplexity/pplx-decider-v1-27b",
|
|
state="review",
|
|
questions={"is_defect": {"type": "noul", "instructions": "Is this a defect?"}},
|
|
api_key="caller-key",
|
|
)
|
|
await _drain_logging_worker()
|
|
finally:
|
|
litellm.callbacks = original_callbacks
|
|
|
|
assert recording_logger.standard_logging_object is not None
|
|
perplexity_cost: Final = litellm.model_cost["perplexity/pplx-decider-v1-27b"]
|
|
expected_cost: Final = _INPUT_TOKENS * float(perplexity_cost["input_cost_per_token"]) + _OUTPUT_TOKENS * float(
|
|
perplexity_cost["output_cost_per_token"]
|
|
)
|
|
|
|
assert expected_cost > 0
|
|
assert recording_logger.standard_logging_object["response_cost"] == pytest.approx(expected_cost)
|
|
assert recording_logger.standard_logging_object["prompt_tokens"] == _INPUT_TOKENS
|
|
assert recording_logger.standard_logging_object["completion_tokens"] == _OUTPUT_TOKENS
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_unknown_provider_is_rejected_before_http(respx_mock: respx.MockRouter) -> None:
|
|
with pytest.raises(litellm.BadRequestError, match="Supported providers"):
|
|
await litellm.adecisions(
|
|
model="unknown/jev-1.13",
|
|
state="review",
|
|
questions={"is_defect": {"type": "noul", "instructions": "Is this a defect?"}},
|
|
api_key="caller-key",
|
|
)
|
|
|
|
assert len(respx_mock.calls) == 0
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_empty_custom_provider_is_rejected_before_http(respx_mock: respx.MockRouter) -> None:
|
|
with pytest.raises(litellm.BadRequestError, match="Supported providers"):
|
|
await litellm.adecisions(
|
|
model="perplexity/pplx-decider-v1-27b",
|
|
state="review",
|
|
questions={"is_defect": {"type": "noul", "instructions": "Is this a defect?"}},
|
|
api_key="caller-key",
|
|
custom_llm_provider="",
|
|
)
|
|
|
|
assert len(respx_mock.calls) == 0
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_invalid_question_is_rejected_before_http(respx_mock: respx.MockRouter) -> None:
|
|
with pytest.raises(litellm.BadRequestError, match="Invalid Decisions request"):
|
|
await litellm.adecisions(
|
|
model="perplexity/pplx-decider-v1-27b",
|
|
state="review",
|
|
questions={"sentiment": {"type": "choice"}},
|
|
api_key="caller-key",
|
|
)
|
|
|
|
assert len(respx_mock.calls) == 0
|
|
|
|
|
|
def test_upstream_bad_request_maps_to_litellm_error(respx_mock: respx.MockRouter) -> None:
|
|
respx_mock.post("https://api.perplexity.ai/v1/decisions").respond(
|
|
status_code=400,
|
|
json={"error": {"message": "invalid decision"}},
|
|
)
|
|
|
|
with pytest.raises(litellm.BadRequestError):
|
|
litellm.decisions(
|
|
model="perplexity/pplx-decider-v1-27b",
|
|
state="review",
|
|
questions={"is_defect": {"type": "noul", "instructions": "Is this a defect?"}},
|
|
api_key="caller-key",
|
|
)
|
|
|
|
|
|
def test_server_key_is_sent_to_an_explicit_api_base(
|
|
monkeypatch: pytest.MonkeyPatch,
|
|
respx_mock: respx.MockRouter,
|
|
) -> None:
|
|
monkeypatch.setenv("PERPLEXITYAI_API_KEY", "server-key")
|
|
monkeypatch.delenv("PERPLEXITY_API_KEY", raising=False)
|
|
route: Final = respx_mock.post("https://egress.example/perplexity/v1/decisions").respond(json=_RESPONSE)
|
|
|
|
litellm.decisions(
|
|
model="perplexity/pplx-decider-v1-27b",
|
|
state="review",
|
|
questions={"is_defect": {"type": "noul", "instructions": "Is this a defect?"}},
|
|
api_base="https://egress.example/perplexity",
|
|
)
|
|
|
|
assert route.call_count == 1
|
|
assert route.calls[0].request.headers["authorization"] == "Bearer server-key"
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
@pytest.mark.parametrize("model", ("cloudflare/clef", "cloudflare/@cf/cloudflare/clef"))
|
|
@pytest.mark.parametrize("wrapped", (False, True))
|
|
async def test_cloudflare_clef_resolves_model_and_response_envelope(
|
|
monkeypatch: pytest.MonkeyPatch,
|
|
respx_mock: respx.MockRouter,
|
|
model: str,
|
|
wrapped: bool,
|
|
) -> None:
|
|
monkeypatch.setenv("CLOUDFLARE_ACCOUNT_ID", "acct")
|
|
monkeypatch.setenv("CLOUDFLARE_API_KEY", "cloudflare-key")
|
|
monkeypatch.delenv("CLOUDFLARE_API_BASE", raising=False)
|
|
response_body: Final[Mapping[str, object]] = (
|
|
{"result": _RESPONSE, "success": True, "errors": [], "messages": []} if wrapped else _RESPONSE
|
|
)
|
|
route: Final = respx_mock.post(
|
|
"https://api.cloudflare.com/client/v4/accounts/acct/ai/run/@cf/cloudflare/clef"
|
|
).respond(json=response_body)
|
|
|
|
response: Final = await litellm.adecisions(
|
|
model=model,
|
|
state="review",
|
|
questions={"is_defect": {"type": "noul", "instructions": "Is this a defect?"}},
|
|
)
|
|
|
|
assert route.called
|
|
request: Final = respx_mock.calls[0].request
|
|
assert request.headers["authorization"] == "Bearer cloudflare-key"
|
|
assert json.loads(request.content) == {
|
|
"model": "clef",
|
|
"state": "review",
|
|
"questions": {"is_defect": {"type": "noul", "instructions": "Is this a defect?"}},
|
|
}
|
|
assert response.answers == DecisionsResponse.model_validate(_RESPONSE).answers
|
|
assert response._hidden_params["model"] == "cloudflare/@cf/cloudflare/clef"
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_cloudflare_clef_flash_uses_flash_endpoint_and_request_model(
|
|
monkeypatch: pytest.MonkeyPatch,
|
|
respx_mock: respx.MockRouter,
|
|
) -> None:
|
|
monkeypatch.setenv("CLOUDFLARE_ACCOUNT_ID", "acct")
|
|
monkeypatch.setenv("CLOUDFLARE_API_KEY", "cloudflare-key")
|
|
monkeypatch.delenv("CLOUDFLARE_API_BASE", raising=False)
|
|
route: Final = respx_mock.post(
|
|
"https://api.cloudflare.com/client/v4/accounts/acct/ai/run/@cf/cloudflare/clef-flash"
|
|
).respond(json=_RESPONSE)
|
|
|
|
await litellm.adecisions(
|
|
model="cloudflare/clef-flash",
|
|
state="review",
|
|
questions={"is_defect": {"type": "noul", "instructions": "Is this a defect?"}},
|
|
)
|
|
|
|
assert route.called
|
|
assert json.loads(respx_mock.calls[0].request.content)["model"] == "clef-flash"
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_cloudflare_api_base_from_env_uses_workers_ai_run_path(
|
|
monkeypatch: pytest.MonkeyPatch,
|
|
respx_mock: respx.MockRouter,
|
|
) -> None:
|
|
monkeypatch.setenv("CLOUDFLARE_API_BASE", "https://api.cloudflare.com/client/v4/accounts/acct/ai/v1")
|
|
monkeypatch.setenv("CLOUDFLARE_API_KEY", "cloudflare-key")
|
|
monkeypatch.delenv("CLOUDFLARE_ACCOUNT_ID", raising=False)
|
|
route: Final = respx_mock.post(
|
|
"https://api.cloudflare.com/client/v4/accounts/acct/ai/run/@cf/cloudflare/clef"
|
|
).respond(json=_RESPONSE)
|
|
|
|
await litellm.adecisions(
|
|
model="cloudflare/clef",
|
|
state="review",
|
|
questions={"is_defect": {"type": "noul", "instructions": "Is this a defect?"}},
|
|
)
|
|
|
|
assert route.called
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_cloudflare_requires_account_id_or_api_base_before_http(
|
|
monkeypatch: pytest.MonkeyPatch,
|
|
respx_mock: respx.MockRouter,
|
|
) -> None:
|
|
monkeypatch.delenv("CLOUDFLARE_ACCOUNT_ID", raising=False)
|
|
monkeypatch.delenv("CLOUDFLARE_API_BASE", raising=False)
|
|
monkeypatch.setenv("CLOUDFLARE_API_KEY", "cloudflare-key")
|
|
|
|
with pytest.raises(litellm.BadRequestError, match="Missing CLOUDFLARE_ACCOUNT_ID - set CLOUDFLARE_ACCOUNT_ID"):
|
|
await litellm.adecisions(
|
|
model="cloudflare/clef",
|
|
state="review",
|
|
questions={"is_defect": {"type": "noul", "instructions": "Is this a defect?"}},
|
|
)
|
|
|
|
assert len(respx_mock.calls) == 0
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_cloudflare_clef_cost_uses_the_model_cost_map(
|
|
monkeypatch: pytest.MonkeyPatch,
|
|
respx_mock: respx.MockRouter,
|
|
) -> None:
|
|
monkeypatch.setenv("CLOUDFLARE_ACCOUNT_ID", "acct")
|
|
monkeypatch.setenv("CLOUDFLARE_API_KEY", "cloudflare-key")
|
|
monkeypatch.delenv("CLOUDFLARE_API_BASE", raising=False)
|
|
respx_mock.post("https://api.cloudflare.com/client/v4/accounts/acct/ai/run/@cf/cloudflare/clef").respond(
|
|
json=_RESPONSE
|
|
)
|
|
|
|
response: Final = await litellm.adecisions(
|
|
model="cloudflare/clef",
|
|
state="review",
|
|
questions={"is_defect": {"type": "noul", "instructions": "Is this a defect?"}},
|
|
)
|
|
|
|
cost: Final = litellm.completion_cost(completion_response=response)
|
|
clef_cost: Final = litellm.model_cost["cloudflare/@cf/cloudflare/clef"]
|
|
expected_cost: Final = _INPUT_TOKENS * float(clef_cost["input_cost_per_token"]) + _OUTPUT_TOKENS * float(
|
|
clef_cost["output_cost_per_token"]
|
|
)
|
|
|
|
assert expected_cost > 0
|
|
assert cost == pytest.approx(expected_cost)
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_strands_decider_requires_api_base_before_http(
|
|
monkeypatch: pytest.MonkeyPatch,
|
|
respx_mock: respx.MockRouter,
|
|
) -> None:
|
|
monkeypatch.delenv("STRANDS_DECIDER_API_BASE", raising=False)
|
|
monkeypatch.delenv("STRANDS_DECIDER_API_KEY", raising=False)
|
|
|
|
with pytest.raises(litellm.BadRequestError, match="api_base is required"):
|
|
await litellm.adecisions(
|
|
model="strands_decider/strands-decider-2B-hobson-v19",
|
|
state="review",
|
|
questions={"severity": {"type": "score", "criteria": ["none", "low", "high"]}},
|
|
)
|
|
|
|
assert len(respx_mock.calls) == 0
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_strands_decider_without_key_preserves_response_extras(
|
|
monkeypatch: pytest.MonkeyPatch,
|
|
respx_mock: respx.MockRouter,
|
|
) -> None:
|
|
monkeypatch.delenv("STRANDS_DECIDER_API_BASE", raising=False)
|
|
monkeypatch.delenv("STRANDS_DECIDER_API_KEY", raising=False)
|
|
route: Final = respx_mock.post("https://strands.example/v1/systemone").respond(json=_STRANDS_RESPONSE)
|
|
|
|
response: Final = await litellm.adecisions(
|
|
model="strands_decider/strands-decider-2B-hobson-v19",
|
|
state="review",
|
|
questions={"severity": {"type": "score", "criteria": ["none", "low", "high"]}},
|
|
api_base="https://strands.example",
|
|
)
|
|
|
|
assert route.called
|
|
assert "authorization" not in respx_mock.calls[0].request.headers
|
|
assert response.model_extra["latency_ms"] == _STRANDS_RESPONSE["latency_ms"]
|
|
severity: Final = response.answers["severity"]
|
|
assert isinstance(severity, ScoreAnswer)
|
|
assert severity.legend == {"0": "none", "1": "low", "2": "high"}
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_strands_decider_uses_key_from_matching_environment_base(
|
|
monkeypatch: pytest.MonkeyPatch,
|
|
respx_mock: respx.MockRouter,
|
|
) -> None:
|
|
monkeypatch.setenv("STRANDS_DECIDER_API_BASE", "https://strands.example")
|
|
monkeypatch.setenv("STRANDS_DECIDER_API_KEY", "strands-key")
|
|
route: Final = respx_mock.post("https://strands.example/v1/systemone").respond(json=_STRANDS_RESPONSE)
|
|
|
|
await litellm.adecisions(
|
|
model="strands_decider/strands-decider-2B-hobson-v19",
|
|
state="review",
|
|
questions={"severity": {"type": "score", "criteria": ["none", "low", "high"]}},
|
|
api_base="https://strands.example",
|
|
)
|
|
|
|
assert route.called
|
|
assert respx_mock.calls[0].request.headers["authorization"] == "Bearer strands-key"
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_strands_decider_provider_resolution_and_router_dispatch(
|
|
monkeypatch: pytest.MonkeyPatch,
|
|
respx_mock: respx.MockRouter,
|
|
) -> None:
|
|
monkeypatch.delenv("STRANDS_DECIDER_API_BASE", raising=False)
|
|
monkeypatch.delenv("STRANDS_DECIDER_API_KEY", raising=False)
|
|
provider_resolution: Final = litellm.get_llm_provider("strands_decider/strands-decider-2B-hobson-v19")
|
|
router: Final = litellm.Router(
|
|
model_list=[
|
|
{
|
|
"model_name": "strands",
|
|
"litellm_params": {
|
|
"model": "strands_decider/strands-decider-2B-hobson-v19",
|
|
"api_base": "https://strands.example",
|
|
},
|
|
}
|
|
]
|
|
)
|
|
route: Final = respx_mock.post("https://strands.example/v1/systemone").respond(json=_STRANDS_RESPONSE)
|
|
|
|
response: Final = await router.adecisions(
|
|
model="strands",
|
|
state="review",
|
|
questions={"severity": {"type": "score", "criteria": ["none", "low", "high"]}},
|
|
)
|
|
|
|
assert provider_resolution[:2] == ("strands-decider-2B-hobson-v19", "strands_decider")
|
|
assert route.called
|
|
assert response.model == _STRANDS_RESPONSE["model"]
|