litellm/tests/unit/harness/test_endpoint.py
ishaan-berri 0f6a06a6b9
feat: add litellm.agent() to run claude code, codex, opencode and deep agents through the ai gateway (#43885)
* feat(harness): add litellm/__init__.py

* feat(harness): add litellm/constants.py

* feat(harness): add litellm/harness/__init__.py

* feat(harness): add litellm/harness/adapters/__init__.py

* feat(harness): add litellm/harness/adapters/base.py

* feat(harness): add litellm/harness/adapters/claude_code.py

* feat(harness): add litellm/harness/adapters/codex.py

* feat(harness): add litellm/harness/adapters/opencode.py

* feat(harness): add litellm/harness/endpoint.py

* feat(harness): add litellm/harness/errors.py

* feat(harness): add litellm/harness/options.py

* feat(harness): add litellm/harness/runtime.py

* feat(harness): add litellm/harness/sandbox/__init__.py

* feat(harness): add litellm/harness/sandbox/base.py

* feat(harness): add litellm/harness/sandbox/docker.py

* feat(harness): add litellm/harness/sandbox/local.py

* feat(harness): add litellm/harness/sandbox/snapshot.py

* feat(harness): add litellm/harness/sync.py

* feat(harness): add litellm/harness/types.py

* feat(harness): add litellm/sandbox/__init__.py

* feat(harness): add README.md

* feat(harness): add tests/harness_e2e/__init__.py

* feat(harness): add tests/harness_e2e/conftest.py

* feat(harness): add tests/harness_e2e/test_harness_e2e.py

* feat(harness): add tests/test_litellm/harness/__init__.py

* feat(harness): add tests/test_litellm/harness/adapters/__init__.py

* feat(harness): add tests/test_litellm/harness/adapters/fixtures/claude_code/api_error.jsonl

* feat(harness): add tests/test_litellm/harness/adapters/fixtures/claude_code/max_turns.jsonl

* feat(harness): add tests/test_litellm/harness/adapters/fixtures/claude_code/resume_turn.jsonl

* feat(harness): add tests/test_litellm/harness/adapters/fixtures/claude_code/structured_output.jsonl

* feat(harness): add tests/test_litellm/harness/adapters/fixtures/claude_code/success_tools.jsonl

* feat(harness): add tests/test_litellm/harness/adapters/fixtures/codex/reasoning.jsonl

* feat(harness): add tests/test_litellm/harness/adapters/fixtures/codex/structured_output.jsonl

* feat(harness): add tests/test_litellm/harness/adapters/fixtures/codex/turn_failed.jsonl

* feat(harness): add tests/test_litellm/harness/adapters/fixtures/codex/turn1_bash.jsonl

* feat(harness): add tests/test_litellm/harness/adapters/fixtures/codex/turn2_resume_apply_patch.jsonl

* feat(harness): add tests/test_litellm/harness/adapters/fixtures/opencode/api_error.jsonl

* feat(harness): add tests/test_litellm/harness/adapters/fixtures/opencode/endpoint_requests.jsonl

* feat(harness): add tests/test_litellm/harness/adapters/fixtures/opencode/readonly_denied_bash.jsonl

* feat(harness): add tests/test_litellm/harness/adapters/fixtures/opencode/turn1_write_read.jsonl

* feat(harness): add tests/test_litellm/harness/adapters/fixtures/opencode/turn2_session_skill.jsonl

* feat(harness): add tests/test_litellm/harness/adapters/test_claude_code.py

* feat(harness): add tests/test_litellm/harness/adapters/test_codex.py

* feat(harness): add tests/test_litellm/harness/adapters/test_opencode.py

* feat(harness): add tests/test_litellm/harness/core_fakes.py

* feat(harness): add tests/test_litellm/harness/sandbox/__init__.py

* feat(harness): add tests/test_litellm/harness/sandbox/test_docker.py

* feat(harness): add tests/test_litellm/harness/sandbox/test_local.py

* feat(harness): add tests/test_litellm/harness/sandbox/test_snapshot.py

* feat(harness): add tests/test_litellm/harness/test_endpoint.py

* feat(harness): add tests/test_litellm/harness/test_init.py

* feat(harness): add tests/test_litellm/harness/test_runtime.py

* feat(harness): add tests/test_litellm/harness/test_sync.py

* feat(harness): add tests/test_litellm/harness/test_types.py

* test(harness): use word recall in stream e2e test

* refactor(harness): update litellm/__init__.py

* refactor(harness): update litellm/constants.py

* refactor(harness): update litellm/harness/__init__.py

* refactor(harness): remove litellm/harness/adapters/__init__.py

* refactor(harness): update litellm/harness/context.py

* refactor(harness): update litellm/harness/endpoint.py

* refactor(harness): update litellm/harness/handlers/__init__.py

* refactor(harness): update litellm/harness/handlers/base.py

* refactor(harness): update litellm/harness/handlers/cli_handler.py

* refactor(harness): update litellm/harness/handlers/deepagents_handler.py

* refactor(harness): update litellm/harness/runtime.py

* refactor(harness): update litellm/harness/sandbox/docker.py

* refactor(harness): update litellm/harness/sandbox/local.py

* refactor(harness): update litellm/harness/sync.py

* refactor(harness): update litellm/harness/types.py

* refactor(harness): update litellm/llms/base_llm/harness/__init__.py

* refactor(harness): update litellm/llms/base_llm/harness/transformation.py

* refactor(harness): update litellm/llms/base_llm/harness/utils.py

* refactor(harness): update litellm/llms/claude_code/__init__.py

* refactor(harness): update litellm/llms/claude_code/harness/__init__.py

* refactor(harness): update litellm/llms/claude_code/harness/transformation.py

* refactor(harness): update litellm/llms/codex/__init__.py

* refactor(harness): update litellm/llms/codex/harness/__init__.py

* refactor(harness): update litellm/llms/codex/harness/transformation.py

* refactor(harness): update litellm/llms/deepagents/__init__.py

* refactor(harness): update litellm/llms/deepagents/harness/__init__.py

* refactor(harness): update litellm/llms/deepagents/harness/sandbox_backend.py

* refactor(harness): update litellm/llms/deepagents/harness/transformation.py

* refactor(harness): update litellm/llms/opencode/__init__.py

* refactor(harness): update litellm/llms/opencode/harness/__init__.py

* refactor(harness): update litellm/llms/opencode/harness/transformation.py

* refactor(harness): update litellm/utils.py

* refactor(harness): update README.md

* refactor(harness): update tests/harness_e2e/conftest.py

* refactor(harness): update tests/harness_e2e/test_harness_e2e.py

* refactor(harness): remove tests/test_litellm/harness/__init__.py

* refactor(harness): remove tests/test_litellm/harness/adapters/__init__.py

* refactor(harness): remove tests/test_litellm/harness/adapters/fixtures/claude_code/api_error.jsonl

* refactor(harness): remove tests/test_litellm/harness/adapters/fixtures/claude_code/max_turns.jsonl

* refactor(harness): remove tests/test_litellm/harness/adapters/fixtures/claude_code/resume_turn.jsonl

* refactor(harness): remove tests/test_litellm/harness/adapters/fixtures/claude_code/structured_output.jsonl

* refactor(harness): remove tests/test_litellm/harness/adapters/fixtures/claude_code/success_tools.jsonl

* refactor(harness): remove tests/test_litellm/harness/adapters/fixtures/codex/reasoning.jsonl

* refactor(harness): remove tests/test_litellm/harness/adapters/fixtures/codex/structured_output.jsonl

* refactor(harness): remove tests/test_litellm/harness/adapters/fixtures/codex/turn_failed.jsonl

* refactor(harness): remove tests/test_litellm/harness/adapters/fixtures/codex/turn1_bash.jsonl

* refactor(harness): remove tests/test_litellm/harness/adapters/fixtures/codex/turn2_resume_apply_patch.jsonl

* refactor(harness): remove tests/test_litellm/harness/adapters/fixtures/opencode/api_error.jsonl

* refactor(harness): remove tests/test_litellm/harness/adapters/fixtures/opencode/endpoint_requests.jsonl

* refactor(harness): remove tests/test_litellm/harness/adapters/fixtures/opencode/readonly_denied_bash.jsonl

* refactor(harness): remove tests/test_litellm/harness/adapters/fixtures/opencode/turn1_write_read.jsonl

* refactor(harness): remove tests/test_litellm/harness/adapters/fixtures/opencode/turn2_session_skill.jsonl

* refactor(harness): remove tests/test_litellm/harness/adapters/test_claude_code.py

* refactor(harness): remove tests/test_litellm/harness/adapters/test_codex.py

* refactor(harness): remove tests/test_litellm/harness/adapters/test_opencode.py

* refactor(harness): remove tests/test_litellm/harness/core_fakes.py

* refactor(harness): remove tests/test_litellm/harness/sandbox/__init__.py

* refactor(harness): remove tests/test_litellm/harness/sandbox/test_docker.py

* refactor(harness): remove tests/test_litellm/harness/sandbox/test_local.py

* refactor(harness): remove tests/test_litellm/harness/sandbox/test_snapshot.py

* refactor(harness): remove tests/test_litellm/harness/test_endpoint.py

* refactor(harness): remove tests/test_litellm/harness/test_init.py

* refactor(harness): remove tests/test_litellm/harness/test_runtime.py

* refactor(harness): remove tests/test_litellm/harness/test_sync.py

* refactor(harness): remove tests/test_litellm/harness/test_types.py

* refactor(harness): update tests/unit/harness/__init__.py

* refactor(harness): update tests/unit/harness/core_fakes.py

* refactor(harness): update tests/unit/harness/handlers/__init__.py

* refactor(harness): update tests/unit/harness/handlers/test_deepagents_handler.py

* refactor(harness): update tests/unit/harness/sandbox/__init__.py

* refactor(harness): update tests/unit/harness/sandbox/test_docker.py

* refactor(harness): update tests/unit/harness/sandbox/test_local.py

* refactor(harness): update tests/unit/harness/sandbox/test_snapshot.py

* refactor(harness): update tests/unit/harness/test_endpoint.py

* refactor(harness): update tests/unit/harness/test_init.py

* refactor(harness): update tests/unit/harness/test_runtime.py

* refactor(harness): update tests/unit/harness/test_sync.py

* refactor(harness): update tests/unit/harness/test_types.py

* refactor(harness): update tests/unit/llms/claude_code/__init__.py

* refactor(harness): update tests/unit/llms/claude_code/harness/__init__.py

* refactor(harness): update tests/unit/llms/claude_code/harness/fixtures/api_error.jsonl

* refactor(harness): update tests/unit/llms/claude_code/harness/fixtures/max_turns.jsonl

* refactor(harness): update tests/unit/llms/claude_code/harness/fixtures/resume_turn.jsonl

* refactor(harness): update tests/unit/llms/claude_code/harness/fixtures/structured_output.jsonl

* refactor(harness): update tests/unit/llms/claude_code/harness/fixtures/success_tools.jsonl

* refactor(harness): update tests/unit/llms/claude_code/harness/test_transformation.py

* refactor(harness): update tests/unit/llms/codex/__init__.py

* refactor(harness): update tests/unit/llms/codex/harness/__init__.py

* refactor(harness): update tests/unit/llms/codex/harness/fixtures/reasoning.jsonl

* refactor(harness): update tests/unit/llms/codex/harness/fixtures/structured_output.jsonl

* refactor(harness): update tests/unit/llms/codex/harness/fixtures/turn1_bash.jsonl

* refactor(harness): update tests/unit/llms/codex/harness/fixtures/turn2_resume_apply_patch.jsonl

* refactor(harness): update tests/unit/llms/codex/harness/fixtures/turn_failed.jsonl

* refactor(harness): update tests/unit/llms/codex/harness/test_transformation.py

* refactor(harness): update tests/unit/llms/deepagents/__init__.py

* refactor(harness): update tests/unit/llms/deepagents/harness/__init__.py

* refactor(harness): update tests/unit/llms/deepagents/harness/test_transformation.py

* refactor(harness): update tests/unit/llms/opencode/__init__.py

* refactor(harness): update tests/unit/llms/opencode/harness/__init__.py

* refactor(harness): update tests/unit/llms/opencode/harness/fixtures/api_error.jsonl

* refactor(harness): update tests/unit/llms/opencode/harness/fixtures/endpoint_requests.jsonl

* refactor(harness): update tests/unit/llms/opencode/harness/fixtures/readonly_denied_bash.jsonl

* refactor(harness): update tests/unit/llms/opencode/harness/fixtures/turn1_write_read.jsonl

* refactor(harness): update tests/unit/llms/opencode/harness/fixtures/turn2_session_skill.jsonl

* refactor(harness): update tests/unit/llms/opencode/harness/test_transformation.py

* ci: allowlist tests/harness_e2e, which needs live runtimes and a gateway

* fix(harness): bound the turn event queue

* fix(harness): bound the turn event queue with backpressure

* style: sort imports in utils

* ci: exclude agent-harness config folders from provider docs check

* refactor(harness): update tests/harness_e2e/conftest.py

* refactor(harness): update tests/harness_e2e/test_harness_e2e.py

* refactor(harness): update tests/unit/harness/core_fakes.py

* refactor(harness): update tests/unit/harness/handlers/test_deepagents_handler.py

* refactor(harness): update tests/unit/harness/test_runtime.py

* refactor(harness): update tests/unit/harness/test_sync.py

* refactor(harness): update tests/unit/llms/deepagents/harness/test_transformation.py

* refactor(harness): update tests/unit/llms/opencode/harness/test_transformation.py

* refactor(harness): update litellm/harness/endpoint.py

* refactor(harness): update litellm/harness/handlers/deepagents_handler.py

* refactor(harness): update litellm/harness/options.py

* refactor(harness): update litellm/harness/runtime.py

* refactor(harness): update litellm/harness/sandbox/snapshot.py

* refactor(harness): update litellm/llms/base_llm/harness/transformation.py

* refactor(harness): update litellm/llms/base_llm/harness/utils.py

* refactor(harness): update litellm/llms/claude_code/harness/transformation.py

* refactor(harness): update litellm/llms/codex/harness/transformation.py

* refactor(harness): update litellm/llms/deepagents/harness/sandbox_backend.py

* refactor(harness): update litellm/llms/deepagents/harness/transformation.py

* refactor(harness): update litellm/llms/opencode/harness/transformation.py

* refactor(harness): update litellm/sandbox/__init__.py

* refactor(harness): update tests/unit/llms/claude_code/harness/test_transformation.py

* refactor(harness): update tests/unit/llms/deepagents/harness/test_sandbox_backend_symlinks.py

* refactor(harness): update tests/unit/llms/opencode/harness/test_transformation.py

* refactor(harness): update litellm/llms/base_llm/harness/utils.py

* refactor(harness): update litellm/llms/codex/harness/transformation.py

* refactor(harness): update tests/code_coverage_tests/recursive_detector.py

* refactor(harness): update tests/unit/llms/base_llm/harness/__init__.py

* refactor(harness): update tests/unit/llms/claude_code/harness/fixtures/__init__.py

* refactor(harness): update tests/unit/llms/codex/harness/fixtures/__init__.py

* refactor(harness): update tests/unit/llms/opencode/harness/fixtures/__init__.py

* refactor(harness): update litellm/harness/__init__.py

* refactor(harness): update litellm/harness/context.py

* refactor(harness): update litellm/harness/endpoint.py

* refactor(harness): update litellm/harness/handlers/__init__.py

* refactor(harness): update litellm/harness/handlers/base.py

* refactor(harness): update litellm/harness/handlers/cli_handler.py

* refactor(harness): update litellm/harness/handlers/deepagents_handler.py

* refactor(harness): update litellm/harness/runtime.py

* refactor(harness): update litellm/harness/sandbox/__init__.py

* refactor(harness): update litellm/harness/sandbox/base.py

* refactor(harness): update litellm/harness/sandbox/docker.py

* refactor(harness): update litellm/harness/sandbox/local.py

* refactor(harness): update litellm/harness/sandbox/snapshot.py

* refactor(harness): update litellm/harness/sync.py

* refactor(harness): update litellm/harness/types.py

* refactor(harness): update litellm/llms/base_llm/harness/transformation.py

* refactor(harness): update litellm/llms/base_llm/harness/utils.py

* refactor(harness): update litellm/llms/claude_code/harness/transformation.py

* refactor(harness): update litellm/llms/codex/harness/transformation.py

* refactor(harness): update litellm/llms/deepagents/harness/sandbox_backend.py

* refactor(harness): update litellm/llms/deepagents/harness/transformation.py

* refactor(harness): update litellm/llms/opencode/harness/transformation.py

* refactor(harness): update litellm/types/llms/custom_http.py

* refactor(harness): update tests/unit/harness/test_endpoint.py

* refactor(harness): update litellm/harness/runtime.py

* refactor(harness): update litellm/llms/deepagents/harness/sandbox_backend.py

* refactor(harness): update tests/unit/harness/test_init.py

* refactor(harness): update tests/unit/harness/test_runtime.py

* refactor(harness): update tests/unit/llms/deepagents/harness/test_sandbox_backend_symlinks.py
2026-10-01 22:27:49 +00:00

359 lines
14 KiB
Python

import json
import sys
from collections.abc import AsyncIterator
from typing import Any
import httpx
import pytest
import litellm
from litellm.harness import endpoint as endpoint_module
from litellm.harness.endpoint import (
ModelEndpoint,
SSEUsageParser,
UsageTracker,
compute_cost,
usage_from_body,
)
from litellm.harness.errors import HarnessInstallFailed
from litellm.harness.context import GatewayTarget
from litellm.harness.types import Harness, Usage
from litellm.types.utils import ModelResponse, ModelResponseStream
GATEWAY = GatewayTarget(api_base="https://gw.example.com", api_key="sk-gateway-secret")
ANTHROPIC_SSE = (
b"event: message_start\n"
b'data: {"type":"message_start","message":{"usage":{"input_tokens":11,"output_tokens":1}}}\n\n'
b"event: content_block_delta\n"
b'data: {"type":"content_block_delta","delta":{"type":"text_delta","text":"hi"}}\n\n'
b"event: message_delta\n"
b'data: {"type":"message_delta","usage":{"output_tokens":7}}\n\n'
b"event: message_stop\n"
b'data: {"type":"message_stop"}\n\n'
)
CHAT_SSE = (
b'data: {"choices":[{"delta":{"content":"hi"}}]}\n\n'
b'data: {"choices":[],"usage":{"prompt_tokens":5,"completion_tokens":3}}\n\n'
b"data: [DONE]\n\n"
)
RESPONSES_SSE = (
b"event: response.output_text.delta\n"
b'data: {"type":"response.output_text.delta","delta":"hi"}\n\n'
b"event: response.completed\n"
b'data: {"type":"response.completed","response":{"usage":{"input_tokens":20,"output_tokens":4}}}\n\n'
)
class Recorder:
def __init__(self, response: httpx.Response) -> None:
self.response = response
self.requests: list[httpx.Request] = []
def __call__(self, request: httpx.Request) -> httpx.Response:
self.requests.append(request)
return self.response
def sse_response(body: bytes, headers: dict[str, str] | None = None) -> httpx.Response:
return httpx.Response(
200,
content=body,
headers={"content-type": "text/event-stream", **(headers or {})},
)
def gateway_endpoint(recorder: Recorder, **kwargs: Any) -> ModelEndpoint:
return ModelEndpoint(
Harness.CLAUDE_CODE,
kwargs.pop("model", "claude-sonnet"),
GATEWAY,
client=httpx.AsyncClient(transport=httpx.MockTransport(recorder)),
**kwargs,
)
def auth(ep: ModelEndpoint) -> dict[str, str]:
return {"authorization": f"Bearer {ep.token}"}
@pytest.fixture(autouse=True)
def no_real_cost(monkeypatch: pytest.MonkeyPatch) -> None:
monkeypatch.setattr(
litellm,
"cost_per_token",
lambda model, prompt_tokens, completion_tokens: (
prompt_tokens * 0.001,
completion_tokens * 0.002,
),
)
async def test_rejects_bad_token_and_accepts_both_header_styles() -> None:
recorder = Recorder(httpx.Response(200, json={"usage": {}}))
async with gateway_endpoint(recorder) as ep:
assert ep.url == f"http://127.0.0.1:{ep.port}" and ep.port > 0
async with httpx.AsyncClient(base_url=ep.url) as client:
missing = await client.post("/v1/messages", json={})
wrong = await client.post(
"/v1/messages", json={}, headers={"x-api-key": "nope"}
)
bearer = await client.post("/v1/messages", json={}, headers=auth(ep))
api_key = await client.post(
"/messages", json={}, headers={"x-api-key": ep.token}
)
assert missing.status_code == 401
assert wrong.status_code == 401
assert "error" in wrong.json()
assert bearer.status_code == 200
assert api_key.status_code == 200
assert len(recorder.requests) == 2
async def test_gateway_rewrites_headers_and_model() -> None:
recorder = Recorder(
httpx.Response(
200,
json={"id": "m", "usage": {"input_tokens": 3, "output_tokens": 2}},
)
)
async with gateway_endpoint(recorder, metadata={"run": "abc"}) as ep:
async with httpx.AsyncClient(base_url=ep.url) as client:
resp = await client.post(
"/v1/messages",
json={"model": "whatever", "max_tokens": 5},
headers={
"x-api-key": ep.token,
"anthropic-version": "2023-06-01",
"anthropic-beta": "tools-2024",
},
)
assert resp.status_code == 200
sent = recorder.requests[0]
assert str(sent.url) == "https://gw.example.com/v1/messages"
assert sent.headers["authorization"] == "Bearer sk-gateway-secret"
assert "x-api-key" not in sent.headers
assert sent.headers["x-litellm-tags"] == "harness,claude_code"
assert json.loads(sent.headers["x-litellm-spend-logs-metadata"]) == {"run": "abc"}
assert sent.headers["anthropic-version"] == "2023-06-01"
assert sent.headers["anthropic-beta"] == "tools-2024"
assert json.loads(sent.content)["model"] == "claude-sonnet"
assert ep.usage.input_tokens == 3 and ep.usage.output_tokens == 2
assert ep.usage.calls == 1
@pytest.mark.parametrize(
"path,body,expected",
[
("/v1/messages", ANTHROPIC_SSE, (11, 7)),
("/v1/chat/completions", CHAT_SSE, (5, 3)),
("/responses", RESPONSES_SSE, (20, 4)),
],
)
async def test_gateway_sse_passthrough_and_usage(
path: str, body: bytes, expected: tuple[int, int]
) -> None:
recorder = Recorder(sse_response(body))
async with gateway_endpoint(recorder) as ep:
async with httpx.AsyncClient(base_url=ep.url) as client:
resp = await client.post(path, json={"stream": True}, headers=auth(ep))
assert resp.status_code == 200
assert resp.headers["content-type"].startswith("text/event-stream")
assert resp.content == body
assert (ep.usage.input_tokens, ep.usage.output_tokens) == expected
expected_cost = expected[0] * 0.001 + expected[1] * 0.002
assert ep.usage.cost == pytest.approx(expected_cost)
async def test_cost_header_preferred_over_computed() -> None:
recorder = Recorder(
httpx.Response(
200,
json={"usage": {"prompt_tokens": 100, "completion_tokens": 100}},
headers={"x-litellm-response-cost": "0.42"},
)
)
async with gateway_endpoint(recorder) as ep:
async with httpx.AsyncClient(base_url=ep.url) as client:
await client.post("/v1/chat/completions", json={}, headers=auth(ep))
assert ep.usage.cost == pytest.approx(0.42)
assert ep.usage.snapshot() == Usage(input_tokens=100, output_tokens=100, calls=1)
async def test_gateway_error_status_preserved_and_not_counted() -> None:
recorder = Recorder(httpx.Response(429, json={"error": "rate limited"}))
async with gateway_endpoint(recorder) as ep:
async with httpx.AsyncClient(base_url=ep.url) as client:
resp = await client.post("/v1/chat/completions", json={}, headers=auth(ep))
assert resp.status_code == 429
assert ep.usage.calls == 0
async def test_models_route() -> None:
recorder = Recorder(httpx.Response(200))
async with gateway_endpoint(recorder) as ep:
async with httpx.AsyncClient(base_url=ep.url) as client:
with_model = await client.get("/v1/models", headers=auth(ep))
unauth = await client.get("/models")
assert unauth.status_code == 401
assert with_model.json()["object"] == "list"
assert [m["id"] for m in with_model.json()["data"]] == ["claude-sonnet"]
async with ModelEndpoint(Harness.CODEX, None, None) as ep:
async with httpx.AsyncClient(base_url=ep.url) as client:
empty = await client.get("/models", headers=auth(ep))
assert empty.json() == {"object": "list", "data": []}
async def test_sdk_chat_non_stream(monkeypatch: pytest.MonkeyPatch) -> None:
calls: list[dict[str, Any]] = []
async def fake_acompletion(**kwargs: Any) -> ModelResponse:
calls.append(kwargs)
response = ModelResponse(
model="gpt-x",
choices=[{"message": {"role": "assistant", "content": "hello"}}],
usage={"prompt_tokens": 9, "completion_tokens": 4, "total_tokens": 13},
)
response._hidden_params["response_cost"] = 0.5
return response
monkeypatch.setattr(litellm, "acompletion", fake_acompletion)
async with ModelEndpoint(
Harness.OPENCODE, "openai/gpt-x", None, api_key="sk-real", api_base="https://x"
) as ep:
async with httpx.AsyncClient(base_url=ep.url) as client:
resp = await client.post(
"/v1/chat/completions",
json={
"model": "ignored",
"messages": [{"role": "user", "content": "hi"}],
},
headers=auth(ep),
)
assert resp.status_code == 200
assert resp.json()["choices"][0]["message"]["content"] == "hello"
assert calls[0]["model"] == "openai/gpt-x"
assert calls[0]["api_key"] == "sk-real"
assert calls[0]["api_base"] == "https://x"
assert (ep.usage.input_tokens, ep.usage.output_tokens) == (9, 4)
assert ep.usage.cost == pytest.approx(0.5)
async def fake_chat_stream() -> AsyncIterator[ModelResponseStream]:
yield ModelResponseStream(choices=[{"delta": {"content": "he"}}])
yield ModelResponseStream(choices=[{"delta": {"content": "llo"}}])
final = ModelResponseStream(choices=[])
final.usage = litellm.Usage(prompt_tokens=6, completion_tokens=2, total_tokens=8)
yield final
async def test_sdk_chat_stream(monkeypatch: pytest.MonkeyPatch) -> None:
calls: list[dict[str, Any]] = []
async def fake_acompletion(**kwargs: Any) -> AsyncIterator[ModelResponseStream]:
calls.append(kwargs)
return fake_chat_stream()
monkeypatch.setattr(litellm, "acompletion", fake_acompletion)
async with ModelEndpoint(Harness.OPENCODE, "openai/gpt-x", None) as ep:
async with httpx.AsyncClient(base_url=ep.url) as client:
resp = await client.post(
"/chat/completions",
json={"messages": [], "stream": True},
headers=auth(ep),
)
assert resp.headers["content-type"].startswith("text/event-stream")
lines = [line for line in resp.text.split("\n") if line.startswith("data: ")]
assert lines[-1] == "data: [DONE]"
assert json.loads(lines[0][6:])["choices"][0]["delta"]["content"] == "he"
assert calls[0]["stream_options"] == {"include_usage": True}
assert (ep.usage.input_tokens, ep.usage.output_tokens) == (6, 2)
assert ep.usage.cost == pytest.approx(6 * 0.001 + 2 * 0.002)
async def fake_anthropic_stream() -> AsyncIterator[Any]:
yield {"type": "message_start", "message": {"usage": {"input_tokens": 4}}}
yield b'event: message_delta\ndata: {"type":"message_delta","usage":{"output_tokens":9}}\n\n'
async def test_sdk_messages_stream_handles_dicts_and_bytes(
monkeypatch: pytest.MonkeyPatch,
) -> None:
async def fake_acreate(**kwargs: Any) -> AsyncIterator[Any]:
return fake_anthropic_stream()
monkeypatch.setattr(litellm.anthropic.messages, "acreate", fake_acreate)
async with ModelEndpoint(Harness.CLAUDE_CODE, "anthropic/claude", None) as ep:
async with httpx.AsyncClient(base_url=ep.url) as client:
resp = await client.post(
"/v1/messages", json={"stream": True}, headers=auth(ep)
)
assert "event: message_start" in resp.text
assert "event: message_delta" in resp.text
assert (ep.usage.input_tokens, ep.usage.output_tokens) == (4, 9)
async def test_sdk_error_is_sanitized(monkeypatch: pytest.MonkeyPatch) -> None:
async def failing(**kwargs: Any) -> Any:
raise litellm.RateLimitError(
message="too many requests for key sk-real",
llm_provider="openai",
model="gpt-x",
)
monkeypatch.setattr(litellm, "aresponses", failing)
async with ModelEndpoint(Harness.CODEX, "gpt-x", None, api_key="sk-real") as ep:
async with httpx.AsyncClient(base_url=ep.url) as client:
resp = await client.post("/v1/responses", json={}, headers=auth(ep))
assert resp.status_code == 429
assert "sk-real" not in resp.text
assert resp.json()["error"]["type"] == "RateLimitError"
async def test_missing_server_deps_raises_install_failed(
monkeypatch: pytest.MonkeyPatch,
) -> None:
def missing() -> Any:
raise HarnessInstallFailed(endpoint_module.MISSING_DEPS_MESSAGE)
monkeypatch.setattr(endpoint_module, "_load_server_deps", missing)
with pytest.raises(HarnessInstallFailed, match="pip install starlette uvicorn"):
async with ModelEndpoint(Harness.CODEX, None, None):
pass
def test_load_server_deps_maps_import_error(monkeypatch: pytest.MonkeyPatch) -> None:
monkeypatch.setitem(sys.modules, "uvicorn", None)
with pytest.raises(HarnessInstallFailed, match="starlette and uvicorn"):
endpoint_module._load_server_deps()
def test_usage_helpers() -> None:
assert usage_from_body({"usage": {"prompt_tokens": 1, "completion_tokens": 2}}) == (
1,
2,
)
assert usage_from_body({"response": {"usage": {"input_tokens": 3}}}) == (3, 0)
assert usage_from_body("nope") == (0, 0)
parser = SSEUsageParser()
for i in range(0, len(ANTHROPIC_SSE), 7): # split across arbitrary chunk borders
parser.feed(ANTHROPIC_SSE[i : i + 7])
parser.close()
assert (parser.input_tokens, parser.output_tokens) == (11, 7)
tracker = UsageTracker()
tracker.add(1, 2, 0.1)
tracker.add(3, 4, 0.2)
assert tracker.snapshot() == Usage(input_tokens=4, output_tokens=6, calls=2)
assert tracker.cost == pytest.approx(0.3)
def test_compute_cost_never_raises(monkeypatch: pytest.MonkeyPatch) -> None:
def boom(**kwargs: Any) -> Any:
raise ValueError("unknown model")
monkeypatch.setattr(litellm, "cost_per_token", boom)
assert compute_cost("mystery", 10, 10) == 0.0
assert compute_cost(None, 10, 10) == 0.0