ci(test): replace Railway-hosted mock endpoint with in-repo stdlib server

The exampleopenaiendpoint Railway deployment is a single point of failure
for ~30 tests across 8 CI jobs. Today's outage took the proxy/router/utils
suites offline despite the tests having nothing to do with that endpoint
itself.

Add tests/mock_endpoints/openai_mock_server.py — a stdlib-only HTTP server
that returns static OpenAI-format responses (chat completions incl. SSE,
text completions, embeddings, rerank, Anthropic messages). No new
dependencies; starts in <1s and handles ≥300 concurrent requests in the
self-test.

Wire it into the 8 affected CI jobs via a new start_mock_openai_server
CircleCI command (binds 0.0.0.0:8090). The proxy containers reach it via
host.docker.internal:8090 through the existing --add-host flag; the
direct-pytest jobs reach it on 127.0.0.1:8090. Update affected YAML
configs and test files to point at the new URLs. The single
intentionally-bad URL ("…appxxxx/") becomes http://bad.invalid/ so DNS
failure still tests the fallback path predictably.
This commit is contained in:
Yuneng Jiang 2026-05-19 18:24:41 -07:00
parent e59e34bed3
commit 84d3a807b9
No known key found for this signature in database
16 changed files with 354 additions and 32 deletions

View file

@ -111,6 +111,22 @@ commands:
- wait_for_service:
url: tcp://localhost:6379
timeout: "60"
start_mock_openai_server:
description: "Start the in-repo mock OpenAI server so tests and proxy configs don't depend on an external endpoint. Binds 0.0.0.0:8090 by default — reachable as 127.0.0.1:8090 from the same host and as host.docker.internal:8090 from sibling docker containers."
parameters:
port:
type: string
default: "8090"
steps:
- run:
name: Start mock OpenAI server
background: true
command: |
python3 -m tests.mock_endpoints.openai_mock_server \
--host 0.0.0.0 --port << parameters.port >>
- wait_for_service:
url: http://127.0.0.1:<< parameters.port >>/
timeout: "30"
setup_litellm_enterprise_pip:
steps:
- run:
@ -452,6 +468,7 @@ jobs:
key: v1-uv-cache-{{ checksum "uv.lock" }}
# Run pytest and generate JUnit XML report
- setup_litellm_enterprise_pip
- start_mock_openai_server
- run:
name: Run tests
command: |
@ -984,6 +1001,7 @@ jobs:
name: Install Dependencies
command: |
uv sync --frozen --all-groups --all-extras --python 3.12
- start_mock_openai_server
# Run pytest and generate JUnit XML report
- run:
name: Run tests
@ -1387,6 +1405,7 @@ jobs:
uv sync --frozen --all-groups --all-extras --python 3.12
- start_postgres:
db_name: litellm_test
- start_mock_openai_server
- attach_workspace:
at: ~/project
- run:
@ -1478,6 +1497,7 @@ jobs:
command: |
uv sync --frozen --all-groups --all-extras --python 3.12
- start_postgres
- start_mock_openai_server
- run:
name: Load Docker Database Image
command: |
@ -1643,6 +1663,7 @@ jobs:
command: |
uv sync --frozen --all-groups --all-extras --python 3.12
- start_postgres
- start_mock_openai_server
- attach_workspace:
at: ~/project
- run:
@ -1770,6 +1791,7 @@ jobs:
uv sync --frozen --all-groups --all-extras --python 3.12
- start_postgres
- start_redis
- start_mock_openai_server
- attach_workspace:
at: ~/project
- run:
@ -1852,6 +1874,7 @@ jobs:
command: |
uv sync --frozen --all-groups --all-extras --python 3.12
- start_postgres
- start_mock_openai_server
- attach_workspace:
at: ~/project
- run:
@ -2019,6 +2042,7 @@ jobs:
command: |
docker build -t my-app:latest -f docker/build_from_pip/Dockerfile.build_from_pip .
- start_postgres
- start_mock_openai_server
- run:
name: Run Docker container
# intentionally give bad redis credentials here

View file

@ -3,7 +3,7 @@ model_list:
litellm_params:
model: openai/fake
api_key: fake-key
api_base: https://exampleopenaiendpoint-production.up.railway.app/
api_base: http://host.docker.internal:8090/
general_settings:
alerting: ["slack"]

View file

@ -3,12 +3,12 @@ model_list:
litellm_params:
model: openai/fake
api_key: fake-key
api_base: https://exampleopenaiendpoint-production.up.railway.app/
api_base: http://host.docker.internal:8090/
- model_name: gpt-4
litellm_params:
model: openai/gpt-4
api_key: fake-key
api_base: https://exampleopenaiendpoint-production.up.railway.app/
api_base: http://host.docker.internal:8090/
litellm_settings:
callbacks: ["gcs_bucket"]

View file

@ -3,7 +3,7 @@ model_list:
litellm_params:
model: openai/fake
api_key: fake-key
api_base: https://exampleopenaiendpoint-production.up.railway.app/
api_base: http://host.docker.internal:8090/
tags: ["teamA"]
model_info:
id: "team-a-model"

View file

@ -3,7 +3,7 @@ model_list:
litellm_params:
model: openai/my-fake-model
api_key: my-fake-key
api_base: https://exampleopenaiendpoint-production.up.railway.app/
api_base: http://host.docker.internal:8090/
litellm_settings:
cache: True

View file

@ -3,7 +3,7 @@ model_list:
litellm_params:
model: openai/gpt-5-mini
api_key: fake-key
api_base: https://exampleopenaiendpoint-production.up.railway.app/
api_base: http://host.docker.internal:8090/
tags: ["teamA"]
model_info:
id: "team-a-model"
@ -11,7 +11,7 @@ model_list:
litellm_params:
model: openai/gpt-5-mini
api_key: fake-key
api_base: https://exampleopenaiendpoint-production.up.railway.app/
api_base: http://host.docker.internal:8090/
tags: ["teamB"]
model_info:
id: "team-b-model"
@ -23,7 +23,7 @@ model_list:
litellm_params:
model: openai/429
api_key: fake-key
api_base: https://exampleopenaiendpoint-production.up.railway.app
api_base: http://host.docker.internal:8090
- model_name: llava-hf
litellm_params:
model: openai/llava-hf/llava-v1.6-vicuna-7b-hf
@ -34,12 +34,12 @@ model_list:
- model_name: bedrock/*
litellm_params:
model: bedrock/*
api_base: https://exampleopenaiendpoint-production.up.railway.app/
api_base: http://host.docker.internal:8090/
- model_name: openai/*
litellm_params:
model: openai/*
api_key: os.environ/OPENAI_API_KEY
api_base: https://exampleopenaiendpoint-production.up.railway.app/
api_base: http://host.docker.internal:8090/
litellm_settings:

View file

@ -3,7 +3,7 @@ model_list:
litellm_params:
model: openai/gpt-5-mini
api_key: fake-key
api_base: https://exampleopenaiendpoint-production.up.railway.app/
api_base: http://host.docker.internal:8090/
general_settings:
use_redis_transaction_buffer: true

View file

@ -48,39 +48,39 @@ model_list:
litellm_params:
model: openai/gpt-5-mini
api_key: fake-key
api_base: https://exampleopenaiendpoint-production.up.railway.app/
api_base: http://host.docker.internal:8090/
- model_name: fake-openai-endpoint-2
litellm_params:
model: openai/my-fake-model
api_key: my-fake-key
api_base: https://exampleopenaiendpoint-production.up.railway.app/
api_base: http://host.docker.internal:8090/
stream_timeout: 0.001
rpm: 1
- model_name: fake-openai-endpoint-3
litellm_params:
model: openai/my-fake-model
api_key: my-fake-key
api_base: https://exampleopenaiendpoint-production.up.railway.app/
api_base: http://host.docker.internal:8090/
stream_timeout: 0.001
rpm: 1000
- model_name: fake-openai-endpoint-4
litellm_params:
model: openai/my-fake-model
api_key: my-fake-key
api_base: https://exampleopenaiendpoint-production.up.railway.app/
api_base: http://host.docker.internal:8090/
num_retries: 50
- model_name: fake-openai-endpoint-3
litellm_params:
model: openai/my-fake-model-2
api_key: my-fake-key
api_base: https://exampleopenaiendpoint-production.up.railway.app/
api_base: http://host.docker.internal:8090/
stream_timeout: 0.001
rpm: 1000
- model_name: bad-model
litellm_params:
model: openai/bad-model
api_key: os.environ/OPENAI_API_KEY
api_base: https://exampleopenaiendpoint-production.up.railway.app/
api_base: http://host.docker.internal:8090/
mock_timeout: True
timeout: 60
rpm: 1000
@ -90,7 +90,7 @@ model_list:
litellm_params:
model: openai/bad-model
api_key: os.environ/OPENAI_API_KEY
api_base: https://exampleopenaiendpoint-production.up.railway.app/
api_base: http://host.docker.internal:8090/
rpm: 1000
model_info:
health_check_timeout: 1
@ -132,13 +132,13 @@ model_list:
litellm_params:
model: openai/my-fake-model
api_key: my-fake-key
api_base: https://exampleopenaiendpoint-production.up.railway.app/
api_base: http://host.docker.internal:8090/
timeout: 1
- model_name: badly-configured-openai-endpoint
litellm_params:
model: openai/my-fake-model
api_key: my-fake-key
api_base: https://exampleopenaiendpoint-production.up.railway.appxxxx/
api_base: http://bad.invalid/
- model_name: gemini-2.5-flash
litellm_params:
model: gemini/gemini-2.5-flash

View file

@ -586,7 +586,7 @@ async def test_health_check_bad_model():
"model_name": "openai-gpt-4o",
"litellm_params": {
"api_key": "sk-1234",
"api_base": "https://exampleopenaiendpoint-production.up.railway.app",
"api_base": "http://127.0.0.1:8090",
"model": "openai/my-fake-openai-endpoint",
"mock_timeout": True,
"timeout": 60,

View file

@ -164,7 +164,7 @@ async def test_litellm_overhead_stream(model):
# Specific cases for models
#########################################################
if model == "openai/self_hosted":
kwargs["api_base"] = "https://exampleopenaiendpoint-production.up.railway.app/"
kwargs["api_base"] = "http://127.0.0.1:8090/"
# warmup call for auth validation on vertex_ai models
await litellm.acompletion(**kwargs)

View file

@ -144,14 +144,14 @@ async def test_router_provider_wildcard_routing_regex():
"model_name": "openai/fo::*:static::*",
"litellm_params": {
"model": "openai/fo::*:static::*",
"api_base": "https://exampleopenaiendpoint-production.up.railway.app/",
"api_base": "http://127.0.0.1:8090/",
},
},
{
"model_name": "openai/foo3::hello::*",
"litellm_params": {
"model": "openai/foo3::hello::*",
"api_base": "https://exampleopenaiendpoint-production.up.railway.app/",
"api_base": "http://127.0.0.1:8090/",
},
},
]
@ -1639,7 +1639,7 @@ async def test_router_text_completion_client():
"litellm_params": {
"model": "text-completion-openai/gpt-3.5-turbo-instruct",
"api_key": os.getenv("OPENAI_API_KEY", None),
"api_base": "https://exampleopenaiendpoint-production.up.railway.app/",
"api_base": "http://127.0.0.1:8090/",
},
}
]

View file

@ -26,7 +26,7 @@ def _create_router():
"model_name": "azure-model",
"litellm_params": {
"model": "openai/very-special-endpoint",
"api_base": "https://exampleopenaiendpoint-production.up.railway.app/",
"api_base": "http://127.0.0.1:8090/",
"api_key": "fake-key",
},
"model_info": {"id": "very-special-endpoint"},
@ -35,7 +35,7 @@ def _create_router():
"model_name": "azure-model",
"litellm_params": {
"model": "openai/fast-endpoint",
"api_base": "https://exampleopenaiendpoint-production.up.railway.app/",
"api_base": "http://127.0.0.1:8090/",
"api_key": "fake-key",
},
"model_info": {"id": "fast-endpoint"},

View file

@ -68,7 +68,7 @@ def create_test_router_2():
"litellm_params": {
"model": "openai/fake-openai-endpoint-2",
"api_key": "working-key-since-this-is-fake-endpoint",
"api_base": "https://exampleopenaiendpoint-production.up.railway.app/",
"api_base": "http://127.0.0.1:8090/",
},
},
],
@ -310,5 +310,5 @@ async def test_multiple_fallbacks(function_name):
assert (
result._hidden_params["api_base"]
== "https://exampleopenaiendpoint-production.up.railway.app/"
== "http://127.0.0.1:8090/"
)

View file

@ -1345,7 +1345,7 @@ def test_router_fallbacks_with_custom_model_costs():
"model": "openai/claude-sonnet-4-5-20250929",
"input_cost_per_token": 0.000003, # 3$/M
"output_cost_per_token": 0.000015, # 15$/M
"api_base": "https://exampleopenaiendpoint-production.up.railway.app",
"api_base": "http://127.0.0.1:8090",
"api_key": "my-fake-key",
"mock_response": "Hello! How can I help you today?",
},
@ -1594,14 +1594,14 @@ async def test_router_attempted_fallbacks_in_response(expected_attempted_fallbac
"litellm_params": {
"model": "openai/working-fake-endpoint",
"api_key": "my-fake-key",
"api_base": "https://exampleopenaiendpoint-production.up.railway.app",
"api_base": "http://127.0.0.1:8090",
},
},
{
"model_name": "badly-configured-openai-endpoint",
"litellm_params": {
"model": "openai/my-fake-model",
"api_base": "https://exampleopenaiendpoint-production.up.railway.appzzzzz",
"api_base": "http://bad.invalid/",
},
},
],

View file

View file

@ -0,0 +1,298 @@
"""
In-repo replacement for the Railway-hosted exampleopenaiendpoint that LiteLLM's
CI used to depend on. Returns static OpenAI-format responses so tests can run
without an external dependency.
Run standalone:
python -m tests.mock_endpoints.openai_mock_server --host 0.0.0.0 --port 8090
The server is intentionally implemented with the Python standard library only
(no FastAPI / httpx / pydantic) so CI jobs can start it before installing test
dependencies.
"""
from __future__ import annotations
import argparse
import json
import logging
import time
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
logger = logging.getLogger("openai_mock_server")
_EMBEDDING_DIM = 1536
def _now() -> int:
return int(time.time())
def _chat_completion_body(
model: str, content: str = "Hi! This is a mock response."
) -> dict:
return {
"id": "chatcmpl-mock-0001",
"object": "chat.completion",
"created": _now(),
"model": model,
"system_fingerprint": "fp_mock",
"choices": [
{
"index": 0,
"message": {"role": "assistant", "content": content},
"logprobs": None,
"finish_reason": "stop",
}
],
"usage": {
"prompt_tokens": 10,
"completion_tokens": 20,
"total_tokens": 30,
"prompt_tokens_details": {"cached_tokens": 0, "audio_tokens": 0},
"completion_tokens_details": {
"reasoning_tokens": 0,
"audio_tokens": 0,
"accepted_prediction_tokens": 0,
"rejected_prediction_tokens": 0,
},
},
}
def _text_completion_body(model: str) -> dict:
return {
"id": "cmpl-mock-0001",
"object": "text_completion",
"created": _now(),
"model": model,
"choices": [
{
"text": "Mock completion response.",
"index": 0,
"logprobs": None,
"finish_reason": "stop",
}
],
"usage": {"prompt_tokens": 5, "completion_tokens": 5, "total_tokens": 10},
}
def _embedding_body(model: str, input_count: int) -> dict:
return {
"object": "list",
"model": model,
"data": [
{
"object": "embedding",
"index": i,
"embedding": [0.0] * _EMBEDDING_DIM,
}
for i in range(max(input_count, 1))
],
"usage": {"prompt_tokens": input_count or 1, "total_tokens": input_count or 1},
}
def _rerank_body(query_id: str = "rerank-mock-0001") -> dict:
return {
"id": query_id,
"results": [{"index": 0, "relevance_score": 0.99}],
"meta": {
"api_version": {"version": "1"},
"billed_units": {"search_units": 1},
},
}
def _anthropic_message_body(model: str) -> dict:
return {
"id": "msg_mock_0001",
"type": "message",
"role": "assistant",
"model": model,
"content": [{"type": "text", "text": "Mock Anthropic message."}],
"stop_reason": "end_turn",
"stop_sequence": None,
"usage": {"input_tokens": 10, "output_tokens": 20},
}
def _chat_stream_chunks(model: str) -> list[dict]:
base = {
"id": "chatcmpl-mock-0001",
"object": "chat.completion.chunk",
"created": _now(),
"model": model,
}
return [
{
**base,
"choices": [
{
"index": 0,
"delta": {"role": "assistant", "content": "Hi"},
"finish_reason": None,
}
],
},
{
**base,
"choices": [
{
"index": 0,
"delta": {"content": "! This is a mock"},
"finish_reason": None,
}
],
},
{
**base,
"choices": [
{"index": 0, "delta": {"content": " response."}, "finish_reason": None}
],
},
{**base, "choices": [{"index": 0, "delta": {}, "finish_reason": "stop"}]},
]
class MockOpenAIHandler(BaseHTTPRequestHandler):
server_version = "MockOpenAI/1.0"
def log_message(
self, format: str, *args
) -> None: # noqa: A002 - signature dictated by base
logger.debug(
"%s - - [%s] %s",
self.address_string(),
self.log_date_time_string(),
format % args,
)
def _read_body(self) -> dict:
length = int(self.headers.get("Content-Length") or 0)
if length == 0:
return {}
raw = self.rfile.read(length)
try:
return json.loads(raw.decode("utf-8") or "{}")
except json.JSONDecodeError:
return {}
def _send_json(self, status: int, payload: dict) -> None:
body = json.dumps(payload).encode("utf-8")
self.send_response(status)
self.send_header("Content-Type", "application/json")
self.send_header("Content-Length", str(len(body)))
self.end_headers()
self.wfile.write(body)
def _send_sse(self, chunks: list[dict]) -> None:
self.send_response(200)
self.send_header("Content-Type", "text/event-stream")
self.send_header("Cache-Control", "no-cache")
self.send_header("Connection", "keep-alive")
self.end_headers()
for chunk in chunks:
self.wfile.write(f"data: {json.dumps(chunk)}\n\n".encode("utf-8"))
self.wfile.flush()
self.wfile.write(b"data: [DONE]\n\n")
self.wfile.flush()
def _route(self) -> str:
path = self.path.split("?", 1)[0].rstrip("/") or "/"
for prefix in ("/v1", "/v2"):
if path.startswith(prefix + "/"):
return path[len(prefix) :]
if path == prefix:
return "/"
return path
def do_GET(self) -> None:
route = self._route()
if route == "/models":
self._send_json(
200,
{
"object": "list",
"data": [
{
"id": "gpt-3.5-turbo",
"object": "model",
"created": _now(),
"owned_by": "mock",
},
{
"id": "gpt-4",
"object": "model",
"created": _now(),
"owned_by": "mock",
},
],
},
)
return
self._send_json(200, {"status": "ok", "path": self.path})
def do_POST(self) -> None:
body = self._read_body()
route = self._route()
model = body.get("model") or "gpt-3.5-turbo"
if route == "/chat/completions":
if body.get("stream"):
self._send_sse(_chat_stream_chunks(model))
else:
self._send_json(200, _chat_completion_body(model))
return
if route == "/completions":
self._send_json(200, _text_completion_body(model))
return
if route == "/embeddings":
inputs = body.get("input")
count = len(inputs) if isinstance(inputs, list) else 1
self._send_json(200, _embedding_body(model, count))
return
if route == "/rerank":
self._send_json(200, _rerank_body())
return
if route == "/messages":
self._send_json(200, _anthropic_message_body(model))
return
# Catch-all: respond OK so tests asserting on api_base reachability pass.
self._send_json(200, {"status": "ok", "path": self.path})
def serve(host: str, port: int) -> None:
server = ThreadingHTTPServer((host, port), MockOpenAIHandler)
logger.info("mock openai server listening on http://%s:%d", host, port)
try:
server.serve_forever()
except KeyboardInterrupt:
pass
finally:
server.server_close()
def main() -> None:
parser = argparse.ArgumentParser(
description="Static OpenAI-format mock server for LiteLLM CI."
)
parser.add_argument("--host", default="0.0.0.0")
parser.add_argument("--port", type=int, default=8090)
parser.add_argument("--verbose", action="store_true")
args = parser.parse_args()
logging.basicConfig(
level=logging.DEBUG if args.verbose else logging.INFO,
format="%(asctime)s %(levelname)s %(message)s",
)
serve(args.host, args.port)
if __name__ == "__main__":
main()