test: point router/completion/triton tests at the local fake OpenAI endpoint (#30900)

* test: point router/completion/triton tests at the local fake OpenAI endpoint

The shared Railway-hosted mock (exampleopenaiendpoint-production.up.railway.app)
takes down unrelated CI jobs whenever it is unreachable. #30695 moved the mounted
proxy configs onto a job-local fake server but left these in-Python api_base
literals pointing at the dead host, so litellm_router_testing, local_testing_part1,
local_testing_part2 and llm_translation_testing still fail with a 404
"Application not found" when Railway is down

Resolve the api_base from FAKE_OPENAI_API_BASE (default http://127.0.0.1:8190)
through a shared helper, auto-start the canned server from the local_testing and
llm_translation conftests when nothing is already serving, and extend the server
with a Triton embeddings route and a slow-endpoint delay so the triton and
latency-timeout tests run fully offline. The deliberately broken fallback URL is
left as-is so fallback handling still has a failing upstream

* fix: ignore non-loopback FAKE_OPENAI_API_BASE so the local mock is used in CI

* fix: drop 0.0.0.0 from loopback hosts, an unreliable client connect target

* fix(tests): keep fake OpenAI mock alive across xdist workers

ensure_fake_openai_endpoint registered atexit on the worker that spawned
the subprocess, so under -n 4 the first worker to drain its queue would
terminate the shared mock while siblings were still hitting it. Detach
the child via start_new_session and drop the per-worker teardown; reuse
on /health handles re-runs and CI containers clean up themselves
This commit is contained in:
Mateo Wang 2026-06-20 16:20:35 -07:00 • committed by GitHub
parent c7efa77de3
commit b16cfd7de9
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
13 changed files with 273 additions and 24 deletions

View file

@ -17,6 +17,7 @@ something to trip on.
from __future__ import annotations
import asyncio
import json
import time
import uuid
@ -30,6 +31,8 @@ from starlette.routing import Route
_CANNED_CONTENT: Final = "Hello! This is a mock response from the fake OpenAI endpoint."
_RATE_LIMIT_MODEL: Final = "429"
_SLOW_MODEL: Final = "slow-endpoint"
_SLOW_RESPONSE_SECONDS: Final = 3.0
_PROMPT_TOKENS: Final = 20
_COMPLETION_TOKENS: Final = 20
@ -123,6 +126,8 @@ async def chat_completions(request: Request) -> Response:
model = _requested_model(body)
if model == _RATE_LIMIT_MODEL:
return _rate_limit_response(model)
if model == _SLOW_MODEL:
await asyncio.sleep(_SLOW_RESPONSE_SECONDS)
if _wants_stream(body):
return StreamingResponse(
_chat_completion_stream(model, _wants_stream_usage(body)),
@ -175,6 +180,8 @@ async def completions(request: Request) -> Response:
model = _requested_model(body)
if model == _RATE_LIMIT_MODEL:
return _rate_limit_response(model)
if model == _SLOW_MODEL:
await asyncio.sleep(_SLOW_RESPONSE_SECONDS)
if _wants_stream(body):
return StreamingResponse(
_text_completion_stream(model, _wants_stream_usage(body)),
@ -197,6 +204,22 @@ async def embeddings(request: Request) -> Response:
)
async def triton_embeddings(_request: Request) -> Response:
return JSONResponse(
{
"model_name": "my-triton-model",
"outputs": [
{
"name": "output",
"datatype": "FP32",
"shape": [1, 2],
"data": [0.1, 0.2],
}
],
}
)
async def list_models(_request: Request) -> Response:
return JSONResponse(
{
@ -223,6 +246,7 @@ app = Starlette(
Route("/v1/completions", completions, methods=["POST"]),
Route("/embeddings", embeddings, methods=["POST"]),
Route("/v1/embeddings", embeddings, methods=["POST"]),
Route("/triton/embeddings", triton_embeddings, methods=["POST"]),
Route("/models", list_models, methods=["GET"]),
Route("/v1/models", list_models, methods=["GET"]),
]

View file

@ -0,0 +1,95 @@
"""Shared access to the canned OpenAI mock used across the test suite.
Tests that need a fake OpenAI-shaped endpoint point ``api_base`` at
``FAKE_OPENAI_API_BASE`` instead of a hosted URL, so the suite never depends on
an external service staying up. ``ensure_fake_openai_endpoint`` returns a base
URL that is actually serving: it reuses a server already listening on that base
(it answers ``/health``) and otherwise spawns ``_fake_openai_endpoint_server.py``
detached from the spawning process so it survives until the host goes away.
We deliberately do not register an interpreter-exit teardown. Under
``pytest-xdist`` the spawn happens inside whichever worker wins the race, and
workers exit independently as their queues drain; a per-worker ``atexit`` would
terminate the shared mock mid-session for the workers still running. In CI the
container is ephemeral, and locally the next run reuses the still-healthy server
via ``/health`` (or a developer can free the port manually).
The resolved base honors a ``FAKE_OPENAI_API_BASE`` env var only when it points
at a loopback host; a remote value (CI sets it to the old hosted mock) is ignored
in favor of the local default. Otherwise this helper would try to bind a local
server to a remote host and time out, which is the exact external dependency the
local mock exists to remove.
"""
from __future__ import annotations
import os
import subprocess
import sys
import time
from functools import lru_cache
from pathlib import Path
from typing import Final
from urllib.error import URLError
from urllib.parse import urlsplit
from urllib.request import urlopen
_LOCAL_DEFAULT: Final = "http://127.0.0.1:8190"
_LOOPBACK_HOSTS: Final = frozenset({"127.0.0.1", "localhost", "::1"})
def _resolve_base() -> str:
configured = os.environ.get("FAKE_OPENAI_API_BASE")
if configured and urlsplit(configured).hostname in _LOOPBACK_HOSTS:
return configured
return _LOCAL_DEFAULT
FAKE_OPENAI_API_BASE: Final = _resolve_base()
_SERVER_SCRIPT: Final = (
Path(__file__).resolve().parent / "_fake_openai_endpoint_server.py"
)
_HEALTH_TIMEOUT_SECONDS: Final = 30.0
def _is_healthy(base: str) -> bool:
try:
with urlopen(f"{base.rstrip('/')}/health", timeout=1) as response:
return response.status == 200
except (URLError, OSError):
return False
def _wait_until_healthy(base: str) -> None:
deadline = time.monotonic() + _HEALTH_TIMEOUT_SECONDS
while time.monotonic() < deadline:
if _is_healthy(base):
return
time.sleep(0.5)
raise RuntimeError(
f"fake OpenAI endpoint at {base} did not become healthy within {_HEALTH_TIMEOUT_SECONDS}s"
)
@lru_cache(maxsize=1)
def ensure_fake_openai_endpoint() -> str:
base = FAKE_OPENAI_API_BASE
if _is_healthy(base):
return base
parts = urlsplit(base)
subprocess.Popen(
[
sys.executable,
str(_SERVER_SCRIPT),
"--host",
parts.hostname or "127.0.0.1",
"--port",
str(parts.port or 8190),
],
stdout=subprocess.DEVNULL,
stderr=subprocess.DEVNULL,
start_new_session=True,
)
_wait_until_healthy(base)
return base

View file

@ -31,6 +31,14 @@ from tests._vcr_conftest_common import ( # noqa: E402,F401
reset_vcr_diag_dir,
vcr_config_dict,
)
from tests.fake_openai_endpoint import ensure_fake_openai_endpoint # noqa: E402
@pytest.fixture(scope="session", autouse=True)
def fake_openai_endpoint():
ensure_fake_openai_endpoint()
yield
# Per-item respx detection (``apply_vcr_auto_marker_to_items``) handles
# the vast majority of respx-vs-vcrpy conflicts automatically. The only

View file

@ -19,6 +19,8 @@ import pytest
from litellm.llms.triton.embedding.transformation import TritonEmbeddingConfig
import litellm
from tests.fake_openai_endpoint import FAKE_OPENAI_API_BASE
def test_split_embedding_by_shape_passes():
try:
@ -360,7 +362,7 @@ async def test_triton_embeddings():
litellm.set_verbose = True
response = await litellm.aembedding(
model="triton/my-triton-model",
api_base="https://exampleopenaiendpoint-production.up.railway.app/triton/embeddings",
api_base=f"{FAKE_OPENAI_API_BASE}/triton/embeddings",
input=["good morning from litellm"],
)
print(f"response: {response}")

View file

@ -47,6 +47,14 @@ from tests._vcr_conftest_common import ( # noqa: E402,F401
reset_vcr_diag_dir,
vcr_config_dict,
)
from tests.fake_openai_endpoint import ensure_fake_openai_endpoint # noqa: E402
@pytest.fixture(scope="session", autouse=True)
def fake_openai_endpoint():
ensure_fake_openai_endpoint()
yield
# Per-item respx detection (``apply_vcr_auto_marker_to_items``) auto-skips
# tests whose ``@pytest.mark.respx`` marker or ``respx_mock`` fixture
@ -62,6 +70,8 @@ from tests._vcr_conftest_common import ( # noqa: E402,F401
_VCR_INCOMPATIBLE_FILES = frozenset(
{
"test_router_caching.py",
# Hits the local fake OpenAI endpoint on 127.0.0.1; nothing to record.
"test_fake_openai_endpoint.py",
}
)

View file

@ -24,6 +24,8 @@ from litellm import RateLimitError, Timeout, completion, completion_cost, embedd
from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler, HTTPHandler
from litellm.litellm_core_utils.prompt_templates.factory import anthropic_messages_pt
from tests.fake_openai_endpoint import FAKE_OPENAI_API_BASE
# litellm.num_retries=3
litellm.cache = None
@ -1343,7 +1345,7 @@ def test_lm_studio_completion(monkeypatch):
messages=[
{"role": "user", "content": "What's the weather like in San Francisco?"}
],
api_base="https://exampleopenaiendpoint-production.up.railway.app/",
api_base=FAKE_OPENAI_API_BASE,
)
except litellm.AuthenticationError as e:
pytest.fail(f"Error occurred: {e}")

View file

@ -0,0 +1,102 @@
"""Regression tests for the canned OpenAI mock that decouples the suite from a
hosted endpoint (see ``tests/fake_openai_endpoint.py`` and
``tests/_fake_openai_endpoint_server.py``).
Before this, router/completion/triton tests hardcoded a shared Railway mock as
their ``api_base``; when that single deployment went down, unrelated CI jobs
failed with ``404 Application not found``. These tests pin the two behaviors
those tests now rely on from the local stand-in (the Triton embeddings shape and
the ``slow-endpoint`` delay) and guard against the dead host creeping back in.
"""
from __future__ import annotations
import re
from pathlib import Path
import httpx
import pytest
from tests.fake_openai_endpoint import (
_LOCAL_DEFAULT,
_resolve_base,
ensure_fake_openai_endpoint,
)
_REPO_ROOT = Path(__file__).resolve().parents[2]
_MIGRATED_FILES = (
"tests/llm_translation/test_triton.py",
"tests/local_testing/test_router.py",
"tests/local_testing/test_router_custom_routing.py",
"tests/local_testing/test_router_fallback_handlers.py",
"tests/local_testing/test_router_fallbacks.py",
"tests/local_testing/test_secret_detect_hook.py",
"tests/local_testing/test_lowest_latency_routing.py",
"tests/local_testing/test_completion.py",
)
# A live ``railway.app/`` or ``railway.app"`` api_base reappearing in a migrated
# file would re-couple CI to an external service's uptime. The deliberately
# broken ``...railway.appzzzzz`` fallback URL is not a live host, so it does not
# match.
_LIVE_HOSTED_MOCK = re.compile(r"railway\.app(?:/|\")")
def test_chat_completion_shape():
base = ensure_fake_openai_endpoint()
response = httpx.post(
f"{base}/v1/chat/completions",
json={"model": "gpt-4", "messages": [{"role": "user", "content": "hi"}]},
timeout=10,
)
assert response.status_code == 200
body = response.json()
assert body["choices"][0]["message"]["content"]
assert body["usage"]["total_tokens"] == 40
def test_triton_embeddings_route():
base = ensure_fake_openai_endpoint()
response = httpx.post(f"{base}/triton/embeddings", json={"inputs": []}, timeout=10)
assert response.status_code == 200
output = response.json()["outputs"][0]
assert output["shape"] == [1, 2]
assert output["data"] == [0.1, 0.2]
def test_slow_model_blocks_past_client_timeout():
base = ensure_fake_openai_endpoint()
with pytest.raises(httpx.TimeoutException):
httpx.post(
f"{base}/v1/chat/completions",
json={
"model": "slow-endpoint",
"messages": [{"role": "user", "content": "hi"}],
},
timeout=0.5,
)
def test_remote_env_base_resolves_to_local(monkeypatch):
monkeypatch.setenv(
"FAKE_OPENAI_API_BASE",
"https://exampleopenaiendpoint-production.up.railway.app",
)
assert _resolve_base() == _LOCAL_DEFAULT
@pytest.mark.parametrize("host", ["127.0.0.1", "localhost", "[::1]"])
def test_loopback_env_base_is_honored(monkeypatch, host):
base = f"http://{host}:9191"
monkeypatch.setenv("FAKE_OPENAI_API_BASE", base)
assert _resolve_base() == base
@pytest.mark.parametrize("relative_path", _MIGRATED_FILES)
def test_migrated_files_have_no_live_hosted_mock(relative_path):
source = (_REPO_ROOT / relative_path).read_text()
assert not _LIVE_HOSTED_MOCK.search(source), (
f"{relative_path} references the hosted Railway mock; point api_base at "
"FAKE_OPENAI_API_BASE so CI does not depend on an external service"
)

View file

@ -25,6 +25,8 @@ from litellm import Router
from litellm.caching.caching import DualCache
from litellm.router_strategy.lowest_latency import LowestLatencyLoggingHandler
from tests.fake_openai_endpoint import FAKE_OPENAI_API_BASE
### UNIT TESTS FOR LATENCY ROUTING ###
@ -591,7 +593,7 @@ async def test_lowest_latency_routing_with_timeouts():
"model_name": "azure-model",
"litellm_params": {
"model": "openai/slow-endpoint",
"api_base": "https://exampleopenaiendpoint-production-c715.up.railway.app/", # If you are Krrish, this is OpenAI Endpoint3 on our Railway endpoint :)
"api_base": FAKE_OPENAI_API_BASE,
"api_key": "fake-key",
},
"model_info": {"id": "slow-endpoint"},
@ -600,7 +602,7 @@ async def test_lowest_latency_routing_with_timeouts():
"model_name": "azure-model",
"litellm_params": {
"model": "openai/fast-endpoint",
"api_base": "https://exampleopenaiendpoint-production.up.railway.app/",
"api_base": FAKE_OPENAI_API_BASE,
"api_key": "fake-key",
},
"model_info": {"id": "fast-endpoint"},
@ -666,7 +668,7 @@ async def test_lowest_latency_routing_first_pick():
"model_name": "azure-model",
"litellm_params": {
"model": "openai/fast-endpoint",
"api_base": "https://exampleopenaiendpoint-production.up.railway.app/",
"api_base": FAKE_OPENAI_API_BASE,
"api_key": "fake-key",
},
"model_info": {"id": "fast-endpoint"},
@ -675,7 +677,7 @@ async def test_lowest_latency_routing_first_pick():
"model_name": "azure-model",
"litellm_params": {
"model": "openai/fast-endpoint-2",
"api_base": "https://exampleopenaiendpoint-production.up.railway.app/",
"api_base": FAKE_OPENAI_API_BASE,
"api_key": "fake-key",
},
"model_info": {"id": "fast-endpoint-2"},
@ -684,7 +686,7 @@ async def test_lowest_latency_routing_first_pick():
"model_name": "azure-model",
"litellm_params": {
"model": "openai/fast-endpoint-2",
"api_base": "https://exampleopenaiendpoint-production.up.railway.app/",
"api_base": FAKE_OPENAI_API_BASE,
"api_key": "fake-key",
},
"model_info": {"id": "fast-endpoint-3"},
@ -693,7 +695,7 @@ async def test_lowest_latency_routing_first_pick():
"model_name": "azure-model",
"litellm_params": {
"model": "openai/fast-endpoint-2",
"api_base": "https://exampleopenaiendpoint-production.up.railway.app/",
"api_base": FAKE_OPENAI_API_BASE,
"api_key": "fake-key",
},
"model_info": {"id": "fast-endpoint-4"},

View file

@ -34,6 +34,8 @@ from litellm.router_utils.cooldown_handlers import (
)
from litellm.types.router import DeploymentTypedDict
from tests.fake_openai_endpoint import FAKE_OPENAI_API_BASE
load_dotenv()
@ -144,14 +146,14 @@ async def test_router_provider_wildcard_routing_regex():
"model_name": "openai/fo::*:static::*",
"litellm_params": {
"model": "openai/fo::*:static::*",
"api_base": "https://exampleopenaiendpoint-production.up.railway.app/",
"api_base": FAKE_OPENAI_API_BASE,
},
},
{
"model_name": "openai/foo3::hello::*",
"litellm_params": {
"model": "openai/foo3::hello::*",
"api_base": "https://exampleopenaiendpoint-production.up.railway.app/",
"api_base": FAKE_OPENAI_API_BASE,
},
},
]
@ -1639,10 +1641,7 @@ async def test_router_text_completion_client():
"litellm_params": {
"model": "text-completion-openai/gpt-3.5-turbo-instruct",
"api_key": os.getenv("OPENAI_API_KEY", None),
"api_base": os.getenv(
"FAKE_OPENAI_API_BASE",
"https://exampleopenaiendpoint-production.up.railway.app/",
),
"api_base": FAKE_OPENAI_API_BASE,
},
}
]

View file

@ -18,6 +18,8 @@ import litellm
from litellm import Router
from litellm.router import CustomRoutingStrategyBase
from tests.fake_openai_endpoint import FAKE_OPENAI_API_BASE
def _create_router():
return Router(
@ -26,7 +28,7 @@ def _create_router():
"model_name": "azure-model",
"litellm_params": {
"model": "openai/very-special-endpoint",
"api_base": "https://exampleopenaiendpoint-production.up.railway.app/",
"api_base": FAKE_OPENAI_API_BASE,
"api_key": "fake-key",
},
"model_info": {"id": "very-special-endpoint"},
@ -35,7 +37,7 @@ def _create_router():
"model_name": "azure-model",
"litellm_params": {
"model": "openai/fast-endpoint",
"api_base": "https://exampleopenaiendpoint-production.up.railway.app/",
"api_base": FAKE_OPENAI_API_BASE,
"api_key": "fake-key",
},
"model_info": {"id": "fast-endpoint"},

View file

@ -22,6 +22,8 @@ from litellm.router_utils.fallback_event_handlers import (
log_failure_fallback_event,
)
from tests.fake_openai_endpoint import FAKE_OPENAI_API_BASE
# Helper function to create a Router instance
def create_test_router():
@ -68,7 +70,7 @@ def create_test_router_2():
"litellm_params": {
"model": "openai/fake-openai-endpoint-2",
"api_key": "working-key-since-this-is-fake-endpoint",
"api_base": "https://exampleopenaiendpoint-production.up.railway.app/",
"api_base": FAKE_OPENAI_API_BASE,
},
},
],
@ -308,7 +310,4 @@ async def test_multiple_fallbacks(function_name):
print(result._hidden_params)
assert (
result._hidden_params["api_base"]
== "https://exampleopenaiendpoint-production.up.railway.app/"
)
assert result._hidden_params["api_base"] == FAKE_OPENAI_API_BASE

View file

@ -18,6 +18,8 @@ import litellm
from litellm import Router
from litellm.integrations.custom_logger import CustomLogger
from tests.fake_openai_endpoint import FAKE_OPENAI_API_BASE
class MyCustomHandler(CustomLogger):
success: bool = False
@ -1345,7 +1347,7 @@ def test_router_fallbacks_with_custom_model_costs():
"model": "openai/claude-sonnet-4-5-20250929",
"input_cost_per_token": 0.000003, # 3$/M
"output_cost_per_token": 0.000015, # 15$/M
"api_base": "https://exampleopenaiendpoint-production.up.railway.app",
"api_base": FAKE_OPENAI_API_BASE,
"api_key": "my-fake-key",
"mock_response": "Hello! How can I help you today?",
},
@ -1594,7 +1596,7 @@ async def test_router_attempted_fallbacks_in_response(expected_attempted_fallbac
"litellm_params": {
"model": "openai/working-fake-endpoint",
"api_key": "my-fake-key",
"api_base": "https://exampleopenaiendpoint-production.up.railway.app",
"api_base": FAKE_OPENAI_API_BASE,
},
},
{

View file

@ -36,6 +36,8 @@ from litellm.proxy.proxy_server import chat_completion
from litellm.proxy.utils import ProxyLogging, hash_token
from litellm.router import Router
from tests.fake_openai_endpoint import FAKE_OPENAI_API_BASE
### UNIT TESTS FOR OpenAI Moderation ###
@ -246,7 +248,7 @@ router = Router(
"model_name": "fake-model",
"litellm_params": {
"model": "openai/fake",
"api_base": "https://exampleopenaiendpoint-production.up.railway.app/",
"api_base": FAKE_OPENAI_API_BASE,
"api_key": "sk-12345",
},
}