From 84d3a807b915d3df7e6c11eb4bd1d205d3a10a1a Mon Sep 17 00:00:00 2001 From: Yuneng Jiang Date: Tue, 19 May 2026 18:24:41 -0700 Subject: [PATCH] ci(test): replace Railway-hosted mock endpoint with in-repo stdlib server MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The exampleopenaiendpoint Railway deployment is a single point of failure for ~30 tests across 8 CI jobs. Today's outage took the proxy/router/utils suites offline despite the tests having nothing to do with that endpoint itself. Add tests/mock_endpoints/openai_mock_server.py — a stdlib-only HTTP server that returns static OpenAI-format responses (chat completions incl. SSE, text completions, embeddings, rerank, Anthropic messages). No new dependencies; starts in <1s and handles ≥300 concurrent requests in the self-test. Wire it into the 8 affected CI jobs via a new start_mock_openai_server CircleCI command (binds 0.0.0.0:8090). The proxy containers reach it via host.docker.internal:8090 through the existing --add-host flag; the direct-pytest jobs reach it on 127.0.0.1:8090. Update affected YAML configs and test files to point at the new URLs. The single intentionally-bad URL ("…appxxxx/") becomes http://bad.invalid/ so DNS failure still tests the fallback path predictably. --- .circleci/config.yml | 24 ++ docker/build_from_pip/litellm_config.yaml | 2 +- .../disable_schema_update.yaml | 4 +- .../enterprise_config.yaml | 2 +- .../multi_instance_simple_config.yaml | 2 +- .../example_config_yaml/otel_test_config.yaml | 10 +- .../spend_tracking_config.yaml | 2 +- proxy_server_config.yaml | 18 +- .../litellm_utils_tests/test_health_check.py | 2 +- .../test_litellm_overhead.py | 2 +- tests/local_testing/test_router.py | 6 +- .../test_router_custom_routing.py | 4 +- .../test_router_fallback_handlers.py | 4 +- tests/local_testing/test_router_fallbacks.py | 6 +- tests/mock_endpoints/__init__.py | 0 tests/mock_endpoints/openai_mock_server.py | 298 ++++++++++++++++++ 16 files changed, 354 insertions(+), 32 deletions(-) create mode 100644 tests/mock_endpoints/__init__.py create mode 100644 tests/mock_endpoints/openai_mock_server.py diff --git a/.circleci/config.yml b/.circleci/config.yml index 32983c92e27..f957be50b0c 100644 --- a/.circleci/config.yml +++ b/.circleci/config.yml @@ -111,6 +111,22 @@ commands: - wait_for_service: url: tcp://localhost:6379 timeout: "60" + start_mock_openai_server: + description: "Start the in-repo mock OpenAI server so tests and proxy configs don't depend on an external endpoint. Binds 0.0.0.0:8090 by default — reachable as 127.0.0.1:8090 from the same host and as host.docker.internal:8090 from sibling docker containers." + parameters: + port: + type: string + default: "8090" + steps: + - run: + name: Start mock OpenAI server + background: true + command: | + python3 -m tests.mock_endpoints.openai_mock_server \ + --host 0.0.0.0 --port << parameters.port >> + - wait_for_service: + url: http://127.0.0.1:<< parameters.port >>/ + timeout: "30" setup_litellm_enterprise_pip: steps: - run: @@ -452,6 +468,7 @@ jobs: key: v1-uv-cache-{{ checksum "uv.lock" }} # Run pytest and generate JUnit XML report - setup_litellm_enterprise_pip + - start_mock_openai_server - run: name: Run tests command: | @@ -984,6 +1001,7 @@ jobs: name: Install Dependencies command: | uv sync --frozen --all-groups --all-extras --python 3.12 + - start_mock_openai_server # Run pytest and generate JUnit XML report - run: name: Run tests @@ -1387,6 +1405,7 @@ jobs: uv sync --frozen --all-groups --all-extras --python 3.12 - start_postgres: db_name: litellm_test + - start_mock_openai_server - attach_workspace: at: ~/project - run: @@ -1478,6 +1497,7 @@ jobs: command: | uv sync --frozen --all-groups --all-extras --python 3.12 - start_postgres + - start_mock_openai_server - run: name: Load Docker Database Image command: | @@ -1643,6 +1663,7 @@ jobs: command: | uv sync --frozen --all-groups --all-extras --python 3.12 - start_postgres + - start_mock_openai_server - attach_workspace: at: ~/project - run: @@ -1770,6 +1791,7 @@ jobs: uv sync --frozen --all-groups --all-extras --python 3.12 - start_postgres - start_redis + - start_mock_openai_server - attach_workspace: at: ~/project - run: @@ -1852,6 +1874,7 @@ jobs: command: | uv sync --frozen --all-groups --all-extras --python 3.12 - start_postgres + - start_mock_openai_server - attach_workspace: at: ~/project - run: @@ -2019,6 +2042,7 @@ jobs: command: | docker build -t my-app:latest -f docker/build_from_pip/Dockerfile.build_from_pip . - start_postgres + - start_mock_openai_server - run: name: Run Docker container # intentionally give bad redis credentials here diff --git a/docker/build_from_pip/litellm_config.yaml b/docker/build_from_pip/litellm_config.yaml index 51223026170..467028f06d0 100644 --- a/docker/build_from_pip/litellm_config.yaml +++ b/docker/build_from_pip/litellm_config.yaml @@ -3,7 +3,7 @@ model_list: litellm_params: model: openai/fake api_key: fake-key - api_base: https://exampleopenaiendpoint-production.up.railway.app/ + api_base: http://host.docker.internal:8090/ general_settings: alerting: ["slack"] \ No newline at end of file diff --git a/litellm/proxy/example_config_yaml/disable_schema_update.yaml b/litellm/proxy/example_config_yaml/disable_schema_update.yaml index 5dcbd0dbd57..bde9ad5dc0b 100644 --- a/litellm/proxy/example_config_yaml/disable_schema_update.yaml +++ b/litellm/proxy/example_config_yaml/disable_schema_update.yaml @@ -3,12 +3,12 @@ model_list: litellm_params: model: openai/fake api_key: fake-key - api_base: https://exampleopenaiendpoint-production.up.railway.app/ + api_base: http://host.docker.internal:8090/ - model_name: gpt-4 litellm_params: model: openai/gpt-4 api_key: fake-key - api_base: https://exampleopenaiendpoint-production.up.railway.app/ + api_base: http://host.docker.internal:8090/ litellm_settings: callbacks: ["gcs_bucket"] diff --git a/litellm/proxy/example_config_yaml/enterprise_config.yaml b/litellm/proxy/example_config_yaml/enterprise_config.yaml index 337e85177e5..0011c89bc73 100644 --- a/litellm/proxy/example_config_yaml/enterprise_config.yaml +++ b/litellm/proxy/example_config_yaml/enterprise_config.yaml @@ -3,7 +3,7 @@ model_list: litellm_params: model: openai/fake api_key: fake-key - api_base: https://exampleopenaiendpoint-production.up.railway.app/ + api_base: http://host.docker.internal:8090/ tags: ["teamA"] model_info: id: "team-a-model" diff --git a/litellm/proxy/example_config_yaml/multi_instance_simple_config.yaml b/litellm/proxy/example_config_yaml/multi_instance_simple_config.yaml index f83160a7a22..9add5e21f74 100644 --- a/litellm/proxy/example_config_yaml/multi_instance_simple_config.yaml +++ b/litellm/proxy/example_config_yaml/multi_instance_simple_config.yaml @@ -3,7 +3,7 @@ model_list: litellm_params: model: openai/my-fake-model api_key: my-fake-key - api_base: https://exampleopenaiendpoint-production.up.railway.app/ + api_base: http://host.docker.internal:8090/ litellm_settings: cache: True diff --git a/litellm/proxy/example_config_yaml/otel_test_config.yaml b/litellm/proxy/example_config_yaml/otel_test_config.yaml index c05e2b1b5df..2b9009c1ed4 100644 --- a/litellm/proxy/example_config_yaml/otel_test_config.yaml +++ b/litellm/proxy/example_config_yaml/otel_test_config.yaml @@ -3,7 +3,7 @@ model_list: litellm_params: model: openai/gpt-5-mini api_key: fake-key - api_base: https://exampleopenaiendpoint-production.up.railway.app/ + api_base: http://host.docker.internal:8090/ tags: ["teamA"] model_info: id: "team-a-model" @@ -11,7 +11,7 @@ model_list: litellm_params: model: openai/gpt-5-mini api_key: fake-key - api_base: https://exampleopenaiendpoint-production.up.railway.app/ + api_base: http://host.docker.internal:8090/ tags: ["teamB"] model_info: id: "team-b-model" @@ -23,7 +23,7 @@ model_list: litellm_params: model: openai/429 api_key: fake-key - api_base: https://exampleopenaiendpoint-production.up.railway.app + api_base: http://host.docker.internal:8090 - model_name: llava-hf litellm_params: model: openai/llava-hf/llava-v1.6-vicuna-7b-hf @@ -34,12 +34,12 @@ model_list: - model_name: bedrock/* litellm_params: model: bedrock/* - api_base: https://exampleopenaiendpoint-production.up.railway.app/ + api_base: http://host.docker.internal:8090/ - model_name: openai/* litellm_params: model: openai/* api_key: os.environ/OPENAI_API_KEY - api_base: https://exampleopenaiendpoint-production.up.railway.app/ + api_base: http://host.docker.internal:8090/ litellm_settings: diff --git a/litellm/proxy/example_config_yaml/spend_tracking_config.yaml b/litellm/proxy/example_config_yaml/spend_tracking_config.yaml index dfed2194b58..aaeaa95464d 100644 --- a/litellm/proxy/example_config_yaml/spend_tracking_config.yaml +++ b/litellm/proxy/example_config_yaml/spend_tracking_config.yaml @@ -3,7 +3,7 @@ model_list: litellm_params: model: openai/gpt-5-mini api_key: fake-key - api_base: https://exampleopenaiendpoint-production.up.railway.app/ + api_base: http://host.docker.internal:8090/ general_settings: use_redis_transaction_buffer: true diff --git a/proxy_server_config.yaml b/proxy_server_config.yaml index da37eb34289..a4f45c8a262 100644 --- a/proxy_server_config.yaml +++ b/proxy_server_config.yaml @@ -48,39 +48,39 @@ model_list: litellm_params: model: openai/gpt-5-mini api_key: fake-key - api_base: https://exampleopenaiendpoint-production.up.railway.app/ + api_base: http://host.docker.internal:8090/ - model_name: fake-openai-endpoint-2 litellm_params: model: openai/my-fake-model api_key: my-fake-key - api_base: https://exampleopenaiendpoint-production.up.railway.app/ + api_base: http://host.docker.internal:8090/ stream_timeout: 0.001 rpm: 1 - model_name: fake-openai-endpoint-3 litellm_params: model: openai/my-fake-model api_key: my-fake-key - api_base: https://exampleopenaiendpoint-production.up.railway.app/ + api_base: http://host.docker.internal:8090/ stream_timeout: 0.001 rpm: 1000 - model_name: fake-openai-endpoint-4 litellm_params: model: openai/my-fake-model api_key: my-fake-key - api_base: https://exampleopenaiendpoint-production.up.railway.app/ + api_base: http://host.docker.internal:8090/ num_retries: 50 - model_name: fake-openai-endpoint-3 litellm_params: model: openai/my-fake-model-2 api_key: my-fake-key - api_base: https://exampleopenaiendpoint-production.up.railway.app/ + api_base: http://host.docker.internal:8090/ stream_timeout: 0.001 rpm: 1000 - model_name: bad-model litellm_params: model: openai/bad-model api_key: os.environ/OPENAI_API_KEY - api_base: https://exampleopenaiendpoint-production.up.railway.app/ + api_base: http://host.docker.internal:8090/ mock_timeout: True timeout: 60 rpm: 1000 @@ -90,7 +90,7 @@ model_list: litellm_params: model: openai/bad-model api_key: os.environ/OPENAI_API_KEY - api_base: https://exampleopenaiendpoint-production.up.railway.app/ + api_base: http://host.docker.internal:8090/ rpm: 1000 model_info: health_check_timeout: 1 @@ -132,13 +132,13 @@ model_list: litellm_params: model: openai/my-fake-model api_key: my-fake-key - api_base: https://exampleopenaiendpoint-production.up.railway.app/ + api_base: http://host.docker.internal:8090/ timeout: 1 - model_name: badly-configured-openai-endpoint litellm_params: model: openai/my-fake-model api_key: my-fake-key - api_base: https://exampleopenaiendpoint-production.up.railway.appxxxx/ + api_base: http://bad.invalid/ - model_name: gemini-2.5-flash litellm_params: model: gemini/gemini-2.5-flash diff --git a/tests/litellm_utils_tests/test_health_check.py b/tests/litellm_utils_tests/test_health_check.py index de6f7c38fed..71c62505c3f 100644 --- a/tests/litellm_utils_tests/test_health_check.py +++ b/tests/litellm_utils_tests/test_health_check.py @@ -586,7 +586,7 @@ async def test_health_check_bad_model(): "model_name": "openai-gpt-4o", "litellm_params": { "api_key": "sk-1234", - "api_base": "https://exampleopenaiendpoint-production.up.railway.app", + "api_base": "http://127.0.0.1:8090", "model": "openai/my-fake-openai-endpoint", "mock_timeout": True, "timeout": 60, diff --git a/tests/litellm_utils_tests/test_litellm_overhead.py b/tests/litellm_utils_tests/test_litellm_overhead.py index 3a428e9d588..042c2fa1af6 100644 --- a/tests/litellm_utils_tests/test_litellm_overhead.py +++ b/tests/litellm_utils_tests/test_litellm_overhead.py @@ -164,7 +164,7 @@ async def test_litellm_overhead_stream(model): # Specific cases for models ######################################################### if model == "openai/self_hosted": - kwargs["api_base"] = "https://exampleopenaiendpoint-production.up.railway.app/" + kwargs["api_base"] = "http://127.0.0.1:8090/" # warmup call for auth validation on vertex_ai models await litellm.acompletion(**kwargs) diff --git a/tests/local_testing/test_router.py b/tests/local_testing/test_router.py index 6d04e6ecaa5..20a18225ca6 100644 --- a/tests/local_testing/test_router.py +++ b/tests/local_testing/test_router.py @@ -144,14 +144,14 @@ async def test_router_provider_wildcard_routing_regex(): "model_name": "openai/fo::*:static::*", "litellm_params": { "model": "openai/fo::*:static::*", - "api_base": "https://exampleopenaiendpoint-production.up.railway.app/", + "api_base": "http://127.0.0.1:8090/", }, }, { "model_name": "openai/foo3::hello::*", "litellm_params": { "model": "openai/foo3::hello::*", - "api_base": "https://exampleopenaiendpoint-production.up.railway.app/", + "api_base": "http://127.0.0.1:8090/", }, }, ] @@ -1639,7 +1639,7 @@ async def test_router_text_completion_client(): "litellm_params": { "model": "text-completion-openai/gpt-3.5-turbo-instruct", "api_key": os.getenv("OPENAI_API_KEY", None), - "api_base": "https://exampleopenaiendpoint-production.up.railway.app/", + "api_base": "http://127.0.0.1:8090/", }, } ] diff --git a/tests/local_testing/test_router_custom_routing.py b/tests/local_testing/test_router_custom_routing.py index 3f829a13c02..b69f1a2b48f 100644 --- a/tests/local_testing/test_router_custom_routing.py +++ b/tests/local_testing/test_router_custom_routing.py @@ -26,7 +26,7 @@ def _create_router(): "model_name": "azure-model", "litellm_params": { "model": "openai/very-special-endpoint", - "api_base": "https://exampleopenaiendpoint-production.up.railway.app/", + "api_base": "http://127.0.0.1:8090/", "api_key": "fake-key", }, "model_info": {"id": "very-special-endpoint"}, @@ -35,7 +35,7 @@ def _create_router(): "model_name": "azure-model", "litellm_params": { "model": "openai/fast-endpoint", - "api_base": "https://exampleopenaiendpoint-production.up.railway.app/", + "api_base": "http://127.0.0.1:8090/", "api_key": "fake-key", }, "model_info": {"id": "fast-endpoint"}, diff --git a/tests/local_testing/test_router_fallback_handlers.py b/tests/local_testing/test_router_fallback_handlers.py index bc9b42f5a05..ac08a6dc08c 100644 --- a/tests/local_testing/test_router_fallback_handlers.py +++ b/tests/local_testing/test_router_fallback_handlers.py @@ -68,7 +68,7 @@ def create_test_router_2(): "litellm_params": { "model": "openai/fake-openai-endpoint-2", "api_key": "working-key-since-this-is-fake-endpoint", - "api_base": "https://exampleopenaiendpoint-production.up.railway.app/", + "api_base": "http://127.0.0.1:8090/", }, }, ], @@ -310,5 +310,5 @@ async def test_multiple_fallbacks(function_name): assert ( result._hidden_params["api_base"] - == "https://exampleopenaiendpoint-production.up.railway.app/" + == "http://127.0.0.1:8090/" ) diff --git a/tests/local_testing/test_router_fallbacks.py b/tests/local_testing/test_router_fallbacks.py index a14e53adbc4..16ade420b01 100644 --- a/tests/local_testing/test_router_fallbacks.py +++ b/tests/local_testing/test_router_fallbacks.py @@ -1345,7 +1345,7 @@ def test_router_fallbacks_with_custom_model_costs(): "model": "openai/claude-sonnet-4-5-20250929", "input_cost_per_token": 0.000003, # 3$/M "output_cost_per_token": 0.000015, # 15$/M - "api_base": "https://exampleopenaiendpoint-production.up.railway.app", + "api_base": "http://127.0.0.1:8090", "api_key": "my-fake-key", "mock_response": "Hello! How can I help you today?", }, @@ -1594,14 +1594,14 @@ async def test_router_attempted_fallbacks_in_response(expected_attempted_fallbac "litellm_params": { "model": "openai/working-fake-endpoint", "api_key": "my-fake-key", - "api_base": "https://exampleopenaiendpoint-production.up.railway.app", + "api_base": "http://127.0.0.1:8090", }, }, { "model_name": "badly-configured-openai-endpoint", "litellm_params": { "model": "openai/my-fake-model", - "api_base": "https://exampleopenaiendpoint-production.up.railway.appzzzzz", + "api_base": "http://bad.invalid/", }, }, ], diff --git a/tests/mock_endpoints/__init__.py b/tests/mock_endpoints/__init__.py new file mode 100644 index 00000000000..e69de29bb2d diff --git a/tests/mock_endpoints/openai_mock_server.py b/tests/mock_endpoints/openai_mock_server.py new file mode 100644 index 00000000000..ef7f4ad8b2c --- /dev/null +++ b/tests/mock_endpoints/openai_mock_server.py @@ -0,0 +1,298 @@ +""" +In-repo replacement for the Railway-hosted exampleopenaiendpoint that LiteLLM's +CI used to depend on. Returns static OpenAI-format responses so tests can run +without an external dependency. + +Run standalone: + python -m tests.mock_endpoints.openai_mock_server --host 0.0.0.0 --port 8090 + +The server is intentionally implemented with the Python standard library only +(no FastAPI / httpx / pydantic) so CI jobs can start it before installing test +dependencies. +""" + +from __future__ import annotations + +import argparse +import json +import logging +import time +from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer + +logger = logging.getLogger("openai_mock_server") + +_EMBEDDING_DIM = 1536 + + +def _now() -> int: + return int(time.time()) + + +def _chat_completion_body( + model: str, content: str = "Hi! This is a mock response." +) -> dict: + return { + "id": "chatcmpl-mock-0001", + "object": "chat.completion", + "created": _now(), + "model": model, + "system_fingerprint": "fp_mock", + "choices": [ + { + "index": 0, + "message": {"role": "assistant", "content": content}, + "logprobs": None, + "finish_reason": "stop", + } + ], + "usage": { + "prompt_tokens": 10, + "completion_tokens": 20, + "total_tokens": 30, + "prompt_tokens_details": {"cached_tokens": 0, "audio_tokens": 0}, + "completion_tokens_details": { + "reasoning_tokens": 0, + "audio_tokens": 0, + "accepted_prediction_tokens": 0, + "rejected_prediction_tokens": 0, + }, + }, + } + + +def _text_completion_body(model: str) -> dict: + return { + "id": "cmpl-mock-0001", + "object": "text_completion", + "created": _now(), + "model": model, + "choices": [ + { + "text": "Mock completion response.", + "index": 0, + "logprobs": None, + "finish_reason": "stop", + } + ], + "usage": {"prompt_tokens": 5, "completion_tokens": 5, "total_tokens": 10}, + } + + +def _embedding_body(model: str, input_count: int) -> dict: + return { + "object": "list", + "model": model, + "data": [ + { + "object": "embedding", + "index": i, + "embedding": [0.0] * _EMBEDDING_DIM, + } + for i in range(max(input_count, 1)) + ], + "usage": {"prompt_tokens": input_count or 1, "total_tokens": input_count or 1}, + } + + +def _rerank_body(query_id: str = "rerank-mock-0001") -> dict: + return { + "id": query_id, + "results": [{"index": 0, "relevance_score": 0.99}], + "meta": { + "api_version": {"version": "1"}, + "billed_units": {"search_units": 1}, + }, + } + + +def _anthropic_message_body(model: str) -> dict: + return { + "id": "msg_mock_0001", + "type": "message", + "role": "assistant", + "model": model, + "content": [{"type": "text", "text": "Mock Anthropic message."}], + "stop_reason": "end_turn", + "stop_sequence": None, + "usage": {"input_tokens": 10, "output_tokens": 20}, + } + + +def _chat_stream_chunks(model: str) -> list[dict]: + base = { + "id": "chatcmpl-mock-0001", + "object": "chat.completion.chunk", + "created": _now(), + "model": model, + } + return [ + { + **base, + "choices": [ + { + "index": 0, + "delta": {"role": "assistant", "content": "Hi"}, + "finish_reason": None, + } + ], + }, + { + **base, + "choices": [ + { + "index": 0, + "delta": {"content": "! This is a mock"}, + "finish_reason": None, + } + ], + }, + { + **base, + "choices": [ + {"index": 0, "delta": {"content": " response."}, "finish_reason": None} + ], + }, + {**base, "choices": [{"index": 0, "delta": {}, "finish_reason": "stop"}]}, + ] + + +class MockOpenAIHandler(BaseHTTPRequestHandler): + server_version = "MockOpenAI/1.0" + + def log_message( + self, format: str, *args + ) -> None: # noqa: A002 - signature dictated by base + logger.debug( + "%s - - [%s] %s", + self.address_string(), + self.log_date_time_string(), + format % args, + ) + + def _read_body(self) -> dict: + length = int(self.headers.get("Content-Length") or 0) + if length == 0: + return {} + raw = self.rfile.read(length) + try: + return json.loads(raw.decode("utf-8") or "{}") + except json.JSONDecodeError: + return {} + + def _send_json(self, status: int, payload: dict) -> None: + body = json.dumps(payload).encode("utf-8") + self.send_response(status) + self.send_header("Content-Type", "application/json") + self.send_header("Content-Length", str(len(body))) + self.end_headers() + self.wfile.write(body) + + def _send_sse(self, chunks: list[dict]) -> None: + self.send_response(200) + self.send_header("Content-Type", "text/event-stream") + self.send_header("Cache-Control", "no-cache") + self.send_header("Connection", "keep-alive") + self.end_headers() + for chunk in chunks: + self.wfile.write(f"data: {json.dumps(chunk)}\n\n".encode("utf-8")) + self.wfile.flush() + self.wfile.write(b"data: [DONE]\n\n") + self.wfile.flush() + + def _route(self) -> str: + path = self.path.split("?", 1)[0].rstrip("/") or "/" + for prefix in ("/v1", "/v2"): + if path.startswith(prefix + "/"): + return path[len(prefix) :] + if path == prefix: + return "/" + return path + + def do_GET(self) -> None: + route = self._route() + if route == "/models": + self._send_json( + 200, + { + "object": "list", + "data": [ + { + "id": "gpt-3.5-turbo", + "object": "model", + "created": _now(), + "owned_by": "mock", + }, + { + "id": "gpt-4", + "object": "model", + "created": _now(), + "owned_by": "mock", + }, + ], + }, + ) + return + self._send_json(200, {"status": "ok", "path": self.path}) + + def do_POST(self) -> None: + body = self._read_body() + route = self._route() + model = body.get("model") or "gpt-3.5-turbo" + + if route == "/chat/completions": + if body.get("stream"): + self._send_sse(_chat_stream_chunks(model)) + else: + self._send_json(200, _chat_completion_body(model)) + return + + if route == "/completions": + self._send_json(200, _text_completion_body(model)) + return + + if route == "/embeddings": + inputs = body.get("input") + count = len(inputs) if isinstance(inputs, list) else 1 + self._send_json(200, _embedding_body(model, count)) + return + + if route == "/rerank": + self._send_json(200, _rerank_body()) + return + + if route == "/messages": + self._send_json(200, _anthropic_message_body(model)) + return + + # Catch-all: respond OK so tests asserting on api_base reachability pass. + self._send_json(200, {"status": "ok", "path": self.path}) + + +def serve(host: str, port: int) -> None: + server = ThreadingHTTPServer((host, port), MockOpenAIHandler) + logger.info("mock openai server listening on http://%s:%d", host, port) + try: + server.serve_forever() + except KeyboardInterrupt: + pass + finally: + server.server_close() + + +def main() -> None: + parser = argparse.ArgumentParser( + description="Static OpenAI-format mock server for LiteLLM CI." + ) + parser.add_argument("--host", default="0.0.0.0") + parser.add_argument("--port", type=int, default=8090) + parser.add_argument("--verbose", action="store_true") + args = parser.parse_args() + logging.basicConfig( + level=logging.DEBUG if args.verbose else logging.INFO, + format="%(asctime)s %(levelname)s %(message)s", + ) + serve(args.host, args.port) + + +if __name__ == "__main__": + main()