diff --git a/deploy/charts/litellm-helm/values.yaml b/deploy/charts/litellm-helm/values.yaml index 81558ed5b29..173af282928 100644 --- a/deploy/charts/litellm-helm/values.yaml +++ b/deploy/charts/litellm-helm/values.yaml @@ -161,7 +161,7 @@ proxy_config: litellm_params: model: openai/fake api_key: fake-key - api_base: https://exampleopenaiendpoint-production.up.railway.app/ + api_base: https://api.openai.com/v1/ # replace with your provider's URL general_settings: master_key: os.environ/PROXY_MASTER_KEY diff --git a/litellm/proxy/example_config_yaml/pass_through_config.yaml b/litellm/proxy/example_config_yaml/pass_through_config.yaml index 373ee189f3f..f26ce57e87d 100644 --- a/litellm/proxy/example_config_yaml/pass_through_config.yaml +++ b/litellm/proxy/example_config_yaml/pass_through_config.yaml @@ -3,7 +3,7 @@ model_list: litellm_params: model: openai/fake api_key: fake-key - api_base: https://exampleopenaiendpoint-production.up.railway.app/ + api_base: http://host.docker.internal:8090/ - model_name: claude-sonnet-4-5-20250929 litellm_params: model: anthropic/claude-sonnet-4-5-20250929 diff --git a/litellm/proxy/example_config_yaml/store_model_db_config.yaml b/litellm/proxy/example_config_yaml/store_model_db_config.yaml index b9cd2302046..91dee9f9073 100644 --- a/litellm/proxy/example_config_yaml/store_model_db_config.yaml +++ b/litellm/proxy/example_config_yaml/store_model_db_config.yaml @@ -3,7 +3,7 @@ model_list: litellm_params: model: openai/my-fake-model api_key: my-fake-key - api_base: https://exampleopenaiendpoint-production.up.railway.app/ + api_base: http://host.docker.internal:8090/ general_settings: store_model_in_db: true diff --git a/litellm/proxy/example_config_yaml/test_pipeline_config.yaml b/litellm/proxy/example_config_yaml/test_pipeline_config.yaml index d3a8c56b48a..a027207eb9c 100644 --- a/litellm/proxy/example_config_yaml/test_pipeline_config.yaml +++ b/litellm/proxy/example_config_yaml/test_pipeline_config.yaml @@ -3,12 +3,12 @@ model_list: litellm_params: model: openai/gpt-3.5-turbo api_key: fake-key - api_base: https://exampleopenaiendpoint-production.up.railway.app/ + api_base: http://host.docker.internal:8090/ - model_name: fake-blocked-endpoint litellm_params: model: openai/gpt-3.5-turbo api_key: fake-key - api_base: https://exampleopenaiendpoint-production.up.railway.app/ + api_base: http://host.docker.internal:8090/ guardrails: - guardrail_name: "strict-filter" diff --git a/litellm/proxy/model_config.yaml b/litellm/proxy/model_config.yaml index a0399c0952c..f1918c0298e 100644 --- a/litellm/proxy/model_config.yaml +++ b/litellm/proxy/model_config.yaml @@ -2,9 +2,9 @@ model_list: - model_name: gpt-4o litellm_params: model: openai/gpt-4o - api_base: https://exampleopenaiendpoint-production.up.railway.app/ + api_base: http://host.docker.internal:8090/ - model_name: fake-anthropic-endpoint litellm_params: model: anthropic/fake - api_base: https://exampleanthropicendpoint-production.up.railway.app/ + api_base: http://host.docker.internal:8090/ diff --git a/litellm/proxy/proxy_server.py b/litellm/proxy/proxy_server.py index 879914e5ac6..f654851de64 100644 --- a/litellm/proxy/proxy_server.py +++ b/litellm/proxy/proxy_server.py @@ -12254,7 +12254,7 @@ async def model_info_v1( # noqa: PLR0915 { "model_name": "fake-openai-endpoint", "litellm_params": { - "api_base": "https://exampleopenaiendpoint-production.up.railway.app/", + "api_base": "http://host.docker.internal:8090/", "model": "openai/fake" }, "model_info": { diff --git a/tests/batches_tests/test_hosted_vllm_batches_and_files.py b/tests/batches_tests/test_hosted_vllm_batches_and_files.py index c7a25c71c53..cc8145a80d7 100644 --- a/tests/batches_tests/test_hosted_vllm_batches_and_files.py +++ b/tests/batches_tests/test_hosted_vllm_batches_and_files.py @@ -21,7 +21,7 @@ sys.path.insert(0, os.path.abspath("../..")) import litellm -SERVER_URL = "https://exampleopenaiendpoint-production-0ee2.up.railway.app/v1" +SERVER_URL = "http://127.0.0.1:8090/v1" @pytest.mark.asyncio() diff --git a/tests/llm_translation/test_deepseek_completion.py b/tests/llm_translation/test_deepseek_completion.py index 2ede5d3f3f8..9f3aa01c12d 100644 --- a/tests/llm_translation/test_deepseek_completion.py +++ b/tests/llm_translation/test_deepseek_completion.py @@ -29,7 +29,7 @@ def test_deepseek_mock_completion(stream): response = completion( model="deepseek/deepseek-reasoner", messages=[{"role": "user", "content": "Hello, world!"}], - api_base="https://exampleopenaiendpoint-production.up.railway.app/v1/chat/completions", + api_base="http://127.0.0.1:8090/v1/chat/completions", stream=stream, mock_response="Hello! How can I help you today?", ) diff --git a/tests/llm_translation/test_rerank.py b/tests/llm_translation/test_rerank.py index d784677060a..0910c92d12e 100644 --- a/tests/llm_translation/test_rerank.py +++ b/tests/llm_translation/test_rerank.py @@ -175,7 +175,7 @@ async def test_rerank_custom_api_base(version): "documents": ["hello", "world"], } - api_base = "https://exampleopenaiendpoint-production.up.railway.app/" + api_base = "http://127.0.0.1:8090/" if version == "v1": api_base += "v1/rerank" @@ -202,7 +202,7 @@ async def test_rerank_custom_api_base(version): print("url = ", _url) assert ( _url - == f"https://exampleopenaiendpoint-production.up.railway.app/{version}/rerank" + == f"http://127.0.0.1:8090/{version}/rerank" ) request_data = json.loads(args_to_api) diff --git a/tests/llm_translation/test_triton.py b/tests/llm_translation/test_triton.py index 2d1ca39e1dc..7327be73524 100644 --- a/tests/llm_translation/test_triton.py +++ b/tests/llm_translation/test_triton.py @@ -360,7 +360,7 @@ async def test_triton_embeddings(): litellm.set_verbose = True response = await litellm.aembedding( model="triton/my-triton-model", - api_base="https://exampleopenaiendpoint-production.up.railway.app/triton/embeddings", + api_base="http://127.0.0.1:8090/triton/embeddings", input=["good morning from litellm"], ) print(f"response: {response}") diff --git a/tests/load_tests/test_datadog_load_test.py b/tests/load_tests/test_datadog_load_test.py index f4328b71b1b..dc724d7b013 100644 --- a/tests/load_tests/test_datadog_load_test.py +++ b/tests/load_tests/test_datadog_load_test.py @@ -15,7 +15,7 @@ def test_datadog_logging_async(): # litellm.set_verbose = True os.environ["DD_API_KEY"] = "anything" os.environ["_DATADOG_BASE_URL"] = ( - "https://exampleopenaiendpoint-production.up.railway.app" + "http://127.0.0.1:8090" ) os.environ["DD_SITE"] = "us5.datadoghq.com" diff --git a/tests/load_tests/test_langsmith_load_test.py b/tests/load_tests/test_langsmith_load_test.py index cf9fe526b74..f0532e36a80 100644 --- a/tests/load_tests/test_langsmith_load_test.py +++ b/tests/load_tests/test_langsmith_load_test.py @@ -17,7 +17,7 @@ def test_langsmith_logging_async(): os.environ["LANGSMITH_API_KEY"] = "lsv2_anything" os.environ["LANGSMITH_PROJECT"] = "pr-b" os.environ["LANGSMITH_BASE_URL"] = ( - "https://exampleopenaiendpoint-production.up.railway.app" + "http://127.0.0.1:8090" ) percentage_diffs = [] diff --git a/tests/load_tests/test_memory_usage.py b/tests/load_tests/test_memory_usage.py index f273865a29a..9119f979077 100644 --- a/tests/load_tests/test_memory_usage.py +++ b/tests/load_tests/test_memory_usage.py @@ -72,7 +72,7 @@ async def make_completion_request(): return await litellm.acompletion( model="openai/gpt-4o", messages=[{"role": "user", "content": "Test message for memory usage"}], - api_base="https://exampleopenaiendpoint-production.up.railway.app/", + api_base="http://127.0.0.1:8090/", ) @@ -80,7 +80,7 @@ async def make_text_completion_request(): return await litellm.atext_completion( model="openai/gpt-4o", prompt="Test message for memory usage", - api_base="https://exampleopenaiendpoint-production.up.railway.app/", + api_base="http://127.0.0.1:8090/", ) @@ -108,14 +108,14 @@ litellm_router = Router( "model_name": "text-gpt-4o", "litellm_params": { "model": "text-completion-openai/gpt-3.5-turbo-instruct-unlimited", - "api_base": "https://exampleopenaiendpoint-production.up.railway.app/", + "api_base": "http://127.0.0.1:8090/", }, }, { "model_name": "chat-gpt-4o", "litellm_params": { "model": "openai/gpt-4o", - "api_base": "https://exampleopenaiendpoint-production.up.railway.app/", + "api_base": "http://127.0.0.1:8090/", }, }, ] @@ -128,7 +128,7 @@ async def make_router_atext_completion_request(): temperature=0.5, frequency_penalty=0.5, prompt="<|fim prefix|> Test message for memory usage <|fim prefix|> Test message for memory usage", - api_base="https://exampleopenaiendpoint-production.up.railway.app/", + api_base="http://127.0.0.1:8090/", max_tokens=500, ) @@ -148,7 +148,7 @@ async def make_router_acompletion_request(): return await litellm_router.acompletion( model="chat-gpt-4o", messages=[{"role": "user", "content": "Test message for memory usage"}], - api_base="https://exampleopenaiendpoint-production.up.railway.app/", + api_base="http://127.0.0.1:8090/", ) diff --git a/tests/load_tests/test_otel_load_test.py b/tests/load_tests/test_otel_load_test.py index f5754c0c402..432202a26ea 100644 --- a/tests/load_tests/test_otel_load_test.py +++ b/tests/load_tests/test_otel_load_test.py @@ -16,7 +16,7 @@ def test_otel_logging_async(): try: os.environ["OTEL_EXPORTER"] = "otlp_http" os.environ["OTEL_ENDPOINT"] = ( - "https://exampleopenaiendpoint-production.up.railway.app/traces" + "http://127.0.0.1:8090/traces" ) os.environ["OTEL_HEADERS"] = "Authorization=K0BSwd" diff --git a/tests/load_tests/test_vertex_embeddings_load_test.py b/tests/load_tests/test_vertex_embeddings_load_test.py index 9beee710553..83060ec94bc 100644 --- a/tests/load_tests/test_vertex_embeddings_load_test.py +++ b/tests/load_tests/test_vertex_embeddings_load_test.py @@ -59,7 +59,7 @@ def load_vertex_ai_credentials(): async def create_async_vertex_embedding_task(): load_vertex_ai_credentials() - base_url = "https://exampleopenaiendpoint-production.up.railway.app/v1/projects/pathrise-convert-1606954137718/locations/us-central1/publishers/google/models/textembedding-gecko@001" + base_url = "http://127.0.0.1:8090/v1/projects/pathrise-convert-1606954137718/locations/us-central1/publishers/google/models/textembedding-gecko@001" embedding_args = { "model": "vertex_ai/textembedding-gecko", "input": "This is a test sentence for embedding.", diff --git a/tests/load_tests/test_vertex_load_tests.py b/tests/load_tests/test_vertex_load_tests.py index 9130873b970..b967bb00ada 100644 --- a/tests/load_tests/test_vertex_load_tests.py +++ b/tests/load_tests/test_vertex_load_tests.py @@ -118,7 +118,7 @@ async def make_async_calls(message_type="text"): def create_async_task(message_type): - base_url = "https://exampleopenaiendpoint-production.up.railway.app/v1/projects/pathrise-convert-1606954137718/locations/us-central1/publishers/google/models/gemini-1.0-pro-vision-001" + base_url = "http://127.0.0.1:8090/v1/projects/pathrise-convert-1606954137718/locations/us-central1/publishers/google/models/gemini-1.0-pro-vision-001" if message_type == "text": messages = [{"role": "user", "content": "hi"}] diff --git a/tests/local_testing/test_completion.py b/tests/local_testing/test_completion.py index cce6d33e799..918827cd4f4 100644 --- a/tests/local_testing/test_completion.py +++ b/tests/local_testing/test_completion.py @@ -1343,7 +1343,7 @@ def test_lm_studio_completion(monkeypatch): messages=[ {"role": "user", "content": "What's the weather like in San Francisco?"} ], - api_base="https://exampleopenaiendpoint-production.up.railway.app/", + api_base="http://127.0.0.1:8090/", ) except litellm.AuthenticationError as e: pytest.fail(f"Error occurred: {e}") diff --git a/tests/local_testing/test_completion_cost.py b/tests/local_testing/test_completion_cost.py index cf0c645615d..06a4a11e513 100644 --- a/tests/local_testing/test_completion_cost.py +++ b/tests/local_testing/test_completion_cost.py @@ -2756,7 +2756,7 @@ def model_item(): "litellm_params": { "model": "openai/my-fake-model", "api_key": "my-fake-key", - "api_base": "https://exampleopenaiendpoint-production.up.railway.app/", + "api_base": "http://127.0.0.1:8090/", }, "model_info": {}, } diff --git a/tests/local_testing/test_lakera_ai_prompt_injection.py b/tests/local_testing/test_lakera_ai_prompt_injection.py index 0d6cc20846b..20a1767a669 100644 --- a/tests/local_testing/test_lakera_ai_prompt_injection.py +++ b/tests/local_testing/test_lakera_ai_prompt_injection.py @@ -158,7 +158,7 @@ async def test_moderations_on_embeddings(): "litellm_params": { "model": "text-embedding-ada-002", "api_key": "any", - "api_base": "https://exampleopenaiendpoint-production.up.railway.app/", + "api_base": "http://127.0.0.1:8090/", }, }, ] diff --git a/tests/local_testing/test_lowest_latency_routing.py b/tests/local_testing/test_lowest_latency_routing.py index 90913499e55..23afb20411a 100644 --- a/tests/local_testing/test_lowest_latency_routing.py +++ b/tests/local_testing/test_lowest_latency_routing.py @@ -591,7 +591,7 @@ async def test_lowest_latency_routing_with_timeouts(): "model_name": "azure-model", "litellm_params": { "model": "openai/slow-endpoint", - "api_base": "https://exampleopenaiendpoint-production-c715.up.railway.app/", # If you are Krrish, this is OpenAI Endpoint3 on our Railway endpoint :) + "api_base": "http://127.0.0.1:8090/slow/", # If you are Krrish, this is OpenAI Endpoint3 on our Railway endpoint :) "api_key": "fake-key", }, "model_info": {"id": "slow-endpoint"}, @@ -600,7 +600,7 @@ async def test_lowest_latency_routing_with_timeouts(): "model_name": "azure-model", "litellm_params": { "model": "openai/fast-endpoint", - "api_base": "https://exampleopenaiendpoint-production.up.railway.app/", + "api_base": "http://127.0.0.1:8090/", "api_key": "fake-key", }, "model_info": {"id": "fast-endpoint"}, @@ -666,7 +666,7 @@ async def test_lowest_latency_routing_first_pick(): "model_name": "azure-model", "litellm_params": { "model": "openai/fast-endpoint", - "api_base": "https://exampleopenaiendpoint-production.up.railway.app/", + "api_base": "http://127.0.0.1:8090/", "api_key": "fake-key", }, "model_info": {"id": "fast-endpoint"}, @@ -675,7 +675,7 @@ async def test_lowest_latency_routing_first_pick(): "model_name": "azure-model", "litellm_params": { "model": "openai/fast-endpoint-2", - "api_base": "https://exampleopenaiendpoint-production.up.railway.app/", + "api_base": "http://127.0.0.1:8090/", "api_key": "fake-key", }, "model_info": {"id": "fast-endpoint-2"}, @@ -684,7 +684,7 @@ async def test_lowest_latency_routing_first_pick(): "model_name": "azure-model", "litellm_params": { "model": "openai/fast-endpoint-2", - "api_base": "https://exampleopenaiendpoint-production.up.railway.app/", + "api_base": "http://127.0.0.1:8090/", "api_key": "fake-key", }, "model_info": {"id": "fast-endpoint-3"}, @@ -693,7 +693,7 @@ async def test_lowest_latency_routing_first_pick(): "model_name": "azure-model", "litellm_params": { "model": "openai/fast-endpoint-2", - "api_base": "https://exampleopenaiendpoint-production.up.railway.app/", + "api_base": "http://127.0.0.1:8090/", "api_key": "fake-key", }, "model_info": {"id": "fast-endpoint-4"}, diff --git a/tests/local_testing/test_secret_detect_hook.py b/tests/local_testing/test_secret_detect_hook.py index 8340fb91482..1bedaa4252b 100644 --- a/tests/local_testing/test_secret_detect_hook.py +++ b/tests/local_testing/test_secret_detect_hook.py @@ -246,7 +246,7 @@ router = Router( "model_name": "fake-model", "litellm_params": { "model": "openai/fake", - "api_base": "https://exampleopenaiendpoint-production.up.railway.app/", + "api_base": "http://127.0.0.1:8090/", "api_key": "sk-12345", }, } diff --git a/tests/logging_callback_tests/test_view_request_resp_logs.py b/tests/logging_callback_tests/test_view_request_resp_logs.py index ea778a44e67..f2bb415e0b7 100644 --- a/tests/logging_callback_tests/test_view_request_resp_logs.py +++ b/tests/logging_callback_tests/test_view_request_resp_logs.py @@ -61,7 +61,7 @@ mock_response_data = { "completion_tokens": 12, "request_tags": [], "end_user": "", - "api_base": "https://exampleopenaiendpoint-production.up.railway.app", + "api_base": "http://127.0.0.1:8090", "model_group": "fake-openai-endpoint", "model_id": "b68d56d76b0c24ac9462ab69541e90886342508212210116e300441155f37865", "requester_ip_address": "127.0.0.1", @@ -100,7 +100,7 @@ mock_response_data = { "hidden_params": { "model_id": "b68d56d76b0c24ac9462ab69541e90886342508212210116e300441155f37865", "cache_key": None, - "api_base": "https://exampleopenaiendpoint-production.up.railway.app/", + "api_base": "http://127.0.0.1:8090/", "response_cost": 3.7500000000000003e-05, "additional_headers": {}, "litellm_overhead_time_ms": 2.126, diff --git a/tests/mock_endpoints/openai_mock_server.py b/tests/mock_endpoints/openai_mock_server.py index ef7f4ad8b2c..0b897c66afc 100644 --- a/tests/mock_endpoints/openai_mock_server.py +++ b/tests/mock_endpoints/openai_mock_server.py @@ -9,6 +9,22 @@ Run standalone: The server is intentionally implemented with the Python standard library only (no FastAPI / httpx / pydantic) so CI jobs can start it before installing test dependencies. + +Response shapes are matched to the historical Railway endpoint to preserve +test intent: + * /chat/completions -> OpenAI chat.completion (non-stream) or SSE + stream of chat.completion.chunk events. + Content: "Hello this is a test response from + a fixed OpenAI endpoint. " (stream) or + "\\n\\nHello there, how may I assist you + today?" (non-stream). + * /completions -> OpenAI text_completion. + * /embeddings -> OpenAI embedding list. + * /rerank, /v2/rerank -> Cohere-shape rerank result. + * /v1/messages -> Anthropic Messages. + +Special model handling matches Railway: + * model == "429" returns HTTP 429 (used by fake-azure-endpoint fixtures). """ from __future__ import annotations @@ -22,81 +38,86 @@ from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer logger = logging.getLogger("openai_mock_server") _EMBEDDING_DIM = 1536 +_CREATED_TS = 1677652288 +_SYSTEM_FINGERPRINT = "fp_44709d6fcb" +_CHAT_CONTENT = "\n\nHello there, how may I assist you today?" +_STREAM_TOKENS = [ + "Hello ", + "this ", + "is ", + "a ", + "test ", + "response ", + "from ", + "a ", + "fixed ", + "OpenAI ", + "endpoint. ", +] -def _now() -> int: - return int(time.time()) - - -def _chat_completion_body( - model: str, content: str = "Hi! This is a mock response." -) -> dict: +def _chat_completion_body(model: str) -> dict: return { - "id": "chatcmpl-mock-0001", + "id": "chatcmpl-c055ccfa83b84490af310d4bb5552422", "object": "chat.completion", - "created": _now(), + "created": _CREATED_TS, "model": model, - "system_fingerprint": "fp_mock", + "system_fingerprint": _SYSTEM_FINGERPRINT, "choices": [ { "index": 0, - "message": {"role": "assistant", "content": content}, + "message": {"role": "assistant", "content": _CHAT_CONTENT}, "logprobs": None, "finish_reason": "stop", } ], - "usage": { - "prompt_tokens": 10, - "completion_tokens": 20, - "total_tokens": 30, - "prompt_tokens_details": {"cached_tokens": 0, "audio_tokens": 0}, - "completion_tokens_details": { - "reasoning_tokens": 0, - "audio_tokens": 0, - "accepted_prediction_tokens": 0, - "rejected_prediction_tokens": 0, - }, - }, + "usage": {"prompt_tokens": 9, "completion_tokens": 12, "total_tokens": 21}, } def _text_completion_body(model: str) -> dict: return { - "id": "cmpl-mock-0001", + "id": "cmpl-9B2ycsf0odECdLmrVzm2y8Q12csjW", "object": "text_completion", - "created": _now(), + "created": _CREATED_TS, "model": model, + "system_fingerprint": None, "choices": [ { - "text": "Mock completion response.", + "text": "\n\nA test request, how intriguing\n" + "An invitation for knowledge bringing\nWith words", "index": 0, "logprobs": None, - "finish_reason": "stop", + "finish_reason": "length", } ], - "usage": {"prompt_tokens": 5, "completion_tokens": 5, "total_tokens": 10}, + "usage": {"prompt_tokens": 10, "completion_tokens": 16, "total_tokens": 26}, } def _embedding_body(model: str, input_count: int) -> dict: + # Match Railway: a constant 1536-d vector tiled from a short pattern. + pattern = [ + -0.006929283495992422, + -0.005336422007530928, + -4.547132266452536e-05, + -0.024047505110502243, + ] + vector = (pattern * ((_EMBEDDING_DIM // len(pattern)) + 1))[:_EMBEDDING_DIM] return { "object": "list", "model": model, "data": [ - { - "object": "embedding", - "index": i, - "embedding": [0.0] * _EMBEDDING_DIM, - } + {"object": "embedding", "index": i, "embedding": vector} for i in range(max(input_count, 1)) ], "usage": {"prompt_tokens": input_count or 1, "total_tokens": input_count or 1}, } -def _rerank_body(query_id: str = "rerank-mock-0001") -> dict: +def _rerank_body() -> dict: return { - "id": query_id, + "id": "rerank-mock-0001", "results": [{"index": 0, "relevance_score": 0.99}], "meta": { "api_version": {"version": "1"}, @@ -107,52 +128,29 @@ def _rerank_body(query_id: str = "rerank-mock-0001") -> dict: def _anthropic_message_body(model: str) -> dict: return { - "id": "msg_mock_0001", + "id": "msg_01G7MsdWPT2JZMUuc1UXRavn", "type": "message", "role": "assistant", "model": model, - "content": [{"type": "text", "text": "Mock Anthropic message."}], + "content": [{"type": "text", "text": _CHAT_CONTENT}], "stop_reason": "end_turn", "stop_sequence": None, - "usage": {"input_tokens": 10, "output_tokens": 20}, + "usage": {"input_tokens": 10, "output_tokens": 20, "total_tokens": 30}, } def _chat_stream_chunks(model: str) -> list[dict]: base = { - "id": "chatcmpl-mock-0001", + "id": "chatcmpl-b0e9cfbf26d148928e8501842b2af4de", "object": "chat.completion.chunk", - "created": _now(), + "created": _CREATED_TS, "model": model, } + # Match Railway: deltas carry only `content` (no role, no finish_reason), + # and the stream ends after the last token without a [DONE] sentinel. return [ - { - **base, - "choices": [ - { - "index": 0, - "delta": {"role": "assistant", "content": "Hi"}, - "finish_reason": None, - } - ], - }, - { - **base, - "choices": [ - { - "index": 0, - "delta": {"content": "! This is a mock"}, - "finish_reason": None, - } - ], - }, - { - **base, - "choices": [ - {"index": 0, "delta": {"content": " response."}, "finish_reason": None} - ], - }, - {**base, "choices": [{"index": 0, "delta": {}, "finish_reason": "stop"}]}, + {**base, "choices": [{"index": 0, "delta": {"content": token}}]} + for token in _STREAM_TOKENS ] @@ -189,18 +187,46 @@ class MockOpenAIHandler(BaseHTTPRequestHandler): def _send_sse(self, chunks: list[dict]) -> None: self.send_response(200) - self.send_header("Content-Type", "text/event-stream") + self.send_header("Content-Type", "text/event-stream; charset=utf-8") self.send_header("Cache-Control", "no-cache") - self.send_header("Connection", "keep-alive") + self.send_header("Connection", "close") self.end_headers() for chunk in chunks: self.wfile.write(f"data: {json.dumps(chunk)}\n\n".encode("utf-8")) self.wfile.flush() + # OpenAI clients (httpx / openai-python) terminate on a `data: [DONE]` + # sentinel; Railway omitted it and relied on connection close. Sending + # it explicitly is safe (LiteLLM tolerates either) and avoids the + # 30s-keep-alive hang seen with the raw stdlib server. self.wfile.write(b"data: [DONE]\n\n") self.wfile.flush() - def _route(self) -> str: + # Path prefixes that simulate latency before responding. Used by tests + # like test_lowest_latency_routing_with_timeouts that historically pointed + # the "slow deployment" at a separate Railway URL which was naturally slow. + _DELAY_PREFIXES = { + "/slow": 2.0, + } + + def _maybe_delay(self) -> None: + path = self.path.split("?", 1)[0] + for prefix, seconds in self._DELAY_PREFIXES.items(): + if path.startswith(prefix + "/") or path == prefix: + time.sleep(seconds) + return + + def _stripped_path(self) -> str: + """Path with `/slow` (delay) prefix removed but `/v1`, `/v2` preserved.""" path = self.path.split("?", 1)[0].rstrip("/") or "/" + for prefix in self._DELAY_PREFIXES: + if path.startswith(prefix + "/"): + return path[len(prefix) :] + if path == prefix: + return "/" + return path + + def _route(self) -> str: + path = self._stripped_path() for prefix in ("/v1", "/v2"): if path.startswith(prefix + "/"): return path[len(prefix) :] @@ -219,13 +245,13 @@ class MockOpenAIHandler(BaseHTTPRequestHandler): { "id": "gpt-3.5-turbo", "object": "model", - "created": _now(), + "created": _CREATED_TS, "owned_by": "mock", }, { "id": "gpt-4", "object": "model", - "created": _now(), + "created": _CREATED_TS, "owned_by": "mock", }, ], @@ -236,10 +262,27 @@ class MockOpenAIHandler(BaseHTTPRequestHandler): def do_POST(self) -> None: body = self._read_body() + self._maybe_delay() + raw_path = self._stripped_path() route = self._route() model = body.get("model") or "gpt-3.5-turbo" + # Railway convention: `model == "429"` triggers a rate-limit response. + # Used by `fake-azure-endpoint` fixtures to exercise retry/fallback paths. + if model == "429" and route in ("/chat/completions", "/completions"): + self._send_json(429, {"detail": "Too many requests"}) + return + if route == "/chat/completions": + # Railway quirk we have to preserve: POST /v1/chat/completions + # returns an Anthropic-shape Messages response (vs. the OpenAI shape + # served at /chat/completions). LiteLLM's OpenAI parser raises + # InternalServerError on that response, and a few tests + # (e.g. test_router_prompt_caching) defensively rely on that path + # via try/except. Matching the quirk keeps their intent intact. + if raw_path.startswith("/v1/"): + self._send_json(200, _anthropic_message_body(model)) + return if body.get("stream"): self._send_sse(_chat_stream_chunks(model)) else: diff --git a/tests/pass_through_unit_tests/test_pass_through_unit_tests.py b/tests/pass_through_unit_tests/test_pass_through_unit_tests.py index 1b16177b755..c722d76a2a6 100644 --- a/tests/pass_through_unit_tests/test_pass_through_unit_tests.py +++ b/tests/pass_through_unit_tests/test_pass_through_unit_tests.py @@ -290,7 +290,7 @@ async def test_pass_through_request_logging_failure( ) response = await pass_through_request( request=request, - target="https://exampleopenaiendpoint-production.up.railway.app/v1/messages", + target="http://127.0.0.1:8090/v1/messages", custom_headers={}, user_api_key_dict=mock_user_api_key_dict, ) @@ -357,7 +357,7 @@ async def test_pass_through_request_logging_failure_with_stream( ) response = await pass_through_request( request=request, - target="https://exampleopenaiendpoint-production.up.railway.app/v1/messages", + target="http://127.0.0.1:8090/v1/messages", custom_headers={}, user_api_key_dict=mock_user_api_key_dict, ) diff --git a/tests/proxy_unit_tests/test_proxy_reject_logging.py b/tests/proxy_unit_tests/test_proxy_reject_logging.py index 51a92fa3b4b..45f0179a854 100644 --- a/tests/proxy_unit_tests/test_proxy_reject_logging.py +++ b/tests/proxy_unit_tests/test_proxy_reject_logging.py @@ -87,7 +87,7 @@ router = Router( "model_name": "fake-model", "litellm_params": { "model": "openai/fake", - "api_base": "https://exampleopenaiendpoint-production.up.railway.app/", + "api_base": "http://127.0.0.1:8090/", "api_key": "sk-12345", }, } diff --git a/tests/proxy_unit_tests/test_proxy_token_counter.py b/tests/proxy_unit_tests/test_proxy_token_counter.py index 1079a5228a1..1b62c5626de 100644 --- a/tests/proxy_unit_tests/test_proxy_token_counter.py +++ b/tests/proxy_unit_tests/test_proxy_token_counter.py @@ -127,7 +127,7 @@ async def test_vLLM_token_counting(): "model_name": "special-alias", "litellm_params": { "model": "openai/wolfram/miquliz-120b-v2.0", - "api_base": "https://exampleopenaiendpoint-production.up.railway.app/", + "api_base": "http://127.0.0.1:8090/", }, } ] diff --git a/tests/router_unit_tests/test_router_prompt_caching.py b/tests/router_unit_tests/test_router_prompt_caching.py index 574eccda162..2cef5e93e4a 100644 --- a/tests/router_unit_tests/test_router_prompt_caching.py +++ b/tests/router_unit_tests/test_router_prompt_caching.py @@ -125,7 +125,7 @@ async def test_router_prompt_caching_same_cacheable_prefix_routes_to_same_deploy "model_name": "test-model", "litellm_params": { "model": "gpt-5-mini", - "api_base": "https://exampleopenaiendpoint-production-0ee2.up.railway.app/v1", + "api_base": "http://127.0.0.1:8090/v1", "api_key": f"test-key-{i}", }, "model_info": {"id": f"deployment-{i}"}, diff --git a/tests/store_model_in_db_tests/test_callbacks_in_db.py b/tests/store_model_in_db_tests/test_callbacks_in_db.py index 4a851251a3e..4ddefaff4e1 100644 --- a/tests/store_model_in_db_tests/test_callbacks_in_db.py +++ b/tests/store_model_in_db_tests/test_callbacks_in_db.py @@ -21,7 +21,7 @@ from openai.types.chat import ChatCompletion load_dotenv() # used for testing -LANGFUSE_BASE_URL = "https://exampleopenaiendpoint-production-c715.up.railway.app" +LANGFUSE_BASE_URL = "http://127.0.0.1:8090/slow" async def config_update(session, routing_strategy=None): diff --git a/tests/test_litellm/llms/azure/test_azure_fine_tuning_api.py b/tests/test_litellm/llms/azure/test_azure_fine_tuning_api.py index 8d008d6c071..0e961d8a011 100644 --- a/tests/test_litellm/llms/azure/test_azure_fine_tuning_api.py +++ b/tests/test_litellm/llms/azure/test_azure_fine_tuning_api.py @@ -36,7 +36,7 @@ def _mock_azure_client( client = AsyncAzureOpenAI( api_key="test-key", api_version="2024-10-21", - azure_endpoint="https://exampleopenaiendpoint-production.up.railway.app", + azure_endpoint="http://127.0.0.1:8090", ) client.fine_tuning.jobs.create = AsyncMock( return_value=( @@ -69,7 +69,7 @@ async def test_azure_acreate_fine_tuning_job_request_and_output_match_expected_j model="gpt-35-turbo-1106", training_file="file-5e4b20ecbd724182b9964f3cd2ab7212", custom_llm_provider="azure", - api_base="https://exampleopenaiendpoint-production.up.railway.app", + api_base="http://127.0.0.1:8090", api_key="test-key", api_version="2024-10-21", ) @@ -100,7 +100,7 @@ async def test_azure_alist_fine_tuning_jobs_request_matches_expected_json(): after=expected_request["after"], limit=expected_request["limit"], custom_llm_provider="azure", - api_base="https://exampleopenaiendpoint-production.up.railway.app", + api_base="http://127.0.0.1:8090", api_key="test-key", api_version="2024-10-21", ) @@ -124,7 +124,7 @@ async def test_azure_acancel_fine_tuning_job_request_and_output_match_expected_j response = await litellm.acancel_fine_tuning_job( fine_tuning_job_id=expected_request["fine_tuning_job_id"], custom_llm_provider="azure", - api_base="https://exampleopenaiendpoint-production.up.railway.app", + api_base="http://127.0.0.1:8090", api_key="test-key", api_version="2024-10-21", ) diff --git a/tests/test_litellm/proxy/management_endpoints/test_delete_callbacks_endpoint.py b/tests/test_litellm/proxy/management_endpoints/test_delete_callbacks_endpoint.py index 4ba656d1286..1b9775ca1c6 100644 --- a/tests/test_litellm/proxy/management_endpoints/test_delete_callbacks_endpoint.py +++ b/tests/test_litellm/proxy/management_endpoints/test_delete_callbacks_endpoint.py @@ -26,7 +26,7 @@ class MockPrismaClient: "environment_variables": { "LANGFUSE_PUBLIC_KEY": "any-public-key", "LANGFUSE_SECRET_KEY": "any-secret-key", - "LANGFUSE_HOST": "https://exampleopenaiendpoint-production-c715.up.railway.app", + "LANGFUSE_HOST": "http://127.0.0.1:8090/slow", }, } diff --git a/tests/test_litellm/router_strategy/test_router_tag_routing.py b/tests/test_litellm/router_strategy/test_router_tag_routing.py index a6e39ec3c0a..74ab4e71076 100644 --- a/tests/test_litellm/router_strategy/test_router_tag_routing.py +++ b/tests/test_litellm/router_strategy/test_router_tag_routing.py @@ -39,7 +39,7 @@ async def test_router_free_paid_tier(): "model_name": "gpt-4", "litellm_params": { "model": "gpt-4o", - "api_base": "https://exampleopenaiendpoint-production.up.railway.app/", + "api_base": "http://127.0.0.1:8090/", "tags": ["free"], }, "model_info": {"id": "very-cheap-model"}, @@ -48,7 +48,7 @@ async def test_router_free_paid_tier(): "model_name": "gpt-4", "litellm_params": { "model": "gpt-4o-mini", - "api_base": "https://exampleopenaiendpoint-production.up.railway.app/", + "api_base": "http://127.0.0.1:8090/", "tags": ["paid"], }, "model_info": {"id": "very-expensive-model"}, @@ -102,7 +102,7 @@ async def test_router_free_paid_tier_embeddings(): "model_name": "gpt-4", "litellm_params": { "model": "gpt-4o", - "api_base": "https://exampleopenaiendpoint-production.up.railway.app/", + "api_base": "http://127.0.0.1:8090/", "tags": ["free"], "mock_response": ["1", "2", "3"], }, @@ -112,7 +112,7 @@ async def test_router_free_paid_tier_embeddings(): "model_name": "gpt-4", "litellm_params": { "model": "gpt-4o-mini", - "api_base": "https://exampleopenaiendpoint-production.up.railway.app/", + "api_base": "http://127.0.0.1:8090/", "tags": ["paid"], "mock_response": ["1", "2", "3"], }, @@ -122,7 +122,7 @@ async def test_router_free_paid_tier_embeddings(): "model_name": "gpt-4", "litellm_params": { "model": "gpt-4o-mini", - "api_base": "https://exampleopenaiendpoint-production.up.railway.app/", + "api_base": "http://127.0.0.1:8090/", "tags": ["default"], "mock_response": ["1", "2", "3"], }, @@ -178,7 +178,7 @@ async def test_default_tagged_deployments(): "model_name": "gpt-4", "litellm_params": { "model": "gpt-4o", - "api_base": "https://exampleopenaiendpoint-production.up.railway.app/", + "api_base": "http://127.0.0.1:8090/", "tags": ["default"], }, "model_info": {"id": "default-model"}, @@ -187,7 +187,7 @@ async def test_default_tagged_deployments(): "model_name": "gpt-4", "litellm_params": { "model": "gpt-4o", - "api_base": "https://exampleopenaiendpoint-production.up.railway.app/", + "api_base": "http://127.0.0.1:8090/", }, "model_info": {"id": "default-model-2"}, }, @@ -195,7 +195,7 @@ async def test_default_tagged_deployments(): "model_name": "gpt-4", "litellm_params": { "model": "gpt-4o-mini", - "api_base": "https://exampleopenaiendpoint-production.up.railway.app/", + "api_base": "http://127.0.0.1:8090/", "tags": ["teamA"], }, "model_info": {"id": "very-expensive-model"}, @@ -268,7 +268,7 @@ async def test_error_from_tag_routing(): "model_name": "gpt-4", "litellm_params": { "model": "gpt-4o", - "api_base": "https://exampleopenaiendpoint-production.up.railway.app/", + "api_base": "http://127.0.0.1:8090/", }, "model_info": {"id": "default-model"}, }, @@ -276,7 +276,7 @@ async def test_error_from_tag_routing(): "model_name": "gpt-4", "litellm_params": { "model": "gpt-4o", - "api_base": "https://exampleopenaiendpoint-production.up.railway.app/", + "api_base": "http://127.0.0.1:8090/", }, "model_info": {"id": "default-model-2"}, }, @@ -284,7 +284,7 @@ async def test_error_from_tag_routing(): "model_name": "gpt-4", "litellm_params": { "model": "gpt-4o-mini", - "api_base": "https://exampleopenaiendpoint-production.up.railway.app/", + "api_base": "http://127.0.0.1:8090/", "tags": ["teamA"], }, "model_info": {"id": "very-expensive-model"}, @@ -386,7 +386,7 @@ async def test_router_free_paid_tier_with_responses_api(): "model_name": "gpt-4", "litellm_params": { "model": "gpt-4o", - "api_base": "https://exampleopenaiendpoint-production.up.railway.app/", + "api_base": "http://127.0.0.1:8090/", "tags": ["free"], }, "model_info": {"id": "very-cheap-model"}, @@ -395,7 +395,7 @@ async def test_router_free_paid_tier_with_responses_api(): "model_name": "gpt-4", "litellm_params": { "model": "gpt-4o-mini", - "api_base": "https://exampleopenaiendpoint-production.up.railway.app/", + "api_base": "http://127.0.0.1:8090/", "tags": ["paid"], }, "model_info": {"id": "very-expensive-model"},