From 3ad5a48b91da40b4418353fce12acdb02c79ffb8 Mon Sep 17 00:00:00 2001 From: Yuneng Jiang Date: Tue, 19 May 2026 22:32:31 -0700 Subject: [PATCH] ci(test): revert Railway URL swap for tests that depend on Railway-only response shapes MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Independent fact-check found the previous follow-up commit (ba1452acb5) silently shifted test intent on 3 surfaces — revert just those: * tests/store_model_in_db_tests/test_callbacks_in_db.py + its proxy YAML — the test GETs {host}/langfuse/trace/{id} and asserts a {"exists": true} body. The Railway endpoint served that route; the in-repo mock doesn't reproduce it. Restoring Railway URL keeps the pre-existing behavior of proxy_store_model_in_db_tests until we decide whether to model that contract in the mock. * tests/llm_translation/test_triton.py — asserts response.data[0]["embedding"] == [0.1, 0.2], a Railway-specific stub. The mock returns a 1536-dim vector at /triton/embeddings, so the assertion would always fail in llm_translation_testing. * tests/load_tests/test_vertex_load_tests.py + test_vertex_embeddings_load_test.py — both call Vertex paths (/v1/projects/.../publishers/google/models/...:generateContent and :predict). Railway returns a Vertex-format response; the mock's catch-all returns {"status":"ok"}. Load tests are not in CI but are broken as standalone scripts after the migration. The 8 originally failing CI jobs are unaffected — their test files and the start_mock_openai_server wiring stay as-is. --- litellm/proxy/example_config_yaml/store_model_db_config.yaml | 2 +- tests/llm_translation/test_triton.py | 2 +- tests/load_tests/test_vertex_embeddings_load_test.py | 2 +- tests/load_tests/test_vertex_load_tests.py | 2 +- tests/store_model_in_db_tests/test_callbacks_in_db.py | 2 +- 5 files changed, 5 insertions(+), 5 deletions(-) diff --git a/litellm/proxy/example_config_yaml/store_model_db_config.yaml b/litellm/proxy/example_config_yaml/store_model_db_config.yaml index 91dee9f9073..b9cd2302046 100644 --- a/litellm/proxy/example_config_yaml/store_model_db_config.yaml +++ b/litellm/proxy/example_config_yaml/store_model_db_config.yaml @@ -3,7 +3,7 @@ model_list: litellm_params: model: openai/my-fake-model api_key: my-fake-key - api_base: http://host.docker.internal:8090/ + api_base: https://exampleopenaiendpoint-production.up.railway.app/ general_settings: store_model_in_db: true diff --git a/tests/llm_translation/test_triton.py b/tests/llm_translation/test_triton.py index 7327be73524..2d1ca39e1dc 100644 --- a/tests/llm_translation/test_triton.py +++ b/tests/llm_translation/test_triton.py @@ -360,7 +360,7 @@ async def test_triton_embeddings(): litellm.set_verbose = True response = await litellm.aembedding( model="triton/my-triton-model", - api_base="http://127.0.0.1:8090/triton/embeddings", + api_base="https://exampleopenaiendpoint-production.up.railway.app/triton/embeddings", input=["good morning from litellm"], ) print(f"response: {response}") diff --git a/tests/load_tests/test_vertex_embeddings_load_test.py b/tests/load_tests/test_vertex_embeddings_load_test.py index 83060ec94bc..9beee710553 100644 --- a/tests/load_tests/test_vertex_embeddings_load_test.py +++ b/tests/load_tests/test_vertex_embeddings_load_test.py @@ -59,7 +59,7 @@ def load_vertex_ai_credentials(): async def create_async_vertex_embedding_task(): load_vertex_ai_credentials() - base_url = "http://127.0.0.1:8090/v1/projects/pathrise-convert-1606954137718/locations/us-central1/publishers/google/models/textembedding-gecko@001" + base_url = "https://exampleopenaiendpoint-production.up.railway.app/v1/projects/pathrise-convert-1606954137718/locations/us-central1/publishers/google/models/textembedding-gecko@001" embedding_args = { "model": "vertex_ai/textembedding-gecko", "input": "This is a test sentence for embedding.", diff --git a/tests/load_tests/test_vertex_load_tests.py b/tests/load_tests/test_vertex_load_tests.py index b967bb00ada..9130873b970 100644 --- a/tests/load_tests/test_vertex_load_tests.py +++ b/tests/load_tests/test_vertex_load_tests.py @@ -118,7 +118,7 @@ async def make_async_calls(message_type="text"): def create_async_task(message_type): - base_url = "http://127.0.0.1:8090/v1/projects/pathrise-convert-1606954137718/locations/us-central1/publishers/google/models/gemini-1.0-pro-vision-001" + base_url = "https://exampleopenaiendpoint-production.up.railway.app/v1/projects/pathrise-convert-1606954137718/locations/us-central1/publishers/google/models/gemini-1.0-pro-vision-001" if message_type == "text": messages = [{"role": "user", "content": "hi"}] diff --git a/tests/store_model_in_db_tests/test_callbacks_in_db.py b/tests/store_model_in_db_tests/test_callbacks_in_db.py index 4ddefaff4e1..4a851251a3e 100644 --- a/tests/store_model_in_db_tests/test_callbacks_in_db.py +++ b/tests/store_model_in_db_tests/test_callbacks_in_db.py @@ -21,7 +21,7 @@ from openai.types.chat import ChatCompletion load_dotenv() # used for testing -LANGFUSE_BASE_URL = "http://127.0.0.1:8090/slow" +LANGFUSE_BASE_URL = "https://exampleopenaiendpoint-production-c715.up.railway.app" async def config_update(session, routing_strategy=None):