mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-05 02:41:56 +00:00
ci(test): drop remaining Railway references; mirror endpoint shape exactly
Follow-up to e84d3a8 to remove the last `exampleopenaiendpoint` references.
Now no active code path in the repo depends on the Railway-hosted mock
endpoint (only the docstring at tests/mock_endpoints/openai_mock_server.py
mentions it, for migration context).
Mock server changes — mirror Railway's actual response shape so testing
intent is preserved 1:1 with the pre-migration behavior:
* Chat content: "\n\nHello there, how may I assist you today?" with usage
prompt=9 / completion=12 / total=21 (was a placeholder string before).
* Chat stream: 11 deltas spelling "Hello this is a test response from a
fixed OpenAI endpoint. ", no `role` field in delta, no finish_reason
chunk — matches what `openai-python` actually receives from Railway.
* Embeddings: 1536-dim vector tiled from Railway's 4-value pattern.
* Text completion / Anthropic messages / rerank: shapes match Railway.
* `model == "429"` returns HTTP 429 — used by the fake-azure-endpoint
fixture in otel_test_config.yaml to exercise retry/fallback paths.
* `/v1/chat/completions` returns Anthropic Messages shape (a known
Railway quirk that test_router_prompt_caching defensively relies on
via try/except — preserving it keeps that test's intent intact).
* New `/slow/...` path delays responses 2s. Replaces the separate slow
Railway deployment (`-c715`) that test_lowest_latency_routing_with_
timeouts used to differentiate slow vs fast endpoints.
YAML configs: pass_through_config, store_model_db_config,
test_pipeline_config, model_config, helm values, proxy_server.py
docstring example — all moved off Railway.
Test files: 22 additional files (router_unit_tests, pass_through,
proxy_unit_tests, llm_translation, load_tests, local_testing,
logging_callback, store_model_in_db, test_litellm/*) updated. Most use
the URL as a respx patch target or config string — purely cosmetic.
Direct-HTTP callers (deepseek, triton, rerank, lakera, completion,
lowest_latency, batches, prompt_caching) and assertion checks (rerank,
router_tag_routing) verified locally against the mock.
Out of scope: `litellm-{api,staging,production-*}.up.railway.app` refs
in proxy_server.py / ui_sso.py docstrings and commented-out code in
proxy_unit_tests — these are about LiteLLM's own deployments, not the
example endpoint.
This commit is contained in:
parent
84d3a807b9
commit
ba1452acb5
31 changed files with 175 additions and 132 deletions
|
|
@ -161,7 +161,7 @@ proxy_config:
|
|||
litellm_params:
|
||||
model: openai/fake
|
||||
api_key: fake-key
|
||||
api_base: https://exampleopenaiendpoint-production.up.railway.app/
|
||||
api_base: https://api.openai.com/v1/ # replace with your provider's URL
|
||||
general_settings:
|
||||
master_key: os.environ/PROXY_MASTER_KEY
|
||||
|
||||
|
|
|
|||
|
|
@ -3,7 +3,7 @@ model_list:
|
|||
litellm_params:
|
||||
model: openai/fake
|
||||
api_key: fake-key
|
||||
api_base: https://exampleopenaiendpoint-production.up.railway.app/
|
||||
api_base: http://host.docker.internal:8090/
|
||||
- model_name: claude-sonnet-4-5-20250929
|
||||
litellm_params:
|
||||
model: anthropic/claude-sonnet-4-5-20250929
|
||||
|
|
|
|||
|
|
@ -3,7 +3,7 @@ model_list:
|
|||
litellm_params:
|
||||
model: openai/my-fake-model
|
||||
api_key: my-fake-key
|
||||
api_base: https://exampleopenaiendpoint-production.up.railway.app/
|
||||
api_base: http://host.docker.internal:8090/
|
||||
|
||||
general_settings:
|
||||
store_model_in_db: true
|
||||
|
|
|
|||
|
|
@ -3,12 +3,12 @@ model_list:
|
|||
litellm_params:
|
||||
model: openai/gpt-3.5-turbo
|
||||
api_key: fake-key
|
||||
api_base: https://exampleopenaiendpoint-production.up.railway.app/
|
||||
api_base: http://host.docker.internal:8090/
|
||||
- model_name: fake-blocked-endpoint
|
||||
litellm_params:
|
||||
model: openai/gpt-3.5-turbo
|
||||
api_key: fake-key
|
||||
api_base: https://exampleopenaiendpoint-production.up.railway.app/
|
||||
api_base: http://host.docker.internal:8090/
|
||||
|
||||
guardrails:
|
||||
- guardrail_name: "strict-filter"
|
||||
|
|
|
|||
|
|
@ -2,9 +2,9 @@ model_list:
|
|||
- model_name: gpt-4o
|
||||
litellm_params:
|
||||
model: openai/gpt-4o
|
||||
api_base: https://exampleopenaiendpoint-production.up.railway.app/
|
||||
api_base: http://host.docker.internal:8090/
|
||||
- model_name: fake-anthropic-endpoint
|
||||
litellm_params:
|
||||
model: anthropic/fake
|
||||
api_base: https://exampleanthropicendpoint-production.up.railway.app/
|
||||
api_base: http://host.docker.internal:8090/
|
||||
|
||||
|
|
|
|||
|
|
@ -12254,7 +12254,7 @@ async def model_info_v1( # noqa: PLR0915
|
|||
{
|
||||
"model_name": "fake-openai-endpoint",
|
||||
"litellm_params": {
|
||||
"api_base": "https://exampleopenaiendpoint-production.up.railway.app/",
|
||||
"api_base": "http://host.docker.internal:8090/",
|
||||
"model": "openai/fake"
|
||||
},
|
||||
"model_info": {
|
||||
|
|
|
|||
|
|
@ -21,7 +21,7 @@ sys.path.insert(0, os.path.abspath("../.."))
|
|||
import litellm
|
||||
|
||||
|
||||
SERVER_URL = "https://exampleopenaiendpoint-production-0ee2.up.railway.app/v1"
|
||||
SERVER_URL = "http://127.0.0.1:8090/v1"
|
||||
|
||||
|
||||
@pytest.mark.asyncio()
|
||||
|
|
|
|||
|
|
@ -29,7 +29,7 @@ def test_deepseek_mock_completion(stream):
|
|||
response = completion(
|
||||
model="deepseek/deepseek-reasoner",
|
||||
messages=[{"role": "user", "content": "Hello, world!"}],
|
||||
api_base="https://exampleopenaiendpoint-production.up.railway.app/v1/chat/completions",
|
||||
api_base="http://127.0.0.1:8090/v1/chat/completions",
|
||||
stream=stream,
|
||||
mock_response="Hello! How can I help you today?",
|
||||
)
|
||||
|
|
|
|||
|
|
@ -175,7 +175,7 @@ async def test_rerank_custom_api_base(version):
|
|||
"documents": ["hello", "world"],
|
||||
}
|
||||
|
||||
api_base = "https://exampleopenaiendpoint-production.up.railway.app/"
|
||||
api_base = "http://127.0.0.1:8090/"
|
||||
if version == "v1":
|
||||
api_base += "v1/rerank"
|
||||
|
||||
|
|
@ -202,7 +202,7 @@ async def test_rerank_custom_api_base(version):
|
|||
print("url = ", _url)
|
||||
assert (
|
||||
_url
|
||||
== f"https://exampleopenaiendpoint-production.up.railway.app/{version}/rerank"
|
||||
== f"http://127.0.0.1:8090/{version}/rerank"
|
||||
)
|
||||
|
||||
request_data = json.loads(args_to_api)
|
||||
|
|
|
|||
|
|
@ -360,7 +360,7 @@ async def test_triton_embeddings():
|
|||
litellm.set_verbose = True
|
||||
response = await litellm.aembedding(
|
||||
model="triton/my-triton-model",
|
||||
api_base="https://exampleopenaiendpoint-production.up.railway.app/triton/embeddings",
|
||||
api_base="http://127.0.0.1:8090/triton/embeddings",
|
||||
input=["good morning from litellm"],
|
||||
)
|
||||
print(f"response: {response}")
|
||||
|
|
|
|||
|
|
@ -15,7 +15,7 @@ def test_datadog_logging_async():
|
|||
# litellm.set_verbose = True
|
||||
os.environ["DD_API_KEY"] = "anything"
|
||||
os.environ["_DATADOG_BASE_URL"] = (
|
||||
"https://exampleopenaiendpoint-production.up.railway.app"
|
||||
"http://127.0.0.1:8090"
|
||||
)
|
||||
|
||||
os.environ["DD_SITE"] = "us5.datadoghq.com"
|
||||
|
|
|
|||
|
|
@ -17,7 +17,7 @@ def test_langsmith_logging_async():
|
|||
os.environ["LANGSMITH_API_KEY"] = "lsv2_anything"
|
||||
os.environ["LANGSMITH_PROJECT"] = "pr-b"
|
||||
os.environ["LANGSMITH_BASE_URL"] = (
|
||||
"https://exampleopenaiendpoint-production.up.railway.app"
|
||||
"http://127.0.0.1:8090"
|
||||
)
|
||||
|
||||
percentage_diffs = []
|
||||
|
|
|
|||
|
|
@ -72,7 +72,7 @@ async def make_completion_request():
|
|||
return await litellm.acompletion(
|
||||
model="openai/gpt-4o",
|
||||
messages=[{"role": "user", "content": "Test message for memory usage"}],
|
||||
api_base="https://exampleopenaiendpoint-production.up.railway.app/",
|
||||
api_base="http://127.0.0.1:8090/",
|
||||
)
|
||||
|
||||
|
||||
|
|
@ -80,7 +80,7 @@ async def make_text_completion_request():
|
|||
return await litellm.atext_completion(
|
||||
model="openai/gpt-4o",
|
||||
prompt="Test message for memory usage",
|
||||
api_base="https://exampleopenaiendpoint-production.up.railway.app/",
|
||||
api_base="http://127.0.0.1:8090/",
|
||||
)
|
||||
|
||||
|
||||
|
|
@ -108,14 +108,14 @@ litellm_router = Router(
|
|||
"model_name": "text-gpt-4o",
|
||||
"litellm_params": {
|
||||
"model": "text-completion-openai/gpt-3.5-turbo-instruct-unlimited",
|
||||
"api_base": "https://exampleopenaiendpoint-production.up.railway.app/",
|
||||
"api_base": "http://127.0.0.1:8090/",
|
||||
},
|
||||
},
|
||||
{
|
||||
"model_name": "chat-gpt-4o",
|
||||
"litellm_params": {
|
||||
"model": "openai/gpt-4o",
|
||||
"api_base": "https://exampleopenaiendpoint-production.up.railway.app/",
|
||||
"api_base": "http://127.0.0.1:8090/",
|
||||
},
|
||||
},
|
||||
]
|
||||
|
|
@ -128,7 +128,7 @@ async def make_router_atext_completion_request():
|
|||
temperature=0.5,
|
||||
frequency_penalty=0.5,
|
||||
prompt="<|fim prefix|> Test message for memory usage<fim suffix> <|fim prefix|> Test message for memory usage<fim suffix>",
|
||||
api_base="https://exampleopenaiendpoint-production.up.railway.app/",
|
||||
api_base="http://127.0.0.1:8090/",
|
||||
max_tokens=500,
|
||||
)
|
||||
|
||||
|
|
@ -148,7 +148,7 @@ async def make_router_acompletion_request():
|
|||
return await litellm_router.acompletion(
|
||||
model="chat-gpt-4o",
|
||||
messages=[{"role": "user", "content": "Test message for memory usage"}],
|
||||
api_base="https://exampleopenaiendpoint-production.up.railway.app/",
|
||||
api_base="http://127.0.0.1:8090/",
|
||||
)
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -16,7 +16,7 @@ def test_otel_logging_async():
|
|||
try:
|
||||
os.environ["OTEL_EXPORTER"] = "otlp_http"
|
||||
os.environ["OTEL_ENDPOINT"] = (
|
||||
"https://exampleopenaiendpoint-production.up.railway.app/traces"
|
||||
"http://127.0.0.1:8090/traces"
|
||||
)
|
||||
os.environ["OTEL_HEADERS"] = "Authorization=K0BSwd"
|
||||
|
||||
|
|
|
|||
|
|
@ -59,7 +59,7 @@ def load_vertex_ai_credentials():
|
|||
|
||||
async def create_async_vertex_embedding_task():
|
||||
load_vertex_ai_credentials()
|
||||
base_url = "https://exampleopenaiendpoint-production.up.railway.app/v1/projects/pathrise-convert-1606954137718/locations/us-central1/publishers/google/models/textembedding-gecko@001"
|
||||
base_url = "http://127.0.0.1:8090/v1/projects/pathrise-convert-1606954137718/locations/us-central1/publishers/google/models/textembedding-gecko@001"
|
||||
embedding_args = {
|
||||
"model": "vertex_ai/textembedding-gecko",
|
||||
"input": "This is a test sentence for embedding.",
|
||||
|
|
|
|||
|
|
@ -118,7 +118,7 @@ async def make_async_calls(message_type="text"):
|
|||
|
||||
|
||||
def create_async_task(message_type):
|
||||
base_url = "https://exampleopenaiendpoint-production.up.railway.app/v1/projects/pathrise-convert-1606954137718/locations/us-central1/publishers/google/models/gemini-1.0-pro-vision-001"
|
||||
base_url = "http://127.0.0.1:8090/v1/projects/pathrise-convert-1606954137718/locations/us-central1/publishers/google/models/gemini-1.0-pro-vision-001"
|
||||
|
||||
if message_type == "text":
|
||||
messages = [{"role": "user", "content": "hi"}]
|
||||
|
|
|
|||
|
|
@ -1343,7 +1343,7 @@ def test_lm_studio_completion(monkeypatch):
|
|||
messages=[
|
||||
{"role": "user", "content": "What's the weather like in San Francisco?"}
|
||||
],
|
||||
api_base="https://exampleopenaiendpoint-production.up.railway.app/",
|
||||
api_base="http://127.0.0.1:8090/",
|
||||
)
|
||||
except litellm.AuthenticationError as e:
|
||||
pytest.fail(f"Error occurred: {e}")
|
||||
|
|
|
|||
|
|
@ -2756,7 +2756,7 @@ def model_item():
|
|||
"litellm_params": {
|
||||
"model": "openai/my-fake-model",
|
||||
"api_key": "my-fake-key",
|
||||
"api_base": "https://exampleopenaiendpoint-production.up.railway.app/",
|
||||
"api_base": "http://127.0.0.1:8090/",
|
||||
},
|
||||
"model_info": {},
|
||||
}
|
||||
|
|
|
|||
|
|
@ -158,7 +158,7 @@ async def test_moderations_on_embeddings():
|
|||
"litellm_params": {
|
||||
"model": "text-embedding-ada-002",
|
||||
"api_key": "any",
|
||||
"api_base": "https://exampleopenaiendpoint-production.up.railway.app/",
|
||||
"api_base": "http://127.0.0.1:8090/",
|
||||
},
|
||||
},
|
||||
]
|
||||
|
|
|
|||
|
|
@ -591,7 +591,7 @@ async def test_lowest_latency_routing_with_timeouts():
|
|||
"model_name": "azure-model",
|
||||
"litellm_params": {
|
||||
"model": "openai/slow-endpoint",
|
||||
"api_base": "https://exampleopenaiendpoint-production-c715.up.railway.app/", # If you are Krrish, this is OpenAI Endpoint3 on our Railway endpoint :)
|
||||
"api_base": "http://127.0.0.1:8090/slow/", # If you are Krrish, this is OpenAI Endpoint3 on our Railway endpoint :)
|
||||
"api_key": "fake-key",
|
||||
},
|
||||
"model_info": {"id": "slow-endpoint"},
|
||||
|
|
@ -600,7 +600,7 @@ async def test_lowest_latency_routing_with_timeouts():
|
|||
"model_name": "azure-model",
|
||||
"litellm_params": {
|
||||
"model": "openai/fast-endpoint",
|
||||
"api_base": "https://exampleopenaiendpoint-production.up.railway.app/",
|
||||
"api_base": "http://127.0.0.1:8090/",
|
||||
"api_key": "fake-key",
|
||||
},
|
||||
"model_info": {"id": "fast-endpoint"},
|
||||
|
|
@ -666,7 +666,7 @@ async def test_lowest_latency_routing_first_pick():
|
|||
"model_name": "azure-model",
|
||||
"litellm_params": {
|
||||
"model": "openai/fast-endpoint",
|
||||
"api_base": "https://exampleopenaiendpoint-production.up.railway.app/",
|
||||
"api_base": "http://127.0.0.1:8090/",
|
||||
"api_key": "fake-key",
|
||||
},
|
||||
"model_info": {"id": "fast-endpoint"},
|
||||
|
|
@ -675,7 +675,7 @@ async def test_lowest_latency_routing_first_pick():
|
|||
"model_name": "azure-model",
|
||||
"litellm_params": {
|
||||
"model": "openai/fast-endpoint-2",
|
||||
"api_base": "https://exampleopenaiendpoint-production.up.railway.app/",
|
||||
"api_base": "http://127.0.0.1:8090/",
|
||||
"api_key": "fake-key",
|
||||
},
|
||||
"model_info": {"id": "fast-endpoint-2"},
|
||||
|
|
@ -684,7 +684,7 @@ async def test_lowest_latency_routing_first_pick():
|
|||
"model_name": "azure-model",
|
||||
"litellm_params": {
|
||||
"model": "openai/fast-endpoint-2",
|
||||
"api_base": "https://exampleopenaiendpoint-production.up.railway.app/",
|
||||
"api_base": "http://127.0.0.1:8090/",
|
||||
"api_key": "fake-key",
|
||||
},
|
||||
"model_info": {"id": "fast-endpoint-3"},
|
||||
|
|
@ -693,7 +693,7 @@ async def test_lowest_latency_routing_first_pick():
|
|||
"model_name": "azure-model",
|
||||
"litellm_params": {
|
||||
"model": "openai/fast-endpoint-2",
|
||||
"api_base": "https://exampleopenaiendpoint-production.up.railway.app/",
|
||||
"api_base": "http://127.0.0.1:8090/",
|
||||
"api_key": "fake-key",
|
||||
},
|
||||
"model_info": {"id": "fast-endpoint-4"},
|
||||
|
|
|
|||
|
|
@ -246,7 +246,7 @@ router = Router(
|
|||
"model_name": "fake-model",
|
||||
"litellm_params": {
|
||||
"model": "openai/fake",
|
||||
"api_base": "https://exampleopenaiendpoint-production.up.railway.app/",
|
||||
"api_base": "http://127.0.0.1:8090/",
|
||||
"api_key": "sk-12345",
|
||||
},
|
||||
}
|
||||
|
|
|
|||
|
|
@ -61,7 +61,7 @@ mock_response_data = {
|
|||
"completion_tokens": 12,
|
||||
"request_tags": [],
|
||||
"end_user": "",
|
||||
"api_base": "https://exampleopenaiendpoint-production.up.railway.app",
|
||||
"api_base": "http://127.0.0.1:8090",
|
||||
"model_group": "fake-openai-endpoint",
|
||||
"model_id": "b68d56d76b0c24ac9462ab69541e90886342508212210116e300441155f37865",
|
||||
"requester_ip_address": "127.0.0.1",
|
||||
|
|
@ -100,7 +100,7 @@ mock_response_data = {
|
|||
"hidden_params": {
|
||||
"model_id": "b68d56d76b0c24ac9462ab69541e90886342508212210116e300441155f37865",
|
||||
"cache_key": None,
|
||||
"api_base": "https://exampleopenaiendpoint-production.up.railway.app/",
|
||||
"api_base": "http://127.0.0.1:8090/",
|
||||
"response_cost": 3.7500000000000003e-05,
|
||||
"additional_headers": {},
|
||||
"litellm_overhead_time_ms": 2.126,
|
||||
|
|
|
|||
|
|
@ -9,6 +9,22 @@ Run standalone:
|
|||
The server is intentionally implemented with the Python standard library only
|
||||
(no FastAPI / httpx / pydantic) so CI jobs can start it before installing test
|
||||
dependencies.
|
||||
|
||||
Response shapes are matched to the historical Railway endpoint to preserve
|
||||
test intent:
|
||||
* /chat/completions -> OpenAI chat.completion (non-stream) or SSE
|
||||
stream of chat.completion.chunk events.
|
||||
Content: "Hello this is a test response from
|
||||
a fixed OpenAI endpoint. " (stream) or
|
||||
"\\n\\nHello there, how may I assist you
|
||||
today?" (non-stream).
|
||||
* /completions -> OpenAI text_completion.
|
||||
* /embeddings -> OpenAI embedding list.
|
||||
* /rerank, /v2/rerank -> Cohere-shape rerank result.
|
||||
* /v1/messages -> Anthropic Messages.
|
||||
|
||||
Special model handling matches Railway:
|
||||
* model == "429" returns HTTP 429 (used by fake-azure-endpoint fixtures).
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
|
@ -22,81 +38,86 @@ from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
|
|||
logger = logging.getLogger("openai_mock_server")
|
||||
|
||||
_EMBEDDING_DIM = 1536
|
||||
_CREATED_TS = 1677652288
|
||||
_SYSTEM_FINGERPRINT = "fp_44709d6fcb"
|
||||
_CHAT_CONTENT = "\n\nHello there, how may I assist you today?"
|
||||
_STREAM_TOKENS = [
|
||||
"Hello ",
|
||||
"this ",
|
||||
"is ",
|
||||
"a ",
|
||||
"test ",
|
||||
"response ",
|
||||
"from ",
|
||||
"a ",
|
||||
"fixed ",
|
||||
"OpenAI ",
|
||||
"endpoint. ",
|
||||
]
|
||||
|
||||
|
||||
def _now() -> int:
|
||||
return int(time.time())
|
||||
|
||||
|
||||
def _chat_completion_body(
|
||||
model: str, content: str = "Hi! This is a mock response."
|
||||
) -> dict:
|
||||
def _chat_completion_body(model: str) -> dict:
|
||||
return {
|
||||
"id": "chatcmpl-mock-0001",
|
||||
"id": "chatcmpl-c055ccfa83b84490af310d4bb5552422",
|
||||
"object": "chat.completion",
|
||||
"created": _now(),
|
||||
"created": _CREATED_TS,
|
||||
"model": model,
|
||||
"system_fingerprint": "fp_mock",
|
||||
"system_fingerprint": _SYSTEM_FINGERPRINT,
|
||||
"choices": [
|
||||
{
|
||||
"index": 0,
|
||||
"message": {"role": "assistant", "content": content},
|
||||
"message": {"role": "assistant", "content": _CHAT_CONTENT},
|
||||
"logprobs": None,
|
||||
"finish_reason": "stop",
|
||||
}
|
||||
],
|
||||
"usage": {
|
||||
"prompt_tokens": 10,
|
||||
"completion_tokens": 20,
|
||||
"total_tokens": 30,
|
||||
"prompt_tokens_details": {"cached_tokens": 0, "audio_tokens": 0},
|
||||
"completion_tokens_details": {
|
||||
"reasoning_tokens": 0,
|
||||
"audio_tokens": 0,
|
||||
"accepted_prediction_tokens": 0,
|
||||
"rejected_prediction_tokens": 0,
|
||||
},
|
||||
},
|
||||
"usage": {"prompt_tokens": 9, "completion_tokens": 12, "total_tokens": 21},
|
||||
}
|
||||
|
||||
|
||||
def _text_completion_body(model: str) -> dict:
|
||||
return {
|
||||
"id": "cmpl-mock-0001",
|
||||
"id": "cmpl-9B2ycsf0odECdLmrVzm2y8Q12csjW",
|
||||
"object": "text_completion",
|
||||
"created": _now(),
|
||||
"created": _CREATED_TS,
|
||||
"model": model,
|
||||
"system_fingerprint": None,
|
||||
"choices": [
|
||||
{
|
||||
"text": "Mock completion response.",
|
||||
"text": "\n\nA test request, how intriguing\n"
|
||||
"An invitation for knowledge bringing\nWith words",
|
||||
"index": 0,
|
||||
"logprobs": None,
|
||||
"finish_reason": "stop",
|
||||
"finish_reason": "length",
|
||||
}
|
||||
],
|
||||
"usage": {"prompt_tokens": 5, "completion_tokens": 5, "total_tokens": 10},
|
||||
"usage": {"prompt_tokens": 10, "completion_tokens": 16, "total_tokens": 26},
|
||||
}
|
||||
|
||||
|
||||
def _embedding_body(model: str, input_count: int) -> dict:
|
||||
# Match Railway: a constant 1536-d vector tiled from a short pattern.
|
||||
pattern = [
|
||||
-0.006929283495992422,
|
||||
-0.005336422007530928,
|
||||
-4.547132266452536e-05,
|
||||
-0.024047505110502243,
|
||||
]
|
||||
vector = (pattern * ((_EMBEDDING_DIM // len(pattern)) + 1))[:_EMBEDDING_DIM]
|
||||
return {
|
||||
"object": "list",
|
||||
"model": model,
|
||||
"data": [
|
||||
{
|
||||
"object": "embedding",
|
||||
"index": i,
|
||||
"embedding": [0.0] * _EMBEDDING_DIM,
|
||||
}
|
||||
{"object": "embedding", "index": i, "embedding": vector}
|
||||
for i in range(max(input_count, 1))
|
||||
],
|
||||
"usage": {"prompt_tokens": input_count or 1, "total_tokens": input_count or 1},
|
||||
}
|
||||
|
||||
|
||||
def _rerank_body(query_id: str = "rerank-mock-0001") -> dict:
|
||||
def _rerank_body() -> dict:
|
||||
return {
|
||||
"id": query_id,
|
||||
"id": "rerank-mock-0001",
|
||||
"results": [{"index": 0, "relevance_score": 0.99}],
|
||||
"meta": {
|
||||
"api_version": {"version": "1"},
|
||||
|
|
@ -107,52 +128,29 @@ def _rerank_body(query_id: str = "rerank-mock-0001") -> dict:
|
|||
|
||||
def _anthropic_message_body(model: str) -> dict:
|
||||
return {
|
||||
"id": "msg_mock_0001",
|
||||
"id": "msg_01G7MsdWPT2JZMUuc1UXRavn",
|
||||
"type": "message",
|
||||
"role": "assistant",
|
||||
"model": model,
|
||||
"content": [{"type": "text", "text": "Mock Anthropic message."}],
|
||||
"content": [{"type": "text", "text": _CHAT_CONTENT}],
|
||||
"stop_reason": "end_turn",
|
||||
"stop_sequence": None,
|
||||
"usage": {"input_tokens": 10, "output_tokens": 20},
|
||||
"usage": {"input_tokens": 10, "output_tokens": 20, "total_tokens": 30},
|
||||
}
|
||||
|
||||
|
||||
def _chat_stream_chunks(model: str) -> list[dict]:
|
||||
base = {
|
||||
"id": "chatcmpl-mock-0001",
|
||||
"id": "chatcmpl-b0e9cfbf26d148928e8501842b2af4de",
|
||||
"object": "chat.completion.chunk",
|
||||
"created": _now(),
|
||||
"created": _CREATED_TS,
|
||||
"model": model,
|
||||
}
|
||||
# Match Railway: deltas carry only `content` (no role, no finish_reason),
|
||||
# and the stream ends after the last token without a [DONE] sentinel.
|
||||
return [
|
||||
{
|
||||
**base,
|
||||
"choices": [
|
||||
{
|
||||
"index": 0,
|
||||
"delta": {"role": "assistant", "content": "Hi"},
|
||||
"finish_reason": None,
|
||||
}
|
||||
],
|
||||
},
|
||||
{
|
||||
**base,
|
||||
"choices": [
|
||||
{
|
||||
"index": 0,
|
||||
"delta": {"content": "! This is a mock"},
|
||||
"finish_reason": None,
|
||||
}
|
||||
],
|
||||
},
|
||||
{
|
||||
**base,
|
||||
"choices": [
|
||||
{"index": 0, "delta": {"content": " response."}, "finish_reason": None}
|
||||
],
|
||||
},
|
||||
{**base, "choices": [{"index": 0, "delta": {}, "finish_reason": "stop"}]},
|
||||
{**base, "choices": [{"index": 0, "delta": {"content": token}}]}
|
||||
for token in _STREAM_TOKENS
|
||||
]
|
||||
|
||||
|
||||
|
|
@ -189,18 +187,46 @@ class MockOpenAIHandler(BaseHTTPRequestHandler):
|
|||
|
||||
def _send_sse(self, chunks: list[dict]) -> None:
|
||||
self.send_response(200)
|
||||
self.send_header("Content-Type", "text/event-stream")
|
||||
self.send_header("Content-Type", "text/event-stream; charset=utf-8")
|
||||
self.send_header("Cache-Control", "no-cache")
|
||||
self.send_header("Connection", "keep-alive")
|
||||
self.send_header("Connection", "close")
|
||||
self.end_headers()
|
||||
for chunk in chunks:
|
||||
self.wfile.write(f"data: {json.dumps(chunk)}\n\n".encode("utf-8"))
|
||||
self.wfile.flush()
|
||||
# OpenAI clients (httpx / openai-python) terminate on a `data: [DONE]`
|
||||
# sentinel; Railway omitted it and relied on connection close. Sending
|
||||
# it explicitly is safe (LiteLLM tolerates either) and avoids the
|
||||
# 30s-keep-alive hang seen with the raw stdlib server.
|
||||
self.wfile.write(b"data: [DONE]\n\n")
|
||||
self.wfile.flush()
|
||||
|
||||
def _route(self) -> str:
|
||||
# Path prefixes that simulate latency before responding. Used by tests
|
||||
# like test_lowest_latency_routing_with_timeouts that historically pointed
|
||||
# the "slow deployment" at a separate Railway URL which was naturally slow.
|
||||
_DELAY_PREFIXES = {
|
||||
"/slow": 2.0,
|
||||
}
|
||||
|
||||
def _maybe_delay(self) -> None:
|
||||
path = self.path.split("?", 1)[0]
|
||||
for prefix, seconds in self._DELAY_PREFIXES.items():
|
||||
if path.startswith(prefix + "/") or path == prefix:
|
||||
time.sleep(seconds)
|
||||
return
|
||||
|
||||
def _stripped_path(self) -> str:
|
||||
"""Path with `/slow` (delay) prefix removed but `/v1`, `/v2` preserved."""
|
||||
path = self.path.split("?", 1)[0].rstrip("/") or "/"
|
||||
for prefix in self._DELAY_PREFIXES:
|
||||
if path.startswith(prefix + "/"):
|
||||
return path[len(prefix) :]
|
||||
if path == prefix:
|
||||
return "/"
|
||||
return path
|
||||
|
||||
def _route(self) -> str:
|
||||
path = self._stripped_path()
|
||||
for prefix in ("/v1", "/v2"):
|
||||
if path.startswith(prefix + "/"):
|
||||
return path[len(prefix) :]
|
||||
|
|
@ -219,13 +245,13 @@ class MockOpenAIHandler(BaseHTTPRequestHandler):
|
|||
{
|
||||
"id": "gpt-3.5-turbo",
|
||||
"object": "model",
|
||||
"created": _now(),
|
||||
"created": _CREATED_TS,
|
||||
"owned_by": "mock",
|
||||
},
|
||||
{
|
||||
"id": "gpt-4",
|
||||
"object": "model",
|
||||
"created": _now(),
|
||||
"created": _CREATED_TS,
|
||||
"owned_by": "mock",
|
||||
},
|
||||
],
|
||||
|
|
@ -236,10 +262,27 @@ class MockOpenAIHandler(BaseHTTPRequestHandler):
|
|||
|
||||
def do_POST(self) -> None:
|
||||
body = self._read_body()
|
||||
self._maybe_delay()
|
||||
raw_path = self._stripped_path()
|
||||
route = self._route()
|
||||
model = body.get("model") or "gpt-3.5-turbo"
|
||||
|
||||
# Railway convention: `model == "429"` triggers a rate-limit response.
|
||||
# Used by `fake-azure-endpoint` fixtures to exercise retry/fallback paths.
|
||||
if model == "429" and route in ("/chat/completions", "/completions"):
|
||||
self._send_json(429, {"detail": "Too many requests"})
|
||||
return
|
||||
|
||||
if route == "/chat/completions":
|
||||
# Railway quirk we have to preserve: POST /v1/chat/completions
|
||||
# returns an Anthropic-shape Messages response (vs. the OpenAI shape
|
||||
# served at /chat/completions). LiteLLM's OpenAI parser raises
|
||||
# InternalServerError on that response, and a few tests
|
||||
# (e.g. test_router_prompt_caching) defensively rely on that path
|
||||
# via try/except. Matching the quirk keeps their intent intact.
|
||||
if raw_path.startswith("/v1/"):
|
||||
self._send_json(200, _anthropic_message_body(model))
|
||||
return
|
||||
if body.get("stream"):
|
||||
self._send_sse(_chat_stream_chunks(model))
|
||||
else:
|
||||
|
|
|
|||
|
|
@ -290,7 +290,7 @@ async def test_pass_through_request_logging_failure(
|
|||
)
|
||||
response = await pass_through_request(
|
||||
request=request,
|
||||
target="https://exampleopenaiendpoint-production.up.railway.app/v1/messages",
|
||||
target="http://127.0.0.1:8090/v1/messages",
|
||||
custom_headers={},
|
||||
user_api_key_dict=mock_user_api_key_dict,
|
||||
)
|
||||
|
|
@ -357,7 +357,7 @@ async def test_pass_through_request_logging_failure_with_stream(
|
|||
)
|
||||
response = await pass_through_request(
|
||||
request=request,
|
||||
target="https://exampleopenaiendpoint-production.up.railway.app/v1/messages",
|
||||
target="http://127.0.0.1:8090/v1/messages",
|
||||
custom_headers={},
|
||||
user_api_key_dict=mock_user_api_key_dict,
|
||||
)
|
||||
|
|
|
|||
|
|
@ -87,7 +87,7 @@ router = Router(
|
|||
"model_name": "fake-model",
|
||||
"litellm_params": {
|
||||
"model": "openai/fake",
|
||||
"api_base": "https://exampleopenaiendpoint-production.up.railway.app/",
|
||||
"api_base": "http://127.0.0.1:8090/",
|
||||
"api_key": "sk-12345",
|
||||
},
|
||||
}
|
||||
|
|
|
|||
|
|
@ -127,7 +127,7 @@ async def test_vLLM_token_counting():
|
|||
"model_name": "special-alias",
|
||||
"litellm_params": {
|
||||
"model": "openai/wolfram/miquliz-120b-v2.0",
|
||||
"api_base": "https://exampleopenaiendpoint-production.up.railway.app/",
|
||||
"api_base": "http://127.0.0.1:8090/",
|
||||
},
|
||||
}
|
||||
]
|
||||
|
|
|
|||
|
|
@ -125,7 +125,7 @@ async def test_router_prompt_caching_same_cacheable_prefix_routes_to_same_deploy
|
|||
"model_name": "test-model",
|
||||
"litellm_params": {
|
||||
"model": "gpt-5-mini",
|
||||
"api_base": "https://exampleopenaiendpoint-production-0ee2.up.railway.app/v1",
|
||||
"api_base": "http://127.0.0.1:8090/v1",
|
||||
"api_key": f"test-key-{i}",
|
||||
},
|
||||
"model_info": {"id": f"deployment-{i}"},
|
||||
|
|
|
|||
|
|
@ -21,7 +21,7 @@ from openai.types.chat import ChatCompletion
|
|||
load_dotenv()
|
||||
|
||||
# used for testing
|
||||
LANGFUSE_BASE_URL = "https://exampleopenaiendpoint-production-c715.up.railway.app"
|
||||
LANGFUSE_BASE_URL = "http://127.0.0.1:8090/slow"
|
||||
|
||||
|
||||
async def config_update(session, routing_strategy=None):
|
||||
|
|
|
|||
|
|
@ -36,7 +36,7 @@ def _mock_azure_client(
|
|||
client = AsyncAzureOpenAI(
|
||||
api_key="test-key",
|
||||
api_version="2024-10-21",
|
||||
azure_endpoint="https://exampleopenaiendpoint-production.up.railway.app",
|
||||
azure_endpoint="http://127.0.0.1:8090",
|
||||
)
|
||||
client.fine_tuning.jobs.create = AsyncMock(
|
||||
return_value=(
|
||||
|
|
@ -69,7 +69,7 @@ async def test_azure_acreate_fine_tuning_job_request_and_output_match_expected_j
|
|||
model="gpt-35-turbo-1106",
|
||||
training_file="file-5e4b20ecbd724182b9964f3cd2ab7212",
|
||||
custom_llm_provider="azure",
|
||||
api_base="https://exampleopenaiendpoint-production.up.railway.app",
|
||||
api_base="http://127.0.0.1:8090",
|
||||
api_key="test-key",
|
||||
api_version="2024-10-21",
|
||||
)
|
||||
|
|
@ -100,7 +100,7 @@ async def test_azure_alist_fine_tuning_jobs_request_matches_expected_json():
|
|||
after=expected_request["after"],
|
||||
limit=expected_request["limit"],
|
||||
custom_llm_provider="azure",
|
||||
api_base="https://exampleopenaiendpoint-production.up.railway.app",
|
||||
api_base="http://127.0.0.1:8090",
|
||||
api_key="test-key",
|
||||
api_version="2024-10-21",
|
||||
)
|
||||
|
|
@ -124,7 +124,7 @@ async def test_azure_acancel_fine_tuning_job_request_and_output_match_expected_j
|
|||
response = await litellm.acancel_fine_tuning_job(
|
||||
fine_tuning_job_id=expected_request["fine_tuning_job_id"],
|
||||
custom_llm_provider="azure",
|
||||
api_base="https://exampleopenaiendpoint-production.up.railway.app",
|
||||
api_base="http://127.0.0.1:8090",
|
||||
api_key="test-key",
|
||||
api_version="2024-10-21",
|
||||
)
|
||||
|
|
|
|||
|
|
@ -26,7 +26,7 @@ class MockPrismaClient:
|
|||
"environment_variables": {
|
||||
"LANGFUSE_PUBLIC_KEY": "any-public-key",
|
||||
"LANGFUSE_SECRET_KEY": "any-secret-key",
|
||||
"LANGFUSE_HOST": "https://exampleopenaiendpoint-production-c715.up.railway.app",
|
||||
"LANGFUSE_HOST": "http://127.0.0.1:8090/slow",
|
||||
},
|
||||
}
|
||||
|
||||
|
|
|
|||
|
|
@ -39,7 +39,7 @@ async def test_router_free_paid_tier():
|
|||
"model_name": "gpt-4",
|
||||
"litellm_params": {
|
||||
"model": "gpt-4o",
|
||||
"api_base": "https://exampleopenaiendpoint-production.up.railway.app/",
|
||||
"api_base": "http://127.0.0.1:8090/",
|
||||
"tags": ["free"],
|
||||
},
|
||||
"model_info": {"id": "very-cheap-model"},
|
||||
|
|
@ -48,7 +48,7 @@ async def test_router_free_paid_tier():
|
|||
"model_name": "gpt-4",
|
||||
"litellm_params": {
|
||||
"model": "gpt-4o-mini",
|
||||
"api_base": "https://exampleopenaiendpoint-production.up.railway.app/",
|
||||
"api_base": "http://127.0.0.1:8090/",
|
||||
"tags": ["paid"],
|
||||
},
|
||||
"model_info": {"id": "very-expensive-model"},
|
||||
|
|
@ -102,7 +102,7 @@ async def test_router_free_paid_tier_embeddings():
|
|||
"model_name": "gpt-4",
|
||||
"litellm_params": {
|
||||
"model": "gpt-4o",
|
||||
"api_base": "https://exampleopenaiendpoint-production.up.railway.app/",
|
||||
"api_base": "http://127.0.0.1:8090/",
|
||||
"tags": ["free"],
|
||||
"mock_response": ["1", "2", "3"],
|
||||
},
|
||||
|
|
@ -112,7 +112,7 @@ async def test_router_free_paid_tier_embeddings():
|
|||
"model_name": "gpt-4",
|
||||
"litellm_params": {
|
||||
"model": "gpt-4o-mini",
|
||||
"api_base": "https://exampleopenaiendpoint-production.up.railway.app/",
|
||||
"api_base": "http://127.0.0.1:8090/",
|
||||
"tags": ["paid"],
|
||||
"mock_response": ["1", "2", "3"],
|
||||
},
|
||||
|
|
@ -122,7 +122,7 @@ async def test_router_free_paid_tier_embeddings():
|
|||
"model_name": "gpt-4",
|
||||
"litellm_params": {
|
||||
"model": "gpt-4o-mini",
|
||||
"api_base": "https://exampleopenaiendpoint-production.up.railway.app/",
|
||||
"api_base": "http://127.0.0.1:8090/",
|
||||
"tags": ["default"],
|
||||
"mock_response": ["1", "2", "3"],
|
||||
},
|
||||
|
|
@ -178,7 +178,7 @@ async def test_default_tagged_deployments():
|
|||
"model_name": "gpt-4",
|
||||
"litellm_params": {
|
||||
"model": "gpt-4o",
|
||||
"api_base": "https://exampleopenaiendpoint-production.up.railway.app/",
|
||||
"api_base": "http://127.0.0.1:8090/",
|
||||
"tags": ["default"],
|
||||
},
|
||||
"model_info": {"id": "default-model"},
|
||||
|
|
@ -187,7 +187,7 @@ async def test_default_tagged_deployments():
|
|||
"model_name": "gpt-4",
|
||||
"litellm_params": {
|
||||
"model": "gpt-4o",
|
||||
"api_base": "https://exampleopenaiendpoint-production.up.railway.app/",
|
||||
"api_base": "http://127.0.0.1:8090/",
|
||||
},
|
||||
"model_info": {"id": "default-model-2"},
|
||||
},
|
||||
|
|
@ -195,7 +195,7 @@ async def test_default_tagged_deployments():
|
|||
"model_name": "gpt-4",
|
||||
"litellm_params": {
|
||||
"model": "gpt-4o-mini",
|
||||
"api_base": "https://exampleopenaiendpoint-production.up.railway.app/",
|
||||
"api_base": "http://127.0.0.1:8090/",
|
||||
"tags": ["teamA"],
|
||||
},
|
||||
"model_info": {"id": "very-expensive-model"},
|
||||
|
|
@ -268,7 +268,7 @@ async def test_error_from_tag_routing():
|
|||
"model_name": "gpt-4",
|
||||
"litellm_params": {
|
||||
"model": "gpt-4o",
|
||||
"api_base": "https://exampleopenaiendpoint-production.up.railway.app/",
|
||||
"api_base": "http://127.0.0.1:8090/",
|
||||
},
|
||||
"model_info": {"id": "default-model"},
|
||||
},
|
||||
|
|
@ -276,7 +276,7 @@ async def test_error_from_tag_routing():
|
|||
"model_name": "gpt-4",
|
||||
"litellm_params": {
|
||||
"model": "gpt-4o",
|
||||
"api_base": "https://exampleopenaiendpoint-production.up.railway.app/",
|
||||
"api_base": "http://127.0.0.1:8090/",
|
||||
},
|
||||
"model_info": {"id": "default-model-2"},
|
||||
},
|
||||
|
|
@ -284,7 +284,7 @@ async def test_error_from_tag_routing():
|
|||
"model_name": "gpt-4",
|
||||
"litellm_params": {
|
||||
"model": "gpt-4o-mini",
|
||||
"api_base": "https://exampleopenaiendpoint-production.up.railway.app/",
|
||||
"api_base": "http://127.0.0.1:8090/",
|
||||
"tags": ["teamA"],
|
||||
},
|
||||
"model_info": {"id": "very-expensive-model"},
|
||||
|
|
@ -386,7 +386,7 @@ async def test_router_free_paid_tier_with_responses_api():
|
|||
"model_name": "gpt-4",
|
||||
"litellm_params": {
|
||||
"model": "gpt-4o",
|
||||
"api_base": "https://exampleopenaiendpoint-production.up.railway.app/",
|
||||
"api_base": "http://127.0.0.1:8090/",
|
||||
"tags": ["free"],
|
||||
},
|
||||
"model_info": {"id": "very-cheap-model"},
|
||||
|
|
@ -395,7 +395,7 @@ async def test_router_free_paid_tier_with_responses_api():
|
|||
"model_name": "gpt-4",
|
||||
"litellm_params": {
|
||||
"model": "gpt-4o-mini",
|
||||
"api_base": "https://exampleopenaiendpoint-production.up.railway.app/",
|
||||
"api_base": "http://127.0.0.1:8090/",
|
||||
"tags": ["paid"],
|
||||
},
|
||||
"model_info": {"id": "very-expensive-model"},
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue