mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-11 03:38:38 +00:00
* feat(proxy): serve a Codex-native model catalog with per-model service_tiers from /v1/models
GET /v1/models and /models answer Codex CLI's catalog fetch (the request
carrying its client_version query parameter) with Codex's own
{"models": [...]} shape: a model Codex knows keeps the metadata of its
bundled 0.159.3 catalog (vendored), any other model gets Codex's fallback
entry, and model_info.service_tiers becomes each entry's service tiers so
Codex offers them as slash commands that send service_tier upstream.
Without the parameter the OpenAI list shape is unchanged. The CLI's
litellm agents codex catalog shares the same builder.
* fix(proxy): offer a Codex service tier only when every deployment of the model lists it
* fix(codex-catalog): an invalid service_tiers value offers no tier for the model
* fix(codex-catalog): read service tiers off the deployments the key's team can route to
A tier is offered to Codex only when every deployment of the model name a
request from the key's team can route to lists it, so another team's
deployment of the name and a deployment an admin paused via model_info.blocked
no longer withhold or add tiers for requests that never reach them
The catalog's always-null fields are annotated NoneType so the module imports
under pydantic 2.12.0 on Python 3.14, the lowest pin the MCP resolve job
installs, which rejects a None annotation with a None default
* test(codex-catalog): drop the redundant module docstring and sort the imports
* test(integration): add the Codex catalog audit cells and the multi-worker convergence note
* test(integration): clean up every catalog test model and answer the refresh GET
* fix(proxy): keep tiered models under Codex's catalog cut and resolve alias tiers
Under Codex's 1 MiB catalog limit the entries offering a service tier are kept
ahead of those offering none, each group in model_list order, with every kept
entry at its listing position, so the model an operator configured tiers for
survives a wide key's long listing. A model_group_alias row reads its target's
deployments, so it carries the target's tiers and stock metadata under the
alias name.
* fix(proxy): pick Codex catalog metadata per team and skip entries too large for the cut
The upstream model that selects Codex's stock entry was read off the first deployment of a name
without checking the key's team, so a team whose requests route to a different deployment could be
handed another team's prompt, reasoning levels, and tiers. The upstream model and the tiers now come
from the same team-aware selection routing uses, and a caller with no team reads the deployments no
team owns
The byte cut kept a prefix of the tier-first order, so one entry larger than the whole limit emptied
the catalog. An entry too large for the bytes left is now passed over and the smaller ones after it
are still kept
---------
Co-authored-by: mateo-berri <277851410+mateo-berri@users.noreply.github.com>
137 lines
6.3 KiB
Python
137 lines
6.3 KiB
Python
import ast
|
|
import os
|
|
|
|
|
|
def get_function_names_from_file(file_path):
|
|
"""
|
|
Extracts all function names from a given Python file.
|
|
"""
|
|
with open(file_path, "r", encoding="utf-8") as file:
|
|
tree = ast.parse(file.read())
|
|
|
|
function_names = []
|
|
|
|
for node in tree.body:
|
|
if isinstance(node, (ast.FunctionDef, ast.AsyncFunctionDef)):
|
|
# Top-level functions
|
|
function_names.append(node.name)
|
|
elif isinstance(node, ast.ClassDef):
|
|
# Functions inside classes
|
|
for class_node in node.body:
|
|
if isinstance(class_node, (ast.FunctionDef, ast.AsyncFunctionDef)):
|
|
function_names.append(class_node.name)
|
|
|
|
return function_names
|
|
|
|
|
|
def get_all_functions_called_in_tests(base_dir):
|
|
"""
|
|
Returns a set of function names that are called in test functions
|
|
inside 'local_testing' and 'router_unit_test' directories,
|
|
specifically in files containing the word 'router'.
|
|
"""
|
|
called_functions = set()
|
|
test_dirs = ["local_testing", "router_unit_tests", "test_litellm", "unit"]
|
|
|
|
for test_dir in test_dirs:
|
|
dir_path = os.path.join(base_dir, test_dir)
|
|
if not os.path.exists(dir_path):
|
|
print(f"Warning: Directory {dir_path} does not exist.")
|
|
continue
|
|
|
|
print("dir_path: ", dir_path)
|
|
for root, _, files in os.walk(dir_path):
|
|
for file in files:
|
|
if file.endswith(".py") and "router" in file.lower():
|
|
print("file: ", file)
|
|
file_path = os.path.join(root, file)
|
|
with open(file_path, "r", encoding="utf-8") as f:
|
|
try:
|
|
tree = ast.parse(f.read())
|
|
except SyntaxError:
|
|
print(f"Warning: Syntax error in file {file_path}")
|
|
continue
|
|
if file == "test_router_validate_fallbacks.py":
|
|
print(f"tree: {tree}")
|
|
for node in ast.walk(tree):
|
|
if isinstance(node, ast.Call) and isinstance(node.func, ast.Name):
|
|
called_functions.add(node.func.id)
|
|
elif isinstance(node, ast.Call) and isinstance(node.func, ast.Attribute):
|
|
called_functions.add(node.func.attr)
|
|
|
|
return called_functions
|
|
|
|
|
|
def get_functions_from_router(file_path):
|
|
"""
|
|
Extracts all functions defined in router.py.
|
|
"""
|
|
return get_function_names_from_file(file_path)
|
|
|
|
|
|
ignored_function_names = [
|
|
"_acancel_batch",
|
|
"__init__",
|
|
"avector_store_create", # Tested via proxy vector_store_endpoints (files lack "router" in name)
|
|
"_override_vector_store_methods_for_router", # No-op placeholder, called during Router init
|
|
"_merge_tools_from_deployment", # Tested indirectly via _update_kwargs_with_deployment (test files lack "router" in name)
|
|
"_invalidate_access_groups_cache", # Tested indirectly via set_model_list, upsert_model etc. (test files lack "router" in name)
|
|
"has_buffered_provider_output", # Property, so its reads in test_router.py are never an ast.Call
|
|
"chunks", # Property on FallbackAwareAnthropicMessagesStream, so its reads in tests are never an ast.Call
|
|
"messages", # Property on FallbackAwareAnthropicMessagesStream, so its reads in tests are never an ast.Call
|
|
"model", # Property on FallbackAwareAnthropicMessagesStream, so its reads in tests are never an ast.Call
|
|
"_request_header", # Tested through Claude Code session routing in test_router.py
|
|
"_claude_code_session_router_cache_key", # Tested through Claude Code session routing in test_router.py
|
|
"_delete_claude_code_session_router_binding", # Tested through Redis cleanup failure in test_router.py
|
|
"_resolve_claude_code_session_router", # Tested through Claude Code session routing in test_router.py
|
|
"_get_claude_code_session_router_binding", # Tested through the two-worker session routing test in test_router.py
|
|
"_apply_updated_routing_strategy_args", # Tested via update_settings in test_lowest_latency.py (file lacks "router" in name)
|
|
"arm_routing_read_prefetch", # Tested in tests/unit/caching/test_request_redis_batch_pre_call.py (file lacks "router" in name)
|
|
"_configured_model_info", # Tested through get_configured_service_tiers in test_router.py
|
|
"_routable_deployments", # Tested through get_configured_service_tiers and get_routable_upstream_model in test_router.py
|
|
"_async_get_available_deployment", # Body of the `route {model}` phase wrapper, exercised through async_get_available_deployment in test_router.py
|
|
"_async_get_available_deployment_for_pass_through", # Same, through async_get_available_deployment_for_pass_through in test_router.py
|
|
"_embedding",
|
|
"_aembedding",
|
|
]
|
|
|
|
|
|
def main():
|
|
router_file = [
|
|
"./litellm/router.py",
|
|
"./litellm/router_utils/batch_utils.py",
|
|
"./litellm/router_utils/pattern_match_deployments.py",
|
|
]
|
|
# router_file = [
|
|
# "../../litellm/router.py",
|
|
# "../../litellm/router_utils/pattern_match_deployments.py",
|
|
# "../../litellm/router_utils/batch_utils.py",
|
|
# ] ## LOCAL TESTING
|
|
tests_dir = "./tests/" # Update this path if your tests directory is located elsewhere
|
|
# tests_dir = "../../tests/" # LOCAL TESTING
|
|
|
|
router_functions = []
|
|
for file in router_file:
|
|
router_functions.extend(get_functions_from_router(file))
|
|
print("router_functions: ", router_functions)
|
|
called_functions_in_tests = get_all_functions_called_in_tests(tests_dir)
|
|
untested_functions = [fn for fn in router_functions if fn not in called_functions_in_tests]
|
|
|
|
if untested_functions:
|
|
all_untested_functions = []
|
|
for func in untested_functions:
|
|
if func not in ignored_function_names:
|
|
all_untested_functions.append(func)
|
|
untested_perc = (len(all_untested_functions)) / len(router_functions)
|
|
print("untested_perc: ", untested_perc)
|
|
if untested_perc > 0:
|
|
print("The following functions in router.py are not tested:")
|
|
raise Exception(
|
|
f"{untested_perc * 100:.2f}% of functions in router.py are not tested: {all_untested_functions}"
|
|
)
|
|
else:
|
|
print("All functions in router.py are covered by tests.")
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|