mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-04 02:31:27 +00:00
fix(MLP-6153): reload router cache on /model/info miss to fix Terraform race condition
When the Terraform provider reads back a newly created model via /model/info, the request may land on a pod whose router cache hasn't yet loaded the new model (multi-replica eventual consistency). Instead of immediately returning 400 "not found", trigger a DB reload via proxy_config.add_deployment() and retry the lookup once before failing. Co-Authored-By: Claude Sonnet 4.6 (1M context) <noreply@anthropic.com>
This commit is contained in:
parent
934ecdca78
commit
e9d7728472
1 changed files with 12 additions and 0 deletions
|
|
@ -11182,6 +11182,18 @@ async def model_info_v1( # noqa: PLR0915
|
|||
if litellm_model_id is not None:
|
||||
# user is trying to get specific model from litellm router
|
||||
deployment_info = llm_router.get_deployment(model_id=litellm_model_id)
|
||||
if deployment_info is None and prisma_client is not None:
|
||||
# Cache miss — model may have just been created and the router cache not yet
|
||||
# updated on this pod (multi-replica eventual consistency). Reload from DB
|
||||
# and retry once before returning a 400.
|
||||
try:
|
||||
await proxy_config.add_deployment(
|
||||
prisma_client=prisma_client,
|
||||
proxy_logging_obj=proxy_logging_obj,
|
||||
)
|
||||
deployment_info = llm_router.get_deployment(model_id=litellm_model_id)
|
||||
except Exception:
|
||||
pass
|
||||
if deployment_info is None:
|
||||
raise HTTPException(
|
||||
status_code=400,
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue