mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-02 02:11:58 +00:00
The two SpendLogs index migrations shipped in v1.103.0 each break one table shape: the plain CREATE INDEX holds a SHARE lock on a large unpartitioned table and the CONCURRENTLY one fails with 0A000 on a partitioned parent. Both files are now inert and the indexes are built by a table-driven, shape-aware step after migrate deploy: CONCURRENTLY on a plain table, ON ONLY the parent plus per-partition CONCURRENTLY and ATTACH PARTITION on a partitioned one. The migration job builds them synchronously and exits non-zero on failure; a serving proxy that ran migrate deploy itself builds them in the background off the readiness path. A valid index of the same definition under another name is renamed and reused, an invalid one is rebuilt, and extra copies are reported with their DROP INDEX statement instead of being dropped. The migration checker rejects any CREATE INDEX on LiteLLM_SpendLogs or LiteLLM_ErrorLogs in future migrations Co-authored-by: yassin <yassin@berri.ai> Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
73 lines
3 KiB
Python
73 lines
3 KiB
Python
"""Entrypoint for the migrations Job container.
|
|
|
|
Runs `prisma migrate deploy` against the LiteLLM writer database using the
|
|
recovery logic in `litellm_proxy_extras.ProxyExtrasDBManager.setup_database`
|
|
(P3005 baseline + P3009/P3018 idempotent-error handling, retries, etc.), then
|
|
builds the request-log indexes the migrations leave out
|
|
(`litellm_proxy_extras.request_log_indexes`), waiting for them. The job exits
|
|
non-zero when an index could not be built so that it is rerun. A serving proxy
|
|
that runs the migrations itself builds the same indexes in the background once
|
|
it serves.
|
|
|
|
Env vars:
|
|
DATABASE_URL required unless it can be assembled at
|
|
startup from the discrete DATABASE_* vars
|
|
(password auth) or minted from an IAM token
|
|
(`IAM_TOKEN_DB_AUTH=true`)
|
|
DIRECT_URL optional — used by `migrate diff` when the
|
|
primary URL is a pooler (e.g. Neon -pooler)
|
|
USE_V2_MIGRATION_RESOLVER "false" → fall back to the v1 resolver
|
|
(legacy diff-and-force recovery). Defaults
|
|
to "true": the v2 resolver avoids the schema
|
|
thrashing seen during rolling deploys when
|
|
two LiteLLM versions contend for the same DB.
|
|
USE_PRISMA_DB_PUSH "true" → use `prisma db push` instead of
|
|
`migrate deploy`. Default false.
|
|
"""
|
|
|
|
import os
|
|
import sys
|
|
|
|
from litellm_proxy_extras._logging import logger
|
|
from litellm_proxy_extras.utils import ProxyExtrasDBManager, str_to_bool
|
|
|
|
from litellm.proxy.db.db_url_settings import DatabaseURLSettings
|
|
|
|
|
|
def main() -> int:
|
|
# Assemble DATABASE_URL from the discrete DATABASE_* env vars, matching
|
|
# the gateway/backend startup path (IAM mint or password auth). Leaves an
|
|
# operator-pinned DATABASE_URL untouched.
|
|
DatabaseURLSettings.from_env().apply_to_env()
|
|
|
|
if not os.getenv("DATABASE_URL"):
|
|
logger.error(
|
|
"DATABASE_URL is not set and could not be assembled from the "
|
|
"DATABASE_* env vars — cannot run migrations."
|
|
)
|
|
return 1
|
|
|
|
# v2 is the safer default for componentized deploys: it skips the
|
|
# diff-and-force recovery from v1 that caused schema thrashing during
|
|
# rolling deploys. Set USE_V2_MIGRATION_RESOLVER=false to opt back into v1.
|
|
use_v2 = str_to_bool(os.getenv("USE_V2_MIGRATION_RESOLVER", "true"))
|
|
use_db_push = str_to_bool(os.getenv("USE_PRISMA_DB_PUSH"))
|
|
|
|
logger.info(
|
|
"Starting prisma migration job (use_migrate=%s, use_v2_resolver=%s)",
|
|
not use_db_push,
|
|
use_v2,
|
|
)
|
|
ok = ProxyExtrasDBManager.run_migration_job(
|
|
use_migrate=not use_db_push,
|
|
use_v2_resolver=use_v2,
|
|
)
|
|
if not ok:
|
|
logger.error("Migration job failed after retries.")
|
|
return 1
|
|
logger.info("Migration job completed successfully.")
|
|
return 0
|
|
|
|
|
|
if __name__ == "__main__":
|
|
sys.exit(main())
|