From 6e5319aacb42fe536cde479702e6fa42b070ceba Mon Sep 17 00:00:00 2001 From: Claude Date: Wed, 20 May 2026 19:51:50 +0000 Subject: [PATCH] fix(greptile): drain background refresh + warn on router mode override Address the two new findings from greptile's 19:45 review of the vertex+router surfaces. - vertex_llm_base: when the slow path sees TokenState.INVALID, await any in-flight background refresh task before invoking refresh_auth ourselves. google-auth's Credentials.refresh() is not safe to call concurrently on the same credentials object, and the background task runs outside the per-key lock. After the wait, re-check the cached token so we can short-circuit if the background refresh already restored it. Extracted the helper into _await_in_flight_background_refresh so get_access_token_async stays under ruff's PLR0915 statement budget. - router.py: when alias registration would overwrite the deployment's declared `mode` to keep the shared backend mode stable, emit a verbose_router_logger.warning so the override is visible to operators instead of silently winning. The existing fix (preventing alias registration from downgrading a shared `mode: responses` to chat) is preserved; the warning just surfaces it. --- litellm/llms/vertex_ai/vertex_llm_base.py | 35 +++++++++++++++++++++++ litellm/router.py | 11 +++++++ 2 files changed, 46 insertions(+) diff --git a/litellm/llms/vertex_ai/vertex_llm_base.py b/litellm/llms/vertex_ai/vertex_llm_base.py index 350543f6fa8..48bbcac07c0 100644 --- a/litellm/llms/vertex_ai/vertex_llm_base.py +++ b/litellm/llms/vertex_ai/vertex_llm_base.py @@ -500,6 +500,27 @@ class VertexBase: exc_info=True, ) + async def _await_in_flight_background_refresh( + self, credential_cache_key: tuple + ) -> None: + """Wait for an in-flight background refresh to finish, if any. + + google-auth's ``Credentials.refresh()`` is not safe to invoke + concurrently on the same credentials object. Coroutines that need a + blocking refresh must first drain any background refresh that was + scheduled while a previous STALE token was being served. + """ + existing_task = self._background_refresh_tasks.get(credential_cache_key) + if existing_task is None or existing_task.done(): + return + try: + await existing_task + except Exception: + # Background refresh failures are already logged inside + # _background_refresh_credentials; the caller will fall through + # to its own blocking refresh. + pass + def _schedule_background_refresh( self, credentials: Any, @@ -1021,6 +1042,20 @@ class VertexBase: return current_token, resolved_project if token_state == TokenState.INVALID: + # Drain any in-flight background refresh before invoking + # refresh_auth ourselves; google-auth's + # Credentials.refresh() is not safe to call concurrently + # on the same credentials object, and the background task + # runs outside this lock. + await self._await_in_flight_background_refresh( + credential_cache_key + ) + cached = self._try_get_cached_token( + credential_cache_key, project_id + ) + if cached is not None: + return cached + # Token is expired or missing — must block until refresh completes. try: verbose_logger.debug("Credentials expired, refreshing") diff --git a/litellm/router.py b/litellm/router.py index 17ac728a67c..ba2abe83dc8 100644 --- a/litellm/router.py +++ b/litellm/router.py @@ -7331,6 +7331,17 @@ class Router: # Multiple aliases can point at the same provider/model backend, # but their deployment-level overrides should not downgrade the # backend from responses -> chat via last-write-wins registration. + _deployment_mode = _shared_model_info.get("mode") + if _deployment_mode is not None: + verbose_router_logger.warning( + "Router: preserving existing mode=%s for shared backend " + "key %s instead of the deployment-specified mode=%s " + "(prevents alias registration from downgrading the " + "shared backend mode).", + _existing_shared_mode, + _model_name, + _deployment_mode, + ) _shared_model_info["mode"] = _existing_shared_mode _backend_alias_cost = {_model_name: _shared_model_info} if "responses/" in _model_name: