diff --git a/litellm/llms/vertex_ai/vertex_llm_base.py b/litellm/llms/vertex_ai/vertex_llm_base.py index 350543f6fa8..48bbcac07c0 100644 --- a/litellm/llms/vertex_ai/vertex_llm_base.py +++ b/litellm/llms/vertex_ai/vertex_llm_base.py @@ -500,6 +500,27 @@ class VertexBase: exc_info=True, ) + async def _await_in_flight_background_refresh( + self, credential_cache_key: tuple + ) -> None: + """Wait for an in-flight background refresh to finish, if any. + + google-auth's ``Credentials.refresh()`` is not safe to invoke + concurrently on the same credentials object. Coroutines that need a + blocking refresh must first drain any background refresh that was + scheduled while a previous STALE token was being served. + """ + existing_task = self._background_refresh_tasks.get(credential_cache_key) + if existing_task is None or existing_task.done(): + return + try: + await existing_task + except Exception: + # Background refresh failures are already logged inside + # _background_refresh_credentials; the caller will fall through + # to its own blocking refresh. + pass + def _schedule_background_refresh( self, credentials: Any, @@ -1021,6 +1042,20 @@ class VertexBase: return current_token, resolved_project if token_state == TokenState.INVALID: + # Drain any in-flight background refresh before invoking + # refresh_auth ourselves; google-auth's + # Credentials.refresh() is not safe to call concurrently + # on the same credentials object, and the background task + # runs outside this lock. + await self._await_in_flight_background_refresh( + credential_cache_key + ) + cached = self._try_get_cached_token( + credential_cache_key, project_id + ) + if cached is not None: + return cached + # Token is expired or missing — must block until refresh completes. try: verbose_logger.debug("Credentials expired, refreshing") diff --git a/litellm/router.py b/litellm/router.py index 17ac728a67c..ba2abe83dc8 100644 --- a/litellm/router.py +++ b/litellm/router.py @@ -7331,6 +7331,17 @@ class Router: # Multiple aliases can point at the same provider/model backend, # but their deployment-level overrides should not downgrade the # backend from responses -> chat via last-write-wins registration. + _deployment_mode = _shared_model_info.get("mode") + if _deployment_mode is not None: + verbose_router_logger.warning( + "Router: preserving existing mode=%s for shared backend " + "key %s instead of the deployment-specified mode=%s " + "(prevents alias registration from downgrading the " + "shared backend mode).", + _existing_shared_mode, + _model_name, + _deployment_mode, + ) _shared_model_info["mode"] = _existing_shared_mode _backend_alias_cost = {_model_name: _shared_model_info} if "responses/" in _model_name: