fix(greptile): drain background refresh + warn on router mode override

Address the two new findings from greptile's 19:45 review of the
vertex+router surfaces.

- vertex_llm_base: when the slow path sees TokenState.INVALID, await any
  in-flight background refresh task before invoking refresh_auth
  ourselves. google-auth's Credentials.refresh() is not safe to call
  concurrently on the same credentials object, and the background task
  runs outside the per-key lock. After the wait, re-check the cached
  token so we can short-circuit if the background refresh already
  restored it. Extracted the helper into
  _await_in_flight_background_refresh so get_access_token_async stays
  under ruff's PLR0915 statement budget.
- router.py: when alias registration would overwrite the deployment's
  declared `mode` to keep the shared backend mode stable, emit a
  verbose_router_logger.warning so the override is visible to operators
  instead of silently winning. The existing fix (preventing alias
  registration from downgrading a shared `mode: responses` to chat) is
  preserved; the warning just surfaces it.
This commit is contained in:
Claude 2026-05-20 19:51:50 +00:00
parent f71ab0f1a1
commit 6e5319aacb
No known key found for this signature in database
2 changed files with 46 additions and 0 deletions

View file

@ -500,6 +500,27 @@ class VertexBase:
exc_info=True,
)
async def _await_in_flight_background_refresh(
self, credential_cache_key: tuple
) -> None:
"""Wait for an in-flight background refresh to finish, if any.
google-auth's ``Credentials.refresh()`` is not safe to invoke
concurrently on the same credentials object. Coroutines that need a
blocking refresh must first drain any background refresh that was
scheduled while a previous STALE token was being served.
"""
existing_task = self._background_refresh_tasks.get(credential_cache_key)
if existing_task is None or existing_task.done():
return
try:
await existing_task
except Exception:
# Background refresh failures are already logged inside
# _background_refresh_credentials; the caller will fall through
# to its own blocking refresh.
pass
def _schedule_background_refresh(
self,
credentials: Any,
@ -1021,6 +1042,20 @@ class VertexBase:
return current_token, resolved_project
if token_state == TokenState.INVALID:
# Drain any in-flight background refresh before invoking
# refresh_auth ourselves; google-auth's
# Credentials.refresh() is not safe to call concurrently
# on the same credentials object, and the background task
# runs outside this lock.
await self._await_in_flight_background_refresh(
credential_cache_key
)
cached = self._try_get_cached_token(
credential_cache_key, project_id
)
if cached is not None:
return cached
# Token is expired or missing — must block until refresh completes.
try:
verbose_logger.debug("Credentials expired, refreshing")

View file

@ -7331,6 +7331,17 @@ class Router:
# Multiple aliases can point at the same provider/model backend,
# but their deployment-level overrides should not downgrade the
# backend from responses -> chat via last-write-wins registration.
_deployment_mode = _shared_model_info.get("mode")
if _deployment_mode is not None:
verbose_router_logger.warning(
"Router: preserving existing mode=%s for shared backend "
"key %s instead of the deployment-specified mode=%s "
"(prevents alias registration from downgrading the "
"shared backend mode).",
_existing_shared_mode,
_model_name,
_deployment_mode,
)
_shared_model_info["mode"] = _existing_shared_mode
_backend_alias_cost = {_model_name: _shared_model_info}
if "responses/" in _model_name: