diff --git a/litellm/router_utils/fallback_event_handlers.py b/litellm/router_utils/fallback_event_handlers.py index 4c8688a2d0b..fb390462728 100644 --- a/litellm/router_utils/fallback_event_handlers.py +++ b/litellm/router_utils/fallback_event_handlers.py @@ -48,6 +48,12 @@ def _resolve_dict_fallback_targets( directly rather than calling that helper because the helper pops string entries mid-iteration, which skips the following entry and can drop a dict target when a bare string precedes it in the list. + + Within each tier this matches the helper's pick: exact match breaks on the + first hit (first wins), while stripped and wildcard hits keep overwriting + without a break (last wins). Mirroring last-wins for those tiers keeps the + capability scan aligned with the target the runtime fallback hop will + actually call. """ dict_entries = tuple( (key, entry[key]) @@ -58,18 +64,17 @@ def _resolve_dict_fallback_targets( exact = next((t for k, t in dict_entries if k == model_group), None) if exact is not None: return tuple(exact) - stripped = next( - ( - t - for k, t in dict_entries - if _check_stripped_model_group(model_group=model_group, fallback_key=k) - ), - None, + stripped_matches = tuple( + t + for k, t in dict_entries + if _check_stripped_model_group(model_group=model_group, fallback_key=k) ) - if stripped is not None: - return tuple(stripped) - wildcard = next((t for k, t in dict_entries if k == "*"), None) - return tuple(wildcard or ()) + if stripped_matches: + return tuple(stripped_matches[-1]) + wildcard_matches = tuple(t for k, t in dict_entries if k == "*") + if wildcard_matches: + return tuple(wildcard_matches[-1]) + return () def _candidate_model_groups( diff --git a/tests/test_litellm/router_utils/test_fallback_event_handlers.py b/tests/test_litellm/router_utils/test_fallback_event_handlers.py index f856c601d6b..ea8aa6e6a9a 100644 --- a/tests/test_litellm/router_utils/test_fallback_event_handlers.py +++ b/tests/test_litellm/router_utils/test_fallback_event_handlers.py @@ -275,6 +275,45 @@ def test_stripped_model_group_fallback_is_classified(): assert PARTIAL in result[1]["content"] +def test_duplicate_stripped_keys_pick_last_match(): + """get_fallback_model_group keeps overwriting stripped matches without + breaking, so the last duplicate key in the list is the one the runtime + fallback hop calls. The capability scan must mirror that or it will miss + a prefill-rejecting target hidden behind a prefill-supporting first hit.""" + result = build_mid_stream_continuation_messages( + messages=MESSAGES, + generated_content=PARTIAL, + model_group="openai/gpt-3.5-turbo", + fallbacks=[ + {"gpt-3.5-turbo": ["gpt-4o"]}, + {"gpt-3.5-turbo": ["claude-sonnet-4-6"]}, + ], + ) + assert len(result) == 2 + assert result[1]["role"] == "user" + assert PARTIAL in result[1]["content"] + assert all(m.get("prefix") is not True for m in result) + + +def test_duplicate_wildcard_keys_pick_last_match(): + """Same last-wins semantics for wildcard entries: get_fallback_model_group + rewrites generic_fallback_idx for every ``*`` entry, so the last one is + the runtime target.""" + result = build_mid_stream_continuation_messages( + messages=MESSAGES, + generated_content=PARTIAL, + model_group="some-unmatched-group", + fallbacks=[ + {"*": ["gpt-4o"]}, + {"*": ["claude-sonnet-4-6"]}, + ], + ) + assert len(result) == 2 + assert result[1]["role"] == "user" + assert PARTIAL in result[1]["content"] + assert all(m.get("prefix") is not True for m in result) + + def test_exact_group_match_takes_priority_over_wildcard(): """When both an exact group match and a wildcard fallback are configured, only the exact target is tried, so a prefill-rejecting model behind the