mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-10 03:28:53 +00:00
fix(router): mirror last-wins for stripped/wildcard dict fallback resolution
_resolve_dict_fallback_targets returned the first stripped or wildcard match, but get_fallback_model_group keeps overwriting and uses the last match in each of those tiers. With duplicate stripped keys (or wildcards), the runtime fallback hop calls the last target while the capability scan only saw the first, so a Sonnet 4.6+ deployment hidden behind a prefill-supporting first hit was never detected and the mid-stream resume still sent assistant prefill and got a 400. Mirror the helper's per-tier semantics: exact still first-wins-and-breaks; stripped and wildcard now pick the last match in their tier.
This commit is contained in:
parent
7641db9543
commit
2068377040
2 changed files with 55 additions and 11 deletions
|
|
@ -48,6 +48,12 @@ def _resolve_dict_fallback_targets(
|
|||
directly rather than calling that helper because the helper pops string
|
||||
entries mid-iteration, which skips the following entry and can drop a dict
|
||||
target when a bare string precedes it in the list.
|
||||
|
||||
Within each tier this matches the helper's pick: exact match breaks on the
|
||||
first hit (first wins), while stripped and wildcard hits keep overwriting
|
||||
without a break (last wins). Mirroring last-wins for those tiers keeps the
|
||||
capability scan aligned with the target the runtime fallback hop will
|
||||
actually call.
|
||||
"""
|
||||
dict_entries = tuple(
|
||||
(key, entry[key])
|
||||
|
|
@ -58,18 +64,17 @@ def _resolve_dict_fallback_targets(
|
|||
exact = next((t for k, t in dict_entries if k == model_group), None)
|
||||
if exact is not None:
|
||||
return tuple(exact)
|
||||
stripped = next(
|
||||
(
|
||||
t
|
||||
for k, t in dict_entries
|
||||
if _check_stripped_model_group(model_group=model_group, fallback_key=k)
|
||||
),
|
||||
None,
|
||||
stripped_matches = tuple(
|
||||
t
|
||||
for k, t in dict_entries
|
||||
if _check_stripped_model_group(model_group=model_group, fallback_key=k)
|
||||
)
|
||||
if stripped is not None:
|
||||
return tuple(stripped)
|
||||
wildcard = next((t for k, t in dict_entries if k == "*"), None)
|
||||
return tuple(wildcard or ())
|
||||
if stripped_matches:
|
||||
return tuple(stripped_matches[-1])
|
||||
wildcard_matches = tuple(t for k, t in dict_entries if k == "*")
|
||||
if wildcard_matches:
|
||||
return tuple(wildcard_matches[-1])
|
||||
return ()
|
||||
|
||||
|
||||
def _candidate_model_groups(
|
||||
|
|
|
|||
|
|
@ -275,6 +275,45 @@ def test_stripped_model_group_fallback_is_classified():
|
|||
assert PARTIAL in result[1]["content"]
|
||||
|
||||
|
||||
def test_duplicate_stripped_keys_pick_last_match():
|
||||
"""get_fallback_model_group keeps overwriting stripped matches without
|
||||
breaking, so the last duplicate key in the list is the one the runtime
|
||||
fallback hop calls. The capability scan must mirror that or it will miss
|
||||
a prefill-rejecting target hidden behind a prefill-supporting first hit."""
|
||||
result = build_mid_stream_continuation_messages(
|
||||
messages=MESSAGES,
|
||||
generated_content=PARTIAL,
|
||||
model_group="openai/gpt-3.5-turbo",
|
||||
fallbacks=[
|
||||
{"gpt-3.5-turbo": ["gpt-4o"]},
|
||||
{"gpt-3.5-turbo": ["claude-sonnet-4-6"]},
|
||||
],
|
||||
)
|
||||
assert len(result) == 2
|
||||
assert result[1]["role"] == "user"
|
||||
assert PARTIAL in result[1]["content"]
|
||||
assert all(m.get("prefix") is not True for m in result)
|
||||
|
||||
|
||||
def test_duplicate_wildcard_keys_pick_last_match():
|
||||
"""Same last-wins semantics for wildcard entries: get_fallback_model_group
|
||||
rewrites generic_fallback_idx for every ``*`` entry, so the last one is
|
||||
the runtime target."""
|
||||
result = build_mid_stream_continuation_messages(
|
||||
messages=MESSAGES,
|
||||
generated_content=PARTIAL,
|
||||
model_group="some-unmatched-group",
|
||||
fallbacks=[
|
||||
{"*": ["gpt-4o"]},
|
||||
{"*": ["claude-sonnet-4-6"]},
|
||||
],
|
||||
)
|
||||
assert len(result) == 2
|
||||
assert result[1]["role"] == "user"
|
||||
assert PARTIAL in result[1]["content"]
|
||||
assert all(m.get("prefix") is not True for m in result)
|
||||
|
||||
|
||||
def test_exact_group_match_takes_priority_over_wildcard():
|
||||
"""When both an exact group match and a wildcard fallback are configured,
|
||||
only the exact target is tried, so a prefill-rejecting model behind the
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue