fix(router): mirror last-wins for stripped/wildcard dict fallback resolution

_resolve_dict_fallback_targets returned the first stripped or wildcard match,
but get_fallback_model_group keeps overwriting and uses the last match in each
of those tiers. With duplicate stripped keys (or wildcards), the runtime
fallback hop calls the last target while the capability scan only saw the
first, so a Sonnet 4.6+ deployment hidden behind a prefill-supporting first
hit was never detected and the mid-stream resume still sent assistant prefill
and got a 400.

Mirror the helper's per-tier semantics: exact still first-wins-and-breaks;
stripped and wildcard now pick the last match in their tier.
This commit is contained in:
Cursor Agent 2026-06-18 14:11:28 +00:00
parent 7641db9543
commit 2068377040
No known key found for this signature in database
2 changed files with 55 additions and 11 deletions

View file

@ -48,6 +48,12 @@ def _resolve_dict_fallback_targets(
directly rather than calling that helper because the helper pops string
entries mid-iteration, which skips the following entry and can drop a dict
target when a bare string precedes it in the list.
Within each tier this matches the helper's pick: exact match breaks on the
first hit (first wins), while stripped and wildcard hits keep overwriting
without a break (last wins). Mirroring last-wins for those tiers keeps the
capability scan aligned with the target the runtime fallback hop will
actually call.
"""
dict_entries = tuple(
(key, entry[key])
@ -58,18 +64,17 @@ def _resolve_dict_fallback_targets(
exact = next((t for k, t in dict_entries if k == model_group), None)
if exact is not None:
return tuple(exact)
stripped = next(
(
t
for k, t in dict_entries
if _check_stripped_model_group(model_group=model_group, fallback_key=k)
),
None,
stripped_matches = tuple(
t
for k, t in dict_entries
if _check_stripped_model_group(model_group=model_group, fallback_key=k)
)
if stripped is not None:
return tuple(stripped)
wildcard = next((t for k, t in dict_entries if k == "*"), None)
return tuple(wildcard or ())
if stripped_matches:
return tuple(stripped_matches[-1])
wildcard_matches = tuple(t for k, t in dict_entries if k == "*")
if wildcard_matches:
return tuple(wildcard_matches[-1])
return ()
def _candidate_model_groups(

View file

@ -275,6 +275,45 @@ def test_stripped_model_group_fallback_is_classified():
assert PARTIAL in result[1]["content"]
def test_duplicate_stripped_keys_pick_last_match():
"""get_fallback_model_group keeps overwriting stripped matches without
breaking, so the last duplicate key in the list is the one the runtime
fallback hop calls. The capability scan must mirror that or it will miss
a prefill-rejecting target hidden behind a prefill-supporting first hit."""
result = build_mid_stream_continuation_messages(
messages=MESSAGES,
generated_content=PARTIAL,
model_group="openai/gpt-3.5-turbo",
fallbacks=[
{"gpt-3.5-turbo": ["gpt-4o"]},
{"gpt-3.5-turbo": ["claude-sonnet-4-6"]},
],
)
assert len(result) == 2
assert result[1]["role"] == "user"
assert PARTIAL in result[1]["content"]
assert all(m.get("prefix") is not True for m in result)
def test_duplicate_wildcard_keys_pick_last_match():
"""Same last-wins semantics for wildcard entries: get_fallback_model_group
rewrites generic_fallback_idx for every ``*`` entry, so the last one is
the runtime target."""
result = build_mid_stream_continuation_messages(
messages=MESSAGES,
generated_content=PARTIAL,
model_group="some-unmatched-group",
fallbacks=[
{"*": ["gpt-4o"]},
{"*": ["claude-sonnet-4-6"]},
],
)
assert len(result) == 2
assert result[1]["role"] == "user"
assert PARTIAL in result[1]["content"]
assert all(m.get("prefix") is not True for m in result)
def test_exact_group_match_takes_priority_over_wildcard():
"""When both an exact group match and a wildcard fallback are configured,
only the exact target is tried, so a prefill-rejecting model behind the