From 8fcda9083f74be15b2237d784e17df9c3ff5cfbf Mon Sep 17 00:00:00 2001 From: Ishaan Jaffer Date: Fri, 13 Mar 2026 17:55:47 -0700 Subject: [PATCH] fix(tag_regex): address backwards-compat, metadata overwrite, and warning noise Three issues from code review: 1. Backwards-compat: `has_tag_filter` was widened to activate on any non-empty User-Agent, which would raise ValueError for existing deployments using plain tags without a `default` fallback. Fix: only activate header-based regex filtering when at least one candidate deployment has `tag_regex` configured. 2. Metadata overwrite: `metadata["tag_routing"]` was overwritten for every matching deployment in the loop, leaving inaccurate provenance when multiple deployments match. Fix: write only for the first match. 3. Warning noise: an invalid regex pattern logged one warning per header string rather than once per pattern. Fix: compile first (catching re.error once), then iterate over header strings. Also adds two new tests covering these cases, and adds docs page for tag_regex routing with a Claude Code walk-through. --- .../docs/proxy/tag_regex_routing.md | 188 ++++++++++++++++++ docs/my-website/sidebars.js | 1 + litellm/router_strategy/tag_based_routing.py | 58 ++++-- .../test_router_tag_regex_routing.py | 79 ++++++++ 4 files changed, 308 insertions(+), 18 deletions(-) create mode 100644 docs/my-website/docs/proxy/tag_regex_routing.md diff --git a/docs/my-website/docs/proxy/tag_regex_routing.md b/docs/my-website/docs/proxy/tag_regex_routing.md new file mode 100644 index 00000000000..ea672e49517 --- /dev/null +++ b/docs/my-website/docs/proxy/tag_regex_routing.md @@ -0,0 +1,188 @@ +# Tag Regex Routing + +Route requests to specific deployments based on regex patterns matched against request headers — without requiring per-user tag configuration. + +## Overview + +With standard [tag-based routing](tag_routing), each request must carry a matching tag (e.g. `tags: ["vibe-coding"]`). This works well when you control the client, but becomes impractical at scale. + +**Tag regex routing** lets you match on headers the client already sends automatically — like `User-Agent` — so requests are routed correctly with zero client-side configuration. + +### Use case: route all Claude Code traffic to dedicated AWS accounts + +> "We need to route Claude Code traffic to a dedicated set of AWS accounts. We want to roll out the Claude Code → LiteLLM integration to 5,000 employees — it's not practical to ask every developer to configure a tag. Claude Code always sends a `User-Agent` header that starts with `claude-code/`, so we'd like LiteLLM to use that automatically." + +This is exactly what `tag_regex` is for. + +--- + +## Quick Start + +### 1. Configure `tag_regex` on the target deployment + +Add a `tag_regex` list to `litellm_params`. Each entry is a regex pattern matched against `"Header-Name: value"` strings built from the request metadata. + +```yaml +model_list: + # Claude Code traffic → dedicated Bedrock account, matched by User-Agent + - model_name: claude-sonnet + litellm_params: + model: bedrock/converse/anthropic-claude-sonnet-4-6 + aws_region_name: us-east-1 + aws_role_name: arn:aws:iam::111122223333:role/LiteLLMRole + tag_regex: + - "^User-Agent: claude-code\\/" # matches claude-code/1.x, claude-code/2.x, … + model_info: + id: claude-code-deployment + + # All other traffic → standard deployment (catch-all default) + - model_name: claude-sonnet + litellm_params: + model: bedrock/converse/anthropic-claude-sonnet-4-6 + aws_region_name: us-east-1 + aws_role_name: arn:aws:iam::444455556666:role/LiteLLMRole + tags: + - default + model_info: + id: regular-deployment + +router_settings: + enable_tag_filtering: true + tag_filtering_match_any: true + +general_settings: + master_key: sk-1234 +``` + +### 2. Start the proxy + +```shell +litellm --config config.yaml +``` + +### 3. Send a request from Claude Code + +Claude Code automatically sets `User-Agent: claude-code/`. No extra configuration needed on the client side. + +```shell +curl http://localhost:4000/v1/chat/completions \ + -H "Authorization: Bearer sk-1234" \ + -H "User-Agent: claude-code/1.2.3" \ + -d '{ + "model": "claude-sonnet", + "messages": [{"role": "user", "content": "hello"}] + }' +``` + +Check the response header to confirm routing: + +``` +x-litellm-model-id: claude-code-deployment +``` + +### 4. Send a request from any other client + +No `User-Agent: claude-code/` header → falls through to the `default` deployment. + +```shell +curl http://localhost:4000/v1/chat/completions \ + -H "Authorization: Bearer sk-1234" \ + -d '{ + "model": "claude-sonnet", + "messages": [{"role": "user", "content": "hello"}] + }' +``` + +``` +x-litellm-model-id: regular-deployment +``` + +--- + +## How it works + +When `enable_tag_filtering: true` is set and a deployment has `tag_regex` configured, LiteLLM builds a `"Header-Name: value"` string from the request's `User-Agent` and tests each regex pattern against it using `re.search`. + +**Matching priority (in order):** + +1. **Exact tag match** — if the request includes `tags: ["vibe-coding"]` and a deployment has `tags: ["vibe-coding"]`, that match fires first. +2. **Regex match** — if no exact tag match, `tag_regex` patterns are tested against request headers. +3. **Default fallback** — if nothing matches, deployments tagged `default` are used. +4. **All deployments** — if no `default` tag exists, all healthy deployments are returned (existing behaviour unchanged). + +**Backwards compatibility:** deployments that use only plain `tags` (no `tag_regex`) are unaffected, even when requests carry a `User-Agent` header. + +--- + +## Combining `tags` and `tag_regex` + +You can mix both on the same deployment — a request matches if either the tag or the regex matches: + +```yaml +- model_name: claude-sonnet + litellm_params: + model: bedrock/converse/anthropic-claude-sonnet-4-6 + aws_role_name: arn:aws:iam::111122223333:role/LiteLLMRole + tags: + - vibe-coding # explicit tag still works for teams that set it + tag_regex: + - "^User-Agent: claude-code\\/" # automatic match for everyone else +``` + +--- + +## Observability: `tag_routing` in SpendLogs + +When a regex matches, LiteLLM writes a `tag_routing` block into the request metadata. This flows automatically into SpendLogs so you can see how each request was routed: + +```json +{ + "tag_routing": { + "matched_deployment": "claude-sonnet", + "matched_via": "tag_regex", + "matched_value": "^User-Agent: claude-code\\/", + "user_agent": "claude-code/1.2.3", + "request_tags": [] + } +} +``` + +| Field | Description | +|-------|-------------| +| `matched_via` | `"tag_regex"` or `"tags"` | +| `matched_value` | The specific pattern or tag that matched | +| `user_agent` | The `User-Agent` value from the request | +| `request_tags` | Explicit tags on the request (if any) | + +--- + +## Reference + +### `tag_regex` field + +| | | +|-|-| +| **Location** | `litellm_params` in `config.yaml` | +| **Type** | `list[str]` | +| **Matching** | `re.search(pattern, "User-Agent: ")` | +| **Error handling** | Invalid regex patterns are skipped with a warning at startup and at match time | + +### Supported header sources + +Currently `User-Agent` is the only header source. The pattern format is `"Header-Name: value"`, so for a request with `User-Agent: claude-code/1.2.3` the string tested is `"User-Agent: claude-code/1.2.3"`. + +### Pattern tips + +| Goal | Pattern | +|------|---------| +| Match any Claude Code version | `^User-Agent: claude-code\/` | +| Match specific major version | `^User-Agent: claude-code\/1\.` | +| Match any semver | `^User-Agent: claude-code\/\d+\.\d+` | + +--- + +## Related + +- [Tag Based Routing](tag_routing) — explicit per-request tags +- [Team Based Routing](team_based_routing) — route by team membership +- [Request Tags](request_tags) — how tags flow through requests diff --git a/docs/my-website/sidebars.js b/docs/my-website/sidebars.js index d1eb331f55b..0b5e1c5df64 100644 --- a/docs/my-website/sidebars.js +++ b/docs/my-website/sidebars.js @@ -1018,6 +1018,7 @@ const sidebars = { "proxy/reliability", "proxy/fallback_management", "proxy/tag_routing", + "proxy/tag_regex_routing", "proxy/timeout", "wildcard_routing" ], diff --git a/litellm/router_strategy/tag_based_routing.py b/litellm/router_strategy/tag_based_routing.py index d7e756f3d69..c4940d04aa8 100644 --- a/litellm/router_strategy/tag_based_routing.py +++ b/litellm/router_strategy/tag_based_routing.py @@ -28,18 +28,20 @@ def _is_valid_deployment_tag_regex( Test compiled regex patterns against "Header-Name: value" strings. Returns the first matching pattern string, or None if nothing matches. - Uses re.compile() which has an internal LRU cache — no per-call overhead - after the first compile. + Compiles each pattern once (re's LRU cache) and logs invalid patterns once + per pattern, not once per header string. """ for pattern in tag_regexes: + try: + compiled = re.compile(pattern) + except re.error: + verbose_logger.warning( + "tag_regex: invalid pattern %r — skipping", pattern + ) + continue for header_str in header_strings: - try: - if re.search(pattern, header_str): - return pattern - except re.error: - verbose_logger.warning( - "tag_regex: invalid pattern %r — skipping", pattern - ) + if compiled.search(header_str): + return pattern return None @@ -117,7 +119,23 @@ async def get_deployments_for_tag( new_healthy_deployments: List[Any] = [] default_deployments: List[Any] = [] - has_tag_filter = bool(request_tags) or bool(header_strings) + # Only activate header-based regex filtering when at least one deployment in + # the candidate set has tag_regex configured. This preserves existing + # behaviour for operators who use plain tags: a request that carries a + # User-Agent (all proxy requests do) but targets deployments with no + # tag_regex will continue to use the original tag-only code path. + _healthy_list = ( + healthy_deployments + if isinstance(healthy_deployments, list) + else list(healthy_deployments) + ) + has_regex_deployments = any( + d.get("litellm_params", {}).get("tag_regex") + for d in _healthy_list + ) + has_tag_filter = bool(request_tags) or ( + bool(header_strings) and has_regex_deployments + ) if has_tag_filter: verbose_logger.debug( "get_deployments_for_tag routing: request_tags=%s user_agent=%s", @@ -164,14 +182,18 @@ async def get_deployments_for_tag( matched_via, matched_value, ) - # Record provenance in metadata so it flows to SpendLogs - metadata["tag_routing"] = { - "matched_deployment": deployment.get("model_name"), - "matched_via": matched_via, - "matched_value": matched_value, - "request_tags": request_tags or [], - "user_agent": user_agent, - } + # Record provenance in metadata so it flows to SpendLogs. + # Written only for the first match — load balancer selects one + # deployment from new_healthy_deployments, so overwriting on + # subsequent matches would produce misleading observability data. + if "tag_routing" not in metadata: + metadata["tag_routing"] = { + "matched_deployment": deployment.get("model_name"), + "matched_via": matched_via, + "matched_value": matched_value, + "request_tags": request_tags or [], + "user_agent": user_agent, + } new_healthy_deployments.append(deployment) if deployment_tags and "default" in deployment_tags: diff --git a/tests/test_litellm/router_strategy/test_router_tag_regex_routing.py b/tests/test_litellm/router_strategy/test_router_tag_regex_routing.py index cd762dbcdfe..88f0f5142bb 100644 --- a/tests/test_litellm/router_strategy/test_router_tag_regex_routing.py +++ b/tests/test_litellm/router_strategy/test_router_tag_regex_routing.py @@ -215,3 +215,82 @@ async def test_explicit_tag_match_takes_precedence_over_regex(): assert len(result) == 1 tr = metadata.get("tag_routing", {}) assert tr.get("matched_via") == "tags" + + +@pytest.mark.asyncio +async def test_user_agent_present_no_tag_regex_deployments_does_not_raise(): + """ + Backwards-compat: a request that carries a User-Agent but targets plain-tag + deployments (no tag_regex) must NOT raise ValueError — it should fall + through to the default/all-deployments path just like before. + """ + plain_tag_only_deployments = [ + { + "model_name": "gpt-4", + "litellm_params": { + "model": "openai/premium-deployment", + "api_key": "fake", + "tags": ["premium"], + }, + "model_info": {"id": "premium-deployment"}, + }, + { + "model_name": "gpt-4", + "litellm_params": { + "model": "openai/free-deployment", + "api_key": "fake", + "tags": ["free"], + }, + "model_info": {"id": "free-deployment"}, + }, + ] + router = _make_router_mock() + # The request has a User-Agent (as all proxy requests do) but NO tags and + # neither deployment has tag_regex — must not raise, must return all. + result = await get_deployments_for_tag( + llm_router_instance=router, + model="gpt-4", + healthy_deployments=plain_tag_only_deployments, + request_kwargs={"metadata": {"user_agent": "Mozilla/5.0 (any-client)"}}, + ) + # Falls through to "return healthy_deployments" path unchanged + assert result == plain_tag_only_deployments + + +@pytest.mark.asyncio +async def test_tag_routing_metadata_not_overwritten_for_multiple_matches(): + """ + When multiple deployments match, tag_routing records only the first match + so the provenance reflects what the load balancer likely selected. + """ + deployment_a = { + "model_name": "claude-sonnet", + "litellm_params": { + "model": "openai/cc-deployment-a", + "api_key": "fake", + "tag_regex": [r"^User-Agent: claude-code\/"], + }, + "model_info": {"id": "cc-deployment-a"}, + } + deployment_b = { + "model_name": "claude-sonnet", + "litellm_params": { + "model": "openai/cc-deployment-b", + "api_key": "fake", + "tag_regex": [r"^User-Agent: claude-code\/"], + }, + "model_info": {"id": "cc-deployment-b"}, + } + router = _make_router_mock() + metadata: dict = {"user_agent": "claude-code/1.0"} + result = await get_deployments_for_tag( + llm_router_instance=router, + model="claude-sonnet", + healthy_deployments=[deployment_a, deployment_b], + request_kwargs={"metadata": metadata}, + ) + assert len(result) == 2 + # tag_routing recorded once and reflects the first match + tr = metadata.get("tag_routing", {}) + assert tr.get("matched_deployment") == "claude-sonnet" + assert tr.get("matched_via") == "tag_regex"