fix(windows): keep bundled tiktoken caches byte-exact on checkout

A checkout with core.autocrlf=true, which is Git for Windows' default,
gives the tiktoken cache files in litellm/litellm_core_utils/tokenizers/
CRLF line endings. tiktoken checks their sha256 and rejects them, then
deletes them and downloads them again from
openaipublic.blob.core.windows.net. So on Windows, litellm needs that host
for its bundled encodings and fails without it. The win_amd64 wheel ships
the files that way

Mark the three cache files binary so no checkout converts them, and add a
Windows test that loads each bundled cache with downloads refused

Fixes #41855
This commit is contained in:
MHammett 2026-09-18 15:14:41 -05:00
parent 6e3b6d6d03
commit 476a9647ac
2 changed files with 63 additions and 1 deletions

6
.gitattributes vendored
View file

@ -1,2 +1,6 @@
*.ipynb linguist-vendored
ui/litellm-dashboard/src/lib/http/schema.d.ts linguist-generated
ui/litellm-dashboard/src/lib/http/schema.d.ts linguist-generated
# tiktoken rejects these caches unless they are byte-exact, and Windows checkouts would convert them to CRLF
litellm/litellm_core_utils/tokenizers/9b5ad71b2ce5302211f9c61530b329a4922fc6a4 binary
litellm/litellm_core_utils/tokenizers/ec7223a39ce59f226a68acc30dc1af2788490e15 binary
litellm/litellm_core_utils/tokenizers/fb374d419588a4632f3f557e76b4b70aebbca790 binary

View file

@ -0,0 +1,58 @@
import importlib.util
import re
import shutil
import subprocess
from pathlib import Path
import pytest
import tiktoken.load
from tiktoken_ext import openai_public
REPO_ROOT = Path(__file__).resolve().parents[2]
TOKENIZERS = Path("litellm", "litellm_core_utils", "tokenizers")
TIKTOKEN_CACHE_FILE_NAME = re.compile(r"[0-9a-f]{40}")
def _tokenizers_dir_without_importing_litellm() -> Path:
spec = importlib.util.find_spec("litellm")
assert spec is not None and spec.submodule_search_locations
return Path(next(iter(spec.submodule_search_locations))).joinpath(*TOKENIZERS.parts[1:])
@pytest.mark.parametrize(
"load_encoding",
[openai_public.cl100k_base, openai_public.o200k_base, openai_public.p50k_base],
ids=["cl100k_base", "o200k_base", "p50k_base"],
)
def test_bundled_tiktoken_cache_loads_without_the_network(load_encoding, tmp_path, monkeypatch):
cache_copy = tmp_path / "tokenizers"
shutil.copytree(_tokenizers_dir_without_importing_litellm(), cache_copy)
monkeypatch.setenv("TIKTOKEN_CACHE_DIR", str(cache_copy))
def refuse_download(url):
raise AssertionError(f"tiktoken rejected the bundled cache and tried to download {url}")
monkeypatch.setattr(tiktoken.load, "read_file", refuse_download)
assert load_encoding()["mergeable_ranks"]
def test_gitattributes_keeps_the_bundled_caches_byte_exact():
caches = sorted(
(TOKENIZERS / p.name).as_posix()
for p in (REPO_ROOT / TOKENIZERS).iterdir()
if TIKTOKEN_CACHE_FILE_NAME.fullmatch(p.name)
)
assert caches, f"no tiktoken cache files in {REPO_ROOT / TOKENIZERS}"
try:
out = subprocess.run(
["git", "check-attr", "text", "--", *caches],
cwd=REPO_ROOT,
capture_output=True,
text=True,
check=True,
).stdout
except (OSError, subprocess.CalledProcessError):
pytest.skip("not a git checkout")
assert out.splitlines() == [f"{cache}: text: unset" for cache in caches]