From 476a9647ac3dda10237bd39a5353958870673cce Mon Sep 17 00:00:00 2001 From: MHammett <12503490+MHammett@users.noreply.github.com> Date: Fri, 18 Sep 2026 15:14:41 -0500 Subject: [PATCH] fix(windows): keep bundled tiktoken caches byte-exact on checkout A checkout with core.autocrlf=true, which is Git for Windows' default, gives the tiktoken cache files in litellm/litellm_core_utils/tokenizers/ CRLF line endings. tiktoken checks their sha256 and rejects them, then deletes them and downloads them again from openaipublic.blob.core.windows.net. So on Windows, litellm needs that host for its bundled encodings and fails without it. The win_amd64 wheel ships the files that way Mark the three cache files binary so no checkout converts them, and add a Windows test that loads each bundled cache with downloads refused Fixes #41855 --- .gitattributes | 6 +- .../test_bundled_tiktoken_caches.py | 58 +++++++++++++++++++ 2 files changed, 63 insertions(+), 1 deletion(-) create mode 100644 tests/windows_tests/test_bundled_tiktoken_caches.py diff --git a/.gitattributes b/.gitattributes index 5c9061f52ac..7127d1daae1 100644 --- a/.gitattributes +++ b/.gitattributes @@ -1,2 +1,6 @@ *.ipynb linguist-vendored -ui/litellm-dashboard/src/lib/http/schema.d.ts linguist-generated \ No newline at end of file +ui/litellm-dashboard/src/lib/http/schema.d.ts linguist-generated +# tiktoken rejects these caches unless they are byte-exact, and Windows checkouts would convert them to CRLF +litellm/litellm_core_utils/tokenizers/9b5ad71b2ce5302211f9c61530b329a4922fc6a4 binary +litellm/litellm_core_utils/tokenizers/ec7223a39ce59f226a68acc30dc1af2788490e15 binary +litellm/litellm_core_utils/tokenizers/fb374d419588a4632f3f557e76b4b70aebbca790 binary diff --git a/tests/windows_tests/test_bundled_tiktoken_caches.py b/tests/windows_tests/test_bundled_tiktoken_caches.py new file mode 100644 index 00000000000..042d4969513 --- /dev/null +++ b/tests/windows_tests/test_bundled_tiktoken_caches.py @@ -0,0 +1,58 @@ +import importlib.util +import re +import shutil +import subprocess +from pathlib import Path + +import pytest +import tiktoken.load +from tiktoken_ext import openai_public + +REPO_ROOT = Path(__file__).resolve().parents[2] +TOKENIZERS = Path("litellm", "litellm_core_utils", "tokenizers") +TIKTOKEN_CACHE_FILE_NAME = re.compile(r"[0-9a-f]{40}") + + +def _tokenizers_dir_without_importing_litellm() -> Path: + spec = importlib.util.find_spec("litellm") + assert spec is not None and spec.submodule_search_locations + return Path(next(iter(spec.submodule_search_locations))).joinpath(*TOKENIZERS.parts[1:]) + + +@pytest.mark.parametrize( + "load_encoding", + [openai_public.cl100k_base, openai_public.o200k_base, openai_public.p50k_base], + ids=["cl100k_base", "o200k_base", "p50k_base"], +) +def test_bundled_tiktoken_cache_loads_without_the_network(load_encoding, tmp_path, monkeypatch): + cache_copy = tmp_path / "tokenizers" + shutil.copytree(_tokenizers_dir_without_importing_litellm(), cache_copy) + monkeypatch.setenv("TIKTOKEN_CACHE_DIR", str(cache_copy)) + + def refuse_download(url): + raise AssertionError(f"tiktoken rejected the bundled cache and tried to download {url}") + + monkeypatch.setattr(tiktoken.load, "read_file", refuse_download) + + assert load_encoding()["mergeable_ranks"] + + +def test_gitattributes_keeps_the_bundled_caches_byte_exact(): + caches = sorted( + (TOKENIZERS / p.name).as_posix() + for p in (REPO_ROOT / TOKENIZERS).iterdir() + if TIKTOKEN_CACHE_FILE_NAME.fullmatch(p.name) + ) + assert caches, f"no tiktoken cache files in {REPO_ROOT / TOKENIZERS}" + try: + out = subprocess.run( + ["git", "check-attr", "text", "--", *caches], + cwd=REPO_ROOT, + capture_output=True, + text=True, + check=True, + ).stdout + except (OSError, subprocess.CalledProcessError): + pytest.skip("not a git checkout") + + assert out.splitlines() == [f"{cache}: text: unset" for cache in caches]