From d334c6a3073fecb43517c4f4aa88ad5c8546cf4e Mon Sep 17 00:00:00 2001 From: Aditya Date: Tue, 6 Oct 2026 23:23:52 +0530 Subject: [PATCH] fix: read model text files as UTF-8 regardless of platform locale MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit open() without an encoding uses the locale encoding, which is cp1252 on Windows. Model files are UTF-8, so Bm25(language=...) raised UnicodeDecodeError for Russian, Greek and Arabic stopwords and silently garbled German and French ones ("über" read as "über", so it was never filtered). The same applied to vocab files, IDF files and tokenizer and preprocessor configs. Pass encoding="utf-8" to every text-mode open() in the package. --- fastembed/common/preprocessor_utils.py | 8 +- .../colmodernvbert.py | 6 +- fastembed/sparse/bm25.py | 2 +- fastembed/sparse/bm42.py | 2 +- fastembed/sparse/if_splade.py | 2 +- fastembed/sparse/minicoil.py | 2 +- fastembed/sparse/utils/vocab_resolver.py | 8 +- tests/test_text_file_encoding.py | 75 +++++++++++++++++++ 8 files changed, 90 insertions(+), 15 deletions(-) create mode 100644 tests/test_text_file_encoding.py diff --git a/fastembed/common/preprocessor_utils.py b/fastembed/common/preprocessor_utils.py index fb853924..3b8e013c 100644 --- a/fastembed/common/preprocessor_utils.py +++ b/fastembed/common/preprocessor_utils.py @@ -14,7 +14,7 @@ def load_special_tokens(model_dir: Path) -> dict[str, Any]: if not tokens_map_path.exists(): return {} - with open(str(tokens_map_path)) as tokens_map_file: + with open(str(tokens_map_path), encoding="utf-8") as tokens_map_file: tokens_map = json.load(tokens_map_file) return tokens_map @@ -84,10 +84,10 @@ def load_tokenizer(model_dir: Path) -> tuple[Tokenizer, dict[str, int]]: config_path = model_dir / "config.json" config: dict[str, Any] = {} if config_path.exists(): - with open(str(config_path)) as config_file: + with open(str(config_path), encoding="utf-8") as config_file: config = json.load(config_file) - with open(str(tokenizer_config_path)) as tokenizer_config_file: + with open(str(tokenizer_config_path), encoding="utf-8") as tokenizer_config_file: tokenizer_config = json.load(tokenizer_config_file) max_context = _resolve_max_context(tokenizer_config, model_dir) @@ -153,7 +153,7 @@ def load_preprocessor(model_dir: Path) -> Compose: if not preprocessor_config_path.exists(): raise ValueError(f"Could not find preprocessor_config.json in {model_dir}") - with open(str(preprocessor_config_path)) as preprocessor_config_file: + with open(str(preprocessor_config_path), encoding="utf-8") as preprocessor_config_file: preprocessor_config = json.load(preprocessor_config_file) transforms = Compose.from_config(preprocessor_config) return transforms diff --git a/fastembed/late_interaction_multimodal/colmodernvbert.py b/fastembed/late_interaction_multimodal/colmodernvbert.py index f8fe0154..99f33217 100644 --- a/fastembed/late_interaction_multimodal/colmodernvbert.py +++ b/fastembed/late_interaction_multimodal/colmodernvbert.py @@ -138,12 +138,12 @@ def load_onnx_model(self) -> None: # Load image processing configuration processor_config_path = self._model_dir / "processor_config.json" - with open(processor_config_path) as f: + with open(processor_config_path, encoding="utf-8") as f: processor_config = json.load(f) self.image_seq_len = processor_config.get("image_seq_len", 64) preprocessor_config_path = self._model_dir / "preprocessor_config.json" - with open(preprocessor_config_path) as f: + with open(preprocessor_config_path, encoding="utf-8") as f: preprocessor_config = json.load(f) self.max_image_size = preprocessor_config.get("max_image_size", {}).get( "longest_edge", 512 @@ -151,7 +151,7 @@ def load_onnx_model(self) -> None: # Load model configuration config_path = self._model_dir / "config.json" - with open(config_path) as f: + with open(config_path, encoding="utf-8") as f: model_config = json.load(f) vision_config = model_config.get("vision_config", {}) self.image_size = vision_config.get("image_size", 512) diff --git a/fastembed/sparse/bm25.py b/fastembed/sparse/bm25.py index e095c627..e79b4e96 100644 --- a/fastembed/sparse/bm25.py +++ b/fastembed/sparse/bm25.py @@ -198,7 +198,7 @@ def _load_stopwords(cls, model_dir: Path, language: str) -> list[str]: if not stopwords_path.exists(): return [] - with open(stopwords_path, "r") as f: + with open(stopwords_path, "r", encoding="utf-8") as f: return f.read().splitlines() def _embed_documents( diff --git a/fastembed/sparse/bm42.py b/fastembed/sparse/bm42.py index b4685a88..dc3d0295 100644 --- a/fastembed/sparse/bm42.py +++ b/fastembed/sparse/bm42.py @@ -290,7 +290,7 @@ def _load_stopwords(cls, model_dir: Path) -> list[str]: if not stopwords_path.exists(): return [] - with open(stopwords_path, "r") as f: + with open(stopwords_path, "r", encoding="utf-8") as f: return f.read().splitlines() def embed( diff --git a/fastembed/sparse/if_splade.py b/fastembed/sparse/if_splade.py index 4e1c5bda..25374bca 100644 --- a/fastembed/sparse/if_splade.py +++ b/fastembed/sparse/if_splade.py @@ -164,7 +164,7 @@ def load_onnx_model(self) -> None: ) def _load_idf(self) -> dict[int, float]: - with open(self._model_dir / IDF_FILE) as f: + with open(self._model_dir / IDF_FILE, encoding="utf-8") as f: token_to_idf: dict[str, float] = json.load(f) vocab: dict[str, int] = self.tokenizer.get_vocab() # type: ignore[union-attr] diff --git a/fastembed/sparse/minicoil.py b/fastembed/sparse/minicoil.py index 0aabad7d..b6b85575 100644 --- a/fastembed/sparse/minicoil.py +++ b/fastembed/sparse/minicoil.py @@ -275,7 +275,7 @@ def _load_stopwords(cls, model_dir: Path) -> list[str]: if not stopwords_path.exists(): return [] - with open(stopwords_path, "r") as f: + with open(stopwords_path, "r", encoding="utf-8") as f: return f.read().splitlines() @classmethod diff --git a/fastembed/sparse/utils/vocab_resolver.py b/fastembed/sparse/utils/vocab_resolver.py index 54e037a7..37d4a818 100644 --- a/fastembed/sparse/utils/vocab_resolver.py +++ b/fastembed/sparse/utils/vocab_resolver.py @@ -64,20 +64,20 @@ def vocab_size(self) -> int: return len(self.vocab) + 1 def save_vocab(self, path: str) -> None: - with open(path, "w") as f: + with open(path, "w", encoding="utf-8") as f: for word in self.words: f.write(word + "\n") def save_json_vocab(self, path: str) -> None: import json - with open(path, "w") as f: + with open(path, "w", encoding="utf-8") as f: json.dump({"vocab": self.words, "stem_mapping": self.stem_mapping}, f, indent=2) def load_json_vocab(self, path: str) -> None: import json - with open(path, "r") as f: + with open(path, "r", encoding="utf-8") as f: data = json.load(f) self.words = data["vocab"] self.vocab = {word: idx + 1 for idx, word in enumerate(self.words)} @@ -98,7 +98,7 @@ def add_word(self, word: str) -> None: self.stem_mapping[stem] = word def load_vocab(self, path: str) -> None: - with open(path, "r") as f: + with open(path, "r", encoding="utf-8") as f: for line in f: self.add_word(line.strip()) diff --git a/tests/test_text_file_encoding.py b/tests/test_text_file_encoding.py new file mode 100644 index 00000000..6874874f --- /dev/null +++ b/tests/test_text_file_encoding.py @@ -0,0 +1,75 @@ +"""Model files are UTF-8; reading them with the platform default (cp1252 on Windows) breaks non-ASCII text.""" + +import json +from pathlib import Path + +import pytest +from py_rust_stemmers import SnowballStemmer + +from fastembed.common.preprocessor_utils import load_special_tokens +from fastembed.sparse.bm25 import Bm25 +from fastembed.sparse.utils.vocab_resolver import VocabResolver, VocabTokenizerBase + +# Stopwords from the files Qdrant/bm25 ships for these languages. Under cp1252 the German and +# French ones load as mojibake ("über") and the others raise UnicodeDecodeError. +NON_ASCII_STOPWORDS = { + "german": ["aber", "über", "würde"], + "french": ["à", "été", "au"], + "russian": ["и", "в", "во"], + "greek": ["αλλα", "αν", "αντι"], + "arabic": ["إذ", "إذا", "إذما"], +} + + +@pytest.mark.parametrize(("language", "stopwords"), NON_ASCII_STOPWORDS.items()) +def test_bm25_loads_non_ascii_stopwords( + tmp_path: Path, language: str, stopwords: list[str] +) -> None: + (tmp_path / f"{language}.txt").write_text("\n".join(stopwords), encoding="utf-8") + + assert Bm25._load_stopwords(tmp_path, language) == stopwords + + +def test_vocab_resolver_round_trips_non_ascii_words(tmp_path: Path) -> None: + words = ["café", "naïve", "über", "straße"] + resolver = VocabResolver(VocabTokenizerBase(), set(), SnowballStemmer("english")) + for word in words: + resolver.add_word(word) + + resolver.save_vocab(str(tmp_path / "vocab.txt")) + resolver.save_json_vocab(str(tmp_path / "vocab.json")) + assert (tmp_path / "vocab.txt").read_text(encoding="utf-8").splitlines() == words + + from_txt = VocabResolver(VocabTokenizerBase(), set(), SnowballStemmer("english")) + from_txt.load_vocab(str(tmp_path / "vocab.txt")) + from_json = VocabResolver(VocabTokenizerBase(), set(), SnowballStemmer("english")) + from_json.load_json_vocab(str(tmp_path / "vocab.json")) + + for loaded in (from_txt, from_json): + assert loaded.words == words + assert loaded.vocab == resolver.vocab + assert loaded.stem_mapping == resolver.stem_mapping + + +def test_vocab_resolver_reads_unescaped_utf8_json(tmp_path: Path) -> None: + # json.dump escapes non-ASCII by default, but a vocab file written elsewhere may not + path = tmp_path / "vocab.json" + path.write_text( + json.dumps({"vocab": ["café"], "stem_mapping": {"café": "café"}}, ensure_ascii=False), + encoding="utf-8", + ) + resolver = VocabResolver(VocabTokenizerBase(), set(), SnowballStemmer("english")) + + resolver.load_json_vocab(str(path)) + + assert resolver.vocab == {"café": 1} + + +def test_load_special_tokens_reads_utf8(tmp_path: Path) -> None: + # SentencePiece-based tokenizers use "▁" (U+2581), which cp1252 cannot decode + tokens_map = {"unk_token": "", "additional_special_tokens": ["▁"]} + (tmp_path / "special_tokens_map.json").write_text( + json.dumps(tokens_map, ensure_ascii=False), encoding="utf-8" + ) + + assert load_special_tokens(tmp_path) == tokens_map