diff --git a/fastembed/common/preprocessor_utils.py b/fastembed/common/preprocessor_utils.py index fb853924..3b8e013c 100644 --- a/fastembed/common/preprocessor_utils.py +++ b/fastembed/common/preprocessor_utils.py @@ -14,7 +14,7 @@ def load_special_tokens(model_dir: Path) -> dict[str, Any]: if not tokens_map_path.exists(): return {} - with open(str(tokens_map_path)) as tokens_map_file: + with open(str(tokens_map_path), encoding="utf-8") as tokens_map_file: tokens_map = json.load(tokens_map_file) return tokens_map @@ -84,10 +84,10 @@ def load_tokenizer(model_dir: Path) -> tuple[Tokenizer, dict[str, int]]: config_path = model_dir / "config.json" config: dict[str, Any] = {} if config_path.exists(): - with open(str(config_path)) as config_file: + with open(str(config_path), encoding="utf-8") as config_file: config = json.load(config_file) - with open(str(tokenizer_config_path)) as tokenizer_config_file: + with open(str(tokenizer_config_path), encoding="utf-8") as tokenizer_config_file: tokenizer_config = json.load(tokenizer_config_file) max_context = _resolve_max_context(tokenizer_config, model_dir) @@ -153,7 +153,7 @@ def load_preprocessor(model_dir: Path) -> Compose: if not preprocessor_config_path.exists(): raise ValueError(f"Could not find preprocessor_config.json in {model_dir}") - with open(str(preprocessor_config_path)) as preprocessor_config_file: + with open(str(preprocessor_config_path), encoding="utf-8") as preprocessor_config_file: preprocessor_config = json.load(preprocessor_config_file) transforms = Compose.from_config(preprocessor_config) return transforms diff --git a/fastembed/late_interaction_multimodal/colmodernvbert.py b/fastembed/late_interaction_multimodal/colmodernvbert.py index f8fe0154..99f33217 100644 --- a/fastembed/late_interaction_multimodal/colmodernvbert.py +++ b/fastembed/late_interaction_multimodal/colmodernvbert.py @@ -138,12 +138,12 @@ def load_onnx_model(self) -> None: # Load image processing configuration processor_config_path = self._model_dir / "processor_config.json" - with open(processor_config_path) as f: + with open(processor_config_path, encoding="utf-8") as f: processor_config = json.load(f) self.image_seq_len = processor_config.get("image_seq_len", 64) preprocessor_config_path = self._model_dir / "preprocessor_config.json" - with open(preprocessor_config_path) as f: + with open(preprocessor_config_path, encoding="utf-8") as f: preprocessor_config = json.load(f) self.max_image_size = preprocessor_config.get("max_image_size", {}).get( "longest_edge", 512 @@ -151,7 +151,7 @@ def load_onnx_model(self) -> None: # Load model configuration config_path = self._model_dir / "config.json" - with open(config_path) as f: + with open(config_path, encoding="utf-8") as f: model_config = json.load(f) vision_config = model_config.get("vision_config", {}) self.image_size = vision_config.get("image_size", 512) diff --git a/fastembed/sparse/bm25.py b/fastembed/sparse/bm25.py index e095c627..e79b4e96 100644 --- a/fastembed/sparse/bm25.py +++ b/fastembed/sparse/bm25.py @@ -198,7 +198,7 @@ def _load_stopwords(cls, model_dir: Path, language: str) -> list[str]: if not stopwords_path.exists(): return [] - with open(stopwords_path, "r") as f: + with open(stopwords_path, "r", encoding="utf-8") as f: return f.read().splitlines() def _embed_documents( diff --git a/fastembed/sparse/bm42.py b/fastembed/sparse/bm42.py index b4685a88..dc3d0295 100644 --- a/fastembed/sparse/bm42.py +++ b/fastembed/sparse/bm42.py @@ -290,7 +290,7 @@ def _load_stopwords(cls, model_dir: Path) -> list[str]: if not stopwords_path.exists(): return [] - with open(stopwords_path, "r") as f: + with open(stopwords_path, "r", encoding="utf-8") as f: return f.read().splitlines() def embed( diff --git a/fastembed/sparse/if_splade.py b/fastembed/sparse/if_splade.py index 4e1c5bda..25374bca 100644 --- a/fastembed/sparse/if_splade.py +++ b/fastembed/sparse/if_splade.py @@ -164,7 +164,7 @@ def load_onnx_model(self) -> None: ) def _load_idf(self) -> dict[int, float]: - with open(self._model_dir / IDF_FILE) as f: + with open(self._model_dir / IDF_FILE, encoding="utf-8") as f: token_to_idf: dict[str, float] = json.load(f) vocab: dict[str, int] = self.tokenizer.get_vocab() # type: ignore[union-attr] diff --git a/fastembed/sparse/minicoil.py b/fastembed/sparse/minicoil.py index 0aabad7d..b6b85575 100644 --- a/fastembed/sparse/minicoil.py +++ b/fastembed/sparse/minicoil.py @@ -275,7 +275,7 @@ def _load_stopwords(cls, model_dir: Path) -> list[str]: if not stopwords_path.exists(): return [] - with open(stopwords_path, "r") as f: + with open(stopwords_path, "r", encoding="utf-8") as f: return f.read().splitlines() @classmethod diff --git a/fastembed/sparse/utils/vocab_resolver.py b/fastembed/sparse/utils/vocab_resolver.py index 54e037a7..37d4a818 100644 --- a/fastembed/sparse/utils/vocab_resolver.py +++ b/fastembed/sparse/utils/vocab_resolver.py @@ -64,20 +64,20 @@ def vocab_size(self) -> int: return len(self.vocab) + 1 def save_vocab(self, path: str) -> None: - with open(path, "w") as f: + with open(path, "w", encoding="utf-8") as f: for word in self.words: f.write(word + "\n") def save_json_vocab(self, path: str) -> None: import json - with open(path, "w") as f: + with open(path, "w", encoding="utf-8") as f: json.dump({"vocab": self.words, "stem_mapping": self.stem_mapping}, f, indent=2) def load_json_vocab(self, path: str) -> None: import json - with open(path, "r") as f: + with open(path, "r", encoding="utf-8") as f: data = json.load(f) self.words = data["vocab"] self.vocab = {word: idx + 1 for idx, word in enumerate(self.words)} @@ -98,7 +98,7 @@ def add_word(self, word: str) -> None: self.stem_mapping[stem] = word def load_vocab(self, path: str) -> None: - with open(path, "r") as f: + with open(path, "r", encoding="utf-8") as f: for line in f: self.add_word(line.strip()) diff --git a/tests/test_text_file_encoding.py b/tests/test_text_file_encoding.py new file mode 100644 index 00000000..6874874f --- /dev/null +++ b/tests/test_text_file_encoding.py @@ -0,0 +1,75 @@ +"""Model files are UTF-8; reading them with the platform default (cp1252 on Windows) breaks non-ASCII text.""" + +import json +from pathlib import Path + +import pytest +from py_rust_stemmers import SnowballStemmer + +from fastembed.common.preprocessor_utils import load_special_tokens +from fastembed.sparse.bm25 import Bm25 +from fastembed.sparse.utils.vocab_resolver import VocabResolver, VocabTokenizerBase + +# Stopwords from the files Qdrant/bm25 ships for these languages. Under cp1252 the German and +# French ones load as mojibake ("über") and the others raise UnicodeDecodeError. +NON_ASCII_STOPWORDS = { + "german": ["aber", "über", "würde"], + "french": ["à", "été", "au"], + "russian": ["и", "в", "во"], + "greek": ["αλλα", "αν", "αντι"], + "arabic": ["إذ", "إذا", "إذما"], +} + + +@pytest.mark.parametrize(("language", "stopwords"), NON_ASCII_STOPWORDS.items()) +def test_bm25_loads_non_ascii_stopwords( + tmp_path: Path, language: str, stopwords: list[str] +) -> None: + (tmp_path / f"{language}.txt").write_text("\n".join(stopwords), encoding="utf-8") + + assert Bm25._load_stopwords(tmp_path, language) == stopwords + + +def test_vocab_resolver_round_trips_non_ascii_words(tmp_path: Path) -> None: + words = ["café", "naïve", "über", "straße"] + resolver = VocabResolver(VocabTokenizerBase(), set(), SnowballStemmer("english")) + for word in words: + resolver.add_word(word) + + resolver.save_vocab(str(tmp_path / "vocab.txt")) + resolver.save_json_vocab(str(tmp_path / "vocab.json")) + assert (tmp_path / "vocab.txt").read_text(encoding="utf-8").splitlines() == words + + from_txt = VocabResolver(VocabTokenizerBase(), set(), SnowballStemmer("english")) + from_txt.load_vocab(str(tmp_path / "vocab.txt")) + from_json = VocabResolver(VocabTokenizerBase(), set(), SnowballStemmer("english")) + from_json.load_json_vocab(str(tmp_path / "vocab.json")) + + for loaded in (from_txt, from_json): + assert loaded.words == words + assert loaded.vocab == resolver.vocab + assert loaded.stem_mapping == resolver.stem_mapping + + +def test_vocab_resolver_reads_unescaped_utf8_json(tmp_path: Path) -> None: + # json.dump escapes non-ASCII by default, but a vocab file written elsewhere may not + path = tmp_path / "vocab.json" + path.write_text( + json.dumps({"vocab": ["café"], "stem_mapping": {"café": "café"}}, ensure_ascii=False), + encoding="utf-8", + ) + resolver = VocabResolver(VocabTokenizerBase(), set(), SnowballStemmer("english")) + + resolver.load_json_vocab(str(path)) + + assert resolver.vocab == {"café": 1} + + +def test_load_special_tokens_reads_utf8(tmp_path: Path) -> None: + # SentencePiece-based tokenizers use "▁" (U+2581), which cp1252 cannot decode + tokens_map = {"unk_token": "", "additional_special_tokens": ["▁"]} + (tmp_path / "special_tokens_map.json").write_text( + json.dumps(tokens_map, ensure_ascii=False), encoding="utf-8" + ) + + assert load_special_tokens(tmp_path) == tokens_map