Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
8 changes: 4 additions & 4 deletions fastembed/common/preprocessor_utils.py
Original file line number Diff line number Diff line change
Expand Up @@ -14,7 +14,7 @@ def load_special_tokens(model_dir: Path) -> dict[str, Any]:
if not tokens_map_path.exists():
return {}

with open(str(tokens_map_path)) as tokens_map_file:
with open(str(tokens_map_path), encoding="utf-8") as tokens_map_file:
tokens_map = json.load(tokens_map_file)

return tokens_map
Expand Down Expand Up @@ -84,10 +84,10 @@ def load_tokenizer(model_dir: Path) -> tuple[Tokenizer, dict[str, int]]:
config_path = model_dir / "config.json"
config: dict[str, Any] = {}
if config_path.exists():
with open(str(config_path)) as config_file:
with open(str(config_path), encoding="utf-8") as config_file:
config = json.load(config_file)

with open(str(tokenizer_config_path)) as tokenizer_config_file:
with open(str(tokenizer_config_path), encoding="utf-8") as tokenizer_config_file:
tokenizer_config = json.load(tokenizer_config_file)

max_context = _resolve_max_context(tokenizer_config, model_dir)
Expand Down Expand Up @@ -153,7 +153,7 @@ def load_preprocessor(model_dir: Path) -> Compose:
if not preprocessor_config_path.exists():
raise ValueError(f"Could not find preprocessor_config.json in {model_dir}")

with open(str(preprocessor_config_path)) as preprocessor_config_file:
with open(str(preprocessor_config_path), encoding="utf-8") as preprocessor_config_file:
preprocessor_config = json.load(preprocessor_config_file)
transforms = Compose.from_config(preprocessor_config)
return transforms
6 changes: 3 additions & 3 deletions fastembed/late_interaction_multimodal/colmodernvbert.py
Original file line number Diff line number Diff line change
Expand Up @@ -138,20 +138,20 @@ def load_onnx_model(self) -> None:

# Load image processing configuration
processor_config_path = self._model_dir / "processor_config.json"
with open(processor_config_path) as f:
with open(processor_config_path, encoding="utf-8") as f:
processor_config = json.load(f)
self.image_seq_len = processor_config.get("image_seq_len", 64)

preprocessor_config_path = self._model_dir / "preprocessor_config.json"
with open(preprocessor_config_path) as f:
with open(preprocessor_config_path, encoding="utf-8") as f:
preprocessor_config = json.load(f)
self.max_image_size = preprocessor_config.get("max_image_size", {}).get(
"longest_edge", 512
)

# Load model configuration
config_path = self._model_dir / "config.json"
with open(config_path) as f:
with open(config_path, encoding="utf-8") as f:
model_config = json.load(f)
vision_config = model_config.get("vision_config", {})
self.image_size = vision_config.get("image_size", 512)
Expand Down
2 changes: 1 addition & 1 deletion fastembed/sparse/bm25.py
Original file line number Diff line number Diff line change
Expand Up @@ -198,7 +198,7 @@ def _load_stopwords(cls, model_dir: Path, language: str) -> list[str]:
if not stopwords_path.exists():
return []

with open(stopwords_path, "r") as f:
with open(stopwords_path, "r", encoding="utf-8") as f:
return f.read().splitlines()

def _embed_documents(
Expand Down
2 changes: 1 addition & 1 deletion fastembed/sparse/bm42.py
Original file line number Diff line number Diff line change
Expand Up @@ -290,7 +290,7 @@ def _load_stopwords(cls, model_dir: Path) -> list[str]:
if not stopwords_path.exists():
return []

with open(stopwords_path, "r") as f:
with open(stopwords_path, "r", encoding="utf-8") as f:
return f.read().splitlines()

def embed(
Expand Down
2 changes: 1 addition & 1 deletion fastembed/sparse/if_splade.py
Original file line number Diff line number Diff line change
Expand Up @@ -164,7 +164,7 @@ def load_onnx_model(self) -> None:
)

def _load_idf(self) -> dict[int, float]:
with open(self._model_dir / IDF_FILE) as f:
with open(self._model_dir / IDF_FILE, encoding="utf-8") as f:
token_to_idf: dict[str, float] = json.load(f)

vocab: dict[str, int] = self.tokenizer.get_vocab() # type: ignore[union-attr]
Expand Down
2 changes: 1 addition & 1 deletion fastembed/sparse/minicoil.py
Original file line number Diff line number Diff line change
Expand Up @@ -275,7 +275,7 @@ def _load_stopwords(cls, model_dir: Path) -> list[str]:
if not stopwords_path.exists():
return []

with open(stopwords_path, "r") as f:
with open(stopwords_path, "r", encoding="utf-8") as f:
return f.read().splitlines()

@classmethod
Expand Down
8 changes: 4 additions & 4 deletions fastembed/sparse/utils/vocab_resolver.py
Original file line number Diff line number Diff line change
Expand Up @@ -64,20 +64,20 @@ def vocab_size(self) -> int:
return len(self.vocab) + 1

def save_vocab(self, path: str) -> None:
with open(path, "w") as f:
with open(path, "w", encoding="utf-8") as f:
for word in self.words:
f.write(word + "\n")

def save_json_vocab(self, path: str) -> None:
import json

with open(path, "w") as f:
with open(path, "w", encoding="utf-8") as f:
json.dump({"vocab": self.words, "stem_mapping": self.stem_mapping}, f, indent=2)

def load_json_vocab(self, path: str) -> None:
import json

with open(path, "r") as f:
with open(path, "r", encoding="utf-8") as f:
data = json.load(f)
self.words = data["vocab"]
self.vocab = {word: idx + 1 for idx, word in enumerate(self.words)}
Expand All @@ -98,7 +98,7 @@ def add_word(self, word: str) -> None:
self.stem_mapping[stem] = word

def load_vocab(self, path: str) -> None:
with open(path, "r") as f:
with open(path, "r", encoding="utf-8") as f:
for line in f:
self.add_word(line.strip())

Expand Down
75 changes: 75 additions & 0 deletions tests/test_text_file_encoding.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,75 @@
"""Model files are UTF-8; reading them with the platform default (cp1252 on Windows) breaks non-ASCII text."""

import json
from pathlib import Path

import pytest
from py_rust_stemmers import SnowballStemmer

from fastembed.common.preprocessor_utils import load_special_tokens
from fastembed.sparse.bm25 import Bm25
from fastembed.sparse.utils.vocab_resolver import VocabResolver, VocabTokenizerBase

# Stopwords from the files Qdrant/bm25 ships for these languages. Under cp1252 the German and
# French ones load as mojibake ("über") and the others raise UnicodeDecodeError.
NON_ASCII_STOPWORDS = {
"german": ["aber", "über", "würde"],
"french": ["à", "été", "au"],
"russian": ["и", "в", "во"],
"greek": ["αλλα", "αν", "αντι"],
"arabic": ["إذ", "إذا", "إذما"],
}


@pytest.mark.parametrize(("language", "stopwords"), NON_ASCII_STOPWORDS.items())
def test_bm25_loads_non_ascii_stopwords(
tmp_path: Path, language: str, stopwords: list[str]
) -> None:
(tmp_path / f"{language}.txt").write_text("\n".join(stopwords), encoding="utf-8")

assert Bm25._load_stopwords(tmp_path, language) == stopwords


def test_vocab_resolver_round_trips_non_ascii_words(tmp_path: Path) -> None:
words = ["café", "naïve", "über", "straße"]
resolver = VocabResolver(VocabTokenizerBase(), set(), SnowballStemmer("english"))
for word in words:
resolver.add_word(word)

resolver.save_vocab(str(tmp_path / "vocab.txt"))
resolver.save_json_vocab(str(tmp_path / "vocab.json"))
assert (tmp_path / "vocab.txt").read_text(encoding="utf-8").splitlines() == words

from_txt = VocabResolver(VocabTokenizerBase(), set(), SnowballStemmer("english"))
from_txt.load_vocab(str(tmp_path / "vocab.txt"))
from_json = VocabResolver(VocabTokenizerBase(), set(), SnowballStemmer("english"))
from_json.load_json_vocab(str(tmp_path / "vocab.json"))

for loaded in (from_txt, from_json):
assert loaded.words == words
assert loaded.vocab == resolver.vocab
assert loaded.stem_mapping == resolver.stem_mapping


def test_vocab_resolver_reads_unescaped_utf8_json(tmp_path: Path) -> None:
# json.dump escapes non-ASCII by default, but a vocab file written elsewhere may not
path = tmp_path / "vocab.json"
path.write_text(
json.dumps({"vocab": ["café"], "stem_mapping": {"café": "café"}}, ensure_ascii=False),
encoding="utf-8",
)
resolver = VocabResolver(VocabTokenizerBase(), set(), SnowballStemmer("english"))

resolver.load_json_vocab(str(path))

assert resolver.vocab == {"café": 1}


def test_load_special_tokens_reads_utf8(tmp_path: Path) -> None:
# SentencePiece-based tokenizers use "▁" (U+2581), which cp1252 cannot decode
tokens_map = {"unk_token": "<unk>", "additional_special_tokens": ["▁<extra>"]}
(tmp_path / "special_tokens_map.json").write_text(
json.dumps(tokens_map, ensure_ascii=False), encoding="utf-8"
)

assert load_special_tokens(tmp_path) == tokens_map