From d9e5f9388a8dc131ead0447ce01976bba372b504 Mon Sep 17 00:00:00 2001 From: Pothan Date: Tue, 29 Sep 2026 00:31:02 -0500 Subject: [PATCH 1/3] Avoid full Hamming matrix when filling MUVERA clusters --- fastembed/postprocess/muvera.py | 26 +++++++++++++++----------- 1 file changed, 15 insertions(+), 11 deletions(-) diff --git a/fastembed/postprocess/muvera.py b/fastembed/postprocess/muvera.py index 806cd4ad..1ec5b622 100644 --- a/fastembed/postprocess/muvera.py +++ b/fastembed/postprocess/muvera.py @@ -286,9 +286,9 @@ def process( AssertionError: If input vectors don't have expected dimensionality ValueError: If the input multivector is empty """ - assert ( - vectors.shape[1] == self.dim - ), f"Expected vectors of shape (n, {self.dim}), got {vectors.shape}" + assert vectors.shape[1] == self.dim, ( + f"Expected vectors of shape (n, {self.dim}), got {vectors.shape}" + ) if len(vectors) == 0: raise ValueError("Cannot encode an empty multivector") @@ -299,9 +299,6 @@ def process( # num of space partitions in SimHash num_partitions = 2**self.k_sim cluster_center_ids = np.arange(num_partitions) - precomputed_hamming_matrix = ( - hamming_distance_matrix(cluster_center_ids) if fill_empty_clusters else None - ) for projection_index, simhash in enumerate(self.simhash_projections): # Initialize cluster centers and count vectors assigned to each cluster @@ -331,15 +328,22 @@ def process( # Fill empty clusters using vectors with minimum Hamming distance if fill_empty_clusters: assert empty_mask is not None - assert precomputed_hamming_matrix is not None - masked_hamming = np.where( - empty_mask[None, :], MAX_HAMMING_DISTANCE, precomputed_hamming_matrix + # Compare empty clusters only with occupied clusters. Both ID arrays + # are sorted, preserving the original argmin tie-breaking order. + occupied_ids = cluster_center_ids[~empty_mask] + empty_ids = cluster_center_ids[empty_mask] + distances = np.bitwise_xor(empty_ids[:, None], occupied_ids[None, :]) + bytes_view = ( + distances.astype(np.uint64) + .view(np.uint8) + .reshape(len(empty_ids), len(occupied_ids), 8) ) - nearest_non_empty = np.argmin(masked_hamming, axis=1) + hamming = POPCOUNT_LUT[bytes_view].sum(axis=2) + nearest_non_empty = occupied_ids[np.argmin(hamming, axis=1)] fill_vectors = np.array( [ vectors[cluster_center_id_to_vectors[cluster_id][0]] - for cluster_id in nearest_non_empty[empty_mask] + for cluster_id in nearest_non_empty ] ).reshape(-1, self.dim) cluster_centers[empty_mask] = fill_vectors From 7adee174c2c497e9cda88667483680a9f44972a1 Mon Sep 17 00:00:00 2001 From: Pothan Date: Tue, 29 Sep 2026 00:31:36 -0500 Subject: [PATCH 2/3] Test MUVERA nearest occupied cluster filling Cover fixed SimHash assignments and nearest-cluster choices against the full Hamming matrix reference. --- tests/test_postprocess.py | 28 ++++++++++++++++++++++++++++ 1 file changed, 28 insertions(+) diff --git a/tests/test_postprocess.py b/tests/test_postprocess.py index 3dc80d41..ae1d51a5 100644 --- a/tests/test_postprocess.py +++ b/tests/test_postprocess.py @@ -49,3 +49,31 @@ def test_empty_multivectors_raise_value_error(): with pytest.raises(ValueError, match="Cannot encode an empty multivector"): muvera.process_query(empty) + +def test_muvera_fills_from_nearest_occupied_cluster(): + muvera = Muvera(dim=2, k_sim=2, dim_proj=2, r_reps=1) + muvera.simhash_projections[0].get_cluster_ids = lambda vectors: np.array([0, 3]) + vectors = np.array([[1.0, 2.0], [3.0, 4.0]]) + np.testing.assert_array_equal( + muvera.process_document(vectors).reshape(4, 2), + [vectors[0], vectors[0], vectors[0], vectors[1]], + ) + + +@pytest.mark.parametrize("k_sim", [1, 2, 5, 8]) +@pytest.mark.parametrize("assignments", [[0], [0, 0, 0], [0, 1, 0, 1]]) +def test_muvera_fills_match_full_matrix_reference(k_sim, assignments): + from fastembed.postprocess.muvera import hamming_distance_matrix + + n = 2**k_sim + ids = np.array(assignments) % n + empty = np.bincount(ids, minlength=n) == 0 + full = hamming_distance_matrix(np.arange(n)) + full[:, empty] = 65 + expected_source_ids = np.argmin(full, axis=1)[empty] + vectors = np.arange(len(ids), dtype=np.float64)[:, None] + 1 + muvera = Muvera(dim=1, k_sim=k_sim, dim_proj=1, r_reps=1) + muvera.simhash_projections[0].get_cluster_ids = lambda vectors: ids + result = muvera.process_document(vectors).reshape(n, 1) + for empty_id, nearest in zip(np.flatnonzero(empty), expected_source_ids): + assert result[empty_id, 0] == vectors[np.flatnonzero(ids == nearest)[0], 0] From cee518fd89b71ced2d951e226f665e0dec6923c5 Mon Sep 17 00:00:00 2001 From: George Panchuk Date: Sun, 4 Oct 2026 03:25:26 +0700 Subject: [PATCH 3/3] perf: look up MUVERA cluster distances in a precomputed popcount table --- fastembed/postprocess/muvera.py | 20 ++++++++++---------- tests/test_postprocess.py | 12 +++++------- 2 files changed, 15 insertions(+), 17 deletions(-) diff --git a/fastembed/postprocess/muvera.py b/fastembed/postprocess/muvera.py index 1ec5b622..31d3ad36 100644 --- a/fastembed/postprocess/muvera.py +++ b/fastembed/postprocess/muvera.py @@ -145,6 +145,12 @@ def __init__( ] # Random projection matrices with entries from {-1, +1} for each repetition self.dim_reduction_projections = generator.choice([-1, 1], size=(r_reps, dim, dim_proj)) + # Hamming distance between two cluster ids is the popcount of their XOR, which is + # itself a cluster id, so per-id popcounts are enough to get any pairwise distance + cluster_ids = np.arange(2**k_sim, dtype=np.uint64) + self._cluster_id_popcounts = POPCOUNT_LUT[cluster_ids.view(np.uint8).reshape(-1, 8)].sum( + axis=1 + ) @classmethod def from_multivector_model( @@ -286,9 +292,9 @@ def process( AssertionError: If input vectors don't have expected dimensionality ValueError: If the input multivector is empty """ - assert vectors.shape[1] == self.dim, ( - f"Expected vectors of shape (n, {self.dim}), got {vectors.shape}" - ) + assert ( + vectors.shape[1] == self.dim + ), f"Expected vectors of shape (n, {self.dim}), got {vectors.shape}" if len(vectors) == 0: raise ValueError("Cannot encode an empty multivector") @@ -332,13 +338,7 @@ def process( # are sorted, preserving the original argmin tie-breaking order. occupied_ids = cluster_center_ids[~empty_mask] empty_ids = cluster_center_ids[empty_mask] - distances = np.bitwise_xor(empty_ids[:, None], occupied_ids[None, :]) - bytes_view = ( - distances.astype(np.uint64) - .view(np.uint8) - .reshape(len(empty_ids), len(occupied_ids), 8) - ) - hamming = POPCOUNT_LUT[bytes_view].sum(axis=2) + hamming = self._cluster_id_popcounts[empty_ids[:, None] ^ occupied_ids[None, :]] nearest_non_empty = occupied_ids[np.argmin(hamming, axis=1)] fill_vectors = np.array( [ diff --git a/tests/test_postprocess.py b/tests/test_postprocess.py index ae1d51a5..92f485f1 100644 --- a/tests/test_postprocess.py +++ b/tests/test_postprocess.py @@ -3,6 +3,7 @@ from fastembed import LateInteractionTextEmbedding from fastembed.postprocess import Muvera +from fastembed.postprocess.muvera import MAX_HAMMING_DISTANCE, hamming_distance_matrix CANONICAL_VALUES = [-2.61810007e-04, 1.89005750e00, -2.32070747e00] CANONICAL_QUERY_VALUES = [ @@ -60,16 +61,13 @@ def test_muvera_fills_from_nearest_occupied_cluster(): ) -@pytest.mark.parametrize("k_sim", [1, 2, 5, 8]) -@pytest.mark.parametrize("assignments", [[0], [0, 0, 0], [0, 1, 0, 1]]) -def test_muvera_fills_match_full_matrix_reference(k_sim, assignments): - from fastembed.postprocess.muvera import hamming_distance_matrix - +@pytest.mark.parametrize("k_sim", [1, 5, 8]) +def test_muvera_fills_match_full_matrix_reference(k_sim): n = 2**k_sim - ids = np.array(assignments) % n + ids = np.random.default_rng(0).integers(0, 256, size=20) % n empty = np.bincount(ids, minlength=n) == 0 full = hamming_distance_matrix(np.arange(n)) - full[:, empty] = 65 + full[:, empty] = MAX_HAMMING_DISTANCE expected_source_ids = np.argmin(full, axis=1)[empty] vectors = np.arange(len(ids), dtype=np.float64)[:, None] + 1 muvera = Muvera(dim=1, k_sim=k_sim, dim_proj=1, r_reps=1)