From a6bb7f3d296ee0e312a492407af43be92a4a6ab7 Mon Sep 17 00:00:00 2001 From: George Date: Sun, 4 Oct 2026 03:34:41 +0700 Subject: [PATCH 1/4] new: add weekly ci to test all models --- .github/workflows/weekly-tests.yml | 46 +++++++++++++++++++++++ tests/test_image_onnx_embeddings.py | 4 +- tests/test_late_interaction_embeddings.py | 6 +-- tests/test_sparse_embeddings.py | 4 +- tests/test_text_cross_encoder.py | 4 +- tests/test_text_multitask_embeddings.py | 14 +++---- tests/test_text_onnx_embeddings.py | 6 +-- tests/utils.py | 7 ++++ 8 files changed, 72 insertions(+), 19 deletions(-) create mode 100644 .github/workflows/weekly-tests.yml diff --git a/.github/workflows/weekly-tests.yml b/.github/workflows/weekly-tests.yml new file mode 100644 index 000000000..d0c9a7d84 --- /dev/null +++ b/.github/workflows/weekly-tests.yml @@ -0,0 +1,46 @@ +name: Weekly tests + +# PR runs test one representative model per family, scheduled and manual runs test every model, +# see is_manual_run in tests/utils.py +on: + schedule: + - cron: '17 3 * * 1' # Mondays, 03:17 UTC + workflow_dispatch: + +permissions: + contents: read + +jobs: + test: + + strategy: + fail-fast: false + matrix: + python-version: + - '3.10.x' + - '3.14.x' + + # linux runners have 16 GB of RAM, enough for every model tested in ci (colpali is skipped in ci, + # gpu tests are skipped everywhere) + runs-on: ubuntu-latest + timeout-minutes: 180 + + name: Python ${{ matrix.python-version }} weekly test + + steps: + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + - name: Set up Python + uses: actions/setup-python@5fda3b95a4ea91299a34e894583c3862153e4b97 # v7.0.0 + with: + python-version: ${{ matrix.python-version }} + - name: Install dependencies + run: | + python -m pip install poetry + poetry config virtualenvs.create false + poetry install --no-interaction --no-ansi --without dev,docs + + - name: Run pytest + env: + HF_TOKEN: ${{ secrets.HF_TOKEN }} + run: | + poetry run pytest --durations=20 diff --git a/tests/test_image_onnx_embeddings.py b/tests/test_image_onnx_embeddings.py index 7774055bb..2e5288118 100644 --- a/tests/test_image_onnx_embeddings.py +++ b/tests/test_image_onnx_embeddings.py @@ -10,7 +10,7 @@ from fastembed import ImageEmbedding from tests.config import TEST_MISC_DIR -from tests.utils import delete_model_cache, should_test_model +from tests.utils import delete_model_cache, is_manual_run, should_test_model CANONICAL_VECTOR_VALUES = { "Qdrant/clip-ViT-B-32-vision": np.array([-0.0098, 0.0128, -0.0274, 0.002, -0.0059]), @@ -70,7 +70,7 @@ def get_model(model_name: str): def test_embedding(model_cache, model_name: str) -> None: is_ci = os.getenv("CI") is_mac = platform.system() == "Darwin" - is_manual = os.getenv("GITHUB_EVENT_NAME") == "workflow_dispatch" + is_manual = is_manual_run() for model_desc in ImageEmbedding._list_supported_models(): # quantized int8 ops diverge on macOS; canonical vector is generated on linux/amd64 (CI) diff --git a/tests/test_late_interaction_embeddings.py b/tests/test_late_interaction_embeddings.py index 0e433fce8..a9863aa71 100644 --- a/tests/test_late_interaction_embeddings.py +++ b/tests/test_late_interaction_embeddings.py @@ -7,7 +7,7 @@ from fastembed.late_interaction.late_interaction_text_embedding import ( LateInteractionTextEmbedding, ) -from tests.utils import delete_model_cache, should_test_model +from tests.utils import delete_model_cache, is_manual_run, should_test_model # vectors are abridged and rounded for brevity CANONICAL_COLUMN_VALUES = { @@ -211,7 +211,7 @@ def test_batch_inference_size_same_as_single_inference(model_cache, model_name: @pytest.mark.parametrize("model_name", ["answerdotai/answerai-colbert-small-v1"]) def test_single_embedding(model_cache, model_name: str): is_ci = os.getenv("CI") - is_manual = os.getenv("GITHUB_EVENT_NAME") == "workflow_dispatch" + is_manual = is_manual_run() docs_to_embed = docs for model_desc in LateInteractionTextEmbedding._list_supported_models(): @@ -231,7 +231,7 @@ def test_single_embedding(model_cache, model_name: str): @pytest.mark.parametrize("model_name", ["answerdotai/answerai-colbert-small-v1"]) def test_single_embedding_query(model_cache, model_name: str): is_ci = os.getenv("CI") - is_manual = os.getenv("GITHUB_EVENT_NAME") == "workflow_dispatch" + is_manual = is_manual_run() queries_to_embed = docs for model_desc in LateInteractionTextEmbedding._list_supported_models(): diff --git a/tests/test_sparse_embeddings.py b/tests/test_sparse_embeddings.py index 1a0e2df66..3aafbe32f 100644 --- a/tests/test_sparse_embeddings.py +++ b/tests/test_sparse_embeddings.py @@ -6,7 +6,7 @@ from fastembed.sparse.bm25 import Bm25 from fastembed.sparse.sparse_text_embedding import SparseTextEmbedding -from tests.utils import delete_model_cache, should_test_model +from tests.utils import delete_model_cache, is_manual_run, should_test_model CANONICAL_COLUMN_VALUES = { @@ -175,7 +175,7 @@ def test_batch_embedding(model_cache, model_name: str) -> None: def test_single_embedding(model_cache) -> None: is_ci = os.getenv("CI") - is_manual = os.getenv("GITHUB_EVENT_NAME") == "workflow_dispatch" + is_manual = is_manual_run() for model_desc in SparseTextEmbedding._list_supported_models(): if ( diff --git a/tests/test_text_cross_encoder.py b/tests/test_text_cross_encoder.py index a29fe9d12..85988ca8f 100644 --- a/tests/test_text_cross_encoder.py +++ b/tests/test_text_cross_encoder.py @@ -5,7 +5,7 @@ import pytest from fastembed.rerank.cross_encoder import TextCrossEncoder -from tests.utils import delete_model_cache, should_test_model +from tests.utils import delete_model_cache, is_manual_run, should_test_model CANONICAL_SCORE_VALUES = { "Xenova/ms-marco-MiniLM-L-6-v2": np.array([8.500708, -2.541011]), @@ -54,7 +54,7 @@ def get_model(model_name: str): @pytest.mark.parametrize("model_name", ["Xenova/ms-marco-MiniLM-L-6-v2"]) def test_rerank(model_cache, model_name: str) -> None: is_ci = os.getenv("CI") - is_manual = os.getenv("GITHUB_EVENT_NAME") == "workflow_dispatch" + is_manual = is_manual_run() for model_desc in TextCrossEncoder._list_supported_models(): if not should_test_model(model_desc, model_name, is_ci, is_manual): diff --git a/tests/test_text_multitask_embeddings.py b/tests/test_text_multitask_embeddings.py index 886a71251..fa12c24e3 100644 --- a/tests/test_text_multitask_embeddings.py +++ b/tests/test_text_multitask_embeddings.py @@ -5,7 +5,7 @@ from fastembed import TextEmbedding from fastembed.text.multitask_embedding import JinaEmbeddingV3, Task -from tests.utils import delete_model_cache +from tests.utils import delete_model_cache, is_manual_run CANONICAL_VECTOR_VALUES = { @@ -63,7 +63,7 @@ @pytest.mark.parametrize("dim,model_name", [(1024, "jinaai/jina-embeddings-v3")]) def test_batch_embedding(dim: int, model_name: str): is_ci = os.getenv("CI") - is_manual = os.getenv("GITHUB_EVENT_NAME") == "workflow_dispatch" + is_manual = is_manual_run() if is_ci and not is_manual: pytest.skip("Skipping multitask models in CI non-manual mode") @@ -88,7 +88,7 @@ def test_batch_embedding(dim: int, model_name: str): def test_single_embedding(): is_ci = os.getenv("CI") - is_manual = os.getenv("GITHUB_EVENT_NAME") == "workflow_dispatch" + is_manual = is_manual_run() if is_ci and not is_manual: pytest.skip("Skipping multitask models in CI non-manual mode") @@ -134,7 +134,7 @@ def test_single_embedding(): def test_single_embedding_query(): is_ci = os.getenv("CI") - is_manual = os.getenv("GITHUB_EVENT_NAME") == "workflow_dispatch" + is_manual = is_manual_run() if is_ci and not is_manual: pytest.skip("Skipping multitask models in CI non-manual mode") @@ -165,7 +165,7 @@ def test_single_embedding_query(): def test_single_embedding_passage(): is_ci = os.getenv("CI") - is_manual = os.getenv("GITHUB_EVENT_NAME") == "workflow_dispatch" + is_manual = is_manual_run() if is_ci and not is_manual: pytest.skip("Skipping multitask models in CI non-manual mode") @@ -198,7 +198,7 @@ def test_single_embedding_passage(): @pytest.mark.parametrize("dim,model_name", [(1024, "jinaai/jina-embeddings-v3")]) def test_parallel_processing(dim: int, model_name: str): is_ci = os.getenv("CI") - is_manual = os.getenv("GITHUB_EVENT_NAME") == "workflow_dispatch" + is_manual = is_manual_run() if is_ci and not is_manual: pytest.skip("Skipping in CI non-manual mode") @@ -226,7 +226,7 @@ def test_parallel_processing(dim: int, model_name: str): @pytest.mark.parametrize("model_name", ["jinaai/jina-embeddings-v3"]) def test_lazy_load(model_name: str): is_ci = os.getenv("CI") - is_manual = os.getenv("GITHUB_EVENT_NAME") == "workflow_dispatch" + is_manual = is_manual_run() if is_ci and not is_manual: pytest.skip("Skipping in CI non-manual mode") diff --git a/tests/test_text_onnx_embeddings.py b/tests/test_text_onnx_embeddings.py index 5110a3009..632bb70cd 100644 --- a/tests/test_text_onnx_embeddings.py +++ b/tests/test_text_onnx_embeddings.py @@ -10,7 +10,7 @@ from fastembed.text.last_token_normalized_embedding import LastTokenNormalizedEmbedding from fastembed.text.onnx_embedding import OnnxTextEmbedding from fastembed.text.text_embedding import TextEmbedding -from tests.utils import delete_model_cache, should_test_model +from tests.utils import delete_model_cache, is_manual_run, should_test_model CANONICAL_VECTOR_VALUES = { "BAAI/bge-small-en": np.array([-0.0232, -0.0255, 0.0174, -0.0639, -0.0006]), @@ -166,7 +166,7 @@ def get_model(model_name: str): def test_embedding(model_cache, model_name: str) -> None: is_ci = os.getenv("CI") is_mac = platform.system() == "Darwin" - is_manual = os.getenv("GITHUB_EVENT_NAME") == "workflow_dispatch" + is_manual = is_manual_run() for model_desc in TextEmbedding._list_supported_models(): if model_desc.model in MULTI_TASK_MODELS or ( @@ -196,7 +196,7 @@ def test_embedding(model_cache, model_name: str) -> None: def test_query_embedding(model_cache) -> None: is_ci = os.getenv("CI") is_mac = platform.system() == "Darwin" - is_manual = os.getenv("GITHUB_EVENT_NAME") == "workflow_dispatch" + is_manual = is_manual_run() for model_desc in TextEmbedding._list_supported_models(): if model_desc.model in MULTI_TASK_MODELS or ( diff --git a/tests/utils.py b/tests/utils.py index 481c05045..181c3a0ff 100644 --- a/tests/utils.py +++ b/tests/utils.py @@ -1,3 +1,4 @@ +import os import shutil import traceback @@ -39,6 +40,11 @@ def on_error( shutil.rmtree(model_dir, onerror=on_error) +def is_manual_run() -> bool: + """Whether ci runs the heavyweight tests: on a manual dispatch or on the weekly schedule""" + return os.getenv("GITHUB_EVENT_NAME") in ("workflow_dispatch", "schedule") + + def should_test_model( model_desc: BaseModelDescription, autotest_model_name: str, @@ -55,6 +61,7 @@ def should_test_model( 2) Run heavyweight (manual) tests in ci: - test all models Running tests in ci each time is too expensive, however, it's fine to run it one time with a manual dispatch + or weekly on a schedule 3) Run tests locally: - test all models, which are not too heavy, since network speed might be a bottleneck From 1bfc0f274b50c072357c8f730bab564ad2e3dcd9 Mon Sep 17 00:00:00 2001 From: George Date: Sun, 4 Oct 2026 03:36:53 +0700 Subject: [PATCH 2/4] ci: drop python3.10 from weekly tests --- .github/workflows/weekly-tests.yml | 11 ++--------- 1 file changed, 2 insertions(+), 9 deletions(-) diff --git a/.github/workflows/weekly-tests.yml b/.github/workflows/weekly-tests.yml index d0c9a7d84..b93051517 100644 --- a/.github/workflows/weekly-tests.yml +++ b/.github/workflows/weekly-tests.yml @@ -13,26 +13,19 @@ permissions: jobs: test: - strategy: - fail-fast: false - matrix: - python-version: - - '3.10.x' - - '3.14.x' - # linux runners have 16 GB of RAM, enough for every model tested in ci (colpali is skipped in ci, # gpu tests are skipped everywhere) runs-on: ubuntu-latest timeout-minutes: 180 - name: Python ${{ matrix.python-version }} weekly test + name: Python 3.14 weekly test steps: - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 - name: Set up Python uses: actions/setup-python@5fda3b95a4ea91299a34e894583c3862153e4b97 # v7.0.0 with: - python-version: ${{ matrix.python-version }} + python-version: '3.14.x' - name: Install dependencies run: | python -m pip install poetry From 8ae050903caf77b58c4ec7f48870f0de783b9c13 Mon Sep 17 00:00:00 2001 From: George Panchuk Date: Mon, 5 Oct 2026 21:26:01 +0700 Subject: [PATCH 3/4] ci: change weekly tests time --- .github/workflows/weekly-tests.yml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/.github/workflows/weekly-tests.yml b/.github/workflows/weekly-tests.yml index b93051517..d4c8eaa4a 100644 --- a/.github/workflows/weekly-tests.yml +++ b/.github/workflows/weekly-tests.yml @@ -4,7 +4,7 @@ name: Weekly tests # see is_manual_run in tests/utils.py on: schedule: - - cron: '17 3 * * 1' # Mondays, 03:17 UTC + - cron: '5 9 * * 1' # Mondays, 09:05 UTC workflow_dispatch: permissions: From 57ece03c1f0c9a0453088b287ee50a0f394f4cdb Mon Sep 17 00:00:00 2001 From: George Panchuk Date: Mon, 5 Oct 2026 21:29:27 +0700 Subject: [PATCH 4/4] fix: address security comment --- .github/workflows/weekly-tests.yml | 2 ++ 1 file changed, 2 insertions(+) diff --git a/.github/workflows/weekly-tests.yml b/.github/workflows/weekly-tests.yml index d4c8eaa4a..f84900b41 100644 --- a/.github/workflows/weekly-tests.yml +++ b/.github/workflows/weekly-tests.yml @@ -22,6 +22,8 @@ jobs: steps: - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + persist-credentials: false - name: Set up Python uses: actions/setup-python@5fda3b95a4ea91299a34e894583c3862153e4b97 # v7.0.0 with: