From ae6a88b4438fbf6f823684221dbd26c5e70e39c7 Mon Sep 17 00:00:00 2001 From: Majd Abdallah Date: Sat, 19 Sep 2026 21:25:47 +0200 Subject: [PATCH 1/2] chore: prepare v0.9.1 release --- CHANGELOG.md | 22 ++++++++++++++++++++++ README.md | 4 ++-- docs/index.md | 2 +- pyproject.toml | 2 +- src/trialmatchai/__init__.py | 2 +- uv.lock | 2 +- 6 files changed, 28 insertions(+), 6 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 0c21fee2..b8fbc1df 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -6,7 +6,22 @@ All notable changes to TrialMatchAI are documented here. The format follows ## [Unreleased] +## [0.9.1] — 2026-09-19 + +### Added +- A checksum-pinned, CPU-only `reproduce-paper` command audits the published + TREC 2021/2022 archive against official qrels and separates stored metrics, + ranking recalculation, retrieval recall, and current-evaluator results. +- `trec-evaluate` reuses completed rankings to compare explicit unjudged-trial + policies, with macro means, medians, per-topic metrics, qrels checksums, and + evaluation-input fingerprints. +- Reproducible reports record both unjudged policies for six complete local + configurations across all 125 TREC 2021/2022 topics. + ### Fixed +- Paper archive caches and fresh extractions must contain every summary, + per-topic metric, ranking, candidate list, and recall-cutoff file consumed by + the audit. - Eligibility assessment is controlled by `rag.enabled` independently of the CoT prompt setting. Disabling `use_cot_reasoning` now selects direct JSON assessment. - Ranked output records assessment controls and availability. Reports explicitly @@ -19,6 +34,13 @@ All notable changes to TrialMatchAI are documented here. The format follows - The flowchart places eligibility assessment on the default path and labels the explicit retrieval-only bypass. Both SVGs state its default-on behavior. +### Changed +- The TREC package imports its GPU runner lazily so artifact and metric audits do + not load the model stack. +- Paper-reproduction output reports current tie-aware nDCG under both supported + unjudged-trial policies while retaining the previous condensed result field for + compatibility. + ### Migration - Set `rag.enabled: false` to skip assessment. `use_cot_reasoning: false` alone no longer disables it. No new GPU or clinical-accuracy qualification is claimed. diff --git a/README.md b/README.md index b5205c49..2841457d 100644 --- a/README.md +++ b/README.md @@ -32,7 +32,7 @@ it does not install the optional model stack. ```bash python3.11 -m venv .venv source .venv/bin/activate -python -m pip install trialmatchai==0.9.0 +python -m pip install trialmatchai==0.9.1 trialmatchai --version trialmatchai demo --workdir ./demo-workspace @@ -127,7 +127,7 @@ hardware you select; there is no universal single-GPU capacity promise. For the default model pipeline, install its extras in a dedicated environment: ```bash -python -m pip install 'trialmatchai[llm,gpu,entity]==0.9.0' +python -m pip install 'trialmatchai[llm,gpu,entity]==0.9.1' ``` The optional inference stack has unresolved dependency advisories and has not diff --git a/docs/index.md b/docs/index.md index e1d46529..15aff826 100644 --- a/docs/index.md +++ b/docs/index.md @@ -18,7 +18,7 @@ criterion-level eligibility assessment. Query expansion is separately configurab Install with Python 3.11 in an activated virtual environment: ```bash -python -m pip install trialmatchai==0.9.0 +python -m pip install trialmatchai==0.9.1 trialmatchai demo --workdir ./demo-workspace trialmatchai demo --workdir ./demo-workspace --resume ``` diff --git a/pyproject.toml b/pyproject.toml index ae64cee8..fcbf2192 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -5,7 +5,7 @@ build-backend = "setuptools.build_meta" [project] name = "trialmatchai" -version = "0.9.0" +version = "0.9.1" description = "AI-driven patient-to-clinical-trial matching: hybrid retrieval + LLM eligibility reasoning." readme = "README.md" requires-python = ">=3.11,<3.12" diff --git a/src/trialmatchai/__init__.py b/src/trialmatchai/__init__.py index 3f0fbb08..35dc3cf5 100644 --- a/src/trialmatchai/__init__.py +++ b/src/trialmatchai/__init__.py @@ -1,3 +1,3 @@ from __future__ import annotations -__version__ = "0.9.0" +__version__ = "0.9.1" diff --git a/uv.lock b/uv.lock index 2b0770a2..d40bf45b 100644 --- a/uv.lock +++ b/uv.lock @@ -3080,7 +3080,7 @@ wheels = [ [[package]] name = "trialmatchai" -version = "0.9.0" +version = "0.9.1" source = { editable = "." } dependencies = [ { name = "lancedb", marker = "(platform_machine == 'x86_64' and sys_platform == 'linux') or sys_platform == 'darwin'" }, From 0e1f6ad565e518626a81c37740dc130a8042cbd8 Mon Sep 17 00:00:00 2001 From: Majd Abdallah Date: Sat, 19 Sep 2026 21:28:56 +0200 Subject: [PATCH 2/2] docs: mark assessment controls released --- README.md | 4 ++-- docs/pipeline.md | 2 +- 2 files changed, 3 insertions(+), 3 deletions(-) diff --git a/README.md b/README.md index 2841457d..10507053 100644 --- a/README.md +++ b/README.md @@ -78,8 +78,8 @@ The code calls its generated eligibility explanations **CoT reasoning**; these are model outputs to inspect alongside source evidence, not verified clinical reasoning or a guarantee that every criterion has been covered. -In source checkouts after 0.9.0, eligibility assessment and CoT prompt style have -independent switches (the change is currently unreleased): +Starting with 0.9.1, eligibility assessment and CoT prompt style have independent +switches: | `rag.enabled` | `use_cot_reasoning` | Result | | :--- | :--- | :--- | diff --git a/docs/pipeline.md b/docs/pipeline.md index aeef5707..298b9411 100644 --- a/docs/pipeline.md +++ b/docs/pipeline.md @@ -102,7 +102,7 @@ See the [API reference](api.md) for `StageContext`, `Stage`, `select_stages`, an ## Assessment modes -This section describes the unreleased changes on `main` after 0.9.0. +This section describes the assessment behavior introduced in 0.9.1. Eligibility assessment is enabled by default. `rag.enabled` controls whether it runs; `use_cot_reasoning` selects the CoT prompt (`true`) or direct JSON prompt