diff --git a/.gitignore b/.gitignore
index 2ef7dde1..40d569c1 100644
--- a/.gitignore
+++ b/.gitignore
@@ -10,3 +10,5 @@ null/
.nf-test/
.nf-test.log
.nf-test-*
+.vscode/
+.mypy_cache/
diff --git a/.nf-core.yml b/.nf-core.yml
index 352e6ed3..01c11a32 100644
--- a/.nf-core.yml
+++ b/.nf-core.yml
@@ -10,6 +10,11 @@ lint:
- docs/images/nf-core-spatialaxe_logo_dark.png
- docs/images/nf-core-spatialaxe_logo_light.png
- .github/PULL_REQUEST_TEMPLATE.md
+ # The QC Quarto notebooks use doubled braces as Python f-string escapes (they
+ # emit pandoc callout-note divs) — not Jinja template strings.
+ template_strings:
+ - assets/notebooks/xenium_image_qc_report.qmd
+ - assets/notebooks/transcript_qc.qmd
nf_core_version: 4.0.3
repository_type: pipeline
template:
diff --git a/.vscode/settings.json b/.vscode/settings.json
deleted file mode 100644
index a33b527c..00000000
--- a/.vscode/settings.json
+++ /dev/null
@@ -1,3 +0,0 @@
-{
- "markdown.styles": ["public/vscode_markdown.css"]
-}
diff --git a/CHANGELOG.md b/CHANGELOG.md
index c3dbe4cf..60146d13 100644
--- a/CHANGELOG.md
+++ b/CHANGELOG.md
@@ -9,6 +9,9 @@ Initial release of nf-core/spatialaxe, created with the [nf-core](https://nf-co.
### `Added`
+- Image QC and transcript QC subworkflow (`QC`): runs `IMAGE_QC_ANALYSIS` (focus / SNR / morphology metrics) and `TRANSCRIPT_QC_PROCESSING` (per-transcript and per-cell metrics), each rendering a Quarto HTML report via the nf-core `QUARTONOTEBOOK` module. New `image_qc` and `transcript_qc` local modules with pinned `environment.yml` and Dockerfiles.
+- GPU-optional image QC, and new QC parameters (image + transcript QC) with `nextflow_schema.json` entries.
+
### `Fixed`
### `Dependencies`
diff --git a/README.md b/README.md
index ba9c04fd..75dac779 100644
--- a/README.md
+++ b/README.md
@@ -172,6 +172,9 @@ We thank the following people for their extensive assistance in the development
- Matthias Hörtenhuber (mashehu)
- Maxime Garcia (maxulysse)
- Kübra Narcı (kubranarci)
+- Malwina Prater
+- Nell Nie
+- Christel Krueger
## Contributions and Support
diff --git a/assets/notebooks/transcript_qc.qmd b/assets/notebooks/transcript_qc.qmd
new file mode 100644
index 00000000..af63b97e
--- /dev/null
+++ b/assets/notebooks/transcript_qc.qmd
@@ -0,0 +1,333 @@
+---
+title: "Transcript and cell level QC for one Xenium run"
+author: "Viktor Petukhov; Malwina Prater; Dongze He; Felix Krueger"
+date: "today"
+format:
+ html:
+ embed-resources: true
+ standalone: true
+jupyter: python3
+---
+
+## Introduction
+
+This report provides a comprehensive quality control analysis for a single Xenium spatial transcriptomics run. We analyze molecular detection quality, cellular characteristics, and spatial organization to ensure data integrity and guide downstream analysis decisions.
+
+We generate QC metrics for the molecule and cell level using the following files from the Xenium bundle:
+`transcripts.parquet`, `cells.parquet` and `cell_feature_matrix.h5`.
+
+**Important Note:** The analyses in this report are carried out for a single run (=section on a Xenium slide). For multi-run experiments, each run should be analyzed separately before integration.
+
+```{python parameters}
+#| tags: [parameters]
+#| echo: false
+
+# Parameters - these get overridden by -P command line arguments
+INDIR = "transcript_qc"
+SAMPLE_NAME = ""
+XENIUM_BUNDLE = ""
+SAMPLE_PUBLISHED_OUTDIR = ""
+```
+
+## Notebook parameters
+
+```{python setup-environment}
+#| echo: false
+
+from pathlib import Path
+import json
+import pandas as pd
+import base64
+from IPython.display import HTML
+
+def display_figure(figure_path):
+ """Display a figure as embedded base64 image in HTML output"""
+ full_path = Path(INDIR) / "figures" / figure_path
+
+ if not full_path.exists():
+ raise FileNotFoundError(f"CRITICAL ERROR: Required figure missing: {full_path}")
+
+ with open(full_path, 'rb') as f:
+ img_data = f.read()
+
+ img_base64 = base64.b64encode(img_data).decode('utf-8')
+ return HTML(f'
')
+
+# Validate required files
+metrics_file = Path(INDIR) / "transcript_qc_metrics.json"
+if not metrics_file.exists():
+ raise FileNotFoundError(f"CRITICAL ERROR: Required QC metrics file missing: {metrics_file}")
+
+versions_path = Path(INDIR) / "versions.yml"
+if not versions_path.exists():
+ raise FileNotFoundError(f"CRITICAL ERROR: Required versions file missing: {versions_path}")
+
+# Load QC metrics
+with open(metrics_file, 'r') as f:
+ qc_metrics = json.load(f)
+
+print("=== TRANSCRIPT QC SUMMARY ===")
+print(f"Total molecules: {qc_metrics['total_transcripts']:,}")
+print(f"Selected molecules: {qc_metrics['selected_transcripts']:,}")
+print(f"Total features: {qc_metrics['total_features']:,}")
+print(f"Total cells: {qc_metrics['total_cells']:,}")
+print(f"Analyzed genes: {qc_metrics['analyzed_genes']:,}")
+print(f"Min molecules per cell: {qc_metrics['min_transcripts_per_cell']:,}")
+print(f"Min genes per cell: {qc_metrics['min_genes_per_cell']:,}")
+```
+
+## Load data
+
+The Xenium platform detects and decodes molecules across the tissue sample. Each detected molecule is assigned spatial coordinates, a quality score, and classified into different categories based on the decoding process.
+
+## Molecule overview
+
+In this section, we display basic statistics of the dataset regarding the decoded molecules. These statistics and metrics help us understand the size, shape, and quality of the dataset, which is critical for determining the success of the experiment and appropriate filtering thresholds.
+
+### Detected Molecules
+
+First, we focus on the most important information: the molecules detected from the experiment. We show the total number of detected molecules and their assignment to genes in a cell-agnostic way.
+
+The exact number of detected molecules and corresponding genes depends on the dataset, the assay, and the applied probe sets. **Typical expectations:**
+
+- **Molecule counts:** Range from ~1 million to several billion molecules per sample
+- **Gene counts:** Usually a few hundred to a few thousand genes, depending on the panel used
+- **Quality distribution:** Most molecules should have high quality scores (>20)
+
+#### Molecule Categories
+
+Besides the molecules that come from genes, there are also other molecules that should be excluded from data analysis:
+
+1. **Unassigned molecules:** Molecules not assigned to any features. These are usually background molecules not associated with any predefined feature, often caused by imperfect tissue permeabilization or probe leakage.
+
+2. **Non-gene molecules:** Molecules assigned to non-gene features. In 10x Xenium assays, we expect to see a small number of such molecules, such as those from negative control probes and deprecated probes.
+
+**Quality Assessment:** A high-quality experiment should show:
+
+- Low percentage of unassigned molecules (<5%)
+- Minimal non-gene molecules (<2%)
+- Clear separation between gene and control categories
+
+### Quality Value Distribution
+
+Let's examine the quality value (`qv`) distribution of the detected molecules. The quality value measures the confidence in molecule detection based on signal-to-noise ratios.
+
+**Quality Score Interpretation:**
+
+- **qv ≤ 20:** Considered 'low quality' molecules - should be filtered out
+- **qv > 20:** Considered 'high quality' molecules - reliable for analysis
+- **Typical distribution:** Should show a clear peak at high quality values with relatively few low-quality molecules
+
+```{python display-quality-density}
+#| echo: false
+display_figure("quality_distribution_density.png")
+```
+
+**How to interpret this plot:**
+
+- **X-axis:** Quality value (qv) ranging from 0 to ~40+
+- **Y-axis:** Density of molecules at each quality level
+- **Curves:** Each colored curve represents a different codeword category
+- **Vertical line:** Red dashed line at qv=20 shows the quality threshold
+- **Good pattern:** Most gene molecules should peak well above qv=20
+- **Warning signs:** Large peaks below qv=20 or bimodal distributions may indicate technical issues
+
+```{python display-quality-violin}
+#| echo: false
+display_figure("quality_distributions_comprehensive.png")
+```
+
+**Violin plot interpretation:**
+
+- **Width of violin:** Shows the distribution density at each quality level
+- **Box plot inside:** Shows median, quartiles, and outliers
+- **Good pattern:** Gene categories should show distributions centered well above qv=20
+- **Quality comparison:** Different categories can be compared side-by-side
+
+#### Field of View (FoV) Quality Assessment
+
+We also examine quality value distribution across different Fields of View to identify potential technical issues in specific imaging areas.
+
+```{python display-fov-quality}
+#| echo: false
+display_figure("quality_by_fov.png")
+```
+
+**How to interpret FoV quality:**
+
+- **Consistent quality:** All FoVs should show similar quality distributions
+- **Warning signs:**
+ - Individual FoVs with significantly lower quality may indicate imaging problems
+ - Systematic differences could suggest optical issues or tissue artifacts
+ - Uneven quality across the sample may require FoV-specific filtering
+
+#### Distribution of molecule count of features
+
+Next, we examine the distribution of molecules assigned to each feature across the entire region. This helps us understand gene expression levels and identify lowly expressed genes that might represent noise.
+
+```{python display-molecules-per-feature}
+#| echo: false
+display_figure("num_transcripts_per_feature.png")
+```
+
+**Molecule count distribution interpretation:**
+
+- **X-axis:** Number of molecules per feature (log scale)
+- **Y-axis:** Number of features with that molecule count
+- **Gene vs non-gene:** Different colors distinguish between true genes and control features
+- **Threshold line:** Gray dashed line shows the noise threshold calculated from negative controls
+- **Good pattern:** Clear separation between informative genes (above threshold) and noise (below threshold)
+
+**Quality indicators:**
+
+- **High-quality data:** Should show a clear bimodal distribution with most genes well above the noise threshold
+- **Problem signs:** Uniform distribution or unclear separation may indicate poor signal-to-noise ratio
+
+Next, let's visualize the distribution of gene molecule counts with the calculated threshold overlayed. In downstream analysis, we focus on genes with molecule counts above the threshold, as these are likely to be truly expressed rather than noise.
+
+**Important note:** The threshold was calculated using only negative control features, but the plot shows all gene and non-gene features, including genomic features, negative control features, and deprecated features.
+
+#### Distribution of cell size
+
+Next, we parse the `cells.parquet` file to generate cell-level statistics that help assess segmentation quality and cellular characteristics.
+
+```{python display-cell-size}
+#| echo: false
+display_figure("genes_per_cell_distribution.png")
+```
+
+**Cell size distribution interpretation:**
+
+- **Shape of distribution:** Should typically show a log-normal distribution
+- **Mean vs Median:** Lines show central tendencies - large differences may indicate outliers
+- **Quality assessment:**
+ - Very small cells might be segmentation artifacts
+ - Very large cells might be merged cell artifacts
+ - Reasonable range: typically 50-500 square pixels depending on tissue type
+
+#### Distribution of Nucleus RNA fraction per cell
+
+This plot shows the fraction of detected RNA molecules that are located within the nucleus versus the cytoplasm for each cell.
+
+```{python display-nucleus-rna-fraction}
+#| echo: false
+display_figure("nucleus_transcript_fraction_per_cell_distribution.png")
+```
+
+**Nucleus RNA fraction interpretation:**
+
+- **Biological expectation:** Most RNA should be in the cytoplasm, so nucleus fraction should be relatively low
+- **Typical range:** 0.1-0.4 (10-40% of RNA in nucleus)
+- **Quality indicators:**
+ - Very high nuclear fractions (>0.6) may indicate segmentation issues
+ - Very low nuclear fractions (<0.05) might suggest poor nuclear detection
+
+#### Distribution of nucleus-to-cell area fraction per cell
+
+This shows the ratio of nucleus area to total cell area, which is important for assessing segmentation quality.
+
+```{python display-nucleus-area-fraction}
+#| echo: false
+display_figure("nucleus_to_cell_size_fraction_per_cell_distribution.png")
+```
+
+**Nucleus-to-cell area ratio interpretation:**
+
+- **Biological expectation:** Nucleus typically occupies 10-30% of cell area
+- **Quality assessment:**
+ - Ratios >0.8 suggest over-segmentation of nuclei or under-segmentation of cells
+ - Ratios <0.05 suggest under-segmentation of nuclei or over-segmentation of cells
+ - Consistent ratios across cells indicate good segmentation quality
+
+#### Distribution of the number of molecules per cell
+
+Starting from here, we work with the feature-by-cell count matrix (`cell_feature_matrix.h5`) instead of the molecule table for computational efficiency. This matrix contains the number of molecules assigned to each cell for each gene, with non-gene features excluded.
+
+**Critical Quality Assessment:**
+
+We first show a histogram of molecules per cell. The distribution pattern is crucial for determining data quality:
+
+```{python display-molecules-per-cell}
+#| echo: false
+display_figure("num_transcripts_per_cell.png")
+```
+
+**Distribution patterns and their meanings:**
+
+1. **Single peak (Good):** One clear peak representing healthy, high-confidence cells with hundreds to thousands of molecules
+2. **Two peaks (Warning):**
+ - **Main peak:** Healthy cells with substantial molecule counts
+ - **Left peak:** Poorly-segmented "cells" with very few molecules
+ - **Action needed:** This indicates segmentation issues requiring refinement
+
+**Quality thresholds:**
+
+- **Threshold line:** Gray dashed line shows recommended minimum molecules per cell
+- **Below threshold:** Likely represents segmentation artifacts or low-quality cells
+- **Above threshold:** High-confidence cells suitable for analysis
+
+**When to be concerned:**
+
+- Broad distributions without clear peaks suggest poor segmentation
+- Very low molecule counts (<10-50) dominating the distribution
+- Lack of a clear high-confidence cell population
+
+#### Distribution of the number of detected genes per cell
+
+Similarly, we examine the number of unique genes detected per cell, which is another critical quality metric.
+
+```{python display-genes-per-cell}
+#| echo: false
+display_figure("num_genes_per_cell.png")
+```
+
+**Gene detection patterns:**
+
+- **X-axis:** Number of detected genes per cell (log scale)
+- **Y-axis:** Number of cells with that gene count
+- **Threshold:** Gray dashed line shows minimum genes per cell threshold
+- **Expected pattern:** Should show a clear peak at moderate to high gene counts (50-500+ genes per cell)
+
+**Quality interpretation:**
+
+- **High-quality cells:** Detect hundreds of different genes
+- **Low-quality cells:** Detect very few genes (<10-20), often artifacts
+- **Biological variation:** Different cell types naturally express different numbers of genes
+
+**Red flags:**
+
+- Most cells detecting <20 genes suggests poor sensitivity
+- Very broad distribution without clear peak suggests segmentation issues
+- Large population of cells with <5 genes indicates extensive artifacts
+
+## Summary and Recommendations
+
+Based on the quality control analysis:
+
+**Dataset Quality Indicators:**
+
+- Total molecules detected and their quality distribution
+- Proportion of high-quality vs low-quality molecules
+- Separation between gene and control signals
+- Cell segmentation quality metrics
+- Spatial uniformity across fields of view
+
+**Recommended Next Steps:**
+
+1. **Filter cells** below the calculated thresholds for molecules and genes per cell
+2. **Filter molecules** with quality scores ≤20
+3. **Exclude low-expressed genes** below the noise threshold
+4. **Consider field-of-view specific filtering** if quality varies spatially
+5. **Validate segmentation** if distributions suggest poor cell identification
+
+## Code version
+
+*The information below is shown for reproducibility.*
+
+```{python display-version-info}
+#| echo: false
+# Display version information from processing results
+versions_path = Path(INDIR) / "versions.yml"
+with open(versions_path, 'r') as f:
+ print(f.read())
+```
diff --git a/assets/notebooks/xenium_image_qc_report.qmd b/assets/notebooks/xenium_image_qc_report.qmd
new file mode 100644
index 00000000..4d7f3e94
--- /dev/null
+++ b/assets/notebooks/xenium_image_qc_report.qmd
@@ -0,0 +1,3226 @@
+---
+title: "Xenium Image QC"
+author:
+ - name: "Malwina Prater, Hanneke Okkenaug & Nell Yu Nie"
+ affiliation: "Data Science/Bioinformatics & Imaging"
+date: last-modified
+format:
+ html:
+ embed-resources: true
+ standalone: true
+ toc: true
+ toc-depth: 2
+ fontsize: 0.9rem
+ title-block-style: none
+jupyter: python3
+---
+
+```{python title-banner}
+#| echo: false
+from IPython.display import HTML, display
+from datetime import datetime, timezone
+_LOGO_B64 = "iVBORw0KGgoAAAANSUhEUgAAAZAAAAHDCAYAAAAKpUyVAAEAAElEQVR42uz9d3xd53UmjD7rfXc5/aA3AixgFYskSpRkiVazJVe5xB7bsf3ZmcQp43wzzr3fzZ07+c1kJpPMxPPNJJl8KU6bxLHjeBLJtlwlW7JVLMmiWCT2ApIg0TtOb7u86/6x9z44AAEQbBJFncUffgCB0/be717P+6zyLGJmvGWNXYAkrPw5nj75Fyhl+tB9xx/ATGwgsAJIoG51q1vd6ra4aW918ChOvsSTx/4HXLsMJgHlluqrom51q1vd6gCyPHgURp/iySP/FSzDIKFDhlpgxjcQAICovjrqVre61W0Ze+vFaALmMfFTnj78n0EyBJImXCePaMudEFoYzC6AOoDUrW51q1sdQKrgobycR/YUTx/6bUCYIJKAciCEgXjHAx75qINH3epWt7rVAaQGPQAAysnz1MHfBisLJHQADOWWYSY3I5TcTADXk+d1q1vd6lYHkFr88IBh5uSfwsqdgdCiXjgLAqwsxNrvBUiAWdVXRd3qVre61QEkAA+vJLc4/QpnBx+HNBrAyvH/5kJoUURadgGoh6/qVre61a0OIHPoAYDAysL0yT8DCVnzN4JSNvRQO/RIt199VQ9f1a1udatbHUAALyRFhMzwD7icOQ4hox4jAbxSXXYhzaZqPqRudatb3epWBxAfIwTYrWB24DEIGQbjwhxHNZxVt7rVrW51qwOIxz68fo7c1M+4kjsLkqE59uE9ACQN2MUhuFaGAfKfU7e61a1udXtLA0iQEM+M/miJ5DiDSINrZTB98s8BVl5fCPzQVz2kVbe61a1uS9oNLGXile3a5Wkuzh6C0MLz2UeVhCgILYrc2DNwrBQ3rvsUIs23EgXJ9Gq+pJ5cr1vd6la3twSAsM8mirOvwbFmYehJgO0lHq0g9BhKqaMopv8TzOQWjrXvRrxtN/RQC81/TUJd5qRudatb3d4CYor52ddWiDgKpEUAkijnzqCQOYXpge8h1nonN3Tcj0jD5iorYVZeSKwuuFi3utWtDiA3nnm5DEY5exokjEWrrxYDEUCApAlN6lBuBemRZ5Ae/xlCyU3c0L4bidbboOnxNw0r4Tpfqlvd6lYHkEt3m04lzVZpHGLZHg8CSAJc42aZwXAB0iF1EwoCxcwZ5NNnoA8+iUTLTm5ouwvRxNp5rAQgH0yuIyCtr/G61a1udQC5FPxggAhWeQKunYMm9UUT6N5jXSi7CIgIoEUWJMu52ogoNBMCOhy7gOmR5zE9/goiifXc0HY7Gpq3Q9dj1x0rsRUwWVa8KiLqOFK3utWtDiAr4x8MAmCXxqHcCkgai+/NWUHocURaHkQldw7l/HmwMEBaDETSf5VaVqJApEHTQ3AhkM/2I5M5h7GhZ5Fo2spNLbcgkVwzx0rAXjHY68xKgrDVa7MuH5i28fY2jXc0aVQPZ9WtbnWrA8gKzSpPAlC+ZMkC+CAB1y4gtuq9aNn2/yXlllGc3svZsWdRSJ+AY2dBMgrIsNfNvhgrkSYE6XCcEqbG9mFq8jDCsVXc1LwNTc2bETIbqu0nwez51wNMyGcfQ0WFmE44knKwISER1mjJnAgzw3XrTZR1q9s1uy/JC3ELceO0BNzQAOJa6Yvu1UmYHhMRBmLt91Gs/T5U8oOcnXgB2cl9qJQmwCRBWtQDElqalSgIFArjyOXHMTK6F4nkWm5uvgmNydUka1gQM3upl2vABwKAyNjMZRfQBFC0GEMFlzcll0YQIoKmafW7vG51u8amlAIzQwhx3eVM6wBSCyBO8WJbAiin4OU92PVl3wlmbDW1xj6N5rUfRX7mEKcn96CQ7oNt5wARgtAi8MJU5DOTGlYidEihQ7GLmdnTmEqdgxlq4qbGXrQ2rkc81ka1i4aZ/Wrgq7uQMjbgMGCS98ojeRebktoFb+O9P2F6Zob/nz/+Y1iWNY8xeeFAmkdvgv8TUfX/3s/+XxY+ZgFQLQZelwyUzCv+f7Dzq/2M1f8v8vO8z0RzR7+SY7ks0Ge+5N8t9jP7m5ol/77U90Wet+Lvfpj2wu/BBuvir7308XA1clD7OvM3TDwvurBMqcyi123hdV5sjRARSBAEiSqDWPglpax+aZoGTdOg6zoMw4BpmgiFQ4jH4ti2bRt6enqqH8B1XUgp6wByPZonkkjLBnvYLdV4RlHDLBhChpBou4sSbXfBKk1yduYQMjNHUMqPwLHzABlzzKRaxeWDCSQ0zYQiDZaVx8j4IYxMnUI03MrNDavR0rAasXBDFUyudr4kbQU3KSAFYabMcBUgF7Bn13WhaRqefOIJ/Jf/8l/q28O61e0aWlNTE+677z7+7Gc/iw9+8IMkpYRSat4mpg4g14stKz/iSZ0op1DdgdQyE28Pwt4OHQQj3EYt3Q+jpfthlIvjnEv3ITN7AvncEBy7AJJRkNDmARb7QEQkoUsdLiTypVmki7M4P3kKiUgLtzT0oDXZhbARmZcvudLFlLEZwt+RCQIKDiNnK24wxaLJ9MnJSWiahkQiAcepqxPXrW5XdTPrs6disYhvf/vb+Pa3v4077riD/92/+3f4yEc+Qm9WNnJDA4iQJpYTRCSScCszYHarIooLGcqcI/fBhARCkQ4KRTrQ2nUfSsVxnp06hJnpoyiVs4A0ITWtJrxVG+ISkEKDIAkFQrowjZn8LM6On0ZjvI07G7vREm8hKeTcoqNLy5QQvNBVzmEI8iuaATiKkbMZDcEpoQuZiOM41a+61a1uV9+klIjH4wCA/fv346Mf/Sg+9rGP8Z/8yZ+go6OD3mwgckMDiNRiy+4ISBiwi6Owi8NsRHouMpFwMTAhhCMdtGpNB9pX3YeZqSM8MXkQxdIMGBqEFl4Q3vJCVUESXRO6ByasMJEZx3h2GrFQgjsautDd2EEh3Qy4kic9fxFWEuBC1mYuOh7zUD6AKAaKDi+7sKkuzVK3ul1zJhJUO0ajUQgh8Nhjj2H//v149NFHedeuXW8qELmhAUQzGi4SwpJQThHZgW+iZev/VZNIv1iZ3YVgomlhtHfeSW0dtyOd7ufJqWNIZ4dh20UIGYIQGmgeK6kBEwC6poNJQ9Eq4dTEWZyfHeeuhnb0NLYjboYJ1VyJ/9GX8fXTFQWHGfqC31suzwOaeWxNiGplyKWUGXqaYJd6FwVH/8aEEa7X13ujbCUbh+t9c3El1+JynrvU+Vg0QV/zs1IKSql5PyeTSQwMDOChhx7CE088wffcc8+bBkRuaADRQ61eaGqpBcIupB5DZvgHkOEublz381T1cMwrlHAPwCRgJRKNjRupsXEjiqUZnpw+ialUP0qVAlh4iXkiusB5BvkSQRKG1GArB2enRzCYnkFzrJG7Ek1oicbIlHOVVLwgdBXYWElVczjzwlS89ILftWsXNE1DJpN58znHpUqiFwv/LfbYFf7uUirILqfa7Eqd9FLXbLkKtUuq1lqkeup1RLkLruWyVXELrt/FKugu9dzXnouFPwcAsdQ1jsVi1VJeALBtG7FYDMViER/+8Ifx0ksv8caNG0kpdd33jNCNsouaf3U9FlHOneWzP/sVSCEgmCGhIMCo4jppUKSBScJ1yoi03o3GdZ9AuHFHDZDgklV3F5bmOm4FM6nzPD57Fun8NFwGhAxDSB2K/U8lNDBpACQUeV9EGhQkbAYUNJh6CC2RODrjcbRGImQuskPJO4p/MGJ5x6b8xawYRUvhzjYdt7TopNgLby2048eP89TUFIQQ1QXONaWYvERZ5kod2QU36UInvVjJ7CK/u6zvwXvVlmtewfeV/m25n68EbJY715cKChf7CtZC7ffg65JBZQEQXHCdaY7VEhYvr15pOfbCx6302i91/pc6Z4oZrBRc161+2bYN27ZRLpdRLBaRy+UwPj6OH/zgB9i7dy8ikUj1Pqvu5jUN2WwWd955J55//nnSdf267xW5MQHED9K4do5Pv/hZsJWGFAKSlwIQASYDyi2DhYlQ4y1oXP0hxFpupyogAZcxVGou8R5YtjDJYzPnMJUdQ8WugMmAEAYgdSjIeQACSA9UhAbFEi4EHAgokgjrIbSEI+iIRtAcMhDVJVVcxs+mKjxWBnSpwXXnA8juDh1bm5YGkLrVrW7X1lzXxZ/8yZ/wb/3Wb3lREl2fByK6riOTyeC3f/u38bu/+7vXfSjrBgWQORDp3/PrXE4dhqaFINldGkCggYXusRHXAkMi2noXWtZ+FOHEeqplN1zdoazcCy8szbXsMiYzIzyeGka6mIUDgpQhCKFDkQcWAYAwSSj/ZxIeuLgsYbMAkwZDGojoOsqKUHAFNJ/N1DKQkq3wji4DvUmNeIkcSu2O8i1p10gd4E1597yF8jvX+nzV/i0oVvnRj37EH/vYx+A4TjX/GHzewFe8+uqr2LhxIwW5yTqAvK43gFeaO3rsDzk18CgMPQnB9vIAQr7DFjoYEq5TAckIYm1vQ2Pn/QgneskrDa59n0tT3l2sYTBdmOWx9BimcjMoWpYXvtJMCNLAASMJ2IjPToJwFwsNzBocEJg0CNI89lIDIKwYlqPwvtUmOiKyLqpYt7q9gcBs2zYMw8B3vvMd/shHPoJIJLJoKOuXf/mX8Td/8zfXNQu54QEkPfIkjxz6nUsDEJLeo4QBxeQxEmlCD3cgFFuDaMMWxBo2IxRpp4W7jEvZ8SxkJbZrYzo3y2PZKUwX86g4LkgYkFKH8vMhtQDCAahAA0jAJVk9jloAUS6DmPFz60KI6lQHkLrV7Q0227ah6zq+8IUv8J/+6Z/Oa+AlIriui3A4jCNHj2JVV9d1m1C/cUNYfiK9Uhjgcy/9UhU4hJ9pWBJAhOE5YPi7fghA6FAQUIrher3dID2OcGwNki23oKF5GwyzccE8kJVf7MVYScmuYDKX4pFsGjOlElwISGmAhAYX4gIAYfJyIwsBhJWC5TAadOBDa0P1Vo+61e06sIBxpFIp3rFjB2ZmZqDrenUjGrCQL33pS/j85z9PjuNcl2Kn4oa9Qr4DNyI9ZERXQylrBftuhnKKUE4Bjp0Ds1Mj5c4goUPTItB0r0Exn+nH4NnHcfzQn6O/7zHOpM8yMJc09zSxLg7QtdUmQTlvWDexpqmD7lm7he5ZsxHdiQYwAMtvQroUIHCZ0RwSnqo912/eutXtDXe8fgVWc3Mz/R+f+T9gWda8MFUAJE899VT18dflcdzIF8kLYwlEm3aC3cpFvS6zQtumX8aqW34bjaveDU1PwLXzUL6qb5DgYr8qS2gmdD0GVjZmpg7h1Il/xNEjX+aJiVfZcUpzir0+KKwI96o6XHOLqDkSo12r1tD9q9dhXbIBggiWoy7pXHRERP2urVvdrqc9ru9PPvqRj0LTtHl5EKUUNE3DkSNHUCwWuTbRfl0dw41cdROEkgrTe3lo329A1yIQ7C7aB8JMgBbGmt1/B6nHPXEzO8/5mdeQntiDQuY0HNcBaREIGYZLAuxXQXkhIwNMArZyoZhghprR0roD7W07YBrxebmSS5VvZ7+iLHhGwba5P5PHmUwBChqIlg5huUqBFOPDa8x6/qNudbuu/JOXAy0UCrx9+3YMDw/DNM3q75VS0HUdhw4dwtq1a6/LPMgN3YkehIXCjTtIj3Qzl8YBqS0Sx2GAdCg7j0q2D+GmWwFWkHqMkh33ItlxL0r5QU5PvIz09CFY5bQ3+lZG/BAXgVmBQRDSgCQdtl3E0MjLGJs6juamjdzechPi0dpZIOyPbl+BjARqZEwARHWddrQ0Im4YvHcyA10sHijzRBSB7rCog0fd6nadMpBoNEqdnZ18/vx5hEKhOabxJtjb3+Aj6AjMLoQMI9ZyJzLnH4UmEwDcxa4mmF3kJ19GpPl2fx56jWhibDWFY6vRtvoRTs8cxOzEfuRzw1CuBdIiIKEjSDIEUwp1acJxbYxOHMb49GnEY13c0tSL5kQ3Qmasih0rnQUS/FX5fRymlMuvMT/n0ZuQcwtykbdwXXdRCfmLsdPlPu/VfK1r8bwr3Tm+GV7zWr5u3ZZej5fTPW6a5gXrmcHVYVV1AHnDIMS7IPHOdyI7+K2lcxE+0OSnXkHz+s+wNJJUmxAPhkxJPUrNHbvR3LEb+Ww/z0y+hkyqD5VK3mMlWqj6HMU+kGhhMATS+XHM5iag6VHEo23ckuxGS7ILoQWzQC7W0ObhFOPoTBpimce5CkgahJ6opOB5i9mbeSLaG8ls61a3Je89110RkAQbt6UqrC5V3LQOIFf9bvfGKoUbbyYzvpGd/BlAmouWI5HQ4ZSnkB1/Do2rP+TnUGTVa88bMkUCsUQvxRK9sO08Z2ZPYWb2JHK5Udh2ESADwgcTVSPfzkJCsYvZ3DhmchM4O3EKjbF27mjsRnO8tToLpHZx1VogQ3JwapZnyxYMzfAOZcE6FQTYCtjRIKohLlpk4QLAt771LX7++ecxOTkJ27YhhKiO5NQ0rQowgRS167o1ekguXNf72XVdzM7OVhd9bWUZEUFKWW2kchwHyWSyqidERAiFQtXpbMHvA+nrgCXVajEBi3fPB5pddFlRAK5St6WmxC28oYPH8YLnrGRsbvBz8JpL/W6xxyz8qn1M9bFCzNOCEvO+z1UAzmlHAYLEij5T9TjEXOFH7bmp/X31PZc5jwtf90qZ18VEJIPrtVDvbanHK6XgKgXlunCVC+UquMoF2GMQLS0tuOmmm7B79250dHRc0kEsBiCLrac6gLwBxj4TiHc9jJkTRwEZXjyMxQokQ0iP/AjJVe+BkMYicZ+aBe6zEl2PUUv77Whpvx3lcoozmXOYSZ1FLj8B2y6CpAkpNDAFi1JAkzpAEo5yMJ4ewVhmAtFwktuTnehIdiAeitKizouAM6k0n0lnYUodaolQl8tATAc2J+QFKftaxdBPfvKT/Nhjj121c/3Zz34W73//+xGPx6HrnqC8N6TKBZHn8HO5HJ566il85Stfmffc9evXI5VKVROIwc0agFMAFq5SYKVQt7pdyzAUCBBCQgpRvW8c14VSLlgtvTVpaWnBr/7ar/Lv/KffoQAYLub8l2Ig1c9SZyBv5ILwLkC8693I9H8VUBUAcpG9J0NIE5X8EDJjz3Bj93tpTqpk8VgSLciVhEKNFAo1or39NhRLMzw9ewZTqXMolDJgUj4rISh/JgZBQJcaWHizQM5M9OPc9AgaY83c1dCOlmiSTM1zxLZy0Z9K87GZFIwlwCMIVVkKuLlRwpR0AQQqpSClxBe/+EV+7LHHqjv/pVjPSnZPxWKx2vS0kud86lOfwrZt2/jf/tt/i3A4DNu20d/fD9M0IISc9zlqKf7FFFPrIbK6XQ5bWe53tSoTwaZoKVbAzCgUCvj9//r7mJyY5L/5m7+hgFUv995LhZGv91np2ltjiRDAClqojaJt93Fh+NuQesOi4Q1mhtBCmBn4NhId97LUorRk9nkxVlIDJpFwM61e1Yzuzl1IZYd5fOYMZnPjsJwKhCRIIUEgTw6avVkgUtPggjCRn8VEPgNDD3PEiICEhoKtUHBcSKkvGZohP3TVbACbExfOPw9KAUdHR/mLX/wiNE2DbduXnWw1DAPFYhH/5t/8G3z+85+nIATGDGiaB9yDE3lmMFa3xchjP97x/uZv/iZ95zvf4ZdeeqlavmhZNkxzftx3qbBC3er2egL6YmGxpTZUuq7jueeeq845v9jGbFkAEXUAuW4sseajKI4+iaVBgSGEAbs0gZnzj6Ntw2ewLAtZAZgIIdHcsIaaG9agWM7w+Ox5TKaHUah4fRxSmtUhU0ESXZeeWKLjukiVCnAhQUKHLrTFgm8LjgC4vUlC0oU5gKBB6Y//+I+Ry+VgGMYlzUCvjYUTEcrlMj74wQ/ij//4j8lxHEhN88BDEmZzFf7aT87i5FAalu1i18Zm/uX3bSHvhvIYz1e/+lXceeedmJmZgaZpPohYME2zGp9+Kzinul0/7ONyHrPw8Uop6IaOlQohLgsgqAPIdXCnCoAVzORWCrfcyaWpn4H0hiUWgILUYpgd+iES7fdyKL6WVjbqdmkwCYAhEkpSb9ctWNOxDbO5CR5Pj2I6OwXLsSG1hVIGfoUGCQgIqEUmGdaaAFB0GVsbJDpCF/Z9MDOklJienua//du/rcopBI77YjdK7VChILn9i7/4i/jLv/xLL88iRDU5OzRV4L984hQmZkuIGBLCAH56eBx3bGrl2ze3EsPLc/T29tITTzzBn/jEJzAwMFDNf1QqlSqIXI8OxxtPzJf9GrWFAtfC6V3N5y3xYq9bm8KlroErmRK53P+X+rm2qKP277ZlLzud8GIAwssUctQB5I25/T0WsvZTKE29vPzjiMDKxsTpf8Ca2/4DrrQJb2EDoRQaWpOrqDW5CoVKns+M92EiNwNBctFPzRd9faCigLaQxK2N+qJNg67rQtM0/K//9b8wOzsLwzBgWdYlh6za2tpwxx134Fd+5Vfw3ve+d97b5EsOv9Y/i+/uGUSxbCMe1uE4LgQBuiT0DWdw++ZWgL2bxnEc3HnnnfS9732Pd+/ejVKpVE2aF4vFJUNZAQO6LDfGl+BkaOn93/IjVS+cvLfQISxWzXWxn6/W31by/4s64xXcEITlHWAVRGnpa3UlI3mX+/mikxj9SYMXA14p5bwGwOC7X7G4ItchpFiagdQB5HphId589FDLHRRq3sXl2YMgPbm4C2IFqUWQTx1FavQ5bux6kC5VZXdpVjKfZUTNGN2y5ja82PciFy1rrnR4xeAE2AyEJGF3q7lo6CpgH8Vikb/0pS9V2Uc4HMbnPvc5PPjgg2hubq4mCYOxnAHT0HUd0WgUTU1N6OjoQDQa9eRelIIgwnTO4u/vG8G5iTxSuQokAYYm4boMRzEIDCEIs9lycBqq8WKlFLZu3UqHDx/mSqVyQanwwhv7auyol7oxa+PNizm/eWWqgRPlC1/vYqW9gVepfc4Fr32R11rJ35b67JcCIFcycnclj1sJ850XhmVVXeCLleauBCgWG89b+xWsQcdx5o2otW0blUoFpVIJ+XweuVwOjz32GA4dOgTTNKuMI5BkD+6fizIQ8ebsxXrL5UDgy7Enen8BpZnXlt0aMCtIGcbkuW8h3nIba0ZiBQn1SwgnEEC+nuVoaoTLdtkbCHVJcAQoX1/r7e0RxHSxLPv4+te/jqGhIYTDYZRKJfzGb/wG/uAP/uCSDyi4MQImULIcDE4XMJOrQPigRgBcxWiIGbBtF8WSjXTBYzxiEce5Zs2aekKgbm86Gx4e5v379yMcDl8AICsNYS3s+bmSEF4dQK4pC/FyIeGWOynStpsLU68ARnLJGBEJHXZlBlODP0Dnhk9dYkJ9IXTxBbvVfCnD56fOYjQ9DhLGJcm0B/0eDODtHTG0hrRFwSMYiWnbNv7oj/5oXnJ6y5Yt1R1WwD5WsnNfGLNd3RKl//jxHTg2mOb9p2dwdCCFUtlB2VL4F/euwZq2KH7nKweQylZQrDgcMbVFK8RWcg4JVye5/lZLYL9VjvdqrI2VjKgNZnQUi8UlN1krZSBL9Xpc79fsLchA5iy54ZdQmNmPRVu559wapBZBZuIVtKx+H+tGwyWxkLlkuPCqKQheJ3p2nMdmBzCdm4atGFILgy9hsQgAts9idncm0BkxlhRLDNjHY489xidOnICu6/M6uYMywyuRNAnmrG9f00Db1zRgKlPmb754Hi8enUS+ZKO3M0F3bmnjFw+NYmgij009DV7YRdBFb6K61e16BUpN05ZkDUEobCVA9WZd+2/NO7ZakXUTxbreDWVnANIW3+OT8FnILKzixAp2ON68kKBM1duxe6c5V5zmc6Ov8v4TT/Dh/p9iMj0MANCljksR3RBEsJSCIQXu72pCZ8Sk5fKQQbL6i1/84gWL/WrpYAUvq/w+j9ZkiH7lvZtpdVsUh/pnAQAfuHs1hCD8aN/QgjxQ3ep249lKAGQlm6d6COv63KYAYDRt/GUUpvfBtfIgI+lhqj8fhJUF5VagGIi33I5wbPV8gcUqYABBye380l2FfHGKU5lBzGZGkCulvYSyNKDJEEDSC0Et5/0XhKwAQtlx0RSO4s7ONsR0PxS0xPMDmv29732PDx06BF3Xq4zkagJILbiBANdlSEnYva0N//uZfpwdzfL6rgQ9dHs3/+Bn57G2PcYffPs68m40L8Fet7rdSCyoNgdysUbCNysDeQuHsLzudGm2UPst/4nHD/4unMosWOhQkCAZhRlbi1ByM6JNNyPechuRL8xY7TaHJ0xXOyDKtvOcz48hnRlEJjeKQiXrVSoJEyRD0HUNimU1tHUpjtkGw1Uu1jc04ea2VpJEFw2mBQvzS1/60oqEAa8qPgNojJlwFeOpAyP4fFcCH3twPU4NpvAPT53C0GSOf/6hTWhOhKgOInW70axWc+5KGMj1zNLf0jmQakK98WbquecvOT/xIlw7B2m2IJTYCDPeS/Pk3NmtltgGzth1KygVxjiXHUQ2N4RCYQqWU4ZiAskQhDQhNBOAhAtcMnB43QSEsusiZOjY0daJ7niCgItnYgLZknPnzvELL7xQ/d3rufOpWC7CpoYj52YxNJnnnrYY/fIjW/m/f/0Ann11BCfOpfAv37+Fd21pJxXEg+vd2XW7ARjIpYSwLqaVVQeQ6xxENLOFGlZ/eOHVAysHJDS/MUwCYJQKo5zPnEU+cw6FwhgqVt7TdxIGIE1omgmGBgUB5YeycBn9I0TkjcglYE2yBVvbOimk6XM9BBd5fgAgzz33HCqVyjzZktcreTedq0AKQtlysefEJHraYtjYnaTf/Pmd/OffOoKxqQL+5/9+DZ94aBN/8N5eL6TFQbVaXeqjbm9eU1eJgdQB5M0AIvAYRjAnJJgwSKRBuWUUs/2cSx1HPn0a5eIUHGWBSQfJMIQwIaQOJgEXwp8voIDLcH5B85piBdd10BhtwKbWbrTFAtbBl6yNs3fv3td94QaOf2iyAEEAScKJgZTf4QtsXt1Iv/0Ld/BffecoDvZN4cvfP4YT52f4E+/cjLVdCarP3q3bDRDDuipJ9DqAXPcXOqiY0qrbetfOcTF9AvmZQ8inT6FSnvUk2GUIQoah6TEwJBQkOKi8Al12j2Ew8MdVLmzHRSycxLrmHvQ0tFIg2UG4NGG1YFH29fUtSYevxTTCoKR3NlfhoakCdE3AcRnpvIVC2eFYWCdXMVobw/Tvf+EO/HDPeX70x314+cgYDp2axi0bmnnX1nbcuqkNzQ1hcv38SB1T6vbmwA2uhrDqOZAbGjTYkzcJRtDaeS6mDiE3tReF9EnY5RQYAtDCEFoIAjoUguEyAWBc2c7Ba+oDHNeBgkIs3ICeph50N3ZUpxNeDuuoXZRTk1NLLsRrsfMJbqCD/bPIlWxETc0bxOOX+AJensOLVDHee/daamsM83//h/3QJGH/8XG8dHAEuhT4D79yN+/c0ka1wFS3ur1ZgGSlAFLPgbw5Limqela+42dlo5Q6xPnxF1CcfQ1WeQZMEpARCD0KQEJdYVjqgiCVzzYUK9iuBRIGmmKtWNW8Bu3JdgpGilb1jS4DPILn2raNfCF/UZC5mrkGIoKrGC+fmISuCTAYSgHRkI5ISJs3n10pAASUKg4UA/mSjbAu0dvdgFs3taGnI47XTkzwTb3NFDK1OojU7U1jdQZyI4FG0BToV1HZubNcmHgOhYkXUS4MeKkPLQahxwCSUCxqWMaVeazqe4Og2IXjOoBQCJsN6EquQntjD5KRRlro/AOnXrsQF5MRWc4cx4Ft25e1cC/vpvHCTQdOz/DAZAERQ4IVw3Jc9HYloElvRrzw5VSkIPxozwB/5YnjcF2Fd97Rg3e/bS3WdSVJSoLjKPzbfzzAqzsT/G8/9zaETY3qIFK3N4tdSQ7kSiT/6wByxfQxcLhzoOGUxrk49RJK48+gkj4K5ZTBWhRCiwDQoUBzoHFl4u3zWI6rbCjlAMKEaTagLd6FloYeNMbbSQoNSwFHAB5CLD6h72qwhquZA2H/M1VsF0/sG4ahCT9MBWiCcP8tndUHKngg8pUnT/D3XzwHVgq/+Mg2fMCvxPLAT0HTBD7wwAb83bcO46vfPoJ/9fM7/eOvI0jdrl+7VOdfT6JfVyEqWe0Wd+0cF6ZfQW78GVRmDoAr09CEBpJhSKPBr5pSYLhePuQqAIZSNly2wSQhtBhi0TYkEz1oSPQgEW0lKfX5YECLS4cH4DE4OMhPP/00stksbr/9dtx3331UCzgXW5hLzRq42gtXKY9RPLF/hMdSJUQNCWYgnbNw/44ObOhKeL0e5OVA/uFHp/gHL52HIMJ7dq/DB+7tJVd5ysKCCFIKMID33tdL+46O87N7B3HH9k6+fXtHvfGwbm+CTSxflT6QOgN5XYHDA4H87CFOjz2NwuTLcEsjEAB0aUIaDSAwwMor26XLYBtVfSvPwSlle18kIfQIwuEWRGOrEE+sRjzWhXCogRYuimDRLLVwAvD4xje+wZ/73OeQzWarf/voRz/KX/3qVxEKhehiTERKCV3TrzmAuD54HBvM8DOHJhA1NRAY+ZKDdR0xfOKBXmKGP0yL8PKxcf7+z84jbOpoShj41Lu2UBDaCg6HyAOlkKHh/jt6cPr8LJ5+6Rx2bmuv94fU7U0BILVSJnUAuS6Bg6thKtvK8OzYs5gd+RFKmZMgVYaumdD1pNfSx64HGpfLMOAnhF0LLlfA0CD0OMKRNkRiPYgm1iIWW4VQuJkWDp5aCWjUggcRYXh4mD/72c+iVCpB1/Uq4/jmN7+JTZs24fd///erWlfLAchif7+ajYQBeIylSvy1585BkwQpgHzJQVsyhP/zg1sRNjUoP+dh2S4ef/4swqZEueTg4+/cBtOQUIovLGrzT9X6ngbEogYGRjMYnypwV1uMVsLA6la3NxJAVur8lwQQ1JPo1+jiuD7jIJSLYzw2+D3MjP4YdnEUutChaSakZoJ80GBcioTIXMKdGWBlwXXLYOiQRgLhaBfC8XWIJnsRjvXADLXQwgVQO9McuLSxlMGc8n/+53+ugkeQCBdCQEqJr3/96/id3/kdGIaxaCirdkpdPB5fmoFIcQXXwCvDlYIwni7xXz91BmXLhaFJFEoWWhImvvChrWhOmBR0l4OAY+dTPDZbBAHYvKYRd23rIG9mCS12JQAAsYiBkCFRKtmYmimgqy1Wr8iq25uGgVzMltzIMeP1mz7/FgCQoAyXSKJcmuKhc9/A5MiP4FZmYWgh6EbSCyyx65XdXiJoeNImrqfCCwGhxWFGuxFObkKkYQsi8XUwwq20OBNamAi/Mu92+PDhecOfahfl2NgYxsbGeM2aNUvuxINZH729vXj11VerY2xrF+2llggHoEH+aF4C4ehQhh99aRCFkg1Tl6hYDuIRHf/mg1vRmpwTSnT94zh0ZhrMHnN58PZu77MyIJf5KMI/p0oxsgWr7p3qdl1bbQXlFTGQYONVB5CrxzqUsnH+3Ld46Nw3YJenYeoR6EYDiJ1LD1EFTEO5UHYZTBIy1IpwYgsizbci2rANRrR7fkiKlV9xJGpCXHN5mKu1+PL5/AX0Nfi/ZVnIZDIX3QEBwMc+9jE8+uijIKJ54SzHcdDY2Lhi0AjyEwHojMyW+KcnpvDq2RQIDEMTUIpRcRR+8eH1aGsIVZkFA1VNrINnphEyJIolG8mY4ZX0glGbQF9otquqTYiZXAX1KSJ1u9FDWAsrua7HkO2bBECCHb5EKn2Sjx/9c2TTx2BoERhGsia3cYlsAwA7Ba8vIdSBeNOtiLbejXDTLdDMJlrIfHyo8Hs6gHJxgnOZfuRzAyiVUnCUQjSxGqtXvxO6Hrni+emlUmkeECxcWJa1/E5c0zQopfCRj3yEPvWpT/HXv/71eX9/97vfje3bt1PAVC4EDXg6VjWgMZ2z+OxkAceGs+ifyKNsOTB1AVYKrqvADGiScGo4g5a4yY0xg0xDAgzM5iv8v398BqlcBSFdQNME/veP+gAG37yhhWRNCCsIebmKoWuEmXQR5YoDKQgTMwUQUAeRul33djUYyEpfow4gi0NHtcT17Llv88lTfwehbBhGQzW/QSt1JQFouGWwW4YwGhBuuRPRjnci0nInZC1o+LHHuZkfwSwQQmb6IE8MP4tScRK2awNCAygECB358UmUSils3fpJEGmXFaMPFlOgnHupi26xx/zjP/4jfehDH+If/ehHsCwLu3fvxi/+4i9SkENZHDQAy2WMpMp8dqqIc1NFTKTLKJRsEBiaIEQMDY6jAg1KMAO6FHjm0DhePDqBqKmzqRPAjJlMCbmChZChwXVd6JrA6EwB/+0f9mNNW4zv3NqO27a0Y3VHnDQpvHJfQXBdhadeOg8QYOoaTpyZRqlsIxzS68Oo6nZDA8hiG8g6gKyYAnr5DqUcvHb0T3lg8AmE9Tg0TbsM4GAoJw+wDSO6GtHOhxHpejeMWC9dGJryQWMee/DAwypP88DJv/cqhrQIdD0KJukNiSIJ00ggmx3E6Og+7u6+h6rSKddw8a10YX784x+nj3/844s9CMFUwwA0BmbL3DdZxPnpElIFG7bt5ZQMIkRNDa5y4bqMeTM8hE+6mBANaXBsF5mCBcfxJrNJYoQNDY7jVsHG0AV0SRiYyOHUYArf+MkZdLVEuLsthoaYCWbG+ZE0BkdzCBsSkgipTBlf+dYR/tWf30lCUB1E6vaG2tUowa2X8V4j8LDsPL/86u9jamofwkZDtX9jRe6iFjgARBt2oKHng4h2PAihRakKDMG8Dj80tfyFlpDSBATA8zrX/fdiF7oWxtj4AbS2bmfTTFxxKOtqmeu68xajkNLLa/j/H8/ZfHyiiNNTJczmLTjKAwwCICVBFxLliuvpD/P8cFex4sB1FZTrATGBoZHHRgR5Wlhck8eofa5iRkiXMDUBx2GMTOYxNJb1Z4IwTE0gbBq+FDwjEtax/8g4CoVX+Oc/sBWrOhIrbqysW91eLwC5FPZQB5BrAB7lSop/uvd3kMqcQtRsBCvnUi4rlFOAABBvvRtNaz6KWOs9NWzDxVzllVzR6zEzdLOR4k3beGZiH6SeWJQDEQlYTgkjY/vQu/adl1xuGiyYpfo7gr9fqgxJ8Hh/hHv1M/WnKvzaaBFDqTIqtscUTF0gxIRC2UVrTMf9mxrRHjdw4Hwazx2fhi49SAxyHu++tRPxsIZUzsJkuoSpTBljMwU4rvKICc9JnQhBYPayGEFz4VxfCkHTJYQpIfziBPL/TlVm5oHI6fOz+B9/9TIefnsvv+eB9SSlqLORul1nvuzqJdHrAHIJ4FGqpPjZV/4jstl+hI0klLKx8kAQQXEF8YYd6Nz4S4i17JoPHCsGjYWv6llbz7uQnj6y5KRBBkOTJqamT6Cr43YOhRoui4UYhrH8xdMu/fJVPwUBQ1mb944UcWamDIM8vaqILuC4CmBG0VJY3WTi47e1U0j3jvNt6xt5z+lZ2I6CEECl7OKjd3Xj7k3NtPB9hibz/Fc/OIVUrlItbS5ZDioVG4IATQhI6QGKVWEUSg5YAVFTQApRBRaxyA2mFCMU0kCK8b2n+3Di9BR/8sPbsaojUZc5qdt1Y/Uk+uuM1kQCFSvLP9n7O8hkzyOix6HYwcrdPYHZga4nseFtf+4NY2LXC6uQvAK9K3i5AlYIRTqoofU2nh5/BcJILvFQAdutYHTiEHrX3H9J4ZVgx5FIJC7oXA92JaFQCA0NDZd4fj3WUXIYL48U+fhUGSXLxdu6o7i5PYxvHJlBruxAEKHiKDRGNHzMBw9XcfX3Qb7EdhitCRO71jeSYo9NEOaYzeq2GK3tiPNEqoSQIZAve7Imuza1Ym1HHA0xA7omvc9UcTAwlsX+4xM4dGoCmYKFkCYRMsWS5VZKMSSAeMzAwHAa/8/f7MG/eGQr37mzmwLBxXpEq25vVgZSD2Fd0s7Y6wFw3DKe2f/7mM2crYLHpW0mvXJf185j6Oj/zW1rP4ZQfD3NAw52AYjLHjkLAG3dD2B26qCfP5GLMikpDUzN9qG76w42LqOs96abbvJ24DVdqsHPa9asQWdnJy38+5I7IfZy3BNFl398vojZknde37M+jh3tYSraii3XAwkoL9707puaENYFHBXIsACzBRtlWyGkEYquwppWT6I9kGavZSCuYkxlypDSky/5+AO9eM8dPbQUO1jdHse9t67CxEyBD5yYwJ7DY+gfTiGse7kagRqAqrmnXNdjI6wY//iNQ5iYzPMH3r2lnhep2xtmwZq7GgByPTOQ60RD2AuEMzOee/WPeGzmCEJGAuqydKvmGMDM0Hdx9uVfw8DeL/DM2a9yOXPc94RyzgtdUv9IwEIYoUgnNbTsgOuWlqyyEiRhWQVMTp+6pMUkhOeQP/nJTyIUCsGyLEgpIaWsKnx+4QtfgJRyRWqfAXgM5x3+wbki8raCJoB3ro1hR3uYGMBwxkLRVpAEOIrRGNGwviVMgBfa0nynv+dMqip2rxSjLWHO5Sb8JLervGs5Ml3k0ZkiXJfxyNtW4313rSYiD1iUf71rcyDK/317c5Te9/Ze+p3P30Nb1jYhX7RRrjgoVmxULAeOo/zzRNWmQ6UYUgrEYyZ+8vxZ/P3XX+Vyxbmgk79udXszMJBa+ZI6A7mog1MQJPHikb/ic2MvIRZqgKvsi4atyK+aIq8eyvNENSde0xMAOyjMHEBx6mXoMoRwYiPH2u9HpP1+6LG1c8ykNqm+EsADobXz7ZidPgIscYEZDCE0TM72YVXHLSsu5w0kR3p7e+nxxx/nf/2v/zXOnj3rHZOm4Qtf+AJ+/dd/nQLNrJWBh8tPDc51cO9eFcGWZrM6KvbVsSKkz5GkAHJlF986NMU9DSY0QShZLk6N53FusgBTF3CVgq4J9I3lccuaJDfFzLl7wP/+8slJFMsO2htCeNft3eS6DBKe0OKi3I7mQm2O671+YyKEZNzExp5Gr8u94qBQtJHJVpAv29AFIWTo0CShXHbAroJpaDh8dByFgsW/8KnbEIsa9SFUdXtjfNvl5kB4LlytuJ4DWcbBuRAkcfDs43z03PcRMxuglLMsNSJfr8pxCn5ynaHIKzMVwqj+PZA1EVoUElEQO6hkTsBOHUTm7N8j1Hw7RzrfjUjb3RBarFrWOxf2oKXfH4xoYi1FE+s4nx2E0KIX8hhmCKmjUJxBNj/GyfiqFSvIBiDynve8hw4fPsyvvPIKstkstmzZgs2bN180dMVVFgQM5l1+dsSqki5dELrjGnKW4qmCg9fGihjN2tAlQblcBYHj4wUcHclDuQxXKQgoT67Ebxw0NIHzUwX88Q9OozGqcXsyhLWtUXQ2hjE0lcfLxycRCWmwXQVXMYcM78CVqlUmXpTkQdMEShUH5YqD//GbDyIeMaqPrFgOJqeLfGZgFidOT6F/II1svoJ/+bFbkMtXcPzkJGZmijjVN4V/+Pqr+Nwv3AFdF2CuBam5UmSv7aeOLnV740NeizEYVnUGssTJ8ZhH//grvOfEPyBkJvzxs8udaAnHyUMTEs0ttyORWA9JElZpHKXsaVQKA3DcCjQtDBIaANdvEPTVcbUwJCJgdlGc/BnyU3ugRVYj0nYvx7veCTOxiebil6raib7YhSUiNLfehlzm3NJg47/O1Gw/kvFVlxZf9EEkEonQgw8+OG9Xsxh4cM17BmmCo2mXD0zZ1XilC8Blxrf6cgAzChUXzAqGRlDO/IUa0gUgGcplKEVwFeYAxmcKukZwbMbobAnnJ/J4+fg0BDFsx4Xhh75yRRt/9u1jeP/benjjqiSZupz3Gioo0fUbGV2XoWkCR05P8eh0AfGIMS8pbhoaeroS1NOVwIN3r8XUTIF//EI/0pky3vOOjfTQfesxOVXgF18+j5++eA4/fOoUf/CRreS6QXUWLwperDyGVLe6vd4MZLkQWL0Ka4kTQySQLU7wc4f/ApoMYum8LHhYVgYtjTdh69ZfQ0PjVpp/sWwU0ic4NfYsMhM/hVMagS4NkAyh2jDod5sD8Oef63CsaaTPP4bMyJMwG7ZzvPOdiLXeBakvzUoCkEk2b4U++DRc5QBkLHqcQmhIZUfgKge1I2xXCiK1i8jroxDVXg7UOF6qCVuNlJiPpV2MFV1o/h/cGoBx2XOYpkYeOLh8AQtAVQjUb5jkxa6jx3IMTUAXBKUBrnKhS4LruFDK6zY/PZrFHz12BG3JEG/qTmD7uiZs7G5AY9wkucCTaxrBcRUef+YMJmcKmEoVuaVhrgiB/TcOwlKtzVH65Id34OjJSX7tyBjv3NFJ7W0x+uiHtmNoKM0vvXwe69c387ab2oNRVZgYz/GpExPQNIHungZ09zSSkHXwqNv1x0BUPQeyeJCFGXj2yF+jZOUQ0aNgdpYFj4qVxtruh3DLzf9vksLwBQ7nvKgQOuJNN1O86WY4G3+BU6NPIT38A9jZPggCpAz74S01ByhwQaRDGlEoEIqpIyikjkALr0Ks7W5OtN+LcKJ3bt5HjRIvM0PXYxRPruOZ6eMQurkECEiUK1nkClPcEO9cURiLMV/Guco4/OcR5pMeSzHSlstjJYWREmPW8hLupgBcdSEsCwBM8DrEF1mflsOwHQVWHgORxBAL4F0QLSqYH7CK2v+bugRLwnSmhOGJHH6yfwiJiIGe1hhv6E5iXVcCrckwhCCMT+fx5EvnMDiehXIVBseyaG2MwHYYUs6N/63NmSilsH1LGx0+Ps4HDo3w7bd44cKbd3RicCiN733/OIjBXV0JHHxtBK+8PIBy0YIkwNAk2trjfOuubtxye089mlW36yyEVWcgF4SuiASODDzN5yYOIG4mwewsE7oiWHYWm9Z/DDtu+lWqfY0LQckLVWlGA7Wu/ThaVv8ccpMvcmboO6jMHIDrFiG0KCD0moQ5++NtJYQWAUiDY6UxM/g9zI48g3DDFk603Y14y63QjYbqpWa/HDiWWIeZqaPLfHqCYgfp7Aga4p1YSTkvBbRiEXOUQsF2OGs5SFku0pZCxmYUXILDEkJoMIQHDrXEguDLVZHHRmzX6/JeaC4DqxtNNEc0mJLguIyTY3nM5CuQ/qdnAGXLhe4J7cKyFWzHhXIBU1s85MfM0DUBLaJDKYWS5eBI/wwOnJoAMUOTAsSAZTnQBCEW1lEuKzzx07PYsbEVxrzQ1xwDIX9+ulKMm7d20N/94wFmBd61cxXdfedq6u+f4b5T03j0sUOIhDWUijY0IdDYFIGhCyiHMT2ZxxPfPoKxoTS/+4PbicSbo4ekdtJl3a7f67MCBFkSVOohrIU7ayIUyil+ue9RmHp02bwHkYDj5LFl/cexY8vnfHFCWqKiqbbpzp/PIXQkOh6kRMeDKKePcW74eyhPPAe3PAWSEUCP+5MHaxmGCyINmh6BYkIhdQK51Alo51sRbdzKydbbEWvYTFIL++zAuOgxC5LI5Ceqx3Qxy5TznK+UUXEc2MxwFMNWQMlRyNgOSi7BZoIiDQI6SEhoUkIHQfnBHtSo6jIAhz1mwa7HTFpjOnIVF9myXxZLQNllvHNjA3auis67JNs6o/w3LwzO0Q0GPrSrExs7YnAVI1eykcpb6B/PY3/f9AU3QjVhXSPErkkBGSKEDQHXVZ6qrwKiYd3L17gM09TQd34W//WvXuZ7bu3C+p5GrGqPUzg0X+lYqbmk+H13r8OX/nYPWpoivHZNI33207fTT1/o5/37h1Aq2dCkwK5dPdj99nWQmiBWzNlMGU9+5ygOHxjG5q0d3Lu59bruaA8AOWCm9X6XNzkDWeY61wFkQXyDSOCVM99CvjyLmBFfJnRFcF0b4VAztm/+lzQ3LEqs6JIEw50C0Ak1bKNQwza4Gz7HhfGfoDD2Y5SzZ7wyOS0OiLDPSmiOlUBCaGEIknCdElITryA19Sr0cDtHExugmQ1IzZ6EkCaWE70SQqJYzsCySzD08LIs5PDwSR7KTIEhoSDBpFW/k9AAoYGEDoMEWAiACUxeGMtlhiKvJVwpP0SlAJ2AppBAa0iiKyLRGpGIG4IOjZf4x/05hKT3/OaoVgUPxcEhMWIhCUMKOI5C2Xaxe2MT7qmRL+loCAEA7tzYjHLF4VdOTSHsd7CXLQchXcBxPNFF23aqEiWan9SuFVlUyvv8wtcfC5kaBkYzODvoNRU2NYS5pyOBjWsasbm3CWtWJavNiUoxVncnqakxzP/46EH8+q+8jRsbwvTQOzfSzltX8fe+dxRnTk9j0+Y2xOJmEE6kSNQAwCyEQCF//U88DBQKRkdHub29nYIeoTqIvDkZSL0T/RJCV7P5UT429DxCehSK3WX6PRhCSFh2DmMTr3BXxz3V8NVKd/LzHuc/T4ZaKbH255FY+wmUZl/j/NgzKEwfgF2ZAZMO0qIgoYPJl9HwFXeJdGi6CZAG28pgZmIfmCRIC4MotPTsYmaQELCcMgrlNBt6eNG+hGD+iDedz0XYCMOF8OTifQABBFxffp0xJ1LoMqPBlEjoOkxNQBMCAgydBGIakDQE4vqFq7Rgzw3KclzGqqTHpoL+kUAAMVV0ULYVdOFpWN26JolAviSYKOiNpaV5+Q+lGJ9+x3rsWNcEy3YxPlvEwHgOYzMFjM8UMDpVQKFkQfMv0Zw6MFd3ZgwgHNK9PhXFyOYqODg7gYPHxhA2JFa1J3jn9nbcccsqNDWEyTC8FbXrtm6MjWdh6JKjUYOamyPU2Zngk8cnMTiQwrrepurn/OlPTvP0ZAHRmIHutY1LRRWuG+ZhWRY+//nP87e//W1s3LiR/+mf/gnr1q2jpSr06vbmBZCAgVyPQPKG5ED29X8fFaeEqBED+GIKu1590b6D/x29q9/HG9Z9GOFwG80LU1VZw8Veai7nEUiQhJtuo3DTbWiy0lyY3o/81B6UMqdhWRlAhDwwIb/qyX8/QPkhLs1nB+Kig+8JniZXoZhCY7xzyccAwPauTWQp5rFsCoYfJmP/32KHyWDc3ZbAqqhx0SrU2jVIBEwUHH8WufeHjri+KKidnynBUR5raIzqaEuaJGguT0MgCAIsR2FoughTEyhbLtZ3xvHALZ3VT9XZHMHOjS3V159MFfkvHz+CY2dnYOoC+aINjQDTBwHHZSjXRdHvDTKkgGlIRCO6BzqKMTqew8BwCs//7Dxu3drBmiRMThWQTISwdUs7/c3f7eVNG1p4ZrqAEyfGEU+Y2L9vEL3rm7mlNUqv7h3k/XsGQAA2b+9AY3OErtfdfDA98r/9t//Gf//3f49kMol9+/bh3/3Wb+HRf/7n6zrcUbfLv+ZveQYSlO1mSlN8amwPTD1y0Z6PwIURSRCAM/3fwOjoM+juvJd7ut+FRHIj1YapVs5KaE6/KmAlRgMluh5CoushOOVpzk3vR2biZyhl+6G4AtJjNUAyByYMtaKtagAwhXL6oo+VQmLX6q10amKQz8xMAIJAQi4uHe+Dgik9B+7yInpRVAPFNR+14jJSJRdSeKxBk4S2mD4vJks+2zk9UYAmCZatsKojDF2KeRE7xZ6O1sBkgaezFYQ0gULZxuq2mL+LAoSoATD/87U1RigZNVkxo2Q5uOfmTrx/dy9iER1SeBpauYKF6VQJY5N5jIxnMTFdQDZnQbmu34muIRw2YdsKL+0dgCRCOKRj36tDGJ/I8vmBFM71z0IKr8s+mQjDdRT+4e/3oiER4ny2AkMTiCZDuPv+9XMJpOtwNyulRC6X4y9/+csIhUJgZkQiETz33HOYmpri1tZWqoey3ny2nJx7HUC8KDcIEkeHfoqilUPciK+Afcw9GwAMIwnXKeP8uccxOvQkmhq3cWfXA2hpexsMs5HmbjT3MlmJn9wNtVBj93vQ2P0eFFLHODX6LLIzR+A4eQgtBhK+ECNf6iIRKFdyy1LWWmDY0r6amqJJPjw+jLzteJMYFzkzgoAXJ3K4szXO3VF9jk/UFHLVirwo9rJI6bLLgf6VrRhRQ6IhrFXrhAOAmM7bPJ6xYEgBy3KxtjU8j51U34CAvtEsHJdBvvx7Z1PYj9fPb95T7CWoB8ZzfODUJKQgrG5P4P/z6dvpYucmX7B4ZCKH88Np9A+kMDyaRTpbQkiXiEW8EBwxMDaew/BQBtGIAUlecPDhhzbippvaoRTj+WfO4OihEcRjIRRzFdz/8EZEosYlJ89fL4cdsI/XXnsNw8PDiEajcF0XUkrMzsygr68Pra2tWGzGfd1eX+d/qSGs5Z5fBxB4woKOsnFibA90afqAsryzJT8jQKT8xjFP9kQ3kiB2MDvzKtJTexEJt6K59XZu63oIyebbaliJC1qx6m5txzlXGVO0cRtFG7ehnB/imdHnkJk+BNvOg4UJCvpKSMxVcS2DgYIEKnbR1/4SK1o8bbEk3bs2ilfHhnisUIChmYucW4LLwAsTRWxKhnhHo0HGAgdY+7+gX+70bAUOM0zhdZknIhKhIBnBXl5FgjA4W0LZUQhrBFMXWNMcmRdyq72Jzo7noUlvzKyhCXS3RC/8ADWA881nz8CyXTiuws89sAFEXhNhoOw7r2HSf59Y1KDNvc3Y3NtcBZQTZ6bx9PNnkJotQtcEBAlIKWBoEkQEy3KwvrcZd9yxuvpJdt/byyeOjsG2HCQbw9h6cxfBz+lcb+BR64xOnz5dVSNQyisQcV0Xw8PD122svA4gl/8ab3kACRzm0MxJns4NI6KHwcso7RIJ2E4BjnIgwdCEgCG9yiMvZOP1jGhaFBKAbecwPvQEpod/iERyE7d1vxdNnQ9CM5JUG6bCimeT15QD+88NxXpo1abPoG3NBzgzfRCp6cMoFsZg2wWAJEhGLzJrxAMkx7XguBU/t7F8P0hAYU1Nw9t61tG+0WEezuWhS1ntKmd4eQIhAUMCJzM2RkrgjQkdq8KEkCRiZtgKbCmG4zIqLmMkZ+PoVAWmFGDlTQ0s2gq5istxUxIRoPnn4MRYAZrwHHtjREdTzBMnDFiK8qXcx1MlHpgsIKRLVCwHbQ1hdDb5BQM1x+kq7/EnB1J84NQUpCCsWdWAO7a2kzfhUMwHvoXFBjzXaEkExKIG3XFLF86dn+WXJnIwdYli0UY04pX6lss2ykUb5bLtqQSPZHhyIo/TpyYhBMG2FTo6EzBN7ZLnthARsjMF1k0N4Zj5uiDJuXPnFnU+4+PjdW9+g1mdgdRY38Q+qEBfahnnbTtFdDXfjLbGLWC3gnxhGLlcPyqlSSgwTM30ejfg+npaGjQ9CQkXhUwfzqeOYuLs19Dc+SA39bwPodi6CycSrjTGXa3g8pLYutlALaseQMuqB1AuTnAuew6uU8bU5CGUrRxA5rLH5ioHjmOxoXky6hefwU5VmLm9cxXlrQHO2A4kSdjM0AWhKawj5zBy/jyPvMM4MKtwmBgamF1XeZ3lroLjKjiOC8dlGDQX2tIEIVt28LUDU4ibgqO6QGtMR6Hs4PxMGYZGsGxf/6pW8oO8yitXMb79yjAcV3klu65Cc9xEre5V4Pyl8IDx60/3QQhCpaLwzjt6IIXwwOhi4T2aa7RUypsjc34ozXsPjiAc1lGuOHjH/b248/ZuSCJMTRdw/nwKzX447ekf9eHs6WnEohpMQ4NyXLiuqn6+leBHoJt17ug4v/p0H971C7su6flXstMNAGTh7nZycrLucW9AFvOWBxBBAq6yMTB9Aro0PEayFPOwc9i58ePYufnT886oZed4ZvYoxkafx/TUXlhWGro0oUnDK3z1E9pCC0NDBI6VxkT/PyA19DiSrXdxY/cHEG25k+Ynz3nlYEJUbYILwluhSDuFIu0AAD3UxGf7vgUhQ0umRogIrnLhuva8MM4KgmteApUIt7S34/mhEf+phLs7GtAaNqnsKowUHR4ouJi1gAoDZdd3dAwI9kNXAhCSoBPm6V+xHwqrOIxc2YLjKhwd9UKHuvBCUpogzBYsfPfVcd61rgGxkITlKAzPlPDC8Un0j+UQ0iUcV8HUJc5N5PD84XG+aXUDklGdTN2bPpjKVfh//6QPZ0YyCOsSYA27bmoH+WDk6f8ESf+lO8IDBlAoWvzVbxwCM1AqOfjI+2/C/ffMbRqamiLYvKm1+ryW1ihGhtMwDB1EDMOUGBpIYXQozV09DcT+my9ZWumDx8jZGX7usUPY/cFtiDdd+8qtoDx3YGBg0TknExMTdU98A9pbGkCCMtvJ7DDPFsZhSh2AWiJsVURbwybs3Pxp4qBE17t1YOhx6my/G53td6NQGOax0WcxNfZTlHLnIODCkGGQkACUnyvRII0GgB1kx55BfvxZRBObObHq3Yh2PAgt1F7DStTc1vbinGBBt7t3Ezc1b6PRyEtcLKcgZGiZQBZ7wouXsUNhAM3hMHXGojycLyFq6GgKeTLnISmwPm7Q+jhQcJinKozJEmOqzEiXGWXFHmCwL1/Cc7ImNA/sAUMj6EJASc9ZOg5XQUaTAq+cSWF//6yXVLdd5Es2iBmmIeE6c3LwtqPwtWf6EdYFoiHJUVOCGBifLSCVLSNialWp+O+92I+H7ljNLckwmYbEQl1DtUDSOigeIAL+/tFDmJotghRj952rcf8963zlXe+DBNfIq2LymiGDIValsg1T1+DaLv75q/vx3g9t4y3bO6tzUgLl3oXgkUsV+dlHD2LLHT3YdHs3XWsl3ypYFgo8OjrqFVTUHFctA6n3gdQZyI0DIPCkuodTp2G7FYQ0c85hXxDesbC++4Eq8Ih5OYU5Zx2NdtOGjZ9B7/qfx+z0qzw18mOkp/fCrsxAEzqkDPk7NM9RSz0BgkIlewrT6cPInv0yQk23c6TjnQi37ILQk5dXweWDCfvz1mOxVSgUp4DlVF2vZECMz1i643EM58uwFaPouBzXtXmSVlGNKKoR1kYBQCJnM0+XFCaKDiYKLlIlBwVHwXX8c+p3I9KCt6r9XmthQ8JxXVRsj6FETQ2uOxcGmgMjQtTUYLsuZrIVTLou2PXyLRFTg+N6Mvu6JvC9F/rx9CsDaIia3NMWw6Y1jdjQ3Yju9hgaE6Elx+D+8/eO8bG+KWiSsHVzKz72gW3VKqrgngykVIgIxaLN58/PQtclbNvBPff2YstN7ahUHBx5dRjff+wwTh2Z4Nvetho965qq8FrVnBIEq+Lgx/90EMnWKO5675bXpWw2eI/x8XFMT09D1/UL/jY9PV0HkBsQVN7iISzvJIykz1YHMS1+g7gwtCi6mrf7zxJL7vy9m1lBCB0tbXdRS9tdKJfGeXbsp5gdewblbB/YrUDTDZAwPBBj5c8CCYOdEgpjT6Mw/ixkZBXM5js41vEgwk07iYQ+x0pWxEjmzDAbarrJl3cGl3sqGUCDacKQAhXXRaZiI65r1YTyQqdPAOI6UVyXWJfwALlgK54tuZgoOJjM2x6gVFxULK+c13EZrqvg+lK9+oJzwH7PBwRQtjy13oX6iYIITORVchFB6ALQANcRcB3Xe+1a0AvpcFyFyVQJY1N5vHJ0DKaUSMR0dDRFeFVbHKva4uhojiBkasjkLOw9NIJTZ6YRNjXEojo+/ZGbIfz8Ci3CYKQk/OSZPmSzZehC4P0f2IZbdq6qPnL12iZUig6fOTGB4fMzWLOumXfs6saqNY2k+bmc2ck8v/j94yhkK/jQr94F4c+Dv9ZtI8GaGRoaQj6fRywWm+dYpJSYnZ1FpVKBaZp1WZN6COvGABBBAswK07kRSOFXuVwIDVDKQSyURDTcOhdNWhaZ5TxWEgp3UFfvx9HV+zHkZo9wauwnyE++CKc0Co0IQoar3eAgCaEnANLgWhnkRp5AduTHMOLrOdb5ABKd74QWaq1WcHH1PZe/ITU9fMV0dTkmF9QyVVzXlwshzFZsdMfCi0D2AjZRM30vqguK6gI9CR1AGAyg4iiUbMVlW6HiKJQdBddlnJ0q4dBQDpqYw1OHGRXLhVIKqxpDsGyFyXSpCvlEQKHsAEqBmfykvQtWCiFNXiAJT34joxQEoUsIQ3i/Z6BsOegfTuPMYBrEXmWY5ivv6pKQiOjI5SzsvqMH8ZhJAdNYGPKSkrBv/xDv3z8MAeD+d2zALTtXkeuq6vWolG3OZkoIRw1oUmDw7DSG+qfR3BLlhuYY2HEwOZSBVXLwnk/tRLwxQq/XEKpgnff391dFFGslLjRNQzqTRjab5dbW1jpy1AHkzQ8gwW68YOU4W55dfpgSEZRyQSRorqv8UvMRXigpmAnibvoc56b2IDv2NMozr8Kx0tCkAQQ5CnYB0iH1EBQkrOIwpk9/BbNDP0By1fu4sfvd0MymGvn25bvdpTSrWlZLgg3RinpAFjlSAMBUscgHJ71QhSTCeLGCHc1z7ISW4oA0L4o2D6AJQEgTCGmCsAADp/M2u4phCK9zXbkKMUPDutYIblubxPbuJKUKFv/Bd07CcVyvHNhSuHNzC+7c2AJdCpRtB+m8hf7RDPadmITlzgFNsezAsh0o15s9QiAYkqBJgkYCmhTQQtI7Z74Ao/Cvu4DHlExTw7FTk9i2qZXX9jRQyNQuaATct3+In3jyJAjAxo2t2P32deSxEg+MhCDs/dl5TE/mEY97O/hQ2ABYITNbwuxEAcp2EI2ZeMe/2IFV65vpjZhgePr06UXBxROBLCCVSqG1tbXOQN6k4ao6gCyI94MI2dIMynYBhpSLakYxGJrQUCjP4OTAD3lH74cvQzRxvvouAEg9QQ1d70JD17tg5c9zfvwnKIz9BG7uDAD2WYjwH08gYUBqBpRbwfS5R5EefQaR5ls53noXoo03QWrRmlyJuoCVSGlctIaTIFY8lbAWECbyOT6bnsVksQwmCSF0EAipioNz2RL3JsI0j3EsYCMLWcipWYsrjsK6pI64OT9pU7QVp4o2jo0VsP98FmFDoGK56EiYeHhrCzobTESMueeEdAlTF3BdF6WKi529TfjFh9ZfcCLu29GBlniI/+knpxELa8iXbNy6sQWbVzfAcRSm0yVMpUrIZMsolCxUKi7Klg3XUTA0DRFDeuUX5HWak/CcpKELpLIl/M3X9qMpGeamZAjNjRE0JEMgBoaG0zjXn0I4JFFxXey6o2c+SxYEx1E4c2oShumFA62KA9sf2WuYGpKNEXR0JbD9zh40tce9vMfrCB6Bgzlz5syiYVApJYrFImZmZq4sTFq3OgO5fhiI58Cy5Vk4yvYrsNwlsEZB18I4cPIfUCxN8k1r3o9EbNUCh32J6rvVSi4BI7aWmjZ8Do29n0V5+hXOjzyB0sx+KCsNaBFARgEKQlwaNN2EcsvIjL2I9MTL0EIdiDZt52TrLsQathD5IMCsqj0iFwtxeTtFCSn1xWNNizCHsmPj8PgQj+ZyYNKgSx2qZhKgJggHpwuoKPDauElhKS7KQo7PWPyT8wWAFfYKQtwg1gWBlYLrKswWbOTKDiq2C0MSWHnH9/6b29DTFKpWKAVhp3TRQrHieIOdBOHhnR1g9hoGg3yu6/ohKuGxh2LFwW2bWvF/feLWCz6uYkap7HC+ZCOVKWFwLIvXTkyi79wMTL9T3uu+VqhYNsAEIbyxulPTBUxN5SFJgOAxFl2TCJtatX+klp0EbKxcsrlSdqqNhb2bW9G7qRUNTRFEYyYiMaOaB3kjdveBXPv58+erY44XAozrupiamqp73DqA3CAMxN/y5isZHwBwUf0oKQwc7/8uzg0/g86mbby+5yF0d9xN5DMFuqTE9hwr8Zy8N2Aq3PZ2Cre9HXZhkPNjP0Z+/HlUCiOeNLseB5GAAgOkQeommCQcO4PZ0Z9idnwPzFgPN7S9Dc0db4OmR6t9kY5T8gFrqSQ6QwoNmjRoOfwIoKhsW3h54BRnLQuGZvqqv4szjIMzRZzM2Jw0dDSYGppNiQaDEJYenFZc5lRF4WzaxqmZCoQ/LyRnKeQqCpaj4NguoBgELyke1qWfTGfEQxpa43Md6DVCvJjKVGA5nqZWR2MYq5oiXid7DbEh6VVFHT03Cym9RPcHd68D4JX7CkFVsUdBhGhYp2hYR3tTBFvWNeNd96zDH315Lx/pm0Q8YqJSdtCYNHHLbd2IhnUUihYy2TLKJRvlsoNMpgxdCkiqlYj38jFnz85gw4aWC867IEK54uCeBzbgbff30lJ5iNcbPALAmp2d5ZHREei6viiAAHOlvHUGUg9h3QAA4odErNwl6A4yTCMOZhfDE3sxNrEHnS038/bNn0Vz403+7vdSSm3ncg8UDGT1by49upoaN/wSGtZ9GsWZ/Zwbew6F1BE4VgaQYUDT/ZkbgXx7GIokrOIkRvsfx/T4y2hsu4OTTTcBJDA5vh9C6EuwEc9patKAVtWzWh5Cjo2f51ylCFMPw2HGcnJbpiDYzBgvuRgp+2NewdDJ68qr2ApF2yu1lQSsSehoj0hEdUJLREPFUUiXHKSKDtJFG+NZC7MFCxIe0EQNCVMTi0bo0gWvMdJxGauavbnmgTpv7S5/Kl3i8+OemGRPWwzruhLEDOiaWDT6GZQOKOWNu5XS00ezbBedbTF84V/uQlMyfMEnqlgOTvZN8z9/8xCEJqGYUS47CJkaQmG9KpVSeyyaLogEsRnScfOubq+yWbEvREDV0blvhAUAMjw8jNmZ2UUBJLB6M+GNy0DegvNA/OoWu3RJVY5BA6GhRyHBmJw5jBf2/P+wfs37edOGT8A0GmpyJHyJYFIjmuiXA5M0EW3bTdG23bBL45yfegWZ8ZdQyp8HQ/ozQYJBSR6L0bQwbCuLscGnMT76IhgaGMJvIuRF31YpBVOPLJtED2qtilaZp/IpGFKvVlwtZB21h6z8/+qCPA0rAlwFlBx/2h8DhvBKays2Y01Sx47W+dpNPcn5o3n/6cAE90+VwGAkwpofXvIa+GodfbZk+wCp0NkUnk+jahzga2dmUCw7cF2FnRtbPRl5xUtIWfsH6Df/jUzk+OiZKYRMDbbt4rM/tx1NyTA5rqoClVd9pWAaGnTdS44H1Vj339eL7ds6EI+biMe94659X9PUYJgSggFd97TASF4fSejAcZw9e7Zapuu6i4eC63Im9RDWDcdAHBU4GFwykDAYuh6FYBdn+r+JifGXsG7NI9zV9UDNYKnahPslhLhqyoGrrCTcQY2rP4SGng+gMHOQ0+MvIJ86DtvKAzLsK/B6YOKxEh2eMIsESFt2sBSzQsiMz3OqS0FI0S7DVQqeKnfNhD9mOKygSEExAYIhiefNPq89POnjpfJBRvmSJj8dKmL/aJEbQwItYYmOqI7miERE9+BhPGdxquRACoLjeM8jWoRXETCbt/yudkJbMgTF7IHegnX/yolJSEnQpcTOja1zndTzD3HBrHNASuAnewZQsVw4cPHw7nXoXd1Irs9MAiDzZMwFJqby/OjjR7xKLUH41MdvwaaNS5e2BtcimQxjNJOG6yjWDUnXUtfqcgDk1KlTS4Y7gscEOZB6BdYNENZCvZHwkqhXUALL/ghYj2aoKtMwjCQsK42TJ/4aA/2PoqlpG3d03Ium1l0wfFZSDXGBLkl9F4uUA8dabqNYy22wShOcmdyPzPSrKBXG4LoVsAyBRNj7bDSv1mnZxRANN6zoExlSr3ECXuLZZUbcNLGlpRUMIGe7SFVcZCxG0QVsX4K9diY61QyYqgUAQxDKjsJA2sbZmQrADJMYIUmsFCNVtMHKC3cZmsDATAkvn03x1s4YIqZGuiQIIoyny9w/UYChCRQdF4WyA0EEsWD3vu/UFJ8bz8GQArokrGqN0jydq8UUd9lr/pvNlPjlQyMwdIl4WMcjD26A67Mq9tkZEarg8bdfO4Bi0YahC3z652/FxvUt86RNaJH3IgJaO+I4c2ISuWwZoYiOFYuVvU7x8ZMnT170sdMz03UAuXEQpA4gK1/MBEdZCEkdQgCVShoaEUwZ8jSuqvNAdOhGEq5bweT4S5gZewGRcAuaW27j1s4HkGy5jYQwasCAryDx7gWGjHA7ta55P1pXvw+F7FnOzBxGLtOPUmECWOHgHgZDkEAs3LiiXUfcjFAyFOXZcgm6pkHBA4eEaaIrFp13MIqBouNy0WVUXKCiCCWXkLUZs2VGusKo+KGs4EwQAZoAhEZQ/jwQ12FkK16DoC4ICnMNeQzgiSNTePbEDCIasal5CerxdAll2/WARpf4/r4RnBjKcEgXYD+ElC/ZOHZ+1pvTASBfsvGDlwf4gVu7EAnpRJ6wI/tVXWQa0kt8E2E2U+a/fPQgimUHuhQIhzSETY3kghLafNHig0fH8aNn+lApO2AAH37/Vmxc30K27ULT5EXZROeqJBxHYWw4g9aO+KLKusFM8tdTLiR4rzNnzlTDdItt0oQQSM2m5j2nbtcH+NdDWFdgQd/DUlVYBIKrKmiO9+Ch234TggQmU8cxOr4XUzMHUbFSMDUdUviNeuxAkISmx+bmgQw/ienhHyIaW8MtHfeiqesdCMd7aU7+5HIS72IBEAlEkxsomtzghQtGX+DB/u9B6lEsS7J8vSxdCyMabqSFYZpFg1hE2NG1HnuH+lCwbeh6CLoQGMvnMFtq5KZwyNNE9AEypkuK6Yu9lkTGYp4ouhgvuJgo2EiXgIrDIH/GebDZEX6iWPlsZ2G1V1gXcBxGquLAcV2vNJe8UmKlGIKAkuVi3+kZuL7ulaeGwjAkQfry67om8K3nz+JHrwwgYmislIJlO3AdBhFxPCQRjxoQTBibziGbqyAaMgBmTKdK+C9feolXdya9jUbZQbnsYCZVQDpdRtiQMHQJq+KiWLJhOwq6X3671KTBYI20dyYQjRk4d3oKN+/qvuAaBdP/AiXc1wNIguubTqd5cHAQhmEsyeillMhmsyiXy9Vxt68HE3Fdt/pe9UmIlwcg9SqsJV0hYEhz2dwAEcF2K1jTfgcSUU8JNRZpR++qB1EoTvDg6As4P/A0rMoQXOV480CEVg031c4DqRSHMXL6y5g8/yiSzbdw86r3It6+m4Qwa1gFLiG8NRdCCp4fOPjWrnspk+rjTPoshBZdttJMKRexWAN0LbR4zOZC5opEKEK71271+kDyOWjSBDPw8ugYtrW2cHcsSlqNA1Nc8/yaOegNBlGDoWFzgwaXTcyUXB7N2RjK2hjP2chXvI5uwewtiEU+miDy8ieCIKSAJgAlGY7jzuvskUSIhjSwEmBXwVWAcpUnZVJzgiIhDZajUC6XvVyXz1aUArI5BUzmAQUYOiFsal5VF7z3H5sqYHgsD0FeubEkAU0nRCN6kNGCaWp48uk+vHpwhLff1I633bkasZhJtSGvGnwHGIgnQtS5KslD52aRy5Y5nghV8yDBPPJ0Os3ZbBarV6+mgA1cSxAJHPPg4CCmpqaWrMAKwCyXy6FQKHAoFHpdYljBeVn4ed9Kdq2ro97yDCSkR1Z8IZgZCi6EvzeORtrppg3/AhvWfBCpzFEeG/spZqcOoFIagwDDkCEfTLxciRAGNBkCsYPs5M9QmHwRkXgvN656HxJd75rTuAI8KZMVj7ydYyXefA6vryXRdBPSqb6LAAJBsYNkrMM/zpU1RTKAsG7grp71dHZ2mo9NewlShxX2TUzjxGyeWyJhdETCaI/MjbHlBUBUm6SWBLRFJLVFJG5tD6FgKR7L2xjOWBhOVzCZtbyJgQsqrUqWW3X0CABBAbXzosiXh1euJx3P/rwRteAGE4JgOwpKeUOubNuF67hQisEKEPCaDqOmBinnSoK9L68xkEyAwPBqDAhCeCBYrni7YQmClISxsRzGx3I4dGgM73hgPe/0xRODQVRVxVP/PdZtaEH/qSmcOTGBnXetASsFBc9J/u7v/i7/5V/+JcrlMm6+5Rb+n3/0R9i5cyddSxAJXruvr++iFVhBN3oun0dzc/M1d+bBZ/v+97/Pe/bswdve9jY88sgj9Eb1y9QB5IYDEG8BhY24H3pa3s0WKymvcoZF1cGyXyGl6wbaWm6jtpbbYNs5npl+DVPjLyIzexB2aRIshAcmgbw6AOmHuKzCECZO/DHS5/4R8fZ7Od71HoSabp0bLlVNnF9iiKs2X3LR/IdEY6LrouxjMSbCANY3tVBjOMqvjI6gogBTSpRcF+eyJZzL2YjoOnfHQlgXN5D0K6mCFPAFWliYSxxHDUEbmkxsaDLBHMdI1uJnT6cxkq5A93tLDI3w9g3NaI7q0IRX6lCsuDg5msPB86nqe1guw7EZuvSaCBUTSpY7r62SGSiWbIQNiUTCRFsyjK6WKBJRvz/DD3llCxXsOTSKXKECXXr9J6WKA/bFD01dQ8ggv6yYUCzZSEZN7NzWie6uBKQgTE8XMDiUxvRUAVNTBXzjm4dx6tQkP/zwZjQ3Rxa9COs2tSL0zBmcOjKOW+5YDfbB4z/+x//Iv/d7v4dQKAQpJZ5/7jk89PDD+Onzz/O2bdvoWjORY8eOXeCUg4704IuIUKlUkM1kXpewlZQSv/mbv8l/+Id/WP39r/3ar/Ff/MVfkFcRVw9nXRUAuY6bQl8XBhIzk/5sD15ypy1IolCavsDBUrVCak55V9fj1NF5Hzo674NVmeWZyT2YHv0JCqkjsO2CN1xKGgBcL/EuTEgZAjsFZAe+ieLwd2EkbuJIxzsQaXs79Nhamp84X3mIq1y8eOOWUi7CoUbEI610OTsz8hdRUzhMuzq7+KWRMYDnFpYAkLUcHJot4XTORXfU4K1JDUmdaPGWxgXS7zwHNt1Jgz5yczN/+ZUJlCwHtqvwvptbcfOq2AUf+pY1SVi2y4fOpyGJ0ZYI4YN3rEJz3PS7zYHJdAl/+2QfimXbu86C8Avv2YIdvU1IxkwKGUs7mVzB4uf2DUIPEWyXsXVDC1a1xqBJwqn+WYxP5RDWJSzLxS1bO/CR996Elqb5wKAUY3AozcePT+DkyUkcOz6OocE0du3q5m3bOtDUHCVNE1UZ+KnxHDRNYHa6gInRDHd2N9C+ffv5i1/8IhLxBBjejOrGxkbMzszg87/+eTz7zLPXbLcdvO7x48fn7Xa92SZF6LpeHS4lhECxWEQ6nb6mO+MAPP7pn/6J//AP/xDJZLL6Of/qr/4Kt956K/+rf/WvKHhc3eoM5DL5h19RFGqELo1F5zQE21IhNORKk3BdyxMlXC4PUQMmhtlEnT3vQ2fP+5DPnOKZkaeRmXgRdnEYkhhCC/nz05Un424kQaxgZU6gkj6GVP/XYDbs4GjHA4i0vg2a2bysYOLcjS3AykEufRpCLF3ySURQro3GZA+EkJek6bUwB8HMaA1HKKxpnLFdtEYi6E0mENIkig5jqOBguMQ4m3MxWCTc2SR4XUwsO3udvAjQvOkrUUNSY0TjTMlGxJBY1+z1djDP5Q4UK0gSaE+GwAxYrsKDO9pxU09y3lu1JEyEDMnFso1yxcUnHlyPh3Z1Uy0j4ZqEfSD3n8pV+ODJSZiGBDPwhU/fjp03zU2R/NOv7OexiSws28W6nkb86qdvp4VOM+gDWbumkdauaYQQxK/sOQ/lKrzwfD/27RlAY0OYI2EDkgDbcpGZLUI3JKyyg+mJHDq7G/CHf/AHcBwHJAiu44WPLMtCMpnECz99AU8++SQ/8sgj18RhSinhui76+vqqjEMIgXK5jLvvuRuDA4OYmJiAruvV5H4qlbqm4Zrg/X/v936vmpNxHAdCCJimiS9+8Yv45Cc/yYlEgt7qqsBXg5XyWzaE5a+buNmAkB6FZWchF2EiDIYUOnLFSaTzw9yUWEdzHeZYAZh4TjmW3Eyx5Ga4G/8lZ6b2IDP6NEqpg3DsNCTJqtw6AJAWhiAdLoDi7H4Upg9AhtsRbt7F8Y4HEGm6ef5wqZpekeD9sulTXCqOQ+pxjw3QUjecRGtj7yWFrxZjaUSEc+k0pysVrGtoxB3tLVR7c66Lmzids/nArAsG8NKUiwaDuNGgi4II4KnwFi0XpyZLGM9akAREDIGwIUlc0EXogUmmZFXnh4d06QNNAHrAmdEcT2fLkAREwxru3NJaLQ0mcaFEiFIeS3n8mdNI5SoQYHz6Qzuq4JHNV/h/PXYIx/qmEY9ogKswOVPA//2lF9mQojpZUQpP2FHAF10s2SiXbZim5lWtxUywUshmSsjMlryOd0HQpYBtK1gVG61tDZianuAfPfUUwuHwBbmHAKy+8pWv4JFHHrlmOYbx8XEeGhqaNyjKdV389V/9NX7/938fX/va12CaZvU8XktF3iA09fTTT/Px48eRSCSq50UphVAohMHBQXz3u9/FZz7zGTiOA03TULc6A7lsBhIyopQIN/KElYJXJ4NFd/S2W0b/6ItoTvbCVQ7kimeU18q4M6Qeo6auh9DU9RCs4gjnp/agMPE8KqnDcO00hBYFSPebAKX3f+hQThHZsZ8gM/ECjNg6jre9HYmO3TDCHfOGSwU2OfL88p+PCK5yEIt2IB5tv6zwVW0u49TMFB+ZmkZbJIZdPnioBb0KG+M6lV3iwxmvKutkxsXdrdoSu0nvuZMFh5/pzyJdtFG2FMqW4/WBMBDRJaQ/EnYx2JnJWZCCYDmMQsWZ96FJEI4PpuE4DKERYiEdyahBQSc/+30poGComNc42DeY4mf3D0IThG3rW/Du3esIAPrOz/Lff+swxicLiEUMz5n5AogjYxVowutNYcVQjhdqIuXtmHXhzRgheJViwm9b1aQ3/13TJHTNG1YVjRrYfO9adPTE6atffZzT6RSSySQcx7ngxjZNEy+//DJSqRQ3NjZe1R13AABnzpxBKpVCNBoFM6NSqaCjowNbtmyhrq4uXriuXg9F3m9961tYbD0Hx//444/jM5/5TL0fpQ4gV5oAUhAk0BztxGiqD6SZi6ZCmBUMPYqTA09ibcdd3Nq4ecFMkIuDyaIy7pFV1LTmo2ha81FUcmc4P/wDFEefhFuegtCT3shbZjA8GXehh8AkYRVGMHn265gZ+iGizbdyU897EEmsr94uk8PPcD59BlKP+Z9RLAqgrGy0t2ypmZ0uLgs8jk+N87GpSYT1MHZ1tHshLczXpQoevy2p0UDR5VmXMVFScHnxMe3Ba4/mbJydrSChUbXfw1VeWElKWjTsSAAqtsJktgJdEiwAwzNF3E0t8y7Ta2dmoOsCgoBMwcKJgTTv6G0iuciLBiq9X/7OES/Wrhir2mIYGM3wT/cP4/l9g4BiRCN6te8EgJd8lwTHdqFcRiSkobktgraWKFqbo2hIhiCIYFsOKhUHrqMghUA4rCEeMxGLm4hGDeiGhK5LRHyQA4BnnnlmWQdvGAbGxsZw+PBh3H///biayeMAQI4cPVJlI0op2LaNNWvWgIjQ0NBwwfOuFYAEJbvFYpFffPFF6Lp+AStTSkHXdezduxfpdJobGhqoPtzq6qyDtySABGjRkVyLw0PPYjkFWoKAYgs/3vd7uG3TJ3ndqgdg6NH5EiUrqpSqrY6aC3GZ8Q1k3vQbSK79BOcGHkV+9Ck4ldS8eSCoGS6laQYUu0hP/AzZ2cOINt3KRrgNldIM0qkTEDK07DxspRyEQkm0Nm2k+QB36ZatlMAAbm5vR9wwlr0pBQG9McJMGcjbjIyluMm8MBcSOOAtLSYdHNM5XXAgfBFGZk+YcSJroW+iyG1xPciYwFVeFdbLp2dQKDvQhUDIkDhwdhYdyRBv6UnCdRWeOzyG8VQJhhS+o2F86TtHsb4zwa0NITTFTTTFQ2hOmmhKhOC6jEef7kPfYBqJiA6WwHMHhvDcvkHYtkIspFcnCJIfxiQwCkUb4ZDE1o2t2L6lDb1rmtDSHKFAJ+vydn0M13Vw4MABSCmX3AUGTv3gwYO4//77r8nNfvjQ4Xnvx8xYv349AMwDkOC9A0HFq+20A3A8fvw4BgYGqiG1hc7ONE2MjY3h0KFDVx1U32xW70S/SmGsroZeaNLwZnIsAzZSaHDdCl45+iWc6v82VrXdzqva70Rr0zaS0qzJQaxUnmShNAlDC3dQ45YvILHuU5wb+RFyYz9GJT8EJs2fByL9UlcXgNfxrkDITh2AAoGEedHGQSIB17XQ1XITNM287OR5cIS3dvTQpmaLm8IR4iUWZlCeCwJ6woQjEihYwFhRockUF0hzBHqHIU1gTdLEVNZGaEFfh+0yHt0/AU2g2gfiOi4s24XtKBiagOt6oShXMR59cRAhXcBxXS/n4M8UYT8voRTjSP+M1/uhFKAAQYyQLuE4CoWSjXhEh+0qSHjMRQiCGTEANVeuKgXBshxIAu7Z1Y137F6LVR0JWujMVuLP55c5k99PJHDy5Gk+e/YsQqHQvJs4SFbX2tFjR6/6vRM43WPHjmFh0+KmTZsAAI2NjRfsUq+VoGLwHvv374dt24hEIheE9WpBdd/+/dcMVG80hrDca7ylGUiwiNsSqykeauJyJQ0hlinpZa9nwtDjKJZn0Hfuu+gf+B4aYqt4ddcDWN3zMMJ+M+CljbzFPGkSsII0W6ih99NIrv0YClOvcG7sJyikjsGxMiAZBmQURNJPkBOkFvHKkUmDe5EJJ4pdGHoMnW03X5WbOaRpCGnakoOoqj0fNOf8CZ7e1cm0i01JCV3QvFoxIk+LeDBj8emZMgxtTvuqlqXoGsFxFZTr5S2YPfkSoQu4rpoHOJGQBsevVAobGhy/mVD4EinMDFOX0GXQF8RwHQXbdry5IyEduUIFjXEv1Om6gTCk14lO8AAlV7CwpiuOn//ANmzubaZawAgS8/PEGi/BXNdz1Hv27EGpVLog/6GUmmtA9H8+3Xd6ntO/Gg6FiDA9Pc39/f3V3X7gTDZv3gwAaG5ungdoQghMT097vUdXOf8QHPPevXtX5AwP7N9/TYDsrRZieot3ons7OlMLo6uhF6fG9sCQkYtMJvRuFCl0aNKAIIVCcQzHTv09Bga+i55V7+A1q9+PSHTVvPCWxzZWNj8dQTUYK5AwEGu/l2Lt98IqDHFu8mXkpg+gXBiG65R8MAn78vIE+CXBy7Eu16mgq+tt0PUIXS77WAoklvp93lY8VHSRsoHpCsFWgEZA1lb4yYjFd7frSPrt6hWXMZ53+MRUGadnymC/+1wRLpT5AEESgXwAYia4wAW7eyJvqJTyy7VVtWcBKFVcKOUiamheqFK5sGyFfNH2wMivyCoUbDx05xrs2tqOP//ng/MyS4GjzBUs3H/nanzyka0UMrV5XeVXw1cFDu+FF15Y9GY2DAO2bVfDSbqu4/z588jlchyPx69KzD8I+/T19WFqagqRSMRTafAT9xs3bgQAtLS0VPWxAi2q2dlZFAtFjkajVzX/EITyDh06VGUZy+VKDh8+DNu2q6W+b8U8yNUA8bqcu+9217fegpOjP6uGtVb0TH8miCYM6DIExymj/+yjGB36ITo67uFV3e9CsnEb1XaEX0quZA5IPDdsRHuoeV0Pmtd9DOVsP+dnjyA9uQeV0jQgV3K6yMt9hBvR2bHTL0e+OjfOcuBxLlfhV2fKyLsSLiR0qUMnwPVzGaN5hW/lS4hrYMGMguUiU3bgulwdvMQBw/Cro5gB22FYtgIrNU/KRMCT/lA14FG2FAwhYRoaciXLYwz+7zd2J/C+O3vQ1RyBLoU318RRmEqX8Mf/fBCFkgXbdrGqPYbbbmrD1394Ao6rYOoE+Bpctu1CAPjMh7fjnXevrUqSLCaQeCW7SSklLMvCvn37qk5T0zRkMhn8+3//7/Gud70L73jHOxAOe8OzdF3H+Pg4zp07h5tvvvmqOMtgV3vo0CG4rgshvDySbdtoaWnBmjVrAABNTU2IRCIol8uQUkLTNKTT6WrV1tXcBQshMDIywufOnZuX/6g91gDkQqEQzg8MoL+/nzdv3lxPpF/HDOe6B5Bg972+fSdiZiNct+xNzbtIGIhIVPMW3pwQF4IkdCMJZhsjg09icuRpxONruKn5VrS034tE0/bL6CqfP6WQ4c0DCSXWUyixHi1rPogz+3+HS8UJkBa+6O7VcS2s697tix9eHfaxHCMZyJd5z2QBChIdYYnmkIaiSxjMc3XIlBRASBAkMbIVhbylYEqCxYzephDWJHUYfpmrIQmmLx+SrzgYSVUwnqmgYjsAA5btYipTxkzOgiG9U1exFdZ3xPEv7lmNiCHx598/iZHpAsBASyKE/9dHtpOpX8jazgxnuFTxZohIIZDNW/if//gqiBhh3Su11qRAoWijpTGMX/vYrdiyvpmCSYZXEzxqQ0enTp3is2fPIuyr2gaNg7/+67+Orq4u6u3t5XPnziEcDkMIgUKhgBMnTuDmm2++KgKLgbM98Oqr835nWRbWrFmDpqYmCnIgiUQChUIBmqZBSolcLofJyUl0d3dftZ1/4MSOHz+OdDqNeDwO13WrPSnBMUspqyCczWZx8OBBbN68+ZqLTt7IVp8H4oexYmYDrW3dzieHX4RuROcc/BKsxXZKMIQECc0fLsWYK9El6EYCgl0UcudRTJ/ExPlvIp7YwC2dD6Cx8wGY4c4FFVxyJXdudXY6KxckNOTTJ9iuzHqijcsJepGA45TR0LARrS3bKJCAv5bgMVu2ee9EDpIkbmsOY2PCpMCnHtbAr84oEANJg/C+NSEyBMFyGd8/m+eBdAU3t5p457r4kh6mNaZjXfOFoFm0XH7m6CR+emIKIU3AshXu3tyCjgZPBbYtGeKBiTwkEcKmrI6ddVyFdN7igfEsXjw8hj1HxqFpBPLDXpbtwtCFPw/EO9BMtoJbNrfh1z5xK5qTYXKV1/R3rW5WIQRefOkllMtlJJNJMDMKhQLuve8+dHV1ETNj8+bNOH36dDXXAgCHDx/GJz7xiasW+mBmHDl8uBouCn63ZcsWP1fjIhaLUWNTY7XRUAgB27YxOjqK22677artXoPXOXz4cBWUAkDr7OzEl7/8ZXzuc5/DyMgIDMOonpM9e/ZctXNyI4ewlp9k+hZnILUOb2vXPTg58uJFHy2Fhsb4WpRLkyiXJyFJ+cq7hifO5w+YAhhCmtBkCIIdFDJ9KKaOYuLs19DQehc3d78XsZY75rOSizr1oFNaQ2r0GR47/TUwNJA0l+VMzApSGli79qHX5ZxarsKeiRRcBt7eEUN31PDCOn4ieUejpIG8y1MlBZe9PIZij2G0hCX6Z4FNzSEw/A5wWjzPwtVJ7YF2FiNiSHrktk6cnyrw4FQRpi7x4okpbOiMc7ZooW80C0PzlItHZ4r43a+9ysmIjnSugtlMGdlCBa6rEA55VVqqZpcdTE8slG0YmsBHH96Ej75rCwVVXNcKPGp3/s89++w8RsLMeOD++6uP2bx5M77//e/PS2AfOXrkqsS9A7AYHR3ls2fPXlAuu2PHjiqAGIaB9rb2ajI/+PyDg4NX1flUQfLI4Xk5EcuycM899+DBBx+kRx55hP/0T/8U4XC4+nn27dtXfWzdLupy6gCyJBL7TntNy3ZqinZxoTQJXWoXZGKJJGy7iM622/GOO/8TlSopnpo5jInJvZidPYxKcQKG1KDLEAjKk2T3w04AQ2hhaIiAlY3Z0R8iN/YUYo3buLHnw0h0vcsDEj/hjsVKgasjagkTfX/PM0NPgvQEQHLZng8iAdcpYe36DyIcbqbXI3R1eCbNM2ULuzub0R01yFOm9YHAB5GNCYmpkkK6onAyZfO2Jp087SqGFN5o27lhUksdnj9ACZ6cem2mOqRLKMUwdcLAZB5/8PgJWJaDkuVCCsB1PWXe0ZkiTg1WvIFPQiDiz/hwHVXNoUhBcBSjWLIhBGHnpjZ87F2bsaGnoVplJa4heAShl1wux3v27IFhGFVHKITAPffcU33s1q1b5z1P0zT0nZqTXL+S0FHgMI4dO4bZ2dlquCj4LDfffPO8x3d1dV2Qizh37txVPTdBaKrvVN8FJcW7du2CUgrvete78Gd/9mfVajHTNHHy5EmMjo5yV1cX1cNY9RzIle2sWEGXJjZ23IG9Z74FUxoA1AUnSwgNhfIUXGUhbDbS6q77sbrrflSsNI+Pv4TBge8jn+mDJjSYWshnJG4VADwHK6DpCRAUiuljKM28iuzAN7j1pi8g1HjLPHFzZjUPwMAKo8f/hNNjz0GaLVC8fL4mAL3m1pvR3rHrdQGP6VKZ+1I5bGpIoDcRJub5XemBL1kbE3RwGlxWhFcnLXRHJSdNQY4XDYRa5n143jx17wUtR2EyW+HRVBmnx3Lon/TmoSulYOgCxYoDsCcRohwvLOW6QDJq4DMPbYAA8LWn+rykfQAI7M1nL5UdhAwNu7a24z33rMOtm9vmJcqvdRI2qHzau3cvhoeHEYvFoJRCpVLBqlWrcOutt1Yfu3nz5mqCPXCWw8PDGBgY4E2bNtHVAJD9+/fPCxc5joOmpqZqCCuw1atXX/Dc/nP9F4DKlTgwIsLs7CwPDQ1Vq76qJcVbNkMIgdtvvx2tra3IZrPQNA2GYWBmZgYHDhxAV1fXDd8Pstg1rFdhXdVciL97W7UbB88/AQW1iACIgiZNpLMDmJg5yp0tO4nZAZGEaTTQmtXvR0/3uzA++hwPD34f+fRxCHZgaGE/x+HOy5V4Ia4IpIygnDmO0b3/Gsnu93O042EYiQ0QenJeBZdTmeGJE19CdvoANKMJSrnLluyCCMqtIBRpwdr1j+BqVl0tZ8dm0ghrEre2+Oq3iwo5AiFJ2JSUeHXKhmLgqcES3tkT4pKtAAKcmr6PC0HD+32h4vLZqSL6xgsYmikinbdQtryEui7nQIjZ05myXFXt2/A2DgxD0zA0WcDJgVS1l6NSUShXHACMzqYIbt/cjrff2oV1Xcm5lBeuLetY7Kb/4Q9/OI95WJaF22+/HclkkgJxwN7eXjQ2NqJQKEBKWU0aHzt2DJs2bboiZxk4nSD8UzvrY/v27ejs7KTacbpr165dsAETGDg/cMG0wCsFkKGhIczMzMAwjGoILRKJYH2v1xXf2dlJ27dv52effXZeHuSnL/wUH/jAB95SAHI1WUadgdSEeRiM1sQaWtW4mUdmjiKkmRc2FIDAYPQP/QRdrbcBkP6O3h9hK3R0dT9MXd0PY2ZqH08MPYn09F7YVhqa9CcSzrlEn5UwpBYBsYvcwDeQH/wOZLgDWnwjG4kNEEYLnMo0shM/g11JQdOTUH4n+nKQGAgyrt/0MWha5JqWKwbsI1Wu8Gi+iDs6WmHKCzvMa1kIA9jWqFFf2uGSzUhXFB7vy3vqs+TN2agF+OB1SrbC+Zkyn5oooH+qiHTB9spZ/RnoUb//wlmohcSMhqgBZoV0tlJtOswULPzk1WGAAcf2cletyRC2re3Eri1t2LquiQxdVm8YT+drpWoDVy9MY9s2nnrqqSq7CJz0vffeO+9mbmtro+7ubj58+DCi0Wj1mr/66qv4uZ/7ucu+6QMAKJVKOHLkCHRdrzIjpRRuvfXWKhsJ3nPdunUgP9Ee9KWMjI5iZmaGW1parnhNBscyODgIy7KqnfmWZaG7uxs9PT1VkHnggQfwzDPPzAtzvfjCi1cNzN5sdjUr4N7yAOKdDK9Edmv3fRicOghCaNHH6FoEo5P7kC+McSzaWQ0L1SrvEgk0t95Bza13oFQY5unRZ5AaexaVfD8EO9C0UHV2upcr8fbKQk8CYCgrjeLUKyhMvQyGBhYaoMUh9KiXoCdtWfAgApRbQe+WzyISW3VNQ1fVLT4RzmUyiBsaepMxX9ZkedQxJeH2Vh3PDJVhCsBRcwDTN1PBmqQOECFXcnk0a6F/uoSB2TJSRctrMAQhYkgwU1XKJAhB1YKV5Sh8YNcqbFvdgB+/NoqXUhNeeIsJtt/d3ZYMYfvaRty8vgkbuxsobGo1VL2mIfB1XpfBLI8DBw7w8ePHEYlEvNCBn9/YvXt3lR0Ej920aRMOHjw4L5F+4MCBKwpdBI749OnT8yTcA7vrrrsucE6rV69GPBarzuTQdR0z09MYHBxES0sLrhaADAwMVN83qPZau3YtotFolZm94x3vwH/+z/+5CmahUAjHjh1Df38/r1+//i2XB6kDyFU2Ue0JuQMN0XaUy2kYF0ibeDM0rEoGpwd+gJ1bf3lRNhMACQCEo93Us/Gz6F7/SWRnXuXU6NPIT78CpzwJXRqADM2VDQf5EtIhdN1LkEMDk4SC8F9TXiQcR7DsArrXfxSNLTuuPXj4i9FRCkO5PDY1NVaHTC2HIOQn1DcmNTqf0fhM2kKICLZiGIJwJlXB1BGbmRnZoo1ixa3KpId1b76G63izyyu2C8dWkPC61p1Ajt3HNl0KvHRiGs8cHkc6V0HIkKj48vDrO+J48NZO7NzQMm8KYe1skNcrVLXcTfqNb3yj6oiZGeVKBT09Pdi+ffs8AAG8aqhHH3202gthGAaOHz9+RR3pgYN99dVXUalUEAqF4DgOHMeBaZq48847q58jeO3Ozk5qb2/n8+fPwzTNKpM6efIkbrvttqvWg9HX13eBYwyKCYLzt3PnTqxbt64qtqjrOjKZDJ57/nn09vbeUP0gtXI2b9UQ1htwJQmKFUw9Qhs774bllhYXBmQFXYvi/PAzKJVnmIgWrZX2WImolvWS0JFsvYvW3vIfaPPuv0P3jt9CKLkFys5jsQJVsAKzW/26WHOjV70lYNs5dK15D9pX3fe6gEewiKZLRS47Dpr9LugVaXf4oax7u0xqCgmUXa7Ku2tESJUczBQdKPak3COGhCYIrgJKloui5UISsLY5jHfc1ILP7O7Bb7xnA3ZvakbZdquOnwhI5S0USg7ChkSh7CAW1vBL796I3/rkLXT31nYKGV7VlgrCVIGMyRscf9Y0DcVikb/9nW9XZcqDXfbOnTsRjUZpYansLbfcMhdy86XdR0ZGquNnr8SxvLxnz7ycSKVSQW9vLzZv3lydKxMwn1AohDVr1lTlVQI7cuTI1XES/mueOnXqAocWnIMARCORCO3evRu2bVcrtwDgh08+WWUubyW70Y/3DTm6wFls63kQphaBWqShMKjGKldmcfrcd30vyMttB+aFt5gV9FArNa3+MK296y+oad0noZzilTn6oBnOzqNr7SPoXPPe1wU8ai1bqXjlr75zWnhKeBEIDM53SCO8d20ETSGBgu3DMQG6JBh+57mjgKLlomi70CRhQ2sEj+xowa/c24NfensPPbSthTZ1xsjQCMOzJWj+Tr1KaSVB1wSyRRs3r2vEb33iZuze1u4NkVI1oEFvLGgsDF8BwBNPPomzZ84iHA7PC/s8/PDD8wAh+P327duRSCSq+QgpJRzHwSuvvHLZO0dN0+C6Lvbt3TuvgdBxHNx5150wDKPaAV772QNxxdpw2uHDh6/YiQU5mXw+z6dPn67mZJTrQtO0aklx7fu+5z3vmceoTNPEiy++iJmZGRYL1kvd6gzkMvywFyZqinXT6tZbYDmlRZ0wswtdi+Lc8I9QKk9zkIRfyet7z3f88IgGLdTq94rQZX5mCVYOlFtB94ZPoOMNAI9q8IyA89m8HxKcDxpLpZ2DQuSEIejDG2N0a7sJTRAqDqNkM0q2C9tlhHTC5rYIPritGf9q9yr8/K522rU2Qc0xnRzFGEmV+cdHp/hPfngW/ZMF6JqYB2JEhFzJxjtv7cT/+YGbqCFmUBCmeqOZxsV2iX/9V381jw3bto1EIlF1iMHjgu+rV6+mdevWoVKpzHOggQjj5YSvAODs2bPc19dXbcgL7B3veOeSDmW731wYvI6u6zhx4gQKhcIVOe3geadPn8bo6ChM0xupYNk2Ojo6qsAVyJgAwIMPPojW1lZUKhVv4xIKYXx8HD/+8Y/BzBcMobqh7Sos+HoOZBm7ec27MTD+ypJnWggd5UoaA8NPY8uGT/qNfkvlJ7iapPccu4BVHOGZc19HZug70LWYH6a6dPBw7CKE0YDVW34RieabX3/wCGTxIxGYUmI0X8RrUym+qSlJoZrBSTlbcc5hdIYvnPkXgIgpCff1RGhXR4gnCg4KFReCgLgp0RHTSffjW/mKy6cnizyetTCRKWMiU8HwbAmFoo2YKbwZHjXOQAhCNm/jXTu78HN39xD7ApVvZG5jJexDSomf/exn/OyzzyIajVZ/l8vl8P73vx9r1669IPkbPGbnzp1VdVrXdaHrOvbv3498Ps+xWOyS8iBBiOyFF15AoVCoyshbloVEIoH777vvAkYRvPaO7dvnDb4yTRMjIyPo6+vDzp07LzuRHhz33r17YVlWlZ1VKhXcdNNNaGhomHdulFJob2+nt7/97fz4448jkUjMyy994hOfqCfRbyDT3rgT67GJ7pYd1N64iafTpxHWwtVEt1eJ48mzK2Ujlx9eFjQCKfcgjJWfPcypkR8gN/482E5D12IALhE8fEVfx8og0nQzVm35JZjh9jeEeQTOvyEUog0NjXx8NoPT6RxGCjY3hkyABFIWUHQFdjSa6AwvLktSm/SO6ILWNRj/f/auOz6O6lp/d8r2XfVuS7LlgnvBxgYbDJgeqgMEQoAXIARISIE8CIGElkeAUAKBhOLQQ4AECMWATTFgg3vFNjZusmT1ru27M3PeH7t3PLvalVbNlmEvv0GytJpy597znfqdmN97giqtrfZiV5MfTZ4QfAEFYSUSK9I0DdNKXTiiyIENle3Yvr8Tkhi5s0iPjjBmj83FeUcPZxovgDtMNsI999wTw3rLNb8rr7wyRpDGa4Vz587Fc889p//MYrGgqroK69at63U3Ph4Y/+CDD/TzRVvIYvbs2SgrK4up/zCCyRFHHIH8/Hy0trZCluUYRuFp06b1OXit13J8/rl+T/xnPCMsHkAEQcCCBQvw5ptv6rERq9WKjz/+GDU1NVRSUpKuSk+7sAZk1sCYgKOPuAQgQlgN6PUeihpCKOxBINiGDGc5KsrPxgGRSNGgNweOiMUR9NdTfeXr9PWKn9E3q36J5qq3QRSGFE3b7Z2bSISm+ECagvyR52PE1JvZoQIPo/BXNQ3jsjLZhOxMmEUBnrCCSncQ+70huGQBJxdbMS5D6rGCgteIEEW4s7QovclbW1uxZEc79rQE4AupsMgi7CYRVlmALDJkWGW0ecNo6gxCFCNxKYExBEIqSvPs+MGxZSxSjDj0wUNRFIiiiHcXLaJFixbB4XDoIOLz+TBlyhScccYZjAfZE7m9jjnmGFitVr3hlCAI0FQNH374Ya82PweGlpYWWrZsGcxms26REBHOPPNM3fKJF/BEhJycHHbEEUcgFArFCOYvvviiz5owBzCv10srVqzQ4x8cRObOndvl3BwszzjjDBQVFSEQCAAATCYT2tra8Nprr8W4677t1ofAvt0gKR3aCY7EQopzJrKTj/wNrf76eQQCLTCJIlyOYchxjUBx7lSUFM6GLNmYsc8HtzSCgRZqb16HloZl8LRughpsgSzIkEQLRMkMRLOrWMr3JIJIhaL4Yc2ahIJRl8GaMZpzvR8y8OCbThQEQAAm5mWzEZlOagsqEAURGSYJNiniKyKk2uw38j/+WZPIcN7EHHQEFATCGr6q9eCrGg8kFgEYUWD4/JtWKGEVJhF6GrEWpS754XHl0boP0tl3h+rgGnBnZyf9+te/0qurOQgoioKbbr4ZJpMJvMYhHkCICGPGjGHjx4+njRs2wGa36+f98MMPcdddd6VsfXCX2AcffICGhgbdfRUOh2G323UASaS1q9GA9qxZs7B06VK9iE+SJKxevbrP/Fz8WVauXIl9+/bFULsUFBRgxowZCV1qqqoiOzubnXX22fTUk0/qhYeiKOLFF1/E9ddf/62yPga71iMdA+kRRAgji45mpfnT0eGpIVmywGHNZ4IgdRHuAODz1VFry0a0NK5CZ9tWKIEmSIzBJJohmzIhRAsHOZVJysABQA21Q7QVo3DU+cgadhpDFORSa1A1uCZupE/3dtq5cyeGDx+OqVOnMrssG5x5sW6qPhiEsJkEZjOZ4AuptF6lCMli9HeKSjBLAqwig6apUNQIULiDCs6eUYTibCs7HMCDZzEJgoCfXP0T7Nq5SxfYPPZx7LHH4gcXXsi4IE52Hl48t27dOh14rFYrNm/ejK1bt9LEiRNTctdw99VLL72kWxX8Xk488USMGTMm6Xm4AJs3bx7uu+++mKZOu3fvxldbvqIZR85gvXGnGc/93//+N4baJRgMYubMmcjJyen22a668ko8+8wzUFUVRAS73Y4NGzZg0aJFdM4557BEwJzKPuBWGL+ukZdLd38bjsF2O7Fu67C+3TGQIaEGRDZMhAMrJ2Mkc9mLY8BD0xS0deykb3a/RitW/Za+WP5zfLXpATTUL4eq+CCbMiDJDkSEvdptn5GuF49sKDXUAcYEZJV/H2UzH0TW8DMYog2t2CE2Q/nmvemmm2jSpEk4++yzMW3aNCxYsIBaWlspQsdO6C/xB2NAkydMS3a00VNf1mNbvReKRggqBJPEkGmT4bCIUDQNQYUgCEAgrKI834Z5E/KHLHhwgaqqqq7pC4KA6667jl579TUdPLjmLssyHnnkkZg6hu6Ew9nnnB0TwJYkCYFAAK+88op+zlQAbdOmTfTpp5/CZrPpPyMi/OCiH3Tr9uGCdNasWSgqKtKzwnhB4Scff9JrTZYDmMfjoUWLFum1MfyZeWZaonviczFz5kx24okn6nxhnGblD3/4A0KhUEpzY5wjfn1JkiBJEgRB0LO/+M94Uy1joaWmaXoxpqIo+rm4Oy6VY7Ctk8M1tVkaCps70gEwVkgHQ53U3L4DDU3r0dz6FXzeapDihyxIMIkyTKYMCNDAeCFgwgqIbpw30etp4Q4wyQnXsO8hc8SFMNmHR4n8ovd0iMGDC7wnnniC/vznP8d0fXvzzTfR3NyMDz/8MNp3uo++bkSqyj/b00lb63zwBhVoGiHTKmFUrhVjC2zIc5hgkQWmEaGpM0hvr6tDbZsfAHDmkcUQBab3QO/Vu4/bpMb77+lZEm066tIeICJIjedav349/e53v8PixYv1Og4u+Nvb2/Hggw9i2rRpjM99ssHfw6yjZrEJEybQtm3bYLVaoaoqLBYLnnnmGfzqV7+injR1fp8PPfSQ3sRKVVUEAgEUFhbivHPPi4kvJPpbVVWRlZXF5syZQ//+979jXFbvv/8+brrppl5ZH9y6+uCDD7B37164XC6oWqSlrtPp7JLanOw93HTTTVgSjQdpmgabzYbNmzfjN7/5DT366KNMB3VB7KL9aJqmu774vdfU1NDSpUuxatUqVFVVIRgMwuFwoLi4GKWlpSgtLcWwYcNQVFSE3NxcOJ1OxoFmIJQQDqyp7rM0gAwGaCBK5c3NzChtiNvXQHUtX6G2aT1a2rYj6G8CIwWyJMMkyBBNGRBBACl9AA2AZ2qBFGihTgimLDiHnYmMsgtgclZE+Re1aFHioTfOuNsqFArh/vvv17UqbsKbTCYsW7YMixYtogULFvQo8JK5rRgDPvimg7bV+yAxwGEWMb3EjqklDjjMIosHm2HZVnbsETn01Md7MWdMDkYVOnptfXCBejBMfE3TsG/fPvryyy/xxhtv4P3334ff748IxehcyrKM9vZ2/M///A9uuOGGlN0rPHX3Bz/4AW699VbdjWU2m1FfX48//vGPeOSRR7pUiccrCGvXrqVXX30VjiinlSRJ8Hg8OP/885Gbm9vju+UC6Nxzz8W///1vnXDRZrNh5cqV+Oqrr2jixIkprxF+rwsXLtTfkSiIcHvdOPnkkzFy5MhuQZFbISeeeCI783vfo3feeUe39FwuF/7617/C6XTS//3f/zHjXBj/3ij4P/vsM3rmmWfw/vvvo6mpqdt7F0URDocDWVlZyMvLo+LiYpSUlKCkpARFRUUoKChAbm4uMjMz4XQ6YbVaYTabGbdeEu1Broj0RuAPtgvtOwcgpKfoHhAcbZ4aqmragP2N69HSvhPhUDtEBsiiCSbZDhGAwFQ9GN4nnGaRznikBaCGvRDNWXCVXwhn2QWQHSNjgANDKGuCL969e/dSZWVll0XLN/DWrVuxYMGCXmsxvAHV1sYAbW8OwiwJyLaKOOOITOTa5QPdDbnRZgCc/S1+WE0iTplcwKG5188FABs3bqR9+/bpbg6TyQSLxQKHwwGbzQaz2QyTyQRJknSXjqIoCAaDCAQCCIfD+hEKheDz+eB2u9Ha2or6+nrs27cPu3fvxt69e9HW1gYAsNvtOnhwl0h7ezvOO+88PPXUU6y3qbcAcNlll+GBBx5AIBDQK9IdDgf+9re/4ayzz6aT5s9n4XAYsiFmxd03iqLg5z//uQ48qqrqwv9nP/tZSposv9/TTz8dxcXFaGlp0dN5vV4vnnzySTz22GMpWx+CIGDt2rUxtTGSJEVcaj/4Qcz66+ld33ffffjkk090N6GqqnA6nbjnnnuwbt06+u1vf4tj5hzDTHJsSvmuXbtoyZIleOWVV/Dll1/qll1GRkYXIc7nh1uzvK1v1b4qrKE1XQWfJMFiscBisXAAIeM6M74jURRhs9lQXFyM2bNn47LLLkNeXh4bSKshbYH0aG0Iulbf4W+iPQ3rsKd+NZrad0IJuyEzESbJBLPJCQEUdU1FaNipT7ARzdQiNZqOG4TFWoSM8h/ANfxcSLaSA8CBoQUc8YuqpaUlprVqPIj0dfEJDAiqhNU1XggAsmwSLpyczSySoLfFNdYAcheZN6jQ8m9aMG9cLrIdpl5ZH1zofPbZZ3TjjTdiy5YtesVyvPYmSRLEqFYoCoJeDMPdHoqixPixuxPyZrMZLpcrxp8uSRJCoRA8Hg+uuOIKPPnkk4wL4lQ1R143MmzYMHbxxRfT3/72t5iYiiiKuOzSS7F06VIaO3Ys4+4yLsQA4Nprr6VVq1bpfydJEjo6OvDjH/8YRxxxREpWg9GNde5559LfHv8bLBaL3rPjpZdewo033kiJiiKTne/+++/Xiwe5S624uBjnnntuty61+LkZN24cu+uuu+jGG29EZmYmwuFIawCXy4XFixfjo48+wvjx42ns2LFwOp3weDyorKzEN998g46ODgCA0+nUYyac7LI7DZ+/c7PZ3OVz3B2laRo8Hg863W5QD+uIr7P//Oc/eOKJJ/Dhhx9SaWkp62nvpl1Y/YptcOBgULQw9jRuom01y7G/ZQsCwXbIggCzaIZVj2eoOmiwPoOGEOkmogagKD7Iogm2zPHILD4FrqL5EE1ZUeBQAQhDEjjih8fjidmQffWzJrI+tjcFqNWvQhKBk0a5wMEjUfE4gSCA4dOvmyEyhhMm5EdShnuxSRhjaG5upksuuQQ1NTVwOp36Jjduopj4iKZBMQRbGZgeOE0UM4kXFvzgyQiSJEFRFHR0dCAzMxP3338/fv7zn7PebnrjdYkIN910E15++WW9FoPzQDU1NeGUU07Bc889RyeccAIzvFe68cYb8dRTT8VwaoXDYWRkZOD3v/99r1Jv+ed+evVP8cw/DmQ/cUbcP/zhD3jxxRd1C6O7mNsXX3xBb775ZheX2kUXXYTs7OyUXWGiKEJVVdxwww1szZo19MorryArKwuhUAiqqsLlckHTNGzbti2G/FEQBN3a4MKeZ4ERkW59JhRqhmB6fLyCrwUO7nweUpljxhgsFgt27dqF559/Hrfffju+62PAASTG4gBDZ6CNNu9fji37l6PFXQWRESyiGVaTC0K0p7lGKhioTxlELOqeAqnQtCA0NQASRNjsw5CZdxSyik6ALWuyYQWpEdBgh09zG16MNZCD75ddbSEQAXl2GSUuEyMkAY+o9dHmDdOyHa04a1oBbCaxV9YHd121tbWhrq4OTqczJtOpNwH0eK0t/vt4MOGuLz6XTqcTl19+OW699VaMHj26C9Nuryy5KLCXlZWxm2++mW655RZd0+YWQF1dHU4//XScc845dMwxx6CjowOvvfYatm7dGhOL4e60Bx54ACNGjOhVXIvfx+TJk9nZZ59Nr712IMPM6XTi5ZdfxiWXXEKnnXZawhgPn0NFUXDDDTfo70tVDwTPf/azn/W6noSD6TPPPMPa2tpo8eLFOjDw57bZbDEMw/zgYCdJku62ZIxh1KhRmDRpEkaMGAGn04lgKIjamlpUVVehvq4eLS0t6OzshNfrTQoyHGCMsbjunos0QmtrKwBg9uzZAxoDSSXb71sPIBppusXR4m2glZUfYUvNCngDrbCIMiyyHQIjsH6BBjOARhiK4gfTQpBFE+z2YcjKmYas/GPgzJ7EBNFieEHRIsRBBQ6jCcyiQrr/jXyMbo+BucvIXfkVDe2BCA+WSWQ9KgYCGN7f3IgchwmzR2VH+7D3XpCMGjWK/eY3v6H7779/AAGRdbsJZVlGdnY2Zs2ahZNPPhnfP/98HBGlRu9L8kEy4f2b3/yGLV68mD799NMYEOHFdK+99ppejS1JUsJA/mmnnYZf//rXrDtLoad1c9ttt+Htt9+OsVhlWcbVV1+NlStXUnFxcUxMhgtzWZZx88030+rVq5GRkQH+mY6ODvz85z/HyJEje52swQWg1WrFf//7X3bNNdfQ888/D0mSdOAwgoaR9j0cDusWeF5eHi66+CL86JIfYfbs2bDb7QkXXygUQmtrKzU1NaGurg61tbWoqalBbW0t6uvr0djUiNaWVnR0dMDj8cDn8yW1ZuKtqVGjRuEPf/gDTj311C4xrf4I/++0C4sDh8AEdATa6LPd72Nj9TL4Q52wSRbYTS4wKLp7qi/CgUEEA0HTQggrfjDSYDE5kZE1Edm5U5GdeyQcGWOYIJhiQSMa32CDBhx80fMKeZbg9xiQjnADPYIqUVgjyCJDky8MX1gjmyyweBeWFgWKymY/bazuxE+OK42m7fYeHrlguO+++9j5559Pn376Kerq6pDIAjC6nlRVhaKqUKLB8kAggEAgAJ/Pp8dQTCYT7HY7MjMzkZubi8LCQhQWFiIvLw/5+fkYPnw4CgoKYrJ+4rNr+gNg/FwvvfQS5s2bh927dyMrK0v3+QOAy+WKqU8wZtS1tbVh4sSJeOHFF1LSiLtzGU2aNIld97Pr6KEHH0JmZiZCoZBOsHjeeefh3Xffpby8PGZ8L4Ig4NFHH6X7778fTqdTjzWEQiHk5+fjd7/7XZ9JGY29S5577jl20skn0b1/uhdbt27tVglwuVw4/vjjcd6C83Deuedh+PDhzBiXMNaRcOAxmUwoLCxkhYWFmGRgKTaOcDgMt9tNHR0daG1tRWtrK5qbm9Hc3Iy2tjZ4vV4Eg0E9o6u4uBgTJkzAzJkzWW+q+tMWSIrgoWgKPt/7IX2+ezHcgVbY5QhwcBeV0EvgYFEXGEiBogShaAHIggyHNQ85mUcgL3casrInwmYviU0zJVUX5GyQ3VQH2uxGCiEDgRYKBd3QSIUoWmC2ZMJschref6okI4mF7kAPs8iYxEAqAwJhwqLt7ThtTAY5o6m7GhnfM+HNDfWYOMyJ0YX2fhcNEhFmzpzJZs6ceVAXvNEtMtA9urmFVVJSwhYvXkwLFiwA75nOM5jiBR7v/9HW1oYZM2bgzTffRF5uXr/IBvl93HH7HXj/vfexc+dO2Gw2PTNs9erVmDdvHv785z/T/PnzmcViQWVlJT344IN47LHH4HA4dIHGYx+PPPoIioqKWH+sNaOl8aNLfsQuOP8CLF68mD755BPs2LEDHR0dEEURWVlZGDlyJKZPn46jjz4ao0eP7pLqywEv0Rzxe09UCMhBJmqRsuzsbIwYMaJXz5HqHAxUGu+3EkC4JSEwATtbdtB/t72K6tZdsMlW2E1OMFJ6CRxMZ28lTUFY9UPRFFgkCzJdI1GYOxkFudOQlTkGknTAdI1Unh9o63rQKEeiVkco2E5N9avR3vYN/MHOSLCXidF2uXbYbIWUnzcB+Tlj2AFO3d4FRXtL99DzTEfuwioJyLZKqGoPwiQxVLUH8eK6JkwrttP0YXZmlg5szve3NFOTO4RLZhfzt9XvTRGvQfZ2MyXaoMmEBhc6XGgP1uDCu6Kigi1btoxuu+02PPvcs3o2kSzLMe4ZHmi/7rrrcN9998HhcPSbqZbPrdPpZM8//zzNmzdPp2nhdRg7d+7EWWedhXHjxpHD4cDu3bvR0tICp9OpA53JZEJ7ezsuuOAC/OSqn7CBcPXxd6aqKsxmM84++2x29tlnpwT6xoLCVNZIt/EMA8gY101P5x0MxeM7Z4FwqwMA3t7xFn248z0I0OAwu8BIi2jgKYqySGe6iGtKUQIQSYXN7EJu1mgU505Dcf50ZLlGMKPQjcQzeLqlmEAe903T783LZoyhqe5Lqq1agnDYDyZYwUQzJMkMMAlEIjRNQbt7P1o796OhZSeNKT8eZpOd9fb+eBOfgc07j2BuRZYJe9uCkBGJgwQVwtJd7dhc46GJRTbk2mXsbPBi+a42HFnqQr7TzHpow94rYTvQ1tVQ2GwcRFwuF3v00Ufxs5/9jP7973/js88+Q2VlpV7zUlBQgLlz5+Kyyy/D9GnTGXfLDMSccFfWzJkz2TPPPEMXX3yxbglxenUg0uecx2iM8Riz2Yy2tjYceeSRePrppwe8lzlPnuDpsXwtGN2WxjUy4EpUH12Eg3Wd7wyAcPDoDLrp2Y3P4Kv6Tcgw2aOMrQpS0f8ZEyBEQcMf9kNmDBm2PBRnjcXwvGkozBkPhzU/gWsKUep2EV5vDbk7dqCz7WsEfTVg0OB0jkDRiPNhMuewwQIR7raqq3yXavctgWByQpLt0EiM0qNT1D4jAAIkUQZBRGtHNTZ98x4mjzmDLL0EEQ4gAyrkopcem2tmq2u85Aup0aJNwG4S0BFQ8MnONp3r3SIJmDTMqVufh0+nj0MHIlwQjh07lt1222247bbb4PP5yOfzcVdNjGtmoAGVWxwXXXQRCwQCdPXVV+tFfPzeOJDwf3PtmrvUohXkepbaQAvXeG2eWyiHE1vvYLPxfmsAhINHdWcN/X3tk2j01CHD7AKRAo2oR6sjEi/QEFL8gKYgy5aL8uI5qCiciaLs8TDLxqyK+EZREWhqbFxNu3e9Aq97DzTFDZE0iEyAyAjt9Z+hrf4zTDj6rySbswccRDh4tDasorp970E2ZUGDGL3P5JxABA0myQJfoANf7/0cU8ecGqFLSfG6FotlwC0QboVYJAHHlTnw9vZ2mFnEslA1QBIYRFmAphFUhWCRGIZlWYb8gh5qgoVTehgqmpnNZtM/wwPVg+EW4e5PVVXxP//zP2zEiBH0q1/9Chs3bgQQyYridRKc+oRnPF122WX461//CpfLlW7+1EcASafxGoYaFdRft3xDj695CkHFB4fJATVqdfQ8yQJCig8CA8qzx2Pi8OMwMn86rCYnMwro6Iej3QjFGJeR21NNa9b/EUwNwiRZIMsuiCAIETEOUXbC765E9fYnMXLK76ICf6AmPxLzUMIeqt/7NkTJ1isuLqIIiLS561DbvJNK8sb22O6U/26wAIRF+kFhbK6FnTY6gz7e2QFNi1C4a4gG0SlCtOiyibBFc33T8NF3V118bGYw4zHx7qx58+axlStX4qWXXqKXX34ZmzdvRnt7u14omJWVhfknzcfPfvZznHbqqQPqUkuDS//Wz2ENIBw8Njduo0fXPAUGFRbJDDXFWAcDQyDsQXn2GBw7ZgFG5E3uAhqsWwLDiCXh9zdC08KwyC6AQjqhov4fKZDNmWit+RAFZeeQPXPCgHUQ5MK+vXE1QsEWiKasCPtsL9YHgSAKEmqad6I4d3TK92Ws1O6p3qGvIDKpwMqyrSK9s7UNnkCcUqB3GExDx1AQKH0FER6wv/LKK9mVV16J2tpaqqqqgtfrhd1uR3l5OQoLC3XgONxcSYfz+/7WWiBaFDy+atpOD695CgIAsyBHM6xSszyCYT+ml56AMyZdwURBjGPjFVJ+CZkZo2AxZ0MLuyEmnVQGIgX1u19GxZH/N+ALwd2yBYxJ0RfeuxdLRBAFEb5ABzz+dnLaslkq8QSz2az7swdnkUesjRKXiZ0/OZv+ua4JYSVS5KkBEATAH1YRCKuwmsRBTlFIj8G0hHisQxAEFBcXs+Li4tj9bqAsT4/+ywzWC464w9ECEXoCD4EJ2NW+jx5a8zQAQBZEaNS71EuNVJRkjYpw/uNACmjqenSkzsJkymAlxccjrPiS1ngQqRAlBzqbVsLv3k0s2hSqv+4rgEFTgwgGGhFpdtVXK4BB1VR4/O1IdRJkWR50V4cQjX3k2mU2q8yJoHLA/ScwBl9QRYdfoV6+uPQYgkKNxz14eiw/eDZUGjwO/uiJDPSwAxDOZ9Xka6U/r34aYU2BJMhJwYNXosdr00QaLLINi7e8gNfXPkz7mrcQ/zyLAgOlIOC5MBs58vuwWHKhaeHkejAToCp+tFQv0l1HAzFUNUCaEsRANHIMhv0pP3OiPgWDo6FGsGFSkQ0Os6g3iGIsEgdxB5Q0fnwLwSQR6WB69B8QBiqIftgBCE9CDalhPLjuObQFOmEWzQnBg0UD3v6wH76wFyqpMdTtRoDZXrcar676E/715e30VdXH5A91EjN8lkjtZiIZiFRYzNmsdPipUBRvcvcXaRAlKzobl0FTvBSxVvov9iK0KAOzyXoDagOZ3tld2J/3+7CbRHZEgRX+sAZJiNTqaASYooWFaTGTHumRtkCSAginqfjHljfp65bdsMs2aKQmtDrCahghNYgj8sZjeslsWGUbvCEPgmFf9DMHNBurbIcsmVHT9g0Wb3oCLy+7GUu/eopqWraQFi0OPMCemghMIjXUJcNOhiTZddqSRGKSCWYEfTVwN6/RLaH+QAcAiJKNiZINgNrviTfL1oO3OA2gYeybrlFXMGHR5Oc5IzKQa5fREVDR4VdQkW9DSaaFEQYMQ9MjPdLjMAcQqSt4RILmy2o20PuVy5BpdibMthKZCF/YjSJ7AS6aeBHG509kAOAOdtD2xs3YUrsK1a3fwBvqhEU0wSzK4MSCJskKEQR/qBObKz/A9qoPkecqoxEFMzGi8Chku0YwYwpvJIU2EnAn0mC3D2P5BbOpofYTSLIzeYyDgI76pcgoPL7/ejMRmCDBYitCwN8EJvT1fJFMLKc1G6mq8z01jUralzoOMEIqIaQRCWCwykx/BGNlOac5cZhF9qOZBbS5xgOzKGDKMAeTRJZ2X6VHeiTwwvRX+He3v4dyTEqKfwjGGFoDnbRwyxuwSpbIz7pMiojOUCemF0zCj6f+GE6zk3FB7zRnsJnDj8XM4cei0V1DW+tWY3vdGjR3ViEMBRbJHMliggaRSTCZnBCgoc29Dy3t32Dr7jeQn1lBZYWzMaxgJpy2Ip3GhMdLGGMYXnomGus+i1ZFJxS7ECQrvC3roYbaSTRl9quwkGdLuXIno615IwTG0Nts2khjLRV2aw4c1gymu8V6GD1xACVafMYn3d2p0q4OBa1+FUFFBTSCVQQNc0qYlGeByyywRCDiskhsbkVmAlssPdIjPVIBkO+UBRLp+SDg+a/fRUugA5kme5QS3ThZAjqDbpwy8gT8aNLFjIEZuLGYnqLLGEO+s4TlO8/DcaPOwt7mrbS1Zjn2NW2GL9gKkyBBlEy6BiyJZoiiBYzCaGjdhobmjdjyzb9QmD2eSovnoih/Bkyyk/G4R1b2JGa3l1LQtx+iKKOLNCeCIMhQAk3wNK9GRvEpUfAR+7VIMnKnMYttCQWDnYDYW4oRBlULozB7RNSaSo0SOiMjg2VkZFBnZ2eXOhDGGIyVzUZB71OIljUo2O9RAI3AKGIBairBH9ZQ7wljS4Mfx5XaaVyeJSGI8GsJab9VeqTHoIHLYW+BcBD4qnkXLd2/Fk6TPVJAGCeUvCEfLhx3Ls4ecwYzdh888BkW2+A+6rIZlT+Fjcqfgk5/C+2sX41var9AU8duhNQgrJJFt0oYAFmyQoAVRApqGlahruFLOG35KCmYRSWFc5GRMZqRFiYipWedmDF0NnyGjOJT+lkIF8kYE0QLCkecg71bF0IULb1qiaVpCmxmF4pyR7FUFhdnLTWZTDjqqKNQVVUFk8mEcDgcQzw3f/78GE2FAIQ04ON6BU1+DRaJQVMAVSO917nEGCSJIaxoeG9nJ8Iq0eRCaxcQSWfmpEd69B08vjMAEkmpJbz8zeKEgpYxAZ6QBz8cfw7OHn0a06jnFqA8Q4u7nwDAZc1hR444HUeOOB01rdtox/5lqG5cC5+/Se+RzjOuAAaTbIfAgFCoHbv3/hdV+xbBYcsjgTSEgy2QBDlpgJxIgyha4W/dBDXURpF+6H13Y/EYTGbuVFZYdgbV7nsfoikz+vOe/1ZVFRRnlUESTb1qSENEuP3227F48WK43e4oGEWe+dZbb8WUKVN0viL+dBvaVGoJAlYRULSuwXKKWn4CY7BIDB/v6US+XaJCpzxgbLvpkR7p0X8X1pAHEG59rK7fRlua98AhW3RmXe6+8Ia9OKl8rg4eBAIjlrIs1lN1DVZLSfZ4VpI9Hv7gRVTZsBq7az5HS9t2KKoPJskMSTBF/oJUCEyCyZQBERoC/maIjCAyCZFa6aSvBUyQoQSb4WvbBGfB8f1yYxlBpKj8DCZIVqqt+hCqEoYg2g+QPjIBgBCVwrETJAqSQYT3PHnc0pg4cSJbvnw5PfDAA9i5cyfy8/Nx6aWX4vzzz+8CHu4wUaWXYBaBntpt8KwqTSOsqHLjvAnZ6d2cHulxkEd3lehDHkC4Jvz2nmVRXzd1ETKSIKLN34GtTdvpiNzRTDRkSRF4d7oUNOoYF1dk0qzmDDau9GSMKz0ZTe07qbL2c+xvWAmftw4iALNsjgJQlGZBkBHJB0otNZdA8LWsh7Pg+AEyTQWACAXDTmDOzDFUX7MMHR17EVb8UMEAJkd6gjAJJACCaIq68kQ0te9DaeHkXgXGeIOgyZMnsxdeeKGL5hJ/rlq/hrBGkFPWfiI1HjUdIXQEVMqwiCxNV5Ie6TFEXFgHgXCzzwDCrY8dbVW0pWUPrJKlS80HEcEsmrC5aRu2Nm3FiIwSmlk0DTOKp6PAXqATk2ukxQBEb6wSRF06eZmjWV7maEwdczHVNq5FVc1naGndjGCoA7Joglm0gEFDyilQ0WC6v2NbxCIZqDa30Ta2NkcJGzn2IoSCHeTx7IfP14Kw4oemERRNgS/YCY+/DaIoQhBM8PjasL9pO5UWTGC96S/NmxRxwOAaSyLtpCPc+5oXBiCgEJq9YWRYRKTdWOmRHofWhaW3FR7KFgi/7Y/3r0dIU2BjJqiU+AEtkgUiNOzvrEFl+14s2f0BxuWMpVnDjsL4vInMLFli3FTGGEhPVgnirBJZsrGy4uNQVnwc3N79VFO3HHV1y+Bx74WqhWGSLGCCFAUTtVvrQxBMCHv3Qwk0k2TJG7A+IdwSAQCTOYNlmzOQnRM/byoamnfQ7qovQaRBEk2orN+CvMwyspodrDfNmYyWRvdpvX3CQxARgkq60iM90mMoAIgupIeyBSIyAQElhHWNO2CRTF3oSljcQxIIJtEEq2SCSgo21m/A5vp1KLIX0NSiaZhefBRKMsoY16wjVglSpi4/8DnSGW+d9mHsiFEXYczI89HSsolqaz9Ba9MahALNkAUJJsnSDYgQwCSo4U6EvPsgWfIwoOq1fp74nsq8laWIwrzxDEykHZXLIMkOBJUQdtVswKSRxw5K40Sr1NUNmYobizGGaMuP9EiP9OiFq6q/LqzuYiAcQIZiRqQEANvb9lG9rxVOyQyK0nSwqGWgkgaTIEQC6lFwISJoIAhgsMk2CCC0+Vvw4c538cXejzAyexRNL5mN8QVTYTM0jNIzt1KSmLHpwIAGQZCQl3cky8s7EoFAMzXVf4HG2k/gad8KQTR1+4I1LYyQpxK2nBndFB/2axklfcFEGgpzx7J2dz3VteyByeRAY3s16tv2UWFWWa9cWamMbJMQscx6ASKESE/0HLsUi4vpkR7p0ScA6Y3QP2wtEADY1LwHqqbqZHoRYU/QSIVDNqMz0AGZATbJDJEJYHRAOGnR7yVBhkWUANKwq3kbdjduRo4tB+MKptKUkmNQmj2GCTppohZnbaTygsSYv7VYctnw8nMwvPwc1FUtot3bHolmZSV/EWFv9SFcYISK4bNYu6eJgkoIkihjV80mZDkKYJYtGEhTJMPEYBIi9OypQR8QVgnDnDKyrFI6gJ4e6XGQRyoWyJAFkJ3t+yFFGz3xEmRVU/Hr6T/C+JyR2NT4Nb6sWYNdrXvQGfbBJsqwSDIEkO46IlAkHRiAWbJCAsEXcmPV3g+xsepTDM8cQROLZ2Ns0Uw4LdmxVkkfAu/GnulFpd9jnW2bqXn/+zDLzi7urIiGL0IJ1HNb4WBDCIgIsmTB6NLZ2LTrE8iiBcFwADv2b6LJI2YlrL3oopWk0Eed0DuyeQaAGINGhFnDHQdOkkaQ9EiPgza6C6LLsjxk71tyh3xU522BLEogAkRBQEfIg/NHnYi5JdMYAJxQOhsnlM7Gvo79tLp2HTbUb0SDpxYMGuyiGYIQdZnQAZeNBoLEJJhMDghE2N+2C/tbtmHFrjdRkTeZJpTMxfDcCUxgYpxVklo6MKI90yOsvRryi09C8/7FyS0QJkIJtkW/P/jcMrwgMCdjGCvJG0v7m3fBJDtR374f2S0FNCynnMWDaVKXmEH4JwKPXW4VIS2iHTBEmkVp7ACxon4wQAHgDak4ptSBsixzuogwPdJjiADIYWGB1Hhb0B7yQGIiQISQGkaBLQcXjDk52lCIdGFWljGMlWUMwzljTse2pq9pTe1q7Gj6Gp5gO8yCCJsoR9rURt1aEaskEm8wSWZIsCCk+LGl+lPsqPkcBa4yGlM0G6OLjkaGvTCuTzpLsVI7wsFld42CyZwFTfHGUKtw0cqYAE3xgEgB011dB1dSMhbhCqsomcZa3Y0UUEKQRRN21GyF1WSnHGeefkNBJYTOgI+84RAUjSCJEuwmCzLNFiZHM7AoAXjs9Si0tUOBSZCgaUBYi7inSI2kSoMiNO6qRgirBBmE48udmFliS4NHeqTHwGx0ULS4t79BdMbY0LZA9nsaEVIV2GQZAgBPOIjzR58Ih2xjapTa3YiSPAtrauEUNrVwClr9LbSpbgM21q1BTfseBNQQLKIJFlGKdXERQYMGgYmwmhxgIDS796Gp/Rts2v0mhudOojElx2FY3lQmRUkKjfUhyYV95OcmcyYzWfIo0NkRJTmMd/8IIDUAUkPEJOnQiUkCJNGEscNnYOPu5RAkBhBh4771yM8oJpNkgTvkR1vAj6CiQoMIYhI0iGCiDItkoXyHE+UZGci2mJlxFva6w7SiOQhJlKAhwodVaBWQZxGhqoTWgAp3UIWiAiYmIN8mYnyuGdlWMd3nIz3SY6DwA71vX5fMhSUIwlAHkGb95hXS4DLZceKwGQAAIU5oGzOoeLpvtjWHnTDyJJww8iRUtu2ijTWrsb1hA9p8DZAAWCUTBCZFKsejFo1GBAEEOcrAq1IYe+pWoLLuS+Q4SmhE0SyMKD4WWc5yFl8fkjDwHlWdZVMm/KQe6IoU91pJC4MofIiVk4grK9tVxHIziqmxswGiaIUGoKatFgQRJEhgogxZkEGCBGJiBEggIKQq2NPegUq3D8UOJ1VkOGGVRFR6QtjSFoIsSlAp4raamy9ipMPYuESO1O8TYMzWTVse6ZEeh3Yks0AEQRjaabz1vlYwxiAwAT7Fh9mF45Fvy2a8K2GyISTgtirPGsXKs0YhMPZc2tG4GV/VrsS+lm3whjphFqRI4J2xaDow6VaJCAazbIcAgttXj43fvIIde99GYdZ4Ki85DiX5R8FkSAcmUnXXFb8HBgZJtgPRupPEMK8lbz510A0RQrarEA0ddQfEuyQDiICFaqDG1/9jvFZDhAaGKrcX+70hiExGkBgkUYJCkXTceQUickyMcdLEaPMRMBwAD87MmwaP9EiPQywPklkgoji0LZCWQKcOFCppmFk4/oD7qA/cVgTAItvYlJLZmFIyGy3eetpWuxrb61ahubMSoEhTKUkQo1aJavjbSDqwIJoAUlDXtB71jWvgsuejJH8mDS8+HjnZEwzdCjkYaAAECKIt5fs+9GYugy/ojrlX3pSLGPW82BhgEgRojOk1HBHwAOYXyMgwMRaZleQAIbD+LHgY1glfB2lBkB5p4T+QFog41F1YnSEfRCZA0VQ4TTZMyhmV0H2VmntGiPr/olQmYMixF7JjR5+NOaPORFXLNvq69ktUNm6A198MWWAwi+ZI33RS+V9GGHMByLIdIoBgqAO7K9/Gvur3kZ0xioYVHovCwmNgtRVFb1LQOxV2p/ODiQPHhdWPxcYYg9vXRrXNeyCJMghc1PfWiokdCgFHZZuQYRKYRv0DiGTXo6hleqBnSOxFtOh7T4NJeqQBJBWZyRKegxOoSpI0tAHErwQhMAFBNYQxGcUotGen1OwodauEs/UKKM+dyMpzJ8IX7KA9DevwTd0XaGjdDn/YDbNohkmUI+IozioRmATZ5IJAGjrad6KzdQv27HoZuTmTqaj4eGTnHglJdrCgvzGS2puw0pzARBOYYDq0oo0BihrC1soVkboZIdoat593pREgC0CuRcRgBMRJd3dFTtzuC1OHN4ygokISBLisEnKcZiboFDbJXaADobGlR3p8W6yPZBYIEUEc6gASUsMQmYCwFsbYrDKdvkQcoFqJRE2lbOYMNrH0REwsPRFNnXtpd+2X2Fe/Ch2eaohQYZYsYEw0sO6S7h6TJCtEWEAURkP9MjTXfQaHowQZGaPJ0/41RNGSOM5BGgTRBiaYD0jyQ2R97KpZTx5/G0yyHeoALUDGIim7njDBIfU+C6QncBIYEFQ0rKvsoK+qOtDYEYAvqEBVNZBGkAUgy26i8cNcOOaIPGQ5TEnTgtNdDtPj2zIGai0nAxDJEAMZkkF0Jdq2loFhXHb5IE92fOCdIc81guW5RmDG6POxv2kj7an5HA0tGxEMtsEkmGCWTAbWWw4kkYI7WXZABCHob0KTtzrC0JtwkiMdDkVTRlSiHfw4CQePTm8T1TXvgkm26vGDATJsoBFQ7VVQaB04Nx0Hj52NPlq0uRGN7YFoPlikh4gmRHqsa6qGujY/Kus9+HxLI86YUUzzJhYkBBFVVQ+ZpvddEkrp+Rj49aRpEVc5r/FQVRWBQGDAAMR4X0QEWZaHtgUCRILnTpMNozOHAehb/KM/Li6AIIlmlBfOYuWFs+D2NVBV3ReorluOjo5dYBSGWTKD8UZSOqlj1MUlyJBEGSAFiXRvFu1nLppzdWsEhygWUtO0PWkL3v66mGQB2O9TMUUjmAYgAMLBY31VJ729sREAwW6WoKkqVFWDFjUQ+Zo3SQwmQUYorOKlT3ajsc1PFxxbzjQiaKoKSZLwl7/8hR5//HHYHXaoijroGz4tML8d8zHQc6KR1msznQBoqgrGGCRJgiRJCIVCqK2thdVqhaIoAwpoRDT0YyAiE+BXQxjlLEK+NYsd7AV8oEjwAB2601bAJlQswPiR56KpZQtV1S5FY9NqBP1NEfp2MdqhMBp470ql3lU9J9Ig20r0hcAOwSZQ1BDa3fUQBUnPpBrIITLAq2io92tUau9fV0GKgsfuJj+9s7kZJimSqKCplLTfCFGkNa4gABl2E95fux8ZdplOmV7ClOgf+Xw+7Nq1C2azGeFwOC0d0+OwBHeiAz2PTCZTTK8e/nP+mf64sEwm09CmMrFKZnQGPRibVRoxyQYw/tFru4QdqOtAtHd5fu5klp87GYFgG9XVf4na2qVwt38NVQ1Em0rJANTuuyjxXheOskMyyXxBeXytFAr5wCQLBlOfrg9oKLWL0SlJHIjoIV8NABAIa3hvawskgYEx6rG/uhFISCO4bDLeXlGNSeVZVJhlZUTAZZddhgcffBDBYBAmkykdUE+Pw350R0PSVwDhf2symSDL8pA1SaXhjjzsaq/GzIIjehQsBxFKdBeTTt9uzmIjyr6HEWXfQ1vbNqqr+QjNDV8i5G+ALMoQBVPSsxFUMNEGk2PkgfMfguELtEMjFd2TzvdjIQOQBYb9Pg2TMwkWMTViSo4zZAAAUQCW72mnVl8YVpFBUXvZoAqAwBgCIQUfrKnBj08dDUVRMGzYMHbHHXfQL37xC2RkZEBRlDSIpMe30lJRFCVl6yERCGmaNvQtkPNHHYdiezam5o5mkU0vDLEXEUvfzpiIrKzxLCtrPEKjL6XG+mVoqF6EgKcyCjrURdMmLQyTtQAm2zBOc3uwbRAADMGQd9CvJAAIqIRPmzQaafbBCg9MpgyIggRJkCAyQBIYkwRAZAwCi61S55NW3R6kddUeWGUBmtrHgCMRrGYRm/e2osMbogy7iSmKiuuvv55t2bqFnnryKTgcDoii2G0/hPT49rqC4i31wTjvQReqkoS2tjacdtppGDFiBNOixIo97ZVE82G2WHpk5z6kzzolt4JNya04HJYc4ivQTeYsNqzsbOTlz6Z1y66M8lzFUZkxAaSEYHaNARPN0QC6cNDvHQBUbfB9/gRAYkBrkNDgMwGqM+LK0hSAwhHXoEYkgCAxgsQYJBYJussMMIsMYVXDtnrvgNyMKAhw+0LYU+fGtFE5eoHUk088yXJzcun++++HoigJN9jBtkzSllB6DNQoLS3F3//+95SVI0rgwgIAq8Wir80hCSDGQr/DR3sR+KxHm2AJEAQJUEIJrQsiFbbsI7kdc8jcdAfTdSYxgAkCNJj0lsCqFoljKBpFvlc1qKoGRdOgqgRNUaFpBE3TIDMGkUUysfoLnapGaGjz6xuDsciG+L//+z925pln0l/+8hd8/fXXerYNF+TJsm8EQUgo7PuTrTMYmT6HoxWQ1LLtRoNmXTL+WJci0vj5ZUKU8JRFPqtRpJ6IC1utu0xFij1fou/jD7CIdyXZ7xljYEIk/5Q/a/xXxonjEiTAkBYJmAeDQWRnZ+P2229HeXl5StZHMhcWAFiGOoCk3qN86I3I+hMRCrZAUbyQmYRIJODAQo7Uf7hgz53Jl/ahulOYTLaDYuuEomy7JgEIEyGgEhSVIiChUSSTSos0BiFEXFmRFiORehsiBkXV0IdyjaT35A8pXQSXqqo4+uij2dFHH61vIGOWS6INk05xTY/DZaQCHnw9JwMQm802pK1j6fB+RZFJ9Xv3Q1WDkGVTXBU6g6b64cg5Mhr/OFS85ZFrOm25YEwYtAysCHgQSmwCpmSJsImMqUTkVQieMMEb0uBXNIQVDSGV4A+r8AY1+EIqAmEVYWIIhwmBsAqR9a2vQbK3ZJa71t1w856IIIpiGigO1104xF1/vb2//nJbceUnFcujp2tyABmqQ/o2LGB3xzeRxlOJ0F1T4Co8MfqStENCpsiFodOex6wWF/lDfjBhgO+DASoBLlnAsfkmdqDfB2M2iSHPAgCJr6lSpDthWCUKhDXUdoawfHcHPEGl37Ypz8bKy7B06xpJtoHSQPLtcoWl7y/xiC9C5Pdst9uHNEgLh/fCjdy+u30HBEGOowZh0LQwJEsBXIXHx3z+UGlBoiAhL7MMqqYM+KJmABQNmJRpiold8KbEFG1lyw9jBbnIGCySAKdZZHkOmU0ptrNzJuX0f9GySGGh3SJhRKGz282c1HedHunxLR6CIEDTNNTX1yesG3E4HEP7/g9jwxkAQ8DfSF7P3mgPkdjsK03xIqPoRIimTEakYihUueRkDIPAxAHVKDiRYp5FxHC7FE3HPgAsLPoZwXDEN5Iiw6FqwLBMMxudZ0NQ0fpMCy8whkBYxagSF3IzLCxi2qeFRnqkBwDdfbt3717auXMnLBZLF7ngdDrTADI4Gn0k1tHWsgGhYEckCwtGIjIVouxEbtn5UUF6iCUXi2SAyZIVwgC7r/hTT8+x9FnYM8PBzzG73BXhEevfi8LJ04tj7jM90iM9DhAzvv766/B4PJBlOQZAGGO6CysNIIMhkQE01X8RCUzHeK9EqGE3sktOh8k+jNEhqf3oKkgZGJraKqGq4QFxz3CBH1Q1TMuxIdcisYHg+eKZiiWZZjazzAlPUIXYy+mTBIYOXwhzJxZgXGkmox5aJKdHenzXwEMQBHR0dNDjjz8Ok8nUhaWaiFBcXJwGkIGXxQTGBPh9ddTWsgmiZMWB9F0G0sKQzdkoqLgUwKHPn44E7wU0t++jfXUbIYqmLmmqvb1HIRrnCKoaJmfbMSbDwgaSJJKDyElHZLMJRXZ0+lXd/dXT30kCQ4c3jHHDM/GDeSMY0eHRZjg90uNgDFVVI+0sBAG//NUvUVVVBbPZHFP/FAqFkJmZiXnz5kX2uzA0RfVhmoWlARBRu38JwuEOWEwZUSr3SF2IGmrH8CNuhGzJY1x4DyY4dGchRQBMQFPrbvq68nMwJkdJKwmKEoIgWqCRBkUJQRQtEITkwWPGJTQizZ1kUcSs/EyMcNnYoIhoFgmwX3BkAXOYRFq1tw3QCCYWiW/owRUGkMAAYgioGvyBMKaPysbl8yuYWRaRrLFUKnGggfpMfz7/bR7pRIWDt14YYxBFUU9Z/+1vf0vPP/c8nE5njPUhiiK8Xi8uuugiDBs2jKmq2iXNfcisn8NvM0XuNxz20splP4USaoMsiBBIg8gYEHYjI2cKxs56mEXoT4RBWlSpA1N90zbauW85mGgBBBmKqkGUrCgvmoRsZyHCahgNHQ2o62iEPxwCMQkQZTAmgZgMYgJUEqFChAYRkiSjyOHC+JwsOGSJDaZ+bzz3Nw1eWrajBVVNPgRCCjQtwk+mKpGqdgZCvsuCeRPyMGd8fqQ7cRQ8NE3Tzfahqk2lR3ocDOvjiy++oD/96U/44IMP4HA4EhYRqqqKlStXYurUqUMaQA47C4TXctRUvw+frxZWUwaIFIBFXFeSbEf5pP+N1HuQNlg3EY27aPB07iWPuwqhoDtiF0l2WKw5MFuyAQCNLTtR37wdgmgFYwJCSghmcwYmVRwPhzVTl/uZ9myMyK9AQ2cTNXna4Q4FEFIjvRfBRJglE2xmK3JtThQ5HHCaIr3dB9s5xA48MsYU2NmYAjtq2wK0t8mLxo4gfEEFIgOy7DLK8+yoKHQySWT63xBpkd4iCYCDiKCqKjRNg6qqUFWVol97PJToVy36vRb3e35O41fjoWoaKO5nRHTga5Rag/d0iD/4/VOUxlj/mWGNdOvnA+mJHboVwJiB2PIABQf/PtG/+3UIAgQ2QOdKQifCnykZ7QgMz9vl+1TVScP7iP+ayhH//hMdqqZGqX9i1uuB9ago0XWpQFUi/+ZHOByGoihobm7G5s2bsXnzZqiq2sXy4NaH2+3GaaedhqlTpzJN04YseByGABIR3OGwm/ZV/heSZIsASnTlqaoPo6fcBYt9+CC6riIqdXvTRqrf/zG8vsaIoIQMYiJUJgJMAhNMkX9DhChawRhDWA3DZsnE5NEnwWJyRO+R6XTqJsmE4dklbHh2CTQihFUFWjRWIotiTJ8WihPwg+/qOGBNFGdZWHGWJbmDMVpkYqww37BhA73++utYvXo1WltbI/MR3Vj8SAQUyQS+UdgnEu7pMbRcYwldZaxnqEjI2mvsPdANoCRyQR3q9cEYg81mgyAICVs78+c988wz9TU+lC32wwpAuPVRufdN+Hx1euyDMRHhYDNGjrsGOUUnMCJ1UCrOeeC7Zvd/qHH/p4BoiYAYk0CQQEyEEAUQjQSAiWCCBA2AqmmQZRsmj+LgQTrAMQMS8E6FAmMwS3LizcEOTVIyYweU65iiTTJq46T7egFg7dq1dO+99+Ldd99FMBjsIhS602ARp7nyr3xDGfsk9JU367sSAxhowdnf8w22IO/te0318/1dL7pClIRojs/LsGHDDot1JR0+GyBiUfh89bS38i1IsiP6MwmhYDPKKi7G8FGXDiJ4RK5fs+tVatz/MSRzDjRE21ZCix4syvWrIdLI6kDbS1ULY2TRVFjMzm6to+4WqDGIfmi1qAhocN9tIvfU9u3b6cGHHsJLL76IQCAAu92uZ5ok6//cW8EzlDTL9Ph2A+bBttzWr1+Pc845Z+jf7+Ey0Vzobtj0Z6qtXgKL2QWBVKjBNpSNPB+jJ1yvu4QG2rHDr93euJr2bX0SojkHGgCCGLE+mKhbIFrUAon8ToQW/b2qEaYccTZcjgJG9O3TfD0eD+3btw+bN2/G4sWL8dZbb6G9vR12ux2iKHbh+kmP9DiYFsbhAkS8k6HL5cK6deswfPhwNpTdWIeFBcIFeGvbNtpfsxQWUyY0LQg17MGosVdg5JjLBw08eNxFUwNo2PsWBMmC3oauGRg0UuELdMLlKNQtlMNZs2ttbaV//etf+PTTT7Ft2zZUV1fD4/HonxNFEXa7XQ+UC4KQEr8VdeO//rZomYMlIAfLbTNYwjdVSzTRzxJlLg1IR8OY1pxI6kpNNH+9mU9jDC/+5yaTCc3NzfjRj36EDz/8EJIkDd1+IIfTpqqseg8AIr0/RBMmTLsFxSUnDyJ4HIh7dDZvpKCvDqIpM1L70Us3GQPQ1rkfhbljcDgX1fFFbLPZ2Ny5c6mwsBD79u1DTU0NGhoa0NTUhJaWFrS2tsLj8cDv9yMUCiEcDg/IdZPGSgwbP5UCzYHcjMkEVyrCsDs3XHxWUU+fP9wBs6csM+Pn4hWSnmJoPb2/iCsaQFwGnp6JRQQtmtQxUPMuyzKsVqteWMgHz9D6/PPP8eMrfkz/fOmfzKiEpQGk14srYr61d+yGoniRlzcTEyZeB6dz5KDFPLq4aFq/OuD87yUAEAiCIKHT2wRFDUMS5cN+w1utVkydOpVNnTo14Yb0er3k8XjgdrvhdrvR2dmpf+/xeODxeOD1euHz+eD3++H3+xEIBBAIBBAMBhEMBhEOh3Xwic/WMqbuJkq77C5tM6lwjgqQVACnuyC/UYglEoJG4RefTst/Z/waf/DPi6IY83Pjv/n3oihCEEWI0e8T/btPhyRB6uEzgihAFOJ+Fn9vcfct8J8ZnzX6vPrzxwFIl/lMACSJ3mGidF9jhl98mm44HEYoHEYoGIz5eSgU0tdsMBhEKBSCoii6laRqGkRBgMlkgqZpaG9vxzfffIPly5dj3759MJvNXahMFEVBRkYGXv7ny6gYWUF33XXXkKwHOUxiIBGh3dSykVTFj8KCo5nRtXUwrr973d3k8+wHk2wAIrGOlGMgEIHo99PHnQObJYORoQ7gcB1GQc036GBqSfF5+qqqUjLwSJTqa9QeD2idsXUcqVpB3YEFB4n4OUl0xPzuAJCwpJ9J09x/a0ZraystXLgQf/7zn9Hc3AyXyxUTK+Tv2uv1YtmyZZgzZ86QAxF2uJrBB9MnSFoIO9fcTqFgGyBYomDQewBRIWDqEWfBacv5VgBIKi6d7jT+7rT8ZFpkevTsKuttltvBkAFDLT4zkK7KvsyhMdV99+7ddO211+LDDz9MSmsyZ84cfPbZZ4xzaKVdWH16gZru0jqoC4yJYIIIQv/6WTDGInUih3ADqCk2Ou9tQDBZTUZvfMbd+fcTaV38eVJtRJWsQrmn8ydKAEjkDks2RFFMeE8xvu9eVmAb+w3Hn9/Yc76/4GAUdH1dS93NUTJqG25NDhVA68v5uLBPJvD5HFZUVLDFixfjmmuuoaeeeioGRFRVhd1ux7Jly/DKK6/QxRdfPKSsEJbOn0/NfVa5+SFyt26DIDtBEHpngTARRAJEkx0zxi9gkmj6zmjGh0KTNLqyYsA7zcOVXieHYPSUhmusp7rs8svpxRdeiAERQRAQDAYxYsQIrF+/HlarlQ0VyzwNIClYPYwJaKp6j+p2/xuiKQsE1isAAZMQ1lTkZI7EhFEns4O9YYwxgD//+c+0d+9eCILQReNP9d+MMb0ZTjgcxpgxY3DTTTcxs9ms/54/48KFC2nlypV6t7WefPvxv1dVFYWFhbj++uuZLMsx2mwgEMDDDz9Ma9euhcfrhdfj0YPvqiFjhp9LFEVIkgSz2Qyr1QqHw4GMjAxcdNFFOOOMM/T3wjd8bW0t3XvvvdizZw88Hg9CoRA4N5Esy7BYLLBarbDZbLBarbBYLJAkKVI4qqpob29Hbm4ubr/9drhcLmacO0EQsOi99+j1//wHJnMkgEoAKCpMNC3SHzJCCwNopEVjNdqBbCGNcPTRR+OqK6+EzWbT759/ffbZZ2n58uWwWq0xGn93X/nci6KIcDiMyVMm47JLL+ty7ubmZvrjH/8In8/XZS2lqq1rmoZf/OIXmDhxol7rwL+uXLWSnnv2OTBRiElsSIWuhs9ddyohdWvdRAuESYs19eI+E3knCZJqWOT9mExmXHvtNZg+bTpLFUS8Xi9NmjQJdXV1etAdiLAudHZ24k/3/gm/vfm3Q8cKSZVw7Lt7RIRQyN9MW5ZdT18tu56+Wn4jbfrif2njF7+lDV/eRutX3E7rVtxNa1feQ6tX3UerVz9Iq1b/hVau+St9ufbv9OW6p+nT1X+n5rZK6o7DabAORVFARPjVr35l7Fw7oMf7779P/Fr8ei+88MKAnX/p0qX6+cPhMIgIv//977t8jjFGgiCQIAgkiqJ+8J8xxrr8jSzLtHPnTuIuBX6cdNJJA3Lvv/jFL/R7526lXbt3kdlqGZDzn37G6RQKhfT75m1SZVkekPP//e9/7zL3jz322ICce8SIEdTS0kJGig8iwrzj5w3aWj2Yh93hoJUrV5Kmafq+SHbwuX3ppZcIADmdTrLZbGSz2chut5PFYqHc3Fyqqakh41wdykNK2xg9O5uJNMiWHJZXejrV7v43ZEte92yrMdqWgJASQE7mCORklkaC5wfR+uCayqpVq+jRRx+Fy+UaUOWDa0bc3OYEiqqq4qGHHoIoiglZR1MdnJ3U7XbHnB8AlixZAkmS4HA4eu3v5+9AFEV0dHSgtrYWo0aNgqqqkGUZ27dvp88//zymJ3W8S6wnH7ggCHC73Vi3bp3+b65RdnZ0QtM0uDIzUrvvKOMmJbjm+x98gB07dtDEiRNZOByGIAioqqqCpmnIysrqcyxBkiS0tbVhyZIluOaaa2J+5/P5IElSl8yh3qwdWZaxd+9evPHGG7jqqqugKIqupTscjsi7dfV97RxqxVySJLS3tuHG//0Nln32eY/7XpIkqKqKH/7wh2zhwoX02WefxaxtXmD4xz/+EX/7298GNUaU8hpJA0RqIECkIb/0NBbw1lBrwxqI5mwQsR7+TkRYDcJidmF0+XFA/zuM9zm4d++99+oLbiBpRZSwApvNhvHjx8dc8+uvv6Zt27bBZDLpJIp9AsCoUCkrK4txsYTDYXR2duoWVl82E6eNcDqdGD58eMzvvvnmG4RCIZjN5j4LMA4YPp8vxn0HAMXFxcjIyEBnZ6deadzX9ysIAtwed8zPTSZT5PnC4YgbrI9CkN+/0TUJAGXl5TG1OX29d8YYVqxYgauuuiomXjB16jQseneRXmtxOA5VVeFwOfHlF1/go48/opNPOjkl1xNjDPfffz/mzJkTs64VJbLXnn32WVxzzTU0efLkQ+7KSkcUeymIhx/xY5ZdNBfhUCeIlGhGmNGXL+i1KaGwF1ZLJiaOOQtmk5MRDm5rV74Zv/76a1q8eDFsNtuAbkZRFBEIBjBt2jSMGDEiJp62evVqhEKhmOysvsx5MBTC8OHDMXr0aGZ8D4FAgLxeb6/9713OHwyirKwMw4cPjzl/U1NTUsuit9fgsZN4cBmogD5jDCbZFHO/OTk5sFgsUPuhpXItuqqqCqFQKOaeTzj+eAwbNgyBQKDPc8Rdubt27dLXEz/X3LlzI50uvwUxWiLgL488ktJ64tb7zJkz2eWXXw6v1xuzh0RRRCAQwG233TYkni0NIL1wZXGronTspaxs7I8gmzIQDnuhhH1QlAAUJQhF8SMc9gKMobhwOqaMu4DZrNmRIORBrvvgQuvZZ5+F3+/vlzBPJriICGeffbaucfGxdu3a/i/OaBD9qKOO0ikf+Kirq0NLS0u/tXdN0zBq1CjdfcBHc3PzgM1RopTUsBIeEDAXBAGqoqC5JfZ+8/PzkZWVpac59xVATCYT9u/fj+rqajKuq7y8PDZ9+nQdWPpz/t27d6O9vZ2MFtpRM2eioLAQwWDwsM7QUlUVVpsVH334ETZt2kTJ+oAk2ld33HEHcnJyEAqFYlKzHQ4H3n33XSxZsoQ44KQB5DACEYCQU3g0O2L6jWzkET9CftHRyMwei4ysUcjLm4IRI07D5EmXYWT5iUyKki/2dhMYg2593ZxR/z698sorXagSBmKEw2E4HA6ce+65+sIXRRGhUAifffZZjM+/P1Yf743AXSoA8NWWLV20s74Oi6Vrc6zOzs4B0j4pYW0KQ2ynvn7NEQGbNm2K+XlGRgYrLi6OET59tUDcbjd2Rq0EYxJIeXl5vwo9OYDU1dXhq6++ijl/dnY2mzVrFsL9AKihMiRJQigYxJNPPdUr12dJSQn7zW9+g0AgEOOm4m7L2267DYqixLgW0wBymAAJkQZRtCA7byorrTiHjR73IzZ23CWsYtTZrLDwSGYxZxpcOr0ryuPFazwltK+aD2MML774Iqqrq/U02oF0XwWDQRx77LGoqKhgXLAzxvDxxx/Ttm3bYLPZBgRAEgX+t23dOmAuJq/X2+VcA0nJIklSFyFot9th7ef88PUCBmzZcmA+uKJQUVGhP0t/38HGDRsOXC/68/Hjx/d7TXFh+fnnn3dREr53xhnRBqCHd42IqqqQzSa8/sbraGlpIVEUe5w3Pi/XX389xowZA7/fr79HXly4Zs0avPDiC9RfRS0NIIcCQpgAgKf5RrqDxP+7twufC31RFLFr1y76/e9/T1dddRX1RaiIooja2lp64IEHYLPZ9PMO1MEX7ZVXXhnjLgOAf//73zGEePFHKueXJEk/55lnntlFENbW1fX7GWRZBhHpFohxU5eXl+saeG/PK0mSfjDGdAAx1lK4XC42fPhwhENhmEymfj0HCOjs7IgFFQAnnniivhb6czDG8NFHH+nvgL+H+fPnw2QyQVGUPs0TP7fZbMYTTzwBj8dDkiSBCZF9873vnQlXZka/zn8wD6mbveJwONDU0Ih33n23i7u3OzeW3W5nd9xxB8LhcBeGB5PJhD/e/Ud0dnbSIbNC0nUeh/4wuqv2799P1113HWVlZREAuvrqq8mYI96buo9PPvlkUHPczz//fOJFe8brXnjhhQSAJEnq9zWuvfZavT7DOA+33377gDyDIAi0ePFiMmZzaZqGuro6Ki8vH5Br/P73vyfj/PCv//rXvwbsXbz0z5e6PIPH46Gp06YNyPlvvvnmmPfAn+F3v/vdgJz/iiuuSLiWfjmItUuH4vhbtKYm1f3Ma3vmzJlDgiDE1Ia4XC4CQHfeeWfM+jqYR7oS/RAPo0vgxRdfpJtvvhl1dXV6lfOGDRtQVlbW665k/LyffvopNTU1xQSbjedJRHnd07/5OebPn8+4Fm+sUm5sbKSdO3fGBJC7O+JZdjmgulwunHHGGSxeK2OMoa2tjbhW3FfXjKIoGD58OObMmRPDDsC/r6mpoc8//zzm572Zf03TYLPZcM455zDutoi/xsqVK2nX7l0wyaYI7T8TAAb9K2NCJFYi8K+R6AkTmD6/ebl5mD17dsJnqK+vpzVr1ugWXSpCwWhRhsNhlJeXJ5wjPo9Lly4lr9erv59UD+AAbfn3vvc9Zrxv/vtQKIT333+fQuGQbvX3ZfRUnd5vl7amxVXpxMa7NE2D0+nEKaecojM2pOqVEEURn376Kc2fPz/GLczdlTabDZs2bUJJScnBJ1tMWwCH7jBqW9dddx0BIJPJRDk5OV0qmNPzNbgWYCo/OxjXHarPkOxcB5tV4bt48P1/3oLzCAC5XK4uVgj3VBxsWZG2QA7R4BaFx+Ohiy++GO+++y5cLpfupnE4HNi0aROKiop6rVUY00YHM8UvWaDZGAg1Zo1omgZBFBOmFHBAFUUx2gFOA8MBJth4KyieDTa+J0l3Mab4zye7RqLmVMnYabs8NzuQZWX8vM7gG+1Dwq/BYws8gcIYM4k/v/HnRu6qRA2TjIqK8frxnzfygBkZjvl7TsYobDy/0bKIr3MxWhaJ3kNKc2qwnjknWU/rv9csx8kU7cjN9DthQ5ZlfZ5TLQDka2Lr1q101FFHJWRBUBQFq1atwpQpUw5qcWEaQA4heDQ3N9M555yDL7/8EhkZGQiHwzo1yN13343bbrut14vh285s+l2eg8PtudJrceAGlwPXXHMNPfnkkzEUMpzu56yzzsLbb7+dBpDvAni0trbSqaeeirVr1+rgIQgCQqEQiouLsWnTJjgcjl7RNvNzr1ixgl544QXdx2yMM8THNoysuElboibIpNK1QRZV8+K6AHKKi3A4jIaGBtTU1OCUU07BLbfcwozaLF9/fr+f7rzrLnz88UcoHzECxUXFACLaaigURijEW4aGEA6HdFOdMQEmkwyHwwGX04WzzzkbZ5x+RsJ4wN///nf6z3/+g6KiImRnZ+stc42tSOPbkQqiCIvZDJfLhZycHJxwwgk477zzEp7f6/PRiy+8gPUbNqCtrRVerw9hJdILXhQiWV8mswlmsxkWswUmswmyJIExATW1NaitqcWso47C1VdfjfHjx8fEvfg13nvvPXr//ffR0dEBt9sNv9+fkIGYvydJkmAymWC1WuF0OpGVlYUpU6bghz/8IRwORxeW3X379tEzzzyDqupqdLo74fV4EQgEEAoFETaQQRqtN5Msw2yxwGG3IzMzE8OHDcfFP7wYEydMZPExjc7OTvrTn/6EDRs2wGKxID8/H06nE7Is65lWRrZevpbC4TA8Hg8qKytBRPjhD3+IH//4xwnn6JNPPqHly5dD0zR9fvg5jC2Q49T4mLa5ybIF9Z8b2u/G7FEGvZ86D4DzlsydnZ345ptvoGkaLrroIlx33XWMs1r3tMf556qrq2nq1Knw+/0wpgMLggCv14vFixfj5JNPPnggkvYxHlxfu6qq8Pv9mDcvwjaakZHRxZ/59NNP99qfyTfGtm3b9PMMpYOz4P7jmWdislD4Jjvv+wsinxVYv64jyhJt37FDzxjic/jqq68O2LN8/PHHXTKe/H4/5veXvTf67MfMnUNGZl3+DIsXLx6wZzjrrLOIswMbmXDnzJ3b/d+yuCPJ55wuF23ctElnjY1n8U3Eitzb49FHH415D0SEhoYGPYNxqB+//e1ve7XP4zPfjLEQp9NJgiDQUbOOIiPrc5qN91tmhkqShGuuuYY+++wz3fIwahDTp0/H5Zdf3uusK64R/vKXv0RnZ6dOYzFUXBncNbdxwwbgxz+OMctXrV5Fb77+BhwZTmhqco2sJ2tZkiR0tLVjx/btGDtmTMznX3vtNQiCgIyMDL16N9Xz8iHLMtrb27Fz5069xoL7sjdt2kQff/QRnBmulHpWJIpXcE1z27ZtaGhspKLCQmY815IlS8AYQ3Z2NkKhUMpzE18gqaoqPv74Y9TU1FBpaamurXq9XqrcVwmz1QKTLMeQMPbmeWRZRntrGx559BE8s/AfMetQkiP1MU6ns8t7SHUtCYIAn8+Hvz72GK6++mq9b4YoimhubobP50NGRsaQ5NHilgQQITg97bTTaN68eSlZDDwudsMNN+C5556LofLhFCerV63GP//5T7rssssOihWSBpCDDB5PP/00Pf/883C5XDp48A2oqiruuusuyLKsB8564x99//33ifdV7g8D7qD4SqPPVxKlJTEKpeXLv9ApOfpTUatpGkRJRFFRUcymA4CGhgZomoZwONznayTjtQKA9vb2mCB0f4bP50NLczOKCgtjhKDf79eTLPp6jQN/RzpFvkHwM5vNRqFQSE9m6HPsQ2DYsH6Drjxwf31ZaZnu3uzPM5hMJlTu3YutW7fS9OnT9RTgkSNHslGjRtH27dthtVqHBOV5osHdYb/73e9gTBXvaf0pioKcnBx2ww030G9+8xuYzWZ9bjVNgyzLuPvuu7FgwQIyNhkbrJGuRD+IcY/Kykq66aabuhAD8iDYaaedhu9973u91hz4AvnzA38eskFLvpCPnD69i8a6ddvWAendHQqFUFxSgiOOOEIHD36dgQTURIJPGYD+4/yeQ8EQ2jvau2j+AyUMI3T4SowCAwBmsxkOh6OHbn0pvm8AwVAwpmYBiPCa9Yci37hnwuGwzgHGtXCLxYKTTjqpVwrYoVIo7XY7vvzyS7z77rspkSzy59Y0DT/96U9RUVERQ3GiaRqsVit27dqFxx9/HAeD4iQNIAdReN56661ob2/XKTSMgsFkMuH//u//+rQQBUHAl19+SZ9/9jnsdvuQa8DDGEMgEEBxcTFmzJihC0oOkpV79/bKlZRM8KphBZMmToLT6WTxG6cv7pLevuOBmitQV+sAiPT4GJBNLwgIh8Ooq6vrAk7FxcX95p/iJIk1NTWoqa2JmZiSkhLk5OQM2PtIxPo8Z86cAX0ngyqABQEPPfRQyhlr3Mp1OBzs5t/+Vk++McoDi8WChx56CPX19YPOk5UGkIOgaYiiiGXLltGrr74Kh8MRQ+MtSRK8Xi8uvexSTJ8+vc9+y8cff3zIal1cYB1zzDHIzMxkHOC41bC/pgZMFAZkw7syXF2Eh9frpfb2dqRCYpeq9tv1GQcWnHw+v67J82HsjthvkAJitHcuZMaPG9fFQuwLgMiyjM72DqxauUr/GREhKyuLlZWVdeF26otVL4oili5dimAwGEM8euSRR3bZZ0NVNthsNnzxxRf48ssve22FXHbppWzSpEl6X3ojeDc2NuLee+8ddKbeNIAcJBfWb3/72y5aBhegWVlZuP0Pt/c6b55vot27d9Pbb78Nq9U6JDcNf6aTTz45RpgAQH19PdXV1UE2mQbEjaUY3DL8fHV1dWhsauxi+fV1yLLc9WeSzC86IHMW4s9hOF9ubu6AaO18DjZGAcR4zsmTpwzoO//iyy+6uP6mTJmiKxb92VNWqxU7duzAunXryPhs5eXlbOwRRyAYDA55KngOGgsXLuzV3BIRzGYzfnfrrV2sOd65cOHChdixY8egWiFpADkI1seLL75IX375pd7f2KhJBAIB/PrXv8bw4cP7zHf1yCOPwOPxJBRsQ8WFJ4oipkfjH8YugmvXroXP44Xcj8ZQfC4jQdSKLiBVXV0Nv6//DbV43UOi/iEmsxmCKAADQC8vRHmv4kdxcTEGgutI5/rav19/H1wAjR41CkwUBuSdC4KAb77Z2QUs5syZMyDPwdeRkQqeW+GnnnoqNE0b8CZqgyEjzGYzlixZgqamJkrVSuZWyAXnn8+OOuoo+Hy+GMuYezbuuOOOQbVC0gAyiEKTMYaOjg6644479FRD4+L3+/2oqKjAr371K/QWPLj1UVtbSwsXLoTVao0pIEt2GAsIB/MwLnS3242SkhJUVFR0IcvbvWe3ntqYiA69u0Mv7hIFeDweMIHhoosu6iKweDaOEu2v3ptr8OvwIk9VVZGfn68Le/6sebm50FSti0/a+LlU6Ox5lldhYUEMoADQOzO2t7d3OY+UoOAtEdUMn2eeHWVcq0SEsrIyZGRkwNPpjrnfXr+TqNZrtphjlAgiwumnn46ioiJwt2Jf3ge/BhFhx44dXUDlyiuugN1uh8fjSUgeerCP7vajJEmora3Fnj179L3dG8Xs97//fRfXl6IosNvt+M9//oMVK1YMXufCdIHf4BKg3XrrrV2KfnjhDwB68cUX+0SCZqTsPuWUUyJFdKJIjDH9EAShx0MUxZhDkqQeD1mWYw6TydTlMJvNZDabyWQyUUVFBX366acxdOD8/tva2uiM751Bgij0vUhRFKiouIieefYZiudm0jQNoVAIV199NVmt1j4XsAmCQNnZ2fSLX/yC/H5/DE8WL/q76+67yWK1EhMFkkwyyWYTyWYTSSaZBEnstvAOhkK9c849N+E1iAivv/46VVRUpESVLwgCSZJEZrOZLBYLWSwWMplMJIoiZWZm0ptvvhmz9vg1ln76KU2ZMoUkue90/IIk0pgxY2jV6lV6MaHxGh9++CFVVFSQKIp9LxoVRcrNzaVXX3014XO89NJLlJWVpa9Fvh7jj/j1zI9U9kL8/uFHKnuPH2azmc4//3xyu92UiH8tlQLiefPmkcBYDN270+kkxhjNnz+fEnGWpckUh3DMgzGG3bt30/Tp03XN10gc5/V6cfTRR+Pzzz9nffUHc81RVVWsW7eO4tuXdtE+41wjCVutGn6WrB1rMg070d8QEUaNGgW73c4SkQPyf2/evJlq6+rQ0d4Oj8cDf8AfpRVRow26AEEQdWoOi8UCu80Gp9OJ7OxsjBs/HpkZGd3mve/du5f2VVUduEYcDYiRLFGWZZjNZthsNrhcLmRlZaGiogKFhYXd+qh27d5FjY1NsNttEARRd1OEwyGEgiEEggEEA0EEQ8FoTUrkmiaTCXabDQUFBZgyZQqLnx/jv/1+P7Zs2UItLS3wer16lbd+3xYLLBYLLGYzzGYzZFnW3RuKosDr9aKsvAwlxSVJ34mqqtj81WZqamqG290Jr9cLvz9goHo5kAghiRJkkwyL2QK73Q6ny4XcnBxMmjSJ8U6Yia7h8/noq6++QnNzMzo7O+HxeCK0KeEw1Gh1Odfeje/dFn3vWVlZGDt2LHJzc1miPSgIAhoaGqgu2nwsnhQzXolOpFgn+ll//tawaXT6n/z8fIwZM6ZPvk/uJv/444/p5JNPht1uj7FguAfgrbfewtlnnz3gxYVpABnE2McPfvADeu2112KIz4zuq6VLl+LYY489qORnhxJUE4HkQBY6dTePfans7+01BvI9JpuXgbxGsjkZqLnq7lyH23McDJd3IqWvN89/2mmn0ZIlS2JirVzWTJ48GatWrdJ56AZqz6UBZJDAY+nSpXTSSSd16QvOKT0uvvhivPzyy30Gj3gq85T8pqxvxNY90WZzX65Ry+Ian/H33W0eY4V3dzTxRrpzliDYHE9bbtyU3DUQ//uegtrxvvPuNno83XoyQdHd+0l1vpKdryfhwGNlyehUEv2bP1d3ayF+zvX5MlCqx88jxdGlGJuKJaOQNxYn9iQMU6GV4dT6YBHnWDydfjwV/UDsqXhFIZ76PtmzdydzvvjiC5o3b16XCnwucxYuXIgrr7yS8RbBaQAZgloEP+bMmUNr1qyJKezjRUCyLGP9hg0YOWLEwe8gNkDa8FCg6ubz+m233ob6WujP++M+fP4eh0om4eFGRc9B5LzzzqP//ve/MV4PngBSMmwYNm3cCLvdzgbKCklzYQ2wm0YURTz77LO0atWqLq4rHvu48cYbUTFyZJ+sD67RP/zww7Ry1Uo9u0szaNZEBAZAi/seOmEf6bpXl+8Z9M8pioKrrrwKCxYsSEib3djUSA8/9DDWrVsHn88HWZZht9shyzKqq6tht9uRk5PTpXVtPB24JEm6r764uBi///3vY2Im/PNut5v++Mc/Yu3atejo6EAwGNSb9FitVtjtdtgdDjjsdlitVoiiCEVR4Ha7EQqFDtDPx+tMEXc0yEB0l0h7Nf47UftX4/epfC7eaov/jP4z0nrsxtpdK+JkWUHJMoSICIFAIErjHpk3u92OH/zgB10o1Pn3a9eto0f+8hcoaqQVsSybwFhkDQWDIfgDfvh8Pvi8Xvh8fvgDB+JP/JllSYLNZofL5cKso47CLbfcApfLxYzKlyAIWLZsGf31r3+FPxAAE5huPQhM4EaB3v6WMW5p8e+j8xO9f709MBiINLicLtx3333IzMzsQnXf1NRE99xzD9weD8JKGOFQGIoShqKoUDUVmhpphMb33AELx5CJxxgYEyAIDEwQIBi+F6OZa7m5efjTPffA4XCw3rq1brvtNrz33nsxlpKmabBYLNi7Zw8ee+wx3HLLLQPnQkxnTA0sVXt7ezuVl5eTyWQih8OhZ0Q4HA4ymUxUXl5O7e3tZOzd0NvMrscef/ygUU4LkkhfffUVGTM+NE2D2+2mI488clCu+eSTT8ZQvvPnvuqqq7pQxA8ELXj6SP34xz/+0YXK3u120/Cy0t6di0Xo65kokCCJxESBmCjE0MRf+ZOrYrKr+H6ZNn36oD7jsccdS+0dHfoe5evw5X/966DN82WXX97r7Ez+2YsuuqhL5qfdbieLxUJ5eXlUV1cXkxmXpnMfQtbHww8/jMrKyoSB81AohDvvvBMZGRmst7QjxqySO+64HRarBXJcbcmgmMaKGnMNziq8aNEirFu3DpmZmYbmTqxfLgBZluF2u1FdXd3FH9za2kpvvfUWLBYLZFlO+NzJMtAOdzdtKvc/WO4Wfm1JkuB2u/Hkk0/iiiuu0NcuYxHtWRJFmCyRrK9E90u9oIbne+WDDz6A1+slbo3yPTZr1ix8tXkz7M6BpyuRZRnLPl+G//73TVx+2eUxVd7cqnW4nFAVNRI7GeDBLZUXnn8eP/7x/9Dx847vlaciWjqA//73vzFzwylOmpqacN999+Hhhx8eENmRLiQcIPAQBAFVVVX0yCOPJGTb9Xg8mDNnDn70ox/1yXXFBfKDDz+E5qZmSLKsp58OxkFE8Ho8mD59GiZPntwlVrN+/Xo93ZNrasa/j/93Kgen+S4rL+vi0tlbWYm2tjYIghCj/RqPROfitOGH85HoWbt79sG4Ni+QrKurQ2dnJ3Ghqqoq7DYbO3LGDIQCwZhulMneQU/PwtPem5qa9OI6o6vvuOOOHbT3yp/z/Q8+6OIKnDplCqx2G/x+P1RtcOZbMaQv/zFKrpqqcsALUSdOnMguvvhi+Hy+mGC5qqqwWq1YuHAhdu7cOSAUJ2kAGcCA2913352QbZf//t577+1TwJwDVE1tDS18eiFMFvOgM+4KggAQMH/+/IQB69WrV6eU4dIXK27SxEldNk5Lc7MuWNLj0KxxSZLQ0tKC+vr6LkJ94oSJA2oJSZKEUCCIr776KmYPAMARY4+AJEuDsgc0TQMTBSxfvhydnZ3Es6E0TUN5eTk78sgjEQqGBnUdqqoKi92GT5cuxYoVK1ImWeTzT0S45ZZb4HA4Yij7Ocmlx+PBXXfdNSAUJ+ndOAAvm3eke/HFF2G327uw7Xo8Hlx88cWYO3du3wLnUQB67LHH0NbaCtMAEA+mtJEEpgMI30SMMezZs4fWrFkDi8UyYJuYMYZgMIiioiKMizLCGosvW1pa+kypMZB574frSESr0R2lSjwliiAIkGUZgUAgBkD4KB0+fFDue/2GDV3cdBUVFSgoLER84exAAaXFYkHN/v1YsWKFvhe4pn7O2WcDByFDSxQEqIqKv/39771W/DRNw+jRo9lll10Gvz+WA45TnLz66qtYu3ZtvylO0gAyABsTAO644w4Eg8EYcIg07gkjMzMTd999d5/iAkQEKdKqk5555hnIZtOgWx+MMQQDAZSWlWHGkTNiFiYALFq0CF6vt9cpl0YBFg8GPJts8uTJcLlcjIMVn6/XX3+dJynA7Xb3eHg8Hng8Hni9XgSDwZiUxniOq28DuCSbV6MGHQ6HEQwG4fP54PV69Tnq7vB6vfB6vfD5fAgEAtA0LQZA+NwNLy2NuX5/C9Y4OHHKef6eNE2Dy+ViRxxxBNTw4FikvDsmd2MZ3bfnnHMO7E5Hl2Zcg6GYmixmvP32W6iqqqLedIjklsVNN92EzMzMLtT5vL3C7bff3n9LMQ0B/bc+vvjiC3rnnXe69CDgabu33HILysvL+2R98KD1c88/h8aGRjhczkGnbBdFEaqiYt68ebDb7fp98030gWFj9aTtGov4uI+5OwC85JJLDlhA0TRfzmc1Y8YMjB8/Hrm5uXraL6eEUBQFoVAIfr8fXq8XnZ2d6OjoQHt7O9rb29HR0QG32w2/39/lmrIsQ5blmILI+MK4oWhNGFNpFUVJ2q5XkiRYrVZkZmbCZrPph8VigclsgizJMeuSg00oFEIwGNTTeTXS4PP6YLVauyhQ4444AqIc6UmvCypJ7NYSjE91jk9nFiURO77ZAY/XSw5DIF0QBMyeNRsff/jRoIB/xPoW8PEnH4MX3fFrjxwxks2dO5eWfLAYNsfgNW/jQe/Ojk788+V/4pbf3pJyZT13eZWVlbErr7ySHnzwwZikHt4//f3338eHH35IJ598cp8LmtOFhP1caIIg4JxzzqG3334bTqczhkIgGAyirLwcG9avh81m61PxDk+fPXLmDNry1VewHIQ+z5IkwdPpxrvvvqu32OUCwOPx0KRJk1BbW6t3yDNWFXNhFgqFupzX6XQiNzcXxcXFGD58OEpLS1FaWgqn04m2tjaUlJTg/PPPH/A+zn6/Hx0dHdTS0oK6ujpUVVVhz5492LVrF/bu3Yvq6mo0NzfHaJWiKMJkMkESJYAhJuX6UAIGT2+Ob9HrcrlQWFiI4cOHo7y8HGXl5Rg+fDiKi4qQl5eHrKwsOJ1O2Gw2Zjabe625K4oCVVMBinRGTPR+PvjgA9q5cyd27tqFzV9tRnVVFdrb2+HxehEKdNNSWGC6xcTdZXw9KYqCFV98iWnTpjFjNfzST5fS/BPnw+qwQxsEIR7pLaNgzZo1mDJ5MuNuLEmS8MKLL9Dll10+6MqcIAjw+3yYedRRWPnlil7VhHAFrL6+niZPngyPxxPTUI0rt7NmzcLy5cv7zMeXBpB+WB+CIGDd+vU055hjugTOOYnZK6+8gh/84Af9KhrcsWMHTZ0+rXf0A9Hiqt6+X1EU0dnegXHjx2HD+g3M2EaVMQa3200jK0aiuak5Yc9lSZKQnZ2N4uJijBgxAqNHj8aYMWNQUVGB0tJSFBQUwG639wkdUrUMjAV0qWyK1tZW2rdvH7Zv347Nmzdj8+bN2L59O2pqamIEtclk0oVcfzv2JfqaCDS48A4EAjFgUVFRgUmTJmHatGmYOHEiKioqUFRUxBL1KunuPrqby55oW3oaPp+P2tvb0draiubmZjQ1NaGxqQmNjY1obGxAY2Mjmpqa0NzSgtbWVnR0dCDg88dYMJqi4pVXX8EPLvwBUxRF30PBYBCTpkymvXv2wp6iJcBdU6lwj8iyjLaWVvz+D3/AXXfeGXNtt8dDEyZOQH1dna7Q9VfhYRG+l6RAsG3LVpSWlvaqZxCXObfffjvdddddCQub+yuj0gDST/fVxRdfTK+88krMy+Ev5sQTT8THH33MVK1v5iFfLHsr99K48eMR9AcOyrPZ7Hb8979v4uSTYk1bfj/PPvssPfroo9Gq2VyUl5ejoqICo0ePxsiRIzF8+HDk5eWx7vigEgEBd1kNhjsgkauku+t5vV7avXs3Nm7ciNVrVmP9uvXYvXs3WltbB1TrjK8Kj+cFAwCbzYYJEybgmGOOwbHHHovp06ejvLycJeOJin/GREDQF0u4p7/l6d98D6R6DUVR0NHRQU1NTaitrcWOb3Zg+fLl2LJ1K8rLyrDw6YU64y5PHRdFEc+/8AL9z+WXD+peeOmf/8QlP/yhvg/4V26FHIxRUFSIdWvWori4uFfUR/xdtLa20uTJk9HS0qK747jFEQgEMHbsWKxdu5Zxy7I3ayMNIH0BD02FwARs27aNZsyY0aVhDd/AK1aswJQpU/rFtsvdOV+u+JK2bd0G2SSDc79F+nCzqBcgKoC4UDJ8f4DKgR2geACL0jnQAToHxqAoCiZPmoyxY8f26ErqSRsyCpRkDadSEVjJNPVEANSdBt0doBkJCo1uFONoamqiqqoqVFVXoa62Tqch9/l8erOpeDp4Tj9ut9v1r/Yo1YrVao2hW+drh1scPBlgypQpGDt2LEs2v6nObaJ57HH/xxEh9gaAuotxxINnX/fFJ598Qq2trRFrmDQADKRpepGfTtVDZKD70fROwZHPRkx2TTtAQRIKhTBu3DjMnTs3KeX96jVrqKO9HUxg0DSKnFeLriWK/psIpJFOK8Q/k/j3Wuxno/vr+OOPR2lpaZ9cuzyG8+c//5luuummLlYIJ1r829/+hmuvvbbXRItpAOmH9fHjK66g5559Nual8Bdyyy234J577mGKqkI6DMn+ugOHeKvEKByMAqEnRtpEgqU/QiXV54rX0JMx0xpBZSi0RuVV0d2BRTzVxGDOaaK4UCIOrr4AjTGBortnHewMumTXOFzIFvl8er1emjJ1KvZXV8NsNscoS6FQCEVFRdi0aROcTmevYrVpAOnDpmGMYfv27TRjxoyY3/GXUVhYiM2bN/f6ZfQEWoP2rgxaprEtbl83V7xQMNJvpyrIVFWFz+cjnnLKU0n9fr+eFaQ3ggJBYIIe+DabzTq5osPhgNPphMPh0MnpeprjnkAlWewgEUV5d7GFZD+LtxC6ex9GId7TewuHw/B6vRSfmsvTnI3vSZKkLlaU1WrlX2NiY71Zu8no8fu7Lwaj73dPLtW+ZmDp9PEpjlT3Y09WyJNPPknXXHNNUivk7rvvxm233dYrj0kaQPpofVx66aX00ksvxVofoohOtxsvvPACLr300u9Eo6hkGm93zx0MBtHS0kINDQ2ora3Vj7q6OjQ0NKClpQXt7e26e4gDBqevSHXz8/oSq9UKp9OJzMxM5Ofno6SkRI/bVFRUoLy8HAUFBUldRP3dwAM93xw0EllFHq+HqquqsWfPHuzevRt79+7F/v37UV9fj9bW1pg55QCcLKvPyJYsy7I+l3Z7hDE3MzMTOTk5yM/PR0FBAQoLC1FUVISCggLk5eUhOzu7x4SJRAzNAwUu6RG7R6Op8LRjxw5YLJYYhUFVVTidTmzevBkFBQUpx1rSANIH8Ni4cSPNnj07JiDFA+fz58/Hhx9+yDgtx7dpAcZ/351gJSI0NjZSdXU1du/ejZ07d2L37t3Yt28famtr0dzcrNOsJ9O6jJXQfYmhGAPSRo4v4xBFETk5OSgrK8OECRMwY8YMzJw5ExMmTIgRfjydujfB4YEGDX6/xrFnzx7asGEDVq9ejY0bN2LXrl2or6+Hz+frch4+n6IoRmJmcVZWsiZP8XPJ5zPZkGUZDocDWVlZyMvL09O2y8rKUFZWhuHDh6OoqAi5ubnMWFPSHbik2tArPbqXXa+88gpdfPHFMSUHRivkhhtuwIMPPpiy8psGkD68hO9///v0xhtvdDEFiQirVq3CpEmTDrn1wTd6/KZLJYXUqAn2pIUoioKGhgbat28fdu3ahe3bt2PHjh3Ys2cPamtrE2YtGaul411F8WCVyK3TWzdE/DMlq1sxMs+WlZVh5syZOPnkk3H88cdj5MiRLN51Mdjv11h7wIfb7aYVK1bgww8/xLJly7B9+3Z0dHTECAKTyRQDdInmtLfzmQhoEglzY2sDTqSYCMhcLhdycnJQVFSk166MGDECI0aMQFlZGYqKirq1Xnqqy+mOmTkVBejbOPicddfsTpIkbNi4MeVmd2kA6SV4LP/iCzo+rm0kR++bbroJ991335AAj4HcGB6Ph9ra2tDQ2ID91ftRWVmJ3bt3Y/fu3aiqqkJ9fT3a2tpiXCGMsYR1Ez3FERIJgVSzqZJpsd0BUyKwVFU1hv4kIyMDs2bNwrnnnouzzjoLw4YNY0ZhOdDvmmdz8fsJBAJYunQpvfHGG/j444+xd+9e/bNWqxWSJPVYQd9fDT6VeUz27hKBdncA43A4kJ+fj7KyMowePRpjx47FmDFjUF5ejqKiImRlZbGBWt/GItnvigz74IMP6PTTT09qhfz4xz/GM888k5IcSwNIL4XySSedRJ988oneuF4PnBcVYtPGTXC5XOxQ+m/5fX69fTu98/bb2L59O5qbmyMU1NEFIcsyLBYLrDYbbIZ0UlEUEQ6H4Xa70dHRgba2NrS2tqK1tVXnoIrf8IIgxABFb0AimVvKKGSM7qf+DB4T4dZP/L3GxwGMgfRQKKQX8uXm5uL000/HFVdcgeOPP57Fb86BFGbbt2+nf/7zn3j99dfx9ddf65vcYrHoRZzJsqCMz2cU1v1lMeBxkXiyxUQur74AjJHyJn6tWSwW3S1WUFCA3NxcZGVlweFwQJIknfLG7/frhx7r0VSIggibzYbs7GyMHj0aZ5xxBsaPH68rA98FEOHy4bTTTqMlS5bocix+Ha5avRqTJ03q0RWfBpBeIPe7775LZ511Vgxyc9R+4okn8NOf/nRAG9b39T7/+thf6ab/vSmmerm/AiMR+WCqWUmJhISRvynR38qyDJvNBofDAZfLhYyMDLhcLrhcLjidTr2Wwmw26/dGcZxYHo8HHZ0daG+LVEPz4Lzb7Y4RpKIo6udJBChGMAkGgwgGgxAEAccddxyuv/56LFiwgPHn4qDa201t/LvPPvuM/v73v+O9996D2+2GKIqwWq06x1EywDDSnBg/Y7FY9IB3bm4ucnJykJWdjQyXCw6HI8KJZaAniecVc7vdaO/oQHtbWwy3mNfrTbjGOHNvX61P/vn4vx1IIOTW27XXXot7772X8T37bQcRLiNWrVpFxx57bBdmby7PFixYgNdff71HKyQNICmY7nzxzp49mzZt2qS7rwRBgN/vx+Qpk7Fy5UomidIhyx7hL/rNN9+kBQsWwGq16gy3qfq2++Ky6EmT5CARP2w2m54VVVRUhGHDhmH48OEoKSlBUVER8vPzkZOTg4yMDNjt9l6ljSZ7HrfbTS0tLaitrcWePXvw9ddfY+vWrdi+fTuqqqp0YWgU2PFV4UZh7fF4AABz587FzTffjDPPPJP1xi0S7wL78MMP6aGHHsKSJUugaRqsVqvefTEZoIXD4RiCyOzsbIwcORITJkzAxIkTMXbsWJSXl6OwsBCZmZmstwzKie7Z5/NRR0cHWlpa0NDQgJqaGlRVVelHTU0NGhsb0d7e3uXdcws4mcWaimssVVdcT/ERVVXh9XqxYMECvPLKK8yYsPFdAJFLLrmEXn755YTdU/1+Pz777DPMmTOnWxBJA0iKk/3MM8/QlVdemZCyZNGiRTjjjDMOWeyDb77Ozk6aMmUK6uvrYTYPTNOpZIVhRmBN5G7gvuycnBwUFxejtLQUI0eOxIgRI1BeXo5hw4YhvyAfWZmp+bPj/fu9afPaU/2J3+/Hrl27aM2aNfj888+xYsUK7Nq1SxfuVqtVDzLG850BgNvtBgCcffbZuPPOOzF16lTWk1vL+LtVq1bRH//4RyxatAhEBIfDkfB6XMAZXWpOpxOTJ0/G3LlzMWfuXEyZPBmlpaXdBp85GKW691OdRz7C4TCaW1qotqYG+/btw+7du7Fr1y7s2bMH1dXVaGxsREdHR0IeNX4kWmvxis1ArG1ZltHe3o477rgDt99++3ci9Z4n1+zatYumT5+u/9uYUerxeGIySpPt0TSApCCYvV4vTZ06FdWGKk4+yaeccgo++OCDg77wjAVU4XAYFosFDzzwAP3v//4vMjIyEmp+STdSZDd1eW5j/CERQAiCAKfTiezsbBQVFaG0rAwjR4zQgWL48OEoKCiAy+XqsYAvXsPuT+C8p/cZn4kTPzc+n4/WrFmDRYsW4b333sPWrVt1V5DJZOriRjICic1mwy9/+UvceuutOhV+IjeOIAior6+nO++8E8888wxCoZAOHEbgT2TxZGVlYc6cOTjzzDNx4oknYvTo0Sz+ORPRnAzkPCayHPh1ultrgUBAT+/eu3cvdu3alTC9O5HlHB976SmtO1lv9vi1pmkaTCYTtmzZgqKiIn0vc+bfb6NFwp/xxhtvpIceeigp0eJ7772H008/Pal8SwNICpN8zz330K233hozydwfvnz5csyaNeugAkiigJ/b7aYJEyagoaEhYUvdRH0wuhuCIMBiscBut+v+88LCQpSUlOg07MOHD0dxcTHy8vK6rfI2CrTeuiEOpqLANTFjDCsQCODTTz+lf/7zn1i0aBHa2tpgMpn0bozxQBKtoMfEiRPx8MMP46STTtLjI1wgAcALL7xAv/vd71BTUwO73Y74tqVcEHMXlSiKmDVrFi666CKcedaZGFE+ghk1Sp65lSoD8WDPZyKASQTWxuH1enVCxerqalRVVaG6uho1NTWor69HS0sLOjo64PF4EAgE+kRqKQgCrFZrQr//z3/+c/z1r3/9ThSY8LXe1NREU6ZMQVtbW5e6Nq/Xi6OOOgrLly9nydZVGkB6mOBEfPp8wV144YV49dVXDyp4cHNy0aJF9NZbb4GIkJGRgbVr12L58uUx6cX88yaTCQ//5S8oLirSF4iqqggrCpRwWHdBqaoKWZaRmZmJ7OxsZGVlISsrCxkZGd0WfBldI9+GimKju8wYM9i9ezc999xzeO6557B//34dSIyCjAt+bi3ccMMNuPvuu3Wa9ZaWFrr++uvxr3/9K+Hfc4HGgcPhcOCcc87BT37yE8ybN4/FW22HWy1Dd1xdPT1HMBhEZ2cnJWoUFgwGdRDltTAmk0mPt2gUIUp84YUX8M4778Bms3Vx8TLGcMEFF8DhcKCzsxMjR47EL3/5S2RkZPSqF8fhpiA//PDDdMMNN/SN7j3+haYP0rN5iAg///nPCQC5XC6y2Wz6YbFYaPPmzcSziQ7mPb300kuECJtOzOFwOGLu0eVyEQD68Y9/TP29trFLHT+MwGOsUh6Mg7vSejoG47rhKMjyuWhoaKC7776bioqKCADZbDZyOp0xc+9wOMjhcBAAmjlzJm3bto02bNhAY8aM0ddT/PtyOp1kt9v1d/nTn/6UtmzZQsZ3wCldvi3zG38PPOkifp1xa6+/x9atW8lsNsfMu/GI31Pf+973iN/bt03G8efy+XwYP348SZIUsyYdDgdJkkQTJ02kQCCgv/8Y0su0BZLc+khEmMitj8svvxzPPffcQbM++HsKBAKYMmUKVVZWwm6369ZGfKaO8e/WrFmDI444IqYpTnd+4kRB84HWvnrK8oq3YFK9fqKNMlAWEZ9jbpXU1tbS/fffj6eeegp+vx8ul6uLW4uvl9zcXBARWltb4XQ6E1otnZ2dkCQJF110EW6++WZMnDiRAYhJ2hgord+o+Q/E/CaLXw3kukkWSE8lJZhr3GazGeeeey699dZbXQrp+Bwb04jb29vx+uuvY8GCBd/KADt/pjfeeIO+//3vJy0ufOqpp/CTn/ykS5lCGkC6mdQLL7yQ/v3vf3cx7QBg3bp1es+Mg+FC4C/u6aefpquvvjrhPSUCuosvvhgvv/xyQvA4FH7wVN0V8SNqBZDR4jESHUYJ/1iydqvJ3G29FaDGmA4HkrVr19JNN92EpUuXwm6364FZo1DiSQ284M34nnhtybHHHou77r4Lx887XgeO3sY04vmjUp1rbk0qikLxrshoHRDjGVJ9cQcmSpI42DEwVVUhSRI+/vhjOuWUU2Cz2bpNc+fprFOmTMHKlSsZD95/W+XdySefTB9//HFMcSFvzV1aWoqNGzd2ac2dBpAkk7l8+XI6/vjjE1KWXHnllVi4cOEhsT6mT59Ou3btimHTTKZ5hcNhrFixAtOnTx8UAOmvP7u9vZ1aW1sjrU4bGw+0OG1uRlu0aM3tdusFa8FgsAuDLL+ekS2WM+9GuJYKUVIyDGVlZSgtLUVJSQmys7NZIoDuLfOuEUiICPfccw/deeedICLYbLYuVobxXRqtjry8PNx+++247rrrGG/q1Rvg4AI6WZC6ra2NeK2GMSjd3Nwc6Vnu8cDv9+v0LUYrip+Tzy+ndnc6nXC5XMjKytILFPPy8pCXl6cXK2ZmZvZIoc/3nLEPyMGwfI899lhauXJlDB9UosHjAC+//DIuvvjiQ1ooPNgyb926dXRMgvbcXO498MADuPHGG2PmIA0gCTajIAiYP38+LV26tEupP2MM69atw5gxYw669bFw4UL6yU9+krL1cd555+GNN97QU0kHyk2QSsEVp2yvr6/H/v379SKz6upq1NbWorGpCW2trXC73QgEAj1SivPr8edgSdKOjf7q+GG1WnXm3fHjx2P69Ok48sgjMW7cuBhBx8+RKphwl6cgCFi6LdCQaAAAlWVJREFUdCldddVV2LNnT9L3JAiC3nVwwYIFeOCBBzBixAjG7z8VoDcSLRrnoqOjg77++musX78eGzZswLZt21BdXa3T2SQTkt2lxSZi4+1u8Oy9jIwM5OTkoKCgAMXFxXqx6LBhw/Ri0Z6KG5OleCdbez0BjqIokGUZL7/8Ml1yySUJ3ViJrJBJkyZh9erVLH6+v20gcs0119CTTz7ZJeNUURTk5ORg8+bNugLGGEsDSKJJfPudd+ics89OSFlyxRVX4B//+MdBtz6CwSCmT59OO3fuTMn6CAaD+Pzzz3H00Uf3aH0YBVdvgKa9vZ2amppQEy0Yq6ysRGVlpQ4SvOVrPGW7weWUkO4ingfL6HLqzu/OteVE9QJGmhMj1YfJZEJpaSmOPPJIzJ8/H8cff3xMXUVv4g9cODU0NNAVV1yB9957D3a7vQv48pqP++67D1dddRXjLqSeNFtjqrHxfrZu3UpLly7F0qVLsX79euzfvz9m85vNZp2hl9eUcOvJOMfGGolEsSMOqMYjnneLn9cYEE+0Vm02G1wuF3Jzc1FcXIyysrKY/izDhg1Dbm4u6622ryhKt8oNt3ICgQCmTZ9Ou1Ow5rkV8u9//xvnn3/+t9IK4euqsrKSZh99NDo7OvSsU6P8451W9dYGaQDpqmXNnj2bNm7c2CUl9lDGPv7xj3/QVVdd1aP1wVNITzvtNLz33ns9Wh/xz+H3+9HZ2Ukej0fvBOj2eNDU2Ijq6mr9qK2tRWNjI9ra2uD1ersEZ40plFxoGfmvkvEZcV4qi8Wi9w23Wq2wWCx6//D4c3L+J7/fj/gOhvEFlbzLnizLuiXg9/t1kMvIyMCMGTNwzjnn4KyzzkJ5eTkzuqt6sry4n52I8Itf/IKeeOIJXUAZNbmPPvoI48aNY6m4q/hzGoXWjh076L///S/eeecdbNy4EV6vFwD0ueP3wOcmHsQ5sPB5jp9fI90Hf2c8VsMPnhGWbB0aK8uNwWm+DoxrwThMJhMyMzNRUFCgU72Xl5frdUe5ubnIyMjQaV54urXL5dIpb7qTaxzoH3roIbrxxhtT2lNerxezZ8/GsmXLGFcEvq0K9H333Ue//e1vu1ghqqrCarVi8+bNKCkpYeksrJhFpUKSRLz40ot02aWXJbQ+DlXmVW+sD0EQ4PP58NFHH+GEE07o1vrg4LFhwwZauHAhtmzZgvr6enR0dOhClcccErnJOAsvFw5GgDD21zC6kFwuF7Kzs5GXl4fCwsKYg3exy8zM1AkTo9XfKWmiURp28vl86OzsREtLC+rr61FVVYU9e/boVc/79++P6aHBBSi33LgwzsnJwcknn4z/+Z//wSmnnMKMZIPdvX8uoFpaWmjUqFEIBAL6HHm9XkydOhXr169nxuK/7tYA/0wwGMS7775Lzz//PD799FO43W4wxmC323VSvGAwGNNMym63o6ioSO/AyBkCCgsLdUFss9n0eY4nzIwjviR+fq/Xi87OTrS3t6OlpQVNTU1oaGjQj8bGRp280uPxJORDk2VZByyjdcRTpxMBFC9wtdlsuoZsABCcdNJJ+NOf/gSbzZa0doPPaUtLC02aNKlLEV2yfeX1evHhhx9i/vz538qMLD6XgUCAjjrqKMR3LuRy8H//939x//33R5SfNIAcmDguqBMFqXk67Pjx44e89XHCCSfg448/7tb64JuooaGBpk+fjrq6Ot0FxDc0F3rxjKi8LiIeWEwmE1wZGcjLzdUJEuO70HG/t9ls7tM7SqRd9iY9NxQKYf/+/bR161asXr0aK1euxObNm9HY2Kg/g81mgyAICAQC8Pl8YIxh9uzZuPbaa3HRRRcxrvkmux6Po1VVVdG0adP0SnLGGHw+HyZMmIB169b1eB6+zhRFwTPPPEOPP/44Nm/erAMDp9Xx+Xy6hZGbm4uJEyfiqKOO0jsrlpaW9thaNpny0h9/v9vtptbWVjQ2NqK2thb79+/XK8xramr09sW8EDD+nfJiwHiL01h7FX+/gUBATzntzi3IQf6mm26iP//5zyntLbfbjbPPPhtvvfUWG+ieO0POjf/223TOOefEKNLcgs7IyMDWrVuRm5vL0gBiENSPPfYYXX/99bF9zqOo+6Mf/QgvvvjikLc+PB6Pzl/TnabMYx5r166lmTNnIjs7O8Z3HQqFulzL2KqU9xYvLS1FeXm5DhKFhYXIycnpESDiK9eNwipRBk6qbWzjv0+FSqO2tpZWrlyJ999/H5988gn27Nmj++l5pbjb7QYRYfr06bjvvvtw0kknJQVoPrfV1dU0ZcqULgAyceJErF+/nomimBRA+M/3799Pl1xyCZYvXw5RFOF0OvXALg+KV1RU4Pjjj8dpp52G2bNn6w2v4gVDorqYRHMbn6CQbJ67q99JhXQxEAigtbWVOKPv/v379Vgad5NygEnUitgYP+MV6B0dHXjooYfw61//ulsA4QCwZ88emjZtmp75lkqsYPXq1Zg4ceJBUyQPFYiccsop9NFHH8UkEnF5+Le//Q3XXnttGkCMTLaTJ09OyCWlaipWr1qNSSk0WDmU1ofX68XRRx+Nzz//PCXqBSJCKBTCBRdcQO+++64et8jMzERRURFGjBiBioqKGJr1goIC5OTk9Jie2R21SX8124F43/yITxzo6OigpUuX4pVXXsGSJUvQ1tamxwmiiQOwWq1Yv349xo4dm1ATHQgA4bGUq6++mp5++mnk5uZCURR4PB4oioLc3FycdtppuPDCCzFv3rwYwsp4bqxDRScTbzX2NtXb7XZTQ0ODDix79+7VrZempia0tbXFWF8AMHbsWLzxxhsoKipiPTWJ4nN82WWX0YsvvphyduN1112Hxx9//FvL3Muf6/PPP6cTTzwxJhbMXXnHHXccli5dmgYQPll33HEH3XnnnQmtD16Mdyisj2nTpqVU98FN7DfeeAPnnXceSzWrRxAEhMNhrF27lhhjyMrK4nn8rKe/7Q9BYry1MFA03X2pojdmHxnnbPfu3bRw4UL84x//QFNTE6xWK2w2G1paWvDpp59i3rx5Ca28gQSQ888/n15//XXIsoxwOIzx48fj8ssvx0UXXRRD287XbKp9SBIJ9v7Md2/nPJGV2BuA0TQNXq+XvF4vgsGgzhBQXFzMuLsrlb0vSRJWr15Nc+fO7dJcKdGzKooCp9OJLVu2oKCggH1bOxlyxWjevHm0bNmyGCuEz/XKlSu/21xYPH2xsrKSsrKyyGq1kt1uj+HGMZvNtHHjxoPKecV5l55++umEPFzxh8PhIFEUacaMGWQstEv1SIX7KhxHutgbXqP48xyseTRyiMXzKvWG+6qyspJuuOEGysvLIwB04oknEm8RnOhc/PmqqqooKyuLLBYL2e12cjgcJAgCTZ48mfhnkt0Lz1JbuXIljRs3jmbOnEnPPvsseb1eMq6T7p7H6I7kz57sfQ8Gz1I8r1V/1lBv1k9v1j8/16mnnkqMsS6cZvEH55e7//77ybhXv20Hf67HHnusiwxyOp0EgB599FH6Tlsg3KJIZMJyjX7BggX497//wxQlnLypSh81ECPnTqJgYKpV5/HVsqlYH8k08FRdHt1pj6n0UOAaZGdnJzo6OsAZVjs6OtDZ2alXoPMK6VAoFNPjwtjb3W63w+l0IiMjA9nZ2XoldE5ODrKyslgiq5EDS3cU6PHcV5WVlbRs2TKcddZZyMzMZN3FL/prgcRrvcY5DYfDCefYSGPSHXV6Z2cntbS0oKWlBc3NzYae953wen0IBAIxGVDx8221WmG32+FwOJCRkYGMDBcyMjKj32fwDDqWyrrrqxWbKN4VbwWlKgMkScKiRYvozDPPTKmwMBAIYNSoUdiwYYMe60uW7dVdLCmV36Vi+Q2WBcIYw/r16+mYY46B2WyOoXp3u924+OKLIX3XwWPlypX0r3/9C3a7PSH1xPnnnw9BiGSEHKz7kiQJL730Em3fvr1Hv6yxUvb73/9+n2I0yZoAdUdK2NM1Ojs7qbm5GQ0NDXoGTk1NDerq6tDQ0KDTaLjd7hgajYEYkiTpKcN5eXlUUlKCiooKjBs3DuPGjcPo0aNRXFwcUwGdqPqcf8+BpLy8nJWXl8dssIO1Hjjg8SLMZO43IxC63W7avXs3vv76a2zbtg3ffPMNqqqq0NDQkDB+0J/BwYXTnGRlZVFubi4KCgpQVFSEkpISlJSUoLi4OCaO1p2ik2ocrT/vgYP4qaeeyqZPnx7TsjrZPVmtVmzfvh1vv/02XXjhhUkLC3ubBDKUBl9HfO0ZOxZyGbNjx47vLoDwF3rPPfdAUZQu3EW8j8bjjz+OxsZGcjgcB6qbRRFiVLjwIKwgCMl/nuDfgiDAZDIhOztbz5rhQV2/348HH3xQL5Tq6UUrioJf/epXMJlMSNX6MBLuJWJp7akqPRQKoaWlherq6lBdXa1XoRt7Yre1telB30RC3niYTKYYmpJkFk58emkii4lrtu3t7WhsbMTGjRtjBEZOTg5GjRpF06ZNw5w5c3DUUUehoqJCb63LwcSYxsybRfHNc7AEAgcwfu/x788IGm63m9avX48vvvgCK1euxNatW1FbW6u3v+Xn4Kmx3HJLZlkmiy0Zv8a7jjweDzraO1BZWZnwvcuyzAEG+fn5VFJSoqd6l5WV6TQnOTk5zLgmkoFrvHITf8+pWCScz+y6667DVVddFTPn3b2Xxx9/HBdccEHSe/T5fGQsutRdeKoCVVFjaHeStWtOtM7j2QHiv0/2NRnDdfy75K69trY2/P73v084f6Iooqmp6btZB8IDRFu3bqUjjzwyecP4qMthoPovxy8EURRhs9kwadIkvPnmm8jMzGSCIODJJ5+ka665JiXrIxAIoKKiokdzOv7ZUxGAbrdb7xBXVVWFvXv3orKyUm8/2tTUhI6Oji45/NzdIctyTD2JUdD0VJFuPFc8LYlR8zb2qkg058bGQrzrH+8pzv8mMzMTEydOxCmnnIIzzzwT06ZNiwlO95YRdyBdWImEndFKam9vp6VLl+Kdd97BsmXLsHfvXv25eHU5Vyj4s3NB1t26StSGNz7tOpn1Z6w+Nyoi8e8+Ec2J2WxGRkaGTnHCK9F5mjjvgJmZmclStbR7mmOuSHm8XpoyebLeLKynwkK/349PPvkExx13XExKt6IouPDCC2ndunW8ME+vvE8We+zJlZXImkmWMJLs3z0lmFDkJvR3HQgEEA6Hu9Dx8Dl1uVzfTQuEC9F33nkHwWAwqaAmIr3daF/8lsk+Y+SXaWtrgyzLsNvtLFpFTg899FCvrI9f/OIXsFqtPVofxpaq27dvp46ODh2E6urqdNJD7m7irg63291FQHOA4DEIo/bONwtfgPH3zBlzEzG58k6IUSZX2Gw2WK1WncuJ378xQOzz+eB2u9He3o7m5mbU19ejtrYWNTU1ujXU3t4e4+qx2Wy6WzIYDOLLL7/E8uXL8ac//QkzZ86kCy+8EOeddx6Ki4v7TK0+GC5X/n5XrFhBL7/8MhYtWoS9e/fqwtflckEURX3+jRX3FosF2dnZKCwsQFFRMYqKivSK9KysLLhcLr1AkQOuMV7Egdfr88Lj9uhxq9bWVvCYSktLC9ra2mJazyayQjiljNGC4gqB1+tFe3s7tm3b1mXt2Gw2ZGZmIjc3lwoLC1FcXKy7xoqKipCbm6s/v8PhwLhx4xgvuOyOgFFRFDgdDnbllVfSbbfdBqvV2q3yxqk9/vKXv+C4447rIuC//vprVFdXw2KxdGEZ7i4+1VMspVeyhghaN3U8SZ8NDGCRolo+d0k/+120QPhmvOiii+jVV1/tMXA2GEOSJLjdbkycNAnLPv8cLpeLMcbwxBNP0LXXXpuS9REMBjF8+HBs3LgRDoej25RCrhVv3LiRrrvuOnz11VcIBoP6RkjUWMdINWEkJTRm2CSqSLdarcjMzER+fr6uRfI+6iUlJUYajV5XpPd2BAIB1NfX0969e7F161Zs2LABmzZtwq5du3ThyqvPJUlCKBSCx+OBpmkoLCzEeeedh5/85Ce6VZIKLf5AWyDGOMgbb7xBf//737Fs2TKEQiGYzWa9B0kwGNRb6UqShGHDhmH8+PGYNm0aJk2ahDFjxmDYsGHIyclhgwmEHo+HOjra0dzcooM551Dbv38/6urq0NTUhPb29hjaFb6u4ylOjOvOmCWXzDXKP88Yw5QpU/Dee+8hNze3x/0hCALq6+tp0qRJ8Hg8PdKb8HezatUqTJ48mRndinfeeSfdcccdyMzMjFGiDra87Y+rtbt71TQNNpvtu+3COvXUU2nJkiUHHUAkSYLP50N+fj6WLVuGkSNHMlVVEQgEaOrUqdi3b1+PyB/P0Z+K9UFEOP7442n58uXIzs5OGJw0AkQysjtBEPSK9MLCQgwbNgwjRozAiBEjdNK7goICZGdn98hhlYxltzcByGS+42S1BJqmYc+ePbR69WosXboUy5cvx86dO6GqKkwmk25R+Xw++P1+2Gw2nHvuubj11lsxfvx41pPQH0gA4RlVn3zyCd1+++1Yvnw5GGNwOp2QZRmhUAhutxsAkJ2djSOPPBInnngijj32WEyYMAGZmZkslfhBb4VNd4zIqVCqt7W16RXovEhw3759qK6uRn19PVpaWhJyaHErLN49aly7xnXa0tKCJUuW4OSTT+6RlZoD9fXXX0+PPfZYyoWFRo48Pgf79u2jyZMn65brt03OEhFMJlMaQLoDkL50zktldHZ2wuVyYcmSJZg1axYLBoMwm814/PHH6ec//3mPC5c3i8rPz8fmzZuRmZnZrXbFhVAgEMCYMWOourq6R4Cz2Wx6P4fCwkIYg52lpaV6Nk1GRgbrSUNL1lq2vxpSqgs9podzlPIiPti5evVqvPPOO3jvvfewfft2AIDL5YoR0g6HAw8++CCuvvrqbnnGBgpA+HnuvfdeuuWWWyAIgu6i4szDTqcTc+fOxXnnnYeTTz5ZZw82zj+/xsGadyICgQBCnzL4vF4vNTc369l7vJdMvPXCG431JMM2btyIKVOm9MhMzeXC119/TTNmzEi5GDJKSIpRo0YxI8PBrbfeSvfccw8yMzN7dEcfTMHfW1mZjM5GkqTvtgvr+9//Pr3xxhsJAYSX7A/G/EyaNAlPPfUUZs+erWtFXq+Xpk6diqqqqpStj7vvvhu33XZbSnUfXBi99dZb9J///Edf+BaLBU6nE9nZ2cjPz9cPHo/IyMjoNljZ34r0ZAu7Rz9tN/xNvQEVPp9G4fXRRx/hH//4BxYvXoxQKKQDSSAQgN/vx6pVqzBjxoxB58ISRRH79++n8ePH6zEOn88Hn8+H7OxsXHrppfjJT36CCRMmJKQxSTVZIn6+U13zfU1T7W+7Y6/XSzze1dTUpB/Nzc16vI67ZseMGY2f/eznrDfdHUVRxAUXXED/+c9/UrZCfvazn+Gxxx7T1wQnZz3vvPNo8eLFh62s5AkoidaEIAjfbQC58sor6ZlnnumySPhGnzFjBk455RQUFRXpWRnctaNXu5IWTV+I21z65o1oYoIgwG63o6KiAscedxwzGXoZiKKIRx99lH75y1+mZH0oioLMrEx8tfmrHn27A+ET7c6KSKXYK1mmSXxgsb+aldEd1tv7NPaV4OOLL76ghx9+GG+//baejeL1evHRRx9h/vz5SV0iAwEgHODr6upozJgxemwjKysLl1xyCX75y19i1KhRzGhldCd8E9X06EJggOY/UaZWX6llkqVu9yWFujcyjruxEvFAJdtPqqrCYrFg8+bNGDZsGDMGzVVVxaL33qOmxsakVfKqpkHTVGiq1qXzo34QgZKwSPT4M9JAGiX/dxIWACLCV199hebm5i4gwi2Q72QWFp8Im93W5XeclPDcc8/Fa6+9xgar85jRreB2u+kvf/kLZFmGqnUfi+H3d9WVVyEvL6/XVeecuiEZ42qiDZ+qUOqtqyJ+PqI58xRPO2KkkzZ2MzSbzbwpEuM+8e7Ob6x9MWrnxnvln2OMYc6cOWzOnDlYvnw5Pfzww1ixYgWuvPJKzJs3b9BJNXl/laKiIvbEE0/Q22+/jfHjx+NHP/oRKioqumSGxb8jozBP5LaLH+FwGH6/n/x+f0z/eSNtunHueYaOxWKBxWJhZrM5ZZdvKm7NnmJMiYCmOysh1cFB/dhjj2Vz586lZcuWdds3nccC2tvb8fjjj+Pee+/VwZwrh2efddZhS5a1evVqOuGEExLO8Xe2pS2vHP3Vr35FjzzySIzWz2MFK1aswIwZM1gwGBxQQWEUXvw+Hn74YbrhhhtSsj5UVcX/t/fe4VVV2fv4e26v6b2QAAk1gUSk2MGGwjDYwS52nBHrOI7jqJ/xqz/HMgojNkYRC+qoY1dEQEGkiCKEHlJJ78lNbi/794esM+ece25LggY563nOk3LvaXuvvdZe7V1msxllZWXIzMw8on0JBupq8Pv9sNlsrLOzE+3t7WhtbeUbDpHLoaurCzabDX19fXA4HLzwEubNC+MXJOxJkFH1c1xcHBITE5Gamsr336Z4TU5ODhITE2UhzsPt3KUFfN3d3YyC0pEyVAYzC0v6eaiUYlJ+cmmifr8fTU1NTFjPU1dXh6amJrS3t4vSbmn85QQ9jRUVf+r1ehiNRlgsFsTFxyPx5xRbpKWlISMjg08VTktPQ0ryz5l30fDNrxk3o14h7733Hrv44osjJtkI+2Ts2rULqampPNR7tPUzR5JiGS/pd1UqFSZNmsR++uknmM1mkfLW6/XHbiU6Map08DweD5KTk5GXl/e/TIMjwLC0O+np6WGLFy+GTqeLmAlG1sctt9yCrKwsLpqU0mhcBdLfo1UQbreb7+cgTdWklrcdHR2w2WxwOBwhlaOw8l1YxCZ1cdHzUXaY3W4Xmdxyrgaj0UjzycaOHYvS0lKUlpZi3LhxoviOECpECGUiFNgJCQkRA7FHSqAJ3VrSinRqdyusk2lvb2dlZWX44YcfsH37duzfvx8NDQ3o6uoKjvdxHNSCroA0/lqtVhZvS1gXYrPZRAWdofiWkjKofiMnJ4dP7xYWCIbCLotGwURjvcRihcyZM4crKipi+/fvD4tHR8K0tbUVL730Eu6//35R++OjFfKdxthqtcp6LY5ZF5bQbJcbtMNm+RFNvyNf60svvYTa2tqoM6/i4+Nx6623Itq4h9AlI7WCIglCr9eLzs5O1traioaGBhw6dAi1tbV8RgylW9pstqBqdHIdUD6/xWKRbZUqVQDCat1oLDkSqMJCQxIidN329nY0NDTgu+++458rJycHJSUlbMaMGZgxYwaKi4t5bCwSUFIB8Gt1oZMTQBQoFiqNsrIytmbNGqxbtw47duxAY2OjCABPr9fzDamE40PWhhCtNxoSjr9erw/aBAhdaTQPTU1NImgZIpPZhIT4BKSmpjJhN0tSMJmZmVFXoIeKxQjHIhJv+Xw+6PV63HTTTbj11lsjwptQCviLL76IP/7xjyw+Pv6oh3qnNSDFAaSx1Gq1x6YCEfa2DvX5kU5zVKvV6O7uZkuWLInJ+rjmmmuQn58flfVB9wn1vb6+PtbV1YW2tjY0NTXxSqKurg4NDQ1obm7m3RvhKoqpGj1UwRd1zpM7X6/X89Xm9JMUeLie65QRRf257XY7X5EuFYA6nQ4Wi4XfUft8PjQ1NaGmpgYffvghjEYjJk6cyGbPno3zzjsPRUVFnFCJSlv7/tpWs7Aivby8nH3wwQf46KOPsGPHDn6sDQYD4uPj+d20x+MJ6pfOcRyPCmC1WhEXFweLxcL3ohcqBZpTuo507Ak52eFwhIRJEVafCyFOhNhLra2t2LlzZ9B6lFagZ2dnIycnBzk5OaLi1MTERFgsloiZg9EobcYYLr/8cvzjH/9Aa2tryGwk4cazvr4ey5cvxx133IFQIIu/BTqmFUgkRqId2ZEUAhqNBs8//zzq6+sjWh8kyCwWC26//faorA/aPVRVVbElS5agrq6OF8i0UDs6OtDd3Y2+vj5ZVFZSEDqdDkajUeTTJQVBfcOli89sNiM1NRXJycm8P5wOShNOTEzk4b9NJhP0en1EAD05N5rT6WR9fX0gZdjY2MgXp1VVVaG2thbNzc18wR0JJLPZDOBnYMitW7diy5YteOyxx3DKKaewq6++GnPnzuWMRiPPE7+mK4JcIlQd/fnnn7OXX34Za9asgc1mA8dxMJvNfIGo2+1Gd3c3f35CYiJGFhRgVGEhRo0ahYKCAuTl5VELYlitVhiNRi5WJXkYMoU5HA7YbDZ0d3ejo6MDra2taG5uRlNTExobG9Hc3CxyaYbakFAhp1wFemdnJ1paWoIUjHCTcBjSn6WkpPCwJnaHHW6XGxqNBjNnzsRNN90U0TqgjUZiYiJ39dVXs0ceeYRvbxzO2tdqtVi6dCluvPFGZjKZjmorhJ6bMgCF8jIQCMBsNh/bWVhaGYh2tVr9M6JoTw/i4+Mx2AxAVkFHZwd79tlnodfrIyoryjVfsGABRo0aFdH6oGdua2tj55xzDg4ePCj7nuRioiConAUh3bWS68JsNiM9PZ3vjS4HV5KcnIy4uLiYYTOEgbpwzM1xHGVicQkJCcjJyQlpadXW1mLXrl28oti7dy86Ozv5OElCQgKvTFatWoVVq1ahqKiILViwAFdeeSVSU1O5XwrCXW48aCf79jvvsCWLF2Pz5s0/u35MJl5pOBwOfrEnJydj6tSpOPHEEzFt2jSMLypCbk4OFwmtQJoKHW7sKbvLYrFwFosFaWlpEZV9V1cXD9BZV1eH2tpa1NbW8kWCHR0d6OnpCdrQ0L0IdkZagU7WEaFDhxL0H374IbRaLbvuuusiriPaMN1www149tlnQQk1ocaGoN4rKyvx5ptv4sYbbzxqrRAeYLKvjzU2NoqsL0rmSU1NPbYtkMTDQkNklmm06OnpQVVVFXJycgbd700ZOs8tfQ6NjY1RWR8+nw9GoxF33nlnVAqN4KnXr1+PgwcPIiUlhXcr0GIj37dc7IJA62gXl5mZidzcXL4SXRj0jFSJLowphGr8M5C2qFJlI5cMYLFYuPHjx2P8+PGYP38+AODgwYNs/fr1+Pzzz7Fx40a0tbUBAA/REggEsH//ftx11114+umn8ac//YktWrSIEzbe+qUWslqtxtdff80efPBBfPvtt+A4jndPeTweXhFmZ2fjtOnTMXvWLJx88smilrdCXoqUPhvLuwnjAqHa5NI86PV6ZGRkcBkZGSguLpZVMB0dHay5uVkUc6urq+OTMjo7O9Hb2wubzRZSuWk0GhgMhqCEDIPBgI6ODnz55Ze47rrrIrqyCL05Ly+Pu+iii9jLL78ccb2Ssl+8eDGuvvpqvn7saLNCyOLduXMn6uvrRfUwpLRHjhx5bCuQYcOGyXDgzwvg9ddfx/Tp0/mME6nQCIfbFGoRklupra2NLV26NCbr49JLL0VRUVFUsQ/aOY0aNQopKSno6uriF7HQ4khISOBdTJmZmSJkU/IpE8R8pF2rUJBIxyAaBRxtT/RQFeiR6gaEqcgajQaFhYVcYWEhrr/+etTV1bHPP/8cb7/9NjZt2oS+vj4+NsAYQ3t7O2677TaUl5ezf/3rX9wv3UyKIG5IcRCUuMvlgtFoxLnnnotLL70UM2fORFpammxVen9qc8LNRyjlH62yl1P0er0eWVlZXFZWFo477rig810uF7q7u/lGZc3NzWhububTwskla7PZeJgTqmdxuVy80pkzZ07Mm4A//OEPeP311yNu9ghkcO/evXj33XfZFVdcwUWyQmJFYQi3HgbDM0P8rVKp8Mwzz4RMNZ8yZcqxXYn+3XffsVNPPRUmkylIAHo8Hjz1z3/i1j/+kRvIPeT+98ADD7CHH344KuuDsq82bdqESZMmxZS6y3Ec6uvrWX19PZ+WSR3jrFYrLBYLF23hVygFEUsl+kAA+ITXFILm9bdWIFTNxJYtW9grr7yC999/H52dnXwLVwDo7e1FWVkZxo4d+4tiYZ1wwgls27ZtSE5O5uMHaenpmD9vHhYsWICSkhJR/xKhUI5mLENZDOHmNxwIZn87Bg607kj4bC6Xi5EC8Xg86O7uRllZGfLy8nDqqafGtAmgeZgzZw779NNPo2p763Q6MXHiRGzdupULVehJG8qhSo8++ii7//77RfUfNB56vR7btm07NhWIsO94SUkJq6qqEuV5k4nmcDhw0kkn4eyzz0ZBQQHi4uJ45SIMtOsNephNZhiNRiQmJiI/Px8Wi4WTMiHHcWhpaWETJkxAb29vWH+q0Po4//zz8d///jfmGoRoGFTaNjQWYRxOAMUKOUE+bI/Hw4Q1BSTgDwf0uWhbCwuVXiRhSG49YWZQVVUVe+mll7B8+XK0trYC+Blccc+ePcjJyeGiEfwDVSBkgTz11FPs7rvvBgBkZGTg+uuvx0033cR3spSmHUeap1jm57AAZvScGo0Ger2ei8WSCdU5MFaXWbgNSSywOP1p4qXRaLB69Wp2zjnnBAlUOaK+4e+9/z4uvOCCsFaIEIGBqv+Jf/n35H7u0yFUpuE6ogrcdxzBKnEyPOH3+xllNjqdTvT29mLPnj145ZVXsGrVqqBmUiST5s6diw8//JA7JhUI7dQ0Gg0ef+Jx9ud7/oz4+HjZ5kcUlIx2x28wGJCcnIybbroJf/nLXziacBJQf/nLX9hjjz0WtfXhdruxfv16nHjiif0qHJT2lI7F7RBuRxiNAHI4HIyyctra2tDa2orW1la+Cr2zsxM9PT3o7e3l00DdHjff7lNYOKfTaaHX/6/ndoKg4pkqz7Ozs6leQDZYLESmDSVshT0dAKCuro69/vrraG5uxrnnnotzzz33F4dz5zgOL730EmtsbMSNN96I7OzssE2uhFZaKBiTw645Runb1ESsubkZbW1tPEKA3W6H2+3mXWGUkXc42wmpqakiAM7U1FSkpKQgKSkJCQkJsFqtXDSWUCQ36ECg5qWumf5k09GYnnTSSWzbtm1BXgs5BWK32zF16lRs3LiRI3lCVu/GjRvZAw88wCfsuFwuEXyPbDLDYSUgbVcrd4iKQlUcVJyY3wOMISCIhRJ8EKXHUzxQ+o5qtRq9fb3YsGEDTjn5lGNXgdDkOJ1ONmXKFOzfvx8Wi0W294WwCjqaaxIjbNu2Dccffzwv+BsbG9mECRPgcDgiWh+UDXbOOefg888/P2IV0KGA68J1TSPq7e1lVBwmhN1uaGhAc1Mz2trb+DThcLDbquBdU0iBKGcxEVE71IyMDOTn52PMmDEoKirC+PHjMXLkyKCAfzgoE6kiiWX3eiRa2gq/QxXz0picsOOiNHZQVVXF9u7di127dmHv3r2oqqpCU1MTurq6ZFNqpfMibElMsZVQRKm4hPKcmpqK9PR0PsZGcba0tDQkJycjPj6ei9bleKSqz6PZcGq1Wrz++uvsqquuimoDSFbIZ599hlmzZnFkiWk0Gr5tNXUelUNfiDVmKP1fpLiidPyEyofWh3Reu7u7cd111+Hf//73z/1PjlUFQkypUqnwww8/sOnTp8Pr9UbM9Y5modNi2759O0aNGsV5vV5otVrcdddd7J///GdUzEcNjdasWYMZM2bEZH2EYq5Y3UyHc+9ZS0sL6urqUFNTwx9Uid7Z2Ym+vj5ZxSvsKiftaS5VDKGAGaWMLtx5CRcaCRiv18srcKFiycrKwtixYzFlyhSceOKJKC0tRUpKSlDAWTomwp1xtLvXI6FAaHylz0eLXKg0XC4Xdu3axTZt2oRNmzahrKwMdXV1sNvtIuFGUN3SvuWhDrn4l7QZmbRXfajOgVQrRJZkZmYmnw6em5uLnJwcUfW5ECU5nBUTLiYjp4xjcZ9xHAen08lKSkpQXV0dFt5EuAmcPn061q1bxwldqk1NTWzixIlwOBwRq9wHStF4GUL9TedrNBr09PTghBNOwOrVq2EymbhjFkxRLrC9evVqdvHFF8NmsyE+Pl429TQaIi1NcQsS/IcOHWIlJSVwu90RLRpivBkzZmDt2rVRWx+hds3hqK+vj1EfcaGSqK2tFfUTl6sFEQogYeEXxRQIEDFSYSbtekIpG6FQEl5TjqgYTYjOS+nKtNPmOA7Z2dk8XP8ZZ5yBUaNGiQLRA4E4PxIKRC5mQ+1bAaCrq4tt2LCBT0uuqKjgaykILYAq8UmwSxWtHB+Gmgsh5Hc4wSXtHijtfx6q6yUAUVMzsmAIQ0vgrkRSUhKsVis3kFYF0WwMyAp54okn2D333NOvjSCtA7VajQcffJD9/e9/R2JiIjwez5DqWijcpJF7beY5M7HyzZVISkr6H2T9sa5AhEpk586d7Oabb8aWLVsA/FxgFkswOBAIwG63IyUlBZu3bMbIESM5Yrpo22QKTd/PP/8c5557btSwJbQ4XS4Xenp6mMvlgt/vh9Pp/BmupLkJDfU/+7upcKulpYW3IqTChCrRhbvUaFreAoDBaITFbEZ8fDwSExORnJzM15UkJyfzPvK4uDi+Ep3gM+h+QqElhDAhyBJqKkRgjg0NDXzVc1dXl+h9CCJFrVbjcPtg3tcbHx+PKVOm4IILLsDvf/97ZGVlcULroz9xpyOhQGjsaYMQCATwzTffsLfffhurV69GbW0tb42YTCbePULQI0IsqISEBF4oZ2VlITMzE+np6eRSgsVqhclo5JWxFHaE/OVUvEhV6J2dnejo6EB7ezsf5+rq6uLjXHKN24Q90OVwuggSR26zQFYMNURLT0/nEQ/S09P5mIzFYuE3KBqtBga9ARaLhbdsonVRU4FucXExenp6IvZNl7qiab7J1T179my2bt06GI1G0YYglhjPYFojxGOUKAT8XO5w++234/bbb+dobogfFAUiUSJ+vx+vvvoqW758OXbt2hWyYEmOjEYjSktL8cwzz2Dy5Mm84K+qqmKlpaW87zoSw9ntdkybNg3ffvstF42ZTQJr06ZN7Mknn8S+ffvQ3d0Nj8fDu3WcTqdsQIzgTWjxCgV2qMUrdD+kpqby7gfCJqJ2tykpKT8LI0lG2pH2Vbe1tbFDhw7hwIEDKCsrw86dO3HgwAE0NjbyAsxsNkOv1/OJCpQskZ6ejjlz5uCaa67BSSedxGc5xeJnPxIKROgn7+rqYitXrsSKFSvw448/8mmVBDfj8XhE3TTT0tIwevRoTJw4ERMnTsTYsWORl5eH1NRUTq/XH9H58Hq96O3tZZ2dnTzmWn19PY/c3NDQgJaWFnR0dKC3t1eW16SbGGH1udBNFsrK1ev1PNgmWbuEFZaTk4Obb74Zc+fOjcrSJ8vvzjvvZE8//XS/kmGESL29vb3s7rvvxrvvvouurq4hIQtNJhMyMzNRUlICwoejdgjSokhFgcgsUqLq6mpWW1uLrq4uuN1uUU8K/lBxUKt+hqvOHZaLcWPHcUJ3klqtxo033cSWvfRSTNbHf//7X5x//vkRG0bRM9XX17PjjjsO7e3tMBgMot2c0ByVsyDkdoUEXkcKgirRpeio0QRAw0Fk9NcfLd2JRSqU6+joYLt378amTZuwfv16bN++na8+NxqNfK2H0+mE0+mEVqvFKaecgttvvx1zDjcEikXYD6YCoet5vV4sXbqULVmyBNXV1VRlD41GQ4KaV47FxcU47bTTcOqpp6K0tBSZmZlcNMHpaOckUgvcWGo3HA4H6+jo+BmGpL4etYfdqJSQ0draiq6uLvT19QXdS9jkSorGHC7GRgkUHo8HWVlZOHDgACwWS1QZdiqVChUVFay0tDSq+aPU1/POOw8ffPABJ2w4Rec2NDSw3bt3o6mpiU86oeNwmi/cHjc8Hi+8Hg+85BqWS/kVjI0kDR66w03ATEYjn+hA4KVxcXFITU3lY1Fms5mTbrCD5llRIPLCbiDAecKga3l5OZs0aVJUZicVIJWUlGDLli1cNAV29Kw7d+5kJSUlMBqNfE1FqHsYDAYeeC4lJSXIvywNYA60Ev2XhHEQpmwKXTbSd6ivr2fr16/HJ59+gvXfrEdzczM4joPVaoVWq+UFMmMMl112GZ5//nnekvols7DoWgcPHmRXX301Nm/ezONBcRwHu90Oj8cDg8GAKVOmYO7cuZg5c6aoT7ow5iAXCP8l5iNUi9polEx3dzePn0XZfkLrpb29Hd3d3Xz1eTRuHHKZOZ1O5OXlYc+ePTAajVy0UEEajQaXX345W7lyZcwFwccddxzf0XIw5M2R9MrQJjjUmCgKJIqgdCS/ozRLSKixr7nmGrZixYqYrI8333wTl112WdTtaokJ7733Xvbhhx/CYrHwcCSJiYlISkri4w90JCUlIT4+HiaTKSosq/5Uog/FjYFcU6ampib22Wef4e2338bGjRvhdrt5RcIYQ1dXFy677DK8/vrrXLTFmYOhQEiYtbS0sOnTp+PAgQMi4ES3243hw4fjkksuwbx581BaWsoJzxUmAwzleQqXSh7Ns9vtdr7eiGIvBG1C8b34+Hj4fD60tLSgoaEBbW1tfFzooYcewrXXXht1piMpkM2bN7NTTz0Ver0+qra6NpsN8+fPx1tvvcVJd/TRgFiG2rxEm/IbTQwl1rWtKJAjpLlVKhV2797NpkyZEpUZT610x44di23btvHBvWgXPn3P4/HwZn0sgnWoWBC/tEKhxU30/fffs3/961/4z3/+A4/HA4vFAr/fD4vFgqqqqqjdHINZib5mzRp21llnwWKxwOFwIBAIICcnB7feeituuOEG3j9NVkYssB9Hy1zJpXdHq2Dk6HACADMajZzRaIw5ME1zfOZZZ7F1a9fyfBLNed9//z2Kioq4oWp5xEIqKHREiOM4PProo3C5XFEJc5VKBZ/Ph9tvv50HWYy1+pa6omk0mqAMFgo0CmEShHnpFEgXNvs5miyM/swP+cyFnRCnTJnCvf7669zatWtx0UUXwWQyweVy4brrroPFYuFinZeBED3bySefzN288GYYDAaUlJTg0UcfxY8//oh77rmHS0xM5CiORWmzvyXlIbTsKdVbmvQhTCuW8jzxvfD3QCAAg8GApKQkjly+/XVT3/rHP0atfDQaDVwuF5566qnfzLpSLJAjYH2o1Wps376dnXDCCWG7mEmtj5EjR2L79u0wGAxR+drDmau/VcH/S7ktSenX19czm82GcePGRb1WBjuITp+3t7ezpKQkTtirfai7p4ayVROtCyjSpm3KlCls165dIsjzcOeoVCr8+OOPGDVqFDfUARUVC+RXokceeQQejycqE5Wsj1tvvRUmk2lAu9yhaDXIQZFQCma0hxTG5EhtfGinS/fNycnhxo0bxx3JSuFoBVVKSgqnUqn43uWxAlb2x2Uk7V0f7gjlahqqVs1A1wr13Vm4cCGiLfbVarWw2+345z//KerTrlggCvHWx5YtW9gpp5wSVXCN0H1zcnKwc+dOWCyWo6YNZjToqEdyhyyXHjzYAf7+wG4fyULCgVqX4eBiBnu+wkG+Szc7R6MVReu0t7eXTZgwAY2NjXwDqXDrnYpBd+zYgeHDhx/VVsgx3VDqSOxqAODhhx/muwhG8q9Sfv8f/vAHWK3WqDOvfmnrYSD9GShg6XA4QAc1RKJ+DUIId4pPECwJ9eSgPuaHD46QAiLt3OXqd45Gi64/kCfCTMJoADJpvpxOJ+PrEDweeA9XJgsz2YQ1GDqdjtoLE6IAFwsfhwJLHMoKhvqmx8XFcQsWLGAPPfQQjEZj2GxLxhi0Wi1sNhueeeYZLFmyBL+mdatYIEPM+li/fj07/fTTI8I9EwN6vV6kpaWhrKwMCQkJv4r1IbQkhDUskZSEx+Phe1w3NzejqamJhxKh6uKuri709PT8D679cIc4EkbRLlRSKFRxTXUsycnJyMjIQFZWFt+TnVruJicnyw4ktXU9UhbSkcbCimbHL1f7AvxcKNnS0sKoZSy1i21ububrKXp7e+FwOHgMJJ/PB3/ADxYQ9xQR8ggpElIgBLsfFxeHxMREpKSk8NDv6enpPPz74XTyiGCJwlTyoaRcqCgwFqRtwiMzGAwoKytDTk7OUWuFKBbIYFsf/+/hqHsgE2zJTTfdBMqmOZLWh5w1Id2VSneodrudtba2oq6uDrW1taiqqkJNTQ3q6urQ1NSE9vZ2vkteuPcUZnbRjjWaxS91kxH+UltbW0gwv8MZNsjOzmYFBQUYP348iouLCcIjqCkVKZSjMf1ViNclnbv29nZWXl6O3bt3Y9euXThw4ABqa2vR0tKC3t5eWQUuzHgSKli1Sg1OzYWcH6/XC7fbzV9XmhouJZVKBaPRiLi4OCQlJ7H0tHS+ApoQD7Kzs5GRkYGkpCRO7v2GgnKhXj/Z2dncJZdcwl544YWINV+MMR50dcmSJXjiiSeOWitEsUAG0fqIpWMZmb8JCQnYtWsXUlJSBsX66I/Lye/3o62tjdXX16OqqgoHDx5ERUUFqqurUV9fj7a2Nr4qW6oYhGm/wuuH60kwGGBwoXpBCJGApdX4VqsVw4YNw4QJEzB16lRMnToVRUVFIqyuwailONIWCAlnKfheXV0d++GHH7Bp0yb88MMPOHDgAFpbW0VuVLLk6FwhYOVA5itSr3o5GH+/oKGR3HoxGAxISEhAWloasrOzkZ+fjxEjRmD48OHIy8tDVlYWUlNTw1ovct0QB7sIlqyQPXv2sMmTJ0fFN2SFmM1m7Nq1CxkZGUelFaJYIINkfTDG8PDDD0fNmGR9XHfddUhNTY26CnYgfaP7+vpYc3MzamtrUVFRgfLyclRUVKCmpuZnBNvOTnglOydyTZjN5iAFIZetE06w9KcBUChBFklBk5UjxEXy+XzYt28f9uzZg7feegsajQZ5eXmYPHkyO/3003Haaadh1KhRnBRUcihYJjTGZB2Q8tm+fTv7cvVqrF2zBjt37kRHRwd/DsWPiK+k8xVJEUTb2zxcU6NIVgi1BZDegxRMZ2cnWltbUVZWFjS/8fHxSEtLY7m5ucjPz8fIkSMxYsQIXrmkpKSEbYE8WHEX6udRVFTEnXvuueyDDz6I2DedrJCOjg48++yzeOSRR45KK0SxQAbJ+vj444/Z3LlzIzKO0PowGo0oKytDVlaWyPoYSIaMy+VCW1sba2hoQE1NDSoqKlBZWYmqqiqRNSFdAFI4baHQCrULDbXgIqWBxspzch3TpB3cQllfoZ6XdoBut5t3N1itVkycOBHnnHMOZs2aJYIGidR3/EhZIDRmwl12WVkZ++CDD/Dpp5+irKyMt7SEPT+kAXSpUuDhuGX6fPRnvqRzJNf4S84KkeP3cHMmfFYqEJQKXo1Wg4T4n4FAc3JyMGzYMAwfPhz5+fkYNmwY3w3RarVykdZ2qF7u0vGkFtlff/01O/PMM2XbwYaSA/Hx8di1axdSU1OPOitEUSCDYL4yxnDCCSew7du3w2QyRVQghIvz4IMP4qGHHorJjqY+40Jo7NraWtTW1vIB0ba2NthstqDnEKKWhupAF+0CFrofQr0vNTIitFs6CP1T2hFP2gfC4/HwjaCcTiefvSUM8Ia6L1UrCzsWhkIDpnsT7D3wMwT4pEmTcP755+O8885DQUEBP09erzdsDcZgKBBpQWNPTw/78MMP8cYbb+C7776D0+kEx3EwmUwiUD45tFqpFRayS6BGDaPByGe8GY1GGIxGGA7DoZO7Uihg6XqEGCtEkaUsu3DxgEhu0P7wphBtWsqbBJiZlJSEzMxMXrmMGDEC+fn5yMnJQVpaGhITE7n+CvKZM2eyr776Kip4E5IFDz30EB588EGOFJGiQI4h6+Odd95h8+fPj8r6EPo/ly5diqlTp8JkMvGQ3H19fejs7OSbJDU1NaG1tRVtbW1ob29HR0cH32fc7XbLLkipUI5kTYRTEiQgpOdQ//Hk5GSkZ6QjK/N/jYkyMjKEGTawWCykODhhc6JYyOv1wuV2MafDib6+PvT09PDuDWGPCWqU1d7eLmrhSs9M95cTuMJdtM/n47swxsfH4/TTT8eVV16Jc889lzMYDPyuU65z4UAUiLRpVHV1NXvllVfw5sqVqK6qAgCeX+TchkKFSJ0YhTxpNpuRnp6OnJwc5OXnIz8vTwTPn5iYyDf40uv1nE6ni2q+GGMEF8JI4dvtdlGjqfb2drS2tqKlpQUtLS1obW1Fe3s733BKLhGDrGMhVPtAlIuUr6VEbQzS0tKQkZHB92/PyMhAWno6kgU8TRafsJGbx+PBy6+8jCcefwJmszlqb0RycjLKysqQlJTUbxQKRYEcRUTM6PP5MGXKFLZnz56IPZLl3E3kpxa6VOQUg3DHQotJuqDkYiORFpXQFSBldspmCrVTS09PR2JiYkxNieQQjiP1CIk16NnT08MaGxtRXV2Nffv2YdfuXdi7Zy+qqqqCYgTUVEpOGNOOWGiZFBUV4aqrrsIVV1zB99iQKpL+KBASbOSqOnDgAPvXv/6Ft956C52dnbw1RxsXueek7oMkGPV6PfLy8lBUVITS0lJMmDABo0aNQnZ2dkT3jTTWFI6n5NyZ0ZLT6UR3dzffVpng2oUpxm1tbejp6QlaF0JY9liVi9wzEw+E2jRJ42tanQ4aQadOoUs0FpBEskIeffRR/OUvfzmqrBBFgfSTaJJfffVVtmDBgqjg2uUYWVp7IfUby5nz4WIScguD33V5ffB4xW4fTqVCQnw80tPTMWzYMIwcORKFhYUoKChAXn4esjJ/rqcIJxiEDW2kgj+Uz7g/Clv6u1ysKFT9A2Ps56Y9e3Zj65at2Lx5M3bu3Inm5mbecqOOftL0YKFycDgc8Pv9yMjIwOWXX46FCxdi5MiRvCIhYRKLAhFaHLW1teypp57CihUrYLPZeFef3DPR+U6nk1cq+fn5mDp1KqZPn46pU6di9OjRspD90jkLleAQK6Cn3N/hXIeRBK3H40FnZydrbm5GXV0dqqurUVVVherqatTV1fHti6XWSzTKJZrYXqg1KOcS7a8SJfdpeno6ysrKEB8ff9RYIYoCGYD14Xa7cdxxx7GKioqYrY9QO+5wwcRQglkYAA1lmhuNRiQnJyMrKwv5+fkoKChAYWEhRo4ciWHDhiEtLY2jrnxywkZOgIVK0TziTBtl32ihcJRTLM3Nzez777/H6tWr8fXXX2Pfvn18pXCk3T5ZiomJibjiiitw++23Y8SIERwJPZ1OF1GBkKvsMD4Se+aZZ7B48WK0tbXBaDRCq9UGjbv0/gAwZswYnD1zJmbPmoUpU6YgISGBk7r/pNX4kYT/YM5nLLU+9FNajxLKOuro6GBNTU04dOgQqqqqUFlZGaRc5Cx6qXKRS2CJZrM2WGNGVsiTTz6Ju+6666ixQhQFMgDr4/nnn2e33HJLv6yPcIpBKgSlAISh5sxisSAxMfF/Pu68PD5vftiwYcjMzERSUpIsxARdXy4t+GhFe2WMIcAY2OHdoiCriKnVapHrzel0YsuWLXx2U3V1NR8zIKtEOm9qtRoejwculwuJiYlYuHAh/vKXv/Cw7/X19ay0tFRWgWzbto1PMf3444/Zfffdhz179vBwINJ5pp06tXVNT0/HrFmzMG/ePJxyyim8lUEbmwBjTMVxnFwm1FCH6ZciI0h3/cLxl9bCCC2Xjo4O1tzcjEOHDqGmpgZV1dWoralBfX29qE2uHAndxFKPQKh1Gk3zuXBywOPxIDs7G2VlZVF3v1QUyFEolA4vZFZSUoK6ujro9fqYrA/hd4W9KOSI8uSpfzG1oR02LBfp6Rl8D+PRo0cjMzMTKSkp3GA0qaEMFp/PxyijRdhH/X9/e+H1ij/3+/3w+X3w+/6HpPs/l0kAgQAL6dYQpYOqVVCrDi9kjRoatSaoJwTf6/lwRpf+cMbQz/ENHafRaKOaD6F10t7ezj7/4gusePVVrF+/ni/4ojiDnCuJFElxcTGeeuopnHXWWVxzczMbO3YsXC6XSIGMHz8eZWVlXEtLC7vnnnvw2muvQa1Sw2wxh3SfUep1SUkJrrrqKlx00UXIzc3lhII22nn3eDzwer3s8E8erkTaMyYSIrLQWhA+K6dSQS1IuaZD2G9GOH/S/2k0mrCV54NBfX19rLW1FY2Njairqwtqk9vW1iZqkxttKrMwFT5WRU1WyOLFi7Fo0aKjwgpRFEg/rY+nnnqK3X333f2yPqhvAKVhJiUlISsr63B+eiqSkpKRlJSEhIQEJCUlwWq1IjU1FWazCVqtDgaDnmMMPOAdpVE6nU4epPDnlFcHHA4nnC4nXE4XXC4nnE4X/z36rsvlhMvlhsvlhsfjPgxw6IbH4w3ZkEpOqAQEO/3BhvSWKhehi0OtVkGt1oiA/Sh9mBRvQkI8EhN/bu1LeEwZGRlIT09HSkoyzGaLbNHZ+vXr2eLFi/Hxxx/z+EVarTakIiEL4bbbbsM111yDM888E729vfxO2W6347jjjsPf/vY33HbbbaitrYXVapUt7qOe3V6vFyeddBIWLVqECy64QNaCdDgcjLKcmpub0dzczGfvdXZ2oru7GzabDX12O5yH06CFuGTSRmNBO38wgMVWlS43X8JDqlikG4L/bQT+B9LIH0YDn3JMB6UeC1PF6VzhNQTfDVvF7nK5YLPZWG9vL/r6+tDX1we73Y7e3l709vaC2uhSFmBTUxPvMvP7/TxIaLRErslhw4Zh586dvFU5lK0QRYH0w/ro7u5mEyZMQGtra1QNo4QLy+/3Y9KkSSguLkZ8fBySkpJhtVp55unt7YXNZoMQudblcsFut4t+t9lsvHAJBPzw+wdWrBdOYIdyf4SrVO5P7+ZofMmh/NOhKuPDjYNKpYLZbEZCQgJv1RUXT8CYMWN+jg8VFCD5cFrlrl272FNPPYUPPviAD25TjEJ6TcYY7HY7UlNTeSEt9LHr9Xr02fvAAgxmszloA6JWq+Hz+eB0OpGbm4unnnoKF198MUdCrbKyklVUVGD//v3Yv38/qqur0djYiPb2dvT19cnGwOTiONLi1HBumn4JMcbAIsydXEHhYG5ApIqKLB5SVEJlRUCdpGAI/TmUYqLaGMrU6+vrg81mQ1VVFX748QfUVNeExYgLZ4U899xzWLhw4ZC3QhQFEgNR3cfDDz/MHnjggX7HPrxeL7+I/X5/2JTBoAkDB07FhazGDifgYxXS4b4Ty+dHlIFDvFs4SA6hsKKaAAJUJNLpdEhPT8eIESMwceJEXHjhhTjppJO4Q4cOsRdeeAHLli1DV1cXTCZTSB869aeX85kLCydlrAlYLBbcdtttWLhwITweD95//31s3LgRBw4cQH19fZDvXq4dcTSK99eYy3D8GM3Goz9BebnjSDTBEgblQ8VnIik8l8uFESNGYMeOHXzN0VC1QhQFEkPcguM4tLW1seLiYvT09ECj0fQ7YCb0H0fjK41GqCtz2X9hJpwHEi4UIyAqKirC7NmzMWvWLKjVavzr2X/how8/4nfycjGdWOeEMYY5c+bg0ksvRWdnJ1auXIlvv/2Wr7pXq9X8zjfaugeF+q+cYsVsG2gwXWiFLFu2DNdff/2QtkIUBRKj9XHvvfeyf/zjHwPKvFLo6BE4QkvB4XDwFsPxxx+PWbNmoaOjAytWrMBA2hDTvdxuNwoLC3Haaafhq6++QmVlJYD/VZ4riuLYILJCRo0ahe3bt3PRtj5QFMgQJbIUmpubWXFxMR8UVcbu2FvYlIlFLqSUlBS4XK4BKxChIrHb7VCr1TCZTGGRcxX67RJZIa+88goWLFgwZK0QlTJVkYkW8O7du9He3h6x77FCv10+oJRri8UCq9WK3t7eQVMetFkhnCVhoySFjs1N66effjpkrQ9A6QcSE1mtVn4nGgnmPBJzHAs0GEw/VMeK5v9I1CrEqjSO1kLPo3n+f4m1w3Ecj4gwZC0lRS1EJsIcKi0t5U477TS2fv166PX6oApVYRMdUaEVOICLLt01GmFA14tuBQLiZMrBWbjRfD5Yi3+wgp+hMpNCfX8oxRzCQZcLU4kH6zkHoztmVLw8CN8ZrPsMJcVMlu7s2bOHtCJVYiAxLqiWlhZ27733Ys2aNbDZbPB4PCIwRGHlLaX0UbpouII7OWH1S81NtF3ooqkBiaomhONk9V+4znaRUjKPpKCnegGKew1GnU00RDEXgkoX8prw2agnCD/WgvEdzLqOSJl/Uc3fz38Ifh4uUJT+P0Rr3XAtd6NpyztYrZWPNMXFxeHqq6/GP//5Ty4cHpiiQI5CJQL8DBne0dEBp9PJZ2NRsRKlWVKRkRDHSq6SW/ozFiUTaSctVyMS9lBxUHEq2crv/hwhhdhhIRdKKMkpjIAE00qKEeYPBOD/HwTLz/AcPh98hyvp/QE/wIIrpEnBe71eOBwO2Gw2tLe3o6mpiYe5aGpq4kH5qKo5VBvfwbB4GWOirK+EhATk5eWhoKAABQUFGD58OHJzc5GRkYHk5OSgHuexFnnKKZdYaoQiCfRoiwjD/e9IHDxWGgT/D8TwLAKFF+l9hQpU4CAQ/80Y4uLiUFRUxEPVDGlXm6JAYlciseAOKXT0U29vL6usrMT333+Pr776Chs2bEBrayuAn8EW1Wp1WJDLaK1AqkCnRlZFRUU4/fTTMWPGDJSUlGDYsGHc0dTuVKGBkRSjTVEgvzFFEmnsYoE5H8rzMNSDtEdqbEP1oW9sbGSffvop3nzzTWzatAk+n6/fVokUjNFsNuN3v/sdrr32Wpx22mlBzbqkiMmD0Wvl15yb39p9B+NZhBayYoEopNBvZLNAO0Lhwv7uu+/Y8uXL8dFHH6G9vV1klYSKywgr18kFmpiYiPnz5+OWW25BUVERrwmo2+HRAMOu0LFHigJRSKF+KBRCJiCBXl9fz95991385z//wU8//cTHSyguRkpHin02cuRIzJ8/H9deey3fkIqyqo7mXiwKKQpEIYUUikAU2BfGxLZv387Wrl2L7777DgcOHEB7ezvcbjcP35+VlYVJkybh3HPPxZlnnsn3J6eCRCXOoZCiQBRS6Bi0SqRwE9TT2263Q6VSwWq1IiUlRWRWKIpDIUWBKKSQQrxVIhcvkSoNQHFTKaQoEIUUUiiMZSL8qQTCFVIUiEIKKaSQQsc8KU5XhRRSSCGFFAWikEIKKaSQokAUUkghhRRSFIhCCimkkEKKAlFIIYUUUkghRYEopJBCCimkKBCFFFJIIYUUBaKQQgoppJCiQBRSSCGFFPoNk0YZAgy4l/ZgNPUJ1YRoKEFfhBqnowWeI9SzK6RQf+XDsQ5No0CZHAFmE6KrKgLq1yUCNpSi5CqkkEKKAhkUAdPS0sJiaUMqtwvR6/WwWCyyLUgj9U/3+XxoaWlh0msyxmAymZCYmDgktFBLSwvz+XxB/9dqtUhLS+OG4twK0XB7enoY9RvX6/UwmUycXq9XlPwxbkVEgtF3OBysq6uLX5PC9ZmYmAiTyXTMMtAxq0AYY+A4Dj09PWz8+PHo7OwUMUh/FEhcXByys7MxYcIEnHHGGZg5cyasVitH9wol4KqqqtjEiRN5y4UxBo1GA5/Ph8svvxzLli3jfD7fr7KLpmf3+/0oKSlhlZWVUKlU/LMHAgGMHz8e27Zt44bi/H7//ffs9ddfx6ZNm1BXVweHwwHGGMxmMwwGA+644w7ccccdXDSKXqFji2jNvfHGG+zGG2/k1yQA/vd///vfuOyyy3619flr0zFv1zPG4HA44HQ6B3Qdh8OBrq4u1NbWYtOmTXjhhReQn5+Pe+65hy1cuJAT7lzknqGvrw9yFojL5RoyYxVqnAY6dkdip8lxHG6//Xa2ePHikO8CAE1NTfx5Cv12yWazse7ubtnPUlNTOaPRGFaROJ1OWQtEziI/lkjJwgL43tYUs+jvoVKp+B7YarUaNTU1uOWWW3DttdcysjhCCSqNRiN6Bvp7KO2KaZzkfg41t9VNN93EFi9eLOpJLpwnnU7H/1Tot21FAMDy5ctRWFiIsWPHorCwEIWFhRgzZgwKCwvxzTffMOB/Tb7kPAzCNSn9/VgmJbIIsX9UKuCjCYQzxmSVg0qlgkajwfLly1FQUMDuu+8+WVeJUAjTzkatVoMxNqTanKrVaqjVan5MhD+HAtHYfvzxx2zZsmXQarXw+XxBzZwCgQA8Ho9ieRxjisTj8cDn8/EZj7TWQikO4Tom3id+GYrrU7FAhiAFAgH4/f6wRyAQkBWkgUAAXq8XarUajzzyCBobG5larQ5K2fX5fHC73fD5fPB6vfD5fHC5XPD5fOjp6RkyY9He3i5aiPSzs7NzSDwfKfonnniCVxRCBUGK3mQyITk5GQaDQQmgHyNEmweymqW/hyOn0ylak8Lfh5L7VrFAhhjDMcYwevRoDB8+HHKBcIpd1NfXo7a2lv+OVGhxHAeHw4GPPvoICxcu5N0sdL3U1FT8/e9/5xUR7WwCgQAmTpzI74J+TaGsUqnwwAMPiJIN6Gd6evqQcV01NjayH3/8kVcWwnfQ6/V47rnnMHPmTGi1WrjdbhgMBn5HqdCx4WkQehvCWaC05qZOnYqHHnqIX5P0WSAQwJQpU37V9akokCFKarUaPp8Pt9xyCxYtWsRF2qFs2rSJ3XXXXdi5c6eI0YSCeOvWrVi4cGGQcE5MTOT+9re/RTSjY10s0h24NGYTqyK59dZbuSO5uOWeN9paGjqvtrY2KOBJc3neeedhwYIFXDhFGeszxvqcgyEAhYqxP4pPep1Y3mGg8xSLkJcb5/7yb7/cM4fX3MSJEznayA3G+gz1joP5ftGM42AoPUWBRCCXywW/3w+5ND2aCKPRiDPOOIP76quv2MSJE9Hc3CwSYDSJlPEjnTjGGO+Tl1Nk0aYHkj83muA2+X1jEUAejydkNbdcMDqa6noSRuRjDmVd0HfCCUOy9IQWpJBGjx4Nr9cLv9/PjymNVSSBKUySCGcFkfUYrQCIZozoGYTJHqHGIZxwi+Y6wiJYuXEIN0/94SnhtaOZD+m9Qo21kC/CJa/QHNBB15LyqNfrlT1Xq9VGLYiFRa2R3tHn80GlUsUs5IVzEM049vc+igKJYedNAbNQC4MUQGpqKnfJJZewxYsXi3LGhQI41D2kBYixEikOEozNzc2ssrISjY2N6O3thVqtRmJiInJzc1FQUACr1cpJ3T+RKNaMpUjXFCYUdHd3s927d6O2thZutxtmsxlZWVkoLCxERkYGJ/1+qN2U2WwOeT+DwQCtVgutVhvTmNI9Ozo6WGVlJerr6/nYVHx8PHJycjBy5EgkJydzcu82kDGiuSEe3LlzJysvL4fdbsfo0aNx4oknclKBF4pHiZcDgQB2797NKioq0NPTA71ej5ycHIwbNw4pKSlBYy0ch+7ublZWVoa6ujp4PB5YrVbk5ORg1KhRSEpK4mJ5d7kx7uvrY4cOHUJdXR3a29vhdDqhVqthsViQnJyMrKws5Obmwmw2h+VfGg/i2XDry2QyhRWiKpVqQOtTqHxVKhX8fj8qKytZdXU12tra4PF4YDQakZ6ejuHDh2P48OEcrWO/3x+1ZSiUUR6PB7W1tay+vh5tbW2w2+0AAKvVioyMDOTl5SE3N1d0n/5Ys4oCGUQlEwgEkJubG/J7CQkJIncLLeqmpiZ27bXXimIgarUafr8fZ555Jv70pz+FLHQTMk5PTw9788038Z///Ac//fQTbDab7HNkZmbi5JNPZldeeSXmzJnDEVOHuj4FpK+77jrW2NgYFAMZPnw4XnjhBU74bhzH4Y9//CM7ePAg79IjV9KcOXPwhz/8gVOr1di9ezd75pln8Nlnn6G5uTno/vHx8Tj11FPZHXfcgRkzZnDCAkaVSoUffviB3XffffyzUEBfuOOkXdm///1vrFu3jgkFzrJly5CXlxdU7Enj4fF48N5777GVK1di69ataG9vlx3T5ORkTJkyhV166aW45JJLOL1eH3FMfT4frr76atbW1hYU+1qwYAHmz5/P0d8vvvgie+GFF1BWVsZfJy8vD5WVlVCr1di6dSt74IEH+PPpejqdDsuWLUNGRgbn9/vxwgsvsBdffBG7du2SfYezzjqL/elPf8Jxxx3HCXezbW1t7NFHH8Xbb78tO08pKSk47bTT2KJFi3Dqqady0WxKaHxcLhfef/999t5772Hbtm1obGwMiz2VnZ2NqVOnsksvvRQXXHABJ+QHuuYrr7zC3n77bX4camtrRbwgtP7uvvtupKamMhp/xhiefvppjB8/ngOAr776ij355JP8mqQx8fv9uOeee3DGGWeEXJ/CDcD+/fvZyy+/jM8++wwHDx6UrSHR6/UYPXo0O/fcc3HVVVdh3LhxXKRNnjCmum7dOvb6669jw4YNOHToUMg6FaPRiFGjRrFZs2ZhwYIFKCwsDFurFrWv7Fg5yKzt6upiycnJDADjOI4BYACYRqNhANjjjz/OGGPwer1hr+fxeBAIBHD//feLzhf+/re//U10Lb/fD8YYysvL+e9Kj4svvjjk/YXm+SuvvMKGDRsmOpfjOKZWq5lGo2EajYap1eqg65900knsu+++Y1QUFWqcfD4fMjIyZJ9x+PDhTO6ZxowZI/v9q666ih1epEyv18s+r1qtZiqVSnTefffdxz8njcenn34acuyiOXbt2sWEcyEch88++4xNmDAh4pgK+QYAGzduHPvggw/469J4SMfU7XYjKSlJ9rnoXaurq9kpp5wi+kyr1TKNRsPGjRvHaBw++OCDkO/Y1tbGmpqagq4Taqy1Wi174403+Dldv349y8vL4z9XqVSic6XvL5ynUOuFPvvoo4/Y2LFjg55ZpVLxYxyKHwCwE088ke3evZsfaxqPO++8c0B8sX79ev79ly1bFvJ7L7/8csj1STzlcDhw9913M4PBEPId5d5Pr9ezRYsWsd7e3iAeld6jsbGRzZ07t1/jaDKZ2P/93/+F5Ndwh6JAIiiQxx57jHm9XjidTni9XtmDYgOMMUyaNIlfnNJr/fTTTyJGoJ8VFRVMp9OJFqVer2dqtZpdc801sgxK6cU+nw8LFiwQ3YsWNQk7OlQqVdD/6VmfffbZkPehBT969GimVquZVqsV/SwpKZFVIFOmTGFqtZp/N3qnv/3tb+yNN97gx5yuQ88nJ7DpWZ944glGwpcxhtWrVzOdTseMRiPT6XQixS096FmEx969e0VzQoLtwQcfFJ0nHFMSoOHGFAC75557eEEqXJRCBTJ8+HDRWNIYPfHEE8xms7Hhw4fzQp0WPv0cM2YMEypS4VgTHyQmJrIdO3aw8ePH89cJN9Y0fiqVipWXl7NNmzbxSj6ac+n9H3744ZBKhP73+OOPy/KtdBzl5oDeEQBLSUnhlYjL5QJjDPfffz/T6XTMZDLxYxKKLzQaDc8PBoOB6XQ6flPFGMOKFStEcyP8/fXXX5ddN8RP9fX1bMqUKaJ7yfGMlL+EfFxSUsJqa2uDlAgJ+8bGRlZYWCjaFKhUKv46wrmiz+Xuc+WVV8ryq6JABqBA/vWvf7Forudyufhdj1DD63Q6BoBdffXVQQtKqECIwekZ6P60WxcyaCAQ4K9z8cUXBwkYqQKTLnTp9+jvJUuWBD2jUIEUFBSI3o9+FhcXyyoQqTKl70+ePJlZrVZecMoJeqmAIqGt1+tZTU0No3sMpgVC733vvfcGjY10XuV2etJdHwC2aNGikGPqdrtBViOdT+c9/fTT7KqrruLnVjh3JCDGjh3LK5BPPvlENNY0fnFxcbxw0el0QeMqxyf0v+OOO45lZWXxzyB3bihFolKp2J49e0Jad2+99ZZIMYQaY+nf0vvR2EyaNIkJN3ODaYG8+uqrIb0Kr732muz69Pv96Ojo4K0r6fiFWp/C/9PmCgAbO3Ys6+7uZkILgX4/++yzRbIm1oPjOP7cRx99NKL1KDyUGEiE7Jj169fDaDQyOR8kua4qKirw+eef48CBA0EpvB6PB6eeeiqWLl0adbA6mmdTq9V4+OGH2bvvvgudTicK0JMvWKPRoLS0FNnZ2fB6vaioqMCBAwdEvnbKutFoNFi0aBFKSkrYKaecckTABWlctm3bJvKD63Q6pKenQ6VSobW1VbY4i7JX3G43VqxYgQceeACMMaSmpuLkk0/m36enpyfIv09+8GHDhmHYsGGimh6LxcI/h1arxX/+8x/22GOPBVWxC+d1woQJyM/PBwDU1NSgrKxMBC5JC1yr1WLJkiUoKSlhCxYsiGpMyce+cuVKbN++XeTXJ0FMzxEOh4me22azwWazQaPR8DySkZEBnU6H1tZWWaw14ont27fz/n7KQkpPT4fRaERXV5dskSvdNxAIYOnSpSK+J77r6upiixYt4udFmJYcCAQwbdo0LFy4EBMmTIDJZILD4UBlZSVeeuklrF69WpRhR4W6P/74I7Zs2cJOPvlkjuq3hHzR0NCA6upqWTyrcePGISkpCcIYSGJiIkKBoEa7Pq+55hrs27cPWq1WlMVFc5qcnIyJEyciLi4OnZ2d2LVrF4Sov6SYtFot9u3bh9tuuw2vvvqqKA6zYcMGtnr1atH80vmpqan485//jBNOOAHx8fHwer2orKzERx99hDfeeEM0ZzSO/+///T9cffXVLCsrK6o4lmKBhLBA+nPQbol2iklJSeyuu+5ifX19TChYBmKB0K5j3759sj5ous55553H7wCFzPjFF1+wUaNGBe3u6Lxx48Yxt9vN32cwLRDpztVisbAnn3ySVVZWMgJqPHToEHvsscd4i0r6bhzHsVNOOSWkT3jjxo1B70Zj+cgjj7BQvBAIBNDZ2cnS09N58186r1OnTmWbNm0KusZhwSVriahUKpaQkMCam5uZMF00lAUSzrIxmUxs5MiRrLi4mI0YMYJNnjyZ0Y5baoFIXTQA2KxZs9jGjRsZwdrX1NSwv/71ryHvT1YfADZlyhT2zTffsJ6eHuZ0OtHS0sKWL1/OEhISeNeL8DyO49ioUaOYcGdOvy9fvpzfMdM9yM101llnsXC735kzZ8q6iDmOY08//XTIeMSSJUuCrAi6xurVq2X5gq4TiwVCz/6f//xHZCFJ73nXXXex1tZW0X2bm5vZn//8Z9n5oPN27NghctXdc889ItejcN7Wrl0b0nuycuVK3nVH8RGDwcDUajV7/vnno4r7Ki6sKBSINAgV6pAKDo7j2FlnncUOHjzIpApjIAqEGJTcG3IL4rLLLmNSd5dwUba0tLBRo0YFubPoWm+++SZ/z8FWIHRPnU7Hvvnmm5AM/thjj8meC4BlZmYyu93OhO4Cj8cDv9+PtWvXhlQgDz30EPP7/SAFSeNPY0v3lBvTE044gREUvHBMhcrg9NNPDxn/uv/++4PGNJwCIf84JSm89NJLrLa2llcYXq8XXV1dTJpMIKeshckYcsfNN98c9N5CPh4zZgyz2Wyy53/88cdBz0/zZDAYWH19fVCA+8ILLwy5CduyZQsvIIVwQS6XC4FAgE8WkBvjhx56SDTGwrl+8sknQ87tJ598woQ8JOWLWBQI8UZxcbFoDoX3+/vf/y6SCT6fTyQbyP1GmyhaLyqViv3pT38SKZALLrhA9Dw09haLhdntduZyuXiYJJoDela55AUA7He/+13UCkRxYcWQrRbKnJWmHJJJ/tVXX6GwsBBXXXUVe+qpp5CSksINxI1F6bptbW3sgw8+4Pt0CF0sWVlZePHFF3nmJGRgoUstLS2NW7ZsGZs+fbrsOy5btgyXXXbZEYFnoFTeK664Aqeddhrndruh1WpFRVsAcMstt+Dxxx+X7dPS0dGB1tZW5Ofni9wOkfLlqV5EmvNPz7R8+XLejSJ0qRgMBrz66qswGo28S0E4pl6vFzqdDq+++irGjRsHu90ucs9wHIfXXnsNf/3rX6OuJ6C5PfHEE/HRRx/x9Rl8/r1Gg4SEBC4cECA9f0JCApYuXco/qzD3n+M43HzzzXjxxReDQAXp/AcffBBWq5Vzu918XQW925w5c7iioiK2e/dungdprlwuFzo6OpCdnS2qeUlLS8O0adNEKceBQABxcXEYP348n34sLKSk76SlpcmuuVCFmDTX0fDFQMERydW4adMmtmvXLt5VJUz7nTZtGv72t79xcgV89N0HHngAL730kqjFA7mn1q5dKyrWpMJZYWmASqVCX18fVqxYIUK+oOuTbPjnP/+J/fv3B6V+UylCNC5sRYFEWT0a7aKXCiAAeO211/D9999j9erVLDc3t99KhBTC119/zRcHChWIz+fDtddeC4vFErLBjU6ng9/vx6mnnspNnTqVbdmyhb8OLf6tW7eisbGR94MOJmwEjc28efP4uIZUmAcCAVitVq60tJStXbuWX4i0SDweT8gal/48j0qlwu7du9mBAwdE80f3nT17NkaNGsX5fD7ZIkSKl+Tm5nIXXHABe+211/hCUrr+oUOH8OOPP7KTTjopYvMhWsjJycl4//33kZKSwgkFf7SNz0gxnnXWWUhNTQ16fqpWHjZsGBISEkT+d1JgJpMJ06dPDxLqQrDKoqIiCBWI8Bmlvn8AeO6557hwPE48LQQ8JMW1Zs2a2GsVfsFNJgB88sknvFKSyo67775bFFeTzlcgEEB8fDx38cUXs/Xr1/P/o/E0m818vAIAEhMTgxQkzd8tt9yCn376iV133XUoLS3lpIXA55xzDnfOOeeE5UNFgQygOJAxhuLiYowaNQrhBGlfXx+qqqpQUVEhClATo+h0Ouzfvx8XXXQRNm7cKIKF7g9t2bIlqPKYGPXss8+OGPwj8/Occ87hryW0cJxOJ3bs2IGsrKxBC/wLhY5Op8OYMWNkFxi9C8dxfKBa+C40L4MFx07v9/3334uErvC+s2bNiuo+jDHMmjULr732muj79J5bt27FSSedFPFaQhy2jIwMjqyeWBc3UVFRUdh76vV6PjAuLRLNzMxESkoKF0ppcRyHxMTEfglaYQBdCC0jpc7OTrZ371688cYbWLZsmcjyHkpE62TLli0iaBl63vj4eMyYMSMszBBZQi+//DIn7FAq5Q+qUD/rrLPw1ltvia4n/P6yZcuwbNkyFBYWsuOPPx6TJ0/G8ccfj3HjxomQE8jtF2uPH0WBRFjEN9xwQ1Qggh6PBxs3bmT33nsvtm3bJhKOHo8HWq0W33//PZYvX85uvPFG3oSNVQADwMGDB4MYNBAIwGg0YuTIkVEBpXEch/HjxwcxnPAegyGg5SgxMZEXOqEEIcdxiIuL+8Xmu7y8PGRG1JgxYyJChdAucPTo0SKFHukeoXbhHMfh/PPPH5SeExaLJeyzq9Vq7rAPPYji4uKg0WjCbkpihfkgfqXsP3J3VVZWsv3796O8vBxVVVVoaGhAU1MT6uvreQSA/rad/iWsD5VKBZfLherqatHaIUu2sLAQSUlJXKQNHn0Wzkoly+TSSy/lFi9ezHbu3AmdTsfHLaTfO3jwIA4ePIi33noLACgDjJ1++umYNWsWSktLY4ahURRIFORwOEKCKQonXKfT4fTTT+e+/vprNnXqVOzdu1ekRGjBPPfcc7j++uv7ZYWQIAkFp5GYmMjDpUSjiDIyMkIqidbW1iNm1en1euj1ei7a5/wlSPq+QkgZ8rtH8zwpKSkwGAxwuVxBwq6trS3ideiclJSUqDcDgzEvof4XjTCJdZ6EVu3HH3/M3nnnHWzevBk1NTUh1wSBFrrd7iEtL7q7u1lXV5fsuqJYEO30o7XUQrnKGWMwGAz48MMPMXfuXB7mhqwIoftdGG85XKOCdevWYd26dbj//vsxY8YM9te//hVnnHFGTC52paFUFEKbwN5CHTTYh0EAuT//+c9BO0fy4e/evRuVlZVM6GuPlaTuG1rAOp0OGo0m6tVMvTDkGPVI9mIfSv5repZQgkmj0UQFIimcA+n3aXyjEX50nbS0NB7w8rfU9IqE08GDB9mMGTPY3LlzsXLlSlRXV/+c1aPRwGAwQK/XByUqEMjmUOy9QXPsdDqDQFOlNUfRbhxDtc6Wukbz8/O57777Dn/729+QlZXFb3iF6NDCui8hsCZtir/++muceeaZ+L//+z8mDP4rFsgvSFqtFowxHH/88TxYnpzvsry8HIWFhf02xckfLhUshzsEMp1OF5XEIYEm5xYYKDrw0ULCOJUcUefFaK7DcRwPbRNKuUSrQIxGoyge8VtRHhzHoaamhk2fPh2NjY28e4w+o/RoGouUlBRkZmZi1KhRmDlzJlJSUnDhhRcOWVdWuIwv2pRFM5+hknek8QlSDBaLhfv73/+OO+64g61btw6rVq3Cpk2bcPDgwSAoeo1GwysToWwCgIceeggZGRnspptuiqrwVVEgR2BHazKZoNVq4fF4gipfgZ/TUGPZiQiZSq1WIzk5WVYIdnd3o6enByaTKSqh2dLSEpKhU1NTj6l5k76vMBOpra0NBQUFUc1Xe3u7rPuK3FvRzvtvrdUujSfHcbjpppvQ2NgoQlAgi1yv12PevHmYNWsWiouLkZWVhYSEBH4wqJCTgs1DzZK1Wq0wmUyiTQQ9Z6h+QKEUUaTvSVEJAoEAEhMTuQsvvBAXXnghAoEAqqqq2I4dO7Bx40Zs3LgRO3fu5BW00MVOQXmVSoW//OUvuOSSS1hiYmLEeI2iQI7ALstut8Pr9YbcJQ2kDgQACgoKghoOqVQqOBwOVFVVISMjI6KfldxpUmElvMdvUZCFolGjRgX9jxIp9u/fj2nTpkVsSsRxHA4cOMDPsXCHx3Gc7D2OpXWh1Wqxfft2tnr1ah4qX8hj6enp+OSTTzB58uQgpvN4PHyQeqhuHAEgISGBy8zMZN3d3UFQLQcPHkRfXx+zWCwhBTMJ8lWrVrE333yT91oI4fmfeuopJCYmcuGypchVWFBQwBUUFOCiiy4CAOzbt4999tlnePHFF1FRUSGSUSQzurq6sHHjRsyZMyeiHFEUSBRCO1RrSOl3fD4f9Ho9vvjiC96fK3RjESPRbre/wnnatGlYsmSJbKro6tWrI6aKkvL58ssvRUqDdtx6vR4lJSUDUnZHk8UIAMcffzy/gKX0xRdfYMGCBVFd64svvgiaWwE68TGllKUCDQC+/fbbIF8+KeqHHnoIkydP5txuN9+1j75Hscju7u4hO4bk8iktLRUV6NEGr62tDVu3bsWMGTPCdtjkOA4vv/wy3n///aDPk5KS8Pzzz3PAz+nNtGGRnl9aWsrp9Xpx1bhGg7Fjx3Jjx47FDTfcwGbPno1NmzYF1e9wHIdDhw5FZS0rQfQIRIxM1dJyBzG3Xq/H1q1b2SOPPBKUqy6sgRg7dmy/FgEx3IwZM2A2m0W1KcQAL7/8Mux2e8hAmMfjgVqtxnfffcc2b94s2imT/3by5MmggsffurAjV8jEiRM5srqE2Socx+GTTz5BZWUl02g0sq1NKUOvoaGBvf/++0EIAYwxZGdn8zvrwQapPJpIrlkUjdXkyZNF6AnEj8JY0M6dOwddgVDQmeA+hCCa/fEQ/P73v5ftdw4AixcvDiqMFfKRWq1Ge3s7W7duHTQaDbRaLZ9YoNVqMWvWLD4+uX37dpx44omi46STTsKJJ56I8vJyJmw2R8HyQCAAl8uF+Ph47o477pBNE48ldVxRIBGou7sbLS0trKGhgbW0tAQdTU1NrKKign311VfstttuY9OnT4dcGh8thilTpmDYsGH9qkYnwZSRkcHNnTtXtIshU7OhoQG33norr9QoWEYMq9Pp0NXVxW688UbZ6zPGcP3114uU0m/dAiEk3muuuUa0eIS5/ddddx0PY0JQEDSmFJS8/vrr0dvbK/LP0+9XXHEFjEZjWATdY8kSkbMC7XY7j6ggBJ6kcfd4PHj99dcHjTfpvtu2bYNGo+Ezv6LpWR5qg8cYw+zZs4OKcMk19cknn2DFihWMqvqJj4S90m+//XZ0dXXxlfzUQM3r9eL888/n71dQUACdTieCa6GN7gcffMArKikwKhHJKTklmJWVpSiQgRAt9H/84x8oKCjAmDFjUFBQEHQUFhZi7NixOPvss7FkyZKQAVQSJH/+85+jMg3DMT1jDPfddx/PsMTsZEIvX74c8+fPZ+Xl5UyYhhwIBLBu3Tp22mmnYe/evSLYDnIjjB49GvPnz+fC9YD/rRGN480334yUlBTZhb9+/XqceeaZ7IcffmDke6Z8++3bt7Ozzz6brVq1KsiiI4yn2267bVCKAo92ysjICBLO9Pdzzz0HjuOg1+tFmGUEFXPttdey6urqkAgG0SgLqeXBcRyefvpp3H///ez9999nn3zyCVu+fDmrr69n/d3gWSwW7t577w1yU9H8X3vttXjiiSdYb28vE5YC1NTUsGuvvZa9+eabsnw0YsQIzJo1i6O4yrBhw7hx48aJ5ILP5wPHcfjHP/6Bzz77jOl0Ov76JAsMBgMaGxvZE088IYv9ptPpcNxxx0XlwlZiIBHI4/FElcZJEyQ1TWlX4PF4cNlll+F3v/sdnx7Xn10U3WP8+PHcX//6V/b3v/9dlM1CAu+dd97Bhx9+iJKSEpaVlcX3Ati3b5+IKYV+TwB4/vnnIeznPRRTJY+UFZKcnMwtXryYXX755bylQYtVpVJhw4YNmDx5MkpKSlheXh4A4NChQ/jpp59kx5T6aDz55JPIzMzkjqUxleNbAHwygpD3SZC/8847cLlc7Nprr8XIkSOh1+ths9mwc+dOPPfcc/jhhx+g0WhC1iiEq7OhDDi53bbdbscjjzwi+uybb75BTk5OvzYjfr8fCxcu5N566y22efNmvh+I0K11zz33YPHixSguLmZGoxEtLS3YsWMHHA5HkIIkPnr88cdhMBhElu+tt96K6667DlqtViR7nE4nfve73+Gqq65iF1xwAQoLC/keMOvXr8fSpUvR0NAgu4mcPXs2cnNzlTTewRIu0ZrmwkknAUIw0WeffTb+/e9/c4OBLUVK5KGHHuJ2797N/vvf//IMRM+hVqvhdruxdevWoPeRMg3tXP75z39ixowZR6SZ1NFghfj9flx22WXcTz/9xJ588kneNUUHLewdO3Zgx44dsucLNxNerxe33HILbrjhhmNyTKU8GwgEMHXqVG7ixIls586doiQTsqQ/+ugjfPTRRz8LJ0kSirQxk5y7OZzikuJ8SdeqMD1YDjQzWnlBcdF33nkHJ598Mg4dOhTUoIzczQ0NDWH5iJTHwoULceGFF4r4KBAI4JprruHeeecdtnr1ahGUCb3ja6+9htdee012PIWKij6zWCx4/PHHo64/UlxYhyctXJV5NAf5TgldloSyWq3G3XffjU8//ZSj4jC5iYlU5S5lUrrH22+/zV155ZU8GBoFICkATs8lvBaZ1kJmfeaZZ3DHHXdw9MyxjlMooRHL92M5PxyGVixjKbd4n3jiCe6+++7jffF0Pi0q4XOR71n4PeoJceedd2Lp0qUcWYWxjGl/3CexjtVgPUc080QCTaPRYOnSpXysgyBKpHxJbmSVSgW9Xg9KYCguLkZGRoaIr+kgOBpptfZhHCrummuu4S1r6ZoQ9uOQegZiHVtam7m5udzatWtRXFzMC3aSD7QhET6LkI9o8+L1enHDDTfgueeeC+IjkiXvvPMOTjrpJL6dL8VSVCqVqJ0DubeECNj0Hj6fD0ajEe+++y4KCgq4qN2tx3pDqc7Ozn73Eg53pKens2uvvZZt376dSe8pbShVXl4e8jrUCEiuuYswMLZs2TKWm5sr25SIOhfKdas74YQT2IYNG0L2QRY2lMrIyJB9xhEjRsg2lBozZozs95OTk5nb7ZYdE+G73nLLLREbD9Ez08/Vq1eHPOe+++6LqlEOzcvHH3/MioqKZPtWUyMxuTEdO3Yse++995iwg6TcmLrdbiQlJck+69ixY1mo8REe9N4ffvhhyPf+//6//0/2vQf6HHS9hQsXhrz31q1bRd0j6ecHH3zApPeMNK5XXnkl6+7uZpMnT5a9V2ZmpixfUSKJ3W5nl1xyScw90ZctWxbyey+//HJInqJ3tdlsbNGiRUyv18t2fJTrLAqAZWVlsZdeeikkHwnf0+l04u6772YmkymkDKD7yHWfPPHEE9m2bdti6oeuNJQ6bBbPnj0bNput3/AIHMfBYDAgLS0NhYWFOO644zB58mQkJiZywrhEqOCh2WzGWWedJcL9p93wpEmTQrrShIVK119/PXfRRRext956C++99x5++ukndHV1yfqLc3JycPLJJ+Oyyy7DnDlzokLh5DgOZ599Nu83FZrJw4cPlz1nxowZyM7ODmpYQ/3PI7kNi4uLccYZZ4gsJbqGFM2XfqampuKMM84QzSWdP2bMmKjckrRrnTNnDjdz5ky899577O2338bWrVvR2toqO6ZpaWmYMmUK5s2bh4suuogjX3W4MVWpVJg5cyZaW1v556WxirboUAiMecYZZ4jcEvTekYpC+/sc0cwTgXsK6zn8fj/OO+88rrS0lD399NP4+OOPUV1dHTSuarUaw4cPx/Tp03H11Vfj5JNP5gDgsssuY3FxcUH3OxyTYrSDlvKGyWTi3nnnHdx4443sv//9L8rKytDc3Ay73c7PlcFgQFxcnAiiftiwYUHvR79T86VQlgj1tlm8eDFuvvlmtmLFCnz22Wc4cOCArDvOarViwoQJuPDCC3HllVdGbEInBFV84oknuBtvvJGtXLkSq1atwv79+9Hd3S3LrxqNBjk5OZg2bRrmz5+PuXPn9guNlzsWA3q/FFFw8JfIvJFOfFtbG6uurkZTUxP6+vqgUqmQmJiInJwcDB8+HGazmRPGb4717KBoxrS7u5tVV1ejoaGBb2gVFxeH7OxsDB8+XAS5cazHPGIZW5fLhfLycnbo0CHYbDYermfYsGEYPnw4R/EI2oT0twZECilEz+FyuZjAdcRFg1kW632FFd2MMVRVVbGamhp0dHTA6/XCZDIhPT0dw4cPR2ZmZsx8JL0HyYC6ujo0Nzejp6eHbxCWnJyM7Oxs5ObmckLMu36VFigKZHByyqX9OSK10Yz2GWJZMGTqRoOjI9xJDcY4yd0v1u+HG9Nozw93Tn+EjxDoL9IzC6uOY0m+GMgYDdZ7D+Q5+jNPwvGKxINSXh3omJFrKdr40GDxFLl3w/X5EK7jaJ9P7h7RnistIo7Z+6IokN8mCYOBQsUmBwutUGxCWrpZUMb06B3XUF0Wf433HYggj/Y+gz22igJRSCGFFFKoX6Q4vhVSSCGFFFIUiEIKKaSQQooCUUghhRRSSFEgCimkkEIKKQpEIYUUUkghhQ6TAqaI0HneSmqmQr82Hyo8+Ouv41B1J0Oh+DbUO/9Sz6ak8SqkkEIKKaRYIP3R3hzHoaKigrW2tkKj0fD/8/v9GDduHOLj47looY0VUijUjjXcLpj4q7y8nLW3t/OVyj6fD8nJyRg9erTCfFGs497eXrZ7924RcrLP50NKSgpGjRrVr3XMGENZWRlzOp38rp6qySdOnMj1F/b9SL0zfVZcXCyCK1IUyBGchE2bNmHjxo2wWCw8bIXL5cKiRYsQHx8PRYEoFAvF6j4g/tqwYQO2bdsGs9kM4OdGR5MmTcLo0aMVHoxi/FpbW/H6669Dr9fzkDJ2ux1TpkzBqFGj0F8F8sknn6CpqQmEkeX3+2EwGDB69Gim1Wp/lQ1mqHcmBXfPPffAbDYfcb5RYiAA9Ho9LBYLzGYzr0Cot4dCCsVCXq8X27dvZwSkyXEcvF4vxowZg9TU1LDChvjQZDLxVovBYFAGNUpSq9WwWCzQ6XQiTDIhYGB/yGQywWKx8E2mSIEMBYUufWdSIL9UDESRkICojwUpEGkDeoUUimZH6HK52Pvvvw+n08k39unr68P111+P1NRUxZL4hdaxUJgOdB0LZYPw76H8zr8UKQpEIYUGkTiOg9lsDupYF42v3O12w26383/b7fawfb4VUujXJkWBKKTQIJPQko3GmiUlM3LkSJHLxe12h2zWpZBCigJRSCGFeAUyY8YMbsaMGWG/o5BCigL5DVA0fQTk8P6l/Q6G2r0iPYe0r4Bcn4H+vlN/BGV/rxOqaE96zWjeSXitUM8TrgeE0HIRxkjo9/5kdUnfoT/vNRh8ONB7/lZkQ6y8Pdh9So6UfFAUyAB3jeEmX25ihH+HEiSx3itUFzy5xd0fRqGsjkjvI/1+f99psOYh1rkJdc1IrVSFn2m1Wg5A0OqngHqkHumDIbhCPWus7zUYPC/Hg791i2ow3m0wm0oNpixSFMggkM/nQ29vL5MufmnRYX19PTtw4ABaWlrgcrn4lLuMjAwUFhYiLS2NiyTYGWPo6elh0h1JXFwcJxTqHR0dbP/+/WhoaEBfXx84joPJZEJaWhpGjBiBvLw8jnaEsVgJtANmjOHQoUOstrYWra2tsNvtfNtNi8WCtLQ05OXlITc3lyNBGOpejDF0d3cz6f+0Wi2sVmtMHGyz2ZjP5xPtqkJdR6jYbDYb3y+6q6sLDocDPp8PKpUKRqMRSUlJyMrKQl5eHiwWS8h58vv9sNlsjD5zOByylo3NZkNXVxejZ2CMwWq1igrR+vr6mMfjEb2LTqfj7x/NXAFAe3s7q66uRlNTE2w2G+ia9F65ubnIz8/njEZj1Arf6/Wit7eXCZ9No9EgLi6OE55fW1vLysvL0draCrfbDbVaDavViqysLBQUFCAlJSUizx/t5HA4mMvlioon5cjlcsHhcDDh+KhUKsTFxXGxWjE0L3V1dezgwYNobm4WyaL09HSMGDECWVlZ/ZoXRYHEqM05jkNTUxNbunQpn1nj9XqRmpqKRYsWQavVorW1lX366afYv38/3G43v3sXmpEGgwHjx49ns2fPRlJSUlB9AP3tdrvx/PPPw2az8ZXyAPDHP/6RZWRkcE6nE6tWrWI//PADrziEjEtCKD8/n82aNQvDhw/nohEY9B2/348tW7awLVu2oLm5mRdGdAjdMjqdDllZWWzatGmYOnUqp1KpgoQTfXflypWora3lC6ACgQCMRiPuuOMOZrFYuEhK9bBQZk8//TQ/xiqVCn19fTj33HNx1llnie5Ni6m+vp598803KC8vR29vr2gXLn2fw5sCjBs3jp1xxhlITk7mhAqA4zh0dXWxxYsXi5QG9aWn6xgMBqxatQpffvklr1DcbjduuukmNnLkSI4U8YcffogdO3bwdSAOhwMTJ07ElVdeGVYZ07MfPHiQbdiwAZWVlXA4HKKdp/C91Go1EhMT2cSJE3HqqaciISEh5HjT+1ZXV7Nly5bxdSlutxvDhg3DH/7wB6hUKjQ0NLBPP/0UBw8ehNfrleV5k8mECRMmsNmzZ8Nqtf7mEB5orL766its2LCBL+SjZIiFCxeGFdB0/tatW9lHH30Ei8UCxhi8Xi+SkpJw5513Rl2bRnxdV1fHvvjiC1RUVISURXq9HsOHD2dnnHEGCgsLY5oQRYH0U5EId70+nw8ejwdarRYHDhxgr732GhwOB8xms6iISThpgUAAP/74IyoqKnDdddexYcOGhVxQPp8PPp+Pv0YgEIBer0dXVxd76aWX0NDQAIvFgri4uJB++crKSixduhTz5s1jkydP5qJh5IaGBvaf//wHNTU10Gq10Ol00Ol0QeawUDg1NDTg7bffxrZt29gll1yCjIwMTk6QFxUV4cCBA9BoNPz92tvbUV5ejuOOOw7RKJDy8nJ0dnbCbDbz46PValFcXCwyx+n6a9euZatWrYLP54Ner4fJZArpwqH3cTqd2LRpE3bu3Il58+axCRMmBOHH+f3+iLUGwtx8gsqRu45wrn0+H/x+f8QNjcfjwYcffsi2bt0Kxhj0ej3MZrPsu9F79fX1Ye3atfjhhx8wZ86ciDxBPC98Nq/XC47jsHPnTrZy5Up4vV6YTCaQZSN970AggM2bN6OiogI33HADS09P/03CBAnnkcYt3DyGG+v+nB8IBGA2m7Fz5072xhtvwO/3w2g0hpRFjDGUl5ejvLwc06dPZ3PmzOGidWcpcO4D8FEKF6jJZEJ1dTV75ZVX4Pf7YTKZ4HA4YLPZYLPZ0NvbC4/HI1pUFosFfX19WLFiBfr6+pjQFxnuXjqdDjabDa+88gpaWloQHx8Pt9uN3t5e/l4ul4u/F2MMRqMRGo0Gb7/9Nmpqahill4Yye8vLy9mzzz6L+vp6WK1WHsYhEAjA6XSK7uV0Ovnn1ul0sFqtqKmpwb/+9S9UVlYy2o0LmXLChAmwWq2i+I1KpcLu3bsj+mLps7179/LnqtVqeL1e5OfnIyMjgxdMdP1Vq1axDz/8EFqtllccZD329fWJ5qmvrw9er5d3HVgsFvh8Prz66qs4cOBA0Nj5/X7REWpRS78XCnlXeoRTHna7nb3wwgts48aNMBgMMBqN/DlCnrDZbOjr6+MVALmW3G433njjDXz55ZcheSIUH5rNZpSXl7MVK1ZApVLBYDDAbrfL3o+ua7Va0dHRgddee42vcTnSBbv0vP2JMfXnnFjmMZZrRKt8DAYDtm3bhpUrV0KtVsNsNgfxgsPh4HmV5IPRaMRXX32Ft956i0mVv2KBHEEizJ23334bjDE+NlBUVIT09HSo1Wp0d3ejsrISHR0dMBgM/PeMRiPa2tqwYcMGzJo1izc9I93vvffeQ3NzM7RaLVwuFwoKCpCdnQ2dTge73Y7a2lrU19fzu45AIAC1Wg2Px4NVq1bh5ptvDhlkb2hoYMuXL+chG8glQ77TkSNHIisrCwaDAS6XCw0NDaipqeF3v/ReXq8Xr7zyChYtWiTabTLGkJiYyI0cOZLt3r0bRqMRgUAAOp0OVVVVsNvtzGw2y+5OBbEGVl1dDZ1Ox1sYPp8PEyZMEO2wVCoVDhw4wFatWoW4uDjegqN4RXx8PMaMGYOUlBTodDq43W50dnaitrYW3d3dMBqN8Pv90Gg08Pv9+OCDD3DnnXfy7kuVSoXExETRbq6vr08kFBljMJvNPNwEWQ0DBeLz+XxYsWIFKisrERcXB4JP8fv98Hg8yMzMRH5+Pv9ZW1sbqqqq0Nvby78X+cI//fRTmEwmdsopp0Tl4tRoNOjs7MS7774LrVYLj8cDs9mMkpISpKamknsPFRUV6O7uFvG82WxGXV0dNm/ezKZPn85Fw/MDtQg8Hk+/ID6ONkQKmv+1a9dCrVbD7/fD4XAgOzsbubm5sFqt8Pv9aGlpobUGk8nEK4v4+Hhs3rwZKSkp7Oyzz47IC4oCGQR3llqtRk9PDy/EioqKMGfOHKSmpopWhdPpxMcff8y2bt3KC81AIACDwYBdu3bh7LPPjujjJOHT2toKxhhSUlJw4YUXYsSIEZyU8Tds2MA+/fRTXnDRvWpqatDS0iLrQvD5fHjnnXfg8XhgMBj4RedwOJCfn4+5c+ciPz8/aLVXVVWxjz76CHV1dfy7abVaOJ1OvPPOO7yvXKgESkpKUFZWJhJK3d3dKC8vR2lpKcIpEBJMxPx+vx9WqxXjx48X7ToZY1izZg2f/SSMLZ122mk4/fTTERcXF/Q+vb297PPPP8fWrVv5cdDr9WhpaUFFRQUbN24cFwgEkJiYyN11110ii+Dpp5/mle3h/+HCCy9ESUmJaEHSXMcqPOkaa9asYfv27UN8fDyvPHw+H3Q6Hc4//3xMmjQpCC22q6uLffXVV9i8eTOMRiMvHC0WCz755BOMGDGCZWdnR3RnaTQadHV1QaVSwev1YtKkSZg1axYSExNFJ/X19bH3338fO3fuFPG8TqfDjh07cNpppx0x5UFzVlFRgSeeeKLfWsDhcIiQuo8GmaTRaODxeBAfH4+5c+di/PjxnFQRtLW1sc8++ww7d+4MWkdfffUVioqKWFZWVlheUFxYg0QajQZutxvFxcW49tprudTUVE6IoUNB4nnz5nG5ublwu938blytVqOrqwvt7e0sGpOeXEIpKSlYuHAhRowYwcnheU2fPp2bPHkyhFDUhDRcV1cnch/QOVu3bmU1NTX8YlepVHA6nRgzZgxuueUWLj8/P+hejDGMGDGCu+WWW7iCggL+foFAACaTCZWVlfjxxx95Fwkx45gxY5CUlCSKJ3EcF5Uba+/evfyzq1QquN1ujBw5kg8I0/kdHR2ssbGRD/7SPE2bNg3nnXceFxcXF/Q+gUAAVquVmzdvHpefnw9hZlQgEEBTU5NIoVN8SKvVhrQqNBpN0Pf6I4zIquro6GAbNmyAxWIRWR56vR433XQTpk2bxmm12qD3SkxM5C655BJu9uzZ/DzRNb1eL1atWhWTkHK5XJgyZQouv/xyLjExMYjnLRYLd/nll3Opqal8zITObW9vR09PD5PWiwz2jtzr9aKnp6ffx1DBvYr1na1WKxYuXIji4mI+oUV4pKamctdccw134oknwuFwiGSEz+fDmjVrlBjILzVhPp8PFosFF110kWiXKDzI5zh58mR+MQmtip6enqh9wn6/HxdccAEsFgtHAkR4L3LhTJs2TdQrgK7f3d0dpJT8fj82b94MvV7PC3qv14uEhARcccUVIIEkvRcpBr1ejyuvvBJxcXG8UiBL5LvvvguC9zCZTNzo0aN5ZUo708rKStjtdlnBQsqisrJShEDKGENJSUlQgPDQoUNob2+H0+mE3W5Hb28vfD4fTj31VJE7S26uGGMoLCwUzRUAPpYlFajh5i7S57EoEADYsmUL7Ha7yL3g9Xpx8cUXIzc3lyNek74XvfOZZ57JlZSU8IKDNjiHU8EjCnXijeTkZJx//vm8EpYbR41Gg0mTJvHzLJxHm832i6xPQtfuz3E0ks/nw4UXXojk5OSQvECbvwsvvJAbNmyYaB0aDAbs378fnZ2dYXlBUSCDxKButxslJSWwWq0h/Yb0v5ycHGi12qCKXQrcRuPCysvL45vkyBWnkaBOS0vjhP5xOSFIO6xDhw4xiqsIXT0zZsyA2WwO6w8lhrRardxpp50GyoOn9N7GxkZZwVRaWgphkJ3cWAcPHgwSuPScNTU1rKOjQ9R4KSkpCaNHjxa9OwBkZ2fjyiuvxLx58zB//nzMmzcPV155JZKTkzlplTcpHmFF+FADMyShvHfvXlFMxe12o7CwEBMmTOAo3hWKf+hdzz33XH6zIOQtcitGUiButxvHH3+8qP9GqPvl5uYGPZOQ5490nEGKBBDLcTTKohEjRmD8+PFheUFofc6cOVMkIyiue+DAgbDzoyiQQZy4wsLCqBiO8Pv7YxrTzm/kyJFhJ5YYwWAwwGw2R5UGWFlZKdpt+/1+xMXFYeLEiVH5f0k5lJSU8K4VoWCqrKwUuWEAYMSIEVxGRkaQRRbOjbVnzx6RNeN2uzF69GiYTCZOWl2dnp7OnXzyydzUqVO5adOmcSeccAJ3/PHHc2q1OggGm85TqVTQaDRoaWlhZWVlfAB4KPi2AaClpYXvXCjsoDlp0qSolRAApKWlcSNHjuTrA8i1VFVVFdGFSN8tKCiIyBPAz9lXlLKt0JH3hlAySbS8MGrUKC4jI0Pkrj1c/xPeda8M+eAsbI1Gg/j4+Kj82mq1mt+x95cSEhKiZqhw8BlCamxsFPlBPR4Phg0bBooTRAu5kpiYyKWnpzMqFKTPhLEDsig0Gg2Kiorw5Zdf8jtqcmM5HA5mMpk4YUW8z+fDwYMHRVaSSqXi3VeRdp5Ct5vwfbxeLxwOB7Pb7Whvb0dlZSV++uknOJ1O0b1+bT7jOA4tLS1wu918AzRyP+Xn50cU/NJrjRgxglfWZAF2dHTA6XSKguxy5+t0Or72KNI9hdD2v6S15nQ6MW7cOPz+97/vdxbWihUrILR4h7os0mq1yMvLi5oXaB3m5+fznRfJs9Ha2ipSNIoCOYKa/5dcIIPZcYyuRZlkwl1tWlqaSOBEw4wqlQppaWmoqqoS1WJQjEdoaQDAxIkT8c033/BCntxYFRUVmDBhgkj4Hzp0iLW1tfF1KR6Ph+BaOLlxkasBaG9vZy0tLWhpaUFHRwc6OzvR1dUFu90Oj8cDr9fL59MLXY1Dhbq6ukRWk8/nQ0JCAhISEqIuACNKT08PKvJ0uVzo6+tjRqORi+TG+qU63w1EoBqNRh42qD+k0WjY0eDKoviF0WhEfHx8zOdnZGSIeJ3cWG63m3dTSnlLUSAK8QvN4/EExQQsFku/rkcwDEJmpOJGoQJhjCErK4vLzc1lNTU1vMXCGMOuXbuCTPG9e/fC6/Xy3/N4PBg/fjwf4JcKNGJ6n8+HrVu3sh9//JHHAyJ3l1qtFoEdqtVq+Hw+uFwu6PX6IZe6SeMoVNpGo5FXqrHOkzDJgsZKLlHgaCVyUx4LdSBkgej1eq4/vCAFwjy8oWKhrqcoEEVxiFw9g2XpRHse7XonTpyIiooKPt6g1+tRWVkpcqUEAgGUl5eLrAKdToeJEyeGddO0trayN954A7W1tXwKrclk4pUmwUZQ7EOn0yE9PR2lpaXYvn07WlpaBlz0N9gCMZT7MJbdajjL+bfWznkgFeFH23rurzdEunkMJxcUBaJQkItHzl3jdDr7dV2HwxGElyUnhOk7xcXF+PLLL/lMECpUq6ioYMXFxRxVyDc3N4vcVzk5OcjJyeGkWUD0Hr29vWzZsmVob2+H1Wrlha/dboder0dubi6ysrKQnp6O5ORkxMXFwWKx8O6gsrKyIee+EGIa0Ri63W6+sjwWQeNyueD3+/m5oXEcSgpTodiUwGGcMmYwGGLSIkLLltYlWeSKAlEoojCxWCyiQj+1Wo329vaYdmL0vY6ODlFA/nBRGb+DFn4mhDbZtWsXj0bLGMPu3bt5cMR9+/bB7XaLUJCLi4v5hASpwjqMgcXjhVFtitvtxqRJk3DGGWcgMzOTO9rcF3FxcaJ0XLVajb6+PvT29rJwyLpy1N7ezittIVoBzYFCv55XoL/r2Ol0wmazwWq1xnS+MLZG/H8YhDFkbE1J41WIZ5q0tDQRwJpGo0FTU5OoACwaBnY4HKJ6EmLG9PT0sPcn+BL6n16vx8GDB3kraN++fXzqKjG3FHlXqDzsdjvbvXs3TCYT76JyuVwoKSnBFVdcwZHykFbVCyvchyKlpaWJoP1JgTQ2NkZdu0DjVVNTI1Lmfr8f8fHxfA8SpZXu4K6xaMaToPj7a4F4PB7U19fHzAsNDQ28tUG8kJCQEFSIrCgQhWQpPz9flJGj1WrR0dGB8vLyqNA5ickOHDiArq6uICFHaaahGHjMmDEcQZsA/wPsO3ToELPZbKy+vp6Hk3e73cjPz0daWppsLxUAfOMraZYRVaHLVehK+6kMpboFeq6MjIyg4lDGGHbs2BG1oj/sxmMHDx4UFST6fD7k5uYiEjKvQqFJyPfEX5S0Ec38tra2DmjzQqjWsfBCT08Pq62tFdWn+f1+ZGdnh7WKFAWiEM9oh7GkeFcPCdyvv/46onkttDS++eYbUZEbNcQZPnx4yFRbsijGjBkjsngCgQAOHDiA/fv3izC9/H4/HzwP9UwEWS28lsFg4FMcQy1SskR6e3uZVBH+0m6JUOM0cuRIvuiL/rdz5040NTWxSDVG5O779ttv0d3dLXo/juNQVFSkLIoBkNlsFs07ga329vayUFYB/c/pdEIqyGMh4vF9+/ahpqaGCSGUwvHCxo0b0dfXJ4p3qNVqFBYWhldWynQrJMSmmjhxIlwuF18PYjAYUFVVhVWrVjEh7pUwNVKIJ/XZZ58xYadBwjwqLS0VwWaEIiG0Cd1/z549oH4XZDnExcUFIe/KvZe0LajL5UJPT4/oPaTvQu/5xRdfwG63ywYRw8VH5BCEhf1Z6BgITZs2LSjlUoikLATPk4JFqtVqVFRUsK+//poHzaSUzezsbBQWFnKhoEkUikwpKSmisVOr1ejt7cXevXt515B0XmiztW7dOkZKfaBr+p133oHdbmdS1AU5Xli/fn0QL6SlpYXc9CkKRKEghmOMYfr06UFgiCaTCatXr8ann37KqF2psAkU+V0/+ugjtm7dOh4aWgjGSG6jkLDQhxl0+PDhPLQJLb7u7m6+9wm5rwoKCkJWyNPfCQkJspll//3vf9HR0cEIEUD6Lna7nb377rsi2H0hUX8UOcWi1Wo54T0plrRz5054vV5oNBr+Pv11TzDGMHLkSK64uJhXcKRsDx06hGXLlrG2tjYm924qlQo7duxgy5cvF40VzeGZZ545KBbXsWzJ5+TkiPiG4nlffvklGhoaGPGAdF7Wr1/Pvv76a1F/jv5auzqdDq2trXjxxRfR2NgYkhfKysrYq6++Ktps0abvhBNO4OurQrrrlGlXSGiFJCQkcL///e/Z66+/zmdxkHBas2YN9u/fz0pLS5GTkwO9Xg+Xy4X6+nps374djY2NfH0FMaTL5cL8+fPDgkwKd/Vy0CbCPs7E5ELk3VAKJD09nUtNTWVUx0ELq6GhAUuWLEFJSQnLz8/nYWF6enpQU1ODXbt2obOzk38X6cLct28fRo8ezbRaLTIzMzlhsaFer4fFYkFXV5dIeFRVVWHx4sUsOzubD3qfffbZfApyf4TE3LlzUV1dDafTybs8jEYjqqqqsGTJEpSWlrLDipZvKLV7924+GYEUj0ajQU9PD6ZOnYqSkhLF+hjgJiwpKYnLy8tjBw4c4BWJWq2Gw+HA888/j6lTp7KCggIYjUZ4PB60tLRg586dqKqq4huyhQtcR0NerxdGoxENDQ149tlnUVJSwkaPHo34+HgEAgG0t7djz5492LNnD5+qS/PudDqRm5uLE044ISIvKApEIdHuNhAIYPLkyVxHRwf7/PPPYTab+f+bzWa0tLTg448/5nfS1J5Vp9Px2Ez0/b6+Pvz+979HaWlpVF3uhMrhm2++Ee18pPGU0aNHh80SokU7Y8YMvPrqq7z7jJSAy+XC+vXrsWHDBt6SIAh3UgIul4v/PrnUtFot2tra8MILL4DjONx9992MWujSYsvJyUFVVRWMRiPf11qn06GlpQX19fW8i+Dkk08esKC6+uqr2bJly+B2u/nukQaDAV6vFxs2bMC3337LWxSUiUa9UcjC6+npwZgxY3DRRRdxiuUxMCI+Pf3007Fv3z7RfGk0Gvh8PqxduxZff/01yLVETcCoV0xxcTG2bdsWsxtL6IY0m83Yu3cv4uLi4PF4sGnTJmzevDkkLwh7wmg0GsyfP1/ULiGkzFCmHLL9IGjXG07YSo9YhXW094v12QZ6r0AggHPOOYe76KKL+JaYwh221WrlcaKMRiOsVqsINNHhcIAxhnnz5uHMM8/kooWQoIWWmZnJDRs2DD6fD0I3E/U9HzNmDN8lMNx7MMYwadIk7vTTT0d3dzeE0NbUytVkMkGv1/NKw2q18u1Yc3JyMHPmTLjdbh4MkASBwWCAXq8XvRc9y0knnQSj0cj32qDxprGzWq08hEgkPgwX3wkEAhg5ciS3cOFCJCcno7e3V5T1Ru93GNqC/5usQ4/Hg76+PkyZMgXXX389J5zDaNfIkeL5X2od9/e5w7U1ONxHhjv77LPR09MDYcsFlUoFi8UCo9HIr5/4+Hi+6+cFF1yA8ePHw+v1ing/1P3k3lmtVuPiiy9Geno6enp6oFarYbVa+bbKer0eZrMZRqNRdA2n0wmNRoMFCxYgNzc3qlbDigUCwO12o6+vj9+5kuuF0knldrd2u12E0urz+aL2WzLGYLfb+TgDgZbJ3Y8xBofDAbvdzuM19fX1RdU7hMjhcKCvr4/Hderr6wuLdUSL4NRTT+VGjBjBvvzySxw4cIAXpMTYwuIzskQMBgMmTJiAmTNnIisri4sVf0gIbVJWVsYXNwp3WKGQd0MppPPPP59LTExka9euhc1mE2Ff0T1pJ0ixk5NOOglnnXUWd7jdMKuqqhJl1/j9fni9XlGGC90vIyODu+6669h///tftLS0BPViIZ6TZscQHwrSbMP2IyFln5eXx91+++1Ys2YN27ZtGw+KSXNFz0UJCBTDyc7Oxumnn47S0lIulDuQyOfzoa+vj7fShO8bDdGaEdacyI3BQMjv9/N8LqgFGnBPF1p/VMBKYxguiYIxhlmzZnEGg4GtWbMGvb29smvH7XbD5/MhNTUVl19+OSZMmMBt3ryZ2e12UZ8gEvbh3plcVz09PUhMTOT+8Ic/sA8++AC7du3iN2NSVGS/38/LocLCQpx//vnIzMyMftN3LJustGAqKytZa2trUH+FsWPHIj4+PihQ29fXx/bs2SNCrg0EAhg/fjxfgBVJYe3atYsJBaPP58Po0aORlJQkup/f70dZWRkTAh1SPxC5nuZytGvXLibMJvJ6vcjNzUVubm7Y84VMVF9fz/bs2YPa2lp0dXXxee1kBicmJiI/Px/jx49HdnY2Jz0/1jlxOBxs165dQWOs1WoxceJELlrIDuE1u7u72U8//YSDBw+io6MDbrebdy9ZrVZkZGRg5MiRGDVqFKxWKy9UbTYbW7VqFaqqqvhe51arFcOGDcM555wDs9nMyd3P7/ejoqKCNTY2wmazgRIQjEYjzGYzSkpKRIkABw8e5Pt8kNBOTk7GqFGjuGje73Ach+3ZswcVFRVoa2uDw+HgBYRWq0VcXByys7MxZswYjBkzhhOOb7hr9/T0sH379gUBLxYXF3NCl1gYIcyEsPE0PuPGjYu6XUCk9+/t7WXk0xdu7FJSUvjMsljvcRjUkwlTyMmimDBhAhcO8oXu19nZyX766SdUVVWhs7OTXzsajQYpKSkYP348Jk+ezBHmW3t7O6uoqOD5gLp9TpgwgRM+g/Sd6btGoxFFRUX8d2tqalhZWRnq6upgs9ng8Xh4a9xsNiMrKwsTJ07EuHHjYl63is9ToYgLSOrW8Pl8cLvdjJhQp9NxQn9tLFW3vxRJF4XX64Xb7WaHFQgnxZeSW0Rerxcul4up1WqYTKaohfovNU9yeGCHLWnGcRx0Oh0nReztj5JXqP98J1w7Wq1WpHwHey7k1qHX64XH4+HXrtFo5KRFuLHwraJAEBpxMhyqpZy7KpbJD4WoKne/WL470HtFGqNw6K0DQQKNdk5iHedYnlFoEUoXlVw2SjQLPpxbM9R9BmOewo0TPVOsrsWBojUPBh8O9jru73PH8u7h+E7us1jGOprvRsv3/VlXigJRaEC7m6FmaRzpdxiK1tVvfZ6UtTN0768oEIUUUkghhfpFivNTIYUUUkghRYEopJBCCimkKBCFFFJIIYUUBaKQQgoppJCiQBRSSCGFFFJIUSAKKaSQQgopCkQhhRRSSCFFgSikkEIKKaQoEIUUUkghhX7D9P8DMydEZ4Gi4a4AAAAASUVORK5CYII="
+_now = datetime.now(timezone.utc).strftime("%Y-%m-%d")
+display(HTML(f"""
+
+
+
+
Xenium Image QC
+
Date: {_now}
+
+
+

+
+
+"""))
+display(HTML("""
+
+
+"""))
+```
+
+
+
+
+```{python parameters}
+#| tags: [parameters]
+#| echo: false
+
+# Nextflow QUARTO passes INDIR plus samplesheet-derived paths (see modules/quarto.nf).
+INDIR = "image_qc"
+SAMPLE_NAME = "" # samplesheet `id`; if empty, inferred from INDIR (parent of folder named image_qc)
+XENIUM_BUNDLE = "" # samplesheet `xenium_bundle` path string (local or s3://)
+SAMPLE_PUBLISHED_OUTDIR = "" # expected publish dir: params.outdir / id / image_qc (set by Nextflow)
+ROI_THRESHOLDS_YAML = "" # optional path override; defaults to conf/roi_image_qc_thresholds.yaml if found
+PIPELINE_VERSION = "" # nf-xenium-processing version from workflow.manifest.version; empty for local renders
+```
+
+```{python bootstrap}
+#| echo: false
+
+from __future__ import annotations
+
+import json
+from pathlib import Path
+
+import numpy as np
+import pandas as pd
+from IPython.display import HTML, Markdown, display
+
+base = Path(INDIR)
+
+# Analysis status marker written by modules/image_qc.nf. On a hard failure of the
+# Python analysis (e.g. a very dim sample where no tissue tiles clear the intensity
+# gate), the analysis exits without writing roi_qc_metrics.json. Rather than raise
+# and produce no report at all, we render the report with a prominent QC-FAILED
+# banner and let downstream sections fall back to their empty-input defaults.
+status_path = base / "image_qc_status.json"
+analysis_status: dict = {}
+if status_path.exists():
+ try:
+ with open(status_path, encoding="utf-8") as f:
+ analysis_status = json.load(f)
+ except Exception:
+ analysis_status = {}
+
+roi_json = base / "roi_qc_metrics.json"
+ANALYSIS_FAILED = analysis_status.get("status") == "failed" or not roi_json.exists()
+
+if ANALYSIS_FAILED:
+ roi_metrics: dict = {}
+ _sample_label = SAMPLE_NAME or base.resolve().parent.name or str(INDIR)
+ _exit_code = analysis_status.get("exit_code", "unknown")
+ display(
+ HTML(
+ f''
+ f'
⚠ Image QC FAILED — {_sample_label}
'
+ f'
The image QC analysis did not complete '
+ f'(exit code {_exit_code}), so no metrics are available for this sample. '
+ f'The most common cause is a very dim image where no tissue tiles clear '
+ f'the intensity gate, so the focus/blur model cannot be fit.
'
+ f'
What to check: DAPI/stain intensity and '
+ f'tissue extent for this sample, and whether the correct morphology '
+ f'image was supplied. Sections below render empty because the metrics '
+ f'were never produced.
'
+ f"
"
+ )
+ )
+else:
+ with open(roi_json, encoding="utf-8") as f:
+ roi_metrics = json.load(f)
+
+intensity_path = base / "intensity_assessment.json"
+intensity_stats: dict = {}
+if intensity_path.exists():
+ with open(intensity_path, encoding="utf-8") as f:
+ intensity_stats = json.load(f)
+
+threshold_path = base / "roi_blur_threshold.json"
+threshold_cfg: dict = {}
+if threshold_path.exists():
+ with open(threshold_path, encoding="utf-8") as f:
+ threshold_cfg = json.load(f)
+
+cell_metrics_path = base / "image_qc_metrics.json"
+cell_metrics: dict | None = None
+if cell_metrics_path.exists():
+ with open(cell_metrics_path, encoding="utf-8") as f:
+ cell_metrics = json.load(f)
+
+versions_path = base / "versions.yml"
+versions_text: str | None = None
+if versions_path.exists():
+ versions_text = versions_path.read_text(encoding="utf-8")
+
+
+def load_snr_summary(indir: Path) -> dict | None:
+ """Prefer nested snr under roi_qc_metrics; else standalone snr_metrics.json."""
+ if "snr" in roi_metrics:
+ return roi_metrics["snr"]
+ alt = indir / "snr_metrics.json"
+ if alt.exists():
+ with open(alt, encoding="utf-8") as f:
+ return json.load(f)
+ return None
+
+
+snr_summary = load_snr_summary(base)
+
+
+def load_roi_cutoffs() -> dict:
+ """
+ Load optional report cutoffs from conf/roi_image_qc_thresholds.yaml.
+ Falls back to empty dict when unavailable.
+ """
+ try:
+ import yaml # type: ignore
+ except Exception:
+ return {}
+ candidates = []
+ # Optional explicit override from Quarto parameter if provided
+ try:
+ c = str(ROI_THRESHOLDS_YAML).strip() # type: ignore[name-defined]
+ if c:
+ candidates.append(Path(c))
+ except Exception:
+ pass
+ # Common repo-relative locations for local and Nextflow runs
+ candidates.extend(
+ [
+ Path("conf/roi_image_qc_thresholds.yaml"),
+ Path("../conf/roi_image_qc_thresholds.yaml"),
+ Path("../../conf/roi_image_qc_thresholds.yaml"),
+ ]
+ )
+ for p in candidates:
+ try:
+ if p.exists():
+ with open(p, encoding="utf-8") as f:
+ d = yaml.safe_load(f) or {}
+ if isinstance(d, dict):
+ return d
+ except Exception:
+ continue
+ return {}
+
+
+roi_cutoffs = load_roi_cutoffs()
+
+
+def fig_html(rel: str, *, max_width: str = "88%") -> HTML | str:
+ import base64
+
+ p = base / rel
+ if not p.exists():
+ return HTML(
+ f"Figure not found (optional): {rel}
"
+ )
+ b64 = base64.b64encode(p.read_bytes()).decode("ascii")
+ return HTML(
+ f''
+ f'

'
+ f"
"
+ )
+
+
+def display_table(df: pd.DataFrame) -> None:
+ """Render table without row index for consistent report UX."""
+ display(HTML(df.to_html(index=False, border=0)))
+
+
+def display_kv_table(d: dict) -> None:
+ """Render dict as key/value table without row index."""
+ kv = pd.DataFrame(
+ [{"Metric": str(k), "Value": v} for k, v in d.items()],
+ columns=["Metric", "Value"],
+ )
+ display_table(kv)
+
+
+def _status_theme(status: str) -> tuple[str, str, str]:
+ """Return (container_css, badge_css, label) for report note styling."""
+ s = str(status).strip().upper()
+ if s in {"FAIL", "CRITICAL"}:
+ return (
+ "background:#FEF2F2;border:1px solid #FECACA;color:#7F1D1D;",
+ "background:#DC2626;color:#FFFFFF;",
+ s,
+ )
+ if s in {"WARN", "WARNING"}:
+ return (
+ "background:#FFF7ED;border:1px solid #FED7AA;color:#7C2D12;",
+ "background:#EA580C;color:#FFFFFF;",
+ "WARN",
+ )
+ if s in {"PASS", "GOOD"}:
+ return (
+ "background:#F0FDF4;border:1px solid #BBF7D0;color:#14532D;",
+ "background:#16A34A;color:#FFFFFF;",
+ "PASS",
+ )
+ return (
+ "background:#F8FAFC;border:1px solid #E2E8F0;color:#334155;",
+ "background:#64748B;color:#FFFFFF;",
+ s if s else "N/A",
+ )
+
+
+def _status_cell_css(status: str) -> str:
+ """Cell-level status color styling for PASS/WARN/FAIL-like values."""
+ s = str(status).strip().upper()
+ if s in {"FAIL", "CRITICAL"}:
+ return "background-color:#FEE2E2;color:#991B1B;font-weight:700;"
+ if s in {"WARN", "WARNING"}:
+ return "background-color:#FFEDD5;color:#9A3412;font-weight:700;"
+ if s in {"PASS", "GOOD"}:
+ return "background-color:#DCFCE7;color:#166534;font-weight:700;"
+ return ""
+
+
+def _intensity_display_status(raw: str) -> str:
+ """Map intensity_quality verdict (pass/warn/fail/not_available) to PASS/WARN/FAIL/N/A.
+
+ Also accepts the legacy good/warning/critical values for backwards
+ compatibility with pre-v5 JSON outputs.
+ """
+ if not raw:
+ return "—"
+ sl = str(raw).strip().lower()
+ if sl == "pass" or sl == "good":
+ return "PASS"
+ if sl == "warn" or sl == "warning":
+ return "WARN"
+ if sl == "fail" or sl == "critical":
+ return "FAIL"
+ if sl == "not_available":
+ return "N/A"
+ return str(raw).upper()
+
+
+def metric_cell_with_caveat(name: str, status: str, body_md: str) -> str:
+ """Wrap a metric label in an inline when status is WARN/FAIL.
+
+ The icon is a coloured pill with a chevron (▾) to make it visually
+ obvious that the row is clickable; clicking expands a coloured panel
+ with the explanation. Sample-specific notes that would otherwise sit
+ below the table are anchored to the row that triggered them.
+ """
+ import re as _re
+ s = str(status).strip().upper()
+ if s not in ("WARN", "WARNING", "FAIL", "CRITICAL"):
+ return name
+ is_fail = s in ("FAIL", "CRITICAL")
+ icon = "✕" if is_fail else "⚠"
+ fg = "#7F1D1D" if is_fail else "#9A3412"
+ bg = "#FEE2E2" if is_fail else "#FFF7ED"
+ border = "#DC2626" if is_fail else "#F97316"
+ chip_bg = "#FCA5A5" if is_fail else "#FED7AA"
+ body_html = _re.sub(r"\*\*(.+?)\*\*", r"\1", body_md)
+ chip = (
+ f""
+ f"{icon}▾"
+ f""
+ )
+ return (
+ f""
+ f""
+ f"{name}{chip}"
+ f"
"
+ f""
+ f"{body_html}"
+ f"
"
+ )
+
+
+_SUMMARY_ROW_CSS = (
+ "border-top:2px solid #94A3B8; background:#F1F5F9; font-weight:600;"
+)
+
+
+def _summary_row_styles(row) -> list[str]:
+ """Apply a 'summary row' visual treatment (top divider, tinted bg, bold)
+ to rows whose first column value contains 'Overall' (aggregated rows in
+ [Section 3.4](#sec-3-4) SNR). The 'Tier ' trigger was removed under v5 Phase 8 — the
+ former 8.A tier-divider rows no longer exist now that 8.A split into
+ [Section 4.1](#sec-4-1)/4.2/4.3 separate tables."""
+ label = str(row.iloc[0]) if len(row) else ""
+ if "Overall" in label:
+ return [_SUMMARY_ROW_CSS] * len(row)
+ return [""] * len(row)
+
+
+def display_status_table(df: pd.DataFrame, status_cols: list[str], *, narrow_cols: dict | None = None) -> None:
+ """Render DataFrame without index and with status-colored cells.
+
+ Rows whose first-column value contains 'Overall' get a summary-row visual
+ treatment (top divider line + tinted background + bolder text) so readers
+ can tell aggregated rows apart from per-component rows at a glance.
+
+ narrow_cols (optional) — mapping {column_name: css_width} to apply an
+ explicit width constraint to specific columns. The default cell style is
+ already wrap-friendly (`white-space: normal`, `vertical-align: top`,
+ `overflow-wrap: anywhere`) — `narrow_cols` adds explicit `width` /
+ `max-width` so a column doesn't grow beyond the budget allocated to it."""
+ sty = (
+ df.style
+ .set_table_styles(
+ [
+ {"selector": "th", "props": "text-align:left; vertical-align:top; background-color:#F8FAFC; white-space:normal;"},
+ # Wrap-friendly default (matches the at-a-glance scorecard pattern):
+ # cells wrap on whitespace + break long tokens (e.g.
+ # `low_texture_cell_warn/fail`) at any character so they don't
+ # force horizontal scroll. Previously `white-space:nowrap` here
+ # forced side-scroll in long-content tables (8.A, SNR Summary).
+ {"selector": "td", "props": "text-align:left; vertical-align:top; white-space:normal; line-height:1.5; overflow-wrap:anywhere; word-break:break-word;"},
+ {"selector": "td details[open]", "props": "white-space:normal;"},
+ {"selector": "td details[open] > div", "props": "white-space:normal;"},
+ ]
+ )
+ .hide(axis="index")
+ )
+ for col in status_cols:
+ if col in df.columns:
+ styles = [_status_cell_css(v) for v in df[col]]
+ sty = sty.apply(lambda _col, s=styles: s, subset=[col])
+ sty = sty.apply(_summary_row_styles, axis=1)
+ if narrow_cols:
+ for _col, _width in narrow_cols.items():
+ if _col in df.columns:
+ sty = sty.set_properties(
+ subset=[_col],
+ **{
+ "width": _width,
+ "max-width": _width,
+ "white-space": "normal",
+ # Force long backticked identifiers (e.g. `low_texture_cell_warn/fail`)
+ # to break mid-token rather than overflow into the next column.
+ "overflow-wrap": "anywhere",
+ "word-break": "break-word",
+ },
+ )
+ display(HTML(sty.to_html()))
+
+
+def render_status_note(title: str, status: str, subtitle: str = "") -> None:
+ """Render a compact status note with themed background + badge."""
+ box_css, badge_css, label = _status_theme(status)
+ sub_html = (
+ f"{subtitle}
"
+ if subtitle
+ else ""
+ )
+ display(
+ HTML(
+ f""
+ f"
"
+ f"{title}"
+ f"{label}"
+ f"
"
+ f"{sub_html}"
+ f"
"
+ )
+ )
+```
+
+```{python sample-id-banner}
+#| echo: false
+
+# Resolve sample id (prefer parameter; fall back to INDIR parent dir name).
+_resolved_out = base.resolve()
+_sample_name = str(SAMPLE_NAME).strip() if SAMPLE_NAME else ""
+if not _sample_name:
+ _sample_name = (_resolved_out.parent.name
+ if _resolved_out.name == "image_qc"
+ else _resolved_out.name)
+
+display(HTML(
+ ""
+ f"Sample ID: {_sample_name}"
+ "
"
+))
+```
+
+## 1. At-a-glance Sample Health {#sec-1}
+
+
+::: {.callout-note collapse="true" title="Metric explanation"}
+QC results consolidated at sample level; each reflects the worst sub-verdict in the corresponding subsection.
+
+- **Morphology** — edge / hole burden, usable tissue, contiguous low-quality zones (see [Section 2.1](#sec-2-1)).
+- **Focus** — Tile grid summary, tile focus and blurry (GMM) classification (see [Section 3.2](#sec-3-2)).
+- **Stain intensity** — per-channel mean signal vs the intensity floor (see [Section 3.3](#sec-3-3)).
+- **Signal-to-noise** — image- and transcript-derived signal-to-noise across tiles (see [Section 3.4](#sec-3-4)).
+- **Cell quality** — per-cell focus / blurriness and nuclear texture metrics (see [Section 4](#sec-4), when segmentation is available).
+:::
+
+::: {.callout-tip collapse="true" title="Interpretation help"}
+**Status logic** — each row is the *worst* PASS / WARN / FAIL across its underlying components. Components that aren't available (typically optional dependencies or missing inputs) display as N/A and don't affect the consolidated status. The scorecard is a triage tool — always cross-check with the relevant section before acting.
+
+- An all-PASS scorecard indicates no flagged metrics; still review the per-section figures and tissue-specific caveats before treating the slide as passed.
+- One or two WARN/FAIL rows: consult the linked section to determine whether the failure reflects sample quality, which warrants action, or tissue context, which can be noted without further action.
+- Many WARN/FAIL rows: likely a systemic issue (registration, mounting, optical setup); review pipeline logs and inputs before re-running downstream steps.
+- A FAIL on Morphology with a WARN on Intensity but a PASS on Focus is often a tissue / sample issue rather than imaging — check tissue-type caveats in [Section 2.1](#sec-2-1) first. (Intensity is advisory and never FAILs on its own.)
+
+A FAIL status doesn't always mean a failed experiment — many verdicts are tissue-context dependent. For example, lung tissue naturally produces a high hole-area burden and lower focus scores around airways and bronchioles, which reflect biological structure (lumens, sparse parenchyma) rather than imaging defects. Always cross-check failing rows against the tissue-specific caveats in the linked section before treating the slide as bad.
+:::
+
+```{python at-a-glance-scorecard}
+#| echo: false
+
+
+iq = intensity_stats if intensity_stats else roi_metrics.get("intensity_quality") or {}
+rows: list[dict[str, str]] = []
+# Per-category override for the Category-column expandable "Why" note. When a
+# category sets an entry, the caveat shows that (e.g. only the WARN/FAIL
+# drivers) instead of the full Detail column. The Detail column itself always
+# shows the complete component list.
+_caveat_overrides: dict[str, str] = {}
+
+def _worst_status(labels: list[str]) -> str:
+ rank = {"PASS": 0, "WARN": 1, "FAIL": 2}
+ vals = [str(x).upper() for x in labels if str(x).upper() in rank]
+ if not vals:
+ return "N/A"
+ return max(vals, key=lambda u: rank[u])
+
+def _component_display_verdict(payload: dict) -> str:
+ """Map optional-dependency skips to N/A instead of FAIL."""
+ if not isinstance(payload, dict):
+ return "N/A"
+ reason = str(payload.get("reason", "") or payload.get("moran_note", "")).lower()
+ if "modulenotfounderror" in reason or "no module named" in reason:
+ return "N/A"
+ return str(payload.get("verdict", "")).upper()
+
+# Detect tissue mask failure once — reused by Intensity and Focus rows
+_tm_aag = roi_metrics.get("tissue_mask_qc") if isinstance(roi_metrics.get("tissue_mask_qc"), dict) else {}
+_mask_ok_aag = _tm_aag.get("tissue_mask_generated", True)
+_rit_aag = _tm_aag.get("rois_in_tissue", -1)
+_mask_ok_aag = _mask_ok_aag and (isinstance(_rit_aag, (int, float)) and _rit_aag > 0)
+
+# Spatial morphology category
+morph_statuses = []
+morph_parts = []
+tm = roi_metrics.get("tissue_mask_qc", {})
+if isinstance(tm, dict):
+ s = str(tm.get("status", "")).upper()
+ if s in {"PASS", "WARN", "FAIL"}:
+ morph_statuses.append(s)
+ rin, tr = tm.get("rois_in_tissue"), tm.get("total_rois")
+ if isinstance(rin, (int, float)) and isinstance(tr, (int, float)) and float(tr) > 0:
+ morph_parts.append(f"tissue tiles={int(rin):,}/{int(tr):,} ({100.0*float(rin)/float(tr):.1f}%)")
+
+# Aggregate the §2.1 morphology sub-verdicts into this category, so the
+# at-a-glance reflects "the worst sub-verdict in the subsection" (its stated
+# contract) rather than only the tissue-mask status (2026-06-23, fixes the
+# "§2.1 shows FAIL but §1 Morphology shows PASS" mismatch). Self-contained
+# because the §2.1 helpers/cutoffs live in a later chunk — KEEP THE THRESHOLDS
+# AND METRIC KEYS IN SYNC WITH §2.1. Tissue-coverage, edge and hole are WARN-only
+# (geometry/biology-confounded); usable-tissue is now WARN-only too (2026-06-26), so only
+# cluster-zone (localized physical artefact) can FAIL.
+_mc = (roi_cutoffs.get("image_qc") or {}).get("morphology") or {}
+_sc = (roi_cutoffs.get("image_qc") or {}).get("slide_level") or {}
+_mq = roi_metrics.get("morphology") or {}
+
+def _m_find(*names):
+ for _src in (_mq, roi_metrics):
+ if isinstance(_src, dict):
+ for _n in names:
+ _v = _src.get(_n)
+ if isinstance(_v, (int, float)) and np.isfinite(float(_v)):
+ return float(_v)
+ return None
+
+def _m_status(val, warn, fail, higher_is_better):
+ if val is None:
+ return None
+ if higher_is_better:
+ if fail is not None and val < float(fail):
+ return "FAIL"
+ if warn is not None and val < float(warn):
+ return "WARN"
+ return "PASS"
+ if fail is not None and val > float(fail):
+ return "FAIL"
+ if warn is not None and val > float(warn):
+ return "WARN"
+ return "PASS"
+
+_cov_mean = (roi_metrics.get("tissue_coverage") or {}).get("mean")
+_cov_mean = (
+ float(_cov_mean)
+ if isinstance(_cov_mean, (int, float)) and np.isfinite(float(_cov_mean))
+ else None
+)
+_morph_subs = [
+ ("tissue coverage", _cov_mean, _mc.get("tissue_coverage_warn"), None, True), # WARN-only (geometry/biology)
+ ("edge-zone burden", _m_find("edge_zone_frac", "edge_zone_fraction", "edge_zone", "tissue_edge_fraction"), _mc.get("edge_zone_warn"), None, False),
+ ("hole-area burden", _m_find("hole_area_frac", "hole_area_fraction", "hole_area", "internal_hole_fraction"), _mc.get("hole_area_warn"), None, False),
+ ("usable tissue", _m_find("usable_tissue_frac", "usable_tissue_fraction", "usable_tissue"), _sc.get("usable_tissue_warn"), _sc.get("usable_tissue_fail"), True),
+ ("largest low-quality zone", _m_find("cluster_zone_frac", "cluster_zone_fraction", "cluster_zone_bad_fraction"), None, _sc.get("cluster_zone_fail"), False),
+]
+_morph_driver_parts = []
+for _lbl, _val, _w, _f, _hib in _morph_subs:
+ _st = _m_status(_val, _w, _f, _hib)
+ if _st in ("PASS", "WARN", "FAIL"):
+ morph_statuses.append(_st)
+ if _st in ("WARN", "FAIL") and _val is not None:
+ _morph_driver_parts.append(f"{_lbl}={100.0*_val:.1f}% ({_st})")
+
+_morph_status = _worst_status(morph_statuses)
+# Drivers-only caveat note (same pattern as Cell quality), so the "Why" doesn't
+# list PASS sub-metrics. Detail column keeps the tissue-tile context.
+if _morph_status in ("WARN", "FAIL") and _morph_driver_parts:
+ _caveat_overrides["Morphology"] = "
".join(_morph_driver_parts)
+rows.append(
+ {
+ "Category": "Morphology",
+ "Status": _morph_status,
+ "Detail": ("
".join(morph_parts) if morph_parts else "See section 2.1 for morphology QC summary."),
+ "Links": 'Section 2.1',
+ }
+)
+
+# Tile focus category — composite verdict
+# Strategy: use GMM blur % when component separation is reliable (Cohen's d >= warn);
+# fall back to median Laplacian variance + CCFS when GMM separation is poor.
+roi_focus_status = "N/A"
+roi_detail = "See section 3.2 for tile focus/blurriness metrics."
+_roi_focus_path = "" # which path determined the verdict
+
+_iq_cut = roi_cutoffs.get("image_qc") or {}
+_focus_cut_aag = _iq_cut.get("focus") or {}
+_dapi_cut_aag = ((_iq_cut.get("channels") or {}).get("DAPI") or {})
+
+# Extract component separation and thresholds
+_lap_aag = roi_metrics.get("laplacian_sharpness") or {}
+_comp_sep = _lap_aag.get("component_separation")
+_cs_warn = _focus_cut_aag.get("gmm_2d_laplacian_component_separation_warn", 1.0)
+_cs_fail = _focus_cut_aag.get("gmm_2d_laplacian_component_separation_fail", 0.5)
+_gmm_reliable = (
+ isinstance(_comp_sep, (int, float))
+ and np.isfinite(float(_comp_sep))
+ and float(_comp_sep) >= float(_cs_warn)
+)
+
+blur = roi_metrics.get("blur_gmm_1d") or roi_metrics.get("blur_gmm_2d")
+
+if not _mask_ok_aag:
+ roi_focus_status = "N/A"
+ roi_detail = "no tissue mask — focus QC not performed"
+ _roi_focus_path = "no_mask"
+elif _gmm_reliable and isinstance(blur, dict):
+ # --- Path A: GMM separation is reliable → use GMM blur % ---
+ pct_all = blur.get("pct_blurred_gmm")
+ pct_tissue = blur.get("pct_blurred_gmm_tissue_filtered")
+ pct = pct_tissue if isinstance(pct_tissue, (int, float)) else pct_all
+ fw, ff = _dapi_cut_aag.get("focus_warn"), _dapi_cut_aag.get("focus_fail")
+ if isinstance(pct, (int, float)) and isinstance(fw, (int, float)) and isinstance(ff, (int, float)):
+ if float(pct) >= 100.0 * float(ff):
+ roi_focus_status = "FAIL"
+ elif float(pct) >= 100.0 * float(fw):
+ roi_focus_status = "WARN"
+ else:
+ roi_focus_status = "PASS"
+ if isinstance(pct, (int, float)):
+ roi_detail = f"blurry (GMM)={float(pct):.1f}% (component separation d={float(_comp_sep):.1f})"
+ _roi_focus_path = "GMM"
+else:
+ # --- Path B: GMM unreliable → honest fallback (2026-06-22) ---
+ # When the 2D GMM cannot separate blur from focus (component separation
+ # below the WARN cutoff), we refuse to manufacture a FAIL from a confounded
+ # proxy. The old fallback hard-failed on the absolute Laplacian floor, which
+ # scales with brightness² and is least valid on exactly the dim/degraded
+ # slides that land here (qc_drift_analysis). New behaviour: WARN only when
+ # CCFS low-nuclear-texture is high (a real per-cell signal worth inspecting);
+ # otherwise N/A with the blur % shown for reference. Never FAIL.
+ _cm_aag = cell_metrics or {}
+ _pct_ccfs_aag = _cm_aag.get("pct_low_nuclear_texture")
+ if _pct_ccfs_aag is None and _cm_aag.get("cells_low_nuclear_texture") and _cm_aag.get("total_cells"):
+ _pct_ccfs_aag = round(100.0 * _cm_aag["cells_low_nuclear_texture"] / _cm_aag["total_cells"], 2)
+ _ccfs_warn_thr = _focus_cut_aag.get("low_texture_cell_warn", 0.10)
+ _ccfs_bad = isinstance(_pct_ccfs_aag, (int, float)) and _pct_ccfs_aag >= _ccfs_warn_thr * 100
+
+ _d_str = f"d={float(_comp_sep):.1f}" if isinstance(_comp_sep, (int, float)) and np.isfinite(float(_comp_sep)) else "d=N/A"
+ _pct_blur_fb = blur.get("pct_blurred_gmm_tissue_filtered") if isinstance(blur, dict) else None
+ if not isinstance(_pct_blur_fb, (int, float)):
+ _pct_blur_fb = blur.get("pct_blurred_gmm") if isinstance(blur, dict) else None
+ _blur_str = f"blurry (GMM)={float(_pct_blur_fb):.1f}% (informational)" if isinstance(_pct_blur_fb, (int, float)) else "blur % unavailable"
+
+ if _ccfs_bad:
+ roi_focus_status = "WARN"
+ _ccfs_str = f"{_pct_ccfs_aag:.1f}%" if isinstance(_pct_ccfs_aag, (int, float)) else "N/A"
+ roi_detail = f"GMM unreliable ({_d_str}); low nuclear texture={_ccfs_str} — inspect focus manually"
+ else:
+ roi_focus_status = "N/A"
+ roi_detail = f"GMM unreliable ({_d_str}); {_blur_str} — inspect focus manually"
+ _roi_focus_path = "fallback"
+
+rows.append(
+ {
+ "Category": "Focus",
+ "Status": roi_focus_status,
+ "Detail": roi_detail,
+ "Links": 'Section 3.2',
+ }
+)
+
+# Intensity category
+int_statuses = []
+int_detail_parts = []
+if not _mask_ok_aag:
+ # Tissue-mask generation failed → the sample is unusable and intensity
+ # cannot be measured. FAIL the Intensity category (2026-06-23): a mask
+ # failure is a hard QC FAIL, surfaced in both Morphology and Intensity.
+ # The mask-independent stain p99 in §3.3 explains why it failed.
+ int_statuses = ["FAIL"]
+ int_detail_parts.append(
+ "tissue mask failed — sample unusable; intensity not measurable "
+ "(see the mask-independent stain p99 in Section 3.3)"
+ )
+elif isinstance(iq, dict):
+ oq = iq.get("overall_quality")
+ if oq is not None and str(oq).strip():
+ ov = _intensity_display_status(str(oq))
+ int_statuses.append(ov)
+ int_detail_parts.append(f"overall={str(oq)}")
+ dapi = iq.get("dapi") if isinstance(iq.get("dapi"), dict) else {}
+ if dapi:
+ pbc = dapi.get("pct_tissue_roi_below_critical")
+ if isinstance(pbc, (int, float)):
+ int_detail_parts.append(f"DAPI percentage of tissue tiles below intensity floor={float(pbc):.1f}%")
+rows.append(
+ {
+ "Category": "Stain intensity",
+ "Status": _worst_status(int_statuses),
+ "Detail": ("
".join(int_detail_parts) if int_detail_parts else "See section 3.3 for channel-level details."),
+ "Links": 'Section 3.3',
+ }
+)
+
+# SNR category (split into tile-level and cell/matrix tracks)
+if snr_summary and isinstance(snr_summary.get("components", {}), dict):
+ comp = snr_summary.get("components", {})
+ roi_keys = ("SNR_image_roi_quartile_db", "SNR_image_otsu", "SNR_roi_tx", "SNR_roi_neg_spatial")
+ matrix_keys = () # Plummer/SpatialQM moved to molecule QC (v5 restructure)
+ roi_vals = [_component_display_verdict(comp.get(k) or {}) for k in roi_keys if isinstance(comp.get(k), dict)]
+ mat_vals = [_component_display_verdict(comp.get(k) or {}) for k in matrix_keys if isinstance(comp.get(k), dict)]
+ rows.append(
+ {
+ "Category": "Signal-to-noise",
+ "Status": _worst_status(roi_vals),
+ "Detail": "See section 3.4 for tile/image/transcript SNR components.",
+ "Links": 'Section 3.4',
+ }
+ )
+else:
+ rows.append(
+ {
+ "Category": "Signal-to-noise",
+ "Status": "NOT_COMPUTED",
+ "Detail": "No SNR data found for this run.",
+ "Links": 'Section 3.4',
+ }
+ )
+
+# Cell-level category — composite of CCFS nuclear texture, GMM-ROI blur, and cluster outliers
+cell_status = "N/A"
+cell_detail = "See section 4 for per-cell QC summary and figures."
+if cell_metrics:
+ nc = cell_metrics.get("total_cells")
+ # §1 Cell quality reads the same per-metric advisory cutoffs that [Section 4.1](#sec-4-1)
+ # and [Section 4.2](#sec-4-2) use, so the at-a-glance pill agrees with the §4 advisory
+ # pills. PASS/WARN only (no FAIL) — matches §4.x's advisory tier.
+ _cl_cut = ((roi_cutoffs.get("image_qc") or {}).get("cell_level") or {})
+ _nuc_warn = float(_cl_cut.get("pct_low_nuclear_texture_warn", 5.0))
+ _gmm_warn = float(_cl_cut.get("pct_blurred_gmm_2d_roi_warn", 20.0))
+ _cell_verdicts = []
+ _cell_parts = []
+
+ # (a) Nuclear texture (per-cell contrast — distinct from optical blurriness)
+ _pct_ccfs = cell_metrics.get("pct_low_nuclear_texture")
+ if _pct_ccfs is None:
+ nb = cell_metrics.get("cells_low_nuclear_texture")
+ if isinstance(nb, (int, float)) and isinstance(nc, (int, float)) and float(nc) > 0:
+ _pct_ccfs = 100.0 * float(nb) / float(nc)
+ if isinstance(_pct_ccfs, (int, float)):
+ _cell_parts.append(f"low nuclear texture={_pct_ccfs:.1f}%")
+ _cell_verdicts.append("WARN" if _pct_ccfs >= _nuc_warn else "PASS")
+
+ # (b) GMM-ROI blur (tile-level classification propagated to cells)
+ _pct_gmm_roi = cell_metrics.get("pct_blurred_gmm_2d_roi")
+ if _pct_gmm_roi is None:
+ _n_gmm_roi = cell_metrics.get("cells_blurred_gmm_2d_roi")
+ if isinstance(_n_gmm_roi, (int, float)) and isinstance(nc, (int, float)) and float(nc) > 0:
+ _pct_gmm_roi = 100.0 * float(_n_gmm_roi) / float(nc)
+ if isinstance(_pct_gmm_roi, (int, float)):
+ _cell_parts.append(f"blurry cells (GMM)={_pct_gmm_roi:.1f}%")
+ _cell_verdicts.append("WARN" if _pct_gmm_roi >= _gmm_warn else "PASS")
+
+ # (c) Cluster-outlier blur — any cluster with blur >> sample average
+ _cluster_outliers = cell_metrics.get("cluster_blur_outliers")
+ if isinstance(_cluster_outliers, dict) and _cluster_outliers:
+ worst_cl = max(_cluster_outliers, key=lambda k: _cluster_outliers[k]["pct_blurred"])
+ worst_info = _cluster_outliers[worst_cl]
+ _cell_parts.append(
+ f"cluster outlier: C{worst_cl} "
+ f"({worst_info['pct_blurred']:.0f}% blurry cells (GMM), "
+ f"n={worst_info['n_cells']:,})"
+ )
+ # Outlier cluster is a WARN — spatially biased quality loss
+ _cell_verdicts.append("WARN")
+
+ cell_status = _worst_status(_cell_verdicts)
+ # Detail column keeps ALL components (full context). The Category-column
+ # expandable caveat (the "Why" note) is built from only the WARN/FAIL
+ # drivers, so it doesn't list PASS components (e.g. a blurry-cell % below
+ # its cutoff) alongside the real trigger and make them look causal.
+ cell_detail = (
+ "
".join(_cell_parts)
+ if _cell_parts
+ else "See section 4 for per-cell QC summary and figures."
+ )
+ if cell_status in ("WARN", "FAIL"):
+ _drivers = [
+ _p for _p, _v in zip(_cell_parts, _cell_verdicts) if _v in ("WARN", "FAIL")
+ ]
+ if _drivers:
+ _why = "
".join(_drivers)
+ # Explain why a cell-level number can differ from the tile-level one.
+ if any("blurry cells (GMM)" in _d for _d in _drivers):
+ _why += (
+ "
Note: blurry cells (GMM) is measured over solid-tissue "
+ "cells (tile coverage ≥ 0.5) to match the tile-level 'tiles in "
+ "focus' in Section 3.2; it is per-cell rather than per-tile, so "
+ "small differences from the tile figure are expected."
+ )
+ _caveat_overrides["Cell quality"] = _why
+rows.append(
+ {
+ "Category": "Cell quality",
+ "Status": cell_status,
+ "Detail": cell_detail,
+ "Links": 'Section 4',
+ }
+)
+
+df_glance = pd.DataFrame(rows)
+
+# Anchor a sample-specific caveat to the Category cell when WARN/FAIL — the
+# Detail column already names the offending components; the expandable wraps
+# that in the same icon/colour pattern used by the other section tables.
+for _i, _r in df_glance.iterrows():
+ _s = str(_r["Status"]).strip().upper()
+ if _s not in ("WARN", "FAIL"):
+ continue
+ _cat = str(_r["Category"])
+ # Caveat shows the per-category override (e.g. only the WARN/FAIL drivers)
+ # when set; otherwise the full Detail. The Detail column is unchanged.
+ _detail = (_caveat_overrides.get(_cat) or str(_r["Detail"])).strip()
+ _links = str(_r["Links"]).strip()
+ _body_parts = [f"**{_cat}** is currently **{_s}**."]
+ if _detail and not _detail.lower().startswith("see section"):
+ _body_parts.append(f"Why: {_detail}.")
+ if _links:
+ _body_parts.append(f"Drill into the underlying components in {_links}.")
+ df_glance.at[_i, "Category"] = metric_cell_with_caveat(_cat, _s, " ".join(_body_parts))
+
+_df = df_glance.copy()
+_status = _df["Status"].astype(str).str.upper()
+def _status_css(s: str) -> str:
+ su = str(s).upper()
+ if su == "FAIL":
+ return "background-color:#FEE2E2;color:#991B1B;font-weight:700;"
+ if su == "WARN":
+ return "background-color:#FFEDD5;color:#9A3412;font-weight:700;"
+ if su == "PASS":
+ return "background-color:#DCFCE7;color:#166534;font-weight:700;"
+ return ""
+_status_style = [_status_css(s) for s in _status]
+_sty = (
+ _df.style
+ .set_table_styles(
+ [
+ {"selector": "th", "props": "text-align:left; background-color:#F8FAFC;"},
+ {"selector": "td", "props": "text-align:left; vertical-align:top;"},
+ {"selector": "td.col0, th.col_heading.level0.col0", "props": "text-align:left;"},
+ ]
+ )
+ .set_properties(subset=["Category"], **{"min-width": "12em", "text-align": "left"})
+ .set_properties(subset=["Detail"], **{"min-width": "26em"})
+ .set_properties(subset=["Links"], **{"min-width": "20em"})
+ .apply(lambda _col: _status_style, subset=["Status"])
+ .hide(axis="index")
+)
+display(HTML(_sty.to_html()))
+
+```
+
+## 2. Sample-level metrics {#sec-2}
+
+
+The tissue analysis generates whole-sample masks and identifies problematic regions that could affect data quality.
+
+```{python morphology-metric-explanation}
+#| echo: false
+
+# Pull edge/hole band thresholds from the morphology JSON so the metric
+# explanation tracks the actual values used by the pipeline. Both
+# `edge_distance_threshold_px_ds` and `hole_distance_threshold_px_ds` are
+# negative downsampled-pixel signed-distance thresholds; |value| × 8 × 0.2125
+# converts to µm of inward-band depth from the boundary. Defaults match
+# bin/image_qc.py:4443-4444 (-25.0).
+#
+# TODO (future): `_XENIUM_PX_UM = 0.2125` is hardcoded here, in
+# `bin/image_qc.py:70`, and at `qmd:1167`. The authoritative value for this
+# sample's instrument lives in the bundle's `experiment.xenium` `pixel_size`
+# field. When ready, emit `pixel_size_um` in roi_qc_metrics.json from
+# bin/image_qc.py and read it here instead of hardcoding.
+_morph = roi_metrics.get("morphology") or {}
+_DS_FACTOR = 8
+_XENIUM_PX_UM = 0.2125
+_edge_thr_px = abs(float(_morph.get("edge_distance_threshold_px_ds", -25.0)))
+_hole_thr_px = abs(float(_morph.get("hole_distance_threshold_px_ds", -25.0)))
+_edge_band_um = _edge_thr_px * _DS_FACTOR * _XENIUM_PX_UM
+_hole_band_um = _hole_thr_px * _DS_FACTOR * _XENIUM_PX_UM
+
+# Usable-tissue cutoffs read from YAML so the prose tracks any threshold
+# change without a parallel prose edit. img_qc_cut / slide_cut are initialised
+# later in the doc — read directly from roi_cutoffs here.
+_slide_cut_2_1 = (
+ ((roi_cutoffs.get("image_qc") or {}).get("slide_level") or {})
+ if isinstance(roi_cutoffs, dict)
+ else {}
+)
+_usable_warn_pct = float(_slide_cut_2_1.get("usable_tissue_warn", 0.60)) * 100
+
+display(Markdown(f"""::: {{.callout-note collapse="true" title="Metric explanation"}}
+**Summary table metrics**
+
+- **Edge-zone burden** — fraction of tissue tiles whose centroid lies within ~{_edge_band_um:.1f} µm of the outer tissue boundary. High values indicate edge-dominated samples. **Advisory only: PASS / WARN, never FAIL.** Edge fraction is highly tissue-variable, sparse or branching tissues (lung, intestine, lymph node) legitimately sit high, so a WARN here flags the sample for review rather than indicating poor quality. (WARN calibrated on the 160+ sample cohort, just above the per-tissue p90.)
+- **Hole-area burden** — fraction of tissue tiles within ~{_hole_band_um:.1f} µm of an internal hole/void; high values suggest tears, gaps, or many small voids spread through the tissue. **Advisory only: PASS / WARN, never FAIL.** Hole burden is highly tissue-variable, lumen-rich tissues (lung airways, blood vessels, glandular ducts) produce high hole-adjacency by design, so a WARN is biologically expected for them, not an artefact call. (WARN calibrated on the cohort to clear the lung tissue tail.)
+- **Usable tissue** — fraction of tissue tiles (≥ 20% tissue coverage) that are neither blurry (GMM) nor below the tile DAPI-intensity floor. Dominated by blurry-tile rate — most tissue tiles pass the intensity floor, so usable ≈ 1 − blurry fraction. **Advisory (WARN only, no FAIL):** usable is partly intensity-confounded for dim tissue, so a low value flags the sample for review rather than failing it. Thresholds: PASS ≥ {_usable_warn_pct:.0f}%, WARN < {_usable_warn_pct:.0f}%.
+- **Largest contiguous low-quality zone** — fraction of tissue tiles occupied by the largest single connected region of low-quality tiles (out-of-focus or below intensity floor), 4-neighbour connectivity. High values indicate a spatially concentrated artefact (fold, chip, shadow, coverslip defect) rather than diffuse degradation. Threshold: FAIL > 15%.
+
+**Visualisations**
+
+- **Stainings** — view of DAPI, Boundary, and IntRNA channels (the optically dense region map lives in the Masks panel).
+- **Masks** — view of the three mask products used by the pipeline: tissue extent, holes in sample, and optically dense regions.
+- **Distance to edge and holes** — two-panel map showing per-location distance to the nearest outer tissue boundary (left) and to the nearest internal void/hole (right). Both panels rendered in µm and masked to the tissue extent from all stains (DAPI, Boundary, IntRNA), the same mask shown in the Masks panel.
+:::
+"""))
+```
+
+::: {.callout-tip collapse="true" title="Interpretation help"}
+**Stainings**
+
+This is a **whole-slide overview** at low resolution; individual cells, nuclei or membrane outlines are *not* resolvable here. It shows where signal is present across the slide and how evenly it is distributed.
+
+- *DAPI:* total tissue extent and overall staining uniformity — expect signal that traces the full tissue with broadly even intensity. Watch for missing regions, large dim patches, or hard intensity gradients across the slide.
+- *Boundary:* regional membrane-stain coverage and intensity — expect signal co-located with DAPI tissue extent. A blank or near-blank panel can reflect (a) the channel missing or mis-aligned, (b) a staining failure (antibody cocktail didn't bind / wash properly), or (c) genuinely sparse target biology — Boundary targets epithelial / immune membranes and is expected to be dim in brain and other tissues lacking these cell types (see [Section 3.3](#sec-3-3) for the tissue-specific caveat).
+- *IntRNA:* slide-scale IntRNA intensity — expect coverage that broadly tracks the DAPI footprint with some biologically expected heterogeneity (white matter, adipose and necrotic regions are dim; epithelium and pancreas are bright).
+
+**Masks**
+
+A three-panel view of the labelled mask products. In each panel the **coloured regions are the labelled features**; the dark background is everything *not* in that mask.
+
+- *Tissue mask:* coloured regions are detected tissue, found using all available stains (DAPI, plus Boundary and Interior when the slide has them), so the mask reflects tissue extent. Tissue extent metrics (coverage, edge and hole burden) use this combined mask. Focus, blur and usable-tissue metrics are computed on the DAPI nuclei mask instead, because they judge nuclear sharpness and would be misled by tissue that has few nuclei. On single-stain slides both masks are the DAPI mask.
+- *Holes in sample:* coloured regions are internal voids inside the tissue mask — tears, gaps, folds, or detached patches excluded from analysis. Multiple separate holes appear as separate colours.
+- *Optically dense regions:* coloured regions are unusually bright zones (folds, edge bleed, debris, coverslip contamination) — regions of unusually high pixel intensity not attributable to tissue staining.
+
+Unexpected mask shapes — missing tissue, spurious holes, oversized artefact regions — often account for anomalous downstream values; check the masks before interpreting the metrics.
+
+**Distance to edge and holes**
+
+- Yellow / bright regions are far from edges or holes — generally reliable tissue with good signal support.
+- Dark purple regions on the **left panel** are near edges (partial tissue, sectioning artefacts, weak signal); on the **right panel** they are near holes (potential environmental effects on nearby cells). The thin black line in both panels marks the tissue mask boundary.
+- Cells within ~40–85 µm of the edge are often excluded from downstream analysis.
+- Biological holes (blood vessels, airways, glandular lumens) may be expected; technical holes (tears, folds, processing damage) should be flagged for exclusion.
+- If poor focus / intensity failures align with dark regions in either panel, this usually reflects edge effects or local structural artefacts rather than whole-slide failure.
+:::
+
+### 2.1 Morphology summary {#sec-2-1}
+
+```{python spatial-morphology-qc}
+#| echo: false
+
+img_qc_cut = (roi_cutoffs.get("image_qc") or {}) if isinstance(roi_cutoffs, dict) else {}
+morph_cut = (img_qc_cut.get("morphology") or {}) if isinstance(img_qc_cut, dict) else {}
+slide_cut = (img_qc_cut.get("slide_level") or {}) if isinstance(img_qc_cut, dict) else {}
+
+
+def _find_metric(d: dict, names: tuple[str, ...]) -> float | None:
+ if not isinstance(d, dict):
+ return None
+ for n in names:
+ v = d.get(n)
+ if isinstance(v, (int, float)) and np.isfinite(float(v)):
+ return float(v)
+ return None
+
+
+def _status_from_value(
+ value: float | None,
+ *,
+ warn: float | None = None,
+ fail: float | None = None,
+ higher_is_better: bool = True,
+) -> str:
+ if value is None or fail is None:
+ return "N/A"
+ if warn is None:
+ if higher_is_better:
+ return "FAIL" if value < float(fail) else "PASS"
+ return "FAIL" if value > float(fail) else "PASS"
+ if higher_is_better:
+ if value < float(fail):
+ return "FAIL"
+ if value < float(warn):
+ return "WARN"
+ return "PASS"
+ if value > float(fail):
+ return "FAIL"
+ if value > float(warn):
+ return "WARN"
+ return "PASS"
+
+
+def _advisory_pass_warn(
+ value: float | None,
+ warn: float | None,
+ *,
+ higher_is_better: bool = False,
+) -> str:
+ """PASS / WARN-only verdict for advisory tiers ([Section 4.1](#sec-4-1) cell-level).
+
+ Returns "N/A" if either value or warn is missing. Two-tier by design —
+ these metrics surface flags-to-investigate, not strict FAIL gates.
+ """
+ if value is None or warn is None:
+ return "N/A"
+ try:
+ v, w = float(value), float(warn)
+ except (TypeError, ValueError):
+ return "N/A"
+ if higher_is_better:
+ return "PASS" if v >= w else "WARN"
+ return "WARN" if v >= w else "PASS"
+
+
+rows = []
+
+# Tissue coverage moved to [Section 3.1](#sec-3-1) Tile grid summary (2026-05-15) — was a duplicate
+# of the [Section 3.1](#sec-3-1) "Mean tissue coverage per tile" row. The verdict + cutoffs travel
+# with it. Note: §1 At-a-glance Morphology pill is independent of this row
+# (it consumes `roi_metrics["tissue_mask_qc"]["status"]` directly).
+
+edge_zone = _find_metric(
+ roi_metrics.get("morphology", {}),
+ ("edge_zone_frac", "edge_zone_fraction", "edge_zone", "tissue_edge_fraction"),
+)
+if edge_zone is None:
+ edge_zone = _find_metric(
+ roi_metrics,
+ ("edge_zone_frac", "edge_zone_fraction", "edge_zone", "tissue_edge_fraction"),
+ )
+edge_warn = morph_cut.get("edge_zone_warn")
+# 2026-06-23: Edge-zone burden is WARN-only (no FAIL). Edge-dominance is
+# biology-driven — sparse/branching tissues (lung, intestine, lymph node)
+# legitimately have high edge fractions — so it flags for review, never fails.
+rows.append(
+ {
+ "Metric": "Edge-zone burden",
+ "Status": _advisory_pass_warn(
+ edge_zone,
+ edge_warn if isinstance(edge_warn, (int, float)) else None,
+ higher_is_better=False,
+ ),
+ "Value": f"{100.0*edge_zone:.1f}%" if edge_zone is not None else "not available",
+ "Cutoffs": (
+ f"PASS <= {100.0*float(edge_warn):.1f}%; WARN > {100.0*float(edge_warn):.1f}% (advisory, no FAIL)"
+ if isinstance(edge_warn, (int, float))
+ else "not configured"
+ ),
+ }
+)
+
+hole_area = _find_metric(
+ roi_metrics.get("morphology", {}),
+ ("hole_area_frac", "hole_area_fraction", "hole_area", "internal_hole_fraction"),
+)
+if hole_area is None:
+ hole_area = _find_metric(
+ roi_metrics,
+ ("hole_area_frac", "hole_area_fraction", "hole_area", "internal_hole_fraction"),
+ )
+hole_warn = morph_cut.get("hole_area_warn")
+# 2026-06-23: Hole-area burden is WARN-only (no FAIL). The hole mask does not yet
+# separate anatomical lumens (lung, intestine, glandular epithelia) from
+# artefactual tears, so it flags for review, never fails.
+rows.append(
+ {
+ "Metric": "Hole-area burden",
+ "Status": _advisory_pass_warn(
+ hole_area,
+ hole_warn if isinstance(hole_warn, (int, float)) else None,
+ higher_is_better=False,
+ ),
+ "Value": f"{100.0*hole_area:.1f}%" if hole_area is not None else "not available",
+ "Cutoffs": (
+ f"PASS <= {100.0*float(hole_warn):.1f}%; WARN > {100.0*float(hole_warn):.1f}% (advisory, no FAIL)"
+ if isinstance(hole_warn, (int, float))
+ else "not configured"
+ ),
+ }
+)
+
+usable_tissue = _find_metric(
+ roi_metrics.get("morphology", {}),
+ ("usable_tissue_frac", "usable_tissue_fraction", "usable_tissue"),
+)
+if usable_tissue is None:
+ usable_tissue = _find_metric(
+ roi_metrics,
+ ("usable_tissue_frac", "usable_tissue_fraction", "usable_tissue"),
+ )
+usable_warn = slide_cut.get("usable_tissue_warn")
+# usable_tissue is WARN-only (2026-06-26): no usable_tissue_fail key; row uses _advisory_pass_warn.
+
+# Build a breakdown string showing what drives the usable tissue score
+_usable_detail = ""
+if usable_tissue is not None:
+ _blur_gmm = roi_metrics.get("blur_gmm_2d") or roi_metrics.get("blur_gmm_1d") or {}
+ _pct_blur_tissue = _blur_gmm.get("pct_blurred_gmm_tissue_filtered")
+ if isinstance(_pct_blur_tissue, (int, float)):
+ _usable_detail = f" (out-of-focus {_pct_blur_tissue:.1f}%)"
+
+rows.append(
+ {
+ "Metric": "Usable tissue",
+ # 2026-06-26: WARN-only (advisory). usable is partly intensity-confounded for dim
+ # tissue and the FAIL cutoff was overfit; reintroduce FAIL only with a larger panel.
+ "Status": _advisory_pass_warn(
+ usable_tissue,
+ usable_warn if isinstance(usable_warn, (int, float)) else None,
+ higher_is_better=True,
+ ),
+ "Value": f"{100.0*usable_tissue:.1f}%{_usable_detail}" if usable_tissue is not None else "not available",
+ "Cutoffs": (
+ f"PASS >= {100.0*float(usable_warn):.1f}%; WARN < {100.0*float(usable_warn):.1f}% (advisory, no FAIL)"
+ if isinstance(usable_warn, (int, float))
+ else "not configured"
+ ),
+ }
+)
+
+cluster_bad = _find_metric(
+ roi_metrics.get("morphology", {}),
+ ("cluster_zone_frac", "cluster_zone_fraction", "cluster_zone_bad_fraction"),
+)
+if cluster_bad is None:
+ cluster_bad = _find_metric(
+ roi_metrics,
+ ("cluster_zone_frac", "cluster_zone_fraction", "cluster_zone_bad_fraction"),
+ )
+cluster_fail = slide_cut.get("cluster_zone_fail")
+rows.append(
+ {
+ "Metric": "Largest contiguous low-quality zone",
+ "Status": _status_from_value(
+ cluster_bad,
+ warn=None,
+ fail=cluster_fail if isinstance(cluster_fail, (int, float)) else None,
+ higher_is_better=False,
+ ),
+ "Value": f"{100.0*cluster_bad:.1f}%" if cluster_bad is not None else "not available",
+ "Cutoffs": (
+ f"PASS <= {100.0*float(cluster_fail):.1f}%; FAIL > {100.0*float(cluster_fail):.1f}%"
+ if isinstance(cluster_fail, (int, float))
+ else "not configured"
+ ),
+ }
+)
+
+df_morph = pd.DataFrame(rows)
+
+# Usable tissue: collapsible caveat anchored to its own row when WARN/FAIL
+_usable_status = df_morph.loc[df_morph["Metric"] == "Usable tissue", "Status"].values
+if len(_usable_status) > 0 and str(_usable_status[0]).upper() in ("WARN", "FAIL"):
+ _blur_gmm = roi_metrics.get("blur_gmm_2d") or roi_metrics.get("blur_gmm_1d") or {}
+ _pct_blur_tissue = _blur_gmm.get("pct_blurred_gmm_tissue_filtered")
+ _blur_note = ""
+ if isinstance(_pct_blur_tissue, (int, float)) and isinstance(usable_tissue, (int, float)):
+ _combined_fail_pct = (1.0 - float(usable_tissue)) * 100.0
+ if _combined_fail_pct > 30:
+ _blur_note = (
+ f" In this sample, **{_combined_fail_pct:.1f}%** of tissue tiles failed the "
+ f"combined blurriness-or-intensity gate; **{_pct_blur_tissue:.1f}%** of tissue tiles are "
+ "GMM-blurry (the primary driver). The remaining failing tiles are below the intensity floor."
+ )
+ elif isinstance(_pct_blur_tissue, (int, float)) and _pct_blur_tissue > 30:
+ # Fallback when usable_tissue isn't numeric — surface the GMM rate only.
+ _blur_note = (
+ f" In this sample, **{_pct_blur_tissue:.1f}%** of tissue tiles were classified "
+ "as blurry by the GMM."
+ )
+ _usable_body = (
+ "**Usable tissue** is computed as the fraction of DAPI tissue tiles (those at least "
+ "20% covered by the DAPI nuclei mask) that are neither blurry (GMM) nor below the "
+ "intensity floor. It uses the DAPI mask, not the multi-stain extent mask, because it "
+ "judges nuclear focus quality. "
+ "In most samples, **out-of-focus tiles** dominate — the intensity gate "
+ "rarely excludes additional tiles beyond the blurriness classification."
+ f"{_blur_note} "
+ "See [Section 3.2](#sec-3-2) for spatial blurriness distribution and [Section 4](#sec-4) for cell-level blurriness assessment."
+ )
+ df_morph.loc[df_morph["Metric"] == "Usable tissue", "Metric"] = metric_cell_with_caveat(
+ "Usable tissue", str(_usable_status[0]), _usable_body
+ )
+
+# Edge-zone burden: tissue-type caveat anchored to its row
+_edge_status = df_morph.loc[df_morph["Metric"] == "Edge-zone burden", "Status"].values
+if len(_edge_status) > 0 and str(_edge_status[0]).upper() in ("WARN", "FAIL"):
+ _edge_body = (
+ "**Tissue-type caveat — Edge-zone burden is elevated.** This is often tissue-specific "
+ "rather than a quality problem. Sparse or branching tissues (lung alveoli, intestinal villi, "
+ "lymph node sinuses) have proportionally more internal and external edges than compact "
+ "tissues (liver, brain cortex). High edge-zone burden in these contexts does not indicate "
+ "poor sample quality."
+ )
+ df_morph.loc[df_morph["Metric"] == "Edge-zone burden", "Metric"] = metric_cell_with_caveat(
+ "Edge-zone burden", str(_edge_status[0]), _edge_body
+ )
+
+# Hole-area burden: tissue-type caveat anchored to its row
+_hole_status = df_morph.loc[df_morph["Metric"] == "Hole-area burden", "Status"].values
+if len(_hole_status) > 0 and str(_hole_status[0]).upper() in ("WARN", "FAIL"):
+ _hole_body = (
+ "**Tissue-type caveat — Hole-area burden is elevated.** Distinguish biological lumens "
+ "(airways, blood vessels, glandular ducts) from technical artefacts (tissue tears, "
+ "detachment, processing damage). Lumen-rich tissues (lung, kidney, intestine) routinely "
+ "score high here without any quality concern."
+ )
+ df_morph.loc[df_morph["Metric"] == "Hole-area burden", "Metric"] = metric_cell_with_caveat(
+ "Hole-area burden", str(_hole_status[0]), _hole_body
+ )
+
+display_status_table(df_morph, ["Status"])
+```
+
+### 2.2 Stainings {#sec-2-2}
+
+A compact view of the morphology channels used for quality assessment (DAPI, Boundary, IntRNA) plus optically dense region map.
+
+```{python fig-morphology-overview}
+#| echo: false
+display(fig_html("figures/morphology_overview.png", max_width="95%"))
+```
+
+### 2.3 Masks {#sec-2-3}
+
+Shows the mask products used by the pipeline: tissue mask, holes mask, and optically dense region map.
+
+```{python fig-imageqc-masks}
+#| echo: false
+if _mask_ok_aag:
+ display(fig_html("figures/imageqc_masks.png", max_width="95%"))
+else:
+ _mask_lines = [
+ "> **Tissue mask was not generated** for this sample. The following mask products are unavailable:\n",
+ "> - **Tissue mask:** not generated — automatic tissue segmentation could not find a foreground population",
+ "> - **Hole mask:** not generated (depends on tissue mask)",
+ "> - **Artefact / optically dense mask:** may still be available but cannot be interpreted without tissue context",
+ ">\n> This typically occurs with tissue types that have very low or diffuse DAPI signal "
+ "(e.g. skin, adipose, decalcified bone). Consider lowering the tile intensity floor "
+ "in the thresholds configuration if tissue is genuinely present but below the default gate.",
+ ]
+ display(Markdown("\n".join(_mask_lines)))
+```
+
+### 2.4 Distance to edge and holes {#sec-2-4}
+
+Two-panel map showing per-location distance to the nearest **outer tissue boundary** (left) and to the nearest **internal void/hole** (tears, gaps, folds, detached regions; right). Distances are measured over the tissue extent from all stains (DAPI, Boundary, IntRNA), so dim nuclei-sparse regions (e.g. muscle, brain) are included — the same mask shown in §2.3.
+
+```{python fig-distance-combined}
+#| echo: false
+if _mask_ok_aag:
+ display(fig_html("figures/distance_maps.png"))
+else:
+ display(Markdown(
+ "> **Not available:** tissue mask was not generated for this sample, so "
+ "distance-to-edge and distance-to-holes cannot be computed. This typically "
+ "occurs when none of the available stains (DAPI, Boundary, IntRNA) produce "
+ "a usable tissue mask."
+ ))
+```
+
+## 3. Tile-level metrics {#sec-3}
+
+### 3.1 Tile grid summary {#sec-3-1}
+
+This section assesses image quality at the tile level: a regular grid of square tiles laid over the capture area, where each tile receives focus score, stain intensity and signal-to-noise ratio (SNR) scores independently of cell segmentation.
+
+::: {.callout-note collapse="true" title="Metric explanation"}
+**Tile grid parameters**
+
+- **Tile size (px / µm)** — square tile edge length in pixels, plus the µm equivalent (the Xenium morphology base resolution is 0.2125 µm/pixel).
+- **Total tiles** — count of tiles laid over the capture area before any tissue masking.
+
+**Tissue coverage statistics**
+
+- **Tissue tiles** — count of tiles overlapping the tissue mask. Focus and blur metrics are computed on the DAPI nuclei tiles (those at least 20% covered by DAPI tissue); tissue extent (coverage below) uses all available stains.
+- **Tissue mask generation status** — PASS or FAIL from the mask-generation pipeline step. FAIL means no tissue mask was produced and downstream tissue-filtered metrics are unavailable.
+- **Mean tissue coverage per tile** — average fractional coverage across all tiles, continuous in [0, 1], computed from all available stains (DAPI plus Boundary and Interior when present), so it reflects true tissue extent including nuclei-sparse tissue. **Advisory (WARN only, no FAIL):** low values indicate small / partial or sparse sections, which is tissue geometry / biology rather than a data-quality failure (a genuinely empty slide is caught by the tissue-mask gate). A WARN carries an expandable biology caveat.
+
+:::
+
+::: {.callout-tip collapse="true" title="Interpretation help"}
+**Tile grid sanity checks**
+
+- Tile size much smaller than expected cell diameter → metrics may be noisy at the tile level; cell-level metrics in [Section 4](#sec-4) carry more weight.
+- Mean tissue coverage per tile gives a rough sense of how much of the capture area is empty / off-sample — low values indicate small or partial tissue sections relative to the capture window.
+
+:::
+
+#### Tile grid parameters
+
+```{python roi-grid-parameters}
+#| echo: false
+
+# Xenium morphology base pixel size — see XENIUM_PIXEL_SIZE_UM in bin/image_qc.py
+_XENIUM_PIXEL_SIZE_UM = 0.2125
+
+rows = []
+if "roi_size_pixels" in roi_metrics:
+ _roi_px = roi_metrics["roi_size_pixels"]
+ rows.append({"Metric": "Tile size (px)", "Value": _roi_px})
+ if isinstance(_roi_px, (int, float)):
+ rows.append(
+ {"Metric": "Tile size (µm)", "Value": f"{float(_roi_px) * _XENIUM_PIXEL_SIZE_UM:.1f}"}
+ )
+if "total_rois" in roi_metrics:
+ rows.append({"Metric": "Total tiles", "Value": f"{roi_metrics['total_rois']:,}"})
+
+df_grid = pd.DataFrame(rows, columns=["Metric", "Value"])
+display_table(df_grid)
+```
+
+#### Tissue coverage statistics
+
+```{python roi-tissue-coverage-stats}
+#| echo: false
+
+rows = []
+total = roi_metrics.get("total_rois")
+in_tissue = roi_metrics.get("rois_in_tissue")
+tissue_cov = (roi_metrics.get("tissue_coverage") or {}).get("mean")
+tm = roi_metrics.get("tissue_mask_qc") if isinstance(roi_metrics.get("tissue_mask_qc"), dict) else {}
+
+# Tissue-coverage verdict moved here from [Section 2.1](#sec-2-1) (2026-05-15). The Mean tissue
+# coverage per tile row carries the PASS/WARN/FAIL it used to carry in [Section 2.1](#sec-2-1);
+# informational rows display "—" in Status / Cutoffs.
+tissue_cov_warn = morph_cut.get("tissue_coverage_warn")
+# 2026-06-24: Mean tissue coverage is WARN-only (no FAIL). Low coverage is
+# tissue geometry / biology (small or sparse sections), not a data-quality
+# failure; the genuinely-empty case is caught by the tissue-mask gate. See the
+# biology-driven caveat attached to the row below.
+_tc_cutoffs = (
+ f"PASS >= {100.0*float(tissue_cov_warn):.1f}%; "
+ f"WARN < {100.0*float(tissue_cov_warn):.1f}% (advisory, no FAIL)"
+ if isinstance(tissue_cov_warn, (int, float))
+ else "not configured"
+)
+
+if isinstance(in_tissue, (int, float)):
+ rows.append({
+ "Metric": "Tissue tiles",
+ "Status": "—",
+ "Value": f"{int(in_tissue):,}",
+ "Cutoffs": "—",
+ })
+# Fraction of tiles overlapping tissue (≥ 50%) row dropped 2026-05-15:
+# decided "Mean tissue coverage per tile" is more interpretable for the
+# biology-facing audience (continuous % of capture-area tissue) than the
+# binary tile-count fraction. The mean carries the PASS/WARN/FAIL verdict;
+# the reader can still compute (Tissue tiles / Total tiles) from the
+# values shown above and in the Tile grid parameters table if needed.
+if isinstance(tm, dict):
+ _tm_status = str(tm.get("status", "N/A")).upper()
+ rows.append({
+ "Metric": "Tissue mask generation status",
+ "Status": _tm_status if _tm_status in {"PASS", "WARN", "FAIL"} else "N/A",
+ "Value": "—",
+ # On FAIL the sample is unusable; point the reader to the
+ # mask-independent stain p99 diagnostic in §3.3 (dim vs structural cause).
+ "Cutoffs": (
+ "FAIL → sample unusable; see the mask-independent stain p99 in Section 3.3"
+ if _tm_status == "FAIL"
+ else "—"
+ ),
+ })
+if isinstance(tissue_cov, (int, float)):
+ rows.append({
+ "Metric": "Mean tissue coverage per tile",
+ "Status": _advisory_pass_warn(
+ tissue_cov,
+ tissue_cov_warn if isinstance(tissue_cov_warn, (int, float)) else None,
+ higher_is_better=True,
+ ),
+ "Value": f"{100.0*float(tissue_cov):.1f}%",
+ "Cutoffs": _tc_cutoffs,
+ })
+
+df_cov = pd.DataFrame(rows, columns=["Metric", "Status", "Value", "Cutoffs"])
+# Biology-driven caveat: on WARN, attach an expandable note explaining that low
+# coverage is usually small/sparse tissue, not a quality failure.
+_cov_st = df_cov.loc[df_cov["Metric"] == "Mean tissue coverage per tile", "Status"].values
+if len(_cov_st) and str(_cov_st[0]).upper() in ("WARN", "FAIL"):
+ _cov_body = (
+ "Low mean tissue coverage is usually **tissue geometry / biology**, not a "
+ "quality problem: small or partial sections (biopsy, needle core, TMA core) "
+ "and sparse or branching tissues (lung, intestine, lymph node, adipose) "
+ "naturally occupy little of the capture window. The data on the tissue that "
+ "is present is judged by the focus / intensity / SNR metrics over tissue "
+ "tiles; a genuinely empty slide is caught separately by the tissue-mask "
+ "generation gate. Read this as a 'small / sparse section' flag, not a "
+ "sample failure."
+ )
+ df_cov.loc[df_cov["Metric"] == "Mean tissue coverage per tile", "Metric"] = (
+ metric_cell_with_caveat(
+ "Mean tissue coverage per tile", str(_cov_st[0]), _cov_body
+ )
+ )
+_cov_status_styles = [_status_cell_css(v) for v in df_cov["Status"]]
+_sty_cov = (
+ df_cov.style
+ .set_table_styles(
+ [
+ {"selector": "th", "props": "text-align:left; background-color:#F8FAFC;"},
+ {"selector": "td", "props": "text-align:left; vertical-align:top;"},
+ ]
+ )
+ .hide(axis="index")
+ .apply(lambda _col: _cov_status_styles, subset=["Status"])
+)
+display(HTML(_sty_cov.to_html()))
+```
+
+### 3.2 Focus score {#sec-3-2}
+
+::: {.callout-note collapse="true" title="Metric explanation"}
+**Summary table metrics**
+
+- **Tiles in focus** — primary slide-level verdict. The percentage of tissue tiles classified as in-focus, where the classification comes from a two-component Gaussian Mixture Model on focus score and Laplacian variance. The GMM is trained on tissue tiles only and classifies each tile by posterior probability. Higher in-focus % is better; samples with large out-of-focus regions lose transcripts and disrupt segmentation. Brain samples often show elevated blurriness rates because lower DAPI contrast in neural nuclei mimics the dim-tissue side of the bimodality (biology-confounded — cross-check the per-channel intensity correlation in [Section 4.4.a](#sec-4-4-a) before treating brain values as a quality fail).
+- **Focus score median (tissue tiles)** — per-tile DAPI image contrast (high = sharp, in-focus tissue; low = flat or out-of-focus), summarised as the median across tissue tiles. The underlying formula is **CCFS** (Coefficient of Contrast Focus Score: variance ÷ mean of pixel intensities in a region), measuring image focus — low values indicate optical blurriness.
+- **Median Laplacian variance** (informational) — the median value of **Laplacian variance** (variance of an edge-detection filter's output: high when sharp edges exist, low when the image is smooth or blurry) across tissue tiles. Reported for calibration tracking only; it is no longer a standalone pass/fail gate. Laplacian variance scales with image brightness (roughly the square of it), so an absolute floor is unreliable on dim sections; the GMM above already uses Laplacian variance as one of its two features.
+- **Focus-Laplacian correlation (Spearman)** — rank correlation between per-tile CCFS and per-tile Laplacian — two independent sharpness measures. High values indicate the two metrics agree, giving a reliable focus call; low values indicate they disagree, so one metric may be responding to intensity gradients or saturation rather than focus. `N/A` when the slide is uniformly sharp (low CV on both metrics).
+- **2D GMM Laplacian component separation** — Cohen's *d* between the two GMM components on a `log(1 + Laplacian variance)` scale. PASS → blurry/in-focus split is meaningful; WARN → interpret cautiously; FAIL → heavy overlap, don't use blurry (GMM) % alone. Tissues with diffuse morphology (brain neuropil, adipose) often show reduced separation even when imaging is fine.
+
+**Visualisations**
+
+- **Spatial focus-score map** — per-tile focus score laid out spatially across the Xenium region.
+- **Focus – DAPI intensity dependence** — per-tile focus score vs mean DAPI brightness.
+- **Focus score distribution** — histogram of focus scores across all tiles. The histogram x-axis and the Focus-vs-DAPI-intensity scatter both use a *normalised* focus score (RobustScaler) for cross-run comparability; the spatial map uses *raw* CCFS computed on raw camera counts (16-bit ADU).
+:::
+
+::: {.callout-tip collapse="true" title="Interpretation help"}
+**Spatial focus-score map**
+
+- Relatively uniform warm colours across tissue → consistently sharp imaging.
+- Large contiguous cool regions → systematic defocus (optical tilt, stage drift, tissue fold).
+- Gradients from sharp to soft → uneven tissue thickness, mounting issues, or optical field curvature.
+- Edge-only dips are common and usually reflect partial-tissue tiles rather than genuine blurriness.
+
+**Focus - DAPI intensity dependence**
+
+- No correlation (cloud) → focus and intensity independent; image quality is not confounded by signal level.
+- Positive correlation → dim tiles are also out of focus; understained areas may coincide with physical defocus.
+- A distinct low-intensity / low-focus cluster usually marks background or partial-tissue tiles that should be excluded by the tissue gate.
+
+**Focus score distribution**
+
+- **Single tight peak** → most tiles have similar focus — uniformly sharp (or uniformly blurry); check absolute value to distinguish.
+- **Long left tail** → a minority of tiles are notably softer; check the spatial heatmap to see if they cluster (localised) or scatter (noise).
+- **Bimodal** → two distinct populations — genuine blurry/in-focus split; check the component-separation metric in the summary table.
+:::
+
+#### Summary
+
+```{python roi-focus-blur-summary}
+#| echo: false
+
+rows = []
+
+# Inputs / cutoffs
+blur = roi_metrics.get("blur_gmm_1d") or roi_metrics.get("blur_gmm_2d") or {}
+lap = roi_metrics.get("laplacian_sharpness") or {}
+dapi_cut = (((roi_cutoffs.get("image_qc") or {}).get("channels") or {}).get("DAPI") or {})
+focus_cut = ((roi_cutoffs.get("image_qc") or {}).get("focus") or {})
+
+# Check whether tissue mask was generated — if not, focus metrics are invalid
+_tissue_qc_sec6 = roi_metrics.get("tissue_mask_qc") if isinstance(roi_metrics.get("tissue_mask_qc"), dict) else {}
+_tissue_mask_ok = _tissue_qc_sec6.get("tissue_mask_generated", True)
+_rois_in_tissue_sec6 = _tissue_qc_sec6.get("rois_in_tissue", 0)
+# Also guard against edge case: mask "generated" but 0 tissue tiles
+_tissue_mask_ok = _tissue_mask_ok and (isinstance(_rois_in_tissue_sec6, (int, float)) and _rois_in_tissue_sec6 > 0)
+
+
+def _status_from_thresholds(value, warn=None, fail=None, higher_is_better=True):
+ if not isinstance(value, (int, float)) or not np.isfinite(float(value)):
+ return "N/A"
+ if not isinstance(fail, (int, float)) or not np.isfinite(float(fail)):
+ return "N/A"
+ v = float(value)
+ f = float(fail)
+ if warn is None or not isinstance(warn, (int, float)) or not np.isfinite(float(warn)):
+ if higher_is_better:
+ return "FAIL" if v < f else "PASS"
+ return "FAIL" if v > f else "PASS"
+ w = float(warn)
+ if higher_is_better:
+ if v < f:
+ return "FAIL"
+ if v < w:
+ return "WARN"
+ return "PASS"
+ if v > f:
+ return "FAIL"
+ if v > w:
+ return "WARN"
+ return "PASS"
+
+
+# 1) Tiles in focus (%) — primary status metric
+if not _tissue_mask_ok:
+ # No tissue mask → focus metrics are meaningless
+ pct_blur = np.nan
+ pct_focus = np.nan
+ status_focus = "N/A"
+ den_label = "no tissue mask"
+else:
+ pct_blur_tissue = blur.get("pct_blurred_gmm_tissue_filtered")
+ pct_blur_all = blur.get("pct_blurred_gmm")
+ if isinstance(pct_blur_tissue, (int, float)) and np.isfinite(float(pct_blur_tissue)):
+ pct_blur = float(pct_blur_tissue)
+ den_label = "tiles within the tissue mask"
+ else:
+ pct_blur = float(pct_blur_all) if isinstance(pct_blur_all, (int, float)) and np.isfinite(float(pct_blur_all)) else np.nan
+ den_label = "all tiles"
+ pct_focus = 100.0 - pct_blur if np.isfinite(pct_blur) else np.nan
+focus_warn = dapi_cut.get("focus_warn")
+focus_fail = dapi_cut.get("focus_fail")
+if _tissue_mask_ok:
+ status_focus = _status_from_thresholds(
+ pct_focus if np.isfinite(pct_focus) else None,
+ warn=(100.0 * (1.0 - float(focus_warn))) if isinstance(focus_warn, (int, float)) else None,
+ fail=(100.0 * (1.0 - float(focus_fail))) if isinstance(focus_fail, (int, float)) else None,
+ higher_is_better=True,
+ )
+rows.append(
+ {
+ "Metric": "Tiles in focus",
+ "Status": status_focus,
+ "Value": f"{pct_focus:.1f}%" if np.isfinite(pct_focus) else "not available",
+ "Cutoffs": (
+ f"PASS >= {100.0*(1.0-float(focus_warn)):.1f}%; WARN {100.0*(1.0-float(focus_fail)):.1f}-{100.0*(1.0-float(focus_warn)):.1f}%; FAIL < {100.0*(1.0-float(focus_fail)):.1f}%"
+ if isinstance(focus_warn, (int, float)) and isinstance(focus_fail, (int, float))
+ else "not configured"
+ ),
+ "Notes": f"Denominator: {den_label}",
+ }
+)
+
+# 2) Focus score median (tissue tiles)
+_fs = roi_metrics.get("focus_score") or {}
+focus_med = _fs.get("tissue_median", _fs.get("median"))
+_focus_med_warn = focus_cut.get("focus_median_warn")
+_focus_med_status = "N/A"
+if not _tissue_mask_ok:
+ focus_med = None # suppress misleading value
+elif isinstance(focus_med, (int, float)) and np.isfinite(float(focus_med)):
+ if isinstance(_focus_med_warn, (int, float)) and float(focus_med) < float(_focus_med_warn):
+ _focus_med_status = "WARN"
+ else:
+ _focus_med_status = "PASS"
+rows.append(
+ {
+ "Metric": "Focus score median (tissue tiles)",
+ "Status": _focus_med_status,
+ "Value": f"{float(focus_med):.3g}" if isinstance(focus_med, (int, float)) and np.isfinite(float(focus_med)) else "not available",
+ "Cutoffs": f"WARN < {float(_focus_med_warn):.0f}" if isinstance(_focus_med_warn, (int, float)) else "not configured",
+ "Notes": "Tissue-dependent: brain/small-nucleus tissues score lower. Interpret with the blurry-tile %.",
+ }
+)
+
+# Focus-Laplacian Spearman correlation
+lap_corr = None if not _tissue_mask_ok else lap.get("focus_lap_spearman_corr")
+lap_corr_status = "N/A" if not _tissue_mask_ok else lap.get("focus_lap_corr_status")
+corr_warn = dapi_cut.get("lap_focus_corr_warn")
+corr_fail = dapi_cut.get("lap_focus_corr_fail")
+if lap_corr_status == "uniform":
+ rows.append(
+ {
+ "Metric": "Focus-Laplacian correlation (Spearman)",
+ "Status": "N/A",
+ "Value": "Uniform — N/A",
+ "Cutoffs": "skipped (both metrics have CV < 0.1)",
+ "Notes": "Uniformly sharp slide — correlation is unstable and uninformative",
+ }
+ )
+else:
+ rows.append(
+ {
+ "Metric": "Focus-Laplacian correlation (Spearman)",
+ "Status": _status_from_thresholds(
+ float(lap_corr) if isinstance(lap_corr, (int, float)) and np.isfinite(float(lap_corr)) else None,
+ warn=float(corr_warn) if isinstance(corr_warn, (int, float)) else None,
+ fail=float(corr_fail) if isinstance(corr_fail, (int, float)) else None,
+ higher_is_better=True,
+ ),
+ "Value": f"{float(lap_corr):.3f}" if isinstance(lap_corr, (int, float)) and np.isfinite(float(lap_corr)) else "N/A",
+ "Cutoffs": (
+ f"PASS >= {float(corr_warn):.2f}; WARN {float(corr_fail):.2f}-{float(corr_warn):.2f}; FAIL < {float(corr_fail):.2f}"
+ if isinstance(corr_warn, (int, float)) and isinstance(corr_fail, (int, float))
+ else "not configured"
+ ),
+ "Notes": "Low correlation indicates the two sharpness metrics disagree",
+ }
+ )
+
+# 6) 2D GMM Laplacian component separation
+comp_sep = None if not _tissue_mask_ok else lap.get("component_separation")
+comp_sep_warn = focus_cut.get("gmm_2d_laplacian_component_separation_warn")
+comp_sep_fail = focus_cut.get("gmm_2d_laplacian_component_separation_fail")
+rows.append(
+ {
+ "Metric": "2D GMM Laplacian component separation",
+ "Status": _status_from_thresholds(
+ comp_sep,
+ warn=comp_sep_warn if isinstance(comp_sep_warn, (int, float)) else None,
+ fail=comp_sep_fail if isinstance(comp_sep_fail, (int, float)) else None,
+ higher_is_better=True,
+ ),
+ "Value": f"{float(comp_sep):.2f}" if isinstance(comp_sep, (int, float)) and np.isfinite(float(comp_sep)) else "N/A",
+ "Cutoffs": (
+ f"PASS >= {float(comp_sep_warn):.1f}; WARN {float(comp_sep_fail):.1f}-{float(comp_sep_warn):.1f}; FAIL < {float(comp_sep_fail):.1f}"
+ if isinstance(comp_sep_warn, (int, float)) and isinstance(comp_sep_fail, (int, float))
+ else "not configured"
+ ),
+ "Notes": "Cohen's d between blurry/focused populations on a log(1 + Laplacian variance) scale",
+ }
+)
+
+# 7) Boundary tiles excluded — diagnostic-only; surface the row only when it fires
+n_boundary_excluded = 0 if not _tissue_mask_ok else lap.get("n_boundary_excluded", 0)
+n_rois_used = 0 if not _tissue_mask_ok else lap.get("n_rois_used", 0)
+if isinstance(n_boundary_excluded, (int, float)) and float(n_boundary_excluded) > 0:
+ rows.append(
+ {
+ "Metric": "Boundary tiles excluded",
+ "Status": "info",
+ "Value": f"{n_boundary_excluded} excluded, {n_rois_used} used",
+ "Cutoffs": "N/A",
+ "Notes": "Edge tiles excluded from Laplacian normalisation (reflect-padding artefact)",
+ }
+ )
+
+df_focus = pd.DataFrame(rows, columns=["Metric", "Status", "Value", "Cutoffs", "Notes"])
+
+# Cross-row contradiction notes — anchored to the row whose status drives the panel
+_focus_extra_caveats: dict[str, str] = {}
+if _tissue_mask_ok and _focus_med_status == "PASS" and status_focus in ("WARN", "FAIL"):
+ _focus_extra_caveats["Tiles in focus"] = (
+ f"**Focus score median is PASS but blurry-tile fraction is {status_focus}** "
+ "— this is not contradictory. The **median** is the 50th-percentile focus score "
+ "across all tissue tiles: with "
+ f"{pct_focus:.0f}% of tiles in focus, the median falls inside the sharp population "
+ "and reports a healthy value. The **blurry-tile %** counts how many tiles the GMM placed "
+ "in the blurry component — a substantial minority can be blurry while the majority "
+ "(and therefore the median) remains sharp — the slide has "
+ "**regionally concentrated blurriness** rather than uniform degradation. Check the spatial "
+ "blurriness heatmap below to locate affected zones."
+ )
+elif _tissue_mask_ok and _focus_med_status == "WARN" and status_focus == "PASS":
+ _focus_extra_caveats["Focus score median"] = (
+ "**Blurry-tile fraction is PASS but focus score median is WARN** — the GMM classifies "
+ "few tiles as blurry, yet the overall median focus score is low. This typically means "
+ "the slide is **uniformly soft** rather than having a distinct out-of-focus population: "
+ "most tiles have similar (mediocre) sharpness, so the GMM does not find two separable "
+ "components. Check the component separation metric — "
+ "low separation (d < 1) would confirm a unimodal focus distribution. Tissue type "
+ "matters here: brain and small-nucleus tissues inherently produce lower focus scores."
+ )
+
+# Anchor sample-specific caveats to the Metric cell when Status is WARN/FAIL
+_TILES_IN_FOCUS_TISSUE_CAVEAT = (
+ "**Cross-check morphology metrics in [Section 2.1](#sec-2-1)** — Tiles in focus failures can be "
+ "tissue-context driven, not true imaging defects. Lumen-rich tissues (lung, intestine, "
+ "glandular epithelia) and cell-poor regions (white matter, adipose) naturally produce "
+ "localised low-focus zones because their tile-level CCFS is biased by tissue architecture "
+ "rather than by optical sharpness. Confirm whether the blurriness is spatially concentrated "
+ "(regional, often biology-driven) or diffuse (global, usually imaging-driven) before "
+ "treating this as a sample-level failure."
+)
+
+def _build_status_part_list(name, status, value, cutoffs, notes):
+ parts = [f"**{name}** is currently **{status}**."]
+ val = str(value).strip()
+ if val and val.lower() not in ("n/a", "not available", ""):
+ parts.append(f"Value: **{val}**.")
+ cuts = str(cutoffs).strip()
+ if cuts and cuts.lower() not in ("n/a", "not configured", ""):
+ parts.append(f"Thresholds: {cuts}.")
+ nts = str(notes).strip()
+ if nts:
+ parts.append(nts + ".")
+ return parts
+
+for _i, _r in df_focus.iterrows():
+ _s = str(_r["Status"]).strip().upper()
+ if _s not in ("WARN", "FAIL"):
+ continue
+ _name = str(_r["Metric"])
+ _base_parts = _build_status_part_list(_name, _s, _r["Value"], _r["Cutoffs"], _r["Notes"])
+ _extra = _focus_extra_caveats.get(_name)
+
+ if _name == "Tiles in focus":
+ # Three-bullet caveat: 1) status/value/thresholds, 2) tissue-context cross-check,
+ # 3) cross-row contradiction note (when present, e.g. median-PASS but blurry-tile-FAIL).
+ _bullets = [" ".join(_base_parts), _TILES_IN_FOCUS_TISSUE_CAVEAT]
+ if _extra:
+ _bullets.append(_extra)
+ _body = "" + "".join(
+ f"- {b}
" for b in _bullets
+ ) + "
"
+ df_focus.at[_i, "Metric"] = metric_cell_with_caveat(_name, _s, _body)
+ else:
+ if _extra:
+ _base_parts.append(_extra)
+ df_focus.at[_i, "Metric"] = metric_cell_with_caveat(_name, _s, " ".join(_base_parts))
+
+display_status_table(df_focus, ["Status"], narrow_cols={"Cutoffs": "16em", "Notes": "14em"})
+
+# Warning when tissue mask failed — focus metrics are invalid
+if not _tissue_mask_ok:
+ display(Markdown(
+ "> ⚠ **No tissue mask — focus score analysis is invalid.** "
+ "Tissue mask generation failed (0 tissue tiles detected), so blurry (GMM) classification, "
+ "Laplacian sharpness, and intensity QC could not distinguish tissue from background. "
+ "All focus/blurriness metrics above should be treated as **N/A**. This can occur with tissue "
+ "types that have very low or diffuse DAPI signal (e.g. skin, adipose, decalcified bone) "
+ "where the automatic tissue segmentation cannot find a foreground population. "
+ "Review the morphology overview and quality masks in [Section 2](#sec-2) to understand why mask "
+ "generation failed, and consider lowering the tile intensity floor in the thresholds "
+ "configuration if the tissue is genuinely present but below the default gate."
+ ))
+
+# Conditional warning when GMM component separation is poor
+if isinstance(comp_sep, (int, float)) and np.isfinite(float(comp_sep)):
+ _cs = float(comp_sep)
+ _cs_fail = float(comp_sep_fail) if isinstance(comp_sep_fail, (int, float)) else 0.5
+ _cs_warn = float(comp_sep_warn) if isinstance(comp_sep_warn, (int, float)) else 1.0
+ if _cs < _cs_warn:
+ _severity = "FAIL" if _cs < _cs_fail else "WARN"
+ _gmm_msg = (
+ f"> ⚠ **{'Low' if _severity == 'FAIL' else 'Moderate'} GMM component "
+ f"separation ({_severity}):** Cohen's d = {_cs:.2f} — the GMM cannot "
+ + ("reliably distinguish blurry from focused tiles. "
+ "**Do not use blurry cells (GMM) % for quality decisions on this sample.** "
+ if _severity == "FAIL" else
+ "blurry/focused populations are not well separated. "
+ "blurry cells (GMM) % should be interpreted with caution. ")
+ + "This can occur on uniformly well-focused slides (no real out-of-focus population) "
+ "or on tissues with diffuse morphology (brain, adipose) where Laplacian "
+ "variance is naturally low."
+ )
+ display(Markdown(_gmm_msg))
+
+ # --- Honest fallback slide-level note (2026-06-22) ---
+ # GMM separation is poor, so blur cannot be assessed automatically. We do
+ # NOT issue a FAIL here: the only GMM-independent tile metric was the
+ # absolute Laplacian floor, which scales with brightness² and is least
+ # valid on the dim/degraded slides that reach this branch
+ # (qc_drift_analysis). CCFS low-nuclear-texture is the one per-cell signal
+ # worth surfacing; high CCFS → REVIEW, otherwise inspect manually.
+ # lap_var_median_raw is shown for reference only, not as a verdict.
+ _cm = cell_metrics or {}
+ _pct_ccfs = _cm.get("pct_low_nuclear_texture")
+ if _pct_ccfs is None and _cm.get("cells_low_nuclear_texture") and _cm.get("total_cells"):
+ _pct_ccfs = round(100.0 * _cm["cells_low_nuclear_texture"] / _cm["total_cells"], 2)
+ _ccfs_warn_thr = focus_cut.get("low_texture_cell_warn", 0.10)
+ _ccfs_bad = isinstance(_pct_ccfs, (int, float)) and _pct_ccfs >= _ccfs_warn_thr * 100
+
+ _lap_med_raw = lap.get("lap_var_median_raw") if isinstance(lap, dict) else None
+ _lap_val = f"{float(_lap_med_raw):.1f}" if isinstance(_lap_med_raw, (int, float)) else "N/A"
+ _ccfs_val = f"{_pct_ccfs:.1f}%" if isinstance(_pct_ccfs, (int, float)) else "N/A"
+
+ if _ccfs_bad:
+ _verdict = "REVIEW"
+ _icon = "🔍"
+ _detail = (
+ "A notable fraction of cells show low nuclear contrast (CCFS). This can reflect "
+ "tissue-specific nuclear morphology (e.g. large neurons) or localised defocus. "
+ "**Review spatial CCFS maps and focus heatmaps** to determine whether low-CCFS cells "
+ "cluster in specific regions before drawing a focus conclusion."
+ )
+ else:
+ _verdict = "INSPECT"
+ _icon = "🔍"
+ _detail = (
+ "No elevated cell-level low-texture signal, but blur cannot be assessed automatically "
+ "when GMM separation is this low. **Review the spatial focus heatmaps manually** rather "
+ "than relying on the blurry (GMM) % for this sample."
+ )
+
+ display(Markdown(
+ f"> {_icon} **Slide focus verdict (GMM unreliable):** **{_verdict}**\n>\n"
+ f"> - Median Laplacian variance (informational, not gated): {_lap_val}\n"
+ f"> - Cell-level CCFS nuclear texture: **{_ccfs_val}** low texture "
+ f"(threshold = {focus_cut.get('ccfs_low_texture_threshold', 0.02)})\n>\n"
+ f"> {_detail}"
+ ))
+
+if not lap:
+ tissue_qc = roi_metrics.get("tissue_mask_qc") or {}
+ if not tissue_qc.get("tissue_mask_generated", True):
+ display(Markdown(
+ "> **Note:** Laplacian sharpness metrics are not available for this sample. "
+ "Tissue mask generation failed (0 tissue tiles detected), so there are no "
+ "tissue tiles to compute Laplacian variance on. "
+ "Laplacian sharpness requires a valid tissue mask to distinguish in-tissue "
+ "focus quality from background. Review the tissue mask in section 4 and "
+ "consider whether the sample type or staining quality prevented mask generation."
+ ))
+ else:
+ display(Markdown(
+ "> **Note:** Laplacian sharpness metrics are not available for this sample. "
+ "This may indicate the analysis was run with an older pipeline version "
+ "that did not compute Laplacian variance."
+ ))
+```
+
+#### Spatial focus-score map across the Xenium region
+
+Puts the **per-tile focus score** in **space**: each square of the grid is coloured by how sharp that patch of tissue looks, so you can see **regional** patterns (e.g. edges, folds, or a soft band across the slide).
+
+```{python fig-focus-heatmap}
+#| echo: false
+display(fig_html("figures/grid_roi_focus_heatmap.png"))
+```
+
+#### Focus score - DAPI intensity dependence
+
+Each point is one tissue-overlapping tile: **horizontal axis** is **mean DAPI brightness** in that tile, **vertical axis** is **focus score**. It shows whether **dim or bright** regions systematically differ in sharpness (e.g. out of focus vs simply understained).
+
+```{python fig-focus-vs-intensity}
+#| echo: false
+display(fig_html("figures/roi_focus_vs_intensity.png"))
+```
+
+#### Focus score distribution
+
+Histogram of normalized focus scores across **tissue-overlapping tiles**, with the **GMM 2D classification** overlaid. The spread and skew indicate whether most of the slide is similarly sharp or whether a tail of poor tiles drives concern.
+
+```{python fig-focus-distribution-tissue}
+#| echo: false
+_p_focus_tissue = base / "figures" / "roi_focus_distribution_tissue.png"
+if _p_focus_tissue.exists():
+ display(fig_html("figures/roi_focus_distribution_tissue.png"))
+else:
+ display(Markdown(
+ "*Focus distribution unavailable — tissue mask was not generated for this sample.*"
+ ))
+```
+
+### 3.3 Stain intensity {#sec-3-3}
+
+::: {.callout-note collapse="true" title="Metric explanation"}
+**Per-channel metrics** (DAPI, Boundary, IntRNA).
+
+- **Status** — intensity is **advisory (WARN only, no FAIL)**: brightness does not track data quality, especially after XOA 4.0, so a dim sample is flagged for review, never failed on intensity alone. Flagged **WARN** when the below-floor % exceeds the WARN cutoff *or* the mean tile intensity itself falls below the intensity floor.
+- **Mean intensity** — average raw signal in 16-bit counts across tissue tiles.
+- **Intensity floor** — per-channel value below which a tile is counted as "below floor". The floor is **XOA-version-specific** (XOA 4.0 images are much dimmer than 3.x), selected from the bundle's analysis software version.
+- **Percentage of tissue tiles below threshold** — fraction of tissue tiles whose mean signal falls below the per-channel intensity floor.
+- **Mask-independent stain p99 (whole grid)** — the 99th-percentile per-tile intensity per channel, computed over the **whole tile grid with no tissue mask**, so it survives a tissue-mask-generation failure (when the masked metrics above read N/A). It is a **diagnostic, not a gate**: its Status mirrors the tissue-mask flag, and the values explain *why* a mask failed — a collapsed p99 across channels means the stain was too dim to detect tissue (e.g. faint skin), while a healthy p99 means the stain was fine and the mask failed on tissue geometry (a structural detection failure). It is **not a numeric floor** — an absolute p99 does not separate mask-PASS from mask-FAIL across the cohort, so it is read for cause, not used to pass/fail.
+
+**Threshold rationale** — defaults are tuned **per channel** to tolerate biologically expected heterogeneity (acellular stroma, adipose, necrotic areas, vessel-rich regions) where many tissue-overlapping tiles can be genuinely dim.
+
+```{python intensity-default-thresholds}
+#| echo: false
+# Render the default-threshold line dynamically from intensity_stats so it
+# stays in sync with YAML cutoff changes without prose edits. Uniform cutoffs
+# across DAPI / Boundary / IntRNA (2026-05-15 unification — see YAML comments).
+# If cutoffs ever diverge again in future YAML edits, fall back to per-channel.
+def _fmt_pct(v):
+ return f"{float(v):.0f}%" if isinstance(v, (int, float)) else "?"
+_ch_stats = intensity_stats if isinstance(intensity_stats, dict) else {}
+_d_cut = _ch_stats.get("dapi", {})
+_b_cut = _ch_stats.get("boundary", {})
+_i_cut = _ch_stats.get("intrna", {})
+_w_d, _f_d = _d_cut.get("pct_warn_threshold"), _d_cut.get("pct_fail_threshold")
+_w_b, _f_b = _b_cut.get("pct_warn_threshold"), _b_cut.get("pct_fail_threshold")
+_w_i, _f_i = _i_cut.get("pct_warn_threshold"), _i_cut.get("pct_fail_threshold")
+# Intensity is WARN-only (no FAIL tier) since 2026-06-22: pct_fail_threshold is
+# None. Render warn cutoffs only; the floor itself is XOA-version-specific.
+_uniform = _w_d == _w_b == _w_i
+if _uniform:
+ display(Markdown(
+ f"Intensity is **advisory (WARN only, no FAIL)**. WARN triggers when the "
+ f"**fraction of tissue tiles below the intensity floor** reaches "
+ f"`{_fmt_pct(_w_d)}` (uniform across DAPI / Boundary / IntRNA). The intensity "
+ f"floor itself is XOA-version-specific (XOA 4.0 images are much dimmer than 3.x)."
+ ))
+else:
+ display(Markdown(
+ f"Intensity is **advisory (WARN only, no FAIL)**. WARN triggers on the "
+ f"**fraction of tissue tiles below the intensity floor**: "
+ f"**DAPI** `warn ≥ {_fmt_pct(_w_d)}`; "
+ f"**Boundary** `warn ≥ {_fmt_pct(_w_b)}`; "
+ f"**IntRNA** `warn ≥ {_fmt_pct(_w_i)}`. "
+ f"Floors are XOA-version-specific."
+ ))
+```
+
+Review against tissue context: the appropriate stringency varies significantly by tissue type and by channel, as described below.
+
+- **Boundary** is the most tissue-dependent channel and the most problematic in brain. The boundary stain cocktail (ATP1A1, E-Cadherin, CD45) targets epithelial and immune cell membranes. Brain parenchyma contains very few of these cell types across most regions — neurons, astrocytes, oligodendrocytes, and microglia do not express these markers in a way that produces clean membrane signal. Consequently, boundary stain is expected to be genuinely sparse and dim in brain tissue; this reflects biology, not assay failure. Flagging brain boundary tiles as poor quality based on standard thresholds is likely to produce false positives. Thresholds for this channel should be relaxed when processing brain samples, and QC failures here interpreted with caution.
+- **IntRNA (18S)** signal is diffuse in brain due to the complex morphology of neural cells: cytoplasmic RNA is distributed across long axonal and dendritic processes rather than concentrated in the soma. This causes 18S signal to appear fragmented and low-contrast on a per-tile basis even in well-stained tissue. Dim tiles in this channel should not be interpreted as staining failure without corroborating evidence from the other channels.
+
+:::
+
+::: {.callout-tip collapse="true" title="Interpretation help"}
+**Stain intensity histogram + heatmap**
+
+- A WARN on DAPI is the most actionable: weak DAPI undermines segmentation and propagates downstream. (Intensity is advisory only — it never hard-fails a sample.)
+- A WARN on Boundary or IntRNA in isolation is often tissue-driven; cross-check with the tissue composition you expect.
+
+**Intensity assessment figure**
+
+- Look for channels that are globally too dim, heavily saturated, or spatially inconsistent.
+- Use this figure and table to establish whether weak or uneven staining could affect downstream analysis — most notably cell segmentation.
+- Spatial patchiness (one area noticeably dimmer) often points to uneven illumination or tissue thickness — distinguish from sample-wide failures.
+:::
+
+#### Mask-independent stain p99 (mask-failure diagnostic)
+
+```{python stain-p99-diagnostic}
+#| echo: false
+# Stain p99 over the WHOLE tile grid (mask-independent, so it survives a
+# mask-generation failure when the tissue-filtered summary below is N/A).
+# Status mirrors the tissue-mask generation flag — a mask FAIL is the gate (the
+# sample is unusable and also FAILs Stain intensity in Section 1). The p99
+# values are diagnostic only: they explain WHY a mask failed (collapsed p99 =
+# dim stain, e.g. faint skin; healthy p99 = a structural mask-detection failure
+# on otherwise-bright tissue). p99 is NOT a numeric floor — an absolute p99 does
+# not separate mask-PASS from mask-FAIL across the cohort.
+_sp = roi_metrics.get("stain_percentiles_whole_grid") or {}
+_tm_p99 = roi_metrics.get("tissue_mask_qc") if isinstance(roi_metrics.get("tissue_mask_qc"), dict) else {}
+_mask_status = str(_tm_p99.get("status", "")).upper()
+
+def _fmt_p99(ch):
+ v = (_sp.get(ch) or {}).get("p99")
+ return f"{float(v):.0f}" if isinstance(v, (int, float)) else "N/A"
+
+_p99_rows = [{
+ "Metric": "Mask-independent stain p99 (whole grid)",
+ "Status": _mask_status if _mask_status in ("PASS", "WARN", "FAIL") else "N/A",
+ "p99 DAPI": _fmt_p99("dapi"),
+ "p99 Boundary": _fmt_p99("boundary"),
+ "p99 IntRNA": _fmt_p99("intrna"),
+}]
+display_status_table(pd.DataFrame(_p99_rows), status_cols=["Status"])
+if _mask_status == "FAIL":
+ display(Markdown(
+ "> ⚠ **Tissue-mask generation FAILED — sample unusable.** Intensity could "
+ "not be measured on a mask, so this sample also FAILs **Stain intensity** "
+ "in [Section 1](#sec-1). The p99 values above are mask-independent and "
+ "indicate *why*: a collapsed p99 across channels means the stain was too "
+ "dim to detect tissue (e.g. faint skin); a healthy p99 means the stain "
+ "was fine and the mask failed on tissue geometry (a structural detection "
+ "failure). The p99 is diagnostic, not a threshold."
+ ))
+```
+
+#### Summary
+
+```{python intensity-table}
+#| echo: false
+
+if intensity_stats:
+ def _fmt_num1(v):
+ return (
+ f"{float(v):.1f}"
+ if isinstance(v, (int, float)) and np.isfinite(float(v))
+ else "N/A"
+ )
+ # Display names for channels (case-sensitive user-facing labels).
+ _CHANNEL_DISPLAY = {"dapi": "DAPI", "boundary": "Boundary", "intrna": "IntRNA"}
+ ch_rows = []
+ for ch in ("dapi", "boundary", "intrna"):
+ if ch not in intensity_stats:
+ continue
+ s = intensity_stats[ch]
+ crit_t = s.get("critical_threshold")
+ pct_bc = s.get("pct_tissue_roi_below_critical")
+ pct_warn_t = s.get("pct_warn_threshold")
+ pct_fail_t = s.get("pct_fail_threshold")
+ qs_raw = str(s.get("quality_status", "")).strip().lower()
+ ch_display = _CHANNEL_DISPLAY.get(ch, ch.upper())
+ ch_label = ch_display
+ if qs_raw in ("warn", "fail", "warning", "critical"):
+ display_status = "WARN" if qs_raw in ("warn", "warning") else "FAIL"
+ _parts = [f"**{ch_display}** intensity is currently **{display_status}**."]
+ if isinstance(pct_bc, (int, float)) and np.isfinite(float(pct_bc)):
+ _parts.append(f"{float(pct_bc):.1f}% of tissue tiles are below the intensity floor.")
+ if isinstance(pct_warn_t, (int, float)):
+ _parts.append(f"WARN > {float(pct_warn_t):g}% (advisory; no FAIL tier).")
+ if ch == "dapi":
+ _parts.append(
+ "Low DAPI undermines segmentation; review the spatial intensity figure below "
+ "and the morphology mask in [Section 2](#sec-2) before treating it as a sample failure."
+ )
+ elif ch == "intrna":
+ _parts.append(
+ "IntRNA is the most tissue-dependent channel. White matter, adipose, "
+ "or necrotic tumour areas can produce many low-IntRNA tiles in otherwise valid "
+ "slides — interpret against tissue composition."
+ )
+ else:
+ _parts.append(
+ "Boundary signal can decouple from DAPI in stromal, lymphoid, multinucleated, "
+ "or mixed tissue/background tiles. Cross-check with morphology channel overview "
+ "in [Section 2.1](#sec-2-1)."
+ )
+ ch_label = metric_cell_with_caveat(ch_display, display_status, " ".join(_parts))
+ ch_rows.append(
+ {
+ "Channel": ch_label,
+ "Status": _intensity_display_status(str(s.get("quality_status", ""))),
+ "Mean intensity": _fmt_num1(s.get("mean")),
+ "Intensity floor": f"{int(float(crit_t))}" if isinstance(crit_t, (int, float)) and np.isfinite(float(crit_t)) else "N/A",
+ "Percentage of tissue tiles below threshold": (
+ f"{float(pct_bc):.1f}%"
+ if isinstance(pct_bc, (int, float)) and np.isfinite(float(pct_bc))
+ else "N/A"
+ ),
+ }
+ )
+
+ # Overall footer row — aggregates the per-channel statuses with an escalation rule
+ if ch_rows and "overall_quality" in intensity_stats:
+ overall_status = _intensity_display_status(str(intensity_stats["overall_quality"]))
+ overall_label = "Overall"
+ if overall_status in ("WARN", "FAIL"):
+ _o_parts = [
+ f"**Overall intensity quality** is **{overall_status}**.",
+ "Status logic: **WARN** if any channel is WARN, otherwise PASS. "
+ "Intensity is advisory only — there is no FAIL tier (intensity does "
+ "not track quality after XOA 4.0).",
+ ]
+ overall_label = metric_cell_with_caveat("Overall", overall_status, " ".join(_o_parts))
+ ch_rows.append(
+ {
+ "Channel": overall_label,
+ "Status": overall_status,
+ "Mean intensity": "—",
+ "Intensity floor": "—",
+ "Percentage of tissue tiles below threshold": "—",
+ }
+ )
+
+ if ch_rows:
+ _df_int = pd.DataFrame(ch_rows)
+ _sty_int = (
+ _df_int.style
+ .set_table_styles(
+ [
+ {"selector": "th", "props": "text-align:left; vertical-align:middle; background-color:#F8FAFC; white-space:normal;"},
+ {"selector": "td", "props": "text-align:left; vertical-align:middle; white-space:nowrap; line-height:1.5;"},
+ {"selector": "td details[open]", "props": "white-space:normal;"},
+ {"selector": "td details[open] > div", "props": "white-space:normal;"},
+ ]
+ )
+ .hide(axis="index")
+ )
+ if "Status" in _df_int.columns:
+ _q_styles = [_status_cell_css(v) for v in _df_int["Status"]]
+ _sty_int = _sty_int.apply(lambda _col: _q_styles, subset=["Status"])
+ # Colour % tissue tiles below threshold: green/orange/red based on per-channel YAML thresholds
+ def _pct_below_css(col):
+ styles = []
+ for idx, v in enumerate(col):
+ try:
+ val = float(str(v).rstrip("%"))
+ ch = _df_int.iloc[idx]["Channel"]
+ ch_s = intensity_stats.get(ch, {})
+ # Intensity is WARN-only (pct_fail_threshold is None): no red.
+ fail_t = ch_s.get("pct_fail_threshold")
+ warn_t = ch_s.get("pct_warn_threshold", 15.0)
+ if isinstance(fail_t, (int, float)) and val > fail_t:
+ styles.append(_status_cell_css("CRITICAL"))
+ elif val > warn_t:
+ styles.append(_status_cell_css("WARNING"))
+ else:
+ styles.append(_status_cell_css("PASS"))
+ except (ValueError, TypeError):
+ styles.append("")
+ return styles
+ if "Percentage of tissue tiles below threshold" in _df_int.columns:
+ _sty_int = _sty_int.apply(_pct_below_css, subset=["Percentage of tissue tiles below threshold"])
+ # Mark the Overall summary row visually distinct from per-channel rows
+ _sty_int = _sty_int.apply(_summary_row_styles, axis=1)
+ display(HTML(_sty_int.to_html()))
+ # Check if all channels are not_available — tissue mask failure
+ _int_all_na = all(
+ str(intensity_stats.get(ch, {}).get("quality_status", "")).strip().lower() == "not_available"
+ for ch in ("dapi", "boundary", "intrna")
+ if ch in intensity_stats
+ )
+ if _int_all_na and ch_rows:
+ _tm_int = roi_metrics.get("tissue_mask_qc") if isinstance(roi_metrics.get("tissue_mask_qc"), dict) else {}
+ _mask_gen_int = _tm_int.get("tissue_mask_generated", True)
+ _rit_int = _tm_int.get("rois_in_tissue", -1)
+ if not _mask_gen_int or (isinstance(_rit_int, (int, float)) and _rit_int == 0):
+ display(Markdown(
+ "> ⚠ **All channels are unavailable.** "
+ "Tissue mask generation failed (0 tissue tiles detected), so per-channel "
+ "intensity thresholds could not be evaluated — there are no tissue tiles "
+ "to classify as dim or adequate. This is expected for tissue types with very "
+ "low or diffuse DAPI signal (e.g. skin, adipose, decalcified bone). "
+ "Cross-check with the morphology masks in [Section 2.1](#sec-2-1)."
+ ))
+ else:
+ display(Markdown(
+ "> ⚠ **All channels are unavailable.** "
+ "Intensity QC ran but could not assess any channel. "
+ "Check the pipeline logs and the intensity assessment outputs for details."
+ ))
+else:
+ print("_No intensity_assessment.json — skip table._")
+```
+
+#### Stain intensity: histograms and spatial heatmaps
+
+Multipanel figure:
+*top row* — per-tile intensity histograms for DAPI, Boundary, and IntRNA, with dashed lines marking each channel's intensity floor;
+*bottom row* — spatial heatmaps of the same per-tile intensities laid over the slide footprint, so the reader can see *where* any dim regions concentrate.
+
+```{python fig-intensity}
+#| echo: false
+display(fig_html("figures/intensity_assessment.png", max_width="95%"))
+```
+
+
+
+
+### 3.4 Signal-to-Noise Ratio {#sec-3-4}
+
+Signal-to-noise ratio at tile resolution, computed from image-derived (DAPI contrast) and transcript-derived (decoded spots vs negative-control probes) sources. These two views are complementary: image SNR catches optical-acquisition issues, transcript SNR catches probe-decoding issues. Low SNR compromises downstream expression and neighbourhood analyses.
+
+::: {.callout-note collapse="true" title="Metric explanation"}
+*dB = decibel, a logarithmic unit for expressing ratios. The pipeline uses the amplitude convention `20·log₁₀(signal/noise)`, so +6 dB ≈ 2× signal, +20 dB ≈ 10× signal.*
+
+**Image SNR (tile quartile, dB)** — image-only contrast across the slide. All tissue tiles are sorted by their mean DAPI intensity, then split into the brightest 25% (foreground) and dimmest 25% (background). The score is `20·log₁₀(mean_top_quartile / std_bottom_quartile)`. Using quartile splits limits the influence of a few outlier tiles. Higher value = brighter tissue stands out more cleanly from dim regions.
+*Best used for:* slide-wide staining homogeneity and cross-sample comparison — it's a single global aggregate and will **not** flag localised problems (use the Negative-control spatial uniformity metric and the SNR spatial heatmap for those).
+
+**Image SNR (Otsu split)** — same dB-like scoring as above, but the foreground/background split is done *per tile* on the tile's own pixels, not across the slide. Otsu's method picks a histogram cut-point that maximises separation between dim and bright pixels in that tile, then `20·log₁₀(mean_signal_pixels / std_background_pixels)` is computed. The slide-level value is the median across tiles. Same dB thresholds as above.
+*Best used for:* per-tile contrast quality under uneven staining — adapting the split to each tile's local distribution is more tolerant of inter-tile intensity variation than the global tile-quartile metric, so it surfaces problems that are visible *within* tiles rather than just *between* them. Median-across-tiles still aggregates the slide; for rare localised artefacts use the Negative-control spatial uniformity metric below.
+
+**Transcript SNR (per tile)** — molecular SNR from decoded transcripts (not pixel intensity). Each decoded spot is assigned to a tile by its xy coordinates. Per tile, two quantities: `real / negative` (target-to-noise ratio) and `negative / total` (negative-probe burden). Slide-level uses the median across tiles. *Best used for:* catching molecular-side decoding problems independent of imaging quality — a slide can have high optical quality yet decode poorly (or the reverse); the image-based SNRs above do not capture this difference.
+
+**Negative-control spatial uniformity** — checks whether the negative-probe burden is spread evenly across the slide. Two measurements:
+- *Quadrant spread* (always computed): tiles are split into four spatial quadrants, the mean negative-probe fraction is computed in each, and a spread index summarises the disagreement across the four. High spread = noise concentrated in one region. Defaults: WARN > 0.5, FAIL > 1.0.
+- *Moran's I* (optional, when `esda` / `libpysal` are installed): spatial autocorrelation across all tiles. Detects small contiguous noise blobs that quadrant averaging would smooth over.
+*Best used for:* localised noise concentration — folds, mounting drift, washing artefacts, debris. A high score means the *average* noise level might still pass while the noise is patchy in space; it's the complement to Image SNR (tile quartile, dB), which can't see such localised effects.
+
+**Note on units** — Image SNR (tile quartile, Otsu split) is reported in dB (`20·log₁₀(mean_foreground / std_background)`; same amplitude convention as the [Section 3.4](#sec-3-4) intro). Transcript SNR is a unitless ratio (decoded target count / negative-control count per tile). Negative-control spatial uniformity is a unitless spread index across slide quadrants.
+
+**Visualisations**
+
+- **SNR spatial heatmap** — two-panel slide overview at tile resolution, scoped to tiles inside the tissue mask. *Left:* per-tile fraction of negative-control probes (`negative / total`); dark = clean, bright = high noise burden. *Right:* per-tile real-to-negative ratio on a `log₁₀` scale; dark = high SNR, bright = noise approaching signal. Both panels use fixed colorbars (left 0–40%; right 1×–1000× log) so colours mean the same thing across samples. Tiles past the cap saturate at the deepest red with an `extend` triangle. Dashed and solid lines on each colorbar mark WARN (orange) and FAIL (red) cut-points: 15% / 30% on the left, 60× / 30× on the right.
+
+ *Pseudocount note on the right panel:* the ratio is rendered as `(real+1)/(neg+1)` with an α=1 Laplace pseudocount (display-only — does not affect verdicts or JSON). Without it, tiles with `neg=0` would give `inf` and tiles with `real=0` would give `0`, both dropping out as NaN on the log axis. With it, those tiles paint at the colorbar extremes (green for "no noise detected", red for "all noise") instead of disappearing. Effect on well-populated tiles is negligible.
+:::
+
+::: {.callout-tip collapse="true" title="Interpretation help"}
+**Per-component SNR (image-derived)**
+
+- Lower **Image SNR (tile quartile, dB)** → weaker optical contrast between signal-rich and background-like regions.
+- Lower **Image SNR (Otsu split)** → poorer separation of tissue signal from low-intensity background.
+- Lower **Transcript SNR (per tile)** ratio (or higher negative fraction) → weak molecular separation in decoded transcript space.
+- Higher **Negative-control spatial uniformity** unevenness → localised technical artefacts rather than uniform background.
+
+
+**SNR spatial heatmap**
+
+- Red regions on the **left** panel (negative-control fraction) → high noise burden in those tiles.
+- Red regions on the **right** panel (real-to-negative ratio) → low signal separation.
+- Look for spatial overlap with focus / blurriness maps from [Section 3.2](#sec-3-2); correlated red zones suggest a single root cause (fold, optical issue) rather than independent failures.
+
+**Image-derived quality vs transcript-derived quality**
+
+These three panels test whether image quality predicts transcript quality at the tile level, each using a different image-quality axis. The **top and middle panels** share a y-axis (transcript SNR ratio, high = clean), so healthy tiles cluster in the upper-right quadrant — sharper optics or higher image contrast tend to coincide with stronger molecular signal. The **bottom panel** inverts the y-axis convention (negative-probe fraction, low = clean), so it is interpreted separately in the bottom-panel bullet below. For the top and middle panels, weak overall correlation has two interpretations: both axes uniformly high (expected for good data), or molecular quality varying independently of imaging (a concern); off-pattern clusters point to specific failure modes covered below.
+
+- *Optical sharpness vs transcript quality (top panel):* a positive correlation → out-of-focus regions also have poorer transcript detection; image focus is constraining the molecular readout. A weak correlation is decoupled imaging and transcript quality — check whether that's because both are uniformly high (ideal) or because molecular quality varies independently of imaging (probe panel / hybridisation, not acquisition).
+- *Image SNR vs transcript quality (middle panel):* a positive correlation → tiles with poor local foreground/background contrast also have poor transcript SNR; image acquisition (contrast, staining, exposure) is limiting downstream decoding. A weak correlation → image SNR and transcript SNR are tracking independent quality axes.
+- *Signal strength vs noise contamination (bottom panel):* the y-axis (negative-probe fraction) is bounded above by the small fraction of negative-control codewords in the panel, so most tiles cluster near zero regardless of DAPI level — a roughly flat trend is the expected null. A strong positive slope is anomalous (potential contamination scaling with cell density). A strong negative slope can arise when per-area background noise dominates dim-DAPI tiles.
+
+*The quadrant patterns below apply to the **top and middle panels** only (both with y = transcript SNR ratio). The bottom panel's interpretation is its own bullet above.*
+
+- *Upper-right cluster (high image + high transcript quality):* typical clean tissue. Most tiles fall here in a healthy sample; a sparse upper-right population indicates either poor tissue coverage or widespread low quality.
+- *Lower-left cluster (low image + low transcript quality):* tiles where both optical and molecular quality are degraded. Concentration at tissue periphery is an edge-effect; concentration in interior tissue indicates a real quality problem (folds, debris, focus drift).
+- *Upper-left — high focus but low transcript SNR:* tiles are sharp optically yet molecular noise is high. Points to probe panel design or hybridisation issues rather than acquisition; cross-check [Section 3.3](#sec-3-3) stain intensity for systematic channel underperformance.
+- *Lower-right — low focus but high transcript SNR:* unusual — out-of-focus tiles where transcripts still decode well. This usually indicates the focus metric is responding to low nuclear texture (sparse nuclei, neural tissue with diffuse chromatin) rather than genuine defocus; cross-check [Section 4.4.a](#sec-4-4-a) per-channel intensity correlation for whether DAPI is the **strongest predictor of transcript yield** in this tissue.
+- *Points coloured by blurry (GMM) classification (red = blurry, blue = in-focus):* red cluster overlapping the low-SNR region means the two QC systems agree on the same tiles being problematic. Red scattered across all SNR levels means the GMM is catching something other than what affects molecular quality (often biology, especially in neural tissue).
+- *Threshold lines:* mark WARN (orange dashed) and FAIL (red dashed) from the thresholds configuration. Most tiles should sit safely above WARN.
+:::
+
+#### Summary
+
+```{python snr-overall}
+#| echo: false
+
+if not snr_summary:
+ print("_No SNR block found (run image QC with SNR enabled, or add snr_metrics.json)._")
+else:
+ comp = snr_summary.get("components", {}) if isinstance(snr_summary.get("components", {}), dict) else {}
+ roi_keys = ("SNR_image_roi_quartile_db", "SNR_image_otsu", "SNR_roi_tx", "SNR_roi_neg_spatial")
+ matrix_keys = () # Plummer/SpatialQM moved to molecule QC (v5 restructure)
+
+ rank = {"PASS": 0, "WARN": 1, "FAIL": 2}
+ def _display_verdict(payload):
+ reason = str((payload or {}).get("reason", "") or (payload or {}).get("moran_note", "")).lower()
+ if "modulenotfounderror" in reason or "no module named" in reason:
+ return "N/A"
+ return str((payload or {}).get("verdict", "")).upper()
+ def _rollup(keys):
+ vals = []
+ for k in keys:
+ vu = _display_verdict(comp.get(k) or {}) if isinstance(comp.get(k), dict) else ""
+ if vu in rank:
+ vals.append(vu)
+ if not vals:
+ return "NOT_COMPUTED"
+ return max(vals, key=lambda x: rank[x])
+
+ roi_ov = _rollup(roi_keys)
+ matrix_ov = _rollup(matrix_keys)
+ # Rollup across the four tile-level SNR components — drives the single
+ # [Section 3.4](#sec-3-4) Overall pill in the Summary table. matrix_keys is empty
+ # post-v5 (Plummer/SpatialQM moved to molecule QC).
+ snr_ov = _rollup(roi_keys + matrix_keys)
+```
+
+```{python snr-components}
+#| echo: false
+
+# Image-derived SNR component keys (tile / image / transcript / spatial uniformity).
+_SNR_ROI_KEYS = (
+ "SNR_image_roi_quartile_db",
+ "SNR_image_otsu",
+ "SNR_roi_tx",
+ "SNR_roi_neg_spatial",
+)
+# Matrix-derived SNR component keys (Plummer + SpatialQM) moved to molecule
+# QC under the v5 restructure. Empty tuple preserved so downstream concat
+# expressions (roi_keys + matrix_keys) keep working without further edits.
+_SNR_MATRIX_KEYS = ()
+
+# Cutoff strings per [Section 3.4](#sec-3-4) metric, derived from conf/roi_image_qc_thresholds.yaml.
+# Used by the Detailed-metrics tabs to populate the Cutoffs column.
+_snr_cut = ((roi_cutoffs.get("image_qc") or {}).get("snr") or {}) if isinstance(roi_cutoffs, dict) else {}
+_img_snr_warn = float((_snr_cut.get("image_snr_db") or {}).get("warn", 15.0))
+_img_snr_fail = float((_snr_cut.get("image_snr_db") or {}).get("fail", 10.0))
+_roi_tx_t = _snr_cut.get("roi_tx") or {}
+_roi_tx_rwarn = float(_roi_tx_t.get("ratio_warn", 3.0))
+_roi_tx_rfail = float(_roi_tx_t.get("ratio_fail", 1.5))
+_roi_tx_nwarn = float(_roi_tx_t.get("neg_pct_warn", 0.15))
+_roi_tx_nfail = float(_roi_tx_t.get("neg_pct_fail", 0.30))
+_neg_spat_t = _snr_cut.get("neg_spatial") or {}
+_qs_warn = float(_neg_spat_t.get("quadrant_spread_warn", 0.5))
+_qs_fail = float(_neg_spat_t.get("quadrant_spread_fail", 1.0))
+
+_SNR_TILE_CUTOFFS = {
+ "SNR_image_roi_quartile_db": {
+ "snr_db": f"PASS ≥ {_img_snr_warn:g} dB; WARN {_img_snr_fail:g}–{_img_snr_warn:g} dB; FAIL < {_img_snr_fail:g} dB",
+ },
+ "SNR_image_otsu": {
+ "snr_db_median": f"PASS ≥ {_img_snr_warn:g} dB; WARN {_img_snr_fail:g}–{_img_snr_warn:g} dB; FAIL < {_img_snr_fail:g} dB",
+ },
+ "SNR_roi_tx": {
+ "median_roi_tx_snr_ratio": f"PASS ≥ {_roi_tx_rwarn:g}; WARN {_roi_tx_rfail:g}–{_roi_tx_rwarn:g}; FAIL < {_roi_tx_rfail:g}",
+ "median_neg_pct": f"PASS ≤ {_roi_tx_nwarn:g}; WARN {_roi_tx_nwarn:g}–{_roi_tx_nfail:g}; FAIL > {_roi_tx_nfail:g}",
+ },
+ "SNR_roi_neg_spatial": {
+ "quadrant_spread_index": f"PASS ≤ {_qs_warn:g}; WARN {_qs_warn:g}–{_qs_fail:g}; FAIL > {_qs_fail:g}",
+ },
+}
+
+# Human-readable display names for SNR component JSON IDs rendered in the [Section 3.4](#sec-3-4) Summary table.
+_SNR_DISPLAY_NAMES = {
+ "SNR_image_roi_quartile_db": "Image SNR (tile quartile, dB)",
+ "SNR_image_otsu": "Image SNR (Otsu split)",
+ "SNR_roi_tx": "Transcript SNR (per tile)",
+ "SNR_roi_neg_spatial": "Negative-control spatial uniformity",
+}
+
+# Method-code → short human-readable algorithm name.
+_SNR_METHOD_NAMES = {
+ "per_roi_otsu": "Otsu thresholding",
+ "roi_df_quartiles": "Quartile-based dB ratio",
+ "roi_tx_target_vs_neg": "Decoded target vs negative-control",
+ "quadrant_spread": "Quadrant variance + Moran's I",
+}
+
+# Method-code → one-line description of what the method computes.
+_SNR_METHOD_DESCRIPTIONS = {
+ "per_roi_otsu": "Otsu threshold separates signal from background tiles; their contrast is converted to a dB-like SNR.",
+ "roi_df_quartiles": "Compares robust upper- vs lower-quartile DAPI intensity per tile; ratio converted to a dB-like SNR.",
+ "roi_tx_target_vs_neg": "Per tile, counts decoded target transcripts vs negative-control probes; reports the median target-to-negative ratio and median negative-probe burden across tiles.",
+ "quadrant_spread": "Measures dispersion / clustering of negative-control burden across image quadrants (with optional Moran's I).",
+}
+
+
+def _info_label(name: str, body_md: str) -> str:
+ """Wrap a label in an inline with a neutral info ⓘ chip (no warning colour)."""
+ import re as _re
+ body_html = _re.sub(r"\*\*(.+?)\*\*", r"\1", body_md)
+ body_html = _re.sub(r"`([^`]+)`", r"\1", body_html)
+ chip = (
+ ""
+ "ⓘ▾"
+ ""
+ )
+ return (
+ f""
+ f""
+ f"{name}{chip}"
+ f"
"
+ f""
+ f"{body_html}"
+ f"
"
+ )
+
+
+def display_snr_detailed_metrics(sub: dict, *, show_cutoffs: bool = True, cutoff_overrides: dict | None = None) -> None:
+ """Render one SNR component's payload as a Metric / Value [/ Cutoffs] table.
+
+ Strips status, verdict, reason, and method-machinery keys from the body.
+ Pulls method metadata into an expandable on a synthetic top row, and
+ formats `thresholds_used` (when present) into the Cutoffs column on the
+ row whose key drives the verdict.
+
+ Parameters
+ ----------
+ sub : dict
+ Component payload from snr_summary["components"][...].
+ show_cutoffs : bool, optional
+ If False, omit the Cutoffs column entirely. Use for [Section 3.4](#sec-3-4)
+ components whose thresholds live in external YAML rather than the
+ payload itself, so the column would always be blank.
+ """
+ if not isinstance(sub, dict):
+ print("_Metric not available for this run._")
+ return
+
+ _SUPPRESS = {
+ "status", "verdict", "quadrant_verdict", "reason", "error",
+ "moran_note", "moran_subsample", "moran_p_pseudo", "moran_neighbors",
+ "moran_i", "moran_z", "report_note", "per_roi_table_file",
+ }
+ _METHOD_KEYS = ("method", "method_primary", "intensity_col", "map_key", "formula")
+
+ method_code = str(sub.get("method", sub.get("method_primary", "")) or "")
+ method_name = _SNR_METHOD_NAMES.get(method_code, method_code) if method_code else ""
+ method_desc = _SNR_METHOD_DESCRIPTIONS.get(method_code, "")
+
+ method_extras = [f"**{k}**: `{sub[k]}`" for k in _METHOD_KEYS if k in sub]
+
+ # Threshold formatting → Cutoffs column.
+ _thr = sub.get("thresholds_used") if isinstance(sub.get("thresholds_used"), dict) else None
+ cutoffs_target_key = None
+ cutoffs_str = ""
+ if _thr:
+ if "metric" in _thr:
+ cutoffs_target_key = _thr["metric"]
+ warn = _thr.get("warn")
+ fail = _thr.get("fail")
+ if isinstance(warn, (int, float)) and isinstance(fail, (int, float)):
+ cutoffs_str = f"PASS ≥ {warn:g}; FAIL < {fail:g}"
+ elif "pass_min" in _thr or "fail_below" in _thr:
+ cutoffs_target_key = "snr"
+ pmin = _thr.get("pass_min")
+ fbelow = _thr.get("fail_below")
+ if isinstance(pmin, (int, float)) and isinstance(fbelow, (int, float)):
+ cutoffs_str = f"PASS ≥ {pmin:g}; FAIL < {fbelow:g}"
+
+ rows = []
+
+ # Compose the Method-info body that will live in the Metric column header
+ # (instead of a synthetic top row).
+ _header_body_parts = []
+ if method_name and method_name != method_code:
+ _header_body_parts.append(f"**Method:** {method_name}")
+ elif method_code:
+ _header_body_parts.append(f"**Method:** `{method_code}`")
+ if method_desc:
+ _header_body_parts.append(method_desc)
+ if method_extras:
+ _header_body_parts.append("Internal keys: " + "; ".join(method_extras) + ".")
+ metric_header = (
+ _info_label("Metric", " ".join(_header_body_parts))
+ if _header_body_parts
+ else "Metric"
+ )
+
+ for k, v in sub.items():
+ if k in _SUPPRESS or k in _METHOD_KEYS or k == "thresholds_used":
+ continue
+ if isinstance(v, dict):
+ continue
+ if isinstance(v, list):
+ v_str = ", ".join(f"{x:.3g}" if isinstance(x, float) else str(x) for x in v[:8])
+ if len(v) > 8:
+ v_str += ", …"
+ elif isinstance(v, float):
+ v_str = f"{v:.4g}"
+ else:
+ v_str = str(v)
+ row = {"Metric": k, "Value": v_str}
+ if show_cutoffs:
+ if cutoff_overrides and k in cutoff_overrides:
+ row["Cutoffs"] = cutoff_overrides[k]
+ elif cutoffs_target_key and k == cutoffs_target_key and cutoffs_str:
+ row["Cutoffs"] = cutoffs_str
+ else:
+ row["Cutoffs"] = "—"
+ rows.append(row)
+
+ if not rows:
+ print("_No metrics to display._")
+ return
+
+ columns = ["Metric", "Value", "Cutoffs"] if show_cutoffs else ["Metric", "Value"]
+ df = pd.DataFrame(rows, columns=columns)
+
+ # Build per-row Value-cell coloring using the component-level verdict.
+ # A row is "verdict-bearing" when its Metric key is referenced by the
+ # cutoffs configuration (i.e., appears in cutoff_overrides, or matches
+ # cutoffs_target_key from thresholds_used). The component verdict
+ # (sub["verdict"]) colours those cells; others stay uncoloured.
+ _verdict = str(sub.get("verdict", "")).upper()
+ _verdict_keys = set()
+ if cutoff_overrides:
+ _verdict_keys.update(cutoff_overrides.keys())
+ if cutoffs_target_key:
+ _verdict_keys.add(cutoffs_target_key)
+ _value_cell_styles = [
+ _status_cell_css(_verdict)
+ if (_verdict in {"PASS", "WARN", "FAIL"} and r["Metric"] in _verdict_keys)
+ else ""
+ for r in rows
+ ]
+
+ sty = (
+ df.style
+ .set_table_styles([
+ {"selector": "th", "props": "text-align:left; vertical-align:top; background-color:#F8FAFC;"},
+ {"selector": "td", "props": "text-align:left; vertical-align:top;"},
+ ])
+ .hide(axis="index")
+ .format_index(
+ lambda c: metric_header if c == "Metric" else c,
+ axis=1,
+ escape=None,
+ )
+ .apply(lambda _col: _value_cell_styles, subset=["Value"])
+ )
+ display(HTML(sty.to_html()))
+
+if snr_summary and "components" in snr_summary:
+ comp = snr_summary["components"]
+ recs = []
+ for name in _SNR_ROI_KEYS + _SNR_MATRIX_KEYS:
+ payload = comp.get(name)
+ if not isinstance(payload, dict):
+ continue
+ reason = str(payload.get("reason", "") or payload.get("moran_note", ""))
+ display_verdict = _display_verdict(payload)
+ display_name = _SNR_DISPLAY_NAMES.get(name, name)
+ method_code = str(payload.get("method", payload.get("method_primary", "")))
+ if not method_code:
+ # Fallback for older JSON payloads that didn't emit a method key.
+ method_code = {
+ "SNR_roi_tx": "roi_tx_target_vs_neg",
+ }.get(name, "")
+ method_name = _SNR_METHOD_NAMES.get(method_code, method_code)
+ method_desc = _SNR_METHOD_DESCRIPTIONS.get(method_code, "")
+ if reason:
+ note_text = f"{method_desc} ({reason})" if method_desc else reason
+ else:
+ note_text = method_desc
+ metric_label = display_name
+ if str(display_verdict).upper() in ("WARN", "FAIL"):
+ _parts = [f"**{display_name}** verdict is **{display_verdict}**."]
+ if reason:
+ _parts.append(f"Reason: {reason}.")
+ _hints = {
+ "SNR_image_roi_quartile_db": (
+ "Weak optical contrast between signal-rich and background-like tiles. "
+ "Check intensity ([Section 3.3](#sec-3-3)) and focus ([Section 3.2](#sec-3-2)) before treating it as a noise problem."
+ ),
+ "SNR_image_otsu": (
+ "Otsu split between tissue signal and low-intensity background is poor. "
+ "Often co-occurs with low DAPI intensity — see [Section 3.3](#sec-3-3)."
+ ),
+ "SNR_roi_tx": (
+ "Decoded transcripts vs negative controls are close at tile level. "
+ "Inspect the SNR spatial heatmap below for whether the issue is global or patchy."
+ ),
+ "SNR_roi_neg_spatial": (
+ "Negative-control burden is spatially uneven — likely a localised technical "
+ "artefact rather than uniform background. Check tissue mask and edge zones in [Section 2](#sec-2)."
+ ),
+ }
+ if name in _hints:
+ _parts.append(_hints[name])
+ metric_label = metric_cell_with_caveat(display_name, display_verdict, " ".join(_parts))
+ recs.append(
+ {
+ "Metric": metric_label,
+ "Status": display_verdict,
+ "Method": method_name,
+ "Note": note_text,
+ }
+ )
+ if recs:
+ # Overall footer row — worst PASS/WARN/FAIL across the four tile-level components.
+ overall_label = "Overall"
+ if str(snr_ov).upper() in ("WARN", "FAIL"):
+ _o_parts = [
+ f"**SNR verdict** is **{snr_ov}**.",
+ "Aggregation: worst PASS/WARN/FAIL across **Image SNR (tile quartile, dB)**, "
+ "**Image SNR (Otsu split)**, **Transcript SNR (per tile)**, and "
+ "**Negative-control spatial uniformity**.",
+ "Cross-check against [Section 3.3](#sec-3-3) (intensity) and [Section 3.2](#sec-3-2) (focus); a tile-level "
+ "SNR failure often co-occurs with low DAPI intensity or out-of-focus zones.",
+ ]
+ overall_label = metric_cell_with_caveat("Overall", str(snr_ov), " ".join(_o_parts))
+ recs.append(
+ {
+ "Metric": overall_label,
+ "Status": str(snr_ov),
+ "Method": "—",
+ "Note": "—",
+ }
+ )
+ display_status_table(pd.DataFrame(recs), ["Status"])
+ else:
+ print("_No SNR components in JSON (check SNR enabled and inputs)._")
+
+ # Notes for specific components
+ quartile = comp.get("SNR_image_roi_quartile_db") or {}
+ if quartile.get("status") == "skipped" and quartile.get("reason") in ("too_few_tissue_rois",):
+ tissue_qc = roi_metrics.get("tissue_mask_qc") or {}
+ if not tissue_qc.get("tissue_mask_generated", True):
+ display(Markdown(
+ "> **Note:** Quartile image SNR was skipped because tissue mask generation "
+ "failed (0 tissue tiles detected). This metric requires tissue tiles "
+ "to separate foreground from background intensity. Review the tissue mask in "
+ "section 4."
+ ))
+
+ neg_spatial = comp.get("SNR_roi_neg_spatial") or {}
+ report_note = neg_spatial.get("report_note")
+ if report_note:
+ display(Markdown(f"> **Note:** {report_note}"))
+```
+
+#### SNR Spatial Heatmap
+
+Spatial distribution of transcript-level noise across the slide.
+
+**Left:** fraction of negative-control probes per tile — red regions have high noise burden.
+**Right:** real-to-negative transcript ratio per tile (log scale) — red regions have low signal separation. Dashed/solid lines on each colorbar mark WARN (orange dashed) and FAIL (red solid) thresholds.
+
+```{python snr-heatmap-fig}
+#| echo: false
+display(fig_html("figures/snr_heatmap.png"))
+```
+
+#### Image-derived quality vs transcript-derived quality (per-tile)
+
+Per-tile relationships between image-derived and transcript-derived quality metrics.
+
+**Top:** Focus score vs transcript SNR. **Middle:** Image SNR (Otsu split, per-tile dB) vs transcript SNR. **Bottom:** DAPI intensity vs negative-probe burden.
+
+
+```{python concordance-fig}
+#| echo: false
+_concordance_path = base / "figures" / "cross_section_concordance.png"
+if _concordance_path.exists():
+ display(fig_html("figures/cross_section_concordance.png", max_width="65%"))
+else:
+ display(Markdown("*Image-derived quality vs transcript-derived quality plot not available (requires both focus scores and transcript SNR).*"))
+```
+
+## 4. Cell-level metrics {#sec-4}
+
+Per-cell metrics derived from segmentation masks — rendered only when the bundle supplies segmentation. Verdicts are advisory (PASS / WARN only, no FAIL tier): candidates for review, not sample-level filtering decisions. Metric definitions and biological caveats live in the subsection callouts.
+
+```{python cell-section}
+#| echo: false
+
+if cell_metrics is None:
+ print("_No cell-level image QC metrics for this run (expected for bundle-only / no cells path)._")
+else:
+ display(Markdown("### 4.1 Cell quality scores summary {#sec-4-1}"))
+ display(Markdown(
+ "Use this table to identify cells (or whole clusters) that may need "
+ "verification or filtering before downstream analysis. Inspect flagged "
+ "cells before deciding whether to filter — some will be biologically "
+ "real, not QC failures."
+ ))
+
+ # Cutoffs read from YAML at render time (no new keys — all reuse existing).
+ # Loaded BEFORE the 📖 callout so the callout can interpolate live values.
+ _a_focus_cut = (roi_cutoffs.get("image_qc") or {}).get("focus") or {}
+ _a_cw = _a_focus_cut.get("low_texture_cell_warn", 0.10)
+ _a_cf = _a_focus_cut.get("low_texture_cell_fail", 0.25)
+ _a_n_total = cell_metrics.get("total_cells", 0)
+ _a_ccfs_thr = cell_metrics.get("ccfs_low_texture_threshold", 0.02)
+ _a_ch_cfg = ((roi_cutoffs.get("image_qc") or {}).get("channels") or {})
+ # Show the ACTUAL per-channel floor the cell-level flagging used. The
+ # cell-level code applies the XOA-version-specific floor (same as the tile
+ # level), emitted as intensity_quality..critical_threshold; fall back to
+ # the flat YAML intensity_critical only if that's unavailable. Avoids showing
+ # the bright-era 500/100/300 when a dim XOA-4.0 floor was actually applied.
+ _a_iq = (
+ intensity_stats
+ if intensity_stats
+ else (roi_metrics.get("intensity_quality") or {})
+ )
+ def _a_floor(_iq_key, _cfg_key, _default):
+ _v = (_a_iq.get(_iq_key) or {}).get("critical_threshold")
+ if isinstance(_v, (int, float)):
+ return int(_v)
+ return (_a_ch_cfg.get(_cfg_key) or {}).get("intensity_critical", _default)
+ _a_ic_dapi = _a_floor("dapi", "DAPI", 500)
+ _a_ic_boundary = _a_floor("boundary", "boundary", 100)
+ _a_ic_intrna = _a_floor("intrna", "intRNA", 300)
+ _a_spatial_cfg = ((roi_cutoffs.get("image_qc") or {}).get("spatial_context") or {})
+ _a_artif_warn = _a_spatial_cfg.get("in_artifact_warn")
+ _a_artif_fail = _a_spatial_cfg.get("in_artifact_fail")
+ # Cell-level advisory cutoffs (PASS / WARN only, no FAIL). Read here so the
+ # metric-explanation callout can state them; reused by the §4.1/§4.2 rows.
+ _a_cl_cut = (roi_cutoffs.get("image_qc") or {}).get("cell_level") or {}
+ _a_nuc_warn = _a_cl_cut.get("pct_low_nuclear_texture_warn", 5.0)
+ _a_int_warn = _a_cl_cut.get("pct_cells_below_intensity_warn", 10.0)
+ _a_art_warn = _a_cl_cut.get("pct_cells_in_optically_dense_regions_warn", 30.0)
+ _a_gmm_warn = _a_cl_cut.get("pct_blurred_gmm_2d_roi_warn", 20.0)
+
+ # Pre-formatted threshold strings for the 📖 callout (use Unicode ≤/≥ to
+ # avoid `<` being mistaken for an HTML tag start by Pandoc / browsers).
+ _a_pct_cutoff_str = (
+ f"PASS below {int(_a_cw * 100)}% / WARN ≥ {int(_a_cw * 100)}% / FAIL ≥ {int(_a_cf * 100)}%"
+ )
+ _a_artif_cutoff_str = (
+ f"PASS below {int(float(_a_artif_warn) * 100)}% / WARN ≥ {int(float(_a_artif_warn) * 100)}% / FAIL ≥ {int(float(_a_artif_fail) * 100)}%"
+ if isinstance(_a_artif_warn, (int, float))
+ and isinstance(_a_artif_fail, (int, float))
+ and float(_a_artif_warn) <= float(_a_artif_fail)
+ else "thresholds not configured"
+ )
+
+ display(Markdown(
+ f"""::: {{.callout-note collapse="true" title="Metric explanation"}}
+**Percentage of cells with low nuclear texture** — flags cells where the DAPI signal is too smooth/dim across the nucleus for reliable segmentation. Computed per cell as **CCFS_DAPI** (variance ÷ mean of DAPI pixel intensities inside the nucleus mask; low = smooth/uniform). *Not a blurriness metric* — captures nuclear texture quality, which depends on nuclear morphology, staining, and cell type. Routinely flags different cells to the blurry cells (GMM) metric ([Section 4.2](#sec-4-2)). *Verdict:* **advisory — PASS / WARN only, no FAIL.** A cell is flagged when CCFS_DAPI ≤ {_a_ccfs_thr}; the row turns **WARN** at ≥ {_a_nuc_warn:.0f}% flagged cells, otherwise **PASS**.
+
+**Percentage of cells with low DAPI / Boundary / IntRNA intensity** — three rows, one per channel. A cell is flagged when the average channel signal (measured over the **nucleus mask** for DAPI and the **whole-cell mask** for Boundary and IntRNA) is too dim for confident downstream interpretation.
+*Cutoffs:* DAPI < {_a_ic_dapi}, Boundary < {_a_ic_boundary}, IntRNA < {_a_ic_intrna}.
+
+**Percentage of cells in optically dense regions** — flags cells overlapping regions of unusually high pixel intensity not attributable to tissue staining (e.g. folds, debris, coverslip contamination — sometimes genuine densely-packed tissue). These regions degrade segmentation accuracy and intensity-based metrics for the cells inside them.
+
+**Verdict (all rows here):** advisory — **PASS or WARN only, there is no FAIL tier**. Each row reports the *percentage of cells flagged* by that metric; it turns **WARN** when that percentage reaches low nuclear texture ≥ {_a_nuc_warn:.0f}%, low DAPI/Boundary/IntRNA intensity ≥ {_a_int_warn:.0f}%, or cells in optically dense regions ≥ {_a_art_warn:.0f}%; otherwise **PASS**. (The per-cell intensity floors above decide whether an individual cell is "low intensity"; the percentages here decide the row verdict.) These are flags to investigate, not sample-level gates. **Why this can differ from §3.3 (tiles):** cell-level intensity uses the same per-XOA-version floor as the tile metric, but it flags at ≥ {_a_int_warn:.0f}% of cells (the tile metric needs ≥ 40% of tissue tiles) and is measured per nucleus / whole-cell mask rather than per tile — so a channel can WARN here while §3.3 tiles PASS.
+:::"""
+ ))
+
+ display(Markdown(
+ """::: {.callout-tip collapse="true" title="Interpretation help"}
+- **High percentage of cells with low nuclear texture**: many cells with degraded nuclear contrast. If the sample is **neural**, this can be biology (large neurons → naturally lower nuclear contrast); cross-check the per-channel intensity correlation heatmap in [Section 4.4.a](#sec-4-4-a) to see whether DAPI is the strongest predictor of transcript yield for this tissue. For non-neural tissue, look at the [Section 4.5](#sec-4-5) cell-flagged maps to localise the affected region.
+- **High percentage of cells with low DAPI/Boundary/IntRNA intensity**: a single-channel issue often co-occurs with the same channel's tile-level intensity issue ([Section 3.3](#sec-3-3)); cross-check there. For tissues where DAPI isn't the strongest predictor of transcript yield (brain): consult the correlation heatmap in [Section 4.4.a](#sec-4-4-a) — Boundary or IntRNA may correlate more strongly with transcripts, in which case a low-DAPI flag carries less weight.
+- **High percentage of cells in optically dense regions**: usually tissue folds, debris, or coverslip contamination — cross-check the optically-dense-region map in [Section 2.3](#sec-2-3) Masks.
+- **Tissue-biology context**: a high percentage of low-IntRNA cells in skin, or a high percentage of cells in optically dense regions in spleen or specific brain regions, often reflects tissue biology rather than acquisition failure. Cross-check [Section 4.4.a](#sec-4-4-a) before excluding cells flagged by a single channel.
+:::"""
+ ))
+
+ def _a_value_str(_pct, _n_below, _n_total_local):
+ if not isinstance(_pct, (int, float)):
+ return "—"
+ if isinstance(_n_below, (int, float)) and isinstance(_n_total_local, (int, float)) and _n_total_local > 0:
+ return f"{_pct:.2f}% ({int(_n_below):,}/{int(_n_total_local):,})"
+ return f"{_pct:.2f}%"
+
+ _a_rows = []
+
+ # ---- [Section 4.1](#sec-4-1) Cell quality scores summary — per-cell metrics ----
+
+ # Tier 1 row 1: pct_low_nuclear_texture (per-cell CCFS_DAPI column ≤ threshold)
+ _a_pct_low = cell_metrics.get("pct_low_nuclear_texture")
+ _a_n_low = cell_metrics.get("cells_low_nuclear_texture")
+ _a_rows.append({
+ "Metric": "Percentage of cells with low nuclear texture",
+ "Value": _a_value_str(_a_pct_low, _a_n_low, _a_n_total),
+ "Advisory": _advisory_pass_warn(_a_pct_low, _a_nuc_warn),
+ "Filtering column": "is_low_nuclear_texture",
+ })
+
+ # Tier 1 rows 2–4: per-channel intensity below intensity_critical
+ # The pct_cells_below_intensity_* JSON keys are emitted as % only — derive
+ # the absolute count locally via pct × total / 100 so the Value column
+ # shows "X.XX% (n/total)" matching the nuclear texture / optically-dense rows.
+ for _ch_label, _emit_key, _ic, _filter_col in (
+ ("DAPI", "pct_cells_below_intensity_DAPI", _a_ic_dapi, "mean_intensity"),
+ ("Boundary", "pct_cells_below_intensity_Boundary", _a_ic_boundary, "mean_intensity_Boundary"),
+ ("IntRNA", "pct_cells_below_intensity_IntRNA", _a_ic_intrna, "mean_intensity_IntRNA"),
+ ):
+ _ipct = cell_metrics.get(_emit_key)
+ _i_n_below = (
+ int(round(float(_ipct) * _a_n_total / 100.0))
+ if isinstance(_ipct, (int, float)) and _a_n_total > 0
+ else None
+ )
+ _a_rows.append({
+ "Metric": f"Percentage of cells with low {_ch_label} intensity",
+ "Value": _a_value_str(_ipct, _i_n_below, _a_n_total),
+ "Advisory": _advisory_pass_warn(_ipct, _a_int_warn),
+ "Filtering column": f"{_filter_col}",
+ })
+
+ # Tier 1 row 5: % cells in optically dense regions (derived from cells_with_artifacts)
+ _a_n_artif = cell_metrics.get("cells_with_artifacts")
+ _a_pct_artif = (
+ 100.0 * int(_a_n_artif) / _a_n_total
+ if isinstance(_a_n_artif, (int, float)) and _a_n_total > 0 else None
+ )
+ _a_rows.append({
+ "Metric": "Percentage of cells in optically dense regions",
+ "Value": _a_value_str(_a_pct_artif, _a_n_artif, _a_n_total),
+ "Advisory": _advisory_pass_warn(_a_pct_artif, _a_art_warn),
+ "Filtering column": "has_artifacts",
+ })
+
+ # ---- Render [Section 4.1](#sec-4-1) table; transition to [Section 4.2](#sec-4-2) ----
+ display_status_table(
+ pd.DataFrame(_a_rows),
+ status_cols=["Advisory"],
+ narrow_cols={"Filtering column": "16em"},
+ )
+ display(Markdown("### 4.2 Cell quality (inherited from tile scores summary) {#sec-4-2}"))
+
+ display(Markdown(
+ """::: {.callout-note collapse="true" title="Metric explanation"}
+**Blurry cells (GMM)** — inherits the tile-level blurriness classification (the 2D GMM in [Section 3.2](#sec-3-2)) for each cell via spatial overlap. Computed over **solid-tissue cells only** (tile coverage ≥ 0.5), to match the tile-level "tiles in focus" in [Section 3.2](#sec-3-2). Cells in low-coverage tiles are excluded here (counted in the next row instead) because they are force-labelled blurry for coverage reasons, not optical focus — including them otherwise pushes this % far above the tile figure.
+
+**Cells in low-coverage tiles** — flags cells whose assigned tile lies in low-coverage tissue (`roi_tissue_coverage` < 0.5).
+:::"""
+ ))
+
+ display(Markdown(
+ """::: {.callout-tip collapse="true" title="Interpretation help"}
+If there is a high percentage of blurry cells, check whether Cohen's *d* between the GMM components is a PASS in [Section 3.2](#sec-3-2). If not, the metric may not be detecting real blurriness. Also check whether there is a high percentage of cells in low-coverage tiles, which may indicate a holey tissue where cells receive a 'blurry' label for biological reasons.
+:::"""
+ ))
+
+ _a_rows = []
+
+ # ---- [Section 4.2](#sec-4-2) Tile-mapped per-cell metrics (% cells in flagged tiles) ----
+
+ # [Section 4.2](#sec-4-2) row 1: Blurry cells (GMM), over solid-tissue cells
+ # (denominator = cells_evaluated_for_blur) so it matches the tile-level metric.
+ _a_pct_gmm = cell_metrics.get("pct_blurred_gmm_2d_roi")
+ _a_n_gmm = cell_metrics.get("cells_blurred_gmm_2d_roi")
+ _a_blur_denom = cell_metrics.get("cells_evaluated_for_blur", _a_n_total)
+ if _a_pct_gmm is None and isinstance(_a_n_gmm, (int, float)) and _a_blur_denom:
+ _a_pct_gmm = round(100.0 * float(_a_n_gmm) / float(_a_blur_denom), 2)
+ _a_rows.append({
+ "Metric": "Blurry cells (GMM)",
+ "Value": _a_value_str(_a_pct_gmm, _a_n_gmm, _a_blur_denom),
+ "Advisory": _advisory_pass_warn(_a_pct_gmm, _a_gmm_warn),
+ "Filtering column": "is_blurred_gmm_2d_roi",
+ })
+
+ # Tier 2 row 2: Cells in low-coverage tiles (informational — no Advisory)
+ _a_pct_lc = cell_metrics.get("pct_cells_in_low_coverage_tiles")
+ _a_n_lc = cell_metrics.get("cells_in_low_coverage_tiles")
+ if isinstance(_a_pct_lc, (int, float)) or isinstance(_a_n_lc, (int, float)):
+ _a_rows.append({
+ "Metric": "Cells in low-coverage tiles",
+ "Value": _a_value_str(_a_pct_lc, _a_n_lc, _a_n_total),
+ "Advisory": "—",
+ "Filtering column": "roi_tissue_coverage",
+ })
+
+ # ---- Render [Section 4.2](#sec-4-2) table; transition to [Section 4.3](#sec-4-3) ----
+ display_status_table(
+ pd.DataFrame(_a_rows),
+ status_cols=["Advisory"],
+ narrow_cols={"Filtering column": "16em"},
+ )
+ display(Markdown("### 4.3 Cell quality cluster bias summary {#sec-4-3}"))
+
+ display(Markdown(
+ """::: {.callout-note collapse="true" title="Metric explanation"}
+WARN when an outlier cluster is detected; PASS when none are.
+
+**Outlier cluster** — any cluster where blurry-cell rate or low-nuclear texture rate exceeds max(2× sample-wide rate, 10%) AND n_cells ≥ 50.
+
+**Worst outlier (blurriness / low nuclear texture)** — if one or more clusters with unusually high levels of blurriness or low nuclear texture have been identified, the worst one is shown.
+:::"""
+ ))
+
+ display(Markdown(
+ """::: {.callout-tip collapse="true" title="Interpretation help"}
+A cluster composed largely of blurry or low-nuclear-texture cells is likely defined by that image-quality defect rather than by biology. Because these metrics are sensitive to technical factors, such outlier clusters warrant caution: a cluster with high blurriness reflects a regional optical problem and is unlikely to represent a genuine cell population. A high proportion of low-nuclear-texture cells in a cluster may also indicate a technical problem, though it can also reflect genuine cell-type properties. If a technical cause is likely, the cluster can be filtered out before downstream analysis.
+:::"""
+ ))
+
+ _a_rows = []
+
+ # ---- [Section 4.3](#sec-4-3) Per-cluster outliers ----
+
+ # Tier 3 row 1: cluster_outlier_focus (worst tile-blurriness cluster outlier)
+ # When cluster_blur falls back to is_low_nuclear_texture, the two [Section 4.3](#sec-4-3)
+ # rows compute identical outlier sets — surface that to the reader so
+ # they don't read row 1 + row 2 as independent signals.
+ _a_blur_method = cell_metrics.get("cluster_blur_method", "is_blurred_gmm_2d_roi")
+ _a_blur_outliers = cell_metrics.get("cluster_blur_outliers")
+ _a_blur_fallback_suffix = (
+ " — fallback method (no blurry-(GMM) signal; values equivalent to the low-nuclear texture row below)"
+ if _a_blur_method == "is_low_nuclear_texture" else ""
+ )
+ if isinstance(_a_blur_outliers, dict) and _a_blur_outliers:
+ _a_worst_blur = max(_a_blur_outliers, key=lambda k: _a_blur_outliers[k]["pct_blurred"])
+ _a_worst_blur_info = _a_blur_outliers[_a_worst_blur]
+ _a_rows.append({
+ "Metric": "Worst tile-blurriness cluster outlier",
+ "Value": (
+ f"C{_a_worst_blur} "
+ f"({float(_a_worst_blur_info['pct_blurred']):.0f}% blurry cells (GMM), "
+ f"n={int(_a_worst_blur_info['n_cells']):,})"
+ + _a_blur_fallback_suffix
+ ),
+ "Advisory": "WARN",
+ "Filtering column": "Cluster_kmeans10",
+ })
+ else:
+ _a_rows.append({
+ "Metric": "Worst tile-blurriness cluster outlier",
+ "Value": "no outliers" + _a_blur_fallback_suffix,
+ "Advisory": "PASS",
+ "Filtering column": "—",
+ })
+
+ # Tier 3 row 2: cluster_outlier_CCFS (worst low-nuclear texture cluster outlier)
+ # JSON key kept as `cluster_ccfs_outliers` for backward compatibility; the
+ # user-facing row name now uses "low-nuclear texture" terminology consistent
+ # with the rest of §4.
+ _a_ccfs_outliers = cell_metrics.get("cluster_ccfs_outliers")
+ if isinstance(_a_ccfs_outliers, dict) and _a_ccfs_outliers:
+ _a_worst_ccfs = max(_a_ccfs_outliers, key=lambda k: _a_ccfs_outliers[k]["pct_low_texture"])
+ _a_worst_ccfs_info = _a_ccfs_outliers[_a_worst_ccfs]
+ _a_rows.append({
+ "Metric": "Worst low-nuclear texture cluster outlier",
+ "Value": (
+ f"C{_a_worst_ccfs} "
+ f"({float(_a_worst_ccfs_info['pct_low_texture']):.0f}% low-texture, "
+ f"n={int(_a_worst_ccfs_info['n_cells']):,})"
+ ),
+ "Advisory": "WARN",
+ "Filtering column": "Cluster_kmeans10",
+ })
+ elif "cluster_ccfs" in cell_metrics:
+ _a_rows.append({
+ "Metric": "Worst low-nuclear texture cluster outlier",
+ "Value": "no outliers",
+ "Advisory": "PASS",
+ "Filtering column": "—",
+ })
+
+ # Render [Section 4.3](#sec-4-3) table. Advisory column = WARN when an outlier cluster
+ # fires the upstream rule (>2× sample-wide rate AND ≥50 cells), PASS
+ # otherwise. Verdict is presence/absence of outliers, not a value-vs-
+ # threshold comparison — so no YAML cutoff is involved here.
+ display_status_table(
+ pd.DataFrame(_a_rows),
+ status_cols=["Advisory"],
+ narrow_cols={"Filtering column": "16em"},
+ )
+
+ # =============================================================================
+ # [Section 4.4](#sec-4-4)–4.5 cell-level supporting figures — four sub-subsections + conditional drill-down (Phase 14, extended r6)
+ # =============================================================================
+ # Replaces v3's 9.A figure displays (CCFS spatial, ccfs_thresholded,
+ # nuclear_texture_vs_transcripts_log) + v3's 9.B figures (cell_focus_distribution,
+ # gmm_focus_vs_transcripts, spatial_comparison) + v3's 9.D figures
+ # (nuclear_texture_proportions, nuclear_texture_density, tile_blur_proportions_roi).
+ # v3 9.C / 9.D / 9.E and the UMAP overlay block all dissolved — UMAP figures
+ # remain unsurfaced latent artefacts produced by bin/image_qc.py.
+ # v4 r5: spatial_comparison_nuclei_vs_roi (2-panel CCFS-vs-tile-focus) is
+ # now also latent — split-replaced by single-panel `tile_focus_gmm_spatial`
+ # (right side only; CCFS side was dropped per user feedback as concordance
+ # with tile-focus is known to be low). cell_focus_distribution trimmed
+ # from 2×2 to 1×2 — bottom-left dropped (focus density per cluster, redundant
+ # with nuclear_texture_density), bottom-right extracted as standalone
+ # `blur_prob_density_by_cluster` and moved to Per-cluster subsection.
+ display(Markdown("### 4.4 Technical effects on transcript count {#sec-4-4}"))
+ display(Markdown(
+ "Per-cell scatters of each quality metric against transcript count. "
+ ))
+
+ display(Markdown(
+ """::: {.callout-tip collapse="true" title="Interpretation help"}
+**Nuclear texture vs transcript count**
+
+The correlation is tissue-dependent. Mildly positive in most tissues, but can flip negative in tissues with large nuclei and diffuse chromatin (e.g. brain). A weak or negative correlation here is not automatically a quality problem — read against the tissue context.
+
+**Stain intensity vs transcript count**
+
+- **IntRNA** is the most consistent predictor of transcript yield — expect a strong positive correlation.
+- **DAPI** is tissue-variable. Positive in most tissues, but can flip negative in brain — large pyramidal neurons have diffuse, low-density chromatin (low DAPI signal) but high transcript counts.
+- **Boundary** is a moderate predictor. Can exceed DAPI in samples where membrane staining tracks cell density better than nuclear contrast.
+
+**Focus score vs transcript count**
+
+- The diagnostic signal is the **x-axis spread** within each focus band — imaging defects push blurry cells (red) to lower transcript counts at the same focus level.
+- Biology-driven low-count regions (white matter, adipose) stay co-located with well-focused tissue.
+- Don't read the vertical red/blue split as a finding — the GMM uses focus scores as input.
+:::"""
+ ))
+
+ # ----- [Section 4.4.a](#sec-4-4-a) Per-cell quality metric correlations (heatmap, first) -----
+ # Headline view: 6×6 Spearman ρ matrix across the per-cell quality
+ # axes (nuclear texture, focus score, DAPI/Boundary/IntRNA intensities)
+ # vs transcript counts. Comes first so a reader sees the overall
+ # relationships before drilling into individual scatters below.
+ display(Markdown("#### 4.4.a Per-cell quality metric correlations {#sec-4-4-a}"))
+ display(Markdown(
+ "Spearman correlation matrix across nuclear texture (`CCFS_DAPI`), focus score, "
+ "DAPI / Boundary / IntRNA mean intensity, and transcript counts. "
+ ))
+ display(fig_html("figures/intensity_transcript_correlation.png", max_width="55%"))
+
+ # ----- [Section 4.4.b](#sec-4-4-b) Nuclear texture vs transcript count -----
+ display(Markdown("#### 4.4.b Nuclear texture vs transcript count {#sec-4-4-b}"))
+ display(Markdown(
+ "Points are coloured by local 2D cell density (viridis: bright yellow = "
+ "many overlapping cells, dark purple = sparse outliers)."
+ ))
+ display(fig_html("figures/nuclear_texture_vs_transcripts_log.png"))
+
+ # ----- [Section 4.4.c](#sec-4-4-c) Stain intensity vs transcript count -----
+ # Three per-channel scatters showing DAPI / Boundary / IntRNA intensity
+ # vs transcript count. The high-level Spearman view across all channels
+ # is in [Section 4.4.a](#sec-4-4-a) above.
+ display(Markdown("#### 4.4.c Stain intensity vs transcript count {#sec-4-4-c}"))
+ display(Markdown("**DAPI mean intensity vs transcript count**"))
+ display(fig_html("figures/dapi_intensity_vs_transcripts_log.png"))
+
+ display(Markdown("**Boundary mean intensity vs transcript count**"))
+ display(fig_html("figures/boundary_intensity_vs_transcripts_log.png"))
+
+ display(Markdown("**IntRNA mean intensity vs transcript count**"))
+ display(fig_html("figures/intrna_intensity_vs_transcripts_log.png"))
+
+ # ----- [Section 4.4.d](#sec-4-4-d) Focus score vs transcript count -----
+ display(Markdown("#### 4.4.d Focus score vs transcript count {#sec-4-4-d}"))
+ display(Markdown(
+ "Focus score is computed per tile; each cell inherits the focus score of "
+ "its tile. Points are coloured by GMM classification (blue = in focus, "
+ "red = blurry)."
+ ))
+ display(fig_html("figures/gmm_focus_vs_transcripts.png"))
+
+ # =============================================================================
+ # [Section 4.5](#sec-4-5) Spatial distribution of low-quality flagged cells
+ # =============================================================================
+ display(Markdown("### 4.5 Spatial distribution of low-quality flagged cells {#sec-4-5}"))
+ display(Markdown(
+ "Whole-sample maps marking cells flagged by low nuclear texture or tile "
+ "blurriness. Use these to see whether quality loss is spatially clustered "
+ "(physical optical issue) or scattered (baseline noise)."
+ ))
+
+ display(fig_html("figures/cell_flagged_maps.png"))
+
+ # Cell-level focus score distribution figure removed 2026-05-15 (user
+ # decision). The tile-level focus distribution in [Section 3.2](#sec-3-2) is the canonical
+ # view; the cell-level version was the tile distribution re-weighted by
+ # per-tile cell density, which conflicted with [Section 3.2](#sec-3-2) on samples with
+ # heterogeneous cell density and added no signal for the advisory
+ # filtering use-case of §4. The cell-level GMM classification (verdict
+ # in [Section 4.2](#sec-4-2)'s "Blurry cells (GMM)" row) carries the actionable signal.
+
+ # ----- Cluster bias supporting figures (companion to [Section 4.3](#sec-4-3) table) -----
+ display(Markdown("### 4.6 Cluster bias supporting figures {#sec-4-6}"))
+
+ display(Markdown(
+ """::: {.callout-tip collapse="true" title="Interpretation help"}
+- **Nuclear texture proportions by cluster**: clusters should have similar texture proportions. Cluster-specific bias may reflect nuclear morphology differences between cell types — nuclear texture captures nuclear contrast, not optical blurriness, so cluster separation may be biology rather than quality loss. Cross-check with cluster spatial distribution before excluding cells.
+- **Tile-blurriness proportions by cluster (GMM)**: clusters should have similar blurriness proportions. Cluster-specific enrichment may indicate spatially biased quality loss, likely confounded by imaging artefacts rather than real biology.
+- **Per-cluster blurriness breakdown table**: Cross-check the cell-flagged maps ([Section 4.5](#sec-4-5)) for whether flagged clusters overlap blurry regions before excluding cells.
+ - High blurriness but low low-texture in a cluster: possible imaging issue localised to that cluster's spatial position; nuclear contrast is unaffected.
+ - High low-texture but low blurriness: nuclear texture loss without an imaging defect; likely reflects cell-type-specific nuclear morphology rather than quality loss.
+ - Both high: likely indicates a real spatial quality problem. Consider excluding the affected cluster.
+:::"""
+ ))
+
+ if "clusters_present" in cell_metrics:
+ display(Markdown(f"**Clusters present:** {cell_metrics['clusters_present']}"))
+
+ display(Markdown("**Nuclear texture proportions by cluster**"))
+ display(fig_html("figures/nuclear_texture_proportions.png"))
+
+ display(Markdown("**Tile-blurriness proportions by cluster (GMM)**"))
+ display(fig_html("figures/tile_blur_proportions_roi.png"))
+
+ # ----- Per-cluster blurriness breakdown -----
+ # Two independent signals are shown for this section:
+ # (1) a sample-wide "overall blurriness" advisory line, fired when the
+ # whole-sample blurry-cell rate is high (reuses the §4.2 advisory gate
+ # pct_blurred_gmm_2d_roi_warn, default 20%). It fires even when no
+ # single cluster stands out, which is the broadly-blurry case.
+ # (2) a per-cluster WARN for clusters flagged as outliers by image_qc.py
+ # (cluster_blur_outliers / cluster_ccfs_outliers). The QMD only tests
+ # membership; the detection rule (robust MAD z-score + 15% floor) lives
+ # in bin/image_qc.py so the producer and this report never drift.
+ _cluster_blur = cell_metrics.get("cluster_blur")
+ _cluster_ccfs = cell_metrics.get("cluster_ccfs")
+ _blur_outliers = cell_metrics.get("cluster_blur_outliers") or {}
+ _ccfs_outliers = cell_metrics.get("cluster_ccfs_outliers") or {}
+ display(Markdown("**Per-cluster blurriness breakdown**"))
+
+ # (1) Sample-wide advisory line (PASS/WARN, no FAIL) above the detail table.
+ _cl_cut_46 = (roi_cutoffs.get("image_qc") or {}).get("cell_level") or {}
+ _gmm_warn_46 = float(_cl_cut_46.get("pct_blurred_gmm_2d_roi_warn", 20.0))
+ _sample_blur = cell_metrics.get("pct_blurred_gmm_2d_roi")
+ if isinstance(_sample_blur, (int, float)):
+ _overall_caveat = (
+ f"Overall blurriness is high ({_sample_blur:.1f}% of cells flagged "
+ f"blurry, at or above the {_gmm_warn_46:.0f}% advisory level). A high "
+ "sample-wide rate means image quality is a concern across the whole "
+ "section, so the per-cluster table below may show no single standout "
+ "while the sample is still affected. Review the focus heatmap "
+ "([Section 3.2](#sec-3-2)) and the sample-wide blurry-cells row "
+ "([Section 4.2](#sec-4-2))."
+ )
+ _overall_status = "WARN" if _sample_blur >= _gmm_warn_46 else "PASS"
+ _overall_label = metric_cell_with_caveat(
+ f"{_sample_blur:.1f}%", _overall_status, _overall_caveat,
+ ) if _overall_status == "WARN" else f"{_sample_blur:.1f}%"
+ _overall_df = pd.DataFrame([{
+ "Overall blurriness (sample-wide)": _overall_label,
+ "Status": _overall_status,
+ }])
+ display_status_table(_overall_df, status_cols=["Status"])
+ _cl_rows = []
+ # Iterate the union of cluster IDs from both dicts so a CCFS-only
+ # trigger (where _cluster_blur may be empty) still renders rows. Each
+ # column falls back to "—" when its source dict lacks the cluster key.
+ # Sort by number of cells descending (largest clusters first); clusters
+ # with no cell-count data fall to the end. Secondary key on
+ # int(cluster_id) keeps tied rows deterministic.
+ _blur_keys = set(_cluster_blur.keys()) if isinstance(_cluster_blur, dict) else set()
+ _ccfs_keys = set(_cluster_ccfs.keys()) if isinstance(_cluster_ccfs, dict) else set()
+ def _cluster_sort_key(cid):
+ _b = _cluster_blur.get(cid) if isinstance(_cluster_blur, dict) else None
+ _c = _cluster_ccfs.get(cid) if isinstance(_cluster_ccfs, dict) else None
+ _n = (_b.get("n_cells") if isinstance(_b, dict) else None) \
+ or (_c.get("n_cells") if isinstance(_c, dict) else None)
+ _has = isinstance(_n, (int, float))
+ # Tuple: (no-data clusters sort after; then n_cells desc; then cluster id asc)
+ return (0 if _has else 1, -int(_n) if _has else 0, int(cid))
+ _all_cluster_ids = sorted(_blur_keys | _ccfs_keys, key=_cluster_sort_key)
+ for cl_id in _all_cluster_ids:
+ _blur_info = _cluster_blur.get(cl_id) if isinstance(_cluster_blur, dict) else None
+ _ccfs_info = _cluster_ccfs.get(cl_id) if isinstance(_cluster_ccfs, dict) else None
+ _has_blur_data = isinstance(_blur_info, dict)
+ _has_ccfs_data = isinstance(_ccfs_info, dict)
+ _is_blur_out = cl_id in _blur_outliers
+ _is_ccfs_out = cl_id in _ccfs_outliers
+ is_outlier = _is_blur_out or _is_ccfs_out
+ _cluster_label = f"C{cl_id}"
+ if is_outlier:
+ if _is_blur_out and _is_ccfs_out:
+ _trigger = "high blurriness and low nuclear texture"
+ elif _is_blur_out:
+ _trigger = "high blurriness"
+ else:
+ _trigger = "low nuclear texture"
+ _caveat = (
+ f"**Cluster outlier: {_trigger}.** This cluster's rate is far "
+ "above the other clusters in this sample, which points to "
+ "spatially concentrated quality loss that can affect downstream "
+ "analysis of this cell population. Check the focus heatmap "
+ "([Section 3.2](#sec-3-2)) to see whether the cluster sits over an "
+ "out-of-focus region, then decide whether to exclude or flag its "
+ "cells."
+ )
+ _cluster_label = metric_cell_with_caveat(
+ _cluster_label, "WARN", _caveat,
+ )
+ _n_cells = (
+ _blur_info.get("n_cells") if _has_blur_data
+ else (_ccfs_info.get("n_cells") if _has_ccfs_data else None)
+ )
+ _cl_rows.append({
+ "Cluster": _cluster_label,
+ "Cells": f"{int(_n_cells):,}" if isinstance(_n_cells, (int, float)) else "—",
+ "Blurry Cells": f'{_blur_info["n_blurred"]:,}' if _has_blur_data and "n_blurred" in _blur_info else "—",
+ "Percentage of blurry cells": f'{_blur_info["pct_blurred"]:.1f}%' if _has_blur_data and "pct_blurred" in _blur_info else "—",
+ "Percentage of low-texture cells": f'{float(_ccfs_info["pct_low_texture"]):.1f}%' if _has_ccfs_data and "pct_low_texture" in _ccfs_info else "—",
+ "Median nuclear texture": f'{_blur_info["median_ccfs_dapi"]:.3f}' if _has_blur_data and "median_ccfs_dapi" in _blur_info else "—",
+ "Status": "WARN" if is_outlier else "—",
+ })
+ if _cl_rows:
+ _df_cl = pd.DataFrame(_cl_rows)
+ display_status_table(_df_cl, status_cols=["Status"])
+ else:
+ # No cluster-level data emitted in this run's JSON (cluster_blur and
+ # cluster_ccfs both missing). Render a clarifying note.
+ display(Markdown(
+ "*Per-cluster detail (`cluster_blur` / `cluster_ccfs`) is not "
+ "available in this run's JSON — cluster-level breakdown skipped.*"
+ ))
+
+```
+
+## 5. Metadata {#sec-5}
+
+
+```{python report-metadata}
+#| echo: false
+
+from datetime import datetime
+
+# Set when this HTML is rendered (local time).
+_report_stamp = datetime.now().astimezone().strftime("%Y-%m-%d %H:%M %Z")
+
+_resolved_out = base.resolve()
+_sample = str(SAMPLE_NAME).strip() if SAMPLE_NAME else ""
+if not _sample:
+ if _resolved_out.name == "image_qc":
+ _sample = _resolved_out.parent.name
+ else:
+ _sample = _resolved_out.name
+
+_bundle = str(XENIUM_BUNDLE).strip() if XENIUM_BUNDLE else ""
+_bundle_display = (
+ f"{_bundle}"
+ if _bundle
+ else "— (pass -P XENIUM_BUNDLE:… for local render; Nextflow passes samplesheet path)"
+)
+
+_pub = str(SAMPLE_PUBLISHED_OUTDIR).strip() if SAMPLE_PUBLISHED_OUTDIR else ""
+_pub_display = (
+ f"{_pub}"
+ if _pub
+ else "— (pass -P SAMPLE_PUBLISHED_OUTDIR:… or run via Nextflow)"
+)
+
+# Collapse versions.yml into a single one-line summary (joined key=value pairs).
+if versions_text:
+ _v_pairs = []
+ for _ln in versions_text.splitlines():
+ _s = _ln.strip()
+ if not _s:
+ continue
+ # Skip YAML section-header lines (e.g. "CELLPOSE:") that have no inline value.
+ if _s.endswith(":") and ":" not in _s[:-1]:
+ continue
+ if ":" in _s:
+ _k, _v = _s.split(":", 1)
+ _v = _v.strip()
+ if _v:
+ _v_pairs.append(f"{_k.strip()}={_v}")
+ _versions_one_line = "; ".join(_v_pairs) if _v_pairs else versions_text.strip().replace("\n", " ")
+else:
+ _versions_one_line = "— (versions.yml not found)"
+
+_seg_sw = roi_metrics.get("segmentation_software") or "—"
+# XOA (onboard analysis) version, emitted by image_qc.py from the bundle's
+# experiment.xenium. Strip the "xenium-" prefix for display. Blank on bundles
+# processed before this field was added — reprocess to populate.
+_xoa_raw = roi_metrics.get("xoa_version")
+_xoa_version = (
+ str(_xoa_raw).split("-", 1)[-1]
+ if _xoa_raw
+ else "— (not recorded; reprocess to populate)"
+)
+
+_meta = pd.DataFrame(
+ [
+ {"Field": "Report generated", "Value": _report_stamp},
+ {"Field": "Xenium bundle", "Value": _bundle_display},
+ {"Field": "XOA version", "Value": _xoa_version},
+ {"Field": "Segmentation software", "Value": _seg_sw},
+ {"Field": "Image QC output folder path", "Value": _pub_display},
+ {"Field": "Software versions", "Value": _versions_one_line},
+ ]
+)
+_tbl = _meta.to_html(index=False, header=False, classes="table report-meta", border=0, escape=False)
+display(
+ HTML(
+ ""
+ + _tbl
+ )
+)
+```
+
+
+```{python authors-footer}
+#| echo: false
+_pipeline_version = str(PIPELINE_VERSION).strip() if PIPELINE_VERSION else ""
+_version_clause = f" v{_pipeline_version}" if _pipeline_version else ""
+display(HTML(
+ ''
+ 'Generated by
'
+ f'nf-xenium-processing{_version_clause} — pipeline maintained by Altos '
+ 'Labs Spatial Bioinformatics. Authors: Malwina Prater, Hanneke Okkenaug '
+ 'Nell Yu Nie & Christel Krueger.'
+ '
'
+))
+```
diff --git a/bin/baysor_create_dataset.py b/bin/baysor_create_dataset.py
index 4e5a263a..2ca4d5ae 100755
--- a/bin/baysor_create_dataset.py
+++ b/bin/baysor_create_dataset.py
@@ -13,18 +13,19 @@
from pathlib import Path
-class BaysorPreview():
+class BaysorPreview:
"""
Utility class to generate baysor preview dataset
"""
+
@staticmethod
def generate_dataset(
- transcripts: Path,
- sampled_transcripts: Path,
- sample_fraction: float = 0.3,
- random_state: int = 42,
- prefix: str = ""
- ) -> None:
+ transcripts: Path,
+ sampled_transcripts: Path,
+ sample_fraction: float = 0.3,
+ random_state: int = 42,
+ prefix: str = "",
+ ) -> None:
"""
Reads a csv file & randomly samples a fraction of rows,
and writes the result to a .csv file.
@@ -40,9 +41,10 @@ def generate_dataset(
random.seed(random_state)
output_path = f"{prefix}/{sampled_transcripts}"
os.makedirs(os.path.dirname(output_path), exist_ok=True)
- with open(transcripts, mode='rt', newline='') as infile, \
- open(output_path, mode='wt', newline='') as outfile:
-
+ with (
+ open(transcripts, mode="rt", newline="") as infile,
+ open(output_path, mode="wt", newline="") as outfile,
+ ):
reader = csv.reader(infile)
writer = csv.writer(outfile)
@@ -66,27 +68,25 @@ def main() -> None:
description="Create sampled dataset for Baysor preview"
)
parser.add_argument(
- "--transcripts", required=True,
- help="Path to transcripts CSV file"
- )
- parser.add_argument(
- "--sample-fraction", required=True, type=float,
- help="Fraction of rows to sample"
+ "--transcripts", required=True, help="Path to transcripts CSV file"
)
parser.add_argument(
- "--prefix", required=True,
- help="Output directory prefix"
+ "--sample-fraction",
+ required=True,
+ type=float,
+ help="Fraction of rows to sample",
)
+ parser.add_argument("--prefix", required=True, help="Output directory prefix")
args = parser.parse_args()
- sampled_transcripts = "sampled_transcripts.csv"
+ sampled_transcripts = Path("sampled_transcripts.csv")
# generate dataset
BaysorPreview.generate_dataset(
transcripts=args.transcripts,
sampled_transcripts=sampled_transcripts,
sample_fraction=args.sample_fraction,
- prefix=args.prefix
+ prefix=args.prefix,
)
return None
diff --git a/bin/image_qc.py b/bin/image_qc.py
new file mode 100755
index 00000000..a429d7ea
--- /dev/null
+++ b/bin/image_qc.py
@@ -0,0 +1,14256 @@
+#!/usr/bin/env python3
+"""
+Combined Image QC Script
+
+This script combines three image QC scripts into one:
+- image_qc_roi_processing.py: GPU-accelerated tile/pixel-level focus maps
+- image_qc_processing.py: Cell-based QC with Quarto figures
+- image_qc_mapping_to_cells.py: Maps tile results to cells
+
+Author: Hanneke Okkenhaug, Malwina Prater
+"""
+
+from __future__ import annotations
+
+import logging
+from dataclasses import dataclass
+import math
+import os
+import queue
+import sys
+import threading
+import time
+import traceback
+import warnings
+import click
+import json
+import numpy as np
+import pandas as pd
+import tifffile
+import zarr
+from pathlib import Path
+import matplotlib.pyplot as plt
+from napari_skimage_regionprops import regionprops_table
+from skimage.segmentation import clear_border
+from skimage import measure, color, morphology
+from skimage.filters import apply_hysteresis_threshold, threshold_otsu
+import napari_simpleitk_image_processing as nsitk
+import seaborn as sns
+from sklearn.preprocessing import RobustScaler
+from tifffile import imread
+import multiprocessing
+import multiprocessing.connection
+import shutil
+from concurrent.futures import ThreadPoolExecutor, as_completed
+from typing import Any
+from numpy.typing import NDArray
+from scipy.ndimage import gaussian_laplace as scipy_gaussian_laplace
+from scipy.ndimage import laplace as scipy_laplace
+from scipy.ndimage import uniform_filter as scipy_uniform_filter
+from scipy import ndimage
+from sklearn.mixture import GaussianMixture
+
+import snr_metrics
+
+
+# Set matplotlib to use a non-interactive backend
+import matplotlib
+
+matplotlib.use("Agg")
+
+# GPU backend detection (CuPy)
+try:
+ import cupy as cp # type: ignore[import-untyped]
+ import cupyx.scipy.ndimage # type: ignore[import-untyped] # noqa: F401 (binds `cupyx` for warmup)
+ from cupyx.scipy.ndimage import laplace as cupy_laplace # type: ignore[import-untyped]
+ from cupyx.scipy.ndimage import uniform_filter as cupy_uniform_filter # type: ignore[import-untyped]
+
+ def cupy_gaussian_laplace(image, sigma): # type: ignore[misc]
+ """CuPy Laplacian of Gaussian: Gaussian smooth then Laplacian."""
+ from cupyx.scipy.ndimage import gaussian_filter as _gf # type: ignore[import-untyped]
+
+ return cupy_laplace(_gf(image, sigma=sigma))
+
+ HAS_CUPY = True
+except ImportError:
+ HAS_CUPY = False
+
+# ---------------------------------------------------------------------------
+# Segmentation-software label helpers, inlined from the upstream
+# nf-xenium-processing `xenium_helpers.utils` (dev d71e0cd) so this script is
+# self-contained — this pipeline does not ship the xenium_helpers package.
+# Logic is unchanged from upstream; only annotations were modernised to the
+# `X | None` style (the module uses `from __future__ import annotations`).
+# ---------------------------------------------------------------------------
+
+# Pretty names for pipeline segmentation tools used as a fallback when no
+# component/version breakdown is available.
+SEGMENTATION_PRETTY = {
+ "cellpose": "Cellpose",
+ "cellpose_baysor": "Cellpose + Baysor",
+ "proseg": "Proseg",
+ "segger": "Segger",
+}
+
+# Per-method component tools as (display name, versions.yml key) pairs. The key
+# is the tool name as it appears inside the segmentation modules' versions.yml
+# (e.g. ``cellpose: 3.0.6``). Order defines how multi-tool labels read.
+SEGMENTATION_TOOL_KEYS = {
+ "cellpose": [("Cellpose", "cellpose")],
+ "cellpose_baysor": [("Cellpose", "cellpose"), ("Baysor", "baysor")],
+ "proseg": [("Proseg", "proseg")],
+ "segger": [("Segger", "segger")],
+}
+
+
+def _tool_label(display: str, key: str, tool_versions: dict[str, str] | None) -> str:
+ """``"Cellpose"`` + version -> ``"Cellpose v3.0.6"`` (name only if absent)."""
+ version = (tool_versions or {}).get(key)
+ return f"{display} v{version}" if version else display
+
+
+def read_xenium_analysis_sw_version(bundle_dir) -> str | None:
+ """Read ``analysis_sw_version`` from ``experiment.xenium`` (e.g.
+ ``"xenium-4.0.1.0"``). Returns ``None`` on missing file, missing key, or
+ malformed JSON. Mirrors ``read_xenium_pixel_size_um`` in bin/snr_metrics.py.
+ """
+ exp = Path(bundle_dir) / "experiment.xenium"
+ if not exp.is_file():
+ return None
+ try:
+ with open(exp, encoding="utf-8") as f:
+ meta = json.load(f)
+ version = meta.get("analysis_sw_version")
+ except (OSError, ValueError, TypeError, json.JSONDecodeError):
+ return None
+ if not version or not isinstance(version, str):
+ return None
+ return version
+
+
+def _parse_xenium_version(analysis_sw_version: str | None) -> str | None:
+ """``"xenium-4.0.1.0"`` -> ``"4.0.1"`` (major.minor.patch). Returns ``None``
+ if no leading numeric components can be parsed."""
+ if not analysis_sw_version:
+ return None
+ tail = analysis_sw_version.split("-", 1)[-1] # drop a 'xenium-' style prefix
+ nums = []
+ for part in tail.split("."):
+ if part.isdigit():
+ nums.append(part)
+ else:
+ break
+ if not nums:
+ return None
+ return ".".join(nums[:3])
+
+
+def read_xenium_major_version(bundle_dir) -> int | None:
+ """Major XOA version for a bundle, read from ``experiment.xenium``
+ (``"xenium-4.0.1.0"`` -> ``4``). Returns ``None`` when the file is absent or
+ the version cannot be parsed. Used to pick XOA-version-specific QC floors
+ (e.g. intensity gates differ sharply between XOA 3.x and 4.0)."""
+ parsed = _parse_xenium_version(read_xenium_analysis_sw_version(bundle_dir))
+ if not parsed:
+ return None
+ first = parsed.split(".", 1)[0]
+ return int(first) if first.isdigit() else None
+
+
+def resolve_segmentation_software(
+ bundle_dir,
+ pipeline_segmentation: str = "skip",
+ is_resegmented: bool = False,
+ tool_versions: dict[str, str] | None = None,
+) -> str:
+ """Human-readable label for the segmentation software that produced the
+ bundle a QC report describes.
+
+ - Un-resegmented / pre-seg / ``skip``: the onboard analysis version from the
+ bundle's ``experiment.xenium`` -> ``"Xenium Onboard Analysis v4.0.1"``.
+ - Pipeline ``xr`` resegmentation: the reseg bundle's own
+ ``analysis_sw_version`` -> ``"Xenium Ranger v4.0.1 (resegmentation)"``.
+ - Other pipeline tools (cellpose / cellpose_baysor / proseg / segger): the
+ tool name plus its version from ``tool_versions`` (parsed from the
+ segmentation ``versions.yml``), e.g. ``"Cellpose v3.0.6"`` or
+ ``"Cellpose v3.0.6 + Baysor v0.6.2"``. Falls back to name-only when the
+ version is unavailable. Their reseg bundle is packaged via ``xeniumranger
+ import-segmentation``, so its ``experiment.xenium`` would mislabel them as
+ Xenium Ranger; the pipeline tool name is authoritative here.
+ """
+ seg = (pipeline_segmentation or "skip").strip()
+ parsed = _parse_xenium_version(read_xenium_analysis_sw_version(bundle_dir))
+
+ if not is_resegmented or seg == "skip":
+ if parsed:
+ return f"Xenium Onboard Analysis v{parsed}"
+ return "Xenium Onboard Analysis (version unknown)"
+
+ if seg == "xr":
+ if parsed:
+ return f"Xenium Ranger v{parsed} (resegmentation)"
+ return "Xenium Ranger (resegmentation)"
+
+ components = SEGMENTATION_TOOL_KEYS.get(seg)
+ if components:
+ return " + ".join(
+ _tool_label(display, key, tool_versions) for display, key in components
+ )
+ return SEGMENTATION_PRETTY.get(seg, seg)
+
+
+# Xenium pixel size in micrometers (used for coordinate conversions)
+XENIUM_PIXEL_SIZE_UM = 0.2125
+
+# Default CCFS threshold for classifying cells as low nuclear texture quality.
+# CCFS measures per-cell nuclear contrast (local_var/local_mean), NOT optical blur.
+# Calibrated on 5 samples (2026-04-02): cells below 0.02 lose >50% transcripts
+# relative to top-50% CCFS cells in lung/liver. Brain tissue is confounded by
+# nucleus size (large neurons score lower) — interpret with caution.
+DEFAULT_CCFS_LOW_TEXTURE_THRESHOLD = 0.02
+
+# sensitivity for blurred cell detection (catches more background regions)
+ROI_INTENSITY_THRESHOLD = 100.0
+# 2026-06-26: tissue-tile gate lowered 0.5 -> 0.2. The background-from-nonzero mask is
+# tighter (un-flooded), so real tissue tiles are only thinly covered; 0.5 discarded them.
+# See plans/2026-06-26_PLAN_tissue-mask-recalibration.md.
+ROI_MIN_TISSUE_COVERAGE_FOR_INTENSITY_QC = 0.2
+# Per-tile DAPI floor for the usable_tissue low-intensity term ONLY (decoupled from the
+# is_low_intensity column, which stays at ROI_INTENSITY_THRESHOLD for the 1D-GMM tissue
+# scope). Lowered to 50 to match the relaxed DAPI intensity QC (channels.DAPI.
+# intensity_critical_v3/v4 = 50); a dim-but-real tissue tile should not count as unusable
+# on absolute brightness alone.
+DAPI_LOW_INTENSITY_FLOOR = 50.0
+
+
+# Fallback intensity critical thresholds (used when YAML not provided)
+_INTENSITY_CRITICAL_DEFAULTS = {"dapi": 500, "boundary": 100, "intrna": 300}
+
+# Fixed per-channel colorbar caps for the §3.3 intensity spatial heatmaps.
+# Calibrated against 9 tissues (lung / pancreas / liver / brain) — pooled p99
+# was ~6300 (DAPI), ~7000 (Boundary), ~3600 (IntRNA). Caps were rounded down
+# below pooled p99 so bright tissues (brain, pancreas) saturate to the
+# extend="max" triangle and dim tissues (low-signal lung) genuinely render
+# dim — cross-sample comparability over per-sample auto-scaling.
+_INTENSITY_DISPLAY_CAP = {"dapi": 4000, "boundary": 4000, "intrna": 2000}
+
+
+def _load_qc_thresholds(yaml_path: str | None) -> dict:
+ """Load thresholds from roi_image_qc_thresholds.yaml.
+
+ Returns the ``image_qc`` sub-dict, or an empty dict if the file
+ is missing / unreadable.
+ """
+ if yaml_path is None:
+ return {}
+ try:
+ import yaml
+
+ p = Path(yaml_path)
+ if not p.exists():
+ logging.warning("Thresholds YAML not found: %s", yaml_path)
+ return {}
+ with open(p, encoding="utf-8") as f:
+ d = yaml.safe_load(f) or {}
+ return d.get("image_qc", {}) if isinstance(d, dict) else {}
+ except Exception as exc:
+ logging.warning("Failed to load thresholds YAML: %s", exc)
+ return {}
+
+
+# Focus score percentile: Percentile of raw focus scores to use as threshold
+# Applied to: tiles only after exclusion of low-intensity tiles (intensity >= ROI_INTENSITY_THRESHOLD)
+# Purpose: Separate blurred vs in-focus tiles
+
+ROI_FOCUS_SCORE_PERCENTILE = 5.0
+
+# --- Per-cluster outlier detection (cluster_blur_outliers / cluster_ccfs_outliers) ---
+# A cluster is flagged when its blur (or low-texture) rate is both a robust
+# statistical outlier among the sample's clusters AND above an absolute floor.
+CLUSTER_OUTLIER_MIN_CELLS = 50 # ignore tiny clusters
+CLUSTER_OUTLIER_MIN_CLUSTERS = 5 # need enough clusters for robust stats
+CLUSTER_OUTLIER_FLOOR_PCT = 15.0 # absolute minimum % to ever flag
+CLUSTER_OUTLIER_MAD_Z = 3.5 # modified (MAD-based) z-score cutoff
+
+
+def detect_cluster_outliers(cluster_stats, pct_key):
+ """Flag clusters that stand apart from the rest of the sample.
+
+ A cluster is flagged when its percentage (``pct_key``) is at least
+ ``CLUSTER_OUTLIER_FLOOR_PCT`` AND it is a robust statistical outlier vs the
+ other clusters, measured by a modified z-score ``(x - median) / (1.4826 *
+ MAD) > CLUSTER_OUTLIER_MAD_Z``. When the MAD degenerates to 0 (common for
+ low-texture, where most clusters sit near 0%), the floor alone decides, so a
+ genuine spike is still caught. Needs at least ``CLUSTER_OUTLIER_MIN_CLUSTERS``
+ clusters with ``>= CLUSTER_OUTLIER_MIN_CELLS`` cells; below that, robust
+ stats are unreliable and nothing is flagged.
+
+ Parameters
+ ----------
+ cluster_stats : dict[int, dict]
+ Per-cluster stats, e.g. ``{cluster_id: {"n_cells": int, pct_key: float}}``.
+ pct_key : str
+ Key holding the percentage to test ("pct_blurred" or "pct_low_texture").
+
+ Returns
+ -------
+ dict[int, dict]
+ Subset of ``cluster_stats`` for flagged clusters, full stats retained
+ (downstream report consumers read ``n_cells`` and the percentage).
+ """
+ eligible = {
+ cl: s
+ for cl, s in cluster_stats.items()
+ if s.get("n_cells", 0) >= CLUSTER_OUTLIER_MIN_CELLS
+ }
+ if len(eligible) < CLUSTER_OUTLIER_MIN_CLUSTERS:
+ return {}
+ vals = np.array([eligible[cl][pct_key] for cl in eligible], dtype=float)
+ med = float(np.median(vals))
+ mad = 1.4826 * float(np.median(np.abs(vals - med)))
+ flagged = {}
+ for cl, s in eligible.items():
+ x = s[pct_key]
+ if x < CLUSTER_OUTLIER_FLOOR_PCT:
+ continue
+ if mad <= 0 or (x - med) / mad > CLUSTER_OUTLIER_MAD_Z:
+ flagged[cl] = s
+ return flagged
+
+
+def _run_figure_task(fn):
+ """Wrapper for multiprocessing figure tasks that logs exceptions before exit.
+
+ When using fork-based multiprocessing, child process tracebacks are lost
+ if the child crashes. This wrapper catches and logs the full traceback
+ in the child process before re-raising, so errors are visible in logs.
+ """
+ try:
+ fn()
+ except Exception:
+ logging.error(f"Figure task {fn.__name__} failed:\n{traceback.format_exc()}")
+ raise
+
+
+def open_zarr(path: Path, zarr3: bool = False) -> zarr.Group:
+ if zarr3:
+ store = (
+ zarr.storage.ZipStore(path)
+ if path.suffix == ".zip"
+ else zarr.storage.LocalStore(path)
+ )
+ return zarr.open_group(store=store, mode="r")
+ else:
+ """Open a Zarr file (compatible with zarr < 3)"""
+ store = (
+ zarr.ZipStore(path, mode="r")
+ if path.suffix == ".zip"
+ else zarr.DirectoryStore(path)
+ )
+ return zarr.group(store=store)
+
+
+def load_and_prepare_data(xenium_bundle_dir, outdir):
+ """Load all required data files and prepare paths"""
+
+ xenium_bundle_dir = Path(xenium_bundle_dir)
+ # Resolve outdir to absolute path to avoid nested directory issues
+ # If outdir is already absolute, resolve() returns it as-is
+ # If outdir is relative, resolve() makes it absolute relative to current working directory
+ outdir = Path(outdir).resolve()
+
+ # Required paths
+ cells_parquet_path = xenium_bundle_dir / "cells.parquet"
+ clusters_csv_path = (
+ xenium_bundle_dir
+ / "analysis"
+ / "clustering"
+ / "gene_expression_kmeans_10_clusters"
+ / "clusters.csv"
+ )
+ umap_path = (
+ xenium_bundle_dir
+ / "analysis"
+ / "umap"
+ / "gene_expression_2_components"
+ / "projection.csv"
+ )
+ cell_masks_path = xenium_bundle_dir / "cells.zarr.zip"
+ morphology_focus_dir = xenium_bundle_dir / "morphology_focus"
+
+ # Create output directories
+ figures_dir = outdir / "figures"
+ figures_dir.mkdir(parents=True, exist_ok=True)
+
+ return {
+ "xenium_bundle_dir": xenium_bundle_dir,
+ "outdir": outdir,
+ "figures_dir": figures_dir,
+ "cells_parquet_path": cells_parquet_path,
+ "clusters_csv_path": clusters_csv_path,
+ "umap_path": umap_path,
+ "cell_masks_path": cell_masks_path,
+ "morphology_focus_dir": morphology_focus_dir,
+ }
+
+
+def _load_morphology_channels(
+ xoa_morphology_files,
+ *,
+ level: int = 0,
+):
+ """
+ Load DAPI/Boundary/IntRNA channels robustly from either:
+ - a single multi-channel OME-TIFF, or
+ - separate single-channel files (e.g. *_0000, *_0001, *_0002).
+ """
+
+ def _split_channels(arr):
+ # Support both channel-first (C,Y,X) and channel-last (Y,X,C).
+ # .copy() on each slice so the parent 3D array can be GC'd.
+ if arr.ndim == 2:
+ return arr, None, None
+ if arr.ndim != 3:
+ raise ValueError(f"Unexpected morphology array shape: {arr.shape}")
+ if arr.shape[0] <= 4 and arr.shape[1] > 16 and arr.shape[2] > 16:
+ d = arr[0].copy()
+ b = arr[1].copy() if arr.shape[0] > 1 else None
+ r = arr[2].copy() if arr.shape[0] > 2 else None
+ return d, b, r
+ if arr.shape[2] <= 4 and arr.shape[0] > 16 and arr.shape[1] > 16:
+ d = arr[:, :, 0].copy()
+ b = arr[:, :, 1].copy() if arr.shape[2] > 1 else None
+ r = arr[:, :, 2].copy() if arr.shape[2] > 2 else None
+ return d, b, r
+ # Conservative fallback: prefer first axis as channels.
+ d = arr[0].copy()
+ b = arr[1].copy() if arr.shape[0] > 1 else None
+ r = arr[2].copy() if arr.shape[0] > 2 else None
+ return d, b, r
+
+ primary = tifffile.imread(
+ xoa_morphology_files[0], is_ome=False, level=level, aszarr=False
+ )
+
+ if primary.ndim == 3:
+ dapi, boundary, intrna = _split_channels(primary)
+ return dapi, boundary, intrna
+
+ dapi = primary
+ boundary = None
+ intrna = None
+
+ if len(xoa_morphology_files) > 1 and Path(xoa_morphology_files[1]).exists():
+ try:
+ b = tifffile.imread(
+ xoa_morphology_files[1], is_ome=False, level=level, aszarr=False
+ )
+ boundary = _split_channels(b)[0] if getattr(b, "ndim", 0) == 3 else b
+ except Exception as e:
+ logging.warning(
+ "Boundary channel load failed for %s (continuing with DAPI): %s",
+ xoa_morphology_files[1],
+ e,
+ )
+ boundary = None
+ if len(xoa_morphology_files) > 2 and Path(xoa_morphology_files[2]).exists():
+ try:
+ r = tifffile.imread(
+ xoa_morphology_files[2], is_ome=False, level=level, aszarr=False
+ )
+ intrna = _split_channels(r)[0] if getattr(r, "ndim", 0) == 3 else r
+ except Exception as e:
+ logging.warning(
+ "IntRNA channel load failed for %s (continuing with DAPI): %s",
+ xoa_morphology_files[2],
+ e,
+ )
+ intrna = None
+
+ return dapi, boundary, intrna
+
+
+# Tissue-mask threshold guard (2026-06-23, fixes the generate_tissue_mask bug;
+# see plans/2026-06-23_PLAN_fix-tissue-mask-bug.md and the Otsu spike). The old
+# fixed 60th-percentile threshold assumed ~40% of the field is tissue and
+# degenerated on sparse/dim slides. Otsu is background-aware and adapts to the
+# real tissue fraction, but on a UNIMODAL field (all background or all tissue)
+# Otsu still returns a split, fabricating a mask — so a guard rejects those.
+# Calibrated on the spike's synthetic Gaussian fields: real bimodal class
+# separation 8.9-24.4 sd vs unimodal 2.6-2.8 sd (clean margin around 3.0).
+# CAVEAT: the 3.0 sd cutoff is synthetic-calibrated; confirm on real small0.
+TISSUE_OTSU_GUARD_MIN_FG = 0.02 # foreground < 2% of field -> nothing detected
+TISSUE_OTSU_GUARD_MAX_FG = 0.90 # foreground > 90% of field -> no background
+TISSUE_OTSU_GUARD_MIN_SEP_SD = 3.0 # tissue mean must exceed bg mean by >= 3 bg-sd
+
+
+def otsu_tissue_threshold_with_guard(small0):
+ """Background-aware tissue threshold (Otsu) with a degeneracy guard.
+
+ Returns ``(threshold, ok)``. ``ok=False`` means the field is unimodal /
+ has no separable tissue, so no mask should be formed (the caller returns an
+ empty mask, which downstream becomes ``tissue_mask_qc.status == FAIL``).
+
+ The load-bearing test is class separation, not foreground fraction: on an
+ all-background field Otsu splits the noise at ~40% foreground (inside any
+ fraction band), so only the separation floor catches it.
+ """
+ arr = np.asarray(small0, dtype=np.float64)
+ finite = arr[np.isfinite(arr)]
+ if finite.size == 0 or float(finite.max()) == float(finite.min()):
+ return None, False # empty or uniform field
+ t = float(threshold_otsu(finite))
+ fg = arr >= t
+ fg_frac = float(np.mean(fg))
+ bg_vals = arr[(arr < t) & np.isfinite(arr)]
+ fg_vals = arr[fg & np.isfinite(arr)]
+ if bg_vals.size == 0 or fg_vals.size == 0:
+ return t, False
+ bg_std = float(bg_vals.std())
+ sep_sd = (
+ (float(fg_vals.mean()) - float(bg_vals.mean())) / bg_std
+ if bg_std > 1e-9
+ else np.inf
+ )
+ ok = (
+ TISSUE_OTSU_GUARD_MIN_FG <= fg_frac <= TISSUE_OTSU_GUARD_MAX_FG
+ and sep_sd >= TISSUE_OTSU_GUARD_MIN_SEP_SD
+ )
+ return t, ok
+
+
+# Hysteresis tissue mask (2026-06-25): replaces the global-Otsu threshold, which
+# under-captured dim/sparse tissue (validated non-circularly against decoded
+# transcripts across 14 tissues — Otsu keeps ~20% of transcript tiles, hysteresis
+# ~60%; see plans/2026-06-26_PLAN_tissue-mask-recalibration.md). Background mean + robust
+# SD are estimated from NON-ZERO pixels (the level-3 DAPI overview is 57-80% exact zeros
+# outside the imaged area; including them collapses the robust SD to 0 -> rsd falls back
+# to 1 -> the hysteresis cutoffs become tiny and the mask floods). Estimating from
+# non-zero pixels makes the spread genuinely per-sample. Tissue = pixels connected to a
+# confident-bright seed (bg + SEED*robSD) grown down to a floor (bg + GROW*robSD), so the
+# mask follows dim/uneven tissue without flooding background. The degeneracy guard is
+# STRUCTURAL only (foreground fraction + class separation): verified empty-field-safe in
+# plans/spikes/spike_empty_field_guard.py (an empty/noise field fails on fraction < 2% or
+# separation < 3). No absolute brightness floor on the mask — absolute DAPI quality lives
+# in usable_tissue via is_low_intensity (per-tile means, flood-immune).
+TISSUE_HYST_SEED_SD = 3.0 # seed: confident tissue at bg + 3*robSD
+TISSUE_HYST_GROW_SD = 1.0 # grow connected tissue down to bg + 1*robSD
+
+
+def _robust_background_stats(small0):
+ """``(bg_median, robust_sd)`` of the background, estimated from NON-ZERO pixels
+ (Otsu split on the non-zero values, below-threshold = background). ``(None, None)``
+ on a degenerate (empty / uniform) field. Estimating from non-zero pixels avoids the
+ hard-zero collapse that flooded the mask (see module comment above)."""
+ arr = np.asarray(small0, dtype=np.float64)
+ nz = arr[np.isfinite(arr) & (arr > 0)]
+ if nz.size == 0 or float(nz.max()) == float(nz.min()):
+ return None, None
+ t = float(threshold_otsu(nz))
+ bg = nz[nz < t]
+ if bg.size == 0:
+ bg = nz
+ med = float(np.median(bg))
+ mad = float(np.median(np.abs(bg - med)))
+ rsd = 1.4826 * mad if mad > 0 else 1.0
+ return med, rsd
+
+
+def hysteresis_tissue_mask_with_guard(small0):
+ """Background-relative hysteresis tissue mask with a degeneracy guard.
+
+ Returns ``(mask, low, ok)``:
+ - mask: boolean tissue mask — pixels connected to a ``bg + 3*robSD`` seed,
+ grown down to ``bg + 1*robSD``. Anchored to the field's own (non-zero)
+ background, so it follows dim/uneven tissue instead of flooding.
+ - low: the grow threshold (``bg + 1*robSD``); callers build the background
+ components (edge/hole distance maps) from ``small0 < low``.
+ - ok: ``False`` when the field is degenerate / has no separable tissue, so the
+ caller returns an empty mask -> ``tissue_mask_qc.status == FAIL``.
+
+ Guard (STRUCTURAL, both required): foreground fraction in ``[MIN_FG, MAX_FG]`` AND
+ class separation ``>= MIN_SEP_SD`` background-SD. No absolute brightness floor:
+ verified empty-field-safe in plans/spikes/spike_empty_field_guard.py (empty/noise
+ fields fail on fraction < 2% or separation < 3; real tissue passes). Absolute DAPI
+ quality is judged separately by usable_tissue (is_low_intensity), not by the mask.
+ """
+ bg_med, rsd = _robust_background_stats(small0)
+ if bg_med is None:
+ return np.zeros_like(small0, dtype=bool), None, False
+ low = bg_med + TISSUE_HYST_GROW_SD * rsd
+ high = bg_med + TISSUE_HYST_SEED_SD * rsd
+ arr = np.asarray(small0, dtype=np.float64)
+ mask = apply_hysteresis_threshold(arr, low, high)
+ fg_vals = arr[mask & np.isfinite(arr)]
+ bg_vals = arr[(~mask) & np.isfinite(arr)]
+ if fg_vals.size == 0 or bg_vals.size == 0:
+ return mask, low, False
+ fg_frac = float(np.mean(mask))
+ bg_std = float(bg_vals.std())
+ sep_sd = (
+ (float(fg_vals.mean()) - float(bg_vals.mean())) / bg_std
+ if bg_std > 1e-9
+ else np.inf
+ )
+ ok = (
+ TISSUE_OTSU_GUARD_MIN_FG <= fg_frac <= TISSUE_OTSU_GUARD_MAX_FG
+ and sep_sd >= TISSUE_OTSU_GUARD_MIN_SEP_SD
+ )
+ return mask, low, ok
+
+
+def compute_tissue_mask(small0, min_size_hole=1500):
+ """Shared tissue-mask logic — the single source of truth for the
+ threshold + labelling, used by generate_tissue_mask AND the
+ calculate_roi_focusscore* tissue filters so the three call sites cannot
+ drift (2026-06-23 bug fix; previously the logic was copy-pasted three times,
+ each with the `percentile 60` + `test_mask > 1` defects).
+
+ 2026-06-25: tissue threshold is now background-relative HYSTERESIS (see
+ hysteresis_tissue_mask_with_guard) instead of global Otsu, which under-captured
+ dim/sparse tissue. `test_mask > 0` keeps all foreground components.
+
+ Returns ``(whole_sample, objects, holes)``:
+ - whole_sample: labelled tissue mask. Empty when the guard rejects a
+ degenerate / faint field (no separable tissue) -> downstream
+ ``tissue_mask_qc.status == FAIL`` rather than a fabricated mask.
+ - objects: labelled background components (``label(small0 < low)``, the
+ hysteresis grow threshold) — callers that need the edge/distance map reuse
+ this.
+ - holes: labelled holes (border-cleared background components, small ones
+ removed) — callers that need the hole distance map reuse this.
+ """
+ mask, low, ok = hysteresis_tissue_mask_with_guard(small0)
+ if ok:
+ thresh1 = small0 < low # background (below the hysteresis grow threshold)
+ thresh2 = mask # tissue
+ else:
+ thresh1 = np.ones_like(small0, dtype=bool)
+ thresh2 = np.zeros_like(small0, dtype=bool)
+ objects = measure.label(thresh1)
+ noborder = clear_border(objects)
+ holes = morphology.remove_small_objects(noborder, min_size=min_size_hole)
+ small_objects = noborder ^ holes
+ test_mask = measure.label(thresh2) + small_objects
+ whole_sample = measure.label(test_mask > 0)
+ return whole_sample, objects, holes
+
+
+def compute_multistain_tissue_mask(small0, small1, small2, min_size_hole=1500):
+ """Tissue EXTENT mask combining all available morphology stains.
+
+ DAPI nuclei are sparse in some tissues (muscle fibres, brain neuropil), so a
+ DAPI-only mask under-captures them; the Boundary (membrane) and Interior
+ (cytoplasm/rRNA) stains cover those regions. This builds a per-channel
+ background-relative hysteresis mask (reusing hysteresis_tissue_mask_with_guard, so
+ each channel is guarded against its OWN background) and ORs the channels that pass
+ their guard. The union gets a foreground-fraction sanity check only (the SD-relative
+ separation test is per-channel and undefined on a boolean union).
+
+ Used for tissue EXTENT / coverage ONLY. The DAPI-only mask (compute_tissue_mask)
+ still drives the focus/blur QC, because feeding nuclei-poor tiles into the DAPI
+ focus GMM falsely reads as blurry. See plans/2026-06-26_PLAN_multistain-mask.md.
+
+ DAPI-only bundles (small1 and small2 both None) return EXACTLY
+ compute_tissue_mask(small0) — byte-identical to the DAPI-only behaviour.
+
+ Returns ``(whole_sample, objects, holes)`` like compute_tissue_mask. The union's
+ own background components (``objects``/``holes``) are rebuilt from ``~union`` (a
+ union has no single grow threshold), so edge/hole geometry reflects the combined
+ tissue extent.
+
+ NOTE: dense-intensity-artefact subtraction is intentionally NOT applied — the
+ dense-intensity mask is the brightest p97 of each channel, which includes real
+ bright tissue, so subtracting it would remove real tissue. Artefact handling is a
+ deferred follow-up (the spike measured ~1% false tissue without it).
+ """
+ if small1 is None and small2 is None:
+ return compute_tissue_mask(small0, min_size_hole=min_size_hole)
+
+ union = None
+ for ch in (small0, small1, small2):
+ if ch is None:
+ continue
+ mask, _low, ok = hysteresis_tissue_mask_with_guard(ch)
+ if ok:
+ union = mask if union is None else (union | mask)
+ if union is None:
+ union = np.zeros_like(small0, dtype=bool)
+
+ fg_frac = float(union.mean())
+ if TISSUE_OTSU_GUARD_MIN_FG <= fg_frac <= TISSUE_OTSU_GUARD_MAX_FG:
+ thresh1 = ~union # background = everything outside the combined tissue
+ thresh2 = union
+ else:
+ thresh1 = np.ones_like(small0, dtype=bool)
+ thresh2 = np.zeros_like(small0, dtype=bool)
+ objects = measure.label(thresh1)
+ noborder = clear_border(objects)
+ holes = morphology.remove_small_objects(noborder, min_size=min_size_hole)
+ small_objects = noborder ^ holes
+ test_mask = measure.label(thresh2) + small_objects
+ whole_sample = measure.label(test_mask > 0)
+ return whole_sample, objects, holes
+
+
+def _per_tile_coverage(whole_sample, x1, x2, y1, y2, downsample_factor=8):
+ """Per-tile tissue-coverage fraction for ROI tiles given in full-resolution coords,
+ computed from a labelled `whole_sample` mask at `downsample_factor` resolution via a
+ summed-area table (same logic as the grid builders). Returns a float array aligned to
+ the x1/x2/y1/y2 arrays."""
+ mask = np.asarray(whole_sample) > 0
+ h, w = mask.shape
+ ii = np.zeros((h + 1, w + 1), dtype=np.int64)
+ ii[1:, 1:] = np.cumsum(np.cumsum(mask.astype(np.int64), axis=0), axis=1)
+ r1 = np.clip(np.asarray(y1) // downsample_factor, 0, h)
+ r2 = np.clip((np.asarray(y2) - 1) // downsample_factor + 1, 0, h)
+ c1 = np.clip(np.asarray(x1) // downsample_factor, 0, w)
+ c2 = np.clip((np.asarray(x2) - 1) // downsample_factor + 1, 0, w)
+ area = np.maximum((r2 - r1) * (c2 - c1), 1)
+ s = ii[r2, c2] - ii[r1, c2] - ii[r2, c1] + ii[r1, c1]
+ return s.astype(np.float64) / area
+
+
+def generate_tissue_mask(
+ xoa_morphology_files,
+ small0,
+ small1,
+ small2,
+ threshold_percentile=60,
+ min_size_edge=500000,
+ min_size_hole=1500,
+ dense_intensity_region_percentile=97,
+ min_size_dense_intensity_region=500,
+ downsample_factor=8,
+):
+ """
+ Generate tissue masks and distance maps from morphology images (cell/segmentation independent).
+
+ Parameters:
+ -----------
+ xoa_morphology_files : list
+ List of paths to morphology image files (for compatibility, not used if small0/1/2 provided)
+ small0, small1, small2 : numpy.ndarray
+ Downsampled morphology images (DAPI, Boundary, Interior)
+ threshold_percentile : float, optional
+ Percentile for thresholding (default: 60)
+ min_size_edge : int, optional
+ Minimum size for edge objects in downsampled space (default: 500000)
+ min_size_hole : int, optional
+ Minimum size for holes in downsampled space (default: 1500)
+ dense_intensity_region_percentile : float, optional
+ Percentile for dense intensity region detection (default: 97)
+ min_size_dense_intensity_region : int, optional
+ Minimum size for dense intensity regions (default: 500)
+ downsample_factor : int, optional
+ Downsampling factor (default: 8, for level 3)
+
+ Returns:
+ --------
+ tuple
+ (whole_sample, holes, dense_intensity_regions, distance_map, distance_map2,
+ multistain_whole_sample, multistain_distance_map, multistain_distance_map2)
+ - whole_sample: Labeled DAPI tissue mask (drives focus/blur QC)
+ - holes: Labeled holes mask (DAPI)
+ - dense_intensity_regions: Labeled dense intensity regions mask
+ - distance_map: Distance to edge map (DAPI)
+ - distance_map2: Distance to nearest hole map (DAPI)
+ - multistain_whole_sample: Labeled multi-stain tissue-EXTENT mask (DAPI OR Boundary
+ OR Interior), or None on DAPI-only bundles. Drives the reported coverage / extent.
+ - multistain_distance_map / multistain_distance_map2: edge / hole distance maps for
+ the multi-stain mask (None on DAPI-only bundles)
+ """
+ # Tissue mask via the shared helper (hysteresis + degeneracy guard + `> 0`).
+ # 2026-06-23 bug fix: the old `np.percentile(small0, threshold_percentile)`
+ # (60) assumed ~40% of the field is tissue and degenerated on sparse/dim
+ # slides, and `test_mask > 1` emptied the mask on a single whole-field blob.
+ # `threshold_percentile` is retained in the signature but no longer used.
+ # See compute_tissue_mask and plans/2026-06-23_PLAN_fix-tissue-mask-bug.md.
+ whole_sample, objects, holes = compute_tissue_mask(
+ small0, min_size_hole=min_size_hole
+ )
+
+ # Edge + distance maps (generate_tissue_mask-specific; reuse `objects`/`holes`).
+ mask = morphology.remove_small_objects(objects, min_size=min_size_edge)
+ edge_sample = measure.label(mask)
+ distance_map = nsitk.signed_maurer_distance_map(edge_sample)
+ distance_map2 = nsitk.signed_maurer_distance_map(holes)
+
+ # Detecting dense intensity regions — multi-channel co-thresholding when
+ # Boundary / IntRNA channels are present, DAPI-only fallback otherwise.
+ # `small1` / `small2` are None for DAPI-only bundles (returned from
+ # `_load_morphology_channels`); blindly indexing them would crash.
+ t0 = np.percentile(small0, dense_intensity_region_percentile)
+ thresh_sum = (small0 >= t0).astype(np.int_)
+ if small1 is not None:
+ t1 = np.percentile(small1, dense_intensity_region_percentile)
+ thresh_sum = thresh_sum + (small1 >= t1).astype(np.int_)
+ if small2 is not None:
+ t2 = np.percentile(small2, dense_intensity_region_percentile)
+ thresh_sum = thresh_sum + (small2 >= t2).astype(np.int_)
+ thresh_fill = nsitk.binary_fill_holes(thresh_sum)
+ objects_art = measure.label(thresh_fill)
+ dense_intensity_regions = morphology.remove_small_objects(
+ objects_art, min_size=min_size_dense_intensity_region
+ )
+
+ # Multi-stain tissue-EXTENT mask (DAPI OR Boundary OR Interior) + its edge/hole distance
+ # maps, for the extent metrics and the §2.3 figure. None on DAPI-only bundles so the
+ # extent path falls back to the DAPI coverage and stays byte-identical.
+ if small1 is None and small2 is None:
+ ms_whole_sample = ms_distance_map = ms_distance_map2 = None
+ else:
+ ms_whole_sample, ms_objects, ms_holes = compute_multistain_tissue_mask(
+ small0, small1, small2, min_size_hole=min_size_hole
+ )
+ ms_edge = morphology.remove_small_objects(ms_objects, min_size=min_size_edge)
+ ms_distance_map = nsitk.signed_maurer_distance_map(measure.label(ms_edge))
+ ms_distance_map2 = nsitk.signed_maurer_distance_map(ms_holes)
+
+ return (
+ whole_sample,
+ holes,
+ dense_intensity_regions,
+ distance_map,
+ distance_map2,
+ ms_whole_sample,
+ ms_distance_map,
+ ms_distance_map2,
+ )
+
+
+# ---------------------------------------------------------------------------
+# Internal helpers (from focus_score_maps)
+# ---------------------------------------------------------------------------
+
+_EPSILON: float = 1e-8
+
+
+def _get_backend(
+ use_gpu: bool,
+) -> tuple[Any, Any, Any]:
+ """Return the array module, uniform_filter, and laplace for the chosen backend.
+
+ Args:
+ use_gpu: If ``True`` and CuPy is available, return GPU primitives.
+
+ Returns:
+ Tuple of ``(array_module, uniform_filter_fn, laplace_fn)``.
+
+ Raises:
+ RuntimeError: If ``use_gpu`` is ``True`` but CuPy is not installed.
+ """
+ if use_gpu:
+ if not HAS_CUPY:
+ raise RuntimeError(
+ "use_gpu=True but CuPy is not installed. "
+ "Install CuPy or set use_gpu=False."
+ )
+ return cp, cupy_uniform_filter, cupy_laplace
+ return np, scipy_uniform_filter, scipy_laplace
+
+
+def _to_device(image: NDArray[np.generic], xp: Any) -> Any:
+ """Move a numpy array to the target device (no-op for NumPy).
+
+ Args:
+ image: Input numpy array.
+ xp: Array module (``numpy`` or ``cupy``).
+
+ Returns:
+ Array on the target device.
+ """
+ if xp is np:
+ return image
+ return cp.asarray(image)
+
+
+def _to_numpy(arr: Any, xp: Any) -> NDArray[np.float32]:
+ """Move an array back to host memory as float32 numpy (no-op for NumPy).
+
+ Args:
+ arr: Array on device.
+ xp: Array module that created *arr*.
+
+ Returns:
+ numpy float32 array on the host.
+ """
+ if xp is np:
+ return np.asarray(arr, dtype=np.float32)
+ return cp.asnumpy(arr).astype(np.float32)
+
+
+def _sanitize(arr: NDArray[np.float32]) -> NDArray[np.float32]:
+ """Replace NaN and Inf values with zero.
+
+ Args:
+ arr: Input array (modified in-place).
+
+ Returns:
+ The same array with non-finite values set to zero.
+ """
+ arr[~np.isfinite(arr)] = 0.0
+ return arr
+
+
+#: Longest-axis pixel budget for arrays sent to imshow/contour/label2rgb. A panel
+#: is only ~1600 px at dpi 300, so imshow of an 86-megapixel map (12755x6738)
+#: resamples all of it and discards ~97%: measured 15.9 s vs 0.5 s downsampled
+#: first (32x). 2000 keeps every pixel a >=300-dpi panel can resolve.
+_FIG_DISPLAY_MAX_PX = 2000
+
+
+def _thumb(arr: Any, max_long: int = _FIG_DISPLAY_MAX_PX) -> Any:
+ """Downsample a 2-D array to <= ``max_long`` on its long axis, for DISPLAY only.
+
+ Float arrays are area-averaged (block mean, NaN-safe) to match matplotlib's own
+ antialiased downscale, so the rendered panel is visually identical to
+ ``imshow(arr)``. Integer/bool arrays (labels, masks) are block-MAX reduced (a mean
+ of labels is meaningless): a block stays non-zero if any pixel in it is, so a
+ feature thinner than ``step`` still renders instead of falling between strided
+ rows. No-op on already-small / non-2-D input. Never use for exported data -- only
+ at the draw call. Pair with ``extent`` of the *original* shape (see
+ ``_imshow_thumb``) so axes are unchanged.
+ """
+ a = np.asarray(arr)
+ if a.ndim != 2:
+ return a
+ # Ceil, not floor: floor overshoots the cap (12755 // 2000 = 6 leaves 2125 px;
+ # 3000 // 2000 = 1 would not downsample a 3000 px axis at all).
+ step = max(1, -(-max(a.shape) // max_long))
+ if step == 1:
+ return a
+ ny, nx = (a.shape[0] // step) * step, (a.shape[1] // step) * step
+ blocks = a[:ny, :nx].reshape(ny // step, step, nx // step, step)
+ if np.issubdtype(a.dtype, np.floating):
+ with warnings.catch_warnings(): # all-NaN blocks -> NaN, as intended
+ warnings.simplefilter("ignore", category=RuntimeWarning)
+ return np.nanmean(blocks, axis=(1, 3))
+ return blocks.max(axis=(1, 3))
+
+
+#: Target number of display bins along the long axis for Figure 5's binned focus
+#: and classification heatmaps. ~180 keeps regional signal readable while dropping
+#: the per-pixel speckle of an ~86-megapixel field, and it draws in ~1 s instead of
+#: the ~126 s an imshow of the full field cost.
+_FOCUS_HEATMAP_BINS_LONG = 180
+
+
+def _bin_nanmean(arr: Any, step: int) -> Any:
+ """Block-reduce a 2-D float array by ``step``x``step``, averaging over the finite
+ (non-NaN) pixels of each block. Blocks with no finite pixel return NaN.
+
+ NaN encodes "no tissue here" for the caller: pre-set non-tissue pixels to NaN and
+ this returns the per-bin mean over tissue only. Uses the same NaN-safe block-mean as
+ :func:`_thumb`; the all-NaN ``RuntimeWarning`` is suppressed because an empty
+ (non-tissue) bin returning NaN is the intended result, not an error.
+ """
+ a = np.asarray(arr)
+ ny = (a.shape[0] // step) * step
+ nx = (a.shape[1] // step) * step
+ blocks = a[:ny, :nx].reshape(ny // step, step, nx // step, step)
+ with warnings.catch_warnings(): # all-NaN blocks -> NaN, as intended
+ warnings.simplefilter("ignore", category=RuntimeWarning)
+ return np.nanmean(blocks, axis=(1, 3))
+
+
+def _imshow_thumb(ax: Any, arr: Any, *, rgb: Any = None, **kwargs: Any) -> Any:
+ """``ax.imshow`` of ``arr`` downsampled for speed but with the extent/axes of the
+ FULL ``arr``, so the panel is visually identical to ``ax.imshow(arr)``. ``rgb`` is
+ an optional callable (e.g. ``label2rgb``) applied to the downsampled array."""
+ a = np.asarray(arr)
+ disp = _thumb(a)
+ if rgb is not None:
+ disp = rgb(disp)
+ if a.ndim == 2:
+ kwargs.setdefault("extent", (-0.5, a.shape[1] - 0.5, a.shape[0] - 0.5, -0.5))
+ return ax.imshow(disp, **kwargs)
+
+
+def _contour_thumb(ax: Any, mask: Any, **kwargs: Any) -> Any:
+ """``ax.contour`` of a boolean/label ``mask`` downsampled to display resolution,
+ with X/Y mapped to the FULL-res pixel coordinate space so the boundary overlays an
+ ``_imshow_thumb`` panel exactly. Runs marching-squares on the ~2000-px thumbnail
+ instead of the full ~86-megapixel field (block-MAX keeps every thin boundary), so
+ the traced contour is visually identical to ``ax.contour(mask)`` at >= 300 dpi but
+ costs ~0.1 s instead of several seconds of contouring + tens of thousands of vector
+ segments."""
+ a = np.asarray(mask)
+ m = _thumb(a)
+ ny, nx = m.shape
+ h, w = a.shape
+ xs = np.linspace(0, w - 1, nx)
+ ys = np.linspace(0, h - 1, ny)
+ return ax.contour(xs, ys, m, **kwargs)
+
+
+def _kde_density_grid(kde: Any, xs: Any, ys: Any, gridsize: int = 128) -> Any:
+ """Evaluate a fitted ``gaussian_kde`` at every ``(xs, ys)`` via a coarse grid +
+ bilinear interpolation, instead of ``kde(all points)`` which is O(N * n_fit) and
+ dominates the large density-scatter figures (>100 s at N=531k, ~16 s here). Keeps
+ ALL points -- no outlier dropped -- and the density coloring is visually identical
+ (Spearman 1.000, max normalized-colour diff 6e-4 vs the exact eval)."""
+ from scipy.interpolate import RegularGridInterpolator
+
+ xs = np.asarray(xs)
+ ys = np.asarray(ys)
+ gx = np.linspace(float(xs.min()), float(xs.max()), gridsize)
+ gy = np.linspace(float(ys.min()), float(ys.max()), gridsize)
+ grid_x, grid_y = np.meshgrid(gx, gy)
+ gz = kde(np.vstack([grid_x.ravel(), grid_y.ravel()])).reshape(gridsize, gridsize)
+ interp = RegularGridInterpolator((gy, gx), gz, bounds_error=False, fill_value=None)
+ return interp(np.column_stack([ys, xs]))
+
+
+def detect_gpu_ids() -> list[int]:
+ """Detect available CUDA GPUs.
+
+ Returns:
+ List of GPU device IDs. Empty list if CuPy is not available or no GPUs
+ are detected.
+ """
+ if not HAS_CUPY:
+ logging.info("CuPy is not importable; GPU backend unavailable, using CPU.")
+ return []
+ try:
+ n_devices = cp.cuda.runtime.getDeviceCount()
+ return list(range(n_devices))
+ except Exception as exc:
+ # CuPy is installed but the CUDA runtime could not be queried (driver
+ # missing, GPU not attached to the container, init failure, ...). Surface
+ # it: otherwise this is indistinguishable from "no GPU present" and a
+ # silent CPU fallback on a GPU node looks like correct behaviour.
+ logging.warning(
+ "CuPy is installed but GPU detection failed (%s: %s); "
+ "falling back to CPU backend.",
+ type(exc).__name__,
+ exc,
+ )
+ return []
+
+
+# ---------------------------------------------------------------------------
+# Focus-map computation
+# ---------------------------------------------------------------------------
+
+
+def compute_ccfs_map(
+ image: NDArray[np.generic],
+ window_size: int = 35,
+ use_gpu: bool = False,
+ gpu_id: int = 0,
+) -> tuple[NDArray[np.float32], NDArray[np.float32]]:
+ """Compute a per-pixel CCFS (Coefficient of Contrast Focus Score) map.
+
+ The CCFS focus score at each pixel is defined as::
+
+ focus = local_var / local_mean
+
+ which is equivalent to ``std**2 / mean`` computed over a square window
+ centred on that pixel.
+
+ Args:
+ image: 2-D input image (any numeric dtype).
+ window_size: Side length of the square averaging window.
+ use_gpu: Use CuPy GPU backend when ``True``.
+ gpu_id: CUDA device ID to use when ``use_gpu=True``.
+
+ Returns:
+ Tuple of ``(focus_map, mean_map)`` — both float32 arrays with the
+ same shape as *image*.
+
+ Raises:
+ ValueError: If *image* is not 2-D or is smaller than the window in
+ either dimension.
+ RuntimeError: If *use_gpu* is ``True`` but CuPy is unavailable.
+ """
+ if image.ndim != 2:
+ raise ValueError(f"Expected a 2-D image, got shape {image.shape}.")
+ if image.shape[0] < window_size or image.shape[1] < window_size:
+ raise ValueError(
+ f"Image shape {image.shape} is smaller than window_size "
+ f"{window_size} in at least one dimension."
+ )
+
+ xp, uniform_filter, _ = _get_backend(use_gpu)
+
+ if use_gpu:
+ with cp.cuda.Device(gpu_id):
+ image_f = cp.asarray(image.astype(np.float32, copy=False))
+ local_mean = uniform_filter(image_f, size=window_size)
+ local_sq_mean = uniform_filter(image_f**2, size=window_size)
+ # Promote to float64 for subtraction to avoid catastrophic cancellation
+ local_var = (
+ local_sq_mean.astype(cp.float64) - local_mean.astype(cp.float64) ** 2
+ )
+ local_var = xp.maximum(local_var, 0.0)
+ focus_map_dev = (
+ local_var / (local_mean.astype(cp.float64) + _EPSILON)
+ ).astype(cp.float32)
+ focus_map = _to_numpy(focus_map_dev, xp)
+ mean_map = _to_numpy(local_mean, xp)
+ else:
+ image_f = image.astype(np.float32, copy=False)
+ local_mean = uniform_filter(image_f, size=window_size)
+ sq = image_f**2
+ del image_f
+ local_sq_mean = uniform_filter(sq, size=window_size)
+ del sq
+ # Promote to float64 for the subtraction to avoid catastrophic
+ # cancellation on high-intensity uint16 images.
+ # Explicit steps to limit peak memory (avoid 3+ simultaneous f64 temps).
+ local_var = local_sq_mean.astype(np.float64)
+ del local_sq_mean
+ temp_mean_f64 = local_mean.astype(np.float64)
+ local_var -= temp_mean_f64**2
+ del temp_mean_f64
+ local_var = np.maximum(local_var, 0.0)
+ # Compute focus_map = local_var / (local_mean + eps).
+ # Save mean_map first, then free local_mean before creating f64 temp.
+ mean_map = local_mean.astype(np.float32)
+ temp_denom = local_mean.astype(np.float64)
+ del local_mean
+ temp_denom += _EPSILON
+ local_var /= temp_denom
+ del temp_denom
+ focus_map = local_var.astype(np.float32)
+ del local_var
+
+ return _sanitize(focus_map), _sanitize(mean_map)
+
+
+def compute_laplacian_variance_map(
+ image: NDArray[np.generic],
+ window_size: int = 35,
+ use_gpu: bool = False,
+ gpu_id: int = 0,
+ lap_sigma: float = 1.0,
+) -> NDArray[np.float32]:
+ """Compute a per-pixel windowed Laplacian-variance focus map.
+
+ This is a *local* variant of the standard Laplacian variance focus metric
+ (Pech-Pacheco et al., 2000). Instead of computing a single global
+ variance, it produces a spatial map where each pixel holds the variance
+ of the Laplacian of Gaussian (LoG) response inside the surrounding
+ *window_size* × *window_size* neighbourhood::
+
+ lap_var = uniform_filter(lap**2) - uniform_filter(lap)**2
+
+ Using LoG (``gaussian_laplace`` with *lap_sigma*) rather than the bare
+ 3×3 Laplacian suppresses pixel-level noise and makes the metric more
+ specific to genuine edge content (Sun et al., 2004; Pertuz et al., 2013).
+
+ Variance is computed in float64 to avoid catastrophic cancellation in the
+ ``E[X²] − E[X]²`` formula, then cast back to float32.
+
+ Args:
+ image: 2-D input image (any numeric dtype).
+ window_size: Side length of the square averaging window.
+ use_gpu: Use CuPy GPU backend when ``True``.
+ gpu_id: CUDA device ID to use when ``use_gpu=True``.
+ lap_sigma: Gaussian sigma for LoG pre-smoothing (pixels).
+ Set to 0 to revert to the bare Laplacian.
+
+ Returns:
+ Float32 array with the same shape as *image*.
+
+ Raises:
+ ValueError: If *image* is not 2-D or is smaller than the window in
+ either dimension.
+ RuntimeError: If *use_gpu* is ``True`` but CuPy is unavailable.
+ """
+ if image.ndim != 2:
+ raise ValueError(f"Expected a 2-D image, got shape {image.shape}.")
+ if image.shape[0] < window_size or image.shape[1] < window_size:
+ raise ValueError(
+ f"Image shape {image.shape} is smaller than window_size "
+ f"{window_size} in at least one dimension."
+ )
+
+ xp, uniform_filter, laplace_fn = _get_backend(use_gpu)
+
+ if use_gpu:
+ with cp.cuda.Device(gpu_id):
+ image_f = cp.asarray(image.astype(np.float32, copy=False))
+ if lap_sigma > 0:
+ lap = cupy_gaussian_laplace(image_f, sigma=lap_sigma)
+ else:
+ lap = laplace_fn(image_f)
+ lap = lap.astype(cp.float64)
+ lap_mean = uniform_filter(lap, size=window_size)
+ lap_sq_mean = uniform_filter(lap * lap, size=window_size)
+ lap_var = lap_sq_mean - lap_mean**2
+ lap_var = xp.maximum(lap_var, 0.0)
+ result = _to_numpy(lap_var.astype(cp.float32), xp)
+ else:
+ image_f = image.astype(np.float32, copy=False)
+ if lap_sigma > 0:
+ lap = scipy_gaussian_laplace(image_f, sigma=lap_sigma).astype(np.float64)
+ else:
+ lap = laplace_fn(image_f).astype(np.float64)
+ del image_f
+ lap_sq = lap * lap
+ lap_mean = uniform_filter(lap, size=window_size)
+ del lap
+ lap_sq_mean = uniform_filter(lap_sq, size=window_size)
+ del lap_sq
+ np.square(lap_mean, out=lap_mean) # in-place to avoid temporary
+ lap_var = lap_sq_mean - lap_mean
+ del lap_sq_mean, lap_mean
+ lap_var = np.maximum(lap_var, 0.0)
+ result = lap_var.astype(np.float32)
+ del lap_var
+
+ return _sanitize(result)
+
+
+def _compute_channel_maps_on_gpu(
+ channel: NDArray[np.generic],
+ window_size: int,
+ gpu_id: int,
+ include_laplacian: bool = False,
+ lap_sigma: float = 1.0,
+ keep_mean_device: bool = False,
+ keep_focus_device: bool = False,
+ drop_mean_host: bool = False,
+) -> dict[str, NDArray[np.float32]]:
+ """Compute focus + mean maps for a single channel on a specific GPU.
+
+ This fused implementation opens a single device context and transfers the
+ image to the GPU once, computing CCFS (focus_map, mean_map) and optionally
+ the Laplacian variance map within the same context. This avoids redundant
+ CPU-to-GPU transfers of the full image.
+
+ Args:
+ channel: 2-D image array.
+ window_size: Convolution window size.
+ gpu_id: CUDA device ID.
+ include_laplacian: Also compute Laplacian variance map.
+ lap_sigma: Gaussian sigma for LoG pre-smoothing (0 = bare Laplacian).
+ keep_mean_device: Also return the device (CuPy) mean map under
+ ``mean_map_device`` without copying it to host, so a consumer can fold
+ a reduction (the per-ROI Otsu SNR) on the GPU. Kept alive through the
+ Laplacian phase and freed by the caller once folded.
+ keep_focus_device: Same as *keep_mean_device* for the focus map, returned
+ under ``focus_map_device``. The per-nucleus CCFS (``LabeledSumAccumulator``)
+ and centre-pixel sampling fold on-device from it, so the full focus map
+ never has to be read back for those reductions.
+ drop_mean_host: Skip the ``mean_map`` host copy entirely. Valid only when no
+ consumer reads the host mean map (all mean reductions run on-device from
+ ``mean_map_device``); this is the transfer the device-resident fold
+ eliminates. The host ``focus_map`` copy is always produced -- the Figure 5
+ heatmap reduces it in float32 on the host, which a GPU reduction cannot
+ reproduce to float rounding, so that copy stays.
+
+ Returns:
+ Dict with ``focus_map``, ``mean_map`` (unless *drop_mean_host*), optionally
+ ``lap_var_map``, and -- when the respective flag is set --
+ ``mean_map_device`` / ``focus_map_device`` (CuPy arrays on *gpu_id*).
+ """
+ if not HAS_CUPY:
+ raise RuntimeError(
+ "_compute_channel_maps_on_gpu requires CuPy but it is not installed."
+ )
+
+ pool = cp.get_default_memory_pool()
+
+ with cp.cuda.Device(gpu_id):
+ # Upload image to GPU
+ image_f = cp.asarray(channel.astype(np.float32, copy=False))
+
+ # --- CCFS (focus_map + mean_map) ---
+ # Peak memory: max 3 arrays alive at once (~10.5GB for 34K×26K images)
+ local_mean = cupy_uniform_filter(image_f, size=window_size)
+ sq = image_f**2
+ # Free image_f before allocating local_sq_mean to cap at 3 arrays
+ del image_f
+ pool.free_all_blocks()
+ local_sq_mean = cupy_uniform_filter(sq, size=window_size)
+ del sq
+ # Promote to float64 for subtraction to avoid catastrophic cancellation
+ local_var = (
+ local_sq_mean.astype(cp.float64) - local_mean.astype(cp.float64) ** 2
+ )
+ del local_sq_mean
+ cp.maximum(local_var, 0.0, out=local_var)
+ focus_map_dev = (local_var / (local_mean.astype(cp.float64) + _EPSILON)).astype(
+ cp.float32
+ )
+ del local_var
+
+ # Transfer CCFS results to CPU. The host focus map is always produced --
+ # the Figure 5 heatmap (BlockMeanAccumulator) reduces it in float32 on the
+ # host, and a GPU reduction of float32 does not reproduce numpy's float32
+ # summation order (measured ~2e-7 relative, enough to move Figure 5 pixels;
+ # see docs/failures/2026-07-26_heatmap-16px-dtype-not-summation-order.md), so
+ # that copy stays. The host mean map is skipped when every mean reduction
+ # folds on-device (drop_mean_host) -- that is the D2H transfer this path drops.
+ focus_map = cp.asnumpy(focus_map_dev).astype(np.float32)
+ result: dict[str, NDArray[np.float32]] = {"focus_map": _sanitize(focus_map)}
+ if not drop_mean_host:
+ mean_map = cp.asnumpy(local_mean).astype(np.float32)
+ result["mean_map"] = _sanitize(mean_map)
+
+ # --- Device-resident maps handed to the on-GPU consumers ---
+ # _sanitize each device array in place so it matches the host copy above
+ # bit-for-bit: local_mean/focus are uniform_filter derivatives of a finite
+ # image so this is a no-op in practice, but it keeps the on-device and host
+ # paths structurally identical. Each kept array adds one float32 map of VRAM,
+ # held past this block (through the Laplacian phase); the caller drops it
+ # after folding.
+ if keep_focus_device:
+ focus_map_dev[~cp.isfinite(focus_map_dev)] = 0.0
+ result["focus_map_device"] = focus_map_dev
+ else:
+ del focus_map_dev
+ if keep_mean_device:
+ # Hand RoiOtsuSnrAccumulator / LabeledSumAccumulator the device mean map
+ # so their per-ROI / per-label reductions fold on the GPU (no full-map
+ # D2H copy for those reductions).
+ local_mean[~cp.isfinite(local_mean)] = 0.0
+ result["mean_map_device"] = local_mean
+ else:
+ del local_mean
+ pool.free_all_blocks()
+ if include_laplacian:
+ image_f = cp.asarray(channel.astype(np.float32, copy=False))
+ if lap_sigma > 0:
+ lap = cupy_gaussian_laplace(image_f, sigma=lap_sigma)
+ else:
+ lap = cupy_laplace(image_f)
+ del image_f
+ pool.free_all_blocks()
+ lap = lap.astype(cp.float64)
+ lap_mean = cupy_uniform_filter(lap, size=window_size)
+ sq = lap * lap
+ del lap
+ pool.free_all_blocks()
+ lap_sq_mean = cupy_uniform_filter(sq, size=window_size)
+ del sq
+ lap_var = lap_sq_mean - lap_mean**2
+ del lap_sq_mean, lap_mean
+ cp.maximum(lap_var, 0.0, out=lap_var)
+ lap_var_np = cp.asnumpy(lap_var).astype(np.float32)
+ del lap_var
+ pool.free_all_blocks()
+ result["lap_var_map"] = _sanitize(lap_var_np)
+
+ return result
+
+
+# Shared TIFF-tile decode pool. page.decode (imagecodecs) releases the GIL, so
+# decoding the compressed blobs across a pool sized to the CPU count saturates the
+# cores that would otherwise idle while the 4 GPU workers wait on a single-threaded
+# decoder -- decode was the tile-pass wall (measured). Sized to the CPU count so the
+# total decode concurrency is bounded by cores regardless of GPU count (no
+# oversubscription). Lazily created; lives for the process (cleaned up at exit).
+_DECODE_POOL: ThreadPoolExecutor | None = None
+_DECODE_POOL_LOCK = threading.Lock()
+
+
+def _get_decode_pool() -> ThreadPoolExecutor:
+ global _DECODE_POOL
+ if _DECODE_POOL is None:
+ with _DECODE_POOL_LOCK:
+ if _DECODE_POOL is None:
+ _DECODE_POOL = ThreadPoolExecutor(
+ max_workers=max(2, os.cpu_count() or 4),
+ thread_name_prefix="tiff-decode",
+ )
+ return _DECODE_POOL
+
+
+class _LazyTiffChannel:
+ """Lazy 2-D view into one channel page of a TIFF file.
+
+ Supports ``[y0:y1, x0:x1]`` slicing. For tiled TIFFs (production
+ Xenium images) only the overlapping tiles are decoded, keeping I/O
+ minimal. For non-tiled TIFFs (e.g. test images written by
+ ``tifffile.imwrite``) the full page is cached on first access.
+
+ Thread-safe: a lock serialises file-handle reads so that
+ ``_process_tile_on_gpu`` can call ``[slice]`` from multiple threads.
+ """
+
+ def __init__(self, page, source: tuple[str, int] | None = None):
+ """Accept a ``TiffPage`` or ``TiffFrame``.
+
+ ``TiffFrame`` (non-first pages in a multi-page TIFF) lacks
+ ``imagelength`` / ``is_tiled`` — we fall back to ``.shape`` and
+ always use the cached-full-read path for frames.
+
+ Args:
+ page: The ``TiffPage`` / ``TiffFrame`` to wrap.
+ source: ``(path, page_index)`` this channel can be reopened from in
+ another process. A live ``TiffPage`` holds an OS file handle
+ and a ``threading.Lock``, so it cannot be pickled; the process
+ pool re-opens the channel from this descriptor instead.
+ ``None`` means the channel is not independently reopenable
+ (e.g. it is backed by a cached full read), which disqualifies
+ process-mode tiling.
+ """
+ import threading
+
+ self._page = page
+ # .shape works on both TiffPage and TiffFrame
+ self.shape: tuple[int, int] = (page.shape[0], page.shape[1])
+ self._lock = threading.Lock()
+ self._cached_data: NDArray | None = None
+ # TiffFrame has no is_tiled; treat as non-tiled (fallback path)
+ self._is_tiled: bool = getattr(page, "is_tiled", False)
+ # Profiling accumulators (guarded by _lock). Split the tile pass so we can
+ # tell IO-bound from decode-bound from GPU-convolve-bound: convolve+upload is
+ # then (compute_seconds - read_seconds - decode_seconds) at the caller.
+ self._read_seconds: float = 0.0
+ self._decode_seconds: float = 0.0
+ self._read_bytes: int = 0
+ # tifffile may lazily initialise its decoder on first use; warm it once under
+ # the read lock (see _read_region_tiled) before any off-lock parallel decode.
+ self._decoder_warmed: bool = False
+ self._source: tuple[str, int] | None = source
+
+ # ------------------------------------------------------------------
+ # public API
+ # ------------------------------------------------------------------
+
+ def __getitem__(self, key):
+ if not isinstance(key, tuple) or len(key) != 2:
+ raise ValueError("_LazyTiffChannel supports only [y0:y1, x0:x1] slicing")
+ yslice, xslice = key
+ y0, y1, _ = yslice.indices(self.shape[0])
+ x0, x1, _ = xslice.indices(self.shape[1])
+ height = y1 - y0
+ width = x1 - x0
+ if height <= 0 or width <= 0:
+ return np.empty((0, 0), dtype=self._page.dtype)
+ if self._is_tiled:
+ return self._read_region_tiled(y0, x0, height, width)
+ return self._read_region_fallback(y0, x0, height, width)
+
+ # ------------------------------------------------------------------
+ # tiled path — decode only the tiles that overlap the request
+ # ------------------------------------------------------------------
+
+ def _read_region_tiled(self, y0: int, x0: int, height: int, width: int) -> NDArray:
+ page = self._page
+ tw, th = page.tilewidth, page.tilelength
+ im_w, im_h = page.imagewidth, page.imagelength
+
+ y1 = min(y0 + height, im_h)
+ x1 = min(x0 + width, im_w)
+
+ tile_y0 = y0 // th
+ tile_x0 = x0 // tw
+ tile_y1 = int(np.ceil(y1 / th))
+ tile_x1 = int(np.ceil(x1 / tw))
+ tiles_per_row = int(np.ceil(im_w / tw))
+
+ buf_h = (tile_y1 - tile_y0) * th
+ buf_w = (tile_x1 - tile_x0) * tw
+ out = np.empty((buf_h, buf_w), dtype=page.dtype)
+
+ fh = page.parent.filehandle
+ jpegtables = page.jpegtables
+
+ def _decode_blob(blob: tuple[int, int, int, bytes]) -> None:
+ index, oi, oj, data = blob
+ tile_arr, _indices, _shape = page.decode(data, index, jpegtables=jpegtables)
+ out[oi : oi + th, oj : oj + tw] = tile_arr.squeeze()
+
+ # Phase 1: read the compressed tile blobs under the lock. Only seek+read touch
+ # the shared file handle, so the critical section is just the IO (measured at
+ # ~1-3 s/channel -- negligible). Decoding is pulled OUT of the lock below.
+ blobs: list[tuple[int, int, int, bytes]] = [] # (index, oi, oj, data)
+ read_s = 0.0
+ read_bytes = 0
+ warm_decode_s = 0.0
+ with self._lock:
+ for ti in range(tile_y0, tile_y1):
+ for tj in range(tile_x0, tile_x1):
+ index = ti * tiles_per_row + tj
+ offset = page.dataoffsets[index]
+ bytecount = page.databytecounts[index]
+ _t_io = time.perf_counter()
+ fh.seek(offset)
+ data = fh.read(bytecount)
+ read_s += time.perf_counter() - _t_io
+ read_bytes += bytecount
+ blobs.append(
+ (index, (ti - tile_y0) * th, (tj - tile_x0) * tw, data)
+ )
+ self._read_seconds += read_s
+ self._read_bytes += read_bytes
+ # Warm tifffile's (possibly lazily-initialised) decoder ONCE, single-
+ # threaded under the lock, so the concurrent off-lock decodes below cannot
+ # race on a first-use init. Assembles the first tile of the first call.
+ if not self._decoder_warmed and blobs:
+ _t_warm = time.perf_counter()
+ _decode_blob(blobs[0])
+ warm_decode_s = time.perf_counter() - _t_warm
+ blobs = blobs[1:]
+ self._decoder_warmed = True
+
+ # Phase 2: decode + assemble OFF the lock. page.decode is a pure function of the
+ # read bytes and releases the GIL (imagecodecs), so across the GPU-worker
+ # threads whole strips decode concurrently instead of single-file behind the
+ # reader lock -- the serialized decode was ~91% of the tile-pass wall clock on a
+ # 5.5 GP sample (run bVC3fObZxHK6p). Each tile writes a disjoint region of
+ # `out`, so the assembly is race-free.
+ _t_dec = time.perf_counter()
+ if len(blobs) > 1:
+ # Decode across the shared pool: page.decode releases the GIL, so the
+ # strip's tiles decode on the otherwise-idle cores instead of single-file.
+ # Exhaust the map generator so writes complete and errors propagate.
+ for _ in _get_decode_pool().map(_decode_blob, blobs):
+ pass
+ else:
+ for blob in blobs:
+ _decode_blob(blob)
+ decode_s = warm_decode_s + (time.perf_counter() - _t_dec)
+ with self._lock:
+ self._decode_seconds += decode_s
+
+ ry0 = y0 - tile_y0 * th
+ rx0 = x0 - tile_x0 * tw
+ return out[ry0 : ry0 + (y1 - y0), rx0 : rx0 + (x1 - x0)]
+
+ # ------------------------------------------------------------------
+ # fallback path — cache full page (non-tiled / test images)
+ # ------------------------------------------------------------------
+
+ def _read_region_fallback(
+ self, y0: int, x0: int, height: int, width: int
+ ) -> NDArray:
+ with self._lock:
+ if self._cached_data is None:
+ self._cached_data = self._page.asarray()
+ return self._cached_data[y0 : y0 + height, x0 : x0 + width]
+
+
+def _open_morphology_lazy(
+ xoa_morphology_files,
+ *,
+ level: int = 0,
+) -> tuple[list, tuple[int, int]]:
+ """Open morphology channels as lazy TIFF page wrappers (no pixel data loaded).
+
+ Uses ``tifffile.TiffFile`` directly instead of the zarr store interface,
+ avoiding the zarr v3 dependency introduced by ``tifffile.imread(aszarr=True)``.
+
+ Args:
+ xoa_morphology_files: List of OME-TIFF file paths.
+ level: Resolution level to open (0 = full resolution).
+
+ Returns:
+ (channels, image_shape) where channels is [dapi, boundary, intrna]
+ (None for missing channels) and image_shape is (height, width).
+ """
+ tiff_handles: list[tifffile.TiffFile] = [] # prevent GC
+
+ primary_tif = tifffile.TiffFile(str(xoa_morphology_files[0]))
+ tiff_handles.append(primary_tif)
+
+ # Use PHYSICAL pages in the file, not OME series pages.
+ # Xenium multi-file OME-TIFFs report series.shape=(4,H,W) referencing
+ # all files, but each file has only 1 physical page. series.pages would
+ # include stubs for data in other files that decode to zeros.
+ n_physical = len(primary_tif.pages)
+ if level > 0:
+ # For pyramid levels, use the series API to navigate levels
+ series = primary_tif.series[0]
+ if level < len(series.levels):
+ pages = list(series.levels[level].pages)
+ else:
+ pages = list(primary_tif.pages)
+ else:
+ pages = list(primary_tif.pages)
+
+ n_pages = len(pages)
+
+ # Multi-channel single file: truly multi-page TIFF (each page = channel)
+ # Only enter this path if the file physically contains multiple pages
+ if n_pages >= 2 and n_physical >= 2:
+ # Each page is one channel (C, H, W layout across pages).
+ # At level 0 `pages` is the physical page list, so the list index is the
+ # physical page index and each channel can be reopened elsewhere by
+ # (path, page_index). Pyramid levels come from the series API, where
+ # that identity does not hold — leave those non-reopenable.
+ primary_path = str(xoa_morphology_files[0])
+
+ def _page_source(idx: int) -> tuple[str, int] | None:
+ return (primary_path, idx) if level == 0 else None
+
+ dapi = _LazyTiffChannel(pages[0], source=_page_source(0))
+ shape = dapi.shape
+ boundary = (
+ _LazyTiffChannel(pages[1], source=_page_source(1)) if n_pages > 1 else None
+ )
+ intrna = (
+ _LazyTiffChannel(pages[2], source=_page_source(2)) if n_pages > 2 else None
+ )
+ # Attach TiffFile handles to prevent garbage collection
+ dapi._tiff_handles = tiff_handles # type: ignore[attr-defined]
+ return [dapi, boundary, intrna], shape
+
+ # Single-channel (or single-page) primary
+ # Check if the single page is actually multi-channel (C, H, W) in one page
+ page0 = pages[0]
+ arr_shape = page0.shape # could be (H, W) or (C, H, W)
+ if len(arr_shape) == 3:
+ # Multi-channel packed into one page — fall back to cached full read
+ full = page0.asarray()
+ if arr_shape[0] <= 4 and arr_shape[1] > 16 and arr_shape[2] > 16:
+ # Channel-first (C, H, W)
+ shape = (arr_shape[1], arr_shape[2])
+ ch_axis = 0
+ elif arr_shape[2] <= 4 and arr_shape[0] > 16 and arr_shape[1] > 16:
+ # Channel-last (H, W, C)
+ shape = (arr_shape[0], arr_shape[1])
+ ch_axis = 2
+ else:
+ shape = (arr_shape[1], arr_shape[2])
+ ch_axis = 0
+ n_ch = arr_shape[ch_axis]
+ dapi = _LazyTiffChannel.__new__(_LazyTiffChannel)
+ dapi._page = page0
+ dapi.shape = shape
+ import threading
+
+ dapi._lock = threading.Lock()
+ dapi._source = None # cached full read: not reopenable in another process
+ dapi._cached_data = np.take(full, 0, axis=ch_axis)
+ boundary_ch = _LazyTiffChannel.__new__(_LazyTiffChannel) if n_ch > 1 else None
+ if boundary_ch is not None:
+ boundary_ch._page = page0
+ boundary_ch.shape = shape
+ boundary_ch._lock = threading.Lock()
+ # cached full read: not reopenable in another process
+ boundary_ch._source = None
+ boundary_ch._cached_data = np.take(full, 1, axis=ch_axis)
+ intrna_ch = _LazyTiffChannel.__new__(_LazyTiffChannel) if n_ch > 2 else None
+ if intrna_ch is not None:
+ intrna_ch._page = page0
+ intrna_ch.shape = shape
+ intrna_ch._lock = threading.Lock()
+ # cached full read: not reopenable in another process
+ intrna_ch._source = None
+ intrna_ch._cached_data = np.take(full, 2, axis=ch_axis)
+ dapi._tiff_handles = tiff_handles # type: ignore[attr-defined]
+ return [dapi, boundary_ch, intrna_ch], shape
+
+ # Truly single-channel 2-D page
+ dapi = _LazyTiffChannel(
+ page0, source=(str(xoa_morphology_files[0]), 0) if level == 0 else None
+ )
+ shape = dapi.shape
+ boundary = None
+ intrna = None
+
+ if len(xoa_morphology_files) > 1 and Path(xoa_morphology_files[1]).exists():
+ try:
+ t = tifffile.TiffFile(str(xoa_morphology_files[1]))
+ tiff_handles.append(t)
+ boundary = _LazyTiffChannel(
+ t.pages[0],
+ source=(str(xoa_morphology_files[1]), 0) if level == 0 else None,
+ )
+ except Exception as e:
+ logging.warning(
+ "Boundary TIFF open failed for %s: %s", xoa_morphology_files[1], e
+ )
+
+ if len(xoa_morphology_files) > 2 and Path(xoa_morphology_files[2]).exists():
+ try:
+ t = tifffile.TiffFile(str(xoa_morphology_files[2]))
+ tiff_handles.append(t)
+ intrna = _LazyTiffChannel(
+ t.pages[0],
+ source=(str(xoa_morphology_files[2]), 0) if level == 0 else None,
+ )
+ except Exception as e:
+ logging.warning(
+ "IntRNA TIFF open failed for %s: %s", xoa_morphology_files[2], e
+ )
+
+ # Attach TiffFile handles to prevent garbage collection
+ dapi._tiff_handles = tiff_handles # type: ignore[attr-defined]
+ return [dapi, boundary, intrna], shape
+
+
+def _compute_adaptive_tile_size(
+ height: int,
+ width: int,
+ n_gpus: int,
+ gpu_mem_bytes: int | None = None,
+ target_utilization: float = 0.65,
+ min_tiles_per_gpu: int = 2,
+) -> int:
+ """Compute tile size that maximizes GPU memory utilization.
+
+ Balances two constraints:
+ 1. Each tile's peak GPU memory stays within target_utilization of VRAM.
+ 2. Total tiles >= n_gpus * min_tiles_per_gpu for load balancing.
+
+ Args:
+ height: Image height in pixels.
+ width: Image width in pixels.
+ n_gpus: Number of available GPUs.
+ gpu_mem_bytes: Total GPU VRAM in bytes. Auto-detected if None.
+ target_utilization: Fraction of VRAM to target per tile (default 0.65).
+ min_tiles_per_gpu: Minimum tiles per GPU for load balancing.
+
+ Returns:
+ Tile size in pixels (square tiles).
+ """
+ if gpu_mem_bytes is None:
+ gpu_mem_bytes = cp.cuda.Device(0).mem_info[1]
+
+ # Peak memory per pixel: ~6 concurrent float32 arrays
+ BYTES_PER_PIXEL = 6 * 4 # 24 bytes
+
+ # Max tile dim from GPU memory constraint
+ max_pixels = int(gpu_mem_bytes * target_utilization / BYTES_PER_PIXEL)
+ max_tile_dim = int(math.sqrt(max_pixels))
+
+ # Min tiles needed for load balancing
+ min_tiles = max(n_gpus * min_tiles_per_gpu, 1)
+
+ # Minimum useful tile dimension (must fit the convolution window)
+ min_dim = 256
+
+ # Compute tile_size that produces at least min_tiles
+ # Start from max_tile_dim and shrink until we have enough tiles
+ tile_size = max_tile_dim
+ while tile_size > min_dim:
+ n_y = math.ceil(height / tile_size)
+ n_x = math.ceil(width / tile_size)
+ if n_y * n_x >= min_tiles:
+ break
+ tile_size = int(tile_size * 0.7) # shrink by 30%
+
+ # Clamp to minimum useful dimension
+ tile_size = max(min_dim, tile_size)
+
+ # If image is smaller than tile_size, just use image dims
+ if height <= tile_size and width <= tile_size:
+ tile_size = max(height, width)
+
+ return tile_size
+
+
+def _compute_adaptive_strip_height(
+ height: int,
+ width: int,
+ n_gpus: int,
+ gpu_mem_bytes: int | None = None,
+ target_utilization: float = 0.65,
+ min_strips_per_gpu: int = 2,
+) -> int:
+ """Rows per full-width strip that fit the VRAM budget.
+
+ Full-width strips beat square tiles here for two measured reasons:
+
+ * **Contiguous writes.** A square tile's write region is a rectangle, so a
+ 25,238-wide tile inside a 102,045-wide plane is 25,238 separate
+ non-contiguous row segments. A strip's write region is one byte range. On
+ the Fusion (FUSE/S3) work directory the scattered pattern was pathological
+ -- 6 tiles took ~3 h against a ~46 min whole-run baseline, see
+ docs/failures/2026-07-24_imageqc-mmap-over-fusion.md.
+ * **Half the halo.** A strip needs overlap on top and bottom only, not all
+ four edges, so less redundant convolution.
+
+ At 48 GB VRAM and a 102,045 px width this gives ~12,700 rows, about 5 strips
+ for the 5.5 gigapixel reference image.
+ """
+ if gpu_mem_bytes is None:
+ gpu_mem_bytes = cp.cuda.Device(0).mem_info[1]
+
+ # Same per-pixel budget as the square-tile sizing: ~6 concurrent float32
+ # arrays live at once inside _compute_channel_maps_on_gpu.
+ bytes_per_pixel = 6 * 4
+ max_pixels = int(gpu_mem_bytes * target_utilization / bytes_per_pixel)
+ rows = max(1, max_pixels // max(width, 1))
+
+ # Enough strips to keep every GPU busy, when the image is tall enough.
+ wanted = max(n_gpus * min_strips_per_gpu, 1)
+ if rows * wanted > height:
+ rows = max(1, height // wanted)
+
+ # Round the strip *count* up to a multiple of the GPU count, then re-derive the
+ # height from it. Two problems this fixes, both measured on run 3nkeHOEV1ONlbK
+ # (102045 rows, 4 GPUs): height // wanted gave 12755 rows, so
+ # ceil(102045/12755) = 9 strips -- eight full ones plus a 5-row sliver -- and 9
+ # strips across 4 GPUs left occupancy at 74% / 88% / 100% / 100%. Deriving the
+ # height from a count of 8 gives 12756 rows, no sliver, and two strips per GPU.
+ # Increasing the count only ever shrinks strips, so the VRAM bound still holds.
+ n_strips = max(1, -(-height // rows))
+ n_strips = min(((n_strips + n_gpus - 1) // n_gpus) * n_gpus, height)
+ rows = max(1, -(-height // n_strips))
+
+ return int(min(rows, height))
+
+
+def _compute_tile_grid(
+ height: int,
+ width: int,
+ tile_size: int = 8192,
+ overlap: int = 17,
+ tile_width: int | None = None,
+) -> list[dict[str, int]]:
+ """Compute overlapping tile coordinates for tiled convolution.
+
+ Each tile has a "write" region (non-overlapping, covers full image) and
+ a "read" region (expanded by overlap, clipped to image bounds).
+
+ Args:
+ height: Image height in pixels.
+ width: Image width in pixels.
+ tile_size: Core tile height (before overlap).
+ overlap: Border pixels to add for convolution safety.
+ tile_width: Core tile width; defaults to *tile_size* (square tiles). Pass
+ the image width for full-width row strips, whose write region is one
+ contiguous byte range.
+
+ Returns:
+ List of tile spec dicts with keys: read_y0, read_y1, read_x0, read_x1,
+ write_y0, write_y1, write_x0, write_x1, trim_top, trim_bottom,
+ trim_left, trim_right.
+ """
+ tiles = []
+ step_x = int(tile_width) if tile_width else tile_size
+ for y in range(0, height, tile_size):
+ for x in range(0, width, step_x):
+ wy0 = y
+ wy1 = min(y + tile_size, height)
+ wx0 = x
+ wx1 = min(x + step_x, width)
+
+ ry0 = max(0, wy0 - overlap)
+ ry1 = min(height, wy1 + overlap)
+ rx0 = max(0, wx0 - overlap)
+ rx1 = min(width, wx1 + overlap)
+
+ tiles.append(
+ {
+ "read_y0": ry0,
+ "read_y1": ry1,
+ "read_x0": rx0,
+ "read_x1": rx1,
+ "write_y0": wy0,
+ "write_y1": wy1,
+ "write_x0": wx0,
+ "write_x1": wx1,
+ "trim_top": wy0 - ry0,
+ "trim_bottom": ry1 - wy1,
+ "trim_left": wx0 - rx0,
+ "trim_right": rx1 - wx1,
+ }
+ )
+ return tiles
+
+
+# ---------------------------------------------------------------------------
+# Host memory instrumentation
+# ---------------------------------------------------------------------------
+
+_GIB = float(1024**3)
+
+# High-water marks across the run, reported in the final summary.
+_MEM_PEAK: dict[str, float] = {"working_set": 0.0, "rss": 0.0}
+
+
+# (usage file, stat file, inactive_file key, active_file key) for cgroup v2 then
+# v1. AWS Batch nodes run either depending on the ECS AMI, so try both.
+_CGROUP_SOURCES = (
+ (
+ "/sys/fs/cgroup/memory.current",
+ "/sys/fs/cgroup/memory.stat",
+ "inactive_file",
+ "active_file",
+ ),
+ (
+ "/sys/fs/cgroup/memory/memory.usage_in_bytes",
+ "/sys/fs/cgroup/memory/memory.stat",
+ "total_inactive_file",
+ "total_active_file",
+ ),
+)
+
+
+#: Kernel-maintained high-water usage for the whole cgroup, v2 then v1. Unlike the
+#: sampled figures this is continuous and covers every process, so it cannot miss a
+#: peak that falls between samples. It does include page cache, which makes it an
+#: upper bound rather than an OOM-relevant working set.
+_CGROUP_PEAK_PATHS = (
+ "/sys/fs/cgroup/memory.peak",
+ "/sys/fs/cgroup/memory/memory.max_usage_in_bytes",
+)
+
+
+def _cgroup_peak() -> float | None:
+ """Peak cgroup usage in bytes, or None where the counter is absent."""
+ for path in _CGROUP_PEAK_PATHS:
+ try:
+ with open(path) as fh:
+ return float(fh.read().strip())
+ except (OSError, ValueError):
+ continue
+ return None
+
+
+def _cgroup_memory() -> tuple[float, float] | None:
+ """``(working_set_bytes, page_cache_bytes)`` for this cgroup, or None.
+
+ The usage counter includes reclaimable page cache, and the disk-backed focus
+ planes deliberately generate a lot of it — reading usage alone would show a
+ large number and wrongly suggest nothing improved. The quantity that
+ actually drives an OOM kill is the working set, ``usage - inactive_file``.
+ Both are reported so the two are never confused when sizing the
+ process_gpu_qc memory ladder.
+ """
+ for usage_path, stat_path, inactive_key, active_key in _CGROUP_SOURCES:
+ try:
+ with open(usage_path) as fh:
+ usage = float(fh.read().strip())
+ inactive_file = 0.0
+ active_file = 0.0
+ with open(stat_path) as fh:
+ for line in fh:
+ key, _, value = line.partition(" ")
+ if key == inactive_key:
+ inactive_file = float(value)
+ elif key == active_key:
+ active_file = float(value)
+ except (OSError, ValueError):
+ continue
+ return max(usage - inactive_file, 0.0), inactive_file + active_file
+ return None
+
+
+# Cgroup memory hard limit, v2 then v1. Paired with `_cgroup_memory`'s working-set
+# read to size the figure pool by available RAM headroom, not cores alone.
+_CGROUP_LIMIT_PATHS = (
+ "/sys/fs/cgroup/memory.max", # v2
+ "/sys/fs/cgroup/memory/memory.limit_in_bytes", # v1
+)
+
+
+def _cgroup_memory_limit() -> float | None:
+ """Cgroup memory hard limit in bytes, or None if unlimited/unreadable.
+
+ v2 reports the literal string ``max`` when unlimited; v1 reports a sentinel
+ close to ``2**63``, so an implausibly large limit is also treated as
+ unlimited (a real Batch node is <=1 TB, well under ``2**60``).
+ """
+ for path in _CGROUP_LIMIT_PATHS:
+ try:
+ with open(path) as fh:
+ raw = fh.read().strip()
+ except OSError:
+ continue
+ if raw == "max":
+ return None
+ try:
+ value = float(raw)
+ except ValueError:
+ continue
+ if value >= float(1 << 60):
+ return None
+ return value
+ return None
+
+
+def _tree_rss() -> float | None:
+ """RSS summed over every process in this PID namespace, in bytes.
+
+ ``VmHWM`` from /proc/self/status is the main process only, and the figure phase
+ forks up to ``_FIGURE_WORKERS_MAX`` children whose matplotlib buffers land on top
+ of the parent's resident set. On run 3V53J4ewZt1vsU that made the difference
+ between the 33.4 GB this module reported and the 100.6 GB Tower measured for the
+ same task -- and the summary line told the reader to size the memory request from
+ the smaller number, which would under-provision by 3x.
+
+ Restricted to *our own* descendants rather than all of /proc. Summing everything
+ is right in a container, whose PID namespace holds only our processes, but these
+ modules also run directly on shared servers where it would silently add other
+ users' processes to the total.
+
+ Cgroup ``memory.peak`` would also cover children, but it counts reclaimable page
+ cache — the confusion ``_cgroup_memory`` exists to avoid.
+ """
+ ppid_of: dict[int, int] = {}
+ rss_of: dict[int, float] = {}
+ try:
+ entries = os.listdir("/proc")
+ except OSError:
+ return None
+ for entry in entries:
+ if not entry.isdigit():
+ continue
+ pid = int(entry)
+ try:
+ with open(f"/proc/{pid}/status") as fh:
+ ppid = None
+ rss = None
+ for line in fh:
+ if line.startswith("PPid:"):
+ ppid = int(line.split()[1])
+ elif line.startswith("VmRSS:"):
+ rss = float(line.split()[1]) * 1024.0
+ if ppid is not None and rss is not None:
+ break
+ except (OSError, ValueError, IndexError):
+ continue # exited between listdir and read, or not readable
+ if ppid is None:
+ continue
+ ppid_of[pid] = ppid
+ rss_of[pid] = rss or 0.0
+
+ me = os.getpid()
+ if me not in rss_of:
+ return None
+ # Walk parents to decide membership; the tree is shallow (pool workers are
+ # direct children), so this stays cheap.
+ total = 0.0
+ for pid in rss_of:
+ walker = pid
+ for _ in range(64): # bounded: never loop on a malformed parent chain
+ if walker == me:
+ total += rss_of[pid]
+ break
+ nxt = ppid_of.get(walker)
+ if nxt is None or nxt == walker or nxt <= 1:
+ break
+ walker = nxt
+ return total
+
+
+def _process_rss() -> float | None:
+ """Peak RSS of this process in bytes (VmHWM), excluding children."""
+ try:
+ with open("/proc/self/status") as fh:
+ for line in fh:
+ if line.startswith("VmHWM:"):
+ return float(line.split()[1]) * 1024.0
+ except (OSError, ValueError, IndexError):
+ return None
+ return None
+
+
+def _log_mem(stage: str) -> None:
+ """Log host memory at a stage boundary and update the high-water marks.
+
+ Instrumenting in-code rather than diagnosing after the fact: a Tower run
+ only reports the peak for the whole task, which cannot say *which* stage
+ drove it.
+ """
+ parts = []
+ tree = _tree_rss()
+ if tree is not None and tree > _MEM_PEAK.get("tree_rss", 0.0):
+ _MEM_PEAK["tree_rss"] = tree
+ cgroup = _cgroup_memory()
+ if cgroup is not None:
+ working_set, page_cache = cgroup
+ _MEM_PEAK["working_set"] = max(_MEM_PEAK["working_set"], working_set)
+ parts.append(
+ f"working_set={working_set / _GIB:.1f}GB "
+ f"(+{page_cache / _GIB:.1f}GB reclaimable page cache)"
+ )
+ rss = _process_rss()
+ if rss is not None:
+ _MEM_PEAK["rss"] = max(_MEM_PEAK["rss"], rss)
+ parts.append(f"peak_rss={rss / _GIB:.1f}GB")
+ if parts:
+ logging.info(f" [MEM] {stage}: {' '.join(parts)}")
+
+
+def _log_mem_summary() -> None:
+ """Final high-water summary — the number that sizes the memory request."""
+ tree = _MEM_PEAK.get("tree_rss", 0.0)
+ cgroup_peak = _cgroup_peak()
+ parts = [
+ f"cgroup_peak={cgroup_peak / _GIB:.1f}GB"
+ if cgroup_peak is not None
+ else "cgroup_peak=n/a",
+ f"tree_rss={tree / _GIB:.1f}GB",
+ f"working_set={_MEM_PEAK['working_set'] / _GIB:.1f}GB",
+ f"main_process_rss={_MEM_PEAK['rss'] / _GIB:.1f}GB",
+ ]
+ logging.info("[MEM] PEAK " + " ".join(parts))
+ # Each of these measures something different, and three of the four can understate
+ # the task's real high-water mark. Spelling out which is which, because an earlier
+ # version of this summary pointed at one number and was wrong twice over:
+ # cgroup_peak continuous, all processes, but includes page cache
+ # tree_rss all processes, but SAMPLED at stage boundaries -- it can miss
+ # a peak between samples, and has come in *below*
+ # main_process_rss for exactly that reason
+ # working_set usage - inactive_file, the OOM-relevant quantity, also sampled
+ # main_process_rss VmHWM: continuous, but this process only, so it excludes the
+ # forked figure workers
+ # The authoritative figure for sizing is the peakRss the Nextflow trace reports for
+ # the task, which polls the whole process tree; these are for attributing cost to a
+ # stage, which the trace cannot do.
+ logging.info(
+ "[MEM] For sizing use the Tower/trace peakRss for the task; the figures above "
+ "attribute cost to stages and each understates the total in a different way "
+ "(see the comment in _log_mem_summary)."
+ )
+
+
+# Absolute upper bound on concurrent forked figure workers. Each child inherits
+# the parent's address space copy-on-write, so raising the count re-holds only
+# each child's OWN matplotlib render buffers (~0.5-1 GB for a full-res 86 Mpx
+# imshow), not the shared planes -- but N of those still land on the same cgroup,
+# so `_figure_worker_limit` gates the pool on memory headroom, not cores alone.
+_FIGURE_WORKERS_MAX = 8
+
+#: Peak extra RAM a single figure child adds on top of the shared copy-on-write
+#: planes, dominated by matplotlib's render buffers for a full-resolution imshow.
+#: Deliberately generous (measured ~0.5-1 GB) so the memory gate errs toward
+#: fewer workers rather than an OOM kill.
+_FIGURE_PER_WORKER_GB = 2.0
+
+
+def _cgroup_cpu_quota() -> int | None:
+ """Cores the cgroup actually permits, or None if unlimited/unreadable.
+
+ ``os.cpu_count()`` reports the host: a container given 30 of a 48-vCPU
+ instance's cores still sees 48, so sizing a pool from it oversubscribes.
+ """
+ try:
+ with open("/sys/fs/cgroup/cpu.max") as fh:
+ raw_quota, raw_period = fh.read().split()
+ except (OSError, ValueError):
+ return None
+ if raw_quota == "max":
+ return None
+ try:
+ return max(1, int(int(raw_quota) / int(raw_period)))
+ except (ValueError, ZeroDivisionError):
+ return None
+
+
+def _figure_worker_limit(n_tasks: int) -> int:
+ """Concurrent figure-rendering processes to allow for ``n_tasks`` figures.
+
+ Gated on BOTH the cgroup CPU quota and the cgroup memory headroom, because a
+ high-core node can still OOM if every core forks a full-res-imshow child. The
+ width is::
+
+ min(n_tasks, cpu_quota, floor(available_gb / per_figure_gb), _FIGURE_WORKERS_MAX)
+
+ where ``available_gb`` is the cgroup memory limit minus the working set
+ (``usage - inactive_file``, the OOM-relevant quantity; see ``_cgroup_memory``).
+ When the memory files cannot be read, falls back to the CPU-and-cap width and
+ logs that memory gating was skipped. The cgroup CPU quota is preferred over
+ ``os.cpu_count()`` (which reports the host, not the container's share).
+ Override with IMAGE_QC_FIGURE_WORKERS.
+ """
+ if n_tasks <= 0:
+ return 1
+
+ override = os.environ.get("IMAGE_QC_FIGURE_WORKERS")
+ if override:
+ width = max(1, min(int(override), n_tasks))
+ logging.info(f"[FIGPOOL] width={width} (IMAGE_QC_FIGURE_WORKERS override)")
+ return width
+
+ cpu = _cgroup_cpu_quota() or os.cpu_count() or 4
+ width = min(n_tasks, cpu, _FIGURE_WORKERS_MAX)
+
+ limit = _cgroup_memory_limit()
+ mem = _cgroup_memory()
+ if limit is not None and mem is not None:
+ working_set, _page_cache = mem
+ available_gb = max(0.0, limit - working_set) / _GIB
+ mem_width = max(1, int(available_gb // _FIGURE_PER_WORKER_GB))
+ width = max(1, min(width, mem_width))
+ logging.info(
+ f"[FIGPOOL] width={width} (tasks={n_tasks}, cpu_quota={cpu}, "
+ f"mem_avail={available_gb:.1f}GB / {_FIGURE_PER_WORKER_GB:.0f}GB "
+ f"-> {mem_width}, cap={_FIGURE_WORKERS_MAX})"
+ )
+ else:
+ width = max(1, width)
+ logging.info(
+ f"[FIGPOOL] width={width} (tasks={n_tasks}, cpu_quota={cpu}, "
+ f"cap={_FIGURE_WORKERS_MAX}, memory gate skipped: cgroup mem unreadable)"
+ )
+ return width
+
+
+def _run_figure_pool(tasks, *, phase: str = "") -> None:
+ """Render independent figure tasks concurrently in forked child processes.
+
+ Rolling pool: keeps up to ``width`` children alive at once and starts the
+ next task the instant a slot frees (submit + as-completed semantics), instead
+ of the old batch barrier that started ``width`` children, joined ALL of them,
+ then started the next batch -- so the slowest figure in a batch stalled every
+ idle core until the whole batch drained (measured ~255 s for 11 ROI figures).
+
+ NOT a ``concurrent.futures.ProcessPoolExecutor``: the tasks are closures over
+ large numpy planes, and the executor's work queue pickles every submitted
+ item even under a fork context (verified: "Can't pickle local object"). A raw
+ fork ``Process`` never pickles its target, so children inherit the planes
+ copy-on-write for free -- the whole reason this phase is fork-based. matplotlib
+ is not thread-safe, so a thread pool is not an option either.
+
+ Figure-failure semantics match the pre-pool code exactly: a task that raises
+ logs its full traceback in the child (``_run_figure_task``) and the step
+ CONTINUES, producing every other figure plus all metrics -- a single broken
+ figure never nukes the QC step. The pool additionally logs any non-zero child
+ exit (e.g. a hard crash / OOM-kill the old barrier ignored silently) naming the
+ task, then carries on. This is a performance change only, not a change to what
+ a figure failure does.
+ """
+ tasks = list(tasks)
+ if not tasks:
+ return
+
+ width = _figure_worker_limit(len(tasks))
+ mp_ctx = multiprocessing.get_context("fork")
+ logging.info(f"[FIGPOOL] {phase or 'figures'}: dispatching {len(tasks)} task(s)")
+
+ pending = list(reversed(tasks)) # pop() from the end preserves task order
+ running: dict[Any, tuple[Any, str]] = {} # sentinel fd -> (process, name)
+ failures: list[tuple[str, int]] = []
+
+ def _launch() -> None:
+ fn = pending.pop()
+ p = mp_ctx.Process(target=_run_figure_task, args=(fn,))
+ p.start()
+ running[p.sentinel] = (p, fn.__name__, time.perf_counter())
+
+ while pending and len(running) < width:
+ _launch()
+
+ while running:
+ # Block until at least one child exits; no busy-wait. A process sentinel
+ # is a file descriptor that becomes ready when the process terminates.
+ for sentinel in multiprocessing.connection.wait(list(running)):
+ p, name, started = running.pop(sentinel)
+ p.join()
+ logging.info(
+ f" [TIMING] figure {name} ({phase or 'figures'}): "
+ f"{time.perf_counter() - started:.1f}s"
+ )
+ if p.exitcode != 0:
+ failures.append((name, p.exitcode))
+ if pending:
+ _launch()
+
+ if failures:
+ detail = ", ".join(f"{name} (exit {code})" for name, code in failures)
+ # Match the pre-pool barrier: log loudly, do not abort the step. The child
+ # already logged the Python traceback via _run_figure_task; this covers
+ # non-zero exits (hard crash / OOM-kill) the old p.join() ignored silently.
+ logging.error(
+ f"[FIGPOOL] {phase or 'figures'}: figure task(s) failed: {detail}"
+ )
+
+
+# ---------------------------------------------------------------------------
+# Tile consumers — fold a tile into small state, then drop it
+#
+# The disk-backed planes below bound host RAM, but they are not the right final
+# design: they need ~154 GB of scratch on a 5.5 GP sample, mmap over a FUSE/S3
+# work directory is pathological (see
+# docs/failures/2026-07-24_imageqc-mmap-over-fusion.md), and these modules must
+# run local / server / cloud on their way to nf-core/spatialaxe.
+#
+# No consumer of the full-resolution maps needs a whole map: ROI sampling reads
+# one pixel per tile, SNR already loops per ROI window, the per-cell means are
+# additive (sum, count) reductions, and the focus heatmap needs an 8x block mean.
+# A consumer receives each finished tile and folds it into state that scales with
+# the number of ROIs or cells, never with image size.
+# ---------------------------------------------------------------------------
+
+
+def _is_device_array(array: Any) -> bool:
+ """True when *array* is a CuPy device array (so its reduction folds on-GPU)."""
+ return HAS_CUPY and isinstance(array, cp.ndarray)
+
+
+def _device_or_host(maps: dict[str, Any], key: str) -> Any:
+ """The device variant (``key + "_device"``) of a map if the tile kept one on the
+ GPU, else the host map under *key*. Lets a consumer fold on-device transparently:
+ the streaming GPU path supplies ``focus_map_device`` / ``mean_map_device`` while
+ the CPU / plane path supplies only the host ``focus_map`` / ``mean_map``.
+ """
+ device = maps.get(key + "_device")
+ return maps.get(key) if device is None else device
+
+
+class CentrePixelSampler:
+ """Sample one pixel per ROI from tiles as they are produced.
+
+ Replaces the centre-pixel sampling in ``downsample_maps_to_roi_dataframe``,
+ which indexes ``map[cy, cx]`` on the assembled full-resolution planes. That
+ pinned ~154 GB in order to read one pixel per ROI — 0.02 % of it.
+
+ A tile's *trimmed* arrays cover exactly its write region, so an ROI whose
+ centre lies in ``[write_y0, write_y1) x [write_x0, write_x1)`` is sampled at
+ local offset ``(cy - write_y0, cx - write_x0)``. Write regions are disjoint
+ and cover the image (``_compute_tile_grid``; asserted by
+ ``test_no_write_overlap``), so every ROI is sampled exactly once.
+ """
+
+ wants_untrimmed = False
+ #: This consumer does NOT read the host-side mean map, so the tile pass
+ #: may drop that D2H copy. A consumer that needs it must set this True.
+ reads_host_mean = False
+ #: Row/col multiple this consumer needs its tile write origins to fall on.
+ #: 1 means any split is fine. See BlockMeanAccumulator for why it needs more.
+ write_alignment = 1
+ #: Take the focus / mean maps from the GPU (``*_device``) when the tile pass
+ #: kept them resident, so the one-pixel-per-ROI gather runs on-device and the
+ #: full maps are not read back for it. The gather is a pure index, so the
+ #: sampled float32 pixel is bit-identical to the host read.
+ wants_device_focus = True
+ wants_device_mean = True
+
+ def __init__(self, cy: NDArray[np.integer], cx: NDArray[np.integer], keys):
+ self.cy = np.asarray(cy)
+ self.cx = np.asarray(cx)
+ self.values: dict[str, NDArray[np.float64]] = {
+ key: np.full(self.cy.size, np.nan, dtype=np.float64) for key in keys
+ }
+ # Sanity: every ROI must be claimed by exactly one tile.
+ self._claimed = np.zeros(self.cy.size, dtype=np.int8)
+
+ def consume(self, tile_spec: dict[str, int], maps: dict[str, Any]) -> None:
+ """Fold one tile's trimmed maps into the per-ROI arrays.
+
+ Each value map is taken device-first (``_device_or_host``): on the streaming
+ GPU path the gather runs on the resident ``focus_map_device`` /
+ ``mean_map_device`` and only the sampled pixels come back to host; on the CPU
+ / plane path the host maps are gathered exactly as before. A gather is a pure
+ index with no arithmetic, so the device and host reads return the identical
+ float32 pixel.
+ """
+ wy0, wy1 = tile_spec["write_y0"], tile_spec["write_y1"]
+ wx0, wx1 = tile_spec["write_x0"], tile_spec["write_x1"]
+ inside = (self.cy >= wy0) & (self.cy < wy1) & (self.cx >= wx0) & (self.cx < wx1)
+ if not inside.any():
+ return
+ local_y = self.cy[inside] - wy0
+ local_x = self.cx[inside] - wx0
+ self._claimed[inside] += 1
+ for key in self.values:
+ array = _device_or_host(maps, key)
+ if array is None:
+ continue
+ if _is_device_array(array):
+ # Gather on the array's own device (the fold runs after the GPU was
+ # returned to the pool, so the thread's current device may differ).
+ with array.device:
+ ly = cp.asarray(local_y)
+ lx = cp.asarray(local_x)
+ sampled = cp.asnumpy(array[ly, lx]).astype(np.float64)
+ else:
+ sampled = np.asarray(array)[local_y, local_x].astype(np.float64)
+ self.values[key][inside] = sampled
+
+ def finalize(self) -> dict[str, NDArray[np.float64]]:
+ """Return the per-ROI arrays, checking every ROI was covered once."""
+ unclaimed = int((self._claimed == 0).sum())
+ duplicated = int((self._claimed > 1).sum())
+ if unclaimed or duplicated:
+ raise RuntimeError(
+ f"ROI coverage is wrong: {unclaimed} ROI(s) claimed by no tile, "
+ f"{duplicated} by more than one. Tile write regions must be "
+ "disjoint and cover the image."
+ )
+ return dict(self.values)
+
+ def spawn(self) -> "CentrePixelSampler":
+ """A fresh, empty sampler sharing this one's ROI grid and keys.
+
+ Used by the tile pass to give each worker thread its own accumulator so
+ folds run without a lock; the partials are then reduced with ``merge``.
+ """
+ return CentrePixelSampler(self.cy, self.cx, list(self.values))
+
+ def merge(self, other: "CentrePixelSampler") -> None:
+ """Fold a per-worker partial into this one.
+
+ Each ROI is claimed by exactly one tile, hence by exactly one worker, so the
+ partials hold DISJOINT per-ROI entries: the merge is a pure overlay of the
+ ROIs *other* claimed, with no arithmetic and so bit-identical to the serial
+ fold regardless of order. The claim counts add so ``finalize`` still verifies
+ global coverage.
+ """
+ overlap = (self._claimed > 0) & (other._claimed > 0)
+ assert not overlap.any(), (
+ "CentrePixelSampler partials claim overlapping ROIs; tile write regions "
+ "must be disjoint (each ROI centre lands in exactly one write region)."
+ )
+ claimed = other._claimed > 0
+ self._claimed += other._claimed
+ for key in self.values:
+ self.values[key][claimed] = other.values[key][claimed]
+
+
+class LazyLabelPlane:
+ """Lazily-sliced view over a full-resolution label plane in zarr.
+
+ ``load_spatial_data`` does ``np.array(cell_masks_zarr["masks"]["1"])``, which
+ materialises 22 GB of uint32 on a 5.5 gigapixel sample, and
+ ``calculate_ccfs_from_focus_maps`` did the same for ``masks/0``. Neither needs
+ the whole plane: the consumers of these arrays are row-block reductions and
+ scattered point lookups, both of which this serves from zarr on demand.
+
+ Supports the three access patterns the callers actually use:
+
+ * ``plane[y0:y1]`` and ``plane[y0:y1, x0:x1]`` — row/rect slices, passed
+ straight through to zarr.
+ * ``plane[iy, ix]`` with integer arrays — coordinate lookup, served by reading
+ only the row blocks the points fall in. ``masks[y, x]`` on a zarr array does
+ not do numpy-style pair indexing, so this is why the wrapper exists.
+ """
+
+ def __init__(self, source, rows_per_chunk: int | None = None):
+ self._source = source
+ self.shape: tuple[int, int] = (int(source.shape[0]), int(source.shape[1]))
+ self.dtype = getattr(source, "dtype", None)
+ width = max(self.shape[1], 1)
+ self.rows_per_chunk = rows_per_chunk or max(1, _LABEL_CHUNK_PIXELS // width)
+
+ def __getitem__(self, key):
+ # Coordinate lookup: two integer arrays.
+ if (
+ isinstance(key, tuple)
+ and len(key) == 2
+ and all(
+ isinstance(part, (np.ndarray, list)) or np.isscalar(part)
+ for part in key
+ )
+ and not any(isinstance(part, slice) for part in key)
+ ):
+ return self._gather(np.asarray(key[0]), np.asarray(key[1]))
+ return np.asarray(self._source[key])
+
+ def _gather(
+ self, rows: NDArray[np.integer], cols: NDArray[np.integer]
+ ) -> NDArray[Any]:
+ """Point lookup, reading only the row blocks the points land in."""
+ rows = np.asarray(rows, dtype=np.int64).ravel()
+ cols = np.asarray(cols, dtype=np.int64).ravel()
+ if rows.size != cols.size:
+ raise ValueError("row and column index arrays must be the same length")
+ out = None
+ for start in range(0, self.shape[0], self.rows_per_chunk):
+ stop = min(start + self.rows_per_chunk, self.shape[0])
+ inside = (rows >= start) & (rows < stop)
+ if not inside.any():
+ continue
+ block = np.asarray(self._source[start:stop])
+ if out is None:
+ out = np.zeros(rows.size, dtype=block.dtype)
+ out[inside] = block[rows[inside] - start, cols[inside]]
+ del block
+ if out is None:
+ out = np.zeros(rows.size, dtype=self.dtype or np.int64)
+ return out
+
+ def max(self) -> int:
+ """Largest label value, read in row blocks."""
+ highest = 0
+ for start in range(0, self.shape[0], self.rows_per_chunk):
+ stop = min(start + self.rows_per_chunk, self.shape[0])
+ block = np.asarray(self._source[start:stop])
+ if block.size:
+ highest = max(highest, int(block.max()))
+ del block
+ return highest
+
+
+class LabeledSumAccumulator:
+ """Accumulate per-label ``(count, sum)`` from tiles as they are produced.
+
+ The same additive reduction as ``_labeled_sums_chunked``, keyed on tile write
+ regions instead of row blocks. Counts and sums are additive, so a cell split
+ across tiles contributes partial sums to each and its final ``sum / count`` is
+ exact — this is not an approximation.
+
+ State is ``O(n_labels)``: two arrays per value plane, tens of MB for ~530 k
+ cells, against the 22 GB per plane the whole-map path needed. The label plane
+ is sliced per tile, so it is never materialised either — which also removes
+ the ``cellseg_mask`` array the current path still holds.
+
+ The per-label reduction (``xp.bincount``) is ``xp``-generic. On the streaming GPU
+ path the value maps arrive resident on the device (``focus_map_device`` /
+ ``mean_map_device``); this uploads the int label block — cheaper than reading the
+ float value maps back — and runs every ``bincount`` on the GPU, returning only the
+ per-label ``O(n_labels)`` vectors to host. That moves the largest term of the tile
+ fold (~112 s/channel of host ``bincount``) onto the otherwise-idle GPU. The counts
+ and float64 sums are the same additive reduction either way, and the GPU float64
+ ``bincount`` matches the host result to float64 rounding (tests/test_tile_consumers).
+ On the CPU / plane path the maps are host numpy and the fold is byte-identical to
+ before.
+ """
+
+ wants_untrimmed = False
+ #: This consumer does NOT read the host-side mean map, so the tile pass
+ #: may drop that D2H copy. A consumer that needs it must set this True.
+ reads_host_mean = False
+ #: Row/col multiple this consumer needs its tile write origins to fall on.
+ #: 1 means any split is fine. See BlockMeanAccumulator for why it needs more.
+ write_alignment = 1
+ #: Fold from the GPU-resident maps when the tile pass kept them there, so the
+ #: per-label bincounts run on-device instead of a serial host fold.
+ wants_device_focus = True
+ wants_device_mean = True
+
+ def __init__(
+ self,
+ label_plane,
+ value_keys,
+ include_coords: bool = False,
+ skip_background: bool = False,
+ ):
+ self.label_plane = label_plane
+ self.include_coords = include_coords
+ # Drop label-0 pixels before reducing. Only valid where the caller never reads
+ # index 0: the nuclear reduction does `labels[labels > 0]`, but the *cell*
+ # reduction deliberately reads `cell_counts[0]` as the background pixel count
+ # for CellID 0, so it must keep them.
+ self.skip_background = skip_background
+ self.counts = np.zeros(1, dtype=np.int64)
+ self.sums: dict[str, NDArray[np.float64]] = {
+ key: np.zeros(1, dtype=np.float64) for key in value_keys
+ }
+ if include_coords:
+ self.sums["centroid_y_sum"] = np.zeros(1, dtype=np.float64)
+ self.sums["centroid_x_sum"] = np.zeros(1, dtype=np.float64)
+
+ def _add(self, key: str, block: NDArray[np.float64]) -> None:
+ self.sums[key] = _grow_to(self.sums[key], block.size)
+ self.sums[key][: block.size] += block
+
+ def consume(self, tile_spec: dict[str, int], maps: dict[str, Any]) -> None:
+ """Fold one tile's trimmed maps into the per-label accumulators.
+
+ Runs on the GPU when the tile kept its maps device-resident (``xp=cp``),
+ else on host (``xp=np``); ``_fold_blocks`` is written once against ``xp``.
+ Blocked by rows within the tile, for the same reason
+ ``_labeled_sums_chunked`` blocks: ``bincount`` needs ``intp`` labels and
+ ``float64`` weights, so a whole-tile call casts both. A production tile is
+ a full-width row strip, so that is 2.7 GB of labels plus 5.5 GB per
+ coordinate array — larger than the tile's own maps. One block is ~32 M
+ pixels regardless of tile size, which also keeps the on-device label upload
+ and the ``K*256``-free bincount well under the tile's VRAM budget.
+ """
+ wx0, wx1 = tile_spec["write_x0"], tile_spec["write_x1"]
+ tile_width = wx1 - wx0
+ if tile_width <= 0 or tile_spec["write_y1"] <= tile_spec["write_y0"]:
+ return
+
+ # Take each value map device-first, then decide the fold backend from what the
+ # tile actually handed over. A value-less counts-only accumulator (cells) has no
+ # value map to check, so fall back to any resident device array in the tile.
+ value_maps = {
+ key: _device_or_host(maps, key)
+ for key in self.sums
+ if not key.startswith("centroid_")
+ }
+ device_arr = next((a for a in value_maps.values() if _is_device_array(a)), None)
+ if device_arr is None:
+ device_arr = next((a for a in maps.values() if _is_device_array(a)), None)
+
+ if device_arr is not None:
+ # Run every bincount on the map's own device (the fold happens after the
+ # GPU is returned to the pool, so the thread's current device may differ).
+ with device_arr.device:
+ self._fold_blocks(tile_spec, value_maps, cp, on_gpu=True)
+ else:
+ self._fold_blocks(tile_spec, value_maps, np, on_gpu=False)
+
+ def _fold_blocks(
+ self,
+ tile_spec: dict[str, int],
+ value_maps: dict[str, Any],
+ xp: Any,
+ on_gpu: bool,
+ ) -> None:
+ """Row-blocked per-label reduction over one tile, on ``xp`` (numpy or cupy).
+
+ Reduces on-device when ``xp`` is cupy, pulling only the ``O(n_labels)`` result
+ vectors back to the host accumulators; the intermediate labels/values/coords
+ never leave the device.
+ """
+ wy0, wy1 = tile_spec["write_y0"], tile_spec["write_y1"]
+ wx0, wx1 = tile_spec["write_x0"], tile_spec["write_x1"]
+ tile_width = wx1 - wx0
+ rows_per_chunk = max(1, _LABEL_CHUNK_PIXELS // tile_width)
+
+ def _host(arr: Any) -> Any:
+ return cp.asnumpy(arr) if on_gpu else arr
+
+ for y0 in range(wy0, wy1, rows_per_chunk):
+ y1 = min(y0 + rows_per_chunk, wy1)
+ labels_host = np.asarray(self.label_plane[y0:y1, wx0:wx1]).ravel()
+ if labels_host.size == 0:
+ continue
+ # Upload the int label block to the device (cheaper than reading the float
+ # value maps back), or keep it on host for the numpy path.
+ labels = xp.asarray(labels_host) if on_gpu else labels_host
+ del labels_host
+
+ # Restrict to labelled pixels where index 0 is never read. On the reference
+ # sample the nuclear mask is 9.8 % non-zero (505 k nuclei x ~1066 px of
+ # 5.50 G), so this drops ~90 % of the work from every pass below --
+ # measured at ~169 s for this accumulator, the largest term in the fold
+ # after the Otsu SNR. `flatnonzero` also lets the coordinates be built at
+ # the compressed size instead of materialising a full-chunk repeat/tile.
+ selected = None
+ if self.skip_background:
+ selected = xp.flatnonzero(labels)
+ labels = labels[selected]
+ if labels.size == 0:
+ continue
+
+ block_counts = xp.bincount(labels)
+ n = int(block_counts.size)
+ self.counts = _grow_to(self.counts, n)
+ self.counts[:n] += _host(block_counts)
+
+ # value_maps rows are trimmed to the write region, so map row `y0 - wy0`
+ # is global row `y0`.
+ for key, array in value_maps.items():
+ if array is None:
+ continue
+ values = xp.asarray(
+ array[y0 - wy0 : y1 - wy0], dtype=xp.float64
+ ).ravel()
+ if selected is not None:
+ values = values[selected]
+ self._add(key, _host(xp.bincount(labels, weights=values, minlength=n)))
+ del values
+
+ if self.include_coords:
+ # Global coordinates, so sum / count is regionprops' centroid in
+ # image space rather than tile-local space.
+ if selected is not None:
+ rows = (y0 + selected // tile_width).astype(xp.float64)
+ cols = (wx0 + selected % tile_width).astype(xp.float64)
+ else:
+ rows = xp.repeat(xp.arange(y0, y1, dtype=xp.float64), tile_width)
+ cols = xp.tile(xp.arange(wx0, wx1, dtype=xp.float64), y1 - y0)
+ self._add(
+ "centroid_y_sum",
+ _host(xp.bincount(labels, weights=rows, minlength=n)),
+ )
+ del rows
+ self._add(
+ "centroid_x_sum",
+ _host(xp.bincount(labels, weights=cols, minlength=n)),
+ )
+ del cols
+ del labels, block_counts, selected
+
+ def finalize(self) -> tuple[NDArray[np.int64], dict[str, NDArray[np.float64]]]:
+ """Return ``(counts, sums)`` indexed by raw label value, 0 = background."""
+ return self.counts, dict(self.sums)
+
+ def spawn(self) -> "LabeledSumAccumulator":
+ """A fresh, empty accumulator sharing this one's label plane and keys.
+
+ Used by the tile pass to give each worker thread its own accumulator so
+ folds run without a lock; the partials are then reduced with ``merge``. The
+ label plane is shared by reference (read-only, per-tile slices), not copied.
+ """
+ value_keys = [k for k in self.sums if not k.startswith("centroid_")]
+ return LabeledSumAccumulator(
+ self.label_plane,
+ value_keys,
+ include_coords=self.include_coords,
+ skip_background=self.skip_background,
+ )
+
+ def merge(self, other: "LabeledSumAccumulator") -> None:
+ """Fold a per-worker partial into this one by element-wise addition.
+
+ Counts and per-label float64 sums are additive, so adding a worker's
+ partial totals reduces the same values as the serial per-tile fold -- but
+ it regroups the float64 additions (per-worker subtotals then combined,
+ rather than one running sum in tile order). Reassociation of float64 sums
+ can differ by ULPs; that this reproduces the serial fold byte-for-byte is
+ empirical and is pinned by tests/test_fold_parallel_equivalence, not
+ assumed. Arrays are grown to the longer length before adding.
+ """
+ n = max(self.counts.size, other.counts.size)
+ self.counts = _grow_to(self.counts, n)
+ self.counts[: other.counts.size] += other.counts
+ for key, total in other.sums.items():
+ self.sums[key] = _grow_to(self.sums[key], total.size)
+ self.sums[key][: total.size] += total
+
+ def means(self, labels: NDArray[np.integer]) -> dict[str, NDArray[np.float64]]:
+ """Per-label means for *labels*, NaN where a label has no pixels."""
+ counts = self.counts[labels].astype(np.float64)
+ out = {}
+ with np.errstate(invalid="ignore", divide="ignore"):
+ for key, total in self.sums.items():
+ out[key] = total[labels] / counts
+ return out
+
+
+def _block_sum(
+ array: NDArray[np.generic], y_offset: int, x_offset: int, factor: int
+) -> tuple[NDArray[np.float64], NDArray[np.intp], NDArray[np.intp]]:
+ """Sum *array* into the global factor-grid blocks it overlaps.
+
+ ``array[0, 0]`` sits at global ``(y_offset, x_offset)``. Returns the block
+ sums plus the global block row/column indices they belong to.
+
+ Two ``np.add.reduceat`` passes rather than a flat block-index array: for a
+ 36k x 36k tile the index array alone would be 1.3e9 int64 = 10 GB, whereas
+ the row-reduced intermediate is (n_block_rows, width). ``reduceat`` also
+ handles the ragged first/last groups natively when the tile does not start on
+ a block boundary.
+ """
+ values = np.asarray(array, dtype=np.float64)
+ rows = np.arange(values.shape[0]) + y_offset
+ row_starts = np.flatnonzero(
+ np.r_[True, (rows[1:] // factor) != (rows[:-1] // factor)]
+ )
+ partial = np.add.reduceat(values, row_starts, axis=0)
+
+ cols = np.arange(values.shape[1]) + x_offset
+ col_starts = np.flatnonzero(
+ np.r_[True, (cols[1:] // factor) != (cols[:-1] // factor)]
+ )
+ block = np.add.reduceat(partial, col_starts, axis=1)
+ return block, rows[row_starts] // factor, cols[col_starts] // factor
+
+
+class BlockMeanAccumulator:
+ """Build a factor-``f`` down-sampled canvas from tiles, matching skimage.
+
+ ``_fig5_focus_heatmap`` calls ``downscale_local_mean(dapi_focus, (8, 8))`` on
+ the assembled plane — the only use of that plane, and it pads the array before
+ reducing, so on a 5.5 GP sample it costs another ~22 GB on top of the 22 GB
+ plane it reads.
+
+ ``downscale_local_mean`` zero-pads incomplete blocks and divides by the
+ *full* block size, so accumulating per-block sums and dividing by
+ ``factor**2`` reproduces it **bit-exactly**, including the partial final block, and
+ that is verified against skimage rather than argued (``TestBlockMeanEpsilon``).
+
+ Exactness needs two things, and the *dtype* one is by far the larger:
+
+ First, the same precision. Production focus maps are float32, so the plane-based path
+ reduces in float32; this class reduces in whatever dtype it is handed and must not
+ widen it. An earlier version upcast to float64 and disagreed with skimage in 200 of
+ 200 measured random planes, by up to 1.86e-07 relative.
+
+ Second, *grouping*: every block reduced by one addition chain, never
+ as two partials summed across a tile seam. The dispatcher arranges that by aligning
+ write boundaries to ``factor`` -- it reads ``write_alignment`` off each consumer, so
+ a caller cannot forget -- and ``consume`` refuses an unaligned write rather than
+ silently accumulating one.
+
+ The axis order within a block turns out not to matter -- ``sum(axis=(1, 3))`` and
+ ``sum(axis=3).sum(axis=1)`` agree with skimage bit-for-bit over 300 random
+ shapes/magnitudes, and mutating one to the other is an equivalent mutant. Only the
+ grouping is load-bearing.
+
+ An earlier version summed with two ``np.add.reduceat`` passes over an unaligned grid,
+ so a block straddling a seam *was* accumulated as two partials. That is real, but it
+ contributes only ~3.5e-16 and was **not** the cause of the 16 pixels of Figure 5 that
+ differed from the plane-based path: the float64 upcast above was, at ~1e-7. Fixing
+ the ordering alone left the output byte-for-byte unchanged (16 px, delta 1) --
+ see ``docs/failures/2026-07-26_heatmap-16px-dtype-not-summation-order.md``.
+
+ State is the ``(H/f, W/f)`` canvas: 0.34 GB at f=8 on a 5.5 GP sample,
+ against 44 GB for the plane plus skimage's pad.
+ """
+
+ #: Every consumer takes ``consume(tile_spec, maps)`` where *maps* is the dict
+ #: of this tile's channel maps. A uniform signature lets the tile dispatcher
+ #: feed them all without knowing which is which.
+ wants_untrimmed = False
+ #: This consumer does NOT read the host-side mean map, so the tile pass
+ #: may drop that D2H copy. A consumer that needs it must set this True.
+ reads_host_mean = False
+ #: Row/col multiple this consumer needs its tile write origins to fall on.
+ #: 1 means any split is fine. See BlockMeanAccumulator for why it needs more.
+ write_alignment = 1
+
+ def __init__(
+ self,
+ shape: tuple[int, int],
+ factor: int = 8,
+ map_key: str = "focus_map",
+ ):
+ self.map_key = map_key
+ self.factor = int(factor)
+ # Blocks must not straddle tiles, so the dispatcher must land write origins on
+ # multiples of the factor. Declared here rather than passed in by the caller:
+ # a caller that forgot would only find out via consume()'s RuntimeError.
+ self.write_alignment = self.factor
+ #: dtype of the first array consumed; the canvas is returned in it.
+ self._out_dtype: np.dtype | None = None
+ self.shape = (int(shape[0]), int(shape[1]))
+ rows = -(-self.shape[0] // self.factor) # ceil
+ cols = -(-self.shape[1] // self.factor)
+ self.sums = np.zeros((rows, cols), dtype=np.float64)
+
+ def consume(self, tile_spec: dict[str, int], maps: dict[str, Any]) -> None:
+ """Fold this tile's ``map_key`` plane into the block canvas.
+
+ Reduces each block with a single reshape, matching ``downscale_local_mean``'s own
+ grouping and summation order, which makes the canvas **bit-exact** rather than
+ equal to within a float64 epsilon. That requires every block to lie wholly inside
+ one tile, which the dispatcher guarantees by aligning write boundaries to
+ ``factor``; the check below is that contract, and it raises rather than silently
+ degrading.
+
+ Previously this used two ``np.add.reduceat`` passes, so a block straddling a seam
+ was summed as two partials in a different order from skimage. The resulting
+ ~3.5e-16 discrepancy reached the rendered figure: 16 of Figure 5's 5,553,835
+ pixels landed 1/255 apart from the plane-based path.
+
+ The image's own trailing rows and columns may still form a partial block; those
+ are zero-padded exactly as skimage pads, which is bit-exact for dimensions that
+ are not multiples of the factor.
+ """
+ array = maps.get(self.map_key)
+ if array is None:
+ return
+ # Reduce in the INPUT's dtype, never a wider one. Production focus maps are
+ # float32 and the plane-based path calls downscale_local_mean on them, so it
+ # reduces in float32; accumulating in float64 here disagreed with it by up to
+ # 1.86e-07 relative in 200 of 200 measured cases -- enough to move 16 of Figure
+ # 5's 5.5M pixels across a uint8 boundary. That dtype term is ~5e8 times larger
+ # than the summation-order term this class used to blame.
+ # No integer special-case: numpy already promotes e.g. uint16 sums to uint64,
+ # which lands on skimage's float64 mean exactly. Verified, not assumed --
+ # test_integer_input_follows_skimage_to_float64, and a mutation removing an
+ # explicit promotion survived because it was doing nothing.
+ values = np.asarray(array)
+ if self._out_dtype is None:
+ # The canvas must come out in the dtype downscale_local_mean would have
+ # produced, not merely with the same values. Figure 5 takes its colour scale
+ # from np.percentile of this array, and percentile on a float64 container
+ # returns 2082.678369140625 where float32 returns 2082.6785 -- a different
+ # vmin/vmax, which flips pixels sitting on a colormap boundary. That was the
+ # third and last cause of the differing heatmap pixels.
+ self._out_dtype = (
+ values.dtype
+ if np.issubdtype(values.dtype, np.floating)
+ else np.dtype(np.float64) # skimage's np.mean promotes integers
+ )
+ wy0, wx0 = tile_spec["write_y0"], tile_spec["write_x0"]
+ f = self.factor
+ if wy0 % f or wx0 % f:
+ raise RuntimeError(
+ f"tile write origin ({wy0}, {wx0}) is not aligned to the heatmap block "
+ f"size {f}: blocks would straddle tiles and the canvas would stop "
+ "matching downscale_local_mean exactly. _compute_channel_maps_tiled "
+ "derives this from each consumer's write_alignment, so reaching here "
+ "means the accumulator was driven by something else."
+ )
+ pad_y = (-values.shape[0]) % f
+ pad_x = (-values.shape[1]) % f
+ if pad_y or pad_x:
+ values = np.pad(values, ((0, pad_y), (0, pad_x)), mode="constant")
+ blocks = values.reshape(values.shape[0] // f, f, values.shape[1] // f, f).sum(
+ axis=(1, 3)
+ )
+ r0, c0 = wy0 // f, wx0 // f
+ self.sums[r0 : r0 + blocks.shape[0], c0 : c0 + blocks.shape[1]] += blocks
+
+ def finalize(self) -> NDArray[np.floating]:
+ """Return the block means, identical to ``downscale_local_mean`` in value *and*
+ dtype. The dtype is part of the contract: Figure 5 derives its colour scale from
+ ``np.percentile`` of this array, which answers differently in float32 and float64.
+ """
+ means = self.sums / float(self.factor * self.factor)
+ if self._out_dtype is not None and means.dtype != self._out_dtype:
+ means = means.astype(self._out_dtype)
+ return means
+
+ def spawn(self) -> "BlockMeanAccumulator":
+ """A fresh, empty canvas of the same shape / factor / map key.
+
+ Used by the tile pass to give each worker thread its own accumulator so
+ folds run without a lock; the partials are then reduced with ``merge``.
+ """
+ return BlockMeanAccumulator(
+ self.shape, factor=self.factor, map_key=self.map_key
+ )
+
+ def merge(self, other: "BlockMeanAccumulator") -> None:
+ """Fold a per-worker partial canvas into this one.
+
+ Write origins are aligned to ``factor`` (the dispatcher enforces it via
+ ``write_alignment``), so every block lies wholly inside one tile and hence
+ one worker. The per-worker canvases are therefore DISJOINT -- each block is
+ non-zero in exactly one partial -- so the element-wise add is a pure overlay
+ (``0.0 + x == x``) with no reassociation, bit-identical to the serial fold.
+ """
+ self.sums += other.sums
+ if other._out_dtype is not None:
+ if self._out_dtype is not None and self._out_dtype != other._out_dtype:
+ raise RuntimeError(
+ f"BlockMeanAccumulator partials disagree on output dtype "
+ f"({self._out_dtype} vs {other._out_dtype}); every tile's map "
+ "must have the same dtype."
+ )
+ self._out_dtype = other._out_dtype
+
+
+#: ROIs folded per batched-Otsu call. Two constraints set it:
+#:
+#: 1. **cupy bincount ceiling (hard).** ``roi_snr_db_batch`` builds a per-row
+#: histogram as one flattened ``cp.bincount`` over ``K*1225`` indices into
+#: ``K*256`` bins. In cupy 14.0.1 that bincount raises
+#: ``cudaErrorIllegalAddress`` once K is large: measured OK at K=40_000
+#: (10.2M bins / 49M elems), FAIL at K=50_000 (12.8M / 61.25M). This is what
+#: crashed Tower run 4bBYe2TAP5QEqA on the first fold. Staying an order of
+#: magnitude under that boundary makes the fault structurally impossible and
+#: leaves headroom for other GPUs/cupy builds.
+#: 2. VRAM: bounds the ``(K, N)`` / ``(K, 256)`` float64 temporaries so a tall
+#: strip's owned ROIs stay well under the tile's ~10.5 GB budget.
+#:
+#: 8192 gives ~5x margin on (1) and a small device peak; the extra kernel launches
+#: over a larger chunk are negligible against the fold. See
+#: docs/failures/2026-07-27_gpu-otsu-cuda-illegal-address.md.
+_OTSU_ROI_CHUNK = 8_192
+
+
+class RoiOtsuSnrAccumulator:
+ """Per-ROI Otsu SNR in dB, computed from tiles as they are produced.
+
+ Replaces ``snr_metrics.compute_image_snr_from_pixel_maps``'s loop over
+ ``dapi_mean_map[y1:y2, x1:x2]``, which needed the assembled 22 GB plane.
+
+ The reduction is ``xp``-generic. In the streaming GPU path the tile pass hands
+ over the tile's **device** ``mean_map`` (a CuPy array still resident in VRAM,
+ under key ``"mean_map_device"``) and the batched Otsu SNR
+ (``snr_metrics.roi_snr_db_batch``, ``xp=cp``) folds every owned interior ROI in
+ one vectorized GPU pass, returning only the small per-ROI dB vector to host --
+ instead of the 4.49 M-iteration Python loop over host windows this class used
+ to run (~439 s of CPU). On the CPU / no-GPU path the map is a numpy array and
+ the identical batched kernel runs with ``xp=np``. Either way the result matches
+ the per-ROI scalar ``snr_metrics.roi_snr_db`` to float64 rounding (~1e-14 dB;
+ see ``tests/test_roi_snr_db_batch.py``).
+
+ Windows clipped at the image edge (variable size) or holding a non-finite
+ pixel fall back to the scalar ``roi_snr_db``, which the batched kernel -- it
+ assumes finite, full-size windows -- does not model. Only the full-size
+ interior windows, which share one shape, are batched.
+
+ Unlike the other consumers this one needs the **untrimmed** tile: an ROI is
+ owned by the tile whose write region contains its top-left corner, and its
+ window then extends up to ``roi_size - 1`` px beyond that write region. The
+ read region extends by ``overlap``, so the window is covered exactly when
+ ``overlap >= roi_size``. That is currently true (both are ``window_size``,
+ 35) but it is a real constraint, so a violation raises rather than silently
+ truncating a window and reporting a wrong dB.
+ """
+
+ #: Tells the tile dispatcher to hand this consumer the haloed tile.
+ wants_untrimmed = True
+ #: This consumer does NOT read the host-side mean map, so the tile pass
+ #: may drop that D2H copy. A consumer that needs it must set this True.
+ reads_host_mean = False
+ #: Tells the tile dispatcher to keep the tile's mean map on the GPU and pass
+ #: the device array (``"mean_map_device"``) so the Otsu SNR folds on-device.
+ wants_device_mean = True
+ write_alignment = 1
+
+ def __init__(
+ self,
+ y1: NDArray[np.integer],
+ y2: NDArray[np.integer],
+ x1: NDArray[np.integer],
+ x2: NDArray[np.integer],
+ image_shape: tuple[int, int],
+ map_key: str = "mean_map",
+ ):
+ self.map_key = map_key
+ self.y1 = np.asarray(y1, dtype=np.int64)
+ self.y2 = np.asarray(y2, dtype=np.int64)
+ self.x1 = np.asarray(x1, dtype=np.int64)
+ self.x2 = np.asarray(x2, dtype=np.int64)
+ self.height, self.width = int(image_shape[0]), int(image_shape[1])
+ self.db = np.full(self.y1.size, np.nan, dtype=np.float64)
+ self._claimed = np.zeros(self.y1.size, dtype=np.int8)
+ # The full ROI side; interior windows equal it, edge-clipped windows are
+ # smaller. Windows of exactly this shape are the ones the batched kernel
+ # folds together (it needs one uniform shape); everything else is scalar.
+ self._full_h = int((self.y2 - self.y1).max()) if self.y1.size else 0
+ self._full_w = int((self.x2 - self.x1).max()) if self.x1.size else 0
+
+ def consume(self, tile_spec: dict[str, int], maps: dict[str, Any]) -> None:
+ """Fold this tile's *untrimmed* (haloed) mean map into the per-ROI dB array.
+
+ Reads the device mean map (``"mean_map_device"``) when the tile pass kept
+ one resident, else the host ``map_key``; ``xp`` follows the array type so
+ the same code batches on GPU (``cp``) or CPU (``np``). Full-size interior
+ windows go through ``snr_metrics.roi_snr_db_batch`` in bounded chunks;
+ edge-clipped or non-finite windows fall back to the scalar
+ ``roi_snr_db``.
+ """
+ # The device map the tile pass keeps resident is the mean map specifically
+ # (``mean_map_device``). Only reach for it when this accumulator is actually
+ # configured for the mean map; otherwise fall through to the configured
+ # ``map_key``. Without this guard a focus-configured instance would silently
+ # reduce the mean map on a GPU tile and return plausible but wrong numbers.
+ array = maps.get("mean_map_device") if self.map_key == "mean_map" else None
+ if array is None:
+ array = maps.get(self.map_key)
+ if array is None:
+ return
+
+ on_gpu = HAS_CUPY and isinstance(array, cp.ndarray)
+ xp = cp if on_gpu else np
+
+ ry0, ry1 = tile_spec["read_y0"], tile_spec["read_y1"]
+ rx0, rx1 = tile_spec["read_x0"], tile_spec["read_x1"]
+ wy0, wy1 = tile_spec["write_y0"], tile_spec["write_y1"]
+ wx0, wx1 = tile_spec["write_x0"], tile_spec["write_x1"]
+
+ owned = (self.y1 >= wy0) & (self.y1 < wy1) & (self.x1 >= wx0) & (self.x1 < wx1)
+ owned_idx = np.flatnonzero(owned)
+ if owned_idx.size == 0:
+ return
+
+ # Clip each owned window to the image exactly as the whole-map path does.
+ y_start = np.maximum(0, self.y1[owned_idx])
+ x_start = np.maximum(0, self.x1[owned_idx])
+ y_stop = np.minimum(self.height, self.y2[owned_idx])
+ x_stop = np.minimum(self.width, self.x2[owned_idx])
+ self._claimed[owned_idx] += 1
+
+ valid = (y_stop > y_start) & (x_stop > x_start)
+ # The read region must cover every non-empty window, or a window would be
+ # silently truncated and its dB wrong. Raise, as the per-ROI path did.
+ covered = (
+ (ry0 <= y_start) & (y_stop <= ry1) & (rx0 <= x_start) & (x_stop <= rx1)
+ )
+ bad = np.flatnonzero(valid & ~covered)
+ if bad.size:
+ b = int(bad[0])
+ raise RuntimeError(
+ f"ROI window [{int(y_start[b])}:{int(y_stop[b])}, "
+ f"{int(x_start[b])}:{int(x_stop[b])}] is not covered by its tile's "
+ f"read region [{ry0}:{ry1}, {rx0}:{rx1}]. The tile overlap must be "
+ "at least the ROI size."
+ )
+
+ # Window bounds relative to the read region.
+ y0r = (y_start - ry0).astype(np.int64)
+ x0r = (x_start - rx0).astype(np.int64)
+ y1r = (y_stop - ry0).astype(np.int64)
+ x1r = (x_stop - rx0).astype(np.int64)
+ # Full-size interior windows share one shape and are batched together;
+ # partial (edge-clipped) windows keep the scalar path.
+ full = (
+ valid
+ & (y_stop - y_start == self._full_h)
+ & (x_stop - x_start == self._full_w)
+ )
+
+ def _scalar(positions: NDArray[np.integer]) -> None:
+ for pos in positions:
+ win = array[y0r[pos] : y1r[pos], x0r[pos] : x1r[pos]]
+ if on_gpu:
+ win = cp.asnumpy(win)
+ self.db[owned_idx[pos]] = snr_metrics.roi_snr_db(win)
+
+ def _batch_reduce(st: Any) -> NDArray[np.float64]:
+ # GPU: batched Otsu on-device (CuPy), no map transfer. CPU: numba prange
+ # kernel (~50x the Python loop). The pure-numpy batch is memory-bound and
+ # no faster than the loop, so it is only the fallback when numba is absent.
+ # All three match the scalar roi_snr_db to float64 rounding.
+ if on_gpu:
+ return cp.asnumpy(snr_metrics.roi_snr_db_batch(st, xp=cp))
+ if snr_metrics._HAS_NUMBA:
+ return snr_metrics.roi_snr_db_numba(st)
+ return snr_metrics.roi_snr_db_batch(st, xp=np)
+
+ def _reduce() -> None:
+ full_pos = np.flatnonzero(full)
+ for start in range(0, full_pos.size, _OTSU_ROI_CHUNK):
+ sel = full_pos[start : start + _OTSU_ROI_CHUNK]
+ # xp.stack, not xp.asarray(list): the windows are on-device CuPy
+ # slices, and cp.asarray of a Python list of CuPy arrays is
+ # version-fragile, whereas stack of same-shape arrays is not.
+ stack = xp.stack([array[y0r[p] : y1r[p], x0r[p] : x1r[p]] for p in sel])
+ # The batched/numba kernels assume finite windows; route any
+ # non-finite one to the scalar path (which filters them itself).
+ finite = xp.isfinite(stack.reshape(stack.shape[0], -1)).all(axis=1)
+ finite_host = cp.asnumpy(finite) if on_gpu else finite
+ if bool(finite_host.all()):
+ self.db[owned_idx[sel]] = _batch_reduce(stack)
+ else:
+ self.db[owned_idx[sel[finite_host]]] = _batch_reduce(stack[finite])
+ _scalar(sel[~finite_host])
+ _scalar(np.flatnonzero(valid & ~full))
+
+ if on_gpu:
+ # Run every CuPy op under the array's own device -- the fold happens
+ # after the GPU is returned to the pool, so the thread's current
+ # device may be another one.
+ with array.device:
+ _reduce()
+ else:
+ _reduce()
+
+ def finalize(self) -> NDArray[np.float64]:
+ """Return the per-ROI dB array, checking each ROI was claimed once."""
+ unclaimed = int((self._claimed == 0).sum())
+ duplicated = int((self._claimed > 1).sum())
+ if unclaimed or duplicated:
+ raise RuntimeError(
+ f"ROI coverage is wrong: {unclaimed} claimed by no tile, "
+ f"{duplicated} by more than one."
+ )
+ return self.db
+
+ def spawn(self) -> "RoiOtsuSnrAccumulator":
+ """A fresh, empty accumulator sharing this one's ROI geometry.
+
+ Used by the tile pass to give each worker thread its own accumulator so
+ folds run without a lock; the partials are then reduced with ``merge``.
+ """
+ return RoiOtsuSnrAccumulator(
+ self.y1,
+ self.y2,
+ self.x1,
+ self.x2,
+ (self.height, self.width),
+ map_key=self.map_key,
+ )
+
+ def merge(self, other: "RoiOtsuSnrAccumulator") -> None:
+ """Fold a per-worker partial into this one.
+
+ An ROI is owned by the single tile whose write region holds its top-left
+ corner, so partials hold DISJOINT per-ROI dB entries: the merge overlays the
+ ROIs *other* computed, with no arithmetic and so bit-identical to the serial
+ fold. The claim counts add so ``finalize`` still verifies global coverage.
+ """
+ overlap = (self._claimed > 0) & (other._claimed > 0)
+ assert not overlap.any(), (
+ "RoiOtsuSnrAccumulator partials claim overlapping ROIs; each ROI must be "
+ "owned by exactly one tile (its top-left corner lands in one write region)."
+ )
+ claimed = other._claimed > 0
+ self._claimed += other._claimed
+ self.db[claimed] = other.db[claimed]
+
+
+# ---------------------------------------------------------------------------
+# Full-resolution plane storage (RAM or disk-backed)
+# ---------------------------------------------------------------------------
+
+# Planes at or above this size are backed by a file rather than anonymous RAM.
+_PLANE_SPILL_BYTES = 2 * 1024**3
+
+
+class _PlaneStore:
+ """Owns the full-resolution output planes for one channel.
+
+ A plane that reaches ``_PLANE_SPILL_BYTES`` is backed by an ``np.memmap``
+ file in *plane_dir* (the task working directory by default) instead of
+ anonymous RAM. Nothing declares these files as process outputs, so Nextflow
+ discards them with the work directory; use the ``scratch`` directive to put
+ that directory on node-local NVMe. Two
+ things follow from that, and together they are why this class exists:
+
+ * **Bounded resident set.** Every consumer of these planes reads row blocks
+ or ROI windows, never the whole array (see ``_labeled_sums_chunked`` and
+ the SNR per-tile loop). File backing turns 22 GB of unreclaimable
+ anonymous pages per plane into page cache the kernel can evict under
+ pressure. On a 5.5 gigapixel sample the seven planes come to ~154 GB
+ resident, which is what forced the 180 -> 720 GB retry ladder.
+ * **Real multi-GPU parallelism.** A memmap is shared between processes by
+ *filename*, so tile workers can be separate processes — one per GPU, each
+ with its own interpreter, its own GIL and its own CUDA context — all
+ writing into the same planes with no pixel data ever pickled. Thread
+ workers cannot scale much past a single GPU's throughput no matter how
+ many devices are present, because the host-side numpy in every tile (the
+ input dtype cast, ``_sanitize``, the post-D2H cast) holds the GIL. NVLink
+ is irrelevant to this workload: tiles are independent and nothing is
+ exchanged between devices.
+
+ Planes are always handed out as ordinary ndarrays, so consumers never need
+ to know which backing is in use.
+ """
+
+ def __init__(
+ self,
+ shape: tuple[int, int],
+ keys: list[str],
+ plane_dir: Path | None = None,
+ prefix: str = "plane",
+ dtype: Any = np.float32,
+ ) -> None:
+ self.shape = (int(shape[0]), int(shape[1]))
+ self.dtype = np.dtype(dtype)
+ self.keys = list(keys)
+ plane_bytes = self.shape[0] * self.shape[1] * self.dtype.itemsize
+ # Small planes are not worth a file; large ones always get one.
+ self.on_disk = plane_bytes >= _PLANE_SPILL_BYTES
+ self._dir = Path(plane_dir or Path.cwd()) if self.on_disk else None
+ self.paths: dict[str, str] = {}
+ self._arrays: dict[str, NDArray[Any]] = {}
+
+ if self._dir is not None:
+ self._dir.mkdir(parents=True, exist_ok=True)
+ required = plane_bytes * len(self.keys)
+ free = shutil.disk_usage(self._dir).free
+ if free < required:
+ raise RuntimeError(
+ f"scratch dir {self._dir} has {free / 1024**3:.1f} GB free, but "
+ f"{len(self.keys)} plane(s) of {plane_bytes / 1024**3:.1f} GB "
+ f"need {required / 1024**3:.1f} GB. Point --scratch-dir at a "
+ "larger filesystem, or drop the flag to keep planes in RAM."
+ )
+ logging.info(
+ f" Plane store: {len(self.keys)} x "
+ f"{plane_bytes / 1024**3:.1f} GB on disk at {self._dir}"
+ )
+
+ for key in self.keys:
+ if self._dir is not None:
+ path = self._dir / f"{prefix}_{key}.dat"
+ self._arrays[key] = np.memmap(
+ path, dtype=self.dtype, mode="w+", shape=self.shape
+ )
+ self.paths[key] = str(path)
+ else:
+ self._arrays[key] = np.empty(self.shape, dtype=self.dtype)
+
+ def arrays(self) -> dict[str, Any]:
+ """Plane arrays keyed by name, for in-process use."""
+ return dict(self._arrays)
+
+ def descriptors(self) -> dict[str, str] | None:
+ """``{key: path}`` for reopening in another process, None when in RAM."""
+ return dict(self.paths) if self.on_disk else None
+
+ def flush(self) -> None:
+ """Push dirty memmap pages to the backing files."""
+ for arr in self._arrays.values():
+ if isinstance(arr, np.memmap):
+ arr.flush()
+
+ def release(self) -> None:
+ """Drop references and delete the backing files."""
+ self.flush()
+ self._arrays.clear()
+ for path in self.paths.values():
+ Path(path).unlink(missing_ok=True)
+ self.paths.clear()
+
+
+# ---------------------------------------------------------------------------
+# Process-per-GPU tile workers
+# ---------------------------------------------------------------------------
+
+# Per-process state, populated by _tile_worker_init in each pool worker.
+_WORKER: dict[str, Any] = {}
+
+
+def _tile_worker_init(
+ slot_counter,
+ gpu_ids: tuple[int, ...],
+ channel_source: tuple[str, int],
+ plane_paths: dict[str, str],
+ shape: tuple[int, int],
+ dtype_str: str,
+) -> None:
+ """Pool initializer: claim one GPU, open our own TIFF handle and plane views.
+
+ Each worker claims a distinct device by taking the next slot from a shared
+ counter, so with ``processes == len(gpu_ids)`` every GPU gets exactly one
+ process. Tiles are then pulled from the pool's task queue, which
+ load-balances naturally — unlike static ``gpu_ids[i % n_gpus]`` round-robin,
+ which leaves fast devices idle whenever tile costs differ (the DAPI channel
+ also computes the Laplacian, so its tiles cost roughly 2.5x the others').
+ """
+ # A spawned child never runs main(), so it never runs logging.basicConfig and
+ # the root logger would sit at WARNING -- silencing every worker-side message
+ # exactly where multi-GPU problems would show up.
+ logging.basicConfig(
+ level=logging.INFO,
+ format="%(asctime)s [%(levelname)s] %(message)s",
+ datefmt="%Y-%m-%d %H:%M:%S",
+ force=True,
+ )
+
+ with slot_counter.get_lock():
+ slot = slot_counter.value
+ slot_counter.value += 1
+ gpu_id = gpu_ids[slot % len(gpu_ids)]
+
+ path, page_index = channel_source
+ tif = tifffile.TiffFile(path)
+ channel = _LazyTiffChannel(tif.pages[page_index], source=channel_source)
+ channel._tiff_handles = [tif] # type: ignore[attr-defined]
+
+ dtype = np.dtype(dtype_str)
+ _WORKER.clear()
+ _WORKER.update(
+ gpu_id=gpu_id,
+ channel=channel,
+ planes={
+ key: np.memmap(p, dtype=dtype, mode="r+", shape=tuple(shape))
+ for key, p in plane_paths.items()
+ },
+ )
+ logging.info(f" Tile worker pid={os.getpid()} bound to GPU {gpu_id}")
+
+
+def _tile_worker_run(
+ task: tuple[dict[str, int], int, bool, float],
+) -> tuple[int, float]:
+ """Pool task: compute one tile on this process's GPU. Returns (gpu_id, seconds)."""
+ tile_spec, window_size, include_laplacian, lap_sigma = task
+ t0 = time.perf_counter()
+ _process_tile_on_gpu(
+ _WORKER["channel"],
+ tile_spec,
+ window_size,
+ _WORKER["gpu_id"],
+ _WORKER["planes"],
+ include_laplacian,
+ lap_sigma,
+ )
+ for plane in _WORKER["planes"].values():
+ if isinstance(plane, np.memmap):
+ plane.flush()
+ return int(_WORKER["gpu_id"]), time.perf_counter() - t0
+
+
+def _log_gpu_balance(timings: list[tuple[int, float]], label: str) -> None:
+ """Log per-GPU occupancy so multi-GPU scaling is measurable, not assumed.
+
+ An aggregate wall-clock number cannot reveal load imbalance: it reports
+ max(per-device time), so an idle device looks identical to a saturated one.
+ """
+ if not timings:
+ return
+ per_gpu: dict[int, list[float]] = {}
+ for gpu_id, seconds in timings:
+ per_gpu.setdefault(gpu_id, []).append(seconds)
+ busiest = max(sum(v) for v in per_gpu.values()) or 1.0
+ for gpu_id in sorted(per_gpu):
+ secs = per_gpu[gpu_id]
+ total = sum(secs)
+ logging.info(
+ f" [TIMING] {label} GPU {gpu_id}: {len(secs)} tiles, "
+ f"{total:.1f}s busy ({100.0 * total / busiest:.0f}% of busiest)"
+ )
+
+
+# ---------------------------------------------------------------------------
+# Focus-map computation
+# ---------------------------------------------------------------------------
+
+
+def _process_tile_on_gpu(
+ channel_data,
+ tile_spec: dict[str, int],
+ window_size: int,
+ gpu_id: int,
+ out: dict[str, NDArray[np.float32] | None],
+ include_laplacian: bool = False,
+ lap_sigma: float = 1.0,
+) -> None:
+ """Read a tile, compute focus maps on GPU, trim overlap, write into *out*.
+
+ Writes its trimmed results straight into the caller's preallocated
+ full-resolution planes rather than returning them. This is what bounds
+ host RAM: a returned dict would be pinned by ``Future._result`` until the
+ assembly loop finished, and ``as_completed()`` holds every ``Future`` for
+ the duration of iteration — so returning arrays retains *every* tile's
+ output (roughly one extra full copy of each plane, plus halo). Writing in
+ place means the ``Future`` carries only ``None``.
+
+ Each tile's write region is disjoint from every other tile's (guaranteed by
+ ``_compute_tile_grid``; asserted by ``test_no_write_overlap``), so
+ concurrent in-place writes from the worker threads are safe.
+
+ Args:
+ channel_data: Array-like supporting slicing (_LazyTiffChannel or numpy array).
+ tile_spec: Dict from _compute_tile_grid() with read/write/trim keys.
+ window_size: Convolution window size.
+ gpu_id: CUDA device ID.
+ out: Dict of preallocated full-res planes keyed ``focus_map`` /
+ ``mean_map`` / ``lap_var_map``. A ``None`` value skips that plane.
+ include_laplacian: Also compute Laplacian variance map.
+ lap_sigma: Gaussian sigma for LoG.
+ """
+ # Read tile (triggers actual I/O for lazy TIFF-backed arrays)
+ tile = np.asarray(
+ channel_data[
+ tile_spec["read_y0"] : tile_spec["read_y1"],
+ tile_spec["read_x0"] : tile_spec["read_x1"],
+ ]
+ )
+
+ # Compute on GPU using existing function
+ result = _compute_channel_maps_on_gpu(
+ tile, window_size, gpu_id, include_laplacian, lap_sigma
+ )
+ del tile
+
+ # Trim overlap and write each output straight into the caller's plane
+ tt = tile_spec["trim_top"]
+ tb = tile_spec["trim_bottom"]
+ tl = tile_spec["trim_left"]
+ tr = tile_spec["trim_right"]
+ wy0, wy1 = tile_spec["write_y0"], tile_spec["write_y1"]
+ wx0, wx1 = tile_spec["write_x0"], tile_spec["write_x1"]
+
+ for key in list(result):
+ plane = out.get(key)
+ arr = result.pop(key)
+ if plane is None:
+ continue
+ h, w = arr.shape
+ y1 = h - tb if tb > 0 else h
+ x1 = w - tr if tr > 0 else w
+ plane[wy0:wy1, wx0:wx1] = arr[tt:y1, tl:x1]
+
+
+def _process_tile_for_consumers(
+ channel_data,
+ tile_spec: dict[str, int],
+ window_size: int,
+ gpu_id: int,
+ include_laplacian: bool = False,
+ lap_sigma: float = 1.0,
+ keep_mean_device: bool = False,
+ keep_focus_device: bool = False,
+ drop_mean_host: bool = False,
+) -> tuple[dict[str, NDArray[np.float32]], dict[str, NDArray[np.float32]]]:
+ """Compute one tile and return ``(trimmed, untrimmed)`` maps for consumers.
+
+ Returns rather than writes: with no plane to write into, the parent folds each
+ tile into the consumers and drops it. Both forms are needed —
+ ``RoiOtsuSnrAccumulator`` requires the haloed (untrimmed) tile because an ROI
+ it owns can extend up to ``roi_size - 1`` px past the write region, while the
+ other consumers want the trimmed tile that maps exactly onto the write region.
+
+ Device maps (``focus_map_device`` / ``mean_map_device``) are trimmed too — the
+ write-region view for the trimmed consumers (``CentrePixelSampler``,
+ ``LabeledSumAccumulator``) while the untrimmed dict keeps the full device array
+ for ``RoiOtsuSnrAccumulator``. Slicing a CuPy array is a view, so this adds no
+ device allocation.
+
+ The untrimmed arrays are *views* into the same buffers, so returning both adds
+ no allocation.
+ """
+ tile = np.asarray(
+ channel_data[
+ tile_spec["read_y0"] : tile_spec["read_y1"],
+ tile_spec["read_x0"] : tile_spec["read_x1"],
+ ]
+ )
+ untrimmed = _compute_channel_maps_on_gpu(
+ tile,
+ window_size,
+ gpu_id,
+ include_laplacian,
+ lap_sigma,
+ keep_mean_device=keep_mean_device,
+ keep_focus_device=keep_focus_device,
+ drop_mean_host=drop_mean_host,
+ )
+ del tile
+
+ top = tile_spec["trim_top"]
+ bottom = tile_spec["trim_bottom"]
+ left = tile_spec["trim_left"]
+ right = tile_spec["trim_right"]
+ trimmed = {}
+ for key, arr in untrimmed.items():
+ # Trim every map, host and device alike, to the write region. The trimmed
+ # device slices are what CentrePixelSampler / LabeledSumAccumulator fold;
+ # the untrimmed dict keeps the full device arrays for RoiOtsuSnrAccumulator,
+ # whose owned ROI windows reach past the write region into the halo.
+ height, width = arr.shape
+ y_stop = height - bottom if bottom > 0 else height
+ x_stop = width - right if right > 0 else width
+ trimmed[key] = arr[top:y_stop, left:x_stop]
+ return trimmed, untrimmed
+
+
+def _compute_channel_maps_tiled(
+ channel_data,
+ image_shape: tuple[int, int],
+ window_size: int,
+ gpu_ids: list[int],
+ include_laplacian: bool = False,
+ lap_sigma: float = 1.0,
+ gpu_mem_bytes: int | None = None,
+ plane_dir: Path | None = None,
+ plane_prefix: str = "plane",
+ consumers: list[Any] | None = None,
+) -> dict[str, Any] | None:
+ """Compute focus maps for one channel using tiled multi-GPU processing.
+
+ Tiles the image with convolution-safe overlap, distributes tiles across the
+ available GPUs, and assembles the results into full-resolution planes. Tile
+ size is chosen adaptively to fit GPU memory.
+
+ With more than one GPU and disk-backed planes, tiles are dispatched to one
+ worker **process** per GPU (see ``_tile_worker_init``); otherwise a thread
+ pool is used, which keeps single-GPU behaviour unchanged.
+
+ Args:
+ channel_data: Array-like (_LazyTiffChannel or numpy) with shape matching image_shape.
+ image_shape: (height, width).
+ window_size: Convolution window size.
+ gpu_ids: List of CUDA device IDs.
+ include_laplacian: Also compute Laplacian variance.
+ lap_sigma: Gaussian sigma for LoG.
+ gpu_mem_bytes: Total GPU VRAM in bytes. Auto-detected if None.
+ plane_dir: Directory for the disk-backed output planes. Defaults to
+ the task working directory, which is what Nextflow's ``scratch``
+ directive relocates onto node-local storage.
+ plane_prefix: Filename prefix / log label for this channel's planes.
+
+ Returns:
+ Dict with 'focus_map', 'mean_map', and optionally 'lap_var_map'
+ (full-resolution float32 arrays; memmaps when disk-backed).
+ """
+ H, W = image_shape
+ n_gpus = len(gpu_ids)
+ # Overlap must cover both uniform_filter radius (window_size // 2) and the
+ # Gaussian pre-smoothing in the Laplacian path (~3 * lap_sigma). Using the
+ # full window_size is safe for all kernels and adds negligible I/O overhead.
+ overlap = window_size
+ # Streaming mode needs MORE than the convolution radius. RoiOtsuSnrAccumulator
+ # reads ROI windows out of the *untrimmed* tile, and it claims an ROI by its top
+ # edge y1, so the window reaches `roi_size - 1` px past write_y1 -- while
+ # uniform_filter's reflect padding corrupts the last `window_size // 2` rows of the
+ # read region. With overlap == window_size == roi_size == 35 the padded band starts
+ # at write_y1 + 18 and the ROI reaches write_y1 + 33, so up to 16 of its 35 rows
+ # came from padding.
+ #
+ # Measured before this fix: 2,855 of 335,241 ROIs (0.85 %) had a different
+ # snr_image_otsu_db from the plane-based path, by up to 19.9 % relative, and that
+ # single column was the ONLY difference across 66 output files
+ # (runs 5ZL9TWmJ701ham vs 1AjX0aBBfdbAFQ).
+ #
+ # roi_size is the ROI side, which equals window_size on the production path but is
+ # kept separate here because the requirement is genuinely about the ROI, not the
+ # kernel.
+ if consumers:
+ roi_reach = window_size # ROI side length; grid stride defaults to roi_size
+ needed = roi_reach - 1 + window_size // 2
+ if needed > overlap:
+ overlap = needed
+ # VRAM cost of the larger halo, checked rather than assumed: the strip height from
+ # _compute_adaptive_strip_height does NOT include the overlap, so the read region
+ # always exceeds that budget by 2*overlap rows. At the production geometry (20409-row
+ # strips, 53908 px wide, ~24 B/px of concurrent buffers) going 35 -> 51 adds 32 rows,
+ # i.e. 0.039 GB, taking a tile from 24.68 to 24.71 GB of the 44.5 GB device. The
+ # budget was already optimistic by 70 rows; it is now optimistic by 102.
+
+ # Fail loud if lap_sigma is raised past what the halo can cover: the LoG
+ # needs window_size//2 (uniform_filter) + ~4*lap_sigma+1 (Gaussian+Laplace)
+ # of halo; beyond `overlap` the DAPI lap_var map would differ from the
+ # whole-image result at tile seams (silently). Default lap_sigma=1.0 is safe.
+ if include_laplacian:
+ lap_halo = window_size // 2 + int(math.ceil(4 * lap_sigma)) + 1
+ if lap_halo > overlap:
+ logging.warning(
+ f" lap_sigma={lap_sigma} needs {lap_halo}px halo > overlap "
+ f"{overlap}px; Laplacian tile seams may differ from whole-image. "
+ f"Raise the tile overlap or lower lap_sigma."
+ )
+
+ # Full-width row strips rather than square tiles: a strip's write region is
+ # one contiguous byte range, and it needs halo on top/bottom only. See
+ # _compute_adaptive_strip_height for the measurements that motivated this.
+ strip_height = _compute_adaptive_strip_height(H, W, n_gpus, gpu_mem_bytes)
+ # Derived from the consumers, not supplied by the caller: BlockMeanAccumulator needs
+ # write origins on multiples of its block size, and a caller who had to remember to
+ # say so would discover the omission only as a RuntimeError mid-fold.
+ align_writes_to = 1
+ for consumer in consumers or ():
+ align_writes_to = math.lcm(
+ align_writes_to, int(getattr(consumer, "write_alignment", 1) or 1)
+ )
+ if align_writes_to > 1:
+ # Align write boundaries to the heatmap block size so every block lies wholly
+ # inside one tile; BlockMeanAccumulator then reduces it with one reshape and the
+ # canvas is bit-exact. See that class for the 16-pixel figure difference.
+ # max(), not a conditional skip: a strip shorter than one block would otherwise
+ # be left unaligned and trip the accumulator's contract check at runtime. Rounding
+ # up to one block costs at most align_writes_to rows of halo.
+ strip_height = max(
+ align_writes_to, (strip_height // align_writes_to) * align_writes_to
+ )
+ tiles = _compute_tile_grid(H, W, strip_height, overlap, tile_width=W)
+
+ logging.info(
+ f" Tiled processing: {len(tiles)} row strip(s) "
+ f"({strip_height}x{W}px), {n_gpus} GPU(s), overlap={overlap}px"
+ )
+
+ # Ensure CUDA_PATH is set for CuPy NVRTC kernel compilation in worker
+ # threads. CuPy auto-detects CUDA in the main thread but worker threads
+ # can fail with "Failed to auto-detect CUDA root directory". The pip
+ # package nvidia-cuda-runtime-cu12 installs headers under
+ # site-packages/nvidia/cuda_runtime/include/.
+ if "CUDA_PATH" not in os.environ:
+ try:
+ import nvidia.cuda_runtime as _cr
+
+ _cr_dir = _cr.__path__[0] # .../site-packages/nvidia/cuda_runtime
+ if os.path.isdir(os.path.join(_cr_dir, "include")):
+ os.environ["CUDA_PATH"] = _cr_dir
+ logging.info(f" Set CUDA_PATH={_cr_dir}")
+ except (ImportError, IndexError, AttributeError):
+ pass # CUDA_PATH remains unset; CuPy will try its own detection
+
+ # Warm up CuPy kernel cache in the main thread. NVRTC compiles kernels on
+ # first use; running a tiny operation here populates the disk cache so that
+ # worker threads hit the cache instead of compiling in parallel.
+ try:
+ _warmup = cp.array([1.0], dtype=cp.float32)
+ cupyx.scipy.ndimage.uniform_filter(_warmup.reshape(1, 1), size=1)
+ del _warmup
+ except Exception:
+ pass
+
+ # ------------------------------------------------------------------
+ # Consumer mode: no planes at all. Each tile is folded into the consumers and
+ # dropped, so peak host memory is O(max_inflight * tile) + O(n_rois, n_cells) --
+ # NOT O(tile). Results sit in their futures until the parent pops and folds them,
+ # and folding is serialised here, so up to max_inflight tiles can be complete and
+ # resident at once. Measured on a 102045x53908 sample: 12 strips of 8504 rows is
+ # 1.71 GB per map, so DAPI's 3 maps x 8 in flight is ~41 GB worst case against
+ # ~154 GB of planes for the whole image.
+ # ------------------------------------------------------------------
+ if consumers:
+ # Each worker THREAD folds its tiles into its OWN set of consumer
+ # accumulators, then the per-thread partials are reduced into the caller's
+ # consumers after the executor drains. There is no fold lock, so the n_gpus
+ # workers fold CONCURRENTLY -- the fold is the wall clock here (dominated by
+ # RoiOtsuSnrAccumulator's per-ROI Otsu, LabeledSumAccumulator's bincounts),
+ # so a single fold_lock serialised the folds and threw away the multi-GPU
+ # parallelism (the DAPI fold measured ~240 s of the run behind that lock).
+ #
+ # Correctness of the reduce, per consumer:
+ # * RoiOtsuSnrAccumulator / CentrePixelSampler: every ROI is owned by
+ # exactly one tile (the halo guarantees it), so partials hold DISJOINT
+ # per-ROI entries -- merge is an overlay, bit-identical to the serial
+ # fold regardless of order.
+ # * BlockMeanAccumulator: write origins are aligned to the block size, so
+ # every block lies wholly in one tile -> one partial; canvases are
+ # disjoint and add exactly.
+ # * LabeledSumAccumulator: per-label float64 count/sum arrays are added
+ # element-wise. This regroups the additions relative to the serial
+ # tile-order fold; that it still reproduces them byte-for-byte is
+ # empirical, pinned by tests/test_fold_parallel_equivalence.
+ #
+ # Folding in the worker (not the dispatch loop) is also what bounds memory: a
+ # tile is released as soon as it is folded, instead of staying alive in its
+ # Future until the parent catches up.
+ gpu_pool: queue.Queue[int] = queue.Queue()
+ for _gpu in gpu_ids:
+ gpu_pool.put(_gpu)
+
+ # Per-worker-thread consumer sets, created lazily on the thread's first tile
+ # and registered under a lock with a stable creation index so the reduce
+ # order is deterministic (run-to-run reproducible). threading.local() keys
+ # the set to the OS thread the ThreadPoolExecutor reuses, so a thread folds
+ # all its tiles into the one set it created.
+ _thread_state = threading.local()
+ partials: list[tuple[int, list[Any]]] = []
+ partials_lock = threading.Lock()
+ per_consumer_lock = threading.Lock()
+
+ def _thread_consumers() -> list[Any]:
+ local = getattr(_thread_state, "consumers", None)
+ if local is not None:
+ return local
+ local = [c.spawn() for c in consumers]
+ with partials_lock:
+ index = len(partials)
+ partials.append((index, local))
+ _thread_state.consumers = local
+ return local
+
+ # Keep the maps resident on the GPU when a consumer folds a reduction there
+ # (RoiOtsuSnrAccumulator, LabeledSumAccumulator, CentrePixelSampler do their
+ # per-ROI / per-label / centre-pixel reductions on-device). The host mean map
+ # is dropped when every consumer that reads the mean map takes the device one
+ # -- then nothing reads the host copy and the D2H transfer is eliminated. The
+ # host focus map is always produced: BlockMeanAccumulator (Figure 5) reduces
+ # it in float32 on the host, which a GPU reduction cannot match to float
+ # rounding, so it never sets wants_device_focus.
+ keep_mean_device = any(
+ getattr(c, "wants_device_mean", False) for c in consumers
+ )
+ keep_focus_device = any(
+ getattr(c, "wants_device_focus", False) for c in consumers
+ )
+ drop_mean_host = keep_mean_device and not any(
+ getattr(c, "reads_host_mean", False) for c in consumers
+ )
+
+ timings: list[tuple[int, float]] = []
+ fold_seconds = 0.0
+ per_consumer: dict[str, float] = {}
+
+ def _run_tile(spec) -> tuple[int, float, float]:
+ gpu_id = gpu_pool.get()
+ try:
+ t0 = time.perf_counter()
+ trimmed, untrimmed = _process_tile_for_consumers(
+ channel_data,
+ spec,
+ window_size,
+ gpu_id,
+ include_laplacian,
+ lap_sigma,
+ keep_mean_device=keep_mean_device,
+ keep_focus_device=keep_focus_device,
+ drop_mean_host=drop_mean_host,
+ )
+ compute_seconds = time.perf_counter() - t0
+
+ # Fold into THIS thread's own consumer set -- no lock, so the n_gpus
+ # workers fold concurrently.
+ local_consumers = _thread_consumers()
+ t1 = time.perf_counter()
+ local_per: dict[str, float] = {}
+ for consumer in local_consumers:
+ t_c = time.perf_counter()
+ if getattr(consumer, "wants_untrimmed", False):
+ consumer.consume(spec, untrimmed)
+ else:
+ consumer.consume(spec, trimmed)
+ name = type(consumer).__name__
+ local_per[name] = local_per.get(name, 0.0) + (
+ time.perf_counter() - t_c
+ )
+ fold = time.perf_counter() - t1
+ # Aggregate the per-consumer fold time across threads (these overlap
+ # in wall clock now, so this is summed CPU, not wall time).
+ with per_consumer_lock:
+ for name, seconds in local_per.items():
+ per_consumer[name] = per_consumer.get(name, 0.0) + seconds
+ del trimmed, untrimmed
+ return gpu_id, compute_seconds, fold
+ finally:
+ # Release the GPU slot AFTER the fold, not before it. With
+ # keep_mean_device/keep_focus_device this tile's mean/focus maps are
+ # device arrays that the fold reduces on THIS card. Releasing the slot
+ # before the fold (the previous behaviour) let another worker grab this
+ # card and start a second tile's compute while these maps were still
+ # resident -- an unbounded cross-worker pile-up (up to one compute +
+ # n_gpus-1 held maps on one card) that breaks the per-device VRAM bound
+ # #55 established, since _compute_adaptive_strip_height sizes for one
+ # tile's compute arrays only. Holding the slot through the fold bounds
+ # each card to a single tile at a time (its compute arrays, then its
+ # held maps), so the strip-height budget is the real per-card ceiling.
+ # The trade is small: the fold runs on this card regardless (on-device
+ # reduction) and is short since the GPU Otsu change, so the only lost
+ # overlap is a peer tile's compute starting on this card mid-fold.
+ gpu_pool.put(gpu_id)
+
+ t_dispatch = time.perf_counter()
+ logging.info(
+ f" Consumer mode: {len(tiles)} tiles, {n_gpus} GPU(s), "
+ f"per-worker parallel fold, no planes materialised"
+ )
+ with ThreadPoolExecutor(max_workers=n_gpus) as executor:
+ for gpu_id, compute_seconds, fold in executor.map(_run_tile, tiles):
+ timings.append((gpu_id, compute_seconds))
+ fold_seconds += fold
+
+ # Reduce the per-thread partials into the caller's consumers, in a
+ # deterministic order (stable creation index). The caller's consumers never
+ # consumed a tile, so they start empty and this fills them for the
+ # downstream finalize().
+ for _index, local_consumers in sorted(partials, key=lambda item: item[0]):
+ for target, part in zip(consumers, local_consumers):
+ target.merge(part)
+
+ elapsed = time.perf_counter() - t_dispatch
+ logging.info(
+ f" [TIMING] {plane_prefix} tiled compute (consumers, {len(tiles)} "
+ f"tiles, {n_gpus} GPU): {elapsed:.1f}s"
+ )
+ logging.info(
+ f" [TIMING] {plane_prefix} consumer folding: {fold_seconds:.1f}s "
+ f"({100.0 * fold_seconds / max(elapsed, 1e-9):.0f}% of wall clock)"
+ )
+ for name, seconds in sorted(per_consumer.items(), key=lambda kv: -kv[1]):
+ logging.info(
+ f" [TIMING] {plane_prefix} fold {name}: {seconds:.1f}s "
+ f"({100.0 * seconds / max(fold_seconds, 1e-9):.0f}% of folding)"
+ )
+ # Read/decode/convolve split. read+decode are serialized under the reader's
+ # lock, so compare their sum to the wall clock (elapsed): sum ~ elapsed means
+ # the reader is the wall (IO/decode-bound, GPUs starve); sum << elapsed means
+ # the GPU convolution dominates. total_compute is the summed per-tile
+ # read+upload+convolve across all workers, so convolve+upload ~ total_compute
+ # - read - decode.
+ read_s = float(getattr(channel_data, "_read_seconds", 0.0))
+ decode_s = float(getattr(channel_data, "_decode_seconds", 0.0))
+ read_gb = float(getattr(channel_data, "_read_bytes", 0)) / 1024**3
+ total_compute = sum(secs for _gid, secs in timings)
+ logging.info(
+ f" [TIMING] {plane_prefix} read (S3/fh): {read_s:.1f}s "
+ f"({read_gb:.2f} GB, {read_gb / max(read_s, 1e-9):.2f} GB/s), "
+ f"decode: {decode_s:.1f}s, read+decode = "
+ f"{100.0 * (read_s + decode_s) / max(elapsed, 1e-9):.0f}% of wall clock"
+ )
+ logging.info(
+ f" [TIMING] {plane_prefix} convolve+upload ~= "
+ f"{max(total_compute - read_s - decode_s, 0.0):.1f}s "
+ f"(sum tile-compute {total_compute:.1f}s - read {read_s:.1f}s - "
+ f"decode {decode_s:.1f}s)"
+ )
+ _log_gpu_balance(timings, f"{plane_prefix} streamed")
+ return None
+
+ # Allocate the output planes. With a scratch dir configured these are
+ # disk-backed memmaps rather than anonymous RAM, which both bounds the
+ # resident set and lets the tile workers be separate processes — see
+ # _PlaneStore. Workers write their trimmed tiles straight into these, so no
+ # per-tile result is ever retained (`as_completed()` holds every Future for
+ # the duration of iteration, so a worker that *returned* its arrays would
+ # keep all of them alive until the executor block exited).
+ keys = ["focus_map", "mean_map"] + (["lap_var_map"] if include_laplacian else [])
+ store = _PlaneStore(
+ (H, W), keys, plane_dir=plane_dir, prefix=plane_prefix, dtype=np.float32
+ )
+ out_planes: dict[str, Any] = store.arrays()
+ out_planes.setdefault("lap_var_map", None)
+
+ channel_source = getattr(channel_data, "_source", None)
+ plane_paths = store.descriptors()
+ # Process mode needs: more than one GPU to be worth it, a reopenable channel
+ # (a live TiffPage holds an OS handle and a lock, so it cannot be pickled),
+ # and file-backed planes to write into. Single-GPU runs keep the thread
+ # path, so their behaviour is unchanged.
+ use_processes = (
+ n_gpus > 1 and channel_source is not None and plane_paths is not None
+ )
+ tasks = [(spec, window_size, include_laplacian, lap_sigma) for spec in tiles]
+ t_dispatch = time.perf_counter()
+
+ if use_processes:
+ assert channel_source is not None and plane_paths is not None
+ # One process per GPU. Separate interpreters mean the host-side numpy
+ # in each tile (input cast, _sanitize, post-D2H cast) no longer
+ # serialises on a shared GIL, and each process builds its own CUDA
+ # context. Must be "spawn": CUDA does not survive fork().
+ logging.info(
+ f" Dispatching {len(tasks)} tiles to {n_gpus} worker process(es), "
+ f"one per GPU {gpu_ids} (spawn)"
+ )
+ ctx = multiprocessing.get_context("spawn")
+ slot_counter = ctx.Value("i", 0)
+ with ctx.Pool(
+ processes=n_gpus,
+ initializer=_tile_worker_init,
+ initargs=(
+ slot_counter,
+ tuple(gpu_ids),
+ channel_source,
+ plane_paths,
+ (H, W),
+ np.dtype(np.float32).str,
+ ),
+ ) as pool:
+ timings = list(pool.imap_unordered(_tile_worker_run, tasks))
+ else:
+ # Pipelined I/O + GPU: each worker reads its tile from TIFF then
+ # computes on its assigned GPU. The TIFF file handle lock serializes
+ # reads, but I/O for tile N+1 overlaps with GPU compute for tile N.
+ def _run_tile(index: int, spec: dict[str, int]) -> tuple[int, float]:
+ gpu_id = gpu_ids[index % n_gpus]
+ t0 = time.perf_counter()
+ _process_tile_on_gpu(
+ channel_data,
+ spec,
+ window_size,
+ gpu_id,
+ out_planes,
+ include_laplacian,
+ lap_sigma,
+ )
+ return gpu_id, time.perf_counter() - t0
+
+ with ThreadPoolExecutor(max_workers=n_gpus) as executor:
+ futures = [
+ executor.submit(_run_tile, i, spec) for i, spec in enumerate(tiles)
+ ]
+ timings = [f.result() for f in as_completed(futures)]
+
+ store.flush()
+ mode = "process" if use_processes else "thread"
+ logging.info(
+ f" [TIMING] {plane_prefix} tiled compute ({mode}, {len(tiles)} tiles, "
+ f"{n_gpus} GPU): {time.perf_counter() - t_dispatch:.1f}s"
+ )
+ _log_gpu_balance(timings, f"{plane_prefix} tiled")
+
+ # Per-tile results are already sanitized in _compute_channel_maps_on_gpu,
+ # so no redundant final full-array _sanitize here.
+ return {
+ "focus_map": out_planes["focus_map"],
+ "mean_map": out_planes["mean_map"],
+ "lap_var_map": out_planes.get("lap_var_map"),
+ }
+
+
+def compute_all_focus_maps(
+ channels: NDArray[np.generic],
+ window_size: int = 35,
+ use_gpu: bool = False,
+ gpu_ids: list[int] | None = None,
+ lap_sigma: float = 1.0,
+) -> dict[str, NDArray[np.float32] | None]:
+ """Compute focus maps for all available image channels.
+
+ When ``gpu_ids`` contains multiple IDs, channel computations are
+ distributed across GPUs in parallel using a thread pool. Each channel's
+ convolution work runs entirely on one GPU; with 3 channels and 3+ GPUs
+ all channels are processed concurrently.
+
+ Args:
+ channels: Either a 2-D array (single DAPI channel, shape ``(H, W)``)
+ or a 3-D array with channels first (shape ``(C, H, W)``).
+ window_size: Side length of the square averaging window.
+ use_gpu: Use CuPy GPU backend when ``True``. Ignored when *gpu_ids*
+ is provided (GPU is assumed).
+ gpu_ids: List of CUDA device IDs for multi-GPU parallelism. If
+ ``None`` and ``use_gpu=True``, uses device 0 only. If ``None``
+ and ``use_gpu=False``, runs on CPU.
+
+ Returns:
+ Dictionary with the following keys (values are ``None`` when the
+ corresponding channel is not present in *channels*):
+
+ - ``dapi_focus_map`` — CCFS focus map for DAPI (channel 0).
+ - ``dapi_mean_map`` — Local mean map for DAPI.
+ - ``dapi_lap_var_map`` — Laplacian variance map for DAPI.
+ - ``boundary_focus_map`` — CCFS focus map for Boundary (channel 1).
+ - ``boundary_mean_map`` — Local mean map for Boundary.
+ - ``intrna_focus_map`` — CCFS focus map for IntRNA (channel 2).
+ - ``intrna_mean_map`` — Local mean map for IntRNA.
+
+ Raises:
+ ValueError: If *channels* has fewer than 2 or more than 3 dimensions.
+ """
+ # Resolve GPU configuration
+ if gpu_ids is not None and len(gpu_ids) > 0:
+ use_gpu = True
+ elif use_gpu and gpu_ids is None:
+ gpu_ids = [0]
+
+ # Split channels — .copy() so the parent 3-D array can be freed.
+ if channels.ndim == 2:
+ dapi = channels.copy()
+ boundary = None
+ intrna = None
+ elif channels.ndim == 3:
+ n_ch = channels.shape[0]
+ dapi = channels[0].copy()
+ boundary = channels[1].copy() if n_ch > 1 else None
+ intrna = channels[2].copy() if n_ch > 2 else None
+ else:
+ raise ValueError(
+ f"Expected 2-D or 3-D array, got {channels.ndim}-D "
+ f"(shape {channels.shape})."
+ )
+ del channels
+
+ # ---- Multi-GPU path: distribute channels across GPUs ----
+ if use_gpu and gpu_ids is not None and len(gpu_ids) > 0:
+ # Build work items: (channel_name, channel_data, include_laplacian)
+ work_items: list[tuple[str, NDArray[np.generic], bool]] = [
+ ("dapi", dapi, True), # DAPI always gets Laplacian
+ ]
+ if boundary is not None:
+ work_items.append(("boundary", boundary, False))
+ if intrna is not None:
+ work_items.append(("intrna", intrna, False))
+
+ # Assign GPUs round-robin
+ results: dict[str, dict[str, NDArray[np.float32]]] = {}
+ n_gpus = len(gpu_ids)
+ logging.info(
+ f" Distributing {len(work_items)} channel(s) across {n_gpus} GPU(s): {gpu_ids}"
+ )
+
+ t_gpu = time.perf_counter()
+ gpu_failed = False
+ try:
+ with ThreadPoolExecutor(
+ max_workers=min(len(work_items), n_gpus)
+ ) as executor:
+ futures = {}
+ for idx, (name, ch_data, inc_lap) in enumerate(work_items):
+ assigned_gpu = gpu_ids[idx % n_gpus]
+ future = executor.submit(
+ _compute_channel_maps_on_gpu,
+ ch_data,
+ window_size,
+ assigned_gpu,
+ inc_lap,
+ lap_sigma,
+ )
+ futures[future] = name
+
+ for future in as_completed(futures):
+ ch_name = futures[future]
+ results[ch_name] = future.result()
+ logging.info(
+ f" [TIMING] GPU compute (multi-GPU, {len(work_items)} channels): {time.perf_counter() - t_gpu:.1f}s"
+ )
+ except Exception as gpu_err:
+ logging.warning(
+ f" WARNING: GPU computation failed ({gpu_err}), falling back to CPU..."
+ )
+ gpu_failed = True
+
+ if not gpu_failed:
+ # Assemble output dict from GPU results
+ dapi_res = results["dapi"]
+ return {
+ "dapi_focus_map": dapi_res["focus_map"],
+ "dapi_mean_map": dapi_res["mean_map"],
+ "dapi_lap_var_map": dapi_res.get("lap_var_map"),
+ "boundary_focus_map": results["boundary"]["focus_map"]
+ if "boundary" in results
+ else None,
+ "boundary_mean_map": results["boundary"]["mean_map"]
+ if "boundary" in results
+ else None,
+ "intrna_focus_map": results["intrna"]["focus_map"]
+ if "intrna" in results
+ else None,
+ "intrna_mean_map": results["intrna"]["mean_map"]
+ if "intrna" in results
+ else None,
+ }
+ # else: fall through to CPU path below
+
+ # ---- Single-GPU or CPU fallback path ----
+ # Reached when: no multi-GPU available, use_gpu=False, or GPU OOM fallback
+ use_gpu = (
+ False # Force CPU to avoid repeated OOM if we fell through from GPU failure
+ )
+ gpu_id = gpu_ids[0] if gpu_ids else 0
+
+ t_gpu = time.perf_counter()
+ # Process channels sequentially, freeing each input before the next
+ # to keep peak memory at ~1 channel + its output maps.
+
+ # DAPI — always present
+ dapi_focus_map, dapi_mean_map = compute_ccfs_map(
+ dapi, window_size=window_size, use_gpu=use_gpu, gpu_id=gpu_id
+ )
+ dapi_lap_var_map = compute_laplacian_variance_map(
+ dapi,
+ window_size=window_size,
+ use_gpu=use_gpu,
+ gpu_id=gpu_id,
+ lap_sigma=lap_sigma,
+ )
+ del dapi
+
+ # Boundary (channel 1)
+ boundary_focus_map: NDArray[np.float32] | None = None
+ boundary_mean_map: NDArray[np.float32] | None = None
+ if boundary is not None:
+ boundary_focus_map, boundary_mean_map = compute_ccfs_map(
+ boundary, window_size=window_size, use_gpu=use_gpu, gpu_id=gpu_id
+ )
+ del boundary
+
+ # IntRNA (channel 2)
+ intrna_focus_map: NDArray[np.float32] | None = None
+ intrna_mean_map: NDArray[np.float32] | None = None
+ if intrna is not None:
+ intrna_focus_map, intrna_mean_map = compute_ccfs_map(
+ intrna, window_size=window_size, use_gpu=use_gpu, gpu_id=gpu_id
+ )
+ del intrna
+
+ backend_label = "GPU" if use_gpu else "CPU"
+ logging.info(
+ f" [TIMING] {backend_label} compute (single device, all channels): {time.perf_counter() - t_gpu:.1f}s"
+ )
+
+ return {
+ "dapi_focus_map": dapi_focus_map,
+ "dapi_mean_map": dapi_mean_map,
+ "dapi_lap_var_map": dapi_lap_var_map,
+ "boundary_focus_map": boundary_focus_map,
+ "boundary_mean_map": boundary_mean_map,
+ "intrna_focus_map": intrna_focus_map,
+ "intrna_mean_map": intrna_mean_map,
+ }
+
+
+# ---------------------------------------------------------------------------
+# Down-sampling to per-tile DataFrame
+# ---------------------------------------------------------------------------
+
+
+def _build_roi_grid(
+ image_shape: tuple[int, int],
+ roi_size: int,
+ stride: int,
+ tissue_mask: NDArray[np.generic] | None,
+ downsample_factor: int = 8,
+) -> dict[str, Any]:
+ """Build the tile grid and compute tissue coverages (one-time setup).
+
+ Returns a dict with keys: ``x1_arr``, ``x2_arr``, ``y1_arr``, ``y2_arr``,
+ ``cx``, ``cy``, ``n_rois``, ``height``, ``width``, ``tissue_coverages``,
+ ``is_boundary_roi``, ``roi_coords_arr``.
+ """
+ height, width = image_shape
+
+ # Build ROI grid — vectorized
+ x1_arr, x2_arr, y1_arr, y2_arr, cx, cy = compute_roi_grid(
+ height, width, roi_size, stride
+ )
+ n_rois = len(x1_arr)
+ roi_coords_arr = np.stack([x1_arr, x2_arr, y1_arr, y2_arr], axis=1)
+
+ # Tissue coverage
+ if tissue_mask is not None:
+ binary_mask = (tissue_mask > 0).astype(np.float64)
+ mask_h, mask_w = binary_mask.shape
+ x1_ds = np.clip(x1_arr // downsample_factor, 0, mask_w)
+ x2_ds = np.clip((x2_arr - 1) // downsample_factor + 1, 0, mask_w)
+ y1_ds = np.clip(y1_arr // downsample_factor, 0, mask_h)
+ y2_ds = np.clip((y2_arr - 1) // downsample_factor + 1, 0, mask_h)
+
+ integral = np.zeros((mask_h + 1, mask_w + 1), dtype=np.float64)
+ integral[1:, 1:] = np.cumsum(np.cumsum(binary_mask, axis=0), axis=1)
+ block_sums = (
+ integral[y2_ds, x2_ds]
+ - integral[y1_ds, x2_ds]
+ - integral[y2_ds, x1_ds]
+ + integral[y1_ds, x1_ds]
+ )
+ block_w = x2_ds - x1_ds
+ block_h = y2_ds - y1_ds
+ total_pixels_arr = (block_w * block_h).astype(np.float64)
+ total_pixels_arr[total_pixels_arr == 0] = 1.0
+ tissue_coverages = block_sums / total_pixels_arr
+ else:
+ tissue_coverages = np.ones(n_rois, dtype=np.float64)
+
+ half_win = roi_size // 2
+ is_boundary_roi = (
+ (cx < half_win)
+ | (cy < half_win)
+ | (cx >= width - half_win)
+ | (cy >= height - half_win)
+ )
+
+ return {
+ "x1_arr": x1_arr,
+ "x2_arr": x2_arr,
+ "y1_arr": y1_arr,
+ "y2_arr": y2_arr,
+ "cx": cx,
+ "cy": cy,
+ "n_rois": n_rois,
+ "height": height,
+ "width": width,
+ "tissue_coverages": tissue_coverages,
+ "is_boundary_roi": is_boundary_roi,
+ "roi_coords_arr": roi_coords_arr,
+ }
+
+
+def compute_roi_grid(
+ height: int,
+ width: int,
+ roi_size: int = 35,
+ stride: int | None = None,
+) -> tuple[
+ NDArray[np.int64],
+ NDArray[np.int64],
+ NDArray[np.int64],
+ NDArray[np.int64],
+ NDArray[np.int64],
+ NDArray[np.int64],
+]:
+ """The ROI grid and its centre pixels: ``(x1, x2, y1, y2, cx, cy)``.
+
+ Shared by :func:`downsample_maps_to_roi_dataframe` and the streaming path,
+ which needs the centres up front to build :class:`CentrePixelSampler`. The two
+ must agree exactly — the sampler fills a positional array that the DataFrame
+ then labels with these coordinates, so any drift mislabels every ROI silently
+ rather than raising.
+
+ Partial tiles at the right/bottom edge are kept when at least half a tile
+ remains, and their centre is the midpoint of the *clipped* extent, not
+ ``x1 + roi_size // 2``.
+ """
+ if stride is None:
+ stride = roi_size
+ yy, xx = np.meshgrid(
+ np.arange(0, height, stride), np.arange(0, width, stride), indexing="ij"
+ )
+ x1_all, y1_all = xx.ravel(), yy.ravel()
+ x2_all = np.minimum(x1_all + roi_size, width)
+ y2_all = np.minimum(y1_all + roi_size, height)
+
+ valid = (x2_all - x1_all >= roi_size // 2) & (y2_all - y1_all >= roi_size // 2)
+ x1_arr, x2_arr = x1_all[valid], x2_all[valid]
+ y1_arr, y2_arr = y1_all[valid], y2_all[valid]
+ if len(x1_arr) == 0:
+ raise ValueError(
+ f"No tiles generated. Image: {height}x{width}, "
+ f"roi_size={roi_size}, stride={stride}."
+ )
+ return (
+ x1_arr,
+ x2_arr,
+ y1_arr,
+ y2_arr,
+ (x1_arr + x2_arr) // 2,
+ (y1_arr + y2_arr) // 2,
+ )
+
+
+def downsample_maps_to_roi_dataframe(
+ focus_maps: dict[str, NDArray[np.float32] | None],
+ tissue_mask: NDArray[np.generic] | None,
+ roi_size: int = 35,
+ stride: int | None = None,
+ downsample_factor: int = 8,
+ min_tissue_coverage: float = 0.0,
+ image_shape: tuple[int, int] | None = None,
+ presampled: dict[str, NDArray[np.float64]] | None = None,
+) -> pd.DataFrame:
+ """Down-sample pixel-level focus maps to per-tile scalars.
+
+ When *presampled* is given, the centre-pixel sampling step is skipped and those
+ per-ROI arrays are used instead. That is how the streaming path reuses every
+ column, threshold and normalisation below without ever assembling a
+ full-resolution plane: ``CentrePixelSampler`` produces exactly the arrays this
+ function would have read out of ``map[cy, cx]``. Keys are the map names
+ (``dapi_focus_map``, ``boundary_mean_map``, ...); *focus_maps* may then be an
+ empty dict.
+
+ The output DataFrame has **exactly** the same columns as the legacy
+ ``calculate_roi_focusscore()`` function, ensuring full backward
+ compatibility.
+
+ For each tile the focus score is obtained by sampling the **centre pixel**
+ of the corresponding pixel-level map. Because ``uniform_filter`` at pixel
+ ``(cy, cx)`` computes statistics over the surrounding ``window_size x
+ window_size`` window, the centre pixel of a ``roi_size x roi_size`` tile
+ yields the exact block-level statistic (for non-edge tiles). Edge tiles may
+ differ slightly because ``uniform_filter`` uses ``mode='reflect'`` padding
+ while the legacy code clips to the image boundary.
+
+ Args:
+ focus_maps: Dictionary returned by :func:`compute_all_focus_maps`.
+ Required key: ``dapi_focus_map``. All other keys may be ``None``.
+ tissue_mask: Labelled tissue mask at down-sampled resolution (e.g.
+ level 3). Pass ``None`` to skip tissue filtering (all tiles get
+ ``tissue_coverage=1.0``).
+ roi_size: Side length of each square tile in pixels.
+ stride: Grid spacing in pixels. Defaults to *roi_size* (non-
+ overlapping grid).
+ downsample_factor: Scale factor between full-resolution coordinates
+ and *tissue_mask* coordinates (default 8 for level 3).
+ min_tissue_coverage: Minimum tissue fraction for a tile to be
+ included. ``0.0`` means any tissue overlap is sufficient. This
+ parameter is recorded but **not** used for filtering — all tiles
+ are returned so the caller can filter as needed.
+ image_shape: ``(height, width)`` of the full-resolution image. If
+ ``None`` the shape is inferred from ``dapi_focus_map``.
+
+ Returns:
+ :class:`~pandas.DataFrame` with columns identical to the legacy
+ ``calculate_roi_focusscore()`` output.
+
+ Raises:
+ ValueError: If ``dapi_focus_map`` is missing from *focus_maps* or the
+ Tile grid is empty.
+ """
+ # ------------------------------------------------------------------
+ # Unpack maps
+ # ------------------------------------------------------------------
+ # In streaming mode there are no planes to inspect, so channel presence — which
+ # decides both the sampling below and the column set — comes from the same dict
+ # the samples do.
+ source: dict[str, object] = presampled if presampled is not None else focus_maps # type: ignore[assignment]
+ if source.get("dapi_focus_map") is None:
+ which = "presampled" if presampled is not None else "focus_maps"
+ raise ValueError(f"{which} must contain 'dapi_focus_map'.")
+ has_boundary = source.get("boundary_focus_map") is not None
+ has_intrna = source.get("intrna_focus_map") is not None
+
+ dapi_focus_map = focus_maps.get("dapi_focus_map")
+ dapi_mean_map = focus_maps.get("dapi_mean_map")
+ dapi_lap_var_map = focus_maps.get("dapi_lap_var_map")
+ boundary_focus_map = focus_maps.get("boundary_focus_map")
+ boundary_mean_map = focus_maps.get("boundary_mean_map")
+ intrna_focus_map = focus_maps.get("intrna_focus_map")
+ intrna_mean_map = focus_maps.get("intrna_mean_map")
+
+ # ------------------------------------------------------------------
+ # Image shape
+ # ------------------------------------------------------------------
+ if image_shape is not None:
+ height, width = image_shape
+ elif dapi_focus_map is not None:
+ height, width = dapi_focus_map.shape
+ else:
+ raise ValueError("image_shape is required when passing presampled arrays.")
+
+ if stride is None:
+ stride = roi_size
+
+ # ------------------------------------------------------------------
+ # Build ROI grid — vectorized (identical logic to legacy code)
+ # ------------------------------------------------------------------
+ t_grid = time.perf_counter()
+
+ x_starts = np.arange(0, width, stride)
+ y_starts = np.arange(0, height, stride)
+ yy, xx = np.meshgrid(y_starts, x_starts, indexing="ij")
+ x1_all = xx.ravel()
+ y1_all = yy.ravel()
+ x2_all = np.minimum(x1_all + roi_size, width)
+ y2_all = np.minimum(y1_all + roi_size, height)
+
+ # Filter out small edge ROIs (same threshold as legacy code)
+ valid = (x2_all - x1_all >= roi_size // 2) & (y2_all - y1_all >= roi_size // 2)
+ x1_arr = x1_all[valid]
+ x2_arr = x2_all[valid]
+ y1_arr = y1_all[valid]
+ y2_arr = y2_all[valid]
+
+ n_rois = len(x1_arr)
+ if n_rois == 0:
+ raise ValueError(
+ f"No tiles generated from grid. "
+ f"Image: {height}x{width}, roi_size={roi_size}, stride={stride}."
+ )
+
+ # Keep a structured array for backward-compatible DataFrame columns
+ # Columns: x1, x2, y1, y2
+ roi_coords_arr = np.stack([x1_arr, x2_arr, y1_arr, y2_arr], axis=1)
+
+ logging.info(
+ f" [TIMING] Tile grid generation ({n_rois} tiles): {time.perf_counter() - t_grid:.1f}s"
+ )
+
+ # ------------------------------------------------------------------
+ # Tissue coverage — vectorized via block sums
+ # ------------------------------------------------------------------
+ t_tissue = time.perf_counter()
+
+ tissue_coverages_arr: NDArray[np.float64]
+ if tissue_mask is not None:
+ # Convert tissue mask to binary
+ binary_mask = (tissue_mask > 0).astype(np.float64)
+ mask_h, mask_w = binary_mask.shape
+
+ # Compute downsampled ROI coordinates (vectorized)
+ x1_ds = x1_arr // downsample_factor
+ x2_ds = (x2_arr - 1) // downsample_factor + 1
+ y1_ds = y1_arr // downsample_factor
+ y2_ds = (y2_arr - 1) // downsample_factor + 1
+
+ # Clip to mask boundaries
+ x1_ds = np.clip(x1_ds, 0, mask_w)
+ x2_ds = np.clip(x2_ds, 0, mask_w)
+ y1_ds = np.clip(y1_ds, 0, mask_h)
+ y2_ds = np.clip(y2_ds, 0, mask_h)
+
+ # Check if all ROI blocks have uniform downsampled size
+ block_w = x2_ds - x1_ds
+ block_h = y2_ds - y1_ds
+ uniform_w = int(block_w[0]) if len(block_w) > 0 else 0
+ uniform_h = int(block_h[0]) if len(block_h) > 0 else 0
+ all_uniform = bool(
+ np.all(block_w == uniform_w) and np.all(block_h == uniform_h)
+ )
+
+ if all_uniform and uniform_w > 0 and uniform_h > 0:
+ # Fast path: use a 2D integral image (summed-area table)
+ # to compute block sums in O(1) per tile
+ integral = np.zeros((mask_h + 1, mask_w + 1), dtype=np.float64)
+ integral[1:, 1:] = np.cumsum(np.cumsum(binary_mask, axis=0), axis=1)
+ # Block sum = integral[y2,x2] - integral[y1,x2] - integral[y2,x1] + integral[y1,x1]
+ block_sums = (
+ integral[y2_ds, x2_ds]
+ - integral[y1_ds, x2_ds]
+ - integral[y2_ds, x1_ds]
+ + integral[y1_ds, x1_ds]
+ )
+ total_pixels = uniform_w * uniform_h
+ tissue_coverages_arr = block_sums / total_pixels
+ else:
+ # Fallback: still use integral image but handle variable block sizes
+ integral = np.zeros((mask_h + 1, mask_w + 1), dtype=np.float64)
+ integral[1:, 1:] = np.cumsum(np.cumsum(binary_mask, axis=0), axis=1)
+ block_sums = (
+ integral[y2_ds, x2_ds]
+ - integral[y1_ds, x2_ds]
+ - integral[y2_ds, x1_ds]
+ + integral[y1_ds, x1_ds]
+ )
+ total_pixels_arr = (block_w * block_h).astype(np.float64)
+ # Avoid division by zero
+ total_pixels_arr[total_pixels_arr == 0] = 1.0
+ tissue_coverages_arr = block_sums / total_pixels_arr
+ else:
+ tissue_coverages_arr = np.ones(n_rois, dtype=np.float64)
+
+ logging.info(
+ f" [TIMING] Tissue coverage computation: {time.perf_counter() - t_tissue:.1f}s"
+ )
+
+ # ------------------------------------------------------------------
+ # Sample centre pixel for each ROI — vectorized fancy indexing
+ # ------------------------------------------------------------------
+ t_sample = time.perf_counter()
+
+ # Compute centre coordinates for all ROIs at once
+ cx = (x1_arr + x2_arr) // 2
+ cy = (y1_arr + y2_arr) // 2
+
+ if presampled is not None:
+ # Streaming path: the tile pass already read one pixel per ROI.
+ def _sampled(name: str) -> NDArray[np.float64] | None:
+ arr = presampled.get(name)
+ return None if arr is None else np.asarray(arr, dtype=np.float64)
+
+ dapi_focus_scores = _sampled("dapi_focus_map")
+ if dapi_focus_scores is None:
+ raise ValueError("presampled must contain 'dapi_focus_map'")
+ _zeros = np.zeros(n_rois, dtype=np.float64)
+ dapi_intensities = _sampled("dapi_mean_map")
+ if dapi_intensities is None:
+ dapi_intensities = _zeros
+ dapi_lap_vars = _sampled("dapi_lap_var_map")
+ if dapi_lap_vars is None:
+ dapi_lap_vars = _zeros
+ boundary_focus_scores = _sampled("boundary_focus_map") if has_boundary else None
+ boundary_intensities = _sampled("boundary_mean_map") if has_boundary else None
+ intrna_focus_scores = _sampled("intrna_focus_map") if has_intrna else None
+ intrna_intensities = _sampled("intrna_mean_map") if has_intrna else None
+ logging.info(
+ f" [TIMING] Centre-pixel sampling ({n_rois} tiles): from the tile pass"
+ )
+ else:
+ # DAPI channels (always present)
+ dapi_focus_scores = dapi_focus_map[cy, cx].astype(np.float64)
+ dapi_intensities = (
+ dapi_mean_map[cy, cx].astype(np.float64)
+ if dapi_mean_map is not None
+ else np.zeros(n_rois, dtype=np.float64)
+ )
+ dapi_lap_vars = (
+ dapi_lap_var_map[cy, cx].astype(np.float64)
+ if dapi_lap_var_map is not None
+ else np.zeros(n_rois, dtype=np.float64)
+ )
+
+ # Boundary channel
+ if has_boundary:
+ boundary_focus_scores = boundary_focus_map[cy, cx].astype(np.float64) # type: ignore[index]
+ boundary_intensities = boundary_mean_map[cy, cx].astype(np.float64) # type: ignore[index]
+ else:
+ boundary_focus_scores = None
+ boundary_intensities = None
+
+ # IntRNA channel
+ if has_intrna:
+ intrna_focus_scores = intrna_focus_map[cy, cx].astype(np.float64) # type: ignore[index]
+ intrna_intensities = intrna_mean_map[cy, cx].astype(np.float64) # type: ignore[index]
+ else:
+ intrna_focus_scores = None
+ intrna_intensities = None
+
+ logging.info(
+ f" [TIMING] Centre-pixel sampling ({n_rois} tiles): "
+ f"{time.perf_counter() - t_sample:.1f}s"
+ )
+
+ # ------------------------------------------------------------------
+ # Normalize focus scores with RobustScaler
+ # ------------------------------------------------------------------
+ t_scaler = time.perf_counter()
+ scaler_dapi = RobustScaler()
+ dapi_focus_scores_norm = scaler_dapi.fit_transform(
+ dapi_focus_scores.reshape(-1, 1)
+ ).flatten()
+
+ if has_boundary:
+ scaler_boundary = RobustScaler()
+ boundary_focus_scores_norm: NDArray[np.float64] | None = (
+ scaler_boundary.fit_transform(
+ boundary_focus_scores.reshape(-1, 1) # type: ignore[union-attr]
+ ).flatten()
+ )
+ else:
+ boundary_focus_scores_norm = None
+
+ if has_intrna:
+ scaler_intrna = RobustScaler()
+ intrna_focus_scores_norm: NDArray[np.float64] | None = (
+ scaler_intrna.fit_transform(
+ intrna_focus_scores.reshape(-1, 1) # type: ignore[union-attr]
+ ).flatten()
+ )
+ else:
+ intrna_focus_scores_norm = None
+
+ logging.info(
+ f" [TIMING] RobustScaler normalization: {time.perf_counter() - t_scaler:.1f}s"
+ )
+
+ # ------------------------------------------------------------------
+ # Mark boundary tiles (centre within roi_size//2 of image edge)
+ # ------------------------------------------------------------------
+ half_win = roi_size // 2
+ is_boundary_roi = (
+ (cx < half_win)
+ | (cy < half_win)
+ | (cx >= width - half_win)
+ | (cy >= height - half_win)
+ )
+
+ # ------------------------------------------------------------------
+ # Build output DataFrame (column order matches legacy code)
+ # ------------------------------------------------------------------
+ df_data: dict[str, Any] = {
+ "roi_id": np.arange(n_rois),
+ "x1": roi_coords_arr[:, 0],
+ "x2": roi_coords_arr[:, 1],
+ "y1": roi_coords_arr[:, 2],
+ "y2": roi_coords_arr[:, 3],
+ # DAPI focus scores (duplicated for backward compatibility)
+ "focus_score": dapi_focus_scores,
+ "focus_score_norm": dapi_focus_scores_norm,
+ "dapi_focus_score": dapi_focus_scores,
+ "dapi_focus_score_norm": dapi_focus_scores_norm,
+ # Laplacian variance (DAPI)
+ "dapi_lap_var": dapi_lap_vars,
+ # Intensities
+ "dapi_intensity": dapi_intensities,
+ "raw_intensity": dapi_intensities.copy(), # backward compatibility
+ # Boundary channel
+ "boundary_focus_score": (boundary_focus_scores if has_boundary else np.nan),
+ "boundary_focus_score_norm": (
+ boundary_focus_scores_norm if has_boundary else np.nan
+ ),
+ "boundary_intensity": (boundary_intensities if has_boundary else np.nan),
+ # IntRNA channel
+ "intrna_focus_score": (intrna_focus_scores if has_intrna else np.nan),
+ "intrna_focus_score_norm": (intrna_focus_scores_norm if has_intrna else np.nan),
+ "intrna_intensity": (intrna_intensities if has_intrna else np.nan),
+ # Tissue coverage
+ "tissue_coverage": tissue_coverages_arr,
+ "overlaps_tissue": tissue_coverages_arr > 0.0,
+ # Boundary flag: True if ROI centre is near image edge
+ "is_boundary_roi": is_boundary_roi,
+ }
+
+ return pd.DataFrame(df_data)
+
+
+def calculate_roi_focusscore_without_laplace(
+ xoa_morphology_files,
+ roi_size=35,
+ stride=None,
+ tissue_filter=True,
+ min_tissue_coverage=0.0,
+ downsample_factor=8,
+):
+ """
+ Calculate cell-independent tile-based focus scores using a regular grid.
+
+ Creates a regular lattice/grid of tiles across the entire slide and calculates
+ focus scores for each tile independently of cell locations. Calculates focus scores
+ for all available channels (DAPI, Boundary, IntRNA).
+
+ Parameters:
+ -----------
+ xoa_morphology_files : list
+ List of paths to morphology image files
+ roi_size : int, optional
+ Size of square tile in pixels (default: 35)
+ stride : int, optional
+ Grid spacing in pixels. If None, uses non-overlapping grid (stride = roi_size)
+ tissue_filter : bool, optional
+ Enable tissue region filtering (default: True)
+ min_tissue_coverage : float, optional
+ Minimum fraction of tile that must be tissue (default: 0.0, i.e., any tissue overlap)
+ downsample_factor : int, optional
+ Downsampling factor for tissue mask (default: 8, for level 3)
+
+ Returns:
+ --------
+ pandas DataFrame
+ DataFrame with columns:
+ - roi_id: Unique identifier
+ - x1, x2, y1, y2: Tile boundaries (full resolution)
+ - focus_score: Raw DAPI focus score (std² / mean) [backward compatibility]
+ - focus_score_norm: Normalized DAPI focus score [backward compatibility]
+ - dapi_focus_score, dapi_focus_score_norm: DAPI focus scores
+ - boundary_focus_score, boundary_focus_score_norm: Boundary focus scores (if available)
+ - intrna_focus_score, intrna_focus_score_norm: IntRNA focus scores (if available)
+ - dapi_intensity: Mean DAPI intensity per tile
+ - boundary_intensity: Mean Boundary intensity per tile (if available)
+ - intrna_intensity: Mean IntRNA intensity per tile (if available)
+ - raw_intensity: Mean DAPI intensity per tile [backward compatibility]
+ - tissue_coverage: Fraction of tile that is tissue (0.0-1.0)
+ """
+
+ # Set stride (non-overlapping if not specified)
+ if stride is None:
+ stride = roi_size
+
+ # Load channels from either multi-channel stack or split single-channel files.
+ dapi_image, boundary_image, intrna_image = _load_morphology_channels(
+ xoa_morphology_files, level=0
+ )
+ n_channels = 1 + int(boundary_image is not None) + int(intrna_image is not None)
+
+ height, width = dapi_image.shape
+
+ # Check which channels are available
+ has_boundary = n_channels > 1
+ has_intrna = n_channels > 2
+
+ if has_boundary:
+ logging.info(
+ f" Found {n_channels} channels: DAPI, Boundary"
+ + (", IntRNA" if has_intrna else "")
+ )
+ else:
+ logging.info(f" Found {n_channels} channel(s): DAPI only")
+
+ # Generate tissue mask if filtering enabled
+ tissue_mask = None
+ if tissue_filter:
+ # Load downsampled DAPI for tissue mask generation
+ small0_ds, _, _ = _load_morphology_channels(xoa_morphology_files, level=3)
+ # Generate tissue mask (simplified version - just need whole_sample)
+ # Tissue mask via the shared helper (hysteresis + degeneracy guard + `> 0`) —
+ # single source of truth, see compute_tissue_mask. Fixed 2026-06-23 (was
+ # the percentile-60 + `test_mask > 1` defect). This path builds
+ # tissue_coverage, so the fix here is what corrects the §5.5 mask gate
+ # and the GMM tissue selection on sparse/dim slides.
+ tissue_mask = compute_tissue_mask(small0_ds)[0]
+ del small0_ds
+
+ # Create grid of ROI coordinates
+ n_x = (width // stride) + (1 if width % stride > 0 else 0)
+ n_y = (height // stride) + (1 if height % stride > 0 else 0)
+
+ roi_coords = []
+ for y_idx in range(n_y):
+ for x_idx in range(n_x):
+ x1 = x_idx * stride
+ x2 = min(x1 + roi_size, width)
+ y1 = y_idx * stride
+ y2 = min(y1 + roi_size, height)
+
+ # Skip if ROI is too small (at edges)
+ if (x2 - x1) < roi_size // 2 or (y2 - y1) < roi_size // 2:
+ continue
+
+ roi_coords.append((x1, x2, y1, y2))
+
+ # Calculate tissue coverage for ALL tiles (no filtering)
+ # This ensures we always have tiles to process, even if tissue detection fails
+ tissue_coverages = []
+ if tissue_filter and tissue_mask is not None:
+ for x1, x2, y1, y2 in roi_coords:
+ # Scale coordinates to downsampled space
+ x1_ds = x1 // downsample_factor
+ x2_ds = (x2 - 1) // downsample_factor + 1
+ y1_ds = y1 // downsample_factor
+ y2_ds = (y2 - 1) // downsample_factor + 1
+
+ # Clip to mask boundaries
+ x1_ds = max(0, x1_ds)
+ x2_ds = min(tissue_mask.shape[1], x2_ds)
+ y1_ds = max(0, y1_ds)
+ y2_ds = min(tissue_mask.shape[0], y2_ds)
+
+ # Calculate tissue coverage
+ roi_mask = tissue_mask[y1_ds:y2_ds, x1_ds:x2_ds]
+ tissue_pixels = np.sum(roi_mask > 0)
+ total_pixels = roi_mask.size
+ coverage = tissue_pixels / total_pixels if total_pixels > 0 else 0.0
+ tissue_coverages.append(coverage)
+ else:
+ # If tissue filtering is disabled, assume all tiles have full tissue coverage
+ tissue_coverages = [1.0] * len(roi_coords)
+
+ # Calculate focus scores for all ROIs and all available channels
+ n_rois = len(roi_coords)
+
+ # Basic safety check (should never trigger now, but defensive programming)
+ if n_rois == 0:
+ raise ValueError(
+ f"ERROR: No tiles generated from grid!\n"
+ f" - Image dimensions: {height}x{width} (full resolution)\n"
+ f" - Tile size: {roi_size}px, stride: {stride}px\n"
+ f" - This should not happen. Please check image dimensions and tile parameters.\n"
+ )
+
+ # Initialize arrays for all channels
+ dapi_focus_scores = np.empty(n_rois, dtype=np.float64)
+ dapi_intensities = np.empty(n_rois, dtype=np.float64)
+
+ boundary_focus_scores = np.empty(n_rois, dtype=np.float64) if has_boundary else None
+ boundary_intensities = np.empty(n_rois, dtype=np.float64) if has_boundary else None
+
+ intrna_focus_scores = np.empty(n_rois, dtype=np.float64) if has_intrna else None
+ intrna_intensities = np.empty(n_rois, dtype=np.float64) if has_intrna else None
+
+ # Calculate focus scores for each ROI
+ for i, (x1, x2, y1, y2) in enumerate(roi_coords):
+ # DAPI channel
+ roi_dapi = dapi_image[y1:y2, x1:x2]
+ mean_dapi = np.mean(roi_dapi)
+ std_dapi = np.std(roi_dapi)
+ dapi_focus_scores[i] = (
+ (std_dapi * std_dapi) / mean_dapi if mean_dapi > 0 else 0.0
+ )
+ dapi_intensities[i] = mean_dapi
+
+ # Boundary channel (if available)
+ if has_boundary:
+ roi_boundary = boundary_image[y1:y2, x1:x2]
+ mean_boundary = np.mean(roi_boundary)
+ std_boundary = np.std(roi_boundary)
+ boundary_focus_scores[i] = (
+ (std_boundary * std_boundary) / mean_boundary
+ if mean_boundary > 0
+ else 0.0
+ )
+ boundary_intensities[i] = mean_boundary
+
+ # IntRNA channel (if available)
+ if has_intrna:
+ roi_intrna = intrna_image[y1:y2, x1:x2]
+ mean_intrna = np.mean(roi_intrna)
+ std_intrna = np.std(roi_intrna)
+ intrna_focus_scores[i] = (
+ (std_intrna * std_intrna) / mean_intrna if mean_intrna > 0 else 0.0
+ )
+ intrna_intensities[i] = mean_intrna
+
+ # Normalize focus scores for each channel separately
+ # Safety check: Ensure we have data before normalizing
+ if len(dapi_focus_scores) == 0:
+ raise ValueError(
+ f"ERROR: Cannot normalize focus scores - empty array detected!\n"
+ f" - Number of tiles: {n_rois}\n"
+ f" - This should have been caught earlier. Please report this issue.\n"
+ )
+
+ scaler_dapi = RobustScaler()
+ dapi_focus_scores_norm = scaler_dapi.fit_transform(
+ dapi_focus_scores.reshape(-1, 1)
+ ).flatten()
+
+ if has_boundary:
+ if len(boundary_focus_scores) == 0:
+ raise ValueError(
+ "ERROR: Cannot normalize boundary focus scores - empty array detected!"
+ )
+ scaler_boundary = RobustScaler()
+ boundary_focus_scores_norm = scaler_boundary.fit_transform(
+ boundary_focus_scores.reshape(-1, 1)
+ ).flatten()
+ else:
+ boundary_focus_scores_norm = None
+
+ if has_intrna:
+ if len(intrna_focus_scores) == 0:
+ raise ValueError(
+ "ERROR: Cannot normalize IntRNA focus scores - empty array detected!"
+ )
+ scaler_intrna = RobustScaler()
+ intrna_focus_scores_norm = scaler_intrna.fit_transform(
+ intrna_focus_scores.reshape(-1, 1)
+ ).flatten()
+ else:
+ intrna_focus_scores_norm = None
+
+ # Create output DataFrame
+ # roi_id: Simple integer IDs (0, 1, 2, ...) for easy indexing
+ df_data = {
+ "roi_id": range(n_rois),
+ "x1": [coords[0] for coords in roi_coords],
+ "x2": [coords[1] for coords in roi_coords],
+ "y1": [coords[2] for coords in roi_coords],
+ "y2": [coords[3] for coords in roi_coords],
+ # DAPI focus scores (also kept as focus_score/focus_score_norm for backward compatibility)
+ "focus_score": dapi_focus_scores,
+ "focus_score_norm": dapi_focus_scores_norm,
+ "dapi_focus_score": dapi_focus_scores,
+ "dapi_focus_score_norm": dapi_focus_scores_norm,
+ "dapi_intensity": dapi_intensities,
+ "raw_intensity": dapi_intensities, # Backward compatibility
+ "tissue_coverage": tissue_coverages if tissue_filter else [1.0] * n_rois,
+ "overlaps_tissue": [
+ coverage > 0.0
+ for coverage in (tissue_coverages if tissue_filter else [1.0] * n_rois)
+ ],
+ }
+
+ # Add Boundary channel data if available
+ if has_boundary:
+ df_data["boundary_focus_score"] = boundary_focus_scores
+ df_data["boundary_focus_score_norm"] = boundary_focus_scores_norm
+ df_data["boundary_intensity"] = boundary_intensities
+ else:
+ df_data["boundary_focus_score"] = np.nan
+ df_data["boundary_focus_score_norm"] = np.nan
+ df_data["boundary_intensity"] = np.nan
+
+ # Add IntRNA channel data if available
+ if has_intrna:
+ df_data["intrna_focus_score"] = intrna_focus_scores
+ df_data["intrna_focus_score_norm"] = intrna_focus_scores_norm
+ df_data["intrna_intensity"] = intrna_intensities
+ else:
+ df_data["intrna_focus_score"] = np.nan
+ df_data["intrna_focus_score_norm"] = np.nan
+ df_data["intrna_intensity"] = np.nan
+
+ df_grid_roi = pd.DataFrame(df_data)
+
+ # Clean up
+ del dapi_image
+ if has_boundary:
+ del boundary_image
+ if has_intrna:
+ del intrna_image
+
+ return df_grid_roi
+
+
+@dataclass
+class StreamedTileResults:
+ """The small reductions that replace the full-resolution pixel planes.
+
+ Assembling the planes cost ~154 GB of scratch on a 5.5 gigapixel sample, and
+ `mmap` over a FUSE/S3 work directory is pathological (see
+ `docs/failures/2026-07-24_imageqc-mmap-over-fusion.md`). No downstream consumer
+ needs a whole plane, so in streaming mode each tile is folded into these arrays
+ and dropped. Everything here is O(n_ROIs), O(n_cells) or O(pixels / factor^2) —
+ tens of MB, not tens of GB.
+
+ Attributes:
+ presampled: One centre pixel per ROI, keyed by map name
+ (``dapi_focus_map``, ``boundary_mean_map``, ...). Feeds
+ :func:`downsample_maps_to_roi_dataframe`'s ``presampled`` argument.
+ roi_snr_db: Per-ROI Otsu SNR in dB, replacing
+ ``snr_metrics.compute_image_snr_from_pixel_maps``' loop over the DAPI
+ mean plane. Aligned with the ROI grid.
+ focus_heatmap: DAPI focus map down-sampled by ``heatmap_factor``, replacing
+ ``downscale_local_mean`` on the assembled plane in Figure 5.
+ nuclear_counts / nuclear_sums: Per-nucleus pixel counts and value sums over
+ ``masks/0``, including the ``centroid_*_sum`` keys. Replaces the
+ ``_labeled_sums_chunked`` pass in :func:`calculate_ccfs_from_focus_maps`.
+ cell_counts / cell_sums: The same over ``masks/1``, giving cell areas and
+ per-cell boundary/IntRNA means.
+ """
+
+ presampled: dict[str, NDArray[np.float64]]
+ roi_snr_db: NDArray[np.float64] | None = None
+ focus_heatmap: NDArray[np.float64] | None = None
+ heatmap_factor: int = 8
+ nuclear_counts: NDArray[np.int64] | None = None
+ nuclear_sums: dict[str, NDArray[np.float64]] | None = None
+ cell_counts: NDArray[np.int64] | None = None
+ cell_sums: dict[str, NDArray[np.float64]] | None = None
+
+ @property
+ def has_per_cell(self) -> bool:
+ return self.nuclear_counts is not None
+
+
+def _stream_channels(
+ lazy_channels,
+ image_shape: tuple[int, int],
+ roi_size: int,
+ stride: int | None,
+ gpu_ids: list[int],
+ *,
+ lap_sigma: float,
+ gpu_mem_bytes: int | None,
+ cell_masks_path: Path | None = None,
+ heatmap_factor: int = 8,
+) -> StreamedTileResults:
+ """Run the tiled compute per channel, folding each tile into small reductions.
+
+ The plane-based path assembles a full-resolution map per channel and then makes
+ one pass over it per consumer. Here the consumers are attached to the tile
+ dispatch instead, so a tile is reduced and dropped as soon as it is computed
+ and no plane is ever allocated.
+
+ Which consumers attach depends on the channel, because the downstream metrics
+ do not use every channel the same way:
+
+ * every channel needs one centre pixel per ROI, for the ROI DataFrame;
+ * only DAPI feeds the Figure 5 heatmap, the Otsu SNR and the per-nucleus CCFS;
+ * cell areas come from any single channel's pass over ``masks/1``, so they are
+ taken from DAPI, which is always present, while boundary and IntRNA each
+ contribute their own per-cell mean.
+
+ The label planes stay lazy: ``LabeledSumAccumulator`` slices row blocks out of
+ zarr per tile, so neither ``masks/0`` nor ``masks/1`` (22 GB each on a 5.5
+ gigapixel sample) is materialised.
+ """
+ x1, x2, y1, y2, cx, cy = compute_roi_grid(
+ image_shape[0], image_shape[1], roi_size, stride
+ )
+ logging.info(f" Streaming mode: {len(cx):,} ROIs, no pixel planes materialised")
+
+ nuclear_plane = cell_plane = None
+ if cell_masks_path is not None:
+ masks = open_zarr(cell_masks_path).get("masks")
+ nuclear_plane = LazyLabelPlane(masks.get("0"))
+ cell_plane = LazyLabelPlane(masks.get("1"))
+ logging.info(" Per-cell CCFS will be reduced during the tile pass")
+
+ presampled: dict[str, NDArray[np.float64]] = {}
+ nuclear_counts: NDArray[np.int64] | None = None
+ nuclear_sums: dict[str, NDArray[np.float64]] | None = None
+ cell_counts: NDArray[np.int64] | None = None
+ cell_sums: dict[str, NDArray[np.float64]] = {}
+ roi_snr_db: NDArray[np.float64] | None = None
+ focus_heatmap: NDArray[np.float64] | None = None
+
+ for name, channel_data in (
+ ("dapi", lazy_channels[0]),
+ ("boundary", lazy_channels[1]),
+ ("intrna", lazy_channels[2]),
+ ):
+ if channel_data is None:
+ continue
+ is_dapi = name == "dapi"
+
+ # Built by a factory, because the fold runs concurrently on one set of
+ # accumulators per slot and the parent merges them. Order must be identical
+ # across sets -- merging pairs them positionally.
+ def _make_consumers() -> tuple[list[Any], dict[str, Any]]:
+ made: dict[str, Any] = {
+ "sampler": CentrePixelSampler(
+ cy,
+ cx,
+ ["focus_map", "mean_map"] + (["lap_var_map"] if is_dapi else []),
+ )
+ }
+ ordered: list[Any] = [made["sampler"]]
+ if is_dapi:
+ made["heat"] = BlockMeanAccumulator(
+ image_shape, factor=heatmap_factor, map_key="focus_map"
+ )
+ made["snr"] = RoiOtsuSnrAccumulator(
+ y1, y2, x1, x2, image_shape, map_key="mean_map"
+ )
+ ordered += [made["heat"], made["snr"]]
+ if nuclear_plane is not None:
+ made["nuc"] = LabeledSumAccumulator(
+ nuclear_plane,
+ ["focus_map", "mean_map"],
+ include_coords=True,
+ # CCFS reads only labels > 0 from this one.
+ skip_background=True,
+ )
+ made["cells"] = LabeledSumAccumulator(cell_plane, [])
+ ordered += [made["nuc"], made["cells"]]
+ elif cell_plane is not None:
+ made["cells"] = LabeledSumAccumulator(cell_plane, ["mean_map"])
+ ordered.append(made["cells"])
+ return ordered, made
+
+ consumers, named = _make_consumers()
+ sampler = named["sampler"]
+ heat = named.get("heat")
+ snr = named.get("snr")
+ nuc = named.get("nuc")
+ cells = named.get("cells")
+
+ t_ch = time.perf_counter()
+ _compute_channel_maps_tiled(
+ channel_data,
+ image_shape,
+ roi_size,
+ gpu_ids,
+ include_laplacian=is_dapi,
+ lap_sigma=lap_sigma,
+ gpu_mem_bytes=gpu_mem_bytes,
+ plane_prefix=name,
+ consumers=consumers,
+ )
+ logging.info(
+ f" [TIMING] {name} channel streamed: {time.perf_counter() - t_ch:.1f}s"
+ )
+ _log_mem(f"{name} channel streamed")
+
+ # CentrePixelSampler keys are per-channel ("focus_map"); the ROI DataFrame
+ # wants them qualified ("dapi_focus_map").
+ for key, values in sampler.finalize().items():
+ presampled[f"{name}_{key}"] = values
+
+ if is_dapi:
+ focus_heatmap = heat.finalize()
+ roi_snr_db = snr.finalize()
+ if nuc is not None:
+ nuclear_counts, nuclear_sums = nuc.finalize()
+ cell_counts, _ = cells.finalize()
+ elif cells is not None:
+ cell_sums[name] = cells.finalize()[1]["mean_map"]
+
+ # No index alignment needed: every channel's pass covers the whole image, so each
+ # per-label array grows to the same largest label. Asserted in
+ # test_cell_arrays_share_one_label_index.
+
+ return StreamedTileResults(
+ presampled=presampled,
+ roi_snr_db=roi_snr_db,
+ focus_heatmap=focus_heatmap,
+ heatmap_factor=heatmap_factor,
+ nuclear_counts=nuclear_counts,
+ nuclear_sums=nuclear_sums,
+ cell_counts=cell_counts,
+ cell_sums=cell_sums or None,
+ )
+
+
+def calculate_roi_focusscore(
+ xoa_morphology_files,
+ roi_size=35,
+ stride=None,
+ tissue_filter=True,
+ min_tissue_coverage=0.0,
+ downsample_factor=8,
+ use_gpu=False,
+ gpu_ids=None,
+ return_pixel_maps=False,
+ stream_tiles=False,
+ cell_masks_path=None,
+ heatmap_factor=8,
+ lap_sigma: float = 1.0,
+ small0_ds=None,
+ tissue_mask=None,
+):
+ """
+ Calculate cell-independent tile-based focus scores using a regular grid.
+
+ Uses efficient whole-image convolution operations (uniform_filter + laplace)
+ to produce per-pixel focus maps, then samples the centre pixel of each tile
+ for backward-compatible per-tile scalars. GPU acceleration is supported via
+ CuPy when ``use_gpu=True``. Multi-GPU parallelism is available via
+ ``gpu_ids``.
+
+ Parameters:
+ -----------
+ xoa_morphology_files : list
+ List of paths to morphology image files
+ roi_size : int, optional
+ Size of square tile in pixels (default: 35)
+ stride : int, optional
+ Grid spacing in pixels. If None, uses non-overlapping grid (stride = roi_size)
+ tissue_filter : bool, optional
+ Enable tissue region filtering (default: True)
+ min_tissue_coverage : float, optional
+ Minimum fraction of tile that must be tissue (default: 0.0)
+ downsample_factor : int, optional
+ Downsampling factor for tissue mask (default: 8, for level 3)
+ use_gpu : bool, optional
+ Use CuPy GPU backend for convolutions (default: False)
+ gpu_ids : list[int] or None, optional
+ List of CUDA device IDs for multi-GPU parallelism. Channels are
+ distributed across GPUs. If None and use_gpu=True, GPUs are
+ auto-detected (falling back to device 0).
+ return_pixel_maps : bool, optional
+ If True, also return the pixel-level focus map dict (default: False)
+ small0_ds : numpy.ndarray or None, optional
+ Pre-loaded level-3 DAPI plane. When provided (with ``tissue_mask``), the
+ redundant level-3 decode is skipped. ``main`` already holds this array.
+ tissue_mask : numpy.ndarray or None, optional
+ Pre-computed labelled tissue mask (``compute_tissue_mask(small0_ds)[0]``).
+ When provided, the mask recompute is skipped -- ``main`` already produced
+ the identical array via ``generate_tissue_mask`` (same ``small0``, same
+ ``compute_tissue_mask`` default ``min_size_hole=1500``), so the result is
+ bit-identical. Ignored when ``tissue_filter`` is False.
+
+ Returns:
+ --------
+ pandas DataFrame, or ``(DataFrame, focus_maps, streamed)`` when
+ *return_pixel_maps* is set. Exactly one of the last two is not ``None``:
+ ``focus_maps`` on the plane-based path, a :class:`StreamedTileResults` when
+ *stream_tiles* folded each tile into small reductions instead.
+ DataFrame with columns:
+ - roi_id: Unique identifier
+ - x1, x2, y1, y2: Tile boundaries (full resolution)
+ - focus_score: Raw DAPI focus score (std^2 / mean) [backward compat]
+ - focus_score_norm: Normalized DAPI focus score [backward compat]
+ - dapi_focus_score, dapi_focus_score_norm: DAPI focus scores
+ - dapi_lap_var: Laplacian variance (DAPI) per tile
+ - boundary_focus_score, boundary_focus_score_norm: Boundary (if avail)
+ - intrna_focus_score, intrna_focus_score_norm: IntRNA (if available)
+ - dapi_intensity, boundary_intensity, intrna_intensity: Mean per tile
+ - raw_intensity: Mean DAPI intensity per tile [backward compat]
+ - tissue_coverage: Fraction of tile that is tissue (0.0-1.0)
+ - overlaps_tissue: Boolean, tile overlaps any tissue (>0 coverage)
+ """
+ if stride is None:
+ stride = roi_size
+
+ # Channel detection + full-resolution loading happen per-path below: the
+ # GPU path opens lazy TIFF wrappers (no full-res pixels in host RAM) and
+ # computes in memory-bounded tiles; the CPU path eager-loads numpy arrays.
+ # This avoids materialising the whole image (tens of GB on large samples)
+ # before the GPU branch.
+
+ # Generate tissue mask if filtering enabled. `main` already loads the level-3
+ # DAPI plane and computes this exact mask (compute_tissue_mask(small0)[0], the
+ # `whole_sample` output of generate_tissue_mask) — when it threads `small0_ds`
+ # and `tissue_mask` in we skip a redundant level-3 decode + mask recompute
+ # (~20-35s on a 5.5 GP sample). Bit-identical: same small0, same
+ # compute_tissue_mask default min_size_hole=1500.
+ if tissue_filter and tissue_mask is None:
+ if small0_ds is None:
+ small0_ds, _, _ = _load_morphology_channels(xoa_morphology_files, level=3)
+ # Tissue mask via the shared helper (hysteresis + degeneracy guard + `> 0`) —
+ # single source of truth, see compute_tissue_mask. Fixed 2026-06-23 (was
+ # the percentile-60 + `test_mask > 1` defect). This path builds
+ # tissue_coverage, so the fix here is what corrects the §5.5 mask gate
+ # and the GMM tissue selection on sparse/dim slides.
+ tissue_mask = compute_tissue_mask(small0_ds)[0]
+ del small0_ds
+ elif not tissue_filter:
+ tissue_mask = None
+
+ # Resolve GPU configuration
+ _use_gpu = use_gpu
+ if gpu_ids is not None and len(gpu_ids) > 0:
+ _use_gpu = True
+
+ # ------------------------------------------------------------------
+ # GPU path: memory-bounded tiled compute with lazy tile reads. Each
+ # channel is read from disk and computed in tiles so neither a full-res
+ # channel (tens of GB) nor its float64 intermediates land on the GPU or
+ # in host RAM at once. Ports the tiled+lazy path from commit 5039b6a.
+ # ------------------------------------------------------------------
+ if _use_gpu:
+ # Auto-detect GPUs if the caller enabled use_gpu without naming devices
+ # (mem_info + tiled dispatch below index gpu_ids[0]).
+ if not gpu_ids:
+ gpu_ids = detect_gpu_ids() or [0]
+
+ # Lazy TIFF page wrappers — no full-res pixel data loaded into RAM.
+ lazy_channels, img_shape = _open_morphology_lazy(xoa_morphology_files, level=0)
+ has_boundary = lazy_channels[1] is not None
+ has_intrna = lazy_channels[2] is not None
+ n_channels = 1 + int(has_boundary) + int(has_intrna)
+
+ gpu_label = f"gpu_ids={gpu_ids}" if gpu_ids else f"gpu={_use_gpu}"
+ logging.info(
+ f" Found {n_channels} channel(s); computing per-pixel focus maps "
+ f"(window={roi_size}, {gpu_label}, tiled)..."
+ )
+ logging.info(
+ f" Image shape: {img_shape[0]}x{img_shape[1]}, {n_channels} channel(s)"
+ )
+
+ t_gpu = time.perf_counter()
+ gpu_mem = cp.cuda.Device(gpu_ids[0]).mem_info[1] # total VRAM per device
+ logging.info(f" GPU VRAM: {gpu_mem / 1024**3:.1f} GB per device")
+
+ if stream_tiles:
+ streamed = _stream_channels(
+ lazy_channels,
+ img_shape,
+ roi_size,
+ stride,
+ gpu_ids,
+ lap_sigma=lap_sigma,
+ gpu_mem_bytes=gpu_mem,
+ cell_masks_path=cell_masks_path,
+ heatmap_factor=heatmap_factor,
+ )
+ logging.info(
+ f" [TIMING] GPU compute (streamed, {n_channels} ch): "
+ f"{time.perf_counter() - t_gpu:.1f}s"
+ )
+ df_grid_roi = downsample_maps_to_roi_dataframe(
+ {},
+ tissue_mask=tissue_mask,
+ roi_size=roi_size,
+ stride=stride,
+ downsample_factor=downsample_factor,
+ min_tissue_coverage=min_tissue_coverage,
+ image_shape=img_shape,
+ presampled=streamed.presampled,
+ )
+ if return_pixel_maps:
+ return df_grid_roi, None, streamed
+ return df_grid_roi
+
+ dapi_result = _compute_channel_maps_tiled(
+ lazy_channels[0],
+ img_shape,
+ roi_size,
+ gpu_ids,
+ include_laplacian=True,
+ lap_sigma=lap_sigma,
+ gpu_mem_bytes=gpu_mem,
+ plane_prefix="dapi",
+ )
+ boundary_result = None
+ if has_boundary and lazy_channels[1] is not None:
+ boundary_result = _compute_channel_maps_tiled(
+ lazy_channels[1],
+ img_shape,
+ roi_size,
+ gpu_ids,
+ include_laplacian=False,
+ lap_sigma=lap_sigma,
+ gpu_mem_bytes=gpu_mem,
+ plane_prefix="boundary",
+ )
+ intrna_result = None
+ if has_intrna and lazy_channels[2] is not None:
+ intrna_result = _compute_channel_maps_tiled(
+ lazy_channels[2],
+ img_shape,
+ roi_size,
+ gpu_ids,
+ include_laplacian=False,
+ lap_sigma=lap_sigma,
+ gpu_mem_bytes=gpu_mem,
+ plane_prefix="intrna",
+ )
+ logging.info(
+ f" [TIMING] GPU compute (tiled, {n_channels} ch): "
+ f"{time.perf_counter() - t_gpu:.1f}s"
+ )
+
+ focus_maps = {
+ "dapi_focus_map": dapi_result["focus_map"],
+ "dapi_mean_map": dapi_result["mean_map"],
+ "dapi_lap_var_map": dapi_result.get("lap_var_map"),
+ "boundary_focus_map": (
+ boundary_result["focus_map"] if boundary_result else None
+ ),
+ "boundary_mean_map": (
+ boundary_result["mean_map"] if boundary_result else None
+ ),
+ "intrna_focus_map": intrna_result["focus_map"] if intrna_result else None,
+ "intrna_mean_map": intrna_result["mean_map"] if intrna_result else None,
+ }
+ logging.info(" Per-pixel focus maps computed (tiled).")
+
+ df_grid_roi = downsample_maps_to_roi_dataframe(
+ focus_maps,
+ tissue_mask=tissue_mask,
+ roi_size=roi_size,
+ stride=stride,
+ downsample_factor=downsample_factor,
+ min_tissue_coverage=min_tissue_coverage,
+ )
+
+ if return_pixel_maps:
+ return df_grid_roi, focus_maps, None
+ return df_grid_roi
+
+ # ------------------------------------------------------------------
+ # CPU path: incremental per-channel compute→downsample→free to limit
+ # peak memory. At most ~3 maps + convolution intermediates at once.
+ # ------------------------------------------------------------------
+ # Eager-load full-resolution channels (CPU path needs numpy arrays).
+ dapi_image, boundary_image, intrna_image = _load_morphology_channels(
+ xoa_morphology_files, level=0
+ )
+ has_boundary = boundary_image is not None
+ has_intrna = intrna_image is not None
+
+ logging.info(f" Computing per-pixel focus maps (window={roi_size}, gpu=False)...")
+
+ # Build ROI grid once (cheap — just coordinate arrays)
+ image_shape = dapi_image.shape[:2]
+ grid = _build_roi_grid(
+ image_shape, roi_size, stride, tissue_mask, downsample_factor
+ )
+ cx, cy = grid["cx"], grid["cy"]
+ n_rois = grid["n_rois"]
+ logging.info(f" Tile grid: {n_rois:,} tiles")
+
+ # -- DAPI (always present) --
+ dapi_focus_map, dapi_mean_map = compute_ccfs_map(
+ dapi_image, window_size=roi_size, use_gpu=False, gpu_id=0
+ )
+ dapi_lap_var_map = compute_laplacian_variance_map(
+ dapi_image,
+ window_size=roi_size,
+ use_gpu=False,
+ gpu_id=0,
+ lap_sigma=lap_sigma,
+ )
+ del dapi_image
+ # Sample per-tile scalars
+ dapi_focus_scores = dapi_focus_map[cy, cx].astype(np.float64)
+ dapi_intensities = dapi_mean_map[cy, cx].astype(np.float64)
+ dapi_lap_vars = dapi_lap_var_map[cy, cx].astype(np.float64)
+ del dapi_lap_var_map # not needed downstream
+ logging.info(" DAPI maps computed and sampled.")
+
+ # -- Boundary --
+ boundary_focus_scores = None
+ boundary_intensities = None
+ boundary_mean_map = None
+ if has_boundary:
+ b_focus, boundary_mean_map = compute_ccfs_map(
+ boundary_image, window_size=roi_size, use_gpu=False, gpu_id=0
+ )
+ del boundary_image
+ boundary_focus_scores = b_focus[cy, cx].astype(np.float64)
+ boundary_intensities = boundary_mean_map[cy, cx].astype(np.float64)
+ del b_focus # boundary_mean_map kept for CCFS
+ logging.info(" Boundary maps computed and sampled.")
+ else:
+ del boundary_image
+
+ # -- IntRNA --
+ intrna_focus_scores = None
+ intrna_intensities = None
+ intrna_mean_map = None
+ if has_intrna:
+ i_focus, intrna_mean_map = compute_ccfs_map(
+ intrna_image, window_size=roi_size, use_gpu=False, gpu_id=0
+ )
+ del intrna_image
+ intrna_focus_scores = i_focus[cy, cx].astype(np.float64)
+ intrna_intensities = intrna_mean_map[cy, cx].astype(np.float64)
+ del i_focus # intrna_mean_map kept for CCFS
+ logging.info(" IntRNA maps computed and sampled.")
+ else:
+ del intrna_image
+
+ logging.info(" Per-pixel focus maps computed (incremental CPU path).")
+
+ # -- Normalize focus scores with RobustScaler --
+ scaler_dapi = RobustScaler()
+ dapi_focus_scores_norm = scaler_dapi.fit_transform(
+ dapi_focus_scores.reshape(-1, 1)
+ ).flatten()
+
+ boundary_focus_scores_norm = None
+ if boundary_focus_scores is not None:
+ scaler_b = RobustScaler()
+ boundary_focus_scores_norm = scaler_b.fit_transform(
+ boundary_focus_scores.reshape(-1, 1)
+ ).flatten()
+
+ intrna_focus_scores_norm = None
+ if intrna_focus_scores is not None:
+ scaler_i = RobustScaler()
+ intrna_focus_scores_norm = scaler_i.fit_transform(
+ intrna_focus_scores.reshape(-1, 1)
+ ).flatten()
+
+ # -- Assemble DataFrame --
+ rc = grid["roi_coords_arr"]
+ df_data: dict[str, Any] = {
+ "roi_id": np.arange(n_rois),
+ "x1": rc[:, 0],
+ "x2": rc[:, 1],
+ "y1": rc[:, 2],
+ "y2": rc[:, 3],
+ "focus_score": dapi_focus_scores,
+ "focus_score_norm": dapi_focus_scores_norm,
+ "dapi_focus_score": dapi_focus_scores,
+ "dapi_focus_score_norm": dapi_focus_scores_norm,
+ "dapi_lap_var": dapi_lap_vars,
+ "dapi_intensity": dapi_intensities,
+ "raw_intensity": dapi_intensities.copy(),
+ "boundary_focus_score": boundary_focus_scores if has_boundary else np.nan,
+ "boundary_focus_score_norm": boundary_focus_scores_norm
+ if has_boundary
+ else np.nan,
+ "boundary_intensity": boundary_intensities if has_boundary else np.nan,
+ "intrna_focus_score": intrna_focus_scores if has_intrna else np.nan,
+ "intrna_focus_score_norm": intrna_focus_scores_norm if has_intrna else np.nan,
+ "intrna_intensity": intrna_intensities if has_intrna else np.nan,
+ "tissue_coverage": grid["tissue_coverages"],
+ "overlaps_tissue": grid["tissue_coverages"] > 0.0,
+ "is_boundary_roi": grid["is_boundary_roi"],
+ }
+ df_grid_roi = pd.DataFrame(df_data)
+
+ if return_pixel_maps:
+ focus_maps = {
+ "dapi_focus_map": dapi_focus_map,
+ "dapi_mean_map": dapi_mean_map,
+ "dapi_lap_var_map": None, # already freed
+ "boundary_focus_map": None, # already freed
+ "boundary_mean_map": boundary_mean_map,
+ "intrna_focus_map": None, # already freed
+ "intrna_mean_map": intrna_mean_map,
+ }
+ return df_grid_roi, focus_maps, None
+ return df_grid_roi
+
+
+def calculate_roi_blur_threshold(
+ df_grid_roi,
+ intensity_threshold=ROI_INTENSITY_THRESHOLD,
+ focus_percentile=ROI_FOCUS_SCORE_PERCENTILE,
+):
+ """
+ Calculate tile blur detection threshold from raw focus scores.
+
+ Option B: Exclude tiles with intensity < intensity_threshold, then calculate
+ percentile from remaining tissue tiles. This focuses the threshold on tissue regions.
+
+ The threshold is applied to RAW focus scores (not normalized). Classification
+ is: blurred if (focus_score <= threshold) OR (intensity < intensity_threshold)
+
+ Parameters:
+ -----------
+ df_grid_roi : pandas DataFrame
+ DataFrame with 'dapi_focus_score' (raw scores) and 'dapi_intensity'
+ intensity_threshold : float, optional
+ Minimum intensity to include tiles in percentile calculation (default: ROI_INTENSITY_THRESHOLD)
+ focus_percentile : float, optional
+ Percentile of raw scores to use as threshold (default: ROI_FOCUS_SCORE_PERCENTILE)
+
+ Returns:
+ --------
+ float
+ Threshold value in raw score units
+ """
+ # Get intensity column (handle both naming conventions)
+ intensity_col = (
+ "dapi_intensity" if "dapi_intensity" in df_grid_roi.columns else "raw_intensity"
+ )
+ if intensity_col not in df_grid_roi.columns:
+ raise ValueError(
+ "Intensity column not found. Expected 'dapi_intensity' or 'raw_intensity'"
+ )
+
+ # Get focus score column (handle both naming conventions)
+ focus_col = (
+ "dapi_focus_score"
+ if "dapi_focus_score" in df_grid_roi.columns
+ else "focus_score"
+ )
+ if focus_col not in df_grid_roi.columns:
+ raise ValueError(
+ "Focus score column not found. Expected 'dapi_focus_score' or 'focus_score'"
+ )
+
+ # Exclude low-intensity tiles (background) before calculating percentile
+ tissue_rois = df_grid_roi[df_grid_roi[intensity_col] >= intensity_threshold]
+
+ if len(tissue_rois) == 0:
+ # Fallback: if no tissue tiles, use all tiles
+ logging.warning(
+ f" Warning: No tiles with intensity >= {intensity_threshold}, using all tiles for threshold calculation"
+ )
+ tissue_rois = df_grid_roi
+
+ raw_scores = tissue_rois[focus_col].values
+
+ # Calculate threshold on raw scores from tissue ROIs only
+ threshold_raw = np.percentile(raw_scores, focus_percentile)
+
+ return threshold_raw
+
+
+def fit_focus_gmm(
+ df_grid_roi,
+ intensity_threshold: float = ROI_INTENSITY_THRESHOLD,
+ focus_col_name: str = "dapi_focus_score",
+ n_components: int = 2,
+ random_state: int = 0,
+):
+ """
+ Fit a Gaussian Mixture Model (GMM) to tile focus scores from tissue tiles
+ (intensity >= intensity_threshold) in raw focus-score space.
+
+ The model learns 2 components that approximately correspond to:
+ - Lower-focus (blurred) tiles
+ - Higher-focus (in-focus) tiles
+
+ This function does NOT classify tiles directly; it only returns the fitted GMM
+ and identifies which component is the "blur" component (lower mean).
+
+ Parameters
+ ----------
+ df_grid_roi : pandas.DataFrame
+ DataFrame with at least:
+ - focus_col_name (e.g. 'dapi_focus_score' or 'focus_score')
+ - 'dapi_intensity' or 'raw_intensity'
+ intensity_threshold : float, optional
+ Minimum intensity to consider a tile as tissue (default: ROI_INTENSITY_THRESHOLD).
+ Tiles below this are excluded from GMM training.
+ focus_col_name : str, optional
+ Name of the focus-score column to use (default: 'dapi_focus_score').
+ If not present, 'focus_score' will be used.
+ n_components : int, optional
+ Number of Gaussian components for the GMM (default: 2).
+ random_state : int, optional
+ Random seed for reproducibility (default: 0).
+
+ Returns
+ -------
+ gmm : sklearn.mixture.GaussianMixture
+ Fitted GMM model on (log1p(focus_score)) of tissue tiles.
+ blur_component_idx : int
+ Index of the GMM component corresponding to blurred tiles (lower mean).
+ """
+ # Determine intensity column
+ intensity_col = (
+ "dapi_intensity" if "dapi_intensity" in df_grid_roi.columns else "raw_intensity"
+ )
+ if intensity_col not in df_grid_roi.columns:
+ raise ValueError(
+ "Intensity column not found. Expected 'dapi_intensity' or 'raw_intensity'"
+ )
+
+ # Determine focus column
+ if focus_col_name not in df_grid_roi.columns:
+ # Fallback to generic 'focus_score'
+ if "focus_score" not in df_grid_roi.columns:
+ raise ValueError(
+ f"Focus score column '{focus_col_name}' not found and 'focus_score' not present either"
+ )
+ focus_col_name = "focus_score"
+
+ # Select tissue tiles for training
+ tissue_rois = df_grid_roi[df_grid_roi[intensity_col] >= intensity_threshold].copy()
+ if len(tissue_rois) == 0:
+ raise ValueError(
+ f"No tissue tiles found with intensity >= {intensity_threshold}. "
+ f"Cannot fit GMM. Check intensity thresholds or image quality."
+ )
+
+ raw_scores = tissue_rois[focus_col_name].values.astype(np.float64)
+
+ # Log-transform to reduce skewness (handle zeros safely)
+ x_log = np.log1p(raw_scores).reshape(-1, 1)
+
+ # Fit GMM
+ gmm = GaussianMixture(
+ n_components=n_components, covariance_type="full", random_state=random_state
+ )
+ gmm.fit(x_log)
+
+ # Identify which component corresponds to blurred ROIs:
+ # assume lower mean log-focus = blur
+ means = gmm.means_.flatten()
+ blur_component_idx = int(np.argmin(means))
+
+ logging.info(" GMM focus model fitted on tissue tiles")
+ logging.info(f" Number of tissue tiles used: {len(tissue_rois)}")
+ logging.info(f" Component means (log1p focus): {means}")
+ logging.info(f" Blur component index: {blur_component_idx} (lower mean)")
+
+ return gmm, blur_component_idx
+
+
+def classify_roi_blur_by_threshold(
+ df_grid_roi,
+ roi_threshold: float,
+ intensity_threshold: float = ROI_INTENSITY_THRESHOLD,
+ focus_col_name: str = "dapi_focus_score",
+):
+ """
+ Percentile-threshold fallback blur classification.
+
+ Used when the 1D GMM fit fails (e.g. too few tissue tiles on a very dim
+ sample). A tile is classified as blurred if its raw focus score is at or
+ below ``roi_threshold`` OR its intensity is below ``intensity_threshold``.
+ This mirrors the rule documented in ``calculate_roi_blur_threshold`` and
+ produces the same columns as ``classify_roi_blur`` so downstream code is
+ unaffected:
+
+ - 'blur_prob_gmm' : NaN (no posterior probability without a GMM)
+ - 'is_blurred_gmm' : boolean, final classification
+ - 'is_low_intensity' : boolean, intensity < intensity_threshold
+
+ Parameters
+ ----------
+ df_grid_roi : pandas.DataFrame
+ DataFrame with a focus-score column ('dapi_focus_score' or
+ 'focus_score') and an intensity column ('dapi_intensity' or
+ 'raw_intensity').
+ roi_threshold : float
+ Raw focus-score threshold (from ``calculate_roi_blur_threshold``).
+ intensity_threshold : float, optional
+ Tiles below this are auto-blurred (default: ROI_INTENSITY_THRESHOLD).
+ focus_col_name : str, optional
+ Focus-score column to use (default: 'dapi_focus_score'; falls back to
+ 'focus_score').
+ """
+ df = df_grid_roi.copy()
+
+ intensity_col = (
+ "dapi_intensity" if "dapi_intensity" in df.columns else "raw_intensity"
+ )
+ if intensity_col not in df.columns:
+ raise ValueError(
+ "Intensity column not found. Expected 'dapi_intensity' or 'raw_intensity'"
+ )
+
+ if focus_col_name not in df.columns:
+ if "focus_score" not in df.columns:
+ raise ValueError(
+ f"Focus score column '{focus_col_name}' not found and 'focus_score' not present either"
+ )
+ focus_col_name = "focus_score"
+
+ is_low_intensity = df[intensity_col] < intensity_threshold
+ df["is_low_intensity"] = is_low_intensity
+ df["blur_prob_gmm"] = np.nan
+ df["is_blurred_gmm"] = (df[focus_col_name] <= roi_threshold) | is_low_intensity
+
+ total_rois = len(df)
+ n_low_int = int(is_low_intensity.sum())
+ n_blur = int(df["is_blurred_gmm"].sum())
+ pct_low = (n_low_int / total_rois * 100) if total_rois > 0 else 0.0
+ pct_blur = (n_blur / total_rois * 100) if total_rois > 0 else 0.0
+
+ logging.info(" Percentile-threshold fallback blur classification completed")
+ logging.info(f" Total tiles: {total_rois}")
+ logging.info(
+ f" Low-intensity tiles (auto-blurred): {n_low_int} ({pct_low:.1f}%)"
+ )
+ logging.info(
+ f" Blurred tiles (threshold + intensity): {n_blur} ({pct_blur:.1f}%)"
+ )
+
+ return df
+
+
+def classify_roi_blur(
+ df_grid_roi,
+ gmm,
+ blur_component_idx: int,
+ blur_prob_threshold: float = 0.5,
+ intensity_threshold: float = ROI_INTENSITY_THRESHOLD,
+ focus_col_name: str = "dapi_focus_score",
+):
+ """
+ Classify each tile as blurred or in-focus using a fitted GMM and an intensity safeguard.
+
+ Rules:
+ ------
+ - Tiles with intensity < intensity_threshold are always marked as blurred
+ (low-intensity / background or globally problematic tissue).
+ - For tissue tiles (intensity >= threshold), use the GMM posterior probability
+ of belonging to the "blur" component:
+ - blur_prob = P(component == blur_component_idx | focus_score)
+ - Tile is blurred if blur_prob > blur_prob_threshold.
+
+ The classification is added to df_grid_roi in new columns:
+ - 'blur_prob_gmm' : posterior probability of being blurred (NaN for low-intensity tiles)
+ - 'is_blurred_gmm' : boolean, final classification combining intensity + GMM
+ - 'is_low_intensity' : boolean, intensity < intensity_threshold
+
+ Parameters
+ ----------
+ df_grid_roi : pandas.DataFrame
+ DataFrame with at least:
+ - focus_col_name (e.g. 'dapi_focus_score' or 'focus_score')
+ - 'dapi_intensity' or 'raw_intensity'
+ gmm : sklearn.mixture.GaussianMixture
+ Fitted GMM model from fit_focus_gmm()
+ blur_component_idx : int
+ Index of the GMM component corresponding to blurred tiles.
+ blur_prob_threshold : float, optional
+ Threshold on posterior blur probability to classify a tile as blurred
+ (default: 0.5).
+ intensity_threshold : float, optional
+ Intensity safeguard: tiles below this are auto-blurred (default: ROI_INTENSITY_THRESHOLD).
+ focus_col_name : str, optional
+ Name of focus-score column used (default: 'dapi_focus_score').
+
+ Returns
+ -------
+ df_grid_roi : pandas.DataFrame
+ Input DataFrame with new columns:
+ - 'blur_prob_gmm'
+ - 'is_blurred_gmm'
+ - 'is_low_intensity'
+ """
+ df = df_grid_roi.copy()
+
+ # Intensity column
+ intensity_col = (
+ "dapi_intensity" if "dapi_intensity" in df.columns else "raw_intensity"
+ )
+ if intensity_col not in df.columns:
+ raise ValueError(
+ "Intensity column not found. Expected 'dapi_intensity' or 'raw_intensity'"
+ )
+
+ # Focus column
+ if focus_col_name not in df.columns:
+ if "focus_score" not in df.columns:
+ raise ValueError(
+ f"Focus score column '{focus_col_name}' not found and 'focus_score' not present either"
+ )
+ focus_col_name = "focus_score"
+
+ # Intensity-based low-intensity flag
+ is_low_intensity = df[intensity_col] < intensity_threshold
+ df["is_low_intensity"] = is_low_intensity
+
+ # Initialize columns
+ df["blur_prob_gmm"] = np.nan
+ df["is_blurred_gmm"] = False
+
+ # Tissue ROIs: intensity >= threshold
+ tissue_mask = ~is_low_intensity
+ if tissue_mask.any():
+ raw_scores = df.loc[tissue_mask, focus_col_name].values.astype(np.float64)
+ x_log = np.log1p(raw_scores).reshape(-1, 1)
+
+ # Posterior probabilities of components
+ probs = gmm.predict_proba(x_log)
+ blur_prob = probs[:, blur_component_idx]
+
+ df.loc[tissue_mask, "blur_prob_gmm"] = blur_prob
+ df.loc[tissue_mask, "is_blurred_gmm"] = blur_prob > blur_prob_threshold
+
+ # Low-intensity tiles are always blurred according to the safeguard
+ df.loc[is_low_intensity, "is_blurred_gmm"] = True
+
+ # Summary
+ total_rois = len(df)
+ n_low_int = int(is_low_intensity.sum())
+ n_blur = int(df["is_blurred_gmm"].sum())
+ pct_blur = (n_blur / total_rois) * 100 if total_rois > 0 else 0.0
+
+ logging.info(" GMM-based blur classification completed")
+ logging.info(f" Total tiles: {total_rois}")
+ logging.info(
+ f" Low-intensity tiles (auto-blurred): {n_low_int} ({n_low_int / total_rois * 100:.1f}%)"
+ )
+ logging.info(f" Blurred tiles (GMM + intensity): {n_blur} ({pct_blur:.1f}%)")
+
+ return df
+
+
+def fit_focus_gmm_2d(
+ df_grid_roi,
+ intensity_threshold: float = ROI_INTENSITY_THRESHOLD,
+ focus_col_name: str = "dapi_focus_score",
+ n_components: int = 2,
+ random_state: int = 0,
+):
+ """
+ Fit a 2D Gaussian Mixture Model (GMM) to tile focus scores using both
+ focus score (std²/mean) and Laplacian variance as features.
+
+ The model uses log1p-transformed features:
+ - Feature 1: log1p(dapi_focus_score)
+ - Feature 2: log1p(dapi_lap_var)
+
+ This allows the model to use both global contrast (focus_score) and
+ high-frequency content (Laplacian variance) to distinguish blurred vs in-focus tiles.
+
+ Parameters
+ ----------
+ df_grid_roi : pandas.DataFrame
+ DataFrame with at least:
+ - focus_col_name (e.g. 'dapi_focus_score' or 'focus_score')
+ - 'dapi_lap_var' (Laplacian variance)
+ - 'dapi_intensity' or 'raw_intensity'
+ intensity_threshold : float, optional
+ Minimum intensity to consider a tile as tissue (default: ROI_INTENSITY_THRESHOLD).
+ Tiles below this are excluded from GMM training.
+ focus_col_name : str, optional
+ Name of the focus-score column to use (default: 'dapi_focus_score').
+ If not present, 'focus_score' will be used.
+ n_components : int, optional
+ Number of Gaussian components for the GMM (default: 2).
+ random_state : int, optional
+ Random seed for reproducibility (default: 0).
+
+ Returns
+ -------
+ gmm : sklearn.mixture.GaussianMixture
+ Fitted 2D GMM model on [log1p(focus_score), log1p(lap_var)] of tissue tiles.
+ blur_component_idx : int
+ Index of the GMM component corresponding to blurred tiles (lower mean on focus_score dimension).
+ """
+ # Determine intensity column
+ intensity_col = (
+ "dapi_intensity" if "dapi_intensity" in df_grid_roi.columns else "raw_intensity"
+ )
+ if intensity_col not in df_grid_roi.columns:
+ raise ValueError(
+ "Intensity column not found. Expected 'dapi_intensity' or 'raw_intensity'"
+ )
+
+ # Determine focus column
+ if focus_col_name not in df_grid_roi.columns:
+ # Fallback to generic 'focus_score'
+ if "focus_score" not in df_grid_roi.columns:
+ raise ValueError(
+ f"Focus score column '{focus_col_name}' not found and 'focus_score' not present either"
+ )
+ focus_col_name = "focus_score"
+
+ # Check for Laplacian variance column
+ if "dapi_lap_var" not in df_grid_roi.columns:
+ raise ValueError(
+ "dapi_lap_var column not found. 2D GMM requires Laplacian variance. "
+ "Make sure calculate_roi_focusscore() was used (not calculate_roi_focusscore_without_laplace)."
+ )
+
+ # Select tissue tiles for training.
+ # 2026-06-22 (qc_drift_analysis): select by tissue-mask coverage, not a raw
+ # intensity floor. The old `intensity >= ROI_INTENSITY_THRESHOLD` gate broke
+ # on dim XOA-4.0 images (~14x dimmer), dropping most real tissue from the
+ # training set and biasing the blur/focus split. Coverage is
+ # brightness-independent. Falls back to the intensity gate only when
+ # tissue_coverage is unavailable. NB: this still trains the GMM on tissue
+ # only (it is NOT the reverted "all-tiles" Scope B, commit e8e6731), so the
+ # within-tissue blur-vs-focus bimodality is preserved.
+ if "tissue_coverage" in df_grid_roi.columns:
+ tissue_rois = df_grid_roi[
+ df_grid_roi["tissue_coverage"] >= ROI_MIN_TISSUE_COVERAGE_FOR_INTENSITY_QC
+ ].copy()
+ _selection_desc = (
+ f"tissue_coverage >= {ROI_MIN_TISSUE_COVERAGE_FOR_INTENSITY_QC}"
+ )
+ else:
+ tissue_rois = df_grid_roi[
+ df_grid_roi[intensity_col] >= intensity_threshold
+ ].copy()
+ _selection_desc = f"intensity >= {intensity_threshold}"
+ if len(tissue_rois) == 0:
+ raise ValueError(
+ f"No tissue tiles found with {_selection_desc}. "
+ f"Cannot fit GMM. Check tissue mask / intensity thresholds or image quality."
+ )
+
+ # Get both features
+ focus_scores = tissue_rois[focus_col_name].values.astype(np.float64)
+ lap_vars = tissue_rois["dapi_lap_var"].values.astype(np.float64)
+
+ # Filter out NaN values
+ valid_mask = ~(np.isnan(focus_scores) | np.isnan(lap_vars))
+ if valid_mask.sum() == 0:
+ raise ValueError("No valid tile data (all NaN) for 2D GMM training")
+
+ focus_scores_valid = focus_scores[valid_mask]
+ lap_vars_valid = lap_vars[valid_mask]
+
+ # Log-transform both features (clamp negatives to 0 before log1p)
+ x_focus_log = np.log1p(np.maximum(focus_scores_valid, 0.0))
+ x_lap_log = np.log1p(np.maximum(lap_vars_valid, 0.0))
+
+ # Combine into 2D feature matrix
+ x_2d = np.column_stack([x_focus_log, x_lap_log])
+
+ # Fit 2D GMM
+ gmm = GaussianMixture(
+ n_components=n_components, covariance_type="full", random_state=random_state
+ )
+ gmm.fit(x_2d)
+
+ # Identify which component corresponds to blurred ROIs:
+ # Use the focus_score dimension (first column) - lower mean = blur
+ means_focus = gmm.means_[:, 0] # First dimension (focus_score)
+ blur_component_idx = int(np.argmin(means_focus))
+
+ logging.info(" 2D GMM focus model fitted on tissue tiles")
+ logging.info(f" Number of tissue tiles used: {valid_mask.sum()}")
+ logging.info(" Component means (log1p focus_score, log1p lap_var):")
+ for i, mean in enumerate(gmm.means_):
+ logging.info(f" Component {i}: [{mean[0]:.4f}, {mean[1]:.4f}]")
+ logging.info(
+ f" Blur component index: {blur_component_idx} (lower mean on focus_score dimension)"
+ )
+
+ return gmm, blur_component_idx
+
+
+def classify_roi_blur_2d(
+ df_grid_roi,
+ gmm,
+ blur_component_idx: int,
+ blur_prob_threshold: float = 0.5,
+ intensity_threshold: float = ROI_INTENSITY_THRESHOLD,
+ focus_col_name: str = "dapi_focus_score",
+):
+ """
+ Classify each tile as blurred or in-focus using a fitted 2D GMM and an intensity safeguard.
+
+ Uses both focus_score and Laplacian variance as features.
+
+ Rules:
+ ------
+ - Tissue is defined by tissue_coverage >= ROI_MIN_TISSUE_COVERAGE_FOR_INTENSITY_QC
+ when the column is present (brightness-independent); otherwise it falls back
+ to intensity >= intensity_threshold.
+ - Non-tissue tiles (low coverage / background) are marked as blurred. They are
+ excluded from the tissue-filtered blur % regardless.
+ - Tissue tiles missing dapi_lap_var are marked as blurred (cannot use 2D GMM).
+ - For tissue tiles with valid lap_var, use the 2D GMM posterior probability
+ of belonging to the "blur" component:
+ - blur_prob = P(component == blur_component_idx | focus_score, lap_var)
+ - Tile is blurred if blur_prob > blur_prob_threshold.
+
+ The classification is added to df_grid_roi in new columns:
+ - 'blur_prob_gmm_2d' : posterior probability of being blurred (NaN for low-intensity or missing lap_var tiles)
+ - 'is_blurred_gmm_2d' : boolean, final classification combining intensity + 2D GMM
+ - 'is_low_intensity' : boolean, intensity < intensity_threshold (reused if exists)
+
+ Parameters
+ ----------
+ df_grid_roi : pandas.DataFrame
+ DataFrame with at least:
+ - focus_col_name (e.g. 'dapi_focus_score' or 'focus_score')
+ - 'dapi_lap_var' (Laplacian variance)
+ - 'dapi_intensity' or 'raw_intensity'
+ gmm : sklearn.mixture.GaussianMixture
+ Fitted 2D GMM model from fit_focus_gmm_2d()
+ blur_component_idx : int
+ Index of the GMM component corresponding to blurred tiles.
+ blur_prob_threshold : float, optional
+ Threshold on posterior blur probability to classify a tile as blurred
+ (default: 0.5).
+ intensity_threshold : float, optional
+ Intensity safeguard: tiles below this are auto-blurred (default: ROI_INTENSITY_THRESHOLD).
+ focus_col_name : str, optional
+ Name of focus-score column used (default: 'dapi_focus_score').
+
+ Returns
+ -------
+ df_grid_roi : pandas.DataFrame
+ Input DataFrame with new columns:
+ - 'blur_prob_gmm_2d'
+ - 'is_blurred_gmm_2d'
+ - 'is_low_intensity' (if not already present)
+ """
+ df = df_grid_roi.copy()
+
+ # Intensity column
+ intensity_col = (
+ "dapi_intensity" if "dapi_intensity" in df.columns else "raw_intensity"
+ )
+ if intensity_col not in df.columns:
+ raise ValueError(
+ "Intensity column not found. Expected 'dapi_intensity' or 'raw_intensity'"
+ )
+
+ # Focus column
+ if focus_col_name not in df.columns:
+ if "focus_score" not in df.columns:
+ raise ValueError(
+ f"Focus score column '{focus_col_name}' not found and 'focus_score' not present either"
+ )
+ focus_col_name = "focus_score"
+
+ # Check for Laplacian variance
+ if "dapi_lap_var" not in df.columns:
+ raise ValueError(
+ "dapi_lap_var column not found. 2D GMM classification requires Laplacian variance."
+ )
+
+ # Intensity-based low-intensity flag (reuse if exists, otherwise create)
+ if "is_low_intensity" not in df.columns:
+ is_low_intensity = df[intensity_col] < intensity_threshold
+ df["is_low_intensity"] = is_low_intensity
+ else:
+ is_low_intensity = df["is_low_intensity"]
+
+ # Initialize columns
+ df["blur_prob_gmm_2d"] = np.nan
+ df["is_blurred_gmm_2d"] = False
+
+ # Tissue ROIs for blur classification.
+ # 2026-06-22 (qc_drift_analysis): define tissue by mask coverage, not the
+ # intensity floor — matches fit_focus_gmm_2d. On dim XOA-4.0 images many real
+ # tissue tiles fall below ROI_INTENSITY_THRESHOLD; the old rule force-blurred
+ # them and inflated the blur rate (~40% on v4). Coverage is
+ # brightness-independent. is_low_intensity is still computed above for
+ # intensity QC / reporting, just not used to gate blur here.
+ if "tissue_coverage" in df.columns:
+ tissue_mask = df["tissue_coverage"] >= ROI_MIN_TISSUE_COVERAGE_FOR_INTENSITY_QC
+ else:
+ tissue_mask = ~is_low_intensity
+ has_lap_var = df["dapi_lap_var"].notna()
+ valid_mask = tissue_mask & has_lap_var
+
+ if valid_mask.any():
+ # Get both features for valid ROIs
+ focus_scores = df.loc[valid_mask, focus_col_name].values.astype(np.float64)
+ lap_vars = df.loc[valid_mask, "dapi_lap_var"].values.astype(np.float64)
+
+ # Filter out any remaining NaN (shouldn't happen, but safety check)
+ valid_data_mask = ~(np.isnan(focus_scores) | np.isnan(lap_vars))
+ if valid_data_mask.sum() > 0:
+ focus_scores_valid = focus_scores[valid_data_mask]
+ lap_vars_valid = lap_vars[valid_data_mask]
+
+ # Log-transform (clamp negatives to 0, matching fit_focus_gmm_2d)
+ x_focus_log = np.log1p(np.maximum(focus_scores_valid, 0.0))
+ x_lap_log = np.log1p(np.maximum(lap_vars_valid, 0.0))
+
+ # Combine into 2D feature matrix
+ x_2d = np.column_stack([x_focus_log, x_lap_log])
+
+ # Posterior probabilities of components
+ probs = gmm.predict_proba(x_2d)
+ blur_prob = probs[:, blur_component_idx]
+
+ # Map back to original valid_mask indices
+ valid_indices = df.index[valid_mask][valid_data_mask]
+ df.loc[valid_indices, "blur_prob_gmm_2d"] = blur_prob
+ df.loc[valid_indices, "is_blurred_gmm_2d"] = blur_prob > blur_prob_threshold
+
+ # Non-tissue tiles (low mask coverage / background) are not treated as
+ # focused. They are excluded from the tissue-filtered blur % anyway; this
+ # only affects the unfiltered count, preserving the prior convention.
+ df.loc[~tissue_mask, "is_blurred_gmm_2d"] = True
+
+ # Tissue tiles without lap_var cannot be classified by the 2D GMM → blurred
+ missing_lap_var = ~has_lap_var & tissue_mask
+ if missing_lap_var.any():
+ df.loc[missing_lap_var, "is_blurred_gmm_2d"] = True
+
+ # Summary
+ total_rois = len(df)
+ n_low_int = int(is_low_intensity.sum())
+ n_missing_lap = int(missing_lap_var.sum()) if missing_lap_var.any() else 0
+ n_blur = int(df["is_blurred_gmm_2d"].sum())
+ pct_blur = (n_blur / total_rois) * 100 if total_rois > 0 else 0.0
+
+ logging.info(" 2D GMM-based blur classification completed")
+ logging.info(f" Total tiles: {total_rois}")
+ logging.info(
+ f" Low-intensity tiles (auto-blurred): {n_low_int} ({n_low_int / total_rois * 100:.1f}%)"
+ )
+ if n_missing_lap > 0:
+ logging.info(
+ f" Tiles missing lap_var (auto-blurred): {n_missing_lap} ({n_missing_lap / total_rois * 100:.1f}%)"
+ )
+ logging.info(f" Blurred tiles (2D GMM + intensity): {n_blur} ({pct_blur:.1f}%)")
+
+ return df
+
+
+_MAX_SCATTER_POINTS = 10_000
+
+
+def _subsample_idx(mask, max_points=None, rng_seed=42):
+ """Return indices where *mask* is True, randomly subsampled to *max_points*."""
+ if max_points is None:
+ max_points = _MAX_SCATTER_POINTS
+ idx = np.where(mask)[0]
+ if max_points <= 0 or len(idx) <= max_points:
+ return idx
+ rng = np.random.default_rng(rng_seed)
+ return rng.choice(idx, size=max_points, replace=False)
+
+
+def plot_grid_roi_focus_heatmap(
+ df_grid_roi,
+ small0,
+ figures_dir,
+ figures_source_dir,
+ threshold=-1.0,
+ focus_maps=None,
+ focus_heatmap=None,
+):
+ """
+ Create heatmap of grid tile focus scores across the whole tissue.
+
+ When pixel-level ``focus_maps`` are provided, renders smooth per-pixel
+ heatmaps via imshow (much faster and higher resolution than Rectangle
+ patches). Falls back to the legacy Rectangle-patch approach otherwise.
+
+ Parameters:
+ -----------
+ df_grid_roi : pandas DataFrame
+ Grid ROI DataFrame with columns: roi_id, x1, x2, y1, y2, focus_score,
+ focus_score_norm, raw_intensity, tissue_coverage
+ small0 : numpy.ndarray
+ Downsampled DAPI image (level 3, 8x downsampling)
+ figures_dir : Path
+ Directory to save figures
+ figures_source_dir : Path
+ Directory to save source data
+ threshold : float, optional
+ Threshold for normalized focus score (default: -1.0)
+ focus_maps : dict or None, optional
+ Pixel-level focus maps from compute_all_focus_maps(). If provided,
+ uses imshow for smooth rendering instead of Rectangle patches.
+ """
+ from skimage.transform import downscale_local_mean
+ from mpl_toolkits.axes_grid1 import make_axes_locatable
+
+ downsample_factor = 8
+ img_height, img_width = small0.shape
+ img_aspect = img_height / img_width if img_width > 0 else 1.0
+ panel_width = 6
+ # Floor at 50% of width so very wide slides (e.g. brain) don't squash titles / colorbars
+ panel_height = max(panel_width * 0.5, panel_width * img_aspect)
+ fig, axes = plt.subplots(1, 2, figsize=(2 * panel_width + 2, panel_height))
+ fig.suptitle(
+ "Spatial focus-score map across the Xenium region",
+ fontsize=15,
+ fontweight="bold",
+ y=1.02,
+ )
+
+ # DAPI background p99 is identical for both panels -- compute once over the full
+ # ~86 Mpx small0. Draw the background via _imshow_thumb (block-mean to the ~2000px
+ # panel) not a full-res imshow: the raw imshow was ~50 s/panel x2 panels x2 saves
+ # and was the entire residual cost of this figure (the binned panels are ~5 s).
+ _bg_vmax = np.percentile(small0, 99)
+
+ # Plot 1: Focus score heatmap
+ ax = axes[0]
+ _imshow_thumb(
+ ax,
+ small0,
+ cmap="Greys_r",
+ vmax=_bg_vmax,
+ alpha=0.5,
+ aspect="auto",
+ extent=[0, img_width, img_height, 0],
+ origin="upper",
+ )
+
+ # Streaming mode supplies the down-sampled canvas directly, and it reproduces
+ # downscale_local_mean bit-exactly: write boundaries are aligned to the block size so
+ # each block is reduced by a single reshape in skimage's own order. No
+ # full-resolution plane is read (22 GB, plus another 22 GB for skimage's pad).
+ heatmap = focus_heatmap
+ if heatmap is None and focus_maps is not None:
+ dapi_focus = focus_maps.get("dapi_focus_map")
+ if dapi_focus is not None:
+ heatmap = downscale_local_mean(
+ dapi_focus, (downsample_factor, downsample_factor)
+ )
+ if heatmap is not None:
+ # Binned redesign: an imshow of the full ~86-megapixel per-pixel field cost
+ # ~126 s (it measured 4459.8 s on a 102045x53908 sample, runs 1ZyVIlaKBYxJrQ /
+ # 4c0HFuivKDkWXr -- the largest single cost in the step) and rendered millions
+ # of unreadable per-pixel dots. Aggregating to a coarse grid (~180 cells on the
+ # long axis) keeps the regional signal and draws in ~1 s.
+ t_h = time.perf_counter()
+ # Clip to small0 dimensions (in case of rounding: the canvas can be a row/column
+ # larger, ceil vs floor).
+ focus_ds = np.asarray(heatmap)[:img_height, :img_width]
+ positive = focus_ds > 0
+ has_positive = bool(positive.any())
+ # Color range comes from the FULL-resolution positive field, not the binned
+ # means, so viridis maps exactly as the per-pixel figure did.
+ vmin = np.percentile(focus_ds[positive], 1) if has_positive else 0
+ vmax = np.percentile(focus_ds[positive], 99) if has_positive else 1
+ # Per-bin MEAN focus over tissue pixels (focus_ds > 0). Non-tissue pixels are
+ # NaN'd so they do not drag the mean down; all-non-tissue bins come back NaN and
+ # are set to vmin so they render as viridis-min -- the same dark background the
+ # per-pixel field showed where focus_ds == 0.
+ step = max(1, -(-max(img_height, img_width) // _FOCUS_HEATMAP_BINS_LONG))
+ tissue_focus = np.where(positive, focus_ds, np.nan)
+ del positive
+ focus_binned = _bin_nanmean(tissue_focus, step)
+ focus_binned = np.where(np.isnan(focus_binned), vmin, focus_binned)
+ ax.imshow(
+ focus_binned,
+ cmap="viridis",
+ alpha=0.6,
+ aspect="auto",
+ extent=[0, img_width, img_height, 0],
+ origin="upper",
+ vmin=vmin,
+ vmax=vmax,
+ interpolation="nearest",
+ )
+ logging.info(
+ f" [TIMING] fig5 left bin+imshow (step={step}, "
+ f"{focus_binned.shape}): {time.perf_counter() - t_h:.1f}s"
+ )
+ sm = plt.cm.ScalarMappable(
+ cmap="viridis", norm=plt.Normalize(vmin=vmin, vmax=vmax)
+ )
+ sm.set_array([])
+ cax = make_axes_locatable(ax).append_axes("right", size="3%", pad=0.05)
+ cbar = fig.colorbar(sm, cax=cax)
+ cbar.set_label("Focus Score (var/mean)", fontsize=12)
+ ax.set_title("Focus Score (binned)", fontsize=14)
+ else:
+ # Legacy Rectangle-patch approach
+ from matplotlib.patches import Rectangle
+
+ for _, roi in df_grid_roi.iterrows():
+ x1_ds = roi["x1"] / downsample_factor
+ x2_ds = roi["x2"] / downsample_factor
+ y1_ds = roi["y1"] / downsample_factor
+ y2_ds = roi["y2"] / downsample_factor
+ w = x2_ds - x1_ds
+ h = y2_ds - y1_ds
+ if w <= 0 or h <= 0:
+ continue
+ if x1_ds < 0 or y1_ds < 0 or x2_ds > img_width or y2_ds > img_height:
+ continue
+ norm_score = np.clip((roi["focus_score_norm"] + 3) / 6, 0, 1)
+ color = plt.cm.viridis(norm_score)
+ rect = Rectangle(
+ (x1_ds, y1_ds), w, h, facecolor=color, alpha=0.6, edgecolor="none"
+ )
+ ax.add_patch(rect)
+ sm = plt.cm.ScalarMappable(cmap="viridis", norm=plt.Normalize(vmin=-3, vmax=3))
+ sm.set_array([])
+ cax = make_axes_locatable(ax).append_axes("right", size="3%", pad=0.05)
+ cbar = fig.colorbar(sm, cax=cax)
+ cbar.set_label("Focus Score (Normalized)", fontsize=12)
+ cbar.ax.axhline(y=float(threshold), color="red", linewidth=1.5, linestyle="--")
+ ax.set_title("Grid Tile Focus Score Heatmap (Normalized)", fontsize=14)
+
+ ax.set_xlim(0, img_width)
+ ax.set_ylim(img_height, 0)
+ ax.set_aspect("equal")
+ ax.axis("off")
+
+ # Plot 2: GMM 2D classification (blurred vs in-focus)
+ ax = axes[1]
+ _imshow_thumb(
+ ax,
+ small0,
+ cmap="Greys_r",
+ vmax=_bg_vmax,
+ alpha=0.5,
+ aspect="auto",
+ extent=[0, img_width, img_height, 0],
+ origin="upper",
+ )
+
+ has_gmm_2d = "is_blurred_gmm_2d" in df_grid_roi.columns
+
+ # Gated on the GMM column alone, NOT on focus_maps: this branch reads no pixel
+ # data. It rasterises `is_blurred_gmm_2d` from the ROI table's own x1/x2/y1/y2
+ # columns at down-sampled resolution, so the focus_maps check was only ever a
+ # proxy for "not the legacy path".
+ #
+ # It mattered because streaming passes focus_maps=None, which sent this panel to
+ # the Rectangle fallback below -- one patch per ROI through df.iterrows(). On the
+ # 102045x53908 sample that is 4,490,640 patches, and it took Figure 5 from 15.9 s
+ # to 4388-4460 s. Measured on runs 47jub5CwOHb82v (streamed, 436.8 s at 429 k
+ # ROIs) against 3xiurC3181Zgwc (planes, 15.9 s, same bundle).
+ #
+ # `has_gmm_2d` reproduces the old behaviour exactly where it mattered: the 2D GMM
+ # needs dapi_lap_var, which --legacy-focus does not produce, so that path still
+ # falls through to the Rectangle branch.
+ if has_gmm_2d:
+ from matplotlib.colors import LinearSegmentedColormap
+
+ t_h = time.perf_counter()
+ # Rasterise the in-focus indicator at downsampled resolution: 1.0 for an
+ # in-focus ROI pixel, 0.0 for a blurred one, NaN off-tissue (no ROI). ALL ROIs
+ # are painted -- not just blurred ones -- because a bin needs its tissue
+ # denominator; the fraction in focus is meaningless without it.
+ infocus_ds = np.full((img_height, img_width), np.nan, dtype=np.float32)
+ x1_arr = (df_grid_roi["x1"].values // downsample_factor).astype(int)
+ x2_arr = np.minimum(
+ df_grid_roi["x2"].values // downsample_factor, img_width
+ ).astype(int)
+ y1_arr = (df_grid_roi["y1"].values // downsample_factor).astype(int)
+ y2_arr = np.minimum(
+ df_grid_roi["y2"].values // downsample_factor, img_height
+ ).astype(int)
+ infocus_val = np.where(
+ df_grid_roi["is_blurred_gmm_2d"].values, 0.0, 1.0
+ ).astype(np.float32)
+ for i in range(len(x1_arr)):
+ infocus_ds[y1_arr[i] : y2_arr[i], x1_arr[i] : x2_arr[i]] = infocus_val[i]
+ # Per-bin FRACTION in focus: mean of the 0/1 indicator over tissue pixels, on
+ # the SAME coarse grid as the left panel. A continuous 0..1 field is smooth
+ # under binning (no per-pixel speckle) -- the whole point of the redesign.
+ # All-non-tissue bins stay NaN and render transparent (set_bad alpha 0), so the
+ # DAPI background shows through off-tissue.
+ step = max(1, -(-max(img_height, img_width) // _FOCUS_HEATMAP_BINS_LONG))
+ frac_infocus = _bin_nanmean(infocus_ds, step)
+ # Continuous red(0 = blurred) -> blue(1 = in focus): the same two colors the
+ # per-tile ListedColormap used, now as the endpoints of a smooth map.
+ cmap_focus = LinearSegmentedColormap.from_list("focus_frac", ["red", "blue"])
+ cmap_focus.set_bad(alpha=0.0)
+ ax.imshow(
+ frac_infocus,
+ cmap=cmap_focus,
+ alpha=0.5,
+ aspect="auto",
+ extent=[0, img_width, img_height, 0],
+ origin="upper",
+ vmin=0,
+ vmax=1,
+ interpolation="nearest",
+ )
+ sm = plt.cm.ScalarMappable(cmap=cmap_focus, norm=plt.Normalize(vmin=0, vmax=1))
+ sm.set_array([])
+ cax = make_axes_locatable(ax).append_axes("right", size="3%", pad=0.05)
+ cbar = fig.colorbar(sm, cax=cax)
+ cbar.set_label("In-focus fraction (0 = blurred, 1 = in focus)", fontsize=12)
+ logging.info(
+ f" [TIMING] fig5 right bin+imshow (step={step}, "
+ f"{frac_infocus.shape}): {time.perf_counter() - t_h:.1f}s"
+ )
+ ax.set_title("Focus Classification (2D GMM)", fontsize=14)
+ else:
+ # Legacy Rectangle-patch approach
+ from matplotlib.patches import Rectangle
+
+ for _, roi in df_grid_roi.iterrows():
+ x1_ds = roi["x1"] / downsample_factor
+ x2_ds = roi["x2"] / downsample_factor
+ y1_ds = roi["y1"] / downsample_factor
+ y2_ds = roi["y2"] / downsample_factor
+ w = x2_ds - x1_ds
+ h = y2_ds - y1_ds
+ if w <= 0 or h <= 0:
+ continue
+ if x1_ds < 0 or y1_ds < 0 or x2_ds > img_width or y2_ds > img_height:
+ continue
+ if has_gmm_2d:
+ color = "blue" if not roi["is_blurred_gmm_2d"] else "red"
+ else:
+ color = "blue" if roi["focus_score_norm"] > threshold else "red"
+ rect = Rectangle(
+ (x1_ds, y1_ds), w, h, facecolor=color, alpha=0.6, edgecolor="none"
+ )
+ ax.add_patch(rect)
+ if has_gmm_2d:
+ ax.set_title(
+ "Grid Tile Focus Score (2D GMM: Blue=In-Focus, Red=Blurred)",
+ fontsize=14,
+ )
+ else:
+ ax.set_title(f"Grid Tile Focus Score (Threshold={threshold})", fontsize=14)
+
+ ax.set_xlim(0, img_width)
+ ax.set_ylim(img_height, 0)
+ ax.set_aspect("equal")
+ ax.axis("off")
+
+ # The 2D-GMM branch gives the right panel its own real colorbar. Only the legacy
+ # (no-GMM) branch has none, so add a phantom cax there to keep its plotting region
+ # the same width as the left panel (whose width is shrunk by its colorbar).
+ if not has_gmm_2d:
+ cax_r = make_axes_locatable(axes[1]).append_axes("right", size="3%", pad=0.05)
+ cax_r.axis("off")
+
+ plt.tight_layout()
+ plt.savefig(
+ figures_dir / "grid_roi_focus_heatmap.png", dpi=300, bbox_inches="tight"
+ )
+ plt.close(fig)
+
+ # Save data as CSV
+ if figures_source_dir is not None:
+ df_grid_roi_scaled = df_grid_roi.copy()
+ df_grid_roi_scaled["x1_ds"] = df_grid_roi_scaled["x1"] / downsample_factor
+ df_grid_roi_scaled["x2_ds"] = df_grid_roi_scaled["x2"] / downsample_factor
+ df_grid_roi_scaled["y1_ds"] = df_grid_roi_scaled["y1"] / downsample_factor
+ df_grid_roi_scaled["y2_ds"] = df_grid_roi_scaled["y2"] / downsample_factor
+ df_grid_roi_scaled.to_csv(
+ figures_source_dir / "grid_roi_focus_heatmap.csv", index=False
+ )
+
+
+def plot_snr_roi_heatmap(
+ df_grid_roi,
+ small0,
+ figures_dir,
+ figures_source_dir,
+ snr_thresholds=None,
+):
+ """Paint per-tile transcript SNR spatially on the DAPI background.
+
+ Two-panel figure:
+ Left – neg_pct (fraction of negative-control transcripts per tile)
+ Right – roi_tx_snr_ratio (real / negative transcript ratio, log scale)
+
+ Colorbars autoscale to the data's p99 (no fixed floor). When provided,
+ WARN/FAIL threshold lines from `snr_thresholds` are drawn on each
+ colorbar. Lines outside the autoscaled range are clipped by matplotlib —
+ on a clean slide where data is far below FAIL, the line simply doesn't
+ appear, which is the intended visual cue.
+
+ Parameters
+ ----------
+ df_grid_roi : pandas.DataFrame
+ Must contain columns: x1, x2, y1, y2, neg_pct, roi_tx_snr_ratio,
+ snr_total_tx. Optionally: snr_real_tx, snr_neg_tx — when present, the
+ right panel paints (real+1)/(neg+1) (Laplace pseudocount) so tiles
+ with neg=0 or real=0 (currently NaN/0 under raw ratio) render in
+ their correct extremes. The canonical roi_tx_snr_ratio column stays
+ raw — pseudocount applies to *display only*, not to verdicts/JSON.
+ small0 : numpy.ndarray
+ Downsampled DAPI image (level 3, 8× downsampling).
+ figures_dir, figures_source_dir : Path
+ Output directories.
+ snr_thresholds : dict, optional
+ YAML ``snr.roi_tx`` block. Keys: neg_pct_warn, neg_pct_fail,
+ ratio_warn, ratio_fail.
+ """
+ from matplotlib.colors import LogNorm
+
+ _t = snr_thresholds or {}
+
+ downsample_factor = 8
+ img_height, img_width = small0.shape
+
+ # Phase v5: scope to within-tissue tiles with transcripts. Tissue tiles
+ # WITHOUT transcripts (alveolar / bronchiolar airspaces in lung, etc.)
+ # stay transparent so the DAPI background still shows through them.
+ if "tissue_coverage" in df_grid_roi.columns:
+ df_plot = df_grid_roi[
+ (df_grid_roi["snr_total_tx"] > 0) & (df_grid_roi["tissue_coverage"] > 0.5)
+ ].copy()
+ else:
+ df_plot = df_grid_roi[df_grid_roi["snr_total_tx"] > 0].copy()
+ if df_plot.empty:
+ logging.warning("No tiles with transcripts — skipping SNR heatmap.")
+ return
+
+ # Compute figure size from image aspect ratio to avoid empty white space
+ img_aspect = img_height / img_width if img_width > 0 else 1.0
+ panel_width = 6 # width per panel in inches (matches morphology overview)
+ # Floor at 50% of width so very wide slides (e.g. brain) don't squash titles / colorbars
+ panel_height = max(panel_width * 0.5, panel_width * img_aspect)
+ fig, axes = plt.subplots(1, 2, figsize=(2 * panel_width + 2, panel_height))
+
+ # Pre-compute downsampled tile coordinates (vectorized)
+ x1_arr = np.clip(
+ (df_plot["x1"].values / downsample_factor).astype(int), 0, img_width
+ )
+ x2_arr = np.clip(
+ (df_plot["x2"].values / downsample_factor).astype(int), 0, img_width
+ )
+ y1_arr = np.clip(
+ (df_plot["y1"].values / downsample_factor).astype(int), 0, img_height
+ )
+ y2_arr = np.clip(
+ (df_plot["y2"].values / downsample_factor).astype(int), 0, img_height
+ )
+
+ def _draw_background(ax):
+ # _imshow_thumb: downsample the ~86 Mpx DAPI background to the display
+ # resolution before imshow (explicit extent is preserved, so the panel is
+ # visually identical). Cuts ~16 s/panel of full-res resampling.
+ _imshow_thumb(
+ ax,
+ small0,
+ cmap="Greys_r",
+ vmax=np.percentile(small0, 99),
+ alpha=0.5,
+ aspect="auto",
+ extent=[0, img_width, img_height, 0],
+ origin="upper",
+ )
+
+ def _fill_tile_image(values):
+ """Build a 2D float32 array with tile values; NaN = transparent."""
+ img = np.full((img_height, img_width), np.nan, dtype=np.float32)
+ for i in range(len(df_plot)):
+ if x2_arr[i] <= x1_arr[i] or y2_arr[i] <= y1_arr[i]:
+ continue
+ val = values[i]
+ if not np.isfinite(val):
+ continue
+ img[y1_arr[i] : y2_arr[i], x1_arr[i] : x2_arr[i]] = val
+ return img
+
+ def _finish_ax(ax):
+ ax.set_xlim(0, img_width)
+ ax.set_ylim(img_height, 0)
+ ax.set_aspect("equal")
+ ax.axis("off")
+
+ # ── Panel 1: neg_pct ─────────────────────────────────────────────────
+ ax = axes[0]
+ _draw_background(ax)
+
+ # Fixed colorbar (0 → 0.40) — enables cross-sample comparison and ensures
+ # WARN (0.15) / FAIL (0.30) threshold lines always render in frame.
+ # extend="max" tags tiles above 0.40 with the deepest red + a triangle marker.
+ _NEG_VMAX = 0.40
+ neg_vals = df_plot["neg_pct"].values
+
+ neg_img = _fill_tile_image(neg_vals)
+ # _imshow_thumb: NaN-safe block-mean downsample of the full-res tile overlay
+ # (tiles are large blocks, so the mean is visually identical). ~16 s -> ~0.5 s.
+ _imshow_thumb(
+ ax,
+ neg_img,
+ cmap="RdYlGn_r",
+ vmin=0,
+ vmax=_NEG_VMAX,
+ alpha=0.6,
+ aspect="auto",
+ extent=[0, img_width, img_height, 0],
+ origin="upper",
+ interpolation="nearest",
+ )
+ _finish_ax(ax)
+ ax.set_title("Negative Probe Fraction per Tile", fontsize=14)
+
+ sm = plt.cm.ScalarMappable(
+ cmap="RdYlGn_r", norm=plt.Normalize(vmin=0, vmax=_NEG_VMAX)
+ )
+ sm.set_array([])
+ cbar = plt.colorbar(sm, ax=ax, extend="max")
+ cbar.set_label("neg_pct (fraction)", fontsize=12)
+ # Threshold lines (now always in frame thanks to fixed colorbar range)
+ _neg_warn = _t.get("neg_pct_warn")
+ _neg_fail = _t.get("neg_pct_fail")
+ if isinstance(_neg_warn, (int, float)):
+ cbar.ax.axhline(
+ y=float(_neg_warn), color="orange", linewidth=1.5, linestyle="--"
+ )
+ if isinstance(_neg_fail, (int, float)):
+ cbar.ax.axhline(y=float(_neg_fail), color="red", linewidth=1.5)
+
+ # ── Panel 2: roi_tx_snr_ratio (log scale) ───────────────────────────
+ ax = axes[1]
+ _draw_background(ax)
+
+ # Display-only pseudocount: paint (real+1)/(neg+1) instead of raw
+ # real/neg. The raw ratio NaN's whenever neg=0 (divide-by-zero in
+ # snr_metrics.py) AND zeros out whenever real=0 — exactly the extreme
+ # tiles a reader most wants to see. Laplace α=1 regularises both
+ # extremes onto the colorbar without affecting the canonical
+ # roi_tx_snr_ratio column used for sample-level verdicts / JSON.
+ _PSEUDOCOUNT = 1
+ if "snr_real_tx" in df_plot.columns and "snr_neg_tx" in df_plot.columns:
+ ratio_vals = (
+ (df_plot["snr_real_tx"].astype(np.float64) + _PSEUDOCOUNT)
+ / (df_plot["snr_neg_tx"].astype(np.float64) + _PSEUDOCOUNT)
+ ).values
+ _ratio_cbar_label = "(real+1) / (neg+1) — pseudocount-smoothed"
+ _ratio_title_suffix = ", α=1 pseudocount"
+ else:
+ ratio_vals = df_plot["roi_tx_snr_ratio"].values
+ _ratio_cbar_label = "roi_tx_snr_ratio (real / neg)"
+ _ratio_title_suffix = ""
+ # Fixed log colorbar (1× → 1000×) — enables cross-sample comparison and
+ # ensures WARN (30×) / FAIL (10×) threshold lines always render in frame.
+ # extend="min" tags tiles below 1× (the no-signal regime) with the deepest
+ # red + a triangle marker. High end stays uncapped visually — very-good
+ # samples saturate at dark green, which is fine for QC purposes.
+ _RATIO_VMIN = 1.0
+ _RATIO_VMAX = 1000.0
+
+ # For LogNorm, clamp values to [vmin, vmax] range; ≤0 stays NaN
+ ratio_img = _fill_tile_image(ratio_vals)
+ # Replace non-positive finite values with NaN (LogNorm requires > 0)
+ ratio_img[ratio_img <= 0] = np.nan
+ # _imshow_thumb: NaN-safe block-mean downsample (visually identical block overlay).
+ _imshow_thumb(
+ ax,
+ ratio_img,
+ cmap="RdYlGn",
+ norm=LogNorm(vmin=_RATIO_VMIN, vmax=_RATIO_VMAX),
+ alpha=0.6,
+ aspect="auto",
+ extent=[0, img_width, img_height, 0],
+ origin="upper",
+ interpolation="nearest",
+ )
+ _finish_ax(ax)
+ ax.set_title(
+ f"Transcript SNR Ratio per Tile (log scale{_ratio_title_suffix})",
+ fontsize=14,
+ )
+
+ sm = plt.cm.ScalarMappable(
+ cmap="RdYlGn", norm=LogNorm(vmin=_RATIO_VMIN, vmax=_RATIO_VMAX)
+ )
+ sm.set_array([])
+ cbar = plt.colorbar(sm, ax=ax, extend="min")
+ cbar.set_label(_ratio_cbar_label, fontsize=12)
+ # Threshold lines (always in frame thanks to fixed colorbar range)
+ _ratio_warn = _t.get("ratio_warn")
+ _ratio_fail = _t.get("ratio_fail")
+ if isinstance(_ratio_warn, (int, float)) and float(_ratio_warn) > 0:
+ cbar.ax.axhline(
+ y=float(_ratio_warn), color="orange", linewidth=1.5, linestyle="--"
+ )
+ if isinstance(_ratio_fail, (int, float)) and float(_ratio_fail) > 0:
+ cbar.ax.axhline(y=float(_ratio_fail), color="red", linewidth=1.5)
+
+ plt.tight_layout()
+ plt.savefig(figures_dir / "snr_heatmap.png", dpi=300, bbox_inches="tight")
+ plt.close(fig)
+
+ # Save source data
+ _src_cols = [
+ "roi_id",
+ "x1",
+ "x2",
+ "y1",
+ "y2",
+ "neg_pct",
+ "roi_tx_snr_ratio",
+ "roi_tx_snr_log",
+ "tissue_coverage",
+ ]
+ df_src = df_plot[[c for c in _src_cols if c in df_plot.columns]].copy()
+ df_src["x1_ds"] = df_src["x1"] / downsample_factor
+ df_src["x2_ds"] = df_src["x2"] / downsample_factor
+ df_src["y1_ds"] = df_src["y1"] / downsample_factor
+ df_src["y2_ds"] = df_src["y2"] / downsample_factor
+ if figures_source_dir is not None:
+ df_src.to_csv(figures_source_dir / "snr_heatmap.csv", index=False)
+
+
+def plot_cross_section_concordance(
+ df_grid_roi,
+ figures_dir,
+ figures_source_dir,
+ snr_thresholds=None,
+):
+ """Cross-section concordance: image quality vs transcript SNR per tile.
+
+ Two-panel scatter showing how different quality lenses relate:
+ Left – focus_score vs roi_tx_snr_ratio (optical quality → transcript quality)
+ Right – dapi_intensity vs neg_pct (signal strength → noise contamination)
+
+ Only tissue tiles with transcripts are included.
+ """
+ required = {"focus_score", "roi_tx_snr_ratio", "neg_pct", "dapi_intensity"}
+ missing = required - set(df_grid_roi.columns)
+ if missing:
+ logging.warning(
+ "Skipping cross-section concordance plot: missing columns %s", missing
+ )
+ return
+
+ # Filter to tissue ROIs with transcripts
+ df = df_grid_roi.copy()
+ if "overlaps_tissue" in df.columns:
+ df = df[df["overlaps_tissue"]]
+ if "snr_total_tx" in df.columns:
+ df = df[df["snr_total_tx"] > 0]
+ mask = (
+ df["focus_score"].notna()
+ & df["roi_tx_snr_ratio"].notna()
+ & df["neg_pct"].notna()
+ & df["dapi_intensity"].notna()
+ )
+ df = df[mask]
+ if len(df) < 10:
+ logging.warning(
+ "Skipping cross-section concordance: too few valid tiles (%d)", len(df)
+ )
+ return
+
+ from scipy.stats import spearmanr
+
+ _t = snr_thresholds or {}
+
+ # Three stacked panels, all identical size for visual consistency.
+ # top = Focus vs TxSNR, middle = Image-SNR (Otsu) vs TxSNR, bottom =
+ # DAPI vs neg_pct. Each panel 8×4 — matches the §3.2 Focus
+ # distribution / Focus-vs-DAPI scatter dimensions for visual rhythm
+ # across §3.x figures. Middle panel falls back to a placeholder when
+ # per-tile Otsu data is unavailable (legacy samples or upstream SNR
+ # skipped).
+ fig, axes = plt.subplots(3, 1, figsize=(6, 11))
+ plt.subplots_adjust(hspace=0.35)
+
+ # --- Left panel: focus_score vs roi_tx_snr_ratio ---
+ ax = axes[0]
+ focus_vals = df["focus_score"].values
+ snr_vals = df["roi_tx_snr_ratio"].values
+
+ sidx = _subsample_idx(np.ones(len(df), dtype=bool))
+ # Color by GMM 2D classification if available
+ if "is_blurred_gmm_2d" in df.columns and df["is_blurred_gmm_2d"].notna().any():
+ colors = np.where(
+ df["is_blurred_gmm_2d"].values[sidx].astype(bool), "#E57373", "#64B5F6"
+ )
+ else:
+ colors = "#64B5F6"
+
+ ax.scatter(
+ np.log1p(focus_vals[sidx]),
+ np.log1p(snr_vals[sidx]),
+ s=8,
+ alpha=0.4,
+ c=colors,
+ edgecolors="none",
+ rasterized=True,
+ )
+ rho_fs, pval_fs = spearmanr(focus_vals, snr_vals)
+ ax.set_xlabel("log(1 + Focus Score)", fontsize=12)
+ ax.set_ylabel("log(1 + Transcript SNR Ratio)", fontsize=12)
+ ax.set_title("Optical Quality vs Transcript Quality", fontsize=13)
+ ax.text(
+ 0.05,
+ 0.95,
+ f"Spearman ρ = {rho_fs:.3f}\nn = {len(df):,} tiles",
+ transform=ax.transAxes,
+ fontsize=11,
+ verticalalignment="top",
+ bbox=dict(boxstyle="round", facecolor="wheat", alpha=0.7),
+ )
+ # Threshold lines
+ ratio_warn = float(_t.get("ratio_warn", 3.0))
+ ratio_fail = float(_t.get("ratio_fail", 1.5))
+ ax.axhline(
+ np.log1p(ratio_warn),
+ ls="--",
+ lw=1,
+ color="orange",
+ alpha=0.8,
+ label=f"WARN = {ratio_warn}",
+ )
+ ax.axhline(
+ np.log1p(ratio_fail),
+ ls="--",
+ lw=1,
+ color="red",
+ alpha=0.8,
+ label=f"FAIL = {ratio_fail}",
+ )
+ ax.legend(fontsize=9, loc="lower right")
+ ax.grid(True, ls="--", alpha=0.3)
+
+ # --- Middle panel: Image SNR (Otsu) per-tile vs roi_tx_snr_ratio (NEW) ---
+ # Direct test of the section's premise: does per-tile image SNR predict
+ # per-tile transcript SNR? Otsu-split dB is the per-tile Image-SNR axis
+ # (also surfaced as a slide-level metric in §3.4 Detailed metrics).
+ ax = axes[1]
+ rho_o = float("nan")
+ n_valid_otsu = 0
+ if "snr_image_otsu_db" in df.columns and df["snr_image_otsu_db"].notna().any():
+ _otsu_mask = df["snr_image_otsu_db"].notna() & df["roi_tx_snr_ratio"].notna()
+ n_valid_otsu = int(_otsu_mask.sum())
+ if n_valid_otsu >= 10:
+ otsu_db = df["snr_image_otsu_db"].values
+ snr_ratio_arr = df["roi_tx_snr_ratio"].values
+ sidx_o = _subsample_idx(_otsu_mask.values)
+ if (
+ "is_blurred_gmm_2d" in df.columns
+ and df["is_blurred_gmm_2d"].notna().any()
+ ):
+ colors_o = np.where(
+ df["is_blurred_gmm_2d"].values[sidx_o].astype(bool),
+ "#E57373",
+ "#64B5F6",
+ )
+ else:
+ colors_o = "#64B5F6"
+ ax.scatter(
+ otsu_db[sidx_o],
+ np.log1p(snr_ratio_arr[sidx_o]),
+ s=8,
+ alpha=0.4,
+ c=colors_o,
+ edgecolors="none",
+ rasterized=True,
+ )
+ rho_o, _ = spearmanr(otsu_db[_otsu_mask], snr_ratio_arr[_otsu_mask])
+ ax.set_xlabel("Image SNR — Otsu split (dB)", fontsize=12)
+ ax.set_ylabel("log(1 + Transcript SNR Ratio)", fontsize=12)
+ ax.set_title("Image SNR vs Transcript Quality", fontsize=13)
+ ax.text(
+ 0.05,
+ 0.95,
+ f"Spearman ρ = {rho_o:.3f}\nn = {n_valid_otsu:,} tiles",
+ transform=ax.transAxes,
+ fontsize=11,
+ verticalalignment="top",
+ bbox=dict(boxstyle="round", facecolor="wheat", alpha=0.7),
+ )
+ # Reuse Transcript-SNR ratio thresholds from the left panel —
+ # both panels share the same y-axis metric.
+ ax.axhline(
+ np.log1p(ratio_warn),
+ ls="--",
+ lw=1,
+ color="orange",
+ alpha=0.8,
+ label=f"WARN = {ratio_warn}",
+ )
+ ax.axhline(
+ np.log1p(ratio_fail),
+ ls="--",
+ lw=1,
+ color="red",
+ alpha=0.8,
+ label=f"FAIL = {ratio_fail}",
+ )
+ ax.legend(fontsize=9, loc="lower right")
+ ax.grid(True, ls="--", alpha=0.3)
+ else:
+ ax.text(
+ 0.5,
+ 0.5,
+ "Image SNR (Otsu) per-tile data\nunavailable (< 10 valid tiles)",
+ transform=ax.transAxes,
+ ha="center",
+ va="center",
+ fontsize=12,
+ )
+ ax.set_xticks([])
+ ax.set_yticks([])
+ else:
+ ax.text(
+ 0.5,
+ 0.5,
+ "Image SNR (Otsu) per-tile data\nunavailable (legacy sample or SNR not computed)",
+ transform=ax.transAxes,
+ ha="center",
+ va="center",
+ fontsize=12,
+ )
+ ax.set_xticks([])
+ ax.set_yticks([])
+
+ # --- Right panel: dapi_intensity vs neg_pct ---
+ ax = axes[2]
+ int_vals = df["dapi_intensity"].values
+ neg_vals = df["neg_pct"].values
+
+ if "is_blurred_gmm_2d" in df.columns and df["is_blurred_gmm_2d"].notna().any():
+ colors_r = np.where(
+ df["is_blurred_gmm_2d"].values[sidx].astype(bool), "#E57373", "#64B5F6"
+ )
+ else:
+ colors_r = "#64B5F6"
+
+ ax.scatter(
+ np.log1p(int_vals[sidx]),
+ neg_vals[sidx],
+ s=8,
+ alpha=0.4,
+ c=colors_r,
+ edgecolors="none",
+ rasterized=True,
+ )
+ rho_in, pval_in = spearmanr(int_vals, neg_vals)
+ ax.set_xlabel("log(1 + DAPI Intensity)", fontsize=12)
+ ax.set_ylabel("Negative Probe Fraction", fontsize=12)
+ ax.set_title("Signal Strength vs Noise Contamination", fontsize=13)
+ ax.text(
+ 0.05,
+ 0.95,
+ f"Spearman ρ = {rho_in:.3f}\nn = {len(df):,} tiles",
+ transform=ax.transAxes,
+ fontsize=11,
+ verticalalignment="top",
+ bbox=dict(boxstyle="round", facecolor="wheat", alpha=0.7),
+ )
+ neg_warn = float(_t.get("neg_pct_warn", 0.15))
+ neg_fail = float(_t.get("neg_pct_fail", 0.30))
+ ax.axhline(
+ neg_warn, ls="--", lw=1, color="orange", alpha=0.8, label=f"WARN = {neg_warn}"
+ )
+ ax.axhline(
+ neg_fail, ls="--", lw=1, color="red", alpha=0.8, label=f"FAIL = {neg_fail}"
+ )
+ ax.legend(fontsize=9, loc="lower right")
+ ax.grid(True, ls="--", alpha=0.3)
+
+ # Add GMM legend if coloured
+ if "is_blurred_gmm_2d" in df.columns and df["is_blurred_gmm_2d"].notna().any():
+ from matplotlib.patches import Patch
+
+ for a in axes:
+ handles = a.get_legend_handles_labels()[0]
+ handles.extend(
+ [
+ Patch(facecolor="#64B5F6", label="In-Focus (GMM)"),
+ Patch(facecolor="#E57373", label="Blurred (GMM)"),
+ ]
+ )
+ a.legend(
+ handles=handles,
+ fontsize=8,
+ loc="lower right" if a is axes[0] else "upper right",
+ )
+
+ plt.tight_layout()
+ plt.savefig(figures_dir / "cross_section_concordance.png", dpi=200)
+ plt.savefig(
+ figures_dir / "cross_section_concordance.pdf", dpi=200, bbox_inches="tight"
+ )
+ plt.close(fig)
+
+ # Source data
+ src_cols = [
+ "roi_id",
+ "focus_score",
+ "dapi_intensity",
+ "roi_tx_snr_ratio",
+ "neg_pct",
+ "snr_image_otsu_db",
+ ]
+ if "is_blurred_gmm_2d" in df.columns:
+ src_cols.append("is_blurred_gmm_2d")
+ if "tissue_coverage" in df.columns:
+ src_cols.append("tissue_coverage")
+ if figures_source_dir is not None:
+ df[[c for c in src_cols if c in df.columns]].to_csv(
+ figures_source_dir / "cross_section_concordance.csv", index=False
+ )
+ logging.info(
+ "Cross-section concordance: focus-vs-SNR ρ=%.3f, otsu-vs-SNR ρ=%.3f (n=%d), intensity-vs-neg ρ=%.3f (%d tiles)",
+ rho_fs,
+ rho_o,
+ n_valid_otsu,
+ rho_in,
+ len(df),
+ )
+
+
+def plot_roi_focus_vs_intensity(
+ df_grid_roi, figures_dir, figures_source_dir, threshold=-1.0
+):
+ """
+ Create scatter plot showing focus score vs raw DAPI intensity for tile threshold analysis.
+ Uses log scale for intensity to better visualize the relationship.
+
+ Parameters:
+ -----------
+ df_grid_roi : pandas DataFrame
+ Grid ROI DataFrame with columns: focus_score, focus_score_norm, raw_intensity (or dapi_intensity)
+ figures_dir : Path
+ Directory to save figures
+ figures_source_dir : Path
+ Directory to save source data
+ threshold : float, optional
+ Threshold for normalized focus score (default: -1.0)
+ """
+ # Use dapi_intensity if available, otherwise raw_intensity
+ intensity_col = (
+ "dapi_intensity" if "dapi_intensity" in df_grid_roi.columns else "raw_intensity"
+ )
+
+ # Calculate log10 of intensity (add small epsilon to avoid log(0))
+ intensity_values = df_grid_roi[intensity_col].values
+ log_intensity = np.log10(
+ intensity_values + 1e-10
+ ) # Add small epsilon to handle any zeros
+
+ # Phase v5: robust outlier handling for the rendered scatter.
+ # (A) Filter to tissue tiles (tissue_coverage > 0.5) — empty / non-tissue
+ # tiles have intensity ≈ 0 → log10(1e-10) = -10, which dominates the
+ # auto-scaled xlim. CSV still writes unfiltered data.
+ # (B) Compute 1st-99th percentile xlim bounds as a robustness floor so a
+ # single saturated tissue tile (fold, debris) doesn't compress the rest.
+ if "tissue_coverage" in df_grid_roi.columns:
+ _tissue_mask = df_grid_roi["tissue_coverage"].values > 0.5
+ else:
+ _tissue_mask = np.ones(len(df_grid_roi), dtype=bool)
+
+ # Single-panel figure: normalized focus score vs log DAPI intensity,
+ # coloured by GMM 2D blurry/in-focus classification. Figsize matches the
+ # Focus score distribution histogram (figsize=(8, 4)) below for visual
+ # rhythm. Previously this was a 2-panel figure where the left panel
+ # showed raw CCFS vs intensity coloured by *normalised* focus score —
+ # redundant with the right panel since the y-axis there directly
+ # encodes the same metric.
+ fig, ax = plt.subplots(1, 1, figsize=(7, 4))
+ # Filter out invalid values
+ valid_mask_plot2 = (
+ np.isfinite(log_intensity)
+ & np.isfinite(df_grid_roi["focus_score_norm"].values)
+ & _tissue_mask
+ )
+
+ if valid_mask_plot2.sum() == 0:
+ logging.warning(" Warning: No valid data points for plot 2")
+ ax.text(
+ 0.5,
+ 0.5,
+ "No valid data",
+ transform=ax.transAxes,
+ ha="center",
+ va="center",
+ fontsize=14,
+ )
+ else:
+ # Color by GMM 2D classification if available, otherwise fall back to threshold
+ has_gmm_2d = "is_blurred_gmm_2d" in df_grid_roi.columns
+
+ if has_gmm_2d:
+ # Color by GMM 2D classification
+ valid_gmm_mask = valid_mask_plot2 & df_grid_roi["is_blurred_gmm_2d"].notna()
+ if valid_gmm_mask.sum() > 0:
+ sidx = _subsample_idx(valid_gmm_mask)
+ is_blurred_sub = (
+ df_grid_roi["is_blurred_gmm_2d"].values[sidx].astype(bool)
+ )
+ colors = np.where(is_blurred_sub, "red", "blue")
+ ax.scatter(
+ log_intensity[sidx],
+ df_grid_roi["focus_score_norm"].values[sidx],
+ alpha=0.5,
+ s=10,
+ c=colors,
+ edgecolors="none",
+ rasterized=True,
+ )
+
+ # Phase v5: percentile-clipped xlim (see comment at top of function).
+ _x_lo, _x_hi = np.nanpercentile(log_intensity[valid_gmm_mask], [1, 99])
+ ax.set_xlim(_x_lo - 0.1, _x_hi + 0.1)
+ ax.set_ylim(
+ df_grid_roi.loc[valid_gmm_mask, "focus_score_norm"].min() - 0.5,
+ df_grid_roi.loc[valid_gmm_mask, "focus_score_norm"].max() + 0.5,
+ )
+
+ # Add text annotation for GMM 2D (counts from full data, not subsample)
+ n_blurred = df_grid_roi.loc[valid_gmm_mask, "is_blurred_gmm_2d"].sum()
+ n_in_focus = (
+ ~df_grid_roi.loc[valid_gmm_mask, "is_blurred_gmm_2d"]
+ ).sum()
+ pct_blurred = (
+ n_blurred / valid_gmm_mask.sum() * 100
+ if valid_gmm_mask.sum() > 0
+ else 0
+ )
+ ax.text(
+ 0.05,
+ 0.95,
+ f"2D GMM Classification\nRed: Blurred ({n_blurred:,}, {pct_blurred:.1f}%)\nBlue: In-Focus ({n_in_focus:,}, {100 - pct_blurred:.1f}%)",
+ transform=ax.transAxes,
+ fontsize=11,
+ verticalalignment="top",
+ bbox=dict(boxstyle="round", facecolor="wheat", alpha=0.7),
+ )
+ ax.set_title(
+ "Normalized Focus Score vs Log10(Raw DAPI Intensity) (2D GMM Classification)",
+ fontsize=12,
+ )
+ else:
+ # Fallback to threshold method
+ sidx = _subsample_idx(valid_mask_plot2)
+ scores_sub = df_grid_roi["focus_score_norm"].values[sidx]
+ colors = np.where(scores_sub <= threshold, "red", "blue")
+ ax.scatter(
+ log_intensity[sidx],
+ scores_sub,
+ alpha=0.5,
+ s=10,
+ c=colors,
+ edgecolors="none",
+ rasterized=True,
+ )
+
+ # Phase v5: percentile-clipped xlim (see comment at top of function).
+ _x_lo, _x_hi = np.nanpercentile(log_intensity[valid_mask_plot2], [1, 99])
+ ax.set_xlim(_x_lo - 0.1, _x_hi + 0.1)
+ ax.set_ylim(
+ df_grid_roi.loc[valid_mask_plot2, "focus_score_norm"].min() - 0.5,
+ df_grid_roi.loc[valid_mask_plot2, "focus_score_norm"].max() + 0.5,
+ )
+
+ # Add threshold line
+ ax.axhline(
+ y=threshold,
+ color="black",
+ linestyle="--",
+ linewidth=2,
+ label=f"Threshold ({threshold})",
+ )
+
+ # Add text annotation for threshold
+ ax.text(
+ 0.05,
+ 0.95,
+ f"Threshold: {threshold}\nRed: Blurred (≤{threshold})\nBlue: In-Focus (>{threshold})",
+ transform=ax.transAxes,
+ fontsize=11,
+ verticalalignment="top",
+ bbox=dict(boxstyle="round", facecolor="wheat", alpha=0.7),
+ )
+ ax.set_title(
+ "Normalized Focus Score vs Log10(Raw DAPI Intensity) (Thresholded)",
+ fontsize=12,
+ )
+
+ ax.set_xlabel("log₁₀(Raw DAPI Intensity, 16-bit counts)", fontsize=12)
+ ax.set_ylabel("Normalised focus score", fontsize=12)
+ ax.grid(True, axis="y", linestyle="--", alpha=0.7)
+ ax.legend(fontsize=10)
+
+ # Pin axes-box position so the rendered plot box matches the §3.2 Focus
+ # score distribution figure exactly. Both figures use figsize=(8, 4) and
+ # the same subplots_adjust margins; identical (left, right, top, bottom)
+ # ⇒ identical plot-box position regardless of y-tick label width.
+ plt.subplots_adjust(left=0.13, right=0.95, top=0.88, bottom=0.18)
+ plt.savefig(
+ figures_dir / "roi_focus_vs_intensity.pdf", dpi=300, bbox_inches="tight"
+ )
+ plt.savefig(figures_dir / "roi_focus_vs_intensity.png", dpi=300)
+ plt.close(fig)
+
+ # Save data as CSV (include both raw and log intensity)
+ df_scatter = df_grid_roi[
+ ["focus_score", "focus_score_norm", intensity_col, "tissue_coverage"]
+ ].copy()
+ df_scatter["log10_intensity"] = log_intensity
+ df_scatter["is_low_nuclear_texture"] = df_scatter["focus_score_norm"] <= threshold
+ if figures_source_dir is not None:
+ df_scatter.to_csv(
+ figures_source_dir / "roi_focus_vs_intensity.csv", index=False
+ )
+
+ # Print correlation statistics (using log intensity)
+ correlation = np.corrcoef(log_intensity, df_grid_roi["focus_score_norm"])[0, 1]
+ logging.info(
+ f" Correlation (log10(intensity) vs normalized focus score): {correlation:.4f}"
+ )
+
+
+def plot_roi_focus_distribution(
+ df_grid_roi, figures_dir, figures_source_dir, threshold=-1.0
+):
+ """Histogram of normalized focus scores on tissue-filtered tiles, with
+ GMM 2D binary classification overlaid (red = blurred, blue = in-focus).
+
+ Phase v5 (2026-05-15): simplified from the previous dual-panel (raw +
+ normalized) × dual-variant (all-tiles + tissue-filtered) layout. The
+ all-tiles view was misleading — background tiles are force-classified
+ blurry by the intensity-floor rule, so the all-tiles histogram showed
+ a "blurry mass" that was not actually a GMM decision. The raw-score
+ panel was redundant with the normalized panel for QC interpretation.
+ Single panel: tissue-filtered, normalized scores, GMM-colored.
+
+ Emits roi_focus_distribution_tissue.png — only when tissue_coverage
+ column is present with at least one tile above the 0.5 threshold.
+
+ Parameters:
+ -----------
+ df_grid_roi : pandas DataFrame
+ Grid ROI DataFrame with columns: focus_score, focus_score_norm,
+ is_blurred_gmm_2d, tissue_coverage (required).
+ figures_dir : Path
+ Directory to save figures.
+ figures_source_dir : Path
+ Directory to save source data.
+ threshold : float, optional
+ Normalized-score fallback threshold (default: -1.0). Used only when
+ is_blurred_gmm_2d column is absent.
+ """
+ if "tissue_coverage" not in df_grid_roi.columns:
+ logging.info(" No tissue_coverage column; skipping focus score distribution.")
+ return
+
+ df_in = df_grid_roi[df_grid_roi["tissue_coverage"] > 0.5]
+ if len(df_in) == 0:
+ logging.info(
+ " No tissue tiles (tissue_coverage > 0.5); skipping focus score distribution."
+ )
+ return
+
+ save_stem = "roi_focus_distribution_tissue"
+ # figsize chosen to be more landscape-y than the previous (10, 6) so the
+ # histogram doesn't dominate the §3.2 visual flow against the adjacent
+ # spatial focus heatmap. Width reduced ~20%, height reduced ~33%.
+ fig, ax = plt.subplots(figsize=(7, 4))
+
+ has_gmm_2d = "is_blurred_gmm_2d" in df_in.columns
+ if has_gmm_2d:
+ blurred = df_in[df_in["is_blurred_gmm_2d"]]
+ in_focus = df_in[~df_in["is_blurred_gmm_2d"]]
+ title_suffix = "2D GMM Classification"
+ else:
+ blurred = df_in[df_in["focus_score_norm"] <= threshold]
+ in_focus = df_in[df_in["focus_score_norm"] > threshold]
+ title_suffix = f"Threshold: {threshold}"
+
+ ax.hist(
+ [blurred["focus_score_norm"], in_focus["focus_score_norm"]],
+ bins=50,
+ alpha=0.7,
+ edgecolor="black",
+ color=["red", "blue"],
+ label=["Blurred", "In-Focus"],
+ stacked=False,
+ )
+
+ if not has_gmm_2d:
+ ax.axvline(
+ x=threshold,
+ color="black",
+ linestyle="--",
+ linewidth=2,
+ label=f"Threshold ({threshold})",
+ )
+
+ ax.set_xlabel("Focus Score (Normalized)", fontsize=12)
+ ax.set_ylabel("Number of tiles", fontsize=12)
+ ax.set_title(
+ f"Distribution of Normalized Focus Scores ({title_suffix})",
+ fontsize=12,
+ )
+ ax.legend(fontsize=10)
+ ax.grid(True, axis="y", linestyle="--", alpha=0.7)
+
+ mean_norm = df_in["focus_score_norm"].mean()
+ median_norm = df_in["focus_score_norm"].median()
+ pct_blurred = len(blurred) / len(df_in) * 100 if len(df_in) > 0 else 0
+ ax.text(
+ 0.95,
+ 0.95,
+ f"Mean: {mean_norm:.4f}\nMedian: {median_norm:.4f}\nBlurred: {pct_blurred:.1f}%\nn = {len(df_in):,} tiles",
+ transform=ax.transAxes,
+ verticalalignment="top",
+ horizontalalignment="right",
+ bbox=dict(boxstyle="round", facecolor="wheat", alpha=0.7),
+ fontsize=10,
+ )
+
+ # Pin axes-box position so the rendered plot box matches the §3.2 Focus
+ # score vs DAPI intensity figure exactly (same figsize, same margins).
+ plt.subplots_adjust(left=0.13, right=0.95, top=0.88, bottom=0.18)
+ plt.savefig(figures_dir / f"{save_stem}.pdf", dpi=300, bbox_inches="tight")
+ plt.savefig(figures_dir / f"{save_stem}.png", dpi=300)
+ plt.close(fig)
+
+ # Source CSV mirrors the rendered subset.
+ df_dist = df_in[["focus_score", "focus_score_norm"]].copy()
+ df_dist["is_low_nuclear_texture"] = df_dist["focus_score_norm"] <= threshold
+ if figures_source_dir is not None:
+ df_dist.to_csv(figures_source_dir / f"{save_stem}.csv", index=False)
+
+ logging.info(" Focus score distribution summary (tissue tiles):")
+ logging.info(
+ f" Normalized focus score - Mean: {mean_norm:.4f}, Median: {median_norm:.4f}"
+ )
+ logging.info(f" Tiles blurred: {len(blurred)} ({pct_blurred:.1f}%)")
+ logging.info(f" Tiles in-focus: {len(in_focus)} ({100 - pct_blurred:.1f}%)")
+
+
+def calculate_roi_intensities(xoa_morphology_files, df_grid_roi):
+ """
+ Calculate mean intensity per tile for Boundary and IntRNA channels.
+
+ Note: If intensities are already calculated in calculate_roi_focusscore() (new behavior),
+ this function will detect that and return the DataFrame unchanged.
+
+ Parameters:
+ -----------
+ xoa_morphology_files : list
+ List of paths to morphology image files
+ df_grid_roi : pandas DataFrame
+ Grid ROI DataFrame with columns: roi_id, x1, x2, y1, y2, raw_intensity (DAPI)
+ If dapi_intensity, boundary_intensity, intrna_intensity already exist, returns unchanged.
+
+ Returns:
+ --------
+ pandas DataFrame
+ DataFrame with columns:
+ - dapi_intensity: Mean DAPI intensity per tile
+ - boundary_intensity: Mean boundary intensity per tile (if available)
+ - intrna_intensity: Mean IntRNA intensity per tile (if available)
+ """
+
+ # Check if intensities are already calculated (new behavior in calculate_roi_focusscore)
+ # Note: boundary_intensity may be NaN if only DAPI channel is available, but column should exist
+ if "dapi_intensity" in df_grid_roi.columns:
+ # Intensities already calculated, just return the DataFrame
+ logging.info(
+ " Intensities already calculated in calculate_roi_focusscore(), skipping recalculation"
+ )
+ return df_grid_roi.copy()
+
+ # Legacy behavior: calculate intensities if not already present
+ # Load channels robustly from multi-channel stack or split channel files.
+ dapi_image, boundary_image, intrna_image = _load_morphology_channels(
+ xoa_morphology_files, level=0
+ )
+ has_boundary = boundary_image is not None
+ has_intrna = intrna_image is not None
+
+ # Create output DataFrame
+ df_roi_intensities = df_grid_roi.copy()
+ if "dapi_intensity" not in df_roi_intensities.columns:
+ df_roi_intensities["dapi_intensity"] = df_roi_intensities.get(
+ "raw_intensity", np.nan
+ ) # Rename for consistency
+
+ # Calculate intensities for each ROI
+ n_rois = len(df_grid_roi)
+
+ # Pre-extract ROI coordinate arrays for vectorized access
+ x1_arr = df_grid_roi["x1"].values.astype(int)
+ x2_arr = df_grid_roi["x2"].values.astype(int)
+ y1_arr = df_grid_roi["y1"].values.astype(int)
+ y2_arr = df_grid_roi["y2"].values.astype(int)
+
+ # Boundary intensity
+ if has_boundary:
+ boundary_intensities = np.empty(n_rois, dtype=np.float64)
+ for i in range(n_rois):
+ boundary_intensities[i] = np.mean(
+ boundary_image[y1_arr[i] : y2_arr[i], x1_arr[i] : x2_arr[i]]
+ )
+ df_roi_intensities["boundary_intensity"] = boundary_intensities
+ del boundary_image
+ else:
+ df_roi_intensities["boundary_intensity"] = np.nan
+ logging.warning(
+ "Warning: Boundary channel not available, setting boundary_intensity to NaN"
+ )
+
+ # IntRNA intensity
+ if has_intrna:
+ intrna_intensities = np.empty(n_rois, dtype=np.float64)
+ for i in range(n_rois):
+ intrna_intensities[i] = np.mean(
+ intrna_image[y1_arr[i] : y2_arr[i], x1_arr[i] : x2_arr[i]]
+ )
+ df_roi_intensities["intrna_intensity"] = intrna_intensities
+ else:
+ df_roi_intensities["intrna_intensity"] = np.nan
+ logging.warning(
+ "Warning: IntRNA channel not available, setting intrna_intensity to NaN"
+ )
+
+ # Clean up
+ del dapi_image
+
+ return df_roi_intensities
+
+
+def assess_raw_intensity_quality(
+ df_roi_intensities,
+ dapi_threshold_critical=500,
+ boundary_threshold_critical=100,
+ intrna_threshold_critical=300,
+ min_tissue_coverage=ROI_MIN_TISSUE_COVERAGE_FOR_INTENSITY_QC,
+ channel_pct_thresholds=None,
+):
+ """
+ Assess raw intensity quality from per-tile intensities.
+ Handles missing channels gracefully (NaN values).
+
+ Note: This function can apply an additional tissue coverage filter before
+ intensity QC. If ``tissue_coverage`` is present, only tiles with
+ ``tissue_coverage >= min_tissue_coverage`` are used for threshold-based
+ percentage calculations. This reduces false FAILs from low-content/background
+ edge tiles that technically overlap tissue but are mostly non-tissue.
+
+ Threshold Method:
+ -----------------
+ Each channel has a single absolute minimum intensity threshold (``intensity_critical``
+ from YAML). The metric is the fraction of tissue tiles whose mean intensity falls
+ below that threshold: ``pct_tissue_roi_below_critical``.
+
+ Critical thresholds (defaults):
+ - DAPI: 500 — ~0.76% of 16-bit max; typical well-stained range 2000-8000
+ - Boundary: 100 — membrane markers 5-10× lower than DAPI
+ - IntRNA: 300 — rRNA markers 2-3× lower than DAPI
+
+ Quality Status Determination (WARN-only since 2026-06-22):
+ ----------------------------------------------------------
+ Per-channel verdict is driven by ``pct_tissue_roi_below_critical`` compared
+ against the YAML ``intensity_warn`` fraction (converted to %). There is no
+ FAIL tier — intensity does not track quality after XOA 4.0, so a dim sample
+ is flagged for review, never hard-failed on intensity alone:
+ - **warn**: pct_tissue_roi_below_critical > warn% OR mean < critical_threshold
+ - **pass**: otherwise
+ The ``critical_threshold`` is XOA-version-specific (selected by the caller).
+
+ Parameters:
+ -----------
+ df_roi_intensities : pandas DataFrame
+ DataFrame with per-tile intensities (dapi_intensity, boundary_intensity, intrna_intensity).
+ dapi_threshold_critical : float, optional
+ Critical DAPI threshold (default: 500).
+ boundary_threshold_critical : float, optional
+ Critical boundary threshold (default: 100).
+ intrna_threshold_critical : float, optional
+ Critical IntRNA threshold (default: 300).
+
+ Returns:
+ --------
+ dict
+ Per-channel dict with:
+ - mean, median, p10, p25, p75, p90: Intensity statistics
+ - critical_threshold: Absolute minimum intensity threshold
+ - n_tissue_rois_below_critical: Count of tissue tiles below threshold
+ - pct_tissue_roi_below_critical: Percentage of tissue tiles below threshold
+ - pct_warn_threshold, pct_fail_threshold: YAML-derived population thresholds (%)
+ - quality_status: 'pass', 'warn', 'fail', or 'not_available'
+ """
+ stats = {}
+ _cpct = channel_pct_thresholds or {}
+ df_work = df_roi_intensities
+ n_rois_input = int(len(df_roi_intensities))
+ if "tissue_coverage" in df_roi_intensities.columns:
+ df_work = df_roi_intensities[
+ df_roi_intensities["tissue_coverage"] >= float(min_tissue_coverage)
+ ]
+ n_rois_used = int(len(df_work))
+ stats["_intensity_qc_scope"] = {
+ "n_rois_input": n_rois_input,
+ "n_rois_used": n_rois_used,
+ "min_tissue_coverage": float(min_tissue_coverage),
+ "uses_tissue_coverage_filter": "tissue_coverage" in df_roi_intensities.columns,
+ }
+
+ # YAML key → function-local name mapping
+ _yaml_keys = {"dapi": "DAPI", "boundary": "boundary", "intrna": "intRNA"}
+
+ for channel, critical_threshold in [
+ ("dapi", dapi_threshold_critical),
+ ("boundary", boundary_threshold_critical),
+ ("intrna", intrna_threshold_critical),
+ ]:
+ # Per-channel WARN fraction from YAML (or default). WARN-only since
+ # 2026-06-22 (qc_drift_analysis): intensity does not track quality
+ # post-XOA-4.0, so there is no FAIL tier — `intensity_fail` is no longer
+ # read or applied.
+ _ch = _cpct.get(_yaml_keys.get(channel, channel)) or {}
+ pct_warn_frac = float(_ch.get("intensity_warn", 0.15))
+ col_name = f"{channel}_intensity"
+ intensities = df_work[col_name].values
+
+ # Check if channel is available (not all NaN)
+ if np.all(np.isnan(intensities)):
+ stats[channel] = {
+ "mean": np.nan,
+ "median": np.nan,
+ "p10": np.nan,
+ "p25": np.nan,
+ "p75": np.nan,
+ "p90": np.nan,
+ "critical_threshold": float(critical_threshold),
+ "n_tissue_rois_below_critical": 0,
+ "pct_tissue_roi_below_critical": 0.0,
+ "quality_status": "not_available",
+ }
+ continue
+
+ # Filter out NaN values for calculations
+ intensities_valid = intensities[~np.isnan(intensities)]
+
+ if len(intensities_valid) == 0:
+ stats[channel] = {
+ "mean": np.nan,
+ "median": np.nan,
+ "p10": np.nan,
+ "p25": np.nan,
+ "p75": np.nan,
+ "p90": np.nan,
+ "critical_threshold": float(critical_threshold),
+ "n_tissue_rois_below_critical": 0,
+ "pct_tissue_roi_below_critical": 0.0,
+ "quality_status": "not_available",
+ }
+ continue
+
+ # Calculate statistics (using valid values only)
+ mean_int = np.mean(intensities_valid)
+ median_int = np.median(intensities_valid)
+ p10 = np.percentile(intensities_valid, 10)
+ p25 = np.percentile(intensities_valid, 25)
+ p75 = np.percentile(intensities_valid, 75)
+ p90 = np.percentile(intensities_valid, 90)
+
+ # Count tissue tiles below the single critical intensity threshold
+ n_below = int(np.sum(intensities_valid < critical_threshold))
+ pct_tissue_roi_below_critical = (n_below / len(intensities_valid)) * 100
+
+ # Determine quality status — WARN-only (no FAIL tier). Both the
+ # prevalence case (too many tiles below the floor) and the mean-below-floor
+ # case cap at WARN: a dim sample is flagged for review, never hard-failed
+ # on intensity alone.
+ pct_warn = pct_warn_frac * 100.0
+ if pct_tissue_roi_below_critical > pct_warn or mean_int < critical_threshold:
+ quality_status = "warn"
+ else:
+ quality_status = "pass"
+
+ stats[channel] = {
+ "mean": float(mean_int),
+ "median": float(median_int),
+ "p10": float(p10),
+ "p25": float(p25),
+ "p75": float(p75),
+ "p90": float(p90),
+ "critical_threshold": float(critical_threshold),
+ "pct_warn_threshold": float(pct_warn),
+ # None signals "advisory / no FAIL tier" to the report renderer.
+ "pct_fail_threshold": None,
+ "n_tissue_rois_below_critical": n_below,
+ "pct_tissue_roi_below_critical": float(pct_tissue_roi_below_critical),
+ "quality_status": quality_status,
+ }
+
+ # Overall quality — WARN-only (intensity never fails the sample on its own).
+ statuses = [stats[ch]["quality_status"] for ch in ["dapi", "boundary", "intrna"]]
+ if n_rois_used == 0:
+ overall_quality = "not_available"
+ elif "warn" in statuses:
+ overall_quality = "warn"
+ else:
+ overall_quality = "pass"
+
+ stats["overall_quality"] = overall_quality
+
+ return stats
+
+
+def plot_focus_score_vs_laplacian(df_grid_roi, figures_dir, figures_source_dir):
+ """
+ Compare Laplacian variance vs original focus score (std²/mean).
+
+ Creates scatter plot and distribution histograms comparing the two focus metrics.
+
+ Parameters:
+ -----------
+ df_grid_roi : pandas DataFrame
+ Grid ROI DataFrame with columns: dapi_focus_score, dapi_lap_var
+ figures_dir : Path
+ Directory to save figures
+ figures_source_dir : Path
+ Directory to save source data
+ """
+ # Check if required columns exist
+ if "dapi_lap_var" not in df_grid_roi.columns:
+ logging.warning(
+ " Warning: dapi_lap_var column not found, skipping Laplacian comparison plots"
+ )
+ return
+
+ if "dapi_focus_score" not in df_grid_roi.columns:
+ # Fallback to generic focus_score
+ if "focus_score" not in df_grid_roi.columns:
+ logging.warning(
+ " Warning: No focus score column found, skipping Laplacian comparison plots"
+ )
+ return
+ focus_col = "focus_score"
+ else:
+ focus_col = "dapi_focus_score"
+
+ # Filter out NaN values
+ valid_mask = df_grid_roi["dapi_lap_var"].notna() & df_grid_roi[focus_col].notna()
+ df_valid = df_grid_roi[valid_mask].copy()
+
+ if len(df_valid) == 0:
+ logging.warning(
+ " Warning: No valid data for Laplacian comparison, skipping plots"
+ )
+ return
+
+ # Scatter plot: log1p(focus_score) vs log1p(laplacian_variance)
+ logging.info("Generating Figure: Focus score vs Laplacian variance comparison...")
+ fig, ax = plt.subplots(1, 1, figsize=(6, 5))
+
+ # Color by GMM 2D classification if available
+ has_gmm_2d = "is_blurred_gmm_2d" in df_valid.columns
+ focus_vals = df_valid[focus_col].values
+ lap_vals = df_valid["dapi_lap_var"].values
+ if has_gmm_2d:
+ # Filter to valid mask for GMM 2D
+ valid_gmm_mask = df_valid["is_blurred_gmm_2d"].notna().values
+ if valid_gmm_mask.sum() > 0:
+ sidx = _subsample_idx(valid_gmm_mask)
+ is_blurred_sub = df_valid["is_blurred_gmm_2d"].values[sidx].astype(bool)
+ colors = np.where(is_blurred_sub, "red", "blue")
+ ax.scatter(
+ np.log1p(focus_vals[sidx]),
+ np.log1p(lap_vals[sidx]),
+ s=5,
+ alpha=0.3,
+ c=colors,
+ rasterized=True,
+ )
+ from matplotlib.patches import Patch
+
+ legend_elements = [
+ Patch(facecolor="blue", label="In-Focus (2D GMM)"),
+ Patch(facecolor="red", label="Blurred (2D GMM)"),
+ ]
+ ax.legend(handles=legend_elements, fontsize=10)
+ else:
+ sidx = _subsample_idx(np.ones(len(df_valid), dtype=bool))
+ ax.scatter(
+ np.log1p(focus_vals[sidx]),
+ np.log1p(lap_vals[sidx]),
+ s=5,
+ alpha=0.3,
+ rasterized=True,
+ )
+ else:
+ sidx = _subsample_idx(np.ones(len(df_valid), dtype=bool))
+ ax.scatter(
+ np.log1p(focus_vals[sidx]),
+ np.log1p(lap_vals[sidx]),
+ s=5,
+ alpha=0.3,
+ rasterized=True,
+ )
+
+ ax.set_xlabel("log(1 + std²/mean) (DAPI)", fontsize=12)
+ ax.set_ylabel("log(1 + Laplacian variance) (DAPI)", fontsize=12)
+ if has_gmm_2d:
+ ax.set_title(
+ "Focus Score vs Laplacian Variance (2D GMM Classification)", fontsize=14
+ )
+ else:
+ ax.set_title("Focus Score vs Laplacian Variance", fontsize=14)
+ ax.grid(True, alpha=0.3)
+
+ plt.tight_layout()
+ # Save to methodology assessment folder (subfolder of figures_dir)
+ methodology_figures_dir = figures_dir / "figures_methodology_assessment"
+ methodology_figures_dir.mkdir(parents=True, exist_ok=True)
+ plt.savefig(
+ methodology_figures_dir / "focus_score_vs_laplacian.pdf",
+ dpi=300,
+ bbox_inches="tight",
+ )
+ plt.savefig(
+ methodology_figures_dir / "focus_score_vs_laplacian.png",
+ dpi=300,
+ bbox_inches="tight",
+ )
+ plt.close(fig)
+
+ # Distribution comparison
+ fig, axes = plt.subplots(1, 2, figsize=(10, 4))
+
+ axes[0].hist(df_valid[focus_col], bins=50, alpha=0.7, edgecolor="black")
+ axes[0].set_title("std²/mean (DAPI Focus Score)", fontsize=12)
+ axes[0].set_xlabel("Focus Score (std²/mean)", fontsize=11)
+ axes[0].set_ylabel("Number of tiles", fontsize=11)
+ axes[0].grid(True, alpha=0.3, axis="y")
+
+ axes[1].hist(df_valid["dapi_lap_var"], bins=50, alpha=0.7, edgecolor="black")
+ axes[1].set_title("Laplacian Variance (DAPI)", fontsize=12)
+ axes[1].set_xlabel("Laplacian Variance", fontsize=11)
+ axes[1].set_ylabel("Number of tiles", fontsize=11)
+ axes[1].grid(True, alpha=0.3, axis="y")
+
+ plt.tight_layout()
+ # Save to methodology assessment folder
+ plt.savefig(
+ methodology_figures_dir / "focus_score_vs_laplacian_distributions.pdf",
+ dpi=300,
+ bbox_inches="tight",
+ )
+ plt.savefig(
+ methodology_figures_dir / "focus_score_vs_laplacian_distributions.png",
+ dpi=300,
+ bbox_inches="tight",
+ )
+ plt.close(fig)
+
+ # Save source data
+ df_comparison = df_valid[[focus_col, "dapi_lap_var"]].copy()
+ df_comparison["log1p_focus_score"] = np.log1p(df_comparison[focus_col])
+ df_comparison["log1p_lap_var"] = np.log1p(df_comparison["dapi_lap_var"])
+ if figures_source_dir is not None:
+ df_comparison.to_csv(
+ figures_source_dir / "focus_score_vs_laplacian.csv", index=False
+ )
+
+ # Calculate correlation
+ correlation = np.corrcoef(
+ np.log1p(df_valid[focus_col]), np.log1p(df_valid["dapi_lap_var"])
+ )[0, 1]
+ logging.info(f" Correlation (log1p): {correlation:.4f}")
+
+
+def plot_intensity_assessment(
+ df_roi_intensities, intensity_stats, small0, figures_dir, figures_source_dir
+):
+ """
+ Create visualization of per-tile intensity distributions and spatial heatmaps.
+
+ Note: Intensity distributions are calculated from tissue-filtered tiles only
+ (tiles with any tissue overlap, i.e., coverage > 0%). Background/empty regions
+ of the image are excluded from the distributions.
+
+ Parameters:
+ -----------
+ df_roi_intensities : pandas DataFrame
+ DataFrame with per-tile intensities and coordinates (tissue-filtered tiles only)
+ intensity_stats : dict
+ Dictionary from assess_raw_intensity_quality()
+ small0 : numpy.ndarray
+ Downsampled DAPI image (for spatial overlay)
+ figures_dir : Path
+ Directory to save figures
+ figures_source_dir : Path
+ Directory to save source data
+ """
+ downsample_factor = 8
+ fig, axes = plt.subplots(2, 3, figsize=(15, 9))
+
+ # Calculate ROI centers
+ df_roi_intensities["center_x"] = (
+ df_roi_intensities["x1"] + df_roi_intensities["x2"]
+ ) / 2
+ df_roi_intensities["center_y"] = (
+ df_roi_intensities["y1"] + df_roi_intensities["y2"]
+ ) / 2
+
+ channels = [
+ ("dapi", "DAPI", intensity_stats["dapi"]),
+ ("boundary", "Boundary", intensity_stats["boundary"]),
+ ("intrna", "IntRNA", intensity_stats["intrna"]),
+ ]
+
+ # p99 of the DAPI background is the same for all three spatial panels;
+ # compute it once over the full ~86 Mpx small0 instead of once per channel
+ # inside the loop. PIXEL-IDENTICAL.
+ _small0_vmax = np.percentile(small0, 99)
+
+ for col_idx, (channel, channel_name, stats) in enumerate(channels):
+ intensity_col = f"{channel}_intensity"
+ intensities = df_roi_intensities[intensity_col].values
+
+ # Check if channel is available
+ is_available = not np.all(np.isnan(intensities))
+ intensities_valid = (
+ intensities[~np.isnan(intensities)] if is_available else np.array([])
+ )
+
+ # Row 1: Distribution plots
+ ax = axes[0, col_idx]
+
+ if is_available and len(intensities_valid) > 0:
+ # Phase v5: clip the histogram x-range to the 99.9th percentile
+ # to avoid a single saturated/outlier tile compressing the bulk
+ # distribution. Threshold lines (intensity_critical) and stats
+ # text (mean/median/p10/p90) are pulled from the FULL-data
+ # `stats` dict and are unchanged by this clip.
+ _p_hi_intensity = np.nanpercentile(intensities_valid, 99.9)
+ _clipped_intensity = intensities_valid[intensities_valid <= _p_hi_intensity]
+ ax.hist(_clipped_intensity, bins=50, alpha=0.7, edgecolor="black")
+
+ # Add threshold lines (only if thresholds are valid)
+ if not np.isnan(stats["critical_threshold"]):
+ ax.axvline(
+ stats["critical_threshold"],
+ color="red",
+ linestyle="--",
+ linewidth=2,
+ label=f"Min intensity ({stats['critical_threshold']:.0f})",
+ )
+
+ # No green "optimal range" band — only the red critical threshold
+ # line (from YAML intensity_critical) is shown to avoid implying an
+ # uncalibrated optimal window.
+
+ # Add statistics text
+ if not np.isnan(stats["mean"]):
+ stats_text = f"Mean: {stats['mean']:.0f}\nMedian: {stats['median']:.0f}\nP10: {stats['p10']:.0f}\nP90: {stats['p90']:.0f}"
+ ax.text(
+ 0.98,
+ 0.98,
+ stats_text,
+ transform=ax.transAxes,
+ verticalalignment="top",
+ horizontalalignment="right",
+ bbox=dict(boxstyle="round", facecolor="wheat", alpha=0.5),
+ fontsize=9,
+ )
+ else:
+ ax.text(
+ 0.5,
+ 0.5,
+ f"{channel_name} channel\nnot available",
+ transform=ax.transAxes,
+ ha="center",
+ va="center",
+ fontsize=14,
+ bbox=dict(boxstyle="round", facecolor="lightgray", alpha=0.5),
+ )
+
+ ax.set_xlabel(f"{channel_name} Intensity", fontsize=12)
+ ax.set_ylabel("Number of tiles", fontsize=12)
+ status_text = (
+ stats["quality_status"].upper()
+ if stats["quality_status"] != "not_available"
+ else "NOT AVAILABLE"
+ )
+ ax.set_title(
+ f"{channel_name} Intensity Distribution\n(Status: {status_text})",
+ fontsize=14,
+ )
+ if is_available and len(intensities_valid) > 0:
+ ax.legend(fontsize=10)
+ ax.grid(True, alpha=0.3)
+
+ # Row 2: Spatial heatmaps
+ ax = axes[1, col_idx]
+
+ # Get image dimensions for setting axes limits
+ img_height, img_width = small0.shape
+
+ # Show downsampled DAPI as background (optional, light)
+ _imshow_thumb(
+ ax,
+ small0,
+ cmap="Greys_r",
+ vmax=_small0_vmax,
+ alpha=0.3,
+ aspect="auto",
+ origin="upper",
+ extent=[0, img_width, img_height, 0],
+ )
+
+ # Scale coordinates to downsampled space
+ x_scaled = df_roi_intensities["center_x"] / downsample_factor
+ y_scaled = df_roi_intensities["center_y"] / downsample_factor
+
+ if is_available and len(intensities_valid) > 0:
+ # Create scatter plot heatmap (only for valid intensities)
+ valid_mask = ~np.isnan(intensities)
+
+ # Filter coordinates to be within image bounds
+ x_vals = x_scaled.values if hasattr(x_scaled, "values") else x_scaled
+ y_vals = y_scaled.values if hasattr(y_scaled, "values") else y_scaled
+ in_bounds = (
+ (x_vals >= 0)
+ & (x_vals < img_width)
+ & (y_vals >= 0)
+ & (y_vals < img_height)
+ )
+ valid_mask = valid_mask & in_bounds
+
+ if valid_mask.sum() > 0:
+ # Phase v6: fixed per-channel colorbar caps (cross-sample
+ # comparable). Calibrated against 9 tissues — see the
+ # _INTENSITY_DISPLAY_CAP module constant. vmin=0 anchors the
+ # dark end to absolute zero so dim samples render dim and
+ # bright samples fill the range; mirrors the fixed-vmin/vmax
+ # convention already used by plot_snr_roi_heatmap.
+ _vmax_intensity = _INTENSITY_DISPLAY_CAP[channel]
+ # Use hexbin for spatial heatmap — bins ALL data, renders O(bins) not O(N)
+ hb = ax.hexbin(
+ x_vals[valid_mask],
+ y_vals[valid_mask],
+ C=intensities[valid_mask],
+ reduce_C_function=np.mean,
+ gridsize=100,
+ cmap="viridis",
+ mincnt=1,
+ vmin=0,
+ vmax=_vmax_intensity,
+ # rasterized: the PDF embeds a raster of the hex grid instead of
+ # ~30-60k vector polygons per panel. Pixel-identical at dpi 300,
+ # but avoids re-rendering the vector layer for BOTH .pdf and .png.
+ rasterized=True,
+ )
+
+ # extend="max" — upper triangle marks hex cells whose mean
+ # exceeds the cap (bright tissues will saturate routinely).
+ cbar = plt.colorbar(hb, ax=ax, extend="max")
+ cbar.set_label(f"{channel_name} Intensity", fontsize=10)
+ else:
+ ax.text(
+ 0.5,
+ 0.5,
+ "No valid data points\nwithin image bounds",
+ transform=ax.transAxes,
+ ha="center",
+ va="center",
+ fontsize=12,
+ bbox=dict(boxstyle="round", facecolor="lightgray", alpha=0.5),
+ )
+ else:
+ ax.text(
+ 0.5,
+ 0.5,
+ f"{channel_name} channel\nnot available",
+ transform=ax.transAxes,
+ ha="center",
+ va="center",
+ fontsize=14,
+ bbox=dict(boxstyle="round", facecolor="lightgray", alpha=0.5),
+ )
+
+ # Set axes limits explicitly
+ ax.set_xlim(0, img_width)
+ ax.set_ylim(img_height, 0) # Reversed because origin='upper'
+ ax.set_xlabel("X coordinate (downsampled)", fontsize=12)
+ ax.set_ylabel("Y coordinate (downsampled)", fontsize=12)
+ ax.set_title(f"{channel_name} Intensity Spatial Heatmap", fontsize=14)
+ ax.set_aspect("equal")
+
+ plt.tight_layout()
+ plt.savefig(figures_dir / "intensity_assessment.pdf", dpi=300, bbox_inches="tight")
+ plt.savefig(figures_dir / "intensity_assessment.png", dpi=300, bbox_inches="tight")
+ plt.close(fig)
+
+ # Save source data
+ if figures_source_dir is None:
+ return
+ df_roi_intensities.to_csv(
+ figures_source_dir / "intensity_assessment.csv", index=False
+ )
+
+ # Save statistics summary
+ df_stats = pd.DataFrame(
+ {
+ "channel": ["dapi", "boundary", "intrna"],
+ "mean_intensity": [
+ intensity_stats["dapi"]["mean"],
+ intensity_stats["boundary"]["mean"],
+ intensity_stats["intrna"]["mean"],
+ ],
+ "median_intensity": [
+ intensity_stats["dapi"]["median"],
+ intensity_stats["boundary"]["median"],
+ intensity_stats["intrna"]["median"],
+ ],
+ "p10_intensity": [
+ intensity_stats["dapi"]["p10"],
+ intensity_stats["boundary"]["p10"],
+ intensity_stats["intrna"]["p10"],
+ ],
+ "p90_intensity": [
+ intensity_stats["dapi"]["p90"],
+ intensity_stats["boundary"]["p90"],
+ intensity_stats["intrna"]["p90"],
+ ],
+ "critical_threshold": [
+ intensity_stats["dapi"]["critical_threshold"],
+ intensity_stats["boundary"]["critical_threshold"],
+ intensity_stats["intrna"]["critical_threshold"],
+ ],
+ "pct_tissue_roi_below_critical": [
+ intensity_stats["dapi"]["pct_tissue_roi_below_critical"],
+ intensity_stats["boundary"]["pct_tissue_roi_below_critical"],
+ intensity_stats["intrna"]["pct_tissue_roi_below_critical"],
+ ],
+ "quality_status": [
+ intensity_stats["dapi"]["quality_status"],
+ intensity_stats["boundary"]["quality_status"],
+ intensity_stats["intrna"]["quality_status"],
+ ],
+ }
+ )
+ df_stats.to_csv(
+ figures_source_dir / "intensity_assessment_statistics.csv", index=False
+ )
+
+
+def generate_roi_figures(
+ data,
+ small0,
+ small1,
+ small2,
+ distance_map,
+ distance_map2,
+ whole_sample,
+ holes,
+ dense_intensity_regions,
+ df_grid_roi,
+ df_roi_intensities,
+ intensity_stats,
+ xoa_morphology_files,
+ focus_maps=None,
+ focus_heatmap=None,
+ snr_thresholds=None,
+ multistain_whole_sample=None,
+ multistain_distance_map=None,
+ multistain_distance_map2=None,
+ figure_source_tables=False,
+):
+ """
+ Generate cell-independent tile-based figures using multithreading.
+
+ Parameters:
+ -----------
+ data : dict
+ Dictionary with 'figures_dir' key
+ small0, small1, small2 : numpy.ndarray
+ Downsampled morphology images
+ distance_map, distance_map2 : numpy.ndarray
+ Distance maps (edge and holes)
+ whole_sample, holes, dense_intensity_regions : numpy.ndarray
+ Tissue masks
+ df_grid_roi : pandas DataFrame
+ Grid ROI focus scores
+ df_roi_intensities : pandas DataFrame
+ Tile intensity measurements
+ intensity_stats : dict
+ Intensity quality assessment statistics
+ xoa_morphology_files : list
+ List of morphology file paths
+ focus_maps : dict, optional
+ Focus map arrays for heatmap overlay
+ snr_thresholds : dict, optional
+ YAML ``snr.roi_tx`` block for SNR heatmap threshold lines
+ """
+ figures_dir = data["figures_dir"]
+ figures_source_dir = (
+ (figures_dir / "figures_source") if figure_source_tables else None
+ )
+ if figures_source_dir is not None:
+ figures_source_dir.mkdir(parents=True, exist_ok=True)
+ # Create methodology assessment folder for comparison figures
+ methodology_figures_dir = figures_dir / "figures_methodology_assessment"
+ methodology_figures_dir.mkdir(parents=True, exist_ok=True)
+
+ # §2.4 distance figures use the multi-stain extent mask + its distance maps when present,
+ # matching the edge/hole burden metrics in save_roi_qc_metrics. DAPI fallback (None on
+ # DAPI-only bundles) keeps those bundles byte-identical. distance_map/distance_map2 are
+ # referenced only by the distance closures, so rebinding them here is safe; the mask is
+ # aliased as _dm_mask because whole_sample is also used by the masks figure below.
+ if multistain_distance_map is not None:
+ distance_map = multistain_distance_map
+ if multistain_distance_map2 is not None:
+ distance_map2 = multistain_distance_map2
+ _dm_mask = (
+ whole_sample if multistain_whole_sample is None else multistain_whole_sample
+ )
+
+ def _fig1_distance_edge():
+ logging.info("Generating Figure 1: Distance map (edge)...")
+ t_fig = time.time()
+ _h, _w = distance_map.shape
+ _aspect = (_h / _w) if _w > 0 else 1.0
+ # Floor the height at 50% of width so very wide slides (e.g. brain) don't get squashed
+ fig, ax = plt.subplots(1, 1, figsize=(6, max(3, 6 * _aspect)))
+ # Phase v5: distance-to-edge readability fix.
+ # - viridis so near-edge tissue (low |distance|) renders as bright
+ # yellow against the white background — visible diagnostic region.
+ # - Absolute distance in µm: signed maurer is negative inside;
+ # |·| × 8 × 0.2125 → 0-at-boundary → max-deep-inside gradient.
+ # - NaN outside tissue mask → white via cmap.set_bad.
+ # - Thin black tissue outline as unambiguous boundary marker.
+ # TODO: 8 (downsample factor) and 0.2125 (Xenium native µm/px) are
+ # hardcoded here and in three other sites. See task #15 — future
+ # plumbing reads pixel_size from the bundle's experiment.xenium.
+ from copy import copy as _copy_cmap
+
+ _cmap_edge = _copy_cmap(plt.cm.viridis)
+ _cmap_edge.set_bad(color="white")
+ # Cap the colorscale at 300 µm so the near-edge band uses most of the
+ # spectrum; tiles further than 300 µm from the boundary saturate at
+ # yellow and the colorbar shows an "extend max" arrow. Beyond 300 µm
+ # the tile is unambiguously deep-tissue and not edge-affected.
+ _EDGE_VMAX_UM = 300.0
+ if _dm_mask.shape == distance_map.shape:
+ _dist_um = np.abs(distance_map) * 8 * 0.2125
+ _dm_edge = np.where(_dm_mask > 0, _dist_um, np.nan)
+ im = _imshow_thumb(
+ ax, _dm_edge, cmap=_cmap_edge, vmin=0.0, vmax=_EDGE_VMAX_UM
+ )
+ _edge_cbar_extend = "max"
+ else:
+ _dm_edge = (
+ distance_map # fallback: shapes mismatch, preserve prior behaviour
+ )
+ im = _imshow_thumb(ax, _dm_edge, cmap=_cmap_edge)
+ _edge_cbar_extend = "neither"
+ if _dm_mask.shape == distance_map.shape:
+ _contour_thumb(
+ ax,
+ (_dm_mask > 0).astype(np.uint8),
+ levels=[0.5],
+ colors="black",
+ linewidths=0.5,
+ )
+ ax.set_title("Distance to Edge")
+ ax.set_aspect("equal")
+ cbar = fig.colorbar(
+ im, ax=ax, fraction=0.035, pad=0.04, shrink=0.45, extend=_edge_cbar_extend
+ )
+ cbar.set_label("Distance from edge (µm)")
+ # Explicit "300+" label at the cap when extend="max" is active
+ if _edge_cbar_extend == "max":
+ cbar.set_ticks([0, 50, 100, 150, 200, 250, _EDGE_VMAX_UM])
+ cbar.set_ticklabels(
+ ["0", "50", "100", "150", "200", "250", f"{int(_EDGE_VMAX_UM)}+"]
+ )
+ plt.tight_layout()
+ plt.savefig(figures_dir / "distance_map_edge.pdf", dpi=300, bbox_inches="tight")
+ plt.savefig(figures_dir / "distance_map_edge.png", dpi=300, bbox_inches="tight")
+ plt.close(fig)
+ # Full-resolution export. These raster figures_source CSVs are gated behind
+ # --figure-source-tables (off by default), so the ~86-Mpx / ~1 GB dump only
+ # happens when a user explicitly asks for the raw plotting data — and then it
+ # must match the analysis exactly, not a thumbnail. The rendered PNG still uses
+ # the _imshow_thumb / _thumb display-resolution path for speed regardless.
+ if figures_source_dir is not None:
+ pd.DataFrame(distance_map).to_csv(
+ figures_source_dir / "distance_map_edge.csv",
+ index=False,
+ header=False,
+ )
+ logging.info(
+ f"[TIMING] Figure 1 (distance map edge): {time.time() - t_fig:.1f}s"
+ )
+
+ def _fig2_distance_holes():
+ logging.info("Generating Figure 2: Distance map (holes)...")
+ t_fig = time.time()
+ _h, _w = distance_map2.shape
+ _aspect = (_h / _w) if _w > 0 else 1.0
+ # Floor the height at 50% of width so very wide slides (e.g. brain) don't get squashed
+ fig, ax = plt.subplots(1, 1, figsize=(6, max(3, 6 * _aspect)))
+ # Phase v5: distance-to-holes readability fix.
+ # - viridis colormap (same as edge map) for visual consistency.
+ # - Absolute distance in µm: |signed_maurer_distance| × 8 × 0.2125.
+ # - Linear vmin=0 / vmax=300 µm — matches the edge map for visual
+ # consistency across both panels of §2.4 (edge + holes). Tiles beyond 300 µm saturate
+ # at yellow with an "extend max" arrow on the colorbar.
+ # - NaN outside tissue mask → white via cmap.set_bad.
+ # - Thin black tissue outline as boundary marker.
+ # TODO: pixel-size conversion hardcoded — see task #15.
+ from copy import copy as _copy_cmap
+
+ _cmap_holes = _copy_cmap(plt.cm.viridis)
+ _cmap_holes.set_bad(color="white")
+ _HOLES_VMAX_UM = 300.0
+ if _dm_mask.shape == distance_map2.shape:
+ _dist_um_h = np.abs(distance_map2) * 8 * 0.2125
+ _dm_holes = np.where(_dm_mask > 0, _dist_um_h, np.nan)
+ im = _imshow_thumb(
+ ax, _dm_holes, cmap=_cmap_holes, vmin=0.0, vmax=_HOLES_VMAX_UM
+ )
+ _holes_cbar_extend = "max"
+ else:
+ _dm_holes = distance_map2 # fallback: shapes mismatch
+ im = _imshow_thumb(ax, _dm_holes, cmap=_cmap_holes)
+ _holes_cbar_extend = "neither"
+ if _dm_mask.shape == distance_map2.shape:
+ _contour_thumb(
+ ax,
+ (_dm_mask > 0).astype(np.uint8),
+ levels=[0.5],
+ colors="black",
+ linewidths=0.5,
+ )
+ ax.set_title("Distance to Nearest Hole")
+ ax.set_aspect("equal")
+ cbar = fig.colorbar(
+ im, ax=ax, fraction=0.035, pad=0.04, shrink=0.45, extend=_holes_cbar_extend
+ )
+ cbar.set_label("Distance from nearest hole (µm)")
+ # Explicit "300+" label at the cap when extend="max" is active.
+ if _holes_cbar_extend == "max":
+ cbar.set_ticks([0, 50, 100, 150, 200, 250, _HOLES_VMAX_UM])
+ cbar.set_ticklabels(
+ ["0", "50", "100", "150", "200", "250", f"{int(_HOLES_VMAX_UM)}+"]
+ )
+ plt.tight_layout()
+ plt.savefig(
+ figures_dir / "distance_map_holes.pdf", dpi=300, bbox_inches="tight"
+ )
+ plt.savefig(
+ figures_dir / "distance_map_holes.png", dpi=300, bbox_inches="tight"
+ )
+ plt.close(fig)
+ # Full-resolution export (see distance_map_edge.csv note): gated behind
+ # --figure-source-tables (off by default); rendering uses the display-res thumbnail.
+ if figures_source_dir is not None:
+ pd.DataFrame(distance_map2).to_csv(
+ figures_source_dir / "distance_map_holes.csv",
+ index=False,
+ header=False,
+ )
+ logging.info(
+ f"[TIMING] Figure 2 (distance map holes): {time.time() - t_fig:.1f}s"
+ )
+
+ def _fig2b_distance_combined():
+ """Two-panel combined figure: distance to edge (left) + distance to
+ holes (right), sharing coordinate system and colorbar conventions.
+ Standalone _fig1_distance_edge / _fig2_distance_holes continue to
+ render the individual PNGs as latent artefacts; the combined figure
+ is what the QMD §2.4 embeds.
+ """
+ logging.info("Generating Figure 2b: Distance combined (edge + holes)...")
+ t_fig = time.time()
+ _h, _w = distance_map.shape
+ _aspect = (_h / _w) if _w > 0 else 1.0
+ _panel_width = 6
+ _panel_height = max(3, _panel_width * _aspect)
+ fig, axes = plt.subplots(1, 2, figsize=(2 * _panel_width + 2, _panel_height))
+
+ from copy import copy as _copy_cmap
+
+ _VMAX_UM = 300.0
+ _cmap = _copy_cmap(plt.cm.viridis)
+ _cmap.set_bad(color="white")
+
+ # ── Left panel: distance to edge ───────────────────────────────
+ ax = axes[0]
+ if _dm_mask.shape == distance_map.shape:
+ _dist_um = np.abs(distance_map) * 8 * 0.2125
+ _dm_edge = np.where(_dm_mask > 0, _dist_um, np.nan)
+ im_e = _imshow_thumb(ax, _dm_edge, cmap=_cmap, vmin=0.0, vmax=_VMAX_UM)
+ _contour_thumb(
+ ax,
+ (_dm_mask > 0).astype(np.uint8),
+ levels=[0.5],
+ colors="black",
+ linewidths=0.5,
+ )
+ _edge_extend = "max"
+ else:
+ im_e = _imshow_thumb(ax, distance_map, cmap=_cmap)
+ _edge_extend = "neither"
+ ax.set_title("Distance to Edge", fontsize=14)
+ ax.set_aspect("equal")
+ cbar_e = fig.colorbar(
+ im_e, ax=ax, fraction=0.035, pad=0.04, shrink=0.45, extend=_edge_extend
+ )
+ cbar_e.set_label("Distance from edge (µm)")
+ if _edge_extend == "max":
+ cbar_e.set_ticks([0, 50, 100, 150, 200, 250, _VMAX_UM])
+ cbar_e.set_ticklabels(
+ ["0", "50", "100", "150", "200", "250", f"{int(_VMAX_UM)}+"]
+ )
+
+ # ── Right panel: distance to holes ─────────────────────────────
+ ax = axes[1]
+ if _dm_mask.shape == distance_map2.shape:
+ _dist_um_h = np.abs(distance_map2) * 8 * 0.2125
+ _dm_holes = np.where(_dm_mask > 0, _dist_um_h, np.nan)
+ im_h = _imshow_thumb(ax, _dm_holes, cmap=_cmap, vmin=0.0, vmax=_VMAX_UM)
+ _contour_thumb(
+ ax,
+ (_dm_mask > 0).astype(np.uint8),
+ levels=[0.5],
+ colors="black",
+ linewidths=0.5,
+ )
+ _holes_extend = "max"
+ else:
+ im_h = _imshow_thumb(ax, distance_map2, cmap=_cmap)
+ _holes_extend = "neither"
+ ax.set_title("Distance to Nearest Hole", fontsize=14)
+ ax.set_aspect("equal")
+ cbar_h = fig.colorbar(
+ im_h, ax=ax, fraction=0.035, pad=0.04, shrink=0.45, extend=_holes_extend
+ )
+ cbar_h.set_label("Distance from nearest hole (µm)")
+ if _holes_extend == "max":
+ cbar_h.set_ticks([0, 50, 100, 150, 200, 250, _VMAX_UM])
+ cbar_h.set_ticklabels(
+ ["0", "50", "100", "150", "200", "250", f"{int(_VMAX_UM)}+"]
+ )
+
+ plt.tight_layout()
+ plt.savefig(figures_dir / "distance_maps.png", dpi=300, bbox_inches="tight")
+ plt.savefig(figures_dir / "distance_maps.pdf", dpi=300, bbox_inches="tight")
+ plt.close(fig)
+ logging.info(
+ f"[TIMING] Figure 2b (distance combined): {time.time() - t_fig:.1f}s"
+ )
+
+ def _fig3_morphology_overview():
+ logging.info("Generating Figure 3: Morphology overview...")
+ t_fig = time.time()
+ # Phase v5 TODO #12: 1×3 layout (DAPI + Boundary + Interior only). The
+ # artefact / dense-intensity-regions panel that used to live in this
+ # figure's bottom-right is also rendered in imageqc_masks.png (§2.3 Masks),
+ # so showing it here too duplicates the same plot — dropped here.
+ _img_aspect = small0.shape[0] / small0.shape[1] if small0.shape[1] > 0 else 1.0
+ _panel_width = 6
+ _panel_height = max(_panel_width * 0.5, _panel_width * _img_aspect)
+ fig, ax = plt.subplots(1, 3, figsize=(3 * _panel_width, _panel_height))
+ _imshow_thumb(ax[0], small0, cmap="Greys_r", vmax=np.percentile(small0, 99))
+ ax[0].set_title("DAPI", fontsize=14)
+ ax[0].set_aspect("equal")
+ ax[0].axis("off")
+ _imshow_thumb(ax[1], small1, cmap="Greys_r", vmax=np.percentile(small1, 99))
+ ax[1].set_title("Boundary", fontsize=14)
+ ax[1].set_aspect("equal")
+ ax[1].axis("off")
+ _imshow_thumb(ax[2], small2, cmap="Greys_r", vmax=np.percentile(small2, 99))
+ ax[2].set_title("Interior", fontsize=14)
+ ax[2].set_aspect("equal")
+ ax[2].axis("off")
+ plt.tight_layout()
+ plt.savefig(
+ figures_dir / "morphology_overview.pdf", dpi=300, bbox_inches="tight"
+ )
+ plt.savefig(
+ figures_dir / "morphology_overview.png", dpi=300, bbox_inches="tight"
+ )
+ plt.close(fig)
+ # Full-resolution export (see distance_map_edge.csv note): four ~86-Mpx channel
+ # grids, gated behind --figure-source-tables (off by default); rendering uses
+ # the display-res thumbnail so the figure wall time is unaffected.
+ if figures_source_dir is not None:
+ pd.DataFrame(small0).to_csv(
+ figures_source_dir / "morphology_overview_DAPI.csv",
+ index=False,
+ header=False,
+ )
+ pd.DataFrame(small1).to_csv(
+ figures_source_dir / "morphology_overview_Boundary.csv",
+ index=False,
+ header=False,
+ )
+ pd.DataFrame(small2).to_csv(
+ figures_source_dir / "morphology_overview_Interior.csv",
+ index=False,
+ header=False,
+ )
+ pd.DataFrame(dense_intensity_regions).to_csv(
+ figures_source_dir / "morphology_overview_DenseIntensityRegions.csv",
+ index=False,
+ header=False,
+ )
+ logging.info(
+ f"[TIMING] Figure 3 (morphology overview): {time.time() - t_fig:.1f}s"
+ )
+
+ def _fig4_imageqc_masks():
+ logging.info("Generating Figure 4: ImageQC masks...")
+ t_fig = time.time()
+ # Phase v5: 1x3 triplet — three mask panels only. DAPI morphology
+ # image was dropped (it's a staining, already shown under §2.4
+ # Stainings via morphology_overview.png).
+ _img_aspect = small0.shape[0] / small0.shape[1] if small0.shape[1] > 0 else 1.0
+ _panel_width = 6
+ _panel_height = max(_panel_width * 0.5, _panel_width * _img_aspect)
+ # §2.3 tissue mask shows the EXTENT mask (all available stains) so it matches the
+ # reported tissue coverage; falls back to the DAPI mask on single-stain slides.
+ _extent_ws = (
+ multistain_whole_sample
+ if multistain_whole_sample is not None
+ else whole_sample
+ )
+ fig, ax = plt.subplots(1, 3, figsize=(3 * _panel_width, _panel_height))
+ ax[0].set_title("Tissue mask (all stains)", fontsize=14)
+ _imshow_thumb(ax[0], _extent_ws, rgb=lambda d: color.label2rgb(d, bg_label=0))
+ ax[0].set_aspect("equal")
+ ax[0].axis("off")
+ ax[1].set_title("Holes in sample", fontsize=14)
+ _imshow_thumb(ax[1], holes, rgb=lambda d: color.label2rgb(d, bg_label=0))
+ ax[1].set_aspect("equal")
+ ax[1].axis("off")
+ ax[2].set_title("Optically dense regions", fontsize=14)
+ _imshow_thumb(
+ ax[2], dense_intensity_regions, rgb=lambda d: color.label2rgb(d, bg_label=0)
+ )
+ ax[2].set_aspect("equal")
+ ax[2].axis("off")
+ plt.tight_layout()
+ plt.savefig(figures_dir / "imageqc_masks.pdf", dpi=300, bbox_inches="tight")
+ plt.savefig(figures_dir / "imageqc_masks.png", dpi=300, bbox_inches="tight")
+ plt.close(fig)
+ # DAPI csv export dropped — same data exposed by §2.4 Stainings.
+ # Full-resolution export (see distance_map_edge.csv note): three ~86-Mpx mask
+ # grids, gated behind --figure-source-tables (off by default); rendering uses
+ # the display-res thumbnail so the figure wall time is unaffected.
+ if figures_source_dir is not None:
+ pd.DataFrame(_extent_ws).to_csv(
+ figures_source_dir / "imageqc_masks_WholeSample.csv",
+ index=False,
+ header=False,
+ )
+ pd.DataFrame(holes).to_csv(
+ figures_source_dir / "imageqc_masks_Holes.csv",
+ index=False,
+ header=False,
+ )
+ pd.DataFrame(dense_intensity_regions).to_csv(
+ figures_source_dir / "imageqc_masks_DenseIntensityRegions.csv",
+ index=False,
+ header=False,
+ )
+ logging.info(f"[TIMING] Figure 4 (ImageQC masks): {time.time() - t_fig:.1f}s")
+
+ def _fig5_focus_heatmap():
+ logging.info("Generating Figure 5: Grid tile focus heatmap...")
+ t_fig = time.time()
+ plot_grid_roi_focus_heatmap(
+ df_grid_roi,
+ small0,
+ figures_dir,
+ figures_source_dir,
+ threshold=-1.0,
+ focus_maps=focus_maps,
+ focus_heatmap=focus_heatmap,
+ )
+ logging.info(f"[TIMING] Figure 5 (focus heatmap): {time.time() - t_fig:.1f}s")
+
+ def _fig5b_focus_vs_intensity():
+ logging.info("Generating Figure 5b: Tile focus score vs intensity...")
+ t_fig = time.time()
+ plot_roi_focus_vs_intensity(
+ df_grid_roi, figures_dir, figures_source_dir, threshold=-1.0
+ )
+ logging.info(
+ f"[TIMING] Figure 5b (focus vs intensity): {time.time() - t_fig:.1f}s"
+ )
+
+ def _fig5c_focus_distribution():
+ logging.info("Generating Figure 5c: Tile focus score distribution...")
+ t_fig = time.time()
+ plot_roi_focus_distribution(
+ df_grid_roi, figures_dir, figures_source_dir, threshold=-1.0
+ )
+ logging.info(
+ f"[TIMING] Figure 5c (focus distribution): {time.time() - t_fig:.1f}s"
+ )
+
+ def _fig5d_focus_vs_laplacian():
+ logging.info(
+ "Generating Figure 5d: Focus score vs Laplacian variance comparison..."
+ )
+ t_fig = time.time()
+ plot_focus_score_vs_laplacian(df_grid_roi, figures_dir, figures_source_dir)
+ logging.info(
+ f"[TIMING] Figure 5d (focus vs Laplacian): {time.time() - t_fig:.1f}s"
+ )
+
+ def _fig6_intensity_assessment():
+ logging.info("Generating Figure 6: Intensity assessment...")
+ t_fig = time.time()
+ plot_intensity_assessment(
+ df_roi_intensities, intensity_stats, small0, figures_dir, figures_source_dir
+ )
+ logging.info(
+ f"[TIMING] Figure 6 (intensity assessment): {time.time() - t_fig:.1f}s"
+ )
+
+ def _fig7_snr_heatmap():
+ if "neg_pct" not in df_grid_roi.columns:
+ logging.info("Skipping SNR heatmap: no SNR columns in df_grid_roi")
+ return
+ logging.info("Generating Figure 7: SNR spatial heatmap...")
+ t_fig = time.time()
+ plot_snr_roi_heatmap(
+ df_grid_roi,
+ small0,
+ figures_dir,
+ figures_source_dir,
+ snr_thresholds=snr_thresholds,
+ )
+ logging.info(f"[TIMING] Figure 7 (SNR heatmap): {time.time() - t_fig:.1f}s")
+
+ def _fig8_concordance():
+ if "roi_tx_snr_ratio" not in df_grid_roi.columns:
+ logging.info("Skipping concordance plot: no SNR columns in df_grid_roi")
+ return
+ logging.info("Generating Figure 8: Cross-section concordance...")
+ t_fig = time.time()
+ plot_cross_section_concordance(
+ df_grid_roi,
+ figures_dir,
+ figures_source_dir,
+ snr_thresholds=snr_thresholds,
+ )
+ logging.info(f"[TIMING] Figure 8 (concordance): {time.time() - t_fig:.1f}s")
+
+ tasks = [
+ _fig1_distance_edge,
+ _fig2_distance_holes,
+ _fig2b_distance_combined,
+ _fig3_morphology_overview,
+ _fig4_imageqc_masks,
+ _fig5_focus_heatmap,
+ _fig5b_focus_vs_intensity,
+ _fig5c_focus_distribution,
+ # _fig5d_focus_vs_laplacian disabled 2026-05-20 — produces
+ # focus_score_vs_laplacian.png + focus_score_vs_laplacian_distributions.png
+ # in methodology_figures_dir; neither is embedded in the report.
+ # Function definition retained above as latent code.
+ _fig6_intensity_assessment,
+ _fig7_snr_heatmap,
+ _fig8_concordance,
+ ]
+
+ _run_figure_pool(tasks, phase="tile-based figures")
+
+ logging.info("All tile-based figures generated successfully!")
+ logging.info(f"Saved figures to {figures_dir}")
+ if figures_source_dir is not None:
+ logging.info(f"Saved source data to {figures_source_dir}")
+
+
+def compute_whole_grid_stain_percentiles(df_grid_roi):
+ """Mask-independent stain percentiles (p95 / p99) per channel over the WHOLE
+ tile grid (no tissue-mask filter), so they survive a mask-generation failure
+ (qc_threshold_refinement §5.5). DIAGNOSTIC ONLY — used to triage why a mask
+ failed (collapsed p99 = dim stain like skin; healthy p99 = a structural
+ mask-detection failure on bright tissue). NOT a gate: an absolute p99 does
+ not separate mask-PASS from mask-FAIL (XOA-version confound).
+
+ Returns ``{channel: {"p95": float|None, "p99": float|None}}`` for dapi,
+ boundary, intrna. None when the channel column is absent or all-NaN.
+ """
+ out = {}
+ for ch, col in (
+ ("dapi", "dapi_intensity"),
+ ("boundary", "boundary_intensity"),
+ ("intrna", "intrna_intensity"),
+ ):
+ vals = None
+ if col in df_grid_roi.columns:
+ vals = pd.to_numeric(df_grid_roi[col], errors="coerce").to_numpy()
+ vals = vals[np.isfinite(vals)]
+ if vals is not None and vals.size > 0:
+ out[ch] = {
+ "p95": float(np.percentile(vals, 95)),
+ "p99": float(np.percentile(vals, 99)),
+ }
+ else:
+ out[ch] = {"p95": None, "p99": None}
+ return out
+
+
+def save_roi_qc_metrics(
+ df_grid_roi,
+ intensity_stats,
+ outdir,
+ roi_size=None,
+ snr_summary=None,
+ distance_map=None,
+ distance_map2=None,
+ multistain_whole_sample=None,
+ multistain_distance_map=None,
+ multistain_distance_map2=None,
+ edge_distance_threshold: float = -25.0,
+ hole_distance_threshold: float = -25.0,
+ min_tissue_coverage_for_qc: float = ROI_MIN_TISSUE_COVERAGE_FOR_INTENSITY_QC,
+ qc_thresholds: dict | None = None,
+ lap_sigma: float | None = None,
+ segmentation_software: str | None = None,
+ xoa_version: str | None = None,
+):
+ """
+ Save tile-based QC metrics to JSON file.
+
+ Parameters
+ ----------
+ snr_summary : dict or None
+ If provided, stored under ``snr`` (from :mod:`snr_metrics`).
+ segmentation_software : str or None
+ Human-readable label for the segmentation software that produced the
+ bundle this report describes (e.g. ``"Xenium Onboard Analysis v4.0.1"``).
+ """
+ import json
+
+ # Tissue EXTENT coverage: from the multi-stain mask (all available stains) when provided,
+ # else the DAPI tissue_coverage column. EXTENT metrics (coverage stats, mask status,
+ # edge/hole) use this; DAPI-QUALITY metrics (focus, blur, usable, cluster) keep the DAPI
+ # tissue_coverage column. DAPI-only bundles pass multistain_whole_sample=None, so extent
+ # == DAPI and the output is identical. See plans/2026-06-26_PLAN_multistain-mask.md.
+ _has_xy = {"x1", "x2", "y1", "y2"}.issubset(df_grid_roi.columns)
+ if multistain_whole_sample is not None and _has_xy:
+ _ext_cov = _per_tile_coverage(
+ multistain_whole_sample,
+ df_grid_roi["x1"].to_numpy(np.int64, copy=False),
+ df_grid_roi["x2"].to_numpy(np.int64, copy=False),
+ df_grid_roi["y1"].to_numpy(np.int64, copy=False),
+ df_grid_roi["y2"].to_numpy(np.int64, copy=False),
+ )
+ _ext_dmap = (
+ distance_map if multistain_distance_map is None else multistain_distance_map
+ )
+ _ext_dmap2 = (
+ distance_map2
+ if multistain_distance_map2 is None
+ else multistain_distance_map2
+ )
+ else:
+ _ext_cov = (
+ df_grid_roi["tissue_coverage"].to_numpy(np.float64, copy=False)
+ if "tissue_coverage" in df_grid_roi.columns
+ else None
+ )
+ _ext_dmap, _ext_dmap2 = distance_map, distance_map2
+ _ext_rois_in_tissue = int((_ext_cov > 0).sum()) if _ext_cov is not None else 0
+
+ # Existing stats
+ roi_metrics = {
+ "roi_size_pixels": roi_size,
+ "xoa_version": xoa_version,
+ "segmentation_software": segmentation_software,
+ "total_rois": len(df_grid_roi),
+ "rois_in_tissue": _ext_rois_in_tissue,
+ "focus_score": {
+ "mean": float(df_grid_roi["focus_score"].mean()),
+ "median": float(df_grid_roi["focus_score"].median()),
+ "std": float(df_grid_roi["focus_score"].std()),
+ "min": float(df_grid_roi["focus_score"].min()),
+ "max": float(df_grid_roi["focus_score"].max()),
+ "tissue_median": float(
+ df_grid_roi.loc[
+ df_grid_roi["tissue_coverage"] >= min_tissue_coverage_for_qc,
+ "focus_score",
+ ].median()
+ )
+ if "tissue_coverage" in df_grid_roi.columns
+ else float(df_grid_roi["focus_score"].median()),
+ },
+ "focus_score_norm": {
+ "mean": float(df_grid_roi["focus_score_norm"].mean()),
+ "median": float(df_grid_roi["focus_score_norm"].median()),
+ "std": float(df_grid_roi["focus_score_norm"].std()),
+ "min": float(df_grid_roi["focus_score_norm"].min()),
+ "max": float(df_grid_roi["focus_score_norm"].max()),
+ },
+ "raw_intensity": {
+ "mean": float(df_grid_roi["raw_intensity"].mean()),
+ "median": float(df_grid_roi["raw_intensity"].median()),
+ "std": float(df_grid_roi["raw_intensity"].std()),
+ "min": float(df_grid_roi["raw_intensity"].min()),
+ "max": float(df_grid_roi["raw_intensity"].max()),
+ },
+ # tissue_coverage (extent) reflects all available stains (multi-stain mask).
+ "tissue_coverage": {
+ "mean": float(np.mean(_ext_cov)) if _ext_cov is not None else 0.0,
+ "median": float(np.median(_ext_cov)) if _ext_cov is not None else 0.0,
+ "min": float(np.min(_ext_cov)) if _ext_cov is not None else 0.0,
+ "max": float(np.max(_ext_cov)) if _ext_cov is not None else 0.0,
+ },
+ "intensity_quality": intensity_stats,
+ }
+ total_rois = int(len(df_grid_roi))
+ rois_in_tissue = _ext_rois_in_tissue # extent: any tissue tile (all stains)
+ roi_metrics["tissue_mask_qc"] = {
+ "tissue_mask_generated": bool(rois_in_tissue > 0),
+ "status": "PASS" if rois_in_tissue > 0 else "FAIL",
+ "rois_in_tissue": rois_in_tissue,
+ "total_rois": total_rois,
+ "tissue_roi_fraction": float(rois_in_tissue / total_rois)
+ if total_rois > 0
+ else 0.0,
+ }
+
+ # Mask-independent stain percentiles (p95 / p99 over the WHOLE tile grid,
+ # 2026-06-23, qc_threshold_refinement §5.5). See
+ # compute_whole_grid_stain_percentiles for the diagnostic-only rationale.
+ roi_metrics["stain_percentiles_whole_grid"] = compute_whole_grid_stain_percentiles(
+ df_grid_roi
+ )
+
+ # Optional: add GMM-based blur summary if columns are present
+ if (
+ "is_blurred_gmm" in df_grid_roi.columns
+ and "is_low_intensity" in df_grid_roi.columns
+ ):
+ total_rois = len(df_grid_roi)
+ n_blurred = int(df_grid_roi["is_blurred_gmm"].sum())
+ n_low_int = int(df_grid_roi["is_low_intensity"].sum())
+ if "tissue_coverage" in df_grid_roi.columns:
+ tissue_mask_qc = (
+ df_grid_roi["tissue_coverage"] >= min_tissue_coverage_for_qc
+ )
+ total_rois_tissue = int(tissue_mask_qc.sum())
+ n_blurred_tissue = int(
+ df_grid_roi.loc[tissue_mask_qc, "is_blurred_gmm"].sum()
+ )
+ n_low_int_tissue = int(
+ df_grid_roi.loc[tissue_mask_qc, "is_low_intensity"].sum()
+ )
+ else:
+ total_rois_tissue = total_rois
+ n_blurred_tissue = n_blurred
+ n_low_int_tissue = n_low_int
+
+ roi_metrics["blur_gmm_1d"] = {
+ "total_rois": total_rois,
+ "rois_blurred_gmm": n_blurred,
+ "rois_low_intensity": n_low_int,
+ "pct_blurred_gmm": float(n_blurred / total_rois * 100.0)
+ if total_rois > 0
+ else 0.0,
+ "pct_low_intensity": float(n_low_int / total_rois * 100.0)
+ if total_rois > 0
+ else 0.0,
+ # Tissue-aware denominator for report-level interpretation
+ "total_rois_tissue_filtered": total_rois_tissue,
+ "rois_blurred_gmm_tissue_filtered": n_blurred_tissue,
+ "rois_low_intensity_tissue_filtered": n_low_int_tissue,
+ "pct_blurred_gmm_tissue_filtered": float(
+ n_blurred_tissue / total_rois_tissue * 100.0
+ )
+ if total_rois_tissue > 0
+ else 0.0,
+ "pct_low_intensity_tissue_filtered": float(
+ n_low_int_tissue / total_rois_tissue * 100.0
+ )
+ if total_rois_tissue > 0
+ else 0.0,
+ "tissue_coverage_min_for_qc": float(min_tissue_coverage_for_qc),
+ }
+
+ if "blur_prob_gmm" in df_grid_roi.columns:
+ valid_probs = df_grid_roi["blur_prob_gmm"].dropna()
+ if len(valid_probs) > 0:
+ roi_metrics["blur_gmm_1d"]["blur_prob_mean"] = float(valid_probs.mean())
+ roi_metrics["blur_gmm_1d"]["blur_prob_median"] = float(
+ valid_probs.median()
+ )
+
+ # Optional: add 2D GMM-based blur summary if columns are present
+ if "is_blurred_gmm_2d" in df_grid_roi.columns:
+ total_rois = len(df_grid_roi)
+ n_blurred_2d = int(df_grid_roi["is_blurred_gmm_2d"].sum())
+ n_low_int = (
+ int(df_grid_roi["is_low_intensity"].sum())
+ if "is_low_intensity" in df_grid_roi.columns
+ else 0
+ )
+ if "tissue_coverage" in df_grid_roi.columns:
+ tissue_mask_qc = (
+ df_grid_roi["tissue_coverage"] >= min_tissue_coverage_for_qc
+ )
+ total_rois_tissue = int(tissue_mask_qc.sum())
+ n_blurred_tissue = int(
+ df_grid_roi.loc[tissue_mask_qc, "is_blurred_gmm_2d"].sum()
+ )
+ n_low_int_tissue = (
+ int(df_grid_roi.loc[tissue_mask_qc, "is_low_intensity"].sum())
+ if "is_low_intensity" in df_grid_roi.columns
+ else 0
+ )
+ else:
+ total_rois_tissue = total_rois
+ n_blurred_tissue = n_blurred_2d
+ n_low_int_tissue = n_low_int
+
+ roi_metrics["blur_gmm_2d"] = {
+ "total_rois": total_rois,
+ "rois_blurred_gmm": n_blurred_2d,
+ "rois_low_intensity": n_low_int,
+ "pct_blurred_gmm": float(n_blurred_2d / total_rois * 100.0)
+ if total_rois > 0
+ else 0.0,
+ "pct_low_intensity": float(n_low_int / total_rois * 100.0)
+ if total_rois > 0
+ else 0.0,
+ # Tissue-aware denominator for report-level interpretation
+ "total_rois_tissue_filtered": total_rois_tissue,
+ "rois_blurred_gmm_tissue_filtered": n_blurred_tissue,
+ "rois_low_intensity_tissue_filtered": n_low_int_tissue,
+ "pct_blurred_gmm_tissue_filtered": float(
+ n_blurred_tissue / total_rois_tissue * 100.0
+ )
+ if total_rois_tissue > 0
+ else 0.0,
+ "pct_low_intensity_tissue_filtered": float(
+ n_low_int_tissue / total_rois_tissue * 100.0
+ )
+ if total_rois_tissue > 0
+ else 0.0,
+ "tissue_coverage_min_for_qc": float(min_tissue_coverage_for_qc),
+ }
+
+ if "blur_prob_gmm_2d" in df_grid_roi.columns:
+ valid_probs = df_grid_roi["blur_prob_gmm_2d"].dropna()
+ if len(valid_probs) > 0:
+ roi_metrics["blur_gmm_2d"]["blur_prob_mean"] = float(valid_probs.mean())
+ roi_metrics["blur_gmm_2d"]["blur_prob_median"] = float(
+ valid_probs.median()
+ )
+
+ # Compare 1D vs 2D if both exist
+ if "is_blurred_gmm" in df_grid_roi.columns:
+ both_blurred = (
+ df_grid_roi["is_blurred_gmm"] & df_grid_roi["is_blurred_gmm_2d"]
+ ).sum()
+ both_in_focus = (
+ (~df_grid_roi["is_blurred_gmm"]) & (~df_grid_roi["is_blurred_gmm_2d"])
+ ).sum()
+ agreement = (
+ (both_blurred + both_in_focus) / total_rois * 100.0
+ if total_rois > 0
+ else 0.0
+ )
+ roi_metrics["gmm_comparison"] = {
+ "agreement_pct": float(agreement),
+ "both_blurred": int(both_blurred),
+ "both_in_focus": int(both_in_focus),
+ "only_1d_blurred": int(
+ (
+ df_grid_roi["is_blurred_gmm"]
+ & ~df_grid_roi["is_blurred_gmm_2d"]
+ ).sum()
+ ),
+ "only_2d_blurred": int(
+ (
+ ~df_grid_roi["is_blurred_gmm"]
+ & df_grid_roi["is_blurred_gmm_2d"]
+ ).sum()
+ ),
+ }
+
+ if snr_summary is not None:
+ roi_metrics["snr"] = snr_summary
+
+ # Optional: morphology ROI summary metrics (vectorized, memory-friendly).
+ if (
+ distance_map is not None
+ and distance_map2 is not None
+ and {"x1", "x2", "y1", "y2", "tissue_coverage"}.issubset(df_grid_roi.columns)
+ ):
+ x1 = df_grid_roi["x1"].to_numpy(np.int64, copy=False)
+ x2 = df_grid_roi["x2"].to_numpy(np.int64, copy=False)
+ y1 = df_grid_roi["y1"].to_numpy(np.int64, copy=False)
+ y2 = df_grid_roi["y2"].to_numpy(np.int64, copy=False)
+ # Two coverage arrays: DAPI for QUALITY (usable, cluster), multi-stain for EXTENT
+ # (edge/hole, tissue-tile count). ext_cov == dapi_cov on DAPI-only bundles.
+ dapi_cov = df_grid_roi["tissue_coverage"].to_numpy(np.float64, copy=False)
+ ext_cov = _ext_cov if _ext_cov is not None else dapi_cov
+
+ # EXTENT distance maps (multi-stain when present) sampled at ROI centroids.
+ cx_ds = np.clip(((x1 + x2) // 2) // 8, 0, _ext_dmap.shape[1] - 1)
+ cy_ds = np.clip(((y1 + y2) // 2) // 8, 0, _ext_dmap.shape[0] - 1)
+ dist_edge = _ext_dmap[cy_ds, cx_ds]
+ dist_hole = _ext_dmap2[cy_ds, cx_ds]
+
+ # DAPI-quality masks (usable, cluster) and EXTENT masks (edge/hole, counts).
+ dapi_tissue_mask = dapi_cov > 0.0
+ qc_tissue_mask = dapi_cov >= float(min_tissue_coverage_for_qc)
+ n_qc_tissue = int(qc_tissue_mask.sum())
+ ext_tissue_mask = ext_cov > 0.0
+ n_ext_tissue = int(ext_tissue_mask.sum())
+
+ edge_zone_frac = (
+ float(
+ (ext_tissue_mask & (dist_edge > float(edge_distance_threshold))).sum()
+ / n_ext_tissue
+ )
+ if n_ext_tissue > 0
+ else None
+ )
+ hole_area_frac = (
+ float(
+ (ext_tissue_mask & (dist_hole > float(hole_distance_threshold))).sum()
+ / n_ext_tissue
+ )
+ if n_ext_tissue > 0
+ else None
+ )
+
+ if "is_blurred_gmm_2d" in df_grid_roi.columns:
+ blurred_mask = df_grid_roi["is_blurred_gmm_2d"].to_numpy(bool, copy=False)
+ elif "is_blurred_gmm" in df_grid_roi.columns:
+ blurred_mask = df_grid_roi["is_blurred_gmm"].to_numpy(bool, copy=False)
+ else:
+ blurred_mask = np.zeros(len(df_grid_roi), dtype=bool)
+ # usable_tissue uses its OWN low-intensity floor (DAPI_LOW_INTENSITY_FLOOR=50),
+ # independent of the is_low_intensity column (which stays at ROI_INTENSITY_THRESHOLD
+ # for the 1D-GMM tissue scope). A dim-but-real tissue tile shouldn't count as
+ # unusable on absolute brightness alone. See 2026-06-26_PLAN_tissue-mask-recalibration.
+ _intensity_col = (
+ "dapi_intensity"
+ if "dapi_intensity" in df_grid_roi.columns
+ else ("raw_intensity" if "raw_intensity" in df_grid_roi.columns else None)
+ )
+ if _intensity_col is not None:
+ low_intensity_mask = (
+ df_grid_roi[_intensity_col].to_numpy() < DAPI_LOW_INTENSITY_FLOOR
+ )
+ else:
+ low_intensity_mask = np.zeros(len(df_grid_roi), dtype=bool)
+ bad_mask = blurred_mask | low_intensity_mask
+
+ usable_tissue_frac = (
+ float((qc_tissue_mask & (~bad_mask)).sum() / n_qc_tissue)
+ if n_qc_tissue > 0
+ else None
+ )
+
+ # Largest contiguous bad zone (4-neighbour) over ROI grid.
+ x_unique = np.unique(x1)
+ y_unique = np.unique(y1)
+ x_idx = np.searchsorted(x_unique, x1)
+ y_idx = np.searchsorted(y_unique, y1)
+ bad_grid = np.zeros((len(y_unique), len(x_unique)), dtype=bool)
+ tissue_grid = np.zeros((len(y_unique), len(x_unique)), dtype=bool)
+ bad_grid[y_idx, x_idx] = dapi_tissue_mask & bad_mask
+ tissue_grid[y_idx, x_idx] = dapi_tissue_mask
+ n_tissue_grid = int(tissue_grid.sum())
+ if n_tissue_grid > 0 and bad_grid.any():
+ labels, n_comp = ndimage.label(
+ bad_grid,
+ structure=np.array([[0, 1, 0], [1, 1, 1], [0, 1, 0]], dtype=np.uint8),
+ )
+ component_sizes = np.bincount(labels.ravel())
+ largest_bad_component = int(component_sizes[1:].max()) if n_comp > 0 else 0
+ cluster_zone_bad_fraction = float(largest_bad_component / n_tissue_grid)
+ else:
+ cluster_zone_bad_fraction = 0.0 if n_tissue_grid > 0 else None
+
+ roi_metrics["morphology"] = {
+ "edge_zone_frac": edge_zone_frac,
+ "hole_area_frac": hole_area_frac,
+ "usable_tissue_frac": usable_tissue_frac,
+ "cluster_zone_bad_fraction": cluster_zone_bad_fraction,
+ "edge_distance_threshold_px_ds": float(edge_distance_threshold),
+ "hole_distance_threshold_px_ds": float(hole_distance_threshold),
+ "min_tissue_coverage_for_qc": float(min_tissue_coverage_for_qc),
+ "n_tissue_rois": n_ext_tissue,
+ "n_qc_tissue_rois": n_qc_tissue,
+ }
+
+ # Optional: Laplacian sharpness summary for section-6 report metrics.
+ _qc = qc_thresholds or {}
+ _ch_dapi = (_qc.get("channels") or {}).get("DAPI") or {}
+ _focus_cut = _qc.get("focus") or {}
+ # The standalone absolute Laplacian-variance floor verdict was dropped
+ # (2026-06-22, qc_drift_analysis): lap_var scales with brightness² (rho≈0.91
+ # with DAPI) and collapses to near-zero on dim XOA-4.0 images, over-flagging
+ # good dim tissue. It is redundant with the GMM-2D classifier, which already
+ # uses lap_var as a feature. lap_var_median_raw is still emitted as an
+ # informational calibration stat below.
+
+ if {"dapi_lap_var", "tissue_coverage"}.issubset(df_grid_roi.columns):
+ lap_var = df_grid_roi["dapi_lap_var"].to_numpy(np.float64, copy=False)
+ tissue_cov = df_grid_roi["tissue_coverage"].to_numpy(np.float64, copy=False)
+ focus_col = (
+ "dapi_focus_score"
+ if "dapi_focus_score" in df_grid_roi.columns
+ else ("focus_score" if "focus_score" in df_grid_roi.columns else None)
+ )
+ focus_vals = (
+ df_grid_roi[focus_col].to_numpy(np.float64, copy=False)
+ if focus_col is not None
+ else np.full_like(lap_var, np.nan, dtype=np.float64)
+ )
+
+ mask = (
+ (tissue_cov >= float(min_tissue_coverage_for_qc))
+ & np.isfinite(lap_var)
+ & np.isfinite(focus_vals)
+ & (lap_var >= 0.0)
+ )
+ # P5: Exclude boundary ROIs (centre within window_size//2 of image edge)
+ if "is_boundary_roi" in df_grid_roi.columns:
+ boundary_mask = df_grid_roi["is_boundary_roi"].to_numpy(bool, copy=False)
+ mask = mask & ~boundary_mask
+ n_boundary_excluded = int(boundary_mask.sum())
+ else:
+ n_boundary_excluded = 0
+
+ n_used = int(mask.sum())
+ if n_used > 1:
+ lap_used = lap_var[mask]
+ focus_used = focus_vals[mask]
+
+ med = float(np.nanmedian(lap_used))
+ q25, q75 = np.nanpercentile(lap_used, [25, 75])
+ iqr = float(q75 - q25)
+
+ iqr_uniform = iqr <= 1e-12
+
+ # P4: Guard Spearman correlation on uniform slides
+ cv_focus = float(np.std(focus_used) / (np.mean(focus_used) + 1e-12))
+ cv_lap = float(np.std(lap_used) / (np.mean(lap_used) + 1e-12))
+ if cv_focus < 0.1 and cv_lap < 0.1:
+ focus_lap_corr = None
+ focus_lap_corr_status = "uniform"
+ else:
+ focus_lap_corr = float(
+ pd.Series(focus_used).corr(pd.Series(lap_used), method="spearman")
+ )
+ focus_lap_corr_status = None
+
+ # Separation of 2D-GMM blur components in log1p(lap_var) space (Cohen's d).
+ component_separation = None
+ if "is_blurred_gmm_2d" in df_grid_roi.columns:
+ blur2d = df_grid_roi["is_blurred_gmm_2d"].to_numpy(bool, copy=False)[
+ mask
+ ]
+ if blur2d.any() and (~blur2d).any():
+ lap_log = np.log1p(np.maximum(lap_used, 0.0))
+ lap_blur = lap_log[blur2d]
+ lap_focus = lap_log[~blur2d]
+ if lap_blur.size > 1 and lap_focus.size > 1:
+ n_b, n_f = lap_blur.size, lap_focus.size
+ v_blur = float(np.var(lap_blur, ddof=1))
+ v_focus = float(np.var(lap_focus, ddof=1))
+ pooled_sd = np.sqrt(
+ max(
+ ((n_b - 1) * v_blur + (n_f - 1) * v_focus)
+ / (n_b + n_f - 2),
+ 0.0,
+ )
+ + 1e-12
+ )
+ component_separation = float(
+ abs(float(np.mean(lap_focus)) - float(np.mean(lap_blur)))
+ / pooled_sd
+ )
+
+ roi_metrics["laplacian_sharpness"] = {
+ "lap_sigma": float(lap_sigma) if lap_sigma is not None else None,
+ "n_rois_used": n_used,
+ "n_boundary_excluded": n_boundary_excluded,
+ "min_tissue_coverage_for_qc": float(min_tissue_coverage_for_qc),
+ "iqr_uniform": iqr_uniform,
+ "lap_var_median_raw": med,
+ "focus_lap_spearman_corr": focus_lap_corr,
+ "focus_lap_corr_status": focus_lap_corr_status,
+ "component_separation": component_separation,
+ }
+
+ # Save to JSON
+ metrics_file = Path(outdir) / "roi_qc_metrics.json"
+ with open(metrics_file, "w") as f:
+ json.dump(roi_metrics, f, indent=2, default=str)
+ logging.info(f"[OK] Saved tile QC metrics to {metrics_file}")
+
+
+def save_pixel_focus_maps(focus_maps, outdir, compress=True):
+ """
+ Save per-pixel focus maps as compressed tiled TIFF files.
+
+ Uses lz4 compression (5-10x faster than zlib) and parallel I/O via
+ ThreadPoolExecutor to minimize wall-clock time on multi-map datasets.
+
+ Args:
+ focus_maps: dict returned by compute_all_focus_maps()
+ outdir: Path to output directory
+ compress: Use lz4 compression (default: True)
+ """
+ from concurrent.futures import ThreadPoolExecutor
+
+ outdir = Path(outdir)
+ items = [(name, arr) for name, arr in focus_maps.items() if arr is not None]
+
+ if not items:
+ logging.info(" No focus maps to save.")
+ return
+
+ def _save_one(name, array, outdir_path):
+ t_start = time.time()
+ path = outdir_path / f"{name}.tif"
+ tile = (256, 256) if array.shape[0] >= 256 and array.shape[1] >= 256 else None
+ tifffile.imwrite(
+ str(path),
+ np.asarray(array, dtype=np.float32),
+ compression="zstd" if compress else None,
+ tile=tile,
+ )
+ elapsed = time.time() - t_start
+ return name, path, array.shape, elapsed
+
+ with ThreadPoolExecutor(max_workers=min(len(items), 4)) as executor:
+ futures = [executor.submit(_save_one, name, arr, outdir) for name, arr in items]
+ for future in futures:
+ name, path, shape, elapsed = future.result()
+ logging.info(f" Saved {name}: {path} ({shape}, float32, {elapsed:.1f}s)")
+
+
+def save_versions_file(outdir):
+ """Save package versions to YAML file"""
+ import matplotlib
+
+ def get_version(package, package_name=None):
+ """Safely get version of a package"""
+ if package_name is None:
+ package_name = (
+ package.__name__ if hasattr(package, "__name__") else str(package)
+ )
+
+ try:
+ if hasattr(package, "__version__"):
+ return package.__version__
+ else:
+ # Try to get version via importlib
+ import importlib.metadata
+
+ return importlib.metadata.version(package_name)
+ except Exception:
+ return "unknown"
+
+ # Get Python version
+ python_version = (
+ f"{sys.version_info.major}.{sys.version_info.minor}.{sys.version_info.micro}"
+ )
+
+ # Get package versions safely
+ packages = {
+ "python": python_version,
+ "numpy": get_version(np, "numpy"),
+ "pandas": get_version(pd, "pandas"),
+ "matplotlib": get_version(matplotlib, "matplotlib"),
+ "seaborn": get_version(sns, "seaborn"),
+ "click": get_version(click, "click"),
+ "pathlib": "built-in",
+ }
+
+ # Get versions for other packages
+ try:
+ import skimage
+
+ packages["scikit-image"] = get_version(skimage, "scikit-image")
+ except Exception:
+ packages["scikit-image"] = "unknown"
+
+ try:
+ packages["tifffile"] = get_version(tifffile, "tifffile")
+ except Exception:
+ packages["tifffile"] = "unknown"
+
+ try:
+ packages["zarr"] = get_version(zarr, "zarr")
+ except Exception:
+ packages["zarr"] = "unknown"
+
+ try:
+ import napari_skimage_regionprops
+
+ packages["napari-skimage-regionprops"] = get_version(
+ napari_skimage_regionprops, "napari-skimage-regionprops"
+ )
+ except Exception:
+ packages["napari-skimage-regionprops"] = "unknown"
+
+ try:
+ packages["napari-simpleitk-image-processing"] = get_version(
+ nsitk, "napari-simpleitk-image-processing"
+ )
+ except Exception:
+ packages["napari-simpleitk-image-processing"] = "unknown"
+
+ # Create YAML content
+ yaml_content = f"""XENIUM_IMAGE_QC:
+ python: {packages["python"]}
+ numpy: {packages["numpy"]}
+ pandas: {packages["pandas"]}
+ matplotlib: {packages["matplotlib"]}
+ seaborn: {packages["seaborn"]}
+ scikit-image: {packages["scikit-image"]}
+ tifffile: {packages["tifffile"]}
+ zarr: {packages["zarr"]}
+ napari-skimage-regionprops: {packages["napari-skimage-regionprops"]}
+ napari-simpleitk-image-processing: {packages["napari-simpleitk-image-processing"]}
+ click: {packages["click"]}
+ pathlib: {packages["pathlib"]}
+"""
+
+ # Save to file
+ versions_file = outdir / "versions.yml"
+ with open(versions_file, "w") as f:
+ f.write(yaml_content)
+
+ logging.info(f"[OK] Saved package versions to {versions_file}")
+
+
+# ===== FUNCTIONS FROM image_qc_mapping_to_cells.py =====
+
+
+def load_spatial_data(data):
+ """Load and prepare spatial data. This only works for the original Xenium bundle generated from XOA"""
+
+ # Load cell masks
+ cell_masks_zarr = open_zarr(data["cell_masks_path"])
+ # cell_masks_zarr will only have one channel if its the result from xeniumranger import-segmentation
+ # Lazily sliced, not materialised: np.array() here cost 22 GB of uint32 on a
+ # 5.5 gigapixel sample. Its consumers are row-block reductions
+ # (_labeled_sums_chunked, LabeledSumAccumulator) and scattered point lookups,
+ # both of which LazyLabelPlane serves straight from zarr. The legacy
+ # --legacy-focus path calls regionprops_table, which needs a real array, and
+ # materialises it explicitly there.
+ cellseg_mask = LazyLabelPlane(cell_masks_zarr.get("masks").get("1"))
+
+ # Load spatial data
+ df_spatial = pd.read_parquet(
+ data["cells_parquet_path"],
+ columns=[
+ "x_centroid",
+ "y_centroid",
+ "transcript_counts",
+ "cell_area",
+ "nucleus_area",
+ "nucleus_count",
+ "segmentation_method",
+ "cell_id",
+ ],
+ )
+
+ # Rename columns for simplicity
+ df_spatial.rename(columns={"x_centroid": "x", "y_centroid": "y"}, inplace=True)
+
+ # Convert to pixel coordinates
+ df_spatial["x"] = (df_spatial.x / XENIUM_PIXEL_SIZE_UM).astype(int)
+ df_spatial["y"] = (df_spatial.y / XENIUM_PIXEL_SIZE_UM).astype(int)
+
+ # Measure CellID - OPTIMIZED: vectorized NumPy indexing (50-100x faster)
+ x_coords = np.clip(df_spatial["x"].values, 0, cellseg_mask.shape[1] - 1)
+ y_coords = np.clip(df_spatial["y"].values, 0, cellseg_mask.shape[0] - 1)
+ df_spatial["CellID"] = cellseg_mask[y_coords, x_coords]
+
+ # Load and merge clustering and UMAP data
+ df_clusters = pd.read_csv(data["clusters_csv_path"]).rename(
+ columns={"Barcode": "cell_id", "Cluster": "Cluster_kmeans10"}
+ )
+ df_UMAP = pd.read_csv(data["umap_path"]).rename(columns={"Barcode": "cell_id"})
+ df_spatial = df_spatial.merge(df_clusters, on="cell_id").merge(
+ df_UMAP, on="cell_id"
+ )
+ return df_spatial, cellseg_mask, cell_masks_zarr
+
+
+def map_distances_to_cells(
+ df_spatial,
+ distance_map,
+ distance_map2,
+ dense_intensity_regions,
+ downsample_factor=8,
+):
+ """
+ Map distance maps and dense intensity region masks to cells based on cell centroid locations.
+
+ Extracted from generate_sample_masks_and_distances() for cell-independent workflow.
+
+ Parameters:
+ -----------
+ df_spatial : pandas DataFrame
+ Cell DataFrame with 'x' and 'y' centroid columns
+ distance_map : numpy.ndarray
+ Distance to edge map (downsampled)
+ distance_map2 : numpy.ndarray
+ Distance to nearest hole map (downsampled)
+ dense_intensity_regions : numpy.ndarray
+ Dense intensity region mask (downsampled)
+ downsample_factor : int, optional
+ Downsampling factor (default: 8)
+
+ Returns:
+ --------
+ pandas DataFrame
+ df_spatial with added columns:
+ - Distance-to-edge: Distance to sample edge for each cell
+ - Distance-to-nearest-hole: Distance to nearest hole for each cell
+ - In-Area-with-Dense-Intensity-Region: Binary indicator if cell is in dense intensity region area
+ """
+ # Vectorized coordinate mapping -- compute pixel indices once, clipped to array bounds
+ X = np.clip(
+ (df_spatial["x"].values / downsample_factor).astype(int),
+ 0,
+ distance_map.shape[1] - 1,
+ )
+ Y = np.clip(
+ (df_spatial["y"].values / downsample_factor).astype(int),
+ 0,
+ distance_map.shape[0] - 1,
+ )
+
+ # Vectorized lookups -- three lines instead of three loops
+ df_spatial["Distance-to-edge"] = distance_map[Y, X]
+ df_spatial["Distance-to-nearest-hole"] = distance_map2[Y, X]
+ df_spatial["Dense-Intensity-Region-ID"] = dense_intensity_regions[Y, X].astype(int)
+
+ return df_spatial
+
+
+def map_grid_roi_to_cells(df_grid_roi, df_cells, overlapping=False):
+ """
+ Map cell-independent grid tile focus scores to cells based on cell centroid location.
+
+ This function replaces the previous cell-based tile approach with a cell-independent
+ grid-based tile approach that provides uniform coverage across the tissue.
+
+ For each cell, finds which tile its centroid (x, y) falls into and assigns
+ that tile's focus scores to the cell.
+
+ Parameters:
+ -----------
+ df_grid_roi : pandas DataFrame
+ Grid ROI DataFrame with columns: roi_id, x1, x2, y1, y2, focus_score,
+ focus_score_norm, raw_intensity, tissue_coverage, and channel-specific columns
+ df_cells : pandas DataFrame
+ Cell DataFrame with columns: x, y (centroid coordinates)
+ overlapping : bool, optional
+ Whether grid is overlapping. If True and cell falls into multiple ROIs,
+ uses ROI with highest tissue_coverage (default: False)
+
+ Returns:
+ --------
+ pandas DataFrame
+ Cell DataFrame with added columns:
+ - roi_id: Integer ID of ROI containing cell centroid (0, 1, 2, ...) or None if cell outside all ROIs
+ - DAPI_mean_roi: Mean intensity from grid ROI (NaN if cell outside all ROIs)
+ - DAPI_RFS_roi: Raw focus score from grid ROI (NaN if cell outside all ROIs)
+ - DAPI_RFSnorm_roi: Normalized focus score from grid ROI (NaN if cell outside all ROIs)
+ - roi_tissue_coverage: Tissue coverage of assigned ROI (NaN if cell outside all ROIs)
+ - Additional channel-specific columns if available (boundary_*, intrna_*)
+ """
+ # Create output DataFrame
+ df_cells_mapped = df_cells.copy()
+
+ # Initialize columns using old naming convention for compatibility
+ df_cells_mapped["roi_id"] = None
+ df_cells_mapped["DAPI_mean_roi"] = np.nan
+ df_cells_mapped["DAPI_RFS_roi"] = np.nan
+ df_cells_mapped["DAPI_RFSnorm_roi"] = np.nan
+ df_cells_mapped["roi_tissue_coverage"] = np.nan
+
+ # Check for additional channel columns
+ has_boundary = "boundary_focus_score" in df_grid_roi.columns
+ has_intrna = "intrna_focus_score" in df_grid_roi.columns
+
+ if has_boundary:
+ df_cells_mapped["boundary_focus_score_roi"] = np.nan
+ df_cells_mapped["boundary_focus_score_norm_roi"] = np.nan
+ df_cells_mapped["boundary_intensity_roi"] = np.nan
+
+ if has_intrna:
+ df_cells_mapped["intrna_focus_score_roi"] = np.nan
+ df_cells_mapped["intrna_focus_score_norm_roi"] = np.nan
+ df_cells_mapped["intrna_intensity_roi"] = np.nan
+
+ # OPTIMIZED: Fast vectorized lookup using 2D grid
+ logging.info(f" Mapping {len(df_cells):,} cells to {len(df_grid_roi):,} tiles...")
+
+ # Get cell coordinates as arrays (faster than DataFrame access)
+ cell_x = df_cells["x"].values
+ cell_y = df_cells["y"].values
+ n_cells = len(df_cells)
+
+ # Create arrays to store matches
+ cell_roi_matches = np.full(n_cells, -1, dtype=np.int32) # -1 means no match
+ cell_roi_tissue_coverage_arr = np.full(n_cells, np.nan, dtype=np.float64)
+
+ # Always try the fast vectorized approach: build a 2D lookup grid
+ is_regular_grid = False
+ if len(df_grid_roi) > 1 and not overlapping:
+ roi_size = df_grid_roi.iloc[0]["x2"] - df_grid_roi.iloc[0]["x1"]
+ stride = roi_size # non-overlapping grid
+
+ x_min = df_grid_roi["x1"].min()
+ y_min = df_grid_roi["y1"].min()
+ x_max = df_grid_roi["x1"].max()
+ y_max = df_grid_roi["y1"].max()
+
+ n_cols = int(round((x_max - x_min) / stride)) + 1
+ n_rows = int(round((y_max - y_min) / stride)) + 1
+
+ # Build 2D grid: grid[row, col] -> index in df_grid_roi
+ grid_lookup = np.full((n_rows, n_cols), -1, dtype=np.int32)
+ roi_x1_arr = df_grid_roi["x1"].values
+ roi_y1_arr = df_grid_roi["y1"].values
+ gx_arr = np.round((roi_x1_arr - x_min) / stride).astype(int)
+ gy_arr = np.round((roi_y1_arr - y_min) / stride).astype(int)
+
+ # Check if this is a valid grid (no out-of-bounds)
+ valid = (gx_arr >= 0) & (gx_arr < n_cols) & (gy_arr >= 0) & (gy_arr < n_rows)
+ if valid.all():
+ # Populate grid (vectorized)
+ grid_lookup[gy_arr, gx_arr] = np.arange(len(df_grid_roi))
+
+ # Vectorized lookup for all cells -- O(n_cells)
+ cell_gx = np.clip(
+ np.round((cell_x - x_min) / stride).astype(int), 0, n_cols - 1
+ )
+ cell_gy = np.clip(
+ np.round((cell_y - y_min) / stride).astype(int), 0, n_rows - 1
+ )
+ cell_roi_idx = grid_lookup[cell_gy, cell_gx] # vectorized!
+
+ matched_mask = cell_roi_idx >= 0
+ cell_roi_matches[matched_mask] = df_grid_roi["roi_id"].values[
+ cell_roi_idx[matched_mask]
+ ]
+ cell_roi_tissue_coverage_arr[matched_mask] = df_grid_roi[
+ "tissue_coverage"
+ ].values[cell_roi_idx[matched_mask]]
+
+ is_regular_grid = True
+ logging.info(
+ f" Using fast grid lookup (stride={stride}px, grid={n_cols}x{n_rows})..."
+ )
+
+ if not is_regular_grid:
+ # SLOW PATH: Irregular/filtered grid - fallback for grids that don't validate
+ if len(df_grid_roi) > 1000:
+ logging.info(
+ f" Processing {len(df_cells):,} cells against {len(df_grid_roi):,} tiles (irregular grid)..."
+ )
+
+ # Convert ROI boundaries to arrays for fast lookup
+ roi_x1 = df_grid_roi["x1"].values
+ roi_x2 = df_grid_roi["x2"].values
+ roi_y1 = df_grid_roi["y1"].values
+ roi_y2 = df_grid_roi["y2"].values
+ roi_ids = df_grid_roi["roi_id"].values
+ roi_tissue_coverage = df_grid_roi["tissue_coverage"].values
+
+ for cell_idx in range(n_cells):
+ x_cell = cell_x[cell_idx]
+ y_cell = cell_y[cell_idx]
+
+ matches = (
+ (roi_x1 <= x_cell)
+ & (x_cell < roi_x2)
+ & (roi_y1 <= y_cell)
+ & (y_cell < roi_y2)
+ )
+
+ if np.any(matches):
+ match_idx = np.where(matches)[0][0]
+ cell_roi_matches[cell_idx] = roi_ids[match_idx]
+ cell_roi_tissue_coverage_arr[cell_idx] = roi_tissue_coverage[match_idx]
+
+ # Check for GMM columns
+ has_gmm_1d = "is_blurred_gmm" in df_grid_roi.columns
+ has_gmm_2d = "is_blurred_gmm_2d" in df_grid_roi.columns
+ has_blur_prob_1d = "blur_prob_gmm" in df_grid_roi.columns
+ has_blur_prob_2d = "blur_prob_gmm_2d" in df_grid_roi.columns
+
+ # Initialize GMM columns if available
+ if has_gmm_1d:
+ df_cells_mapped["is_blurred_gmm_roi"] = False
+ if has_blur_prob_1d:
+ df_cells_mapped["blur_prob_gmm_roi"] = np.nan
+ if has_gmm_2d:
+ df_cells_mapped["is_blurred_gmm_2d_roi"] = False
+ if has_blur_prob_2d:
+ df_cells_mapped["blur_prob_gmm_2d_roi"] = np.nan
+
+ # Build vectorized column arrays from df_grid_roi for direct indexing
+ # Map roi_id -> index in df_grid_roi for O(1) lookup
+ roi_id_to_idx = pd.Series(
+ range(len(df_grid_roi)), index=df_grid_roi["roi_id"].values
+ )
+
+ # Resolve column names (handle legacy naming)
+ _dapi_intensity_col = (
+ "dapi_intensity" if "dapi_intensity" in df_grid_roi.columns else "raw_intensity"
+ )
+ _dapi_focus_col = (
+ "dapi_focus_score"
+ if "dapi_focus_score" in df_grid_roi.columns
+ else "focus_score"
+ )
+ _dapi_focus_norm_col = (
+ "dapi_focus_score_norm"
+ if "dapi_focus_score_norm" in df_grid_roi.columns
+ else "focus_score_norm"
+ )
+
+ dapi_intensity_arr = df_grid_roi[_dapi_intensity_col].values
+ dapi_focus_arr = df_grid_roi[_dapi_focus_col].values
+ dapi_focus_norm_arr = df_grid_roi[_dapi_focus_norm_col].values
+
+ # Assign ROI values to cells using vectorized operations
+ matched_mask = cell_roi_matches >= 0
+ matched_roi_ids = cell_roi_matches[matched_mask]
+
+ # Map matched roi_ids to df_grid_roi indices for vectorized column access
+ matched_df_indices = roi_id_to_idx[matched_roi_ids].values
+
+ # Assign ROI ID and tissue coverage
+ df_cells_mapped.loc[matched_mask, "roi_id"] = matched_roi_ids
+ df_cells_mapped.loc[matched_mask, "roi_tissue_coverage"] = (
+ cell_roi_tissue_coverage_arr[matched_mask]
+ )
+
+ # Assign DAPI data via direct array indexing (no dict lookups)
+ df_cells_mapped.loc[matched_mask, "DAPI_mean_roi"] = dapi_intensity_arr[
+ matched_df_indices
+ ]
+ df_cells_mapped.loc[matched_mask, "DAPI_RFS_roi"] = dapi_focus_arr[
+ matched_df_indices
+ ]
+ df_cells_mapped.loc[matched_mask, "DAPI_RFSnorm_roi"] = dapi_focus_norm_arr[
+ matched_df_indices
+ ]
+
+ # Assign GMM classifications if available
+ if has_gmm_1d:
+ df_cells_mapped.loc[matched_mask, "is_blurred_gmm_roi"] = df_grid_roi[
+ "is_blurred_gmm"
+ ].values[matched_df_indices]
+ if has_blur_prob_1d:
+ df_cells_mapped.loc[matched_mask, "blur_prob_gmm_roi"] = df_grid_roi[
+ "blur_prob_gmm"
+ ].values[matched_df_indices]
+ if has_gmm_2d:
+ df_cells_mapped.loc[matched_mask, "is_blurred_gmm_2d_roi"] = df_grid_roi[
+ "is_blurred_gmm_2d"
+ ].values[matched_df_indices]
+ if has_blur_prob_2d:
+ df_cells_mapped.loc[matched_mask, "blur_prob_gmm_2d_roi"] = df_grid_roi[
+ "blur_prob_gmm_2d"
+ ].values[matched_df_indices]
+
+ # Assign additional channel data if available
+ if has_boundary:
+ df_cells_mapped.loc[matched_mask, "boundary_focus_score_roi"] = df_grid_roi[
+ "boundary_focus_score"
+ ].values[matched_df_indices]
+ df_cells_mapped.loc[matched_mask, "boundary_focus_score_norm_roi"] = (
+ df_grid_roi["boundary_focus_score_norm"].values[matched_df_indices]
+ )
+ df_cells_mapped.loc[matched_mask, "boundary_intensity_roi"] = df_grid_roi[
+ "boundary_intensity"
+ ].values[matched_df_indices]
+
+ if has_intrna:
+ df_cells_mapped.loc[matched_mask, "intrna_focus_score_roi"] = df_grid_roi[
+ "intrna_focus_score"
+ ].values[matched_df_indices]
+ df_cells_mapped.loc[matched_mask, "intrna_focus_score_norm_roi"] = df_grid_roi[
+ "intrna_focus_score_norm"
+ ].values[matched_df_indices]
+ df_cells_mapped.loc[matched_mask, "intrna_intensity_roi"] = df_grid_roi[
+ "intrna_intensity"
+ ].values[matched_df_indices]
+
+ # Handle overlapping grids: if cell falls into multiple ROIs, use highest tissue_coverage
+ if overlapping:
+ # Find cells with multiple ROI assignments
+ cell_roi_counts = df_cells_mapped.groupby(df_cells_mapped.index)[
+ "roi_id"
+ ].count()
+ cells_with_multiple = cell_roi_counts[cell_roi_counts > 1].index
+
+ if len(cells_with_multiple) > 0:
+ # For each cell with multiple ROIs, find the one with highest tissue_coverage
+ for cell_idx in cells_with_multiple:
+ # Find all ROIs this cell falls into
+ x_cell = df_cells.loc[cell_idx, "x"]
+ y_cell = df_cells.loc[cell_idx, "y"]
+
+ matching_rois = df_grid_roi[
+ (df_grid_roi["x1"] <= x_cell)
+ & (x_cell < df_grid_roi["x2"])
+ & (df_grid_roi["y1"] <= y_cell)
+ & (y_cell < df_grid_roi["y2"])
+ ]
+
+ if len(matching_rois) > 0:
+ # Select ROI with highest tissue_coverage
+ best_roi = matching_rois.loc[
+ matching_rois["tissue_coverage"].idxmax()
+ ]
+
+ # Update cell with best ROI
+ df_cells_mapped.loc[cell_idx, "roi_id"] = best_roi["roi_id"]
+ df_cells_mapped.loc[cell_idx, "DAPI_mean_roi"] = best_roi.get(
+ "dapi_intensity", best_roi.get("raw_intensity", np.nan)
+ )
+ df_cells_mapped.loc[cell_idx, "DAPI_RFS_roi"] = best_roi.get(
+ "dapi_focus_score", best_roi.get("focus_score", np.nan)
+ )
+ df_cells_mapped.loc[cell_idx, "DAPI_RFSnorm_roi"] = best_roi.get(
+ "dapi_focus_score_norm",
+ best_roi.get("focus_score_norm", np.nan),
+ )
+ df_cells_mapped.loc[cell_idx, "roi_tissue_coverage"] = best_roi[
+ "tissue_coverage"
+ ]
+
+ # Update GMM classifications if available
+ if has_gmm_1d:
+ df_cells_mapped.loc[cell_idx, "is_blurred_gmm_roi"] = (
+ best_roi.get("is_blurred_gmm", False)
+ )
+ if has_blur_prob_1d:
+ df_cells_mapped.loc[cell_idx, "blur_prob_gmm_roi"] = (
+ best_roi.get("blur_prob_gmm", np.nan)
+ )
+ if has_gmm_2d:
+ df_cells_mapped.loc[cell_idx, "is_blurred_gmm_2d_roi"] = (
+ best_roi.get("is_blurred_gmm_2d", False)
+ )
+ if has_blur_prob_2d:
+ df_cells_mapped.loc[cell_idx, "blur_prob_gmm_2d_roi"] = (
+ best_roi.get("blur_prob_gmm_2d", np.nan)
+ )
+
+ if has_boundary:
+ df_cells_mapped.loc[cell_idx, "boundary_focus_score_roi"] = (
+ best_roi.get("boundary_focus_score", np.nan)
+ )
+ df_cells_mapped.loc[
+ cell_idx, "boundary_focus_score_norm_roi"
+ ] = best_roi.get("boundary_focus_score_norm", np.nan)
+ df_cells_mapped.loc[cell_idx, "boundary_intensity_roi"] = (
+ best_roi.get("boundary_intensity", np.nan)
+ )
+
+ if has_intrna:
+ df_cells_mapped.loc[cell_idx, "intrna_focus_score_roi"] = (
+ best_roi.get("intrna_focus_score", np.nan)
+ )
+ df_cells_mapped.loc[cell_idx, "intrna_focus_score_norm_roi"] = (
+ best_roi.get("intrna_focus_score_norm", np.nan)
+ )
+ df_cells_mapped.loc[cell_idx, "intrna_intensity_roi"] = (
+ best_roi.get("intrna_intensity", np.nan)
+ )
+
+ return df_cells_mapped
+
+
+def load_roi_blur_threshold(outdir):
+ """
+ Load ROI blur threshold configuration from JSON file saved by image_qc_roi_processing.py.
+
+ Parameters:
+ -----------
+ outdir : Path
+ Output directory where roi_blur_threshold.json should be located
+
+ Returns:
+ --------
+ tuple
+ (roi_focus_score_threshold, roi_intensity_threshold) or (None, None) if not found
+ """
+ threshold_json = Path(outdir) / "roi_blur_threshold.json"
+ if not threshold_json.exists():
+ return None, None
+
+ try:
+ with open(threshold_json, "r") as f:
+ threshold_config = json.load(f)
+ roi_threshold = threshold_config.get("roi_focus_score_threshold")
+ intensity_threshold = threshold_config.get("roi_intensity_threshold")
+ return roi_threshold, intensity_threshold
+ except (json.JSONDecodeError, KeyError) as e:
+ logging.warning(f" Warning: Could not load threshold configuration: {e}")
+ return None, None
+
+
+def create_final_merged_data(
+ df_spatial,
+ myData,
+ roi_data=None,
+ ccfs_threshold=DEFAULT_CCFS_LOW_TEXTURE_THRESHOLD,
+ edge_distance_threshold=-25,
+ hole_distance_threshold=-25,
+ roi_threshold=None,
+ roi_intensity_threshold=None,
+):
+ """Create final merged dataset with boolean columns for thresholds"""
+
+ # Link the Spatial Cell matrix (df_spatial) with the FocusScore table
+ new_df = pd.merge(df_spatial, myData, how="inner", on="CellID")
+
+ # Create new segmentation names with colour palette for plotting.
+ # XR import-segmentation (e.g. from the CROP subworkflow) writes
+ # `segmentation_method = 'Imported Cell Segmentation'` into cells.parquet,
+ # which doesn't substring-match any of the three OBA categories
+ # ("boundary"/"interior"/"nucleus"). Without this 4th row the match-filter
+ # below drops every cell and downstream figures fail on empty arrays.
+ seg = pd.DataFrame(
+ {
+ "segmentation": [
+ "boundary",
+ "interior",
+ "nucleus",
+ "Imported Cell Segmentation",
+ ],
+ "segPal": ["#FFABC3", "#A9A800", "#A9CEFF", "#7FBFFF"],
+ "join": [1, 1, 1, 1],
+ }
+ )
+
+ # Link the Spatial Cell matrix to the new segmentation colour palette
+ new_df["join"] = 1
+ new_df = new_df.merge(seg, on="join").drop("join", axis=1)
+ seg.drop("join", axis=1, inplace=True)
+ new_df["match"] = new_df.apply(
+ lambda x: x.segmentation_method.find(x.segmentation), axis=1
+ ).ge(0)
+ new_df = new_df[new_df["match"]]
+
+ # Add boolean columns for various thresholds (nuclei-based method)
+ new_df["is_low_nuclear_texture"] = new_df["CCFS_DAPI"] <= ccfs_threshold
+ new_df["is_high_nuclear_texture"] = new_df["CCFS_DAPI"] > ccfs_threshold
+ new_df["is_near_edge"] = new_df["Distance-to-edge"] > edge_distance_threshold
+ new_df["is_far_from_edge"] = new_df["Distance-to-edge"] <= edge_distance_threshold
+ new_df["is_near_hole"] = (
+ new_df["Distance-to-nearest-hole"] > hole_distance_threshold
+ )
+ new_df["is_far_from_hole"] = (
+ new_df["Distance-to-nearest-hole"] <= hole_distance_threshold
+ )
+ # Create boolean indicator for cells in any dense intensity region (backward compatibility)
+ new_df["has_dense_intensity_regions"] = new_df["Dense-Intensity-Region-ID"] > 0
+
+ # Merge ROI data if provided
+ calculated_roi_threshold = None
+ if roi_data is not None:
+ # Merge ROI data using x and y coordinates
+ new_df = pd.merge(
+ new_df, roi_data, on=["x", "y"], how="left", suffixes=("", "_roi_merge")
+ )
+
+ # Use new threshold approach: raw score percentile + intensity threshold
+ # If roi_threshold is provided (e.g., for backward compatibility), use it
+ # Otherwise, calculate from raw scores using configured parameters
+ if roi_threshold is None:
+ # roi_threshold should have been calculated from df_grid_roi and passed here
+ # If not provided, we can't calculate it here (need df_grid_roi)
+ # This should not happen in normal flow, but provide fallback
+ logging.warning(
+ " Warning: roi_threshold not provided, using default calculation"
+ )
+ roi_threshold = -1.0 # Old default for backward compatibility
+
+ # Use intensity threshold from configuration if not provided
+ # If None, it means threshold config wasn't loaded (backward compatibility)
+ # In that case, skip intensity check and only use focus score threshold
+ if roi_intensity_threshold is None:
+ roi_intensity_threshold = (
+ None # Will skip intensity check in classification
+ )
+
+ calculated_roi_threshold = roi_threshold
+
+ # Add boolean columns for tile-based blur detection
+ # New approach: blurred if (raw_focus_score <= threshold) OR (intensity < intensity_threshold)
+ # Use raw scores (DAPI_RFS_roi) not normalized (DAPI_RFSnorm_roi)
+ has_intensity = "DAPI_mean_roi" in new_df.columns
+ has_raw_focus = "DAPI_RFS_roi" in new_df.columns
+
+ if has_raw_focus:
+ # Combined threshold: focus score OR intensity (if intensity threshold provided)
+ if has_intensity and roi_intensity_threshold is not None:
+ new_df["is_blurred_roi"] = (new_df["DAPI_RFS_roi"] <= roi_threshold) | (
+ new_df["DAPI_mean_roi"] < roi_intensity_threshold
+ )
+ else:
+ # Fallback: just use focus score if intensity not available or threshold not provided
+ new_df["is_blurred_roi"] = new_df["DAPI_RFS_roi"] <= roi_threshold
+ else:
+ # Fallback: use normalized scores if raw scores not available (backward compatibility)
+ new_df["is_blurred_roi"] = new_df["DAPI_RFSnorm_roi"] <= roi_threshold
+
+ new_df["is_high_focus_roi"] = ~new_df["is_blurred_roi"]
+
+ return new_df, calculated_roi_threshold
+
+
+# ===== CELL-LEVEL PLOTTING FUNCTIONS =====
+
+
+def plot_nuclear_texture_proportions(
+ new_df,
+ figures_dir,
+ figures_source_dir,
+ texture_threshold=DEFAULT_CCFS_LOW_TEXTURE_THRESHOLD,
+ GROUP_BY_COLUMN="Cluster_kmeans10",
+):
+ """
+ Create a stacked bar plot showing proportion of cells with high and low blur scores per cluster.
+
+ Parameters:
+ -----------
+ new_df : pandas DataFrame
+ DataFrame containing CCFS_DAPI and Cluster_kmeans10 columns
+ texture_threshold : float, optional
+ Threshold to classify cells as high/low nuclear texture (default: DEFAULT_CCFS_LOW_TEXTURE_THRESHOLD)
+ """
+ # Create nuclear texture categories (create temporary column to avoid modifying original)
+ df_temp = new_df.copy()
+ # Low CCFS_DAPI values (<= threshold) = low nuclear texture quality
+ # High CCFS_DAPI values (> threshold) = high nuclear texture quality
+ df_temp["texture_category"] = np.where(
+ df_temp["CCFS_DAPI"] <= texture_threshold, "Low Quality", "High Quality"
+ )
+
+ # Calculate proportions
+ proportions = (
+ df_temp.groupby([GROUP_BY_COLUMN, "texture_category"])
+ .size()
+ .unstack(fill_value=0)
+ )
+ proportions = (
+ proportions.div(proportions.sum(axis=1), axis=0) * 100
+ ) # Convert to percentages
+
+ # Reorder so 'Low Quality' (red) is plotted first → at the bottom of
+ # each stacked bar. Matches the convention in plot_tile_blur_proportions_roi
+ # (Blurred at bottom). pandas stacks columns from bottom up in the order
+ # they appear in the DataFrame.
+ proportions = proportions[["Low Quality", "High Quality"]]
+
+ # Set up the plot
+ fig = plt.figure(figsize=(12, 6))
+
+ # Create stacked bar plot with specified colors
+ # Column order is Low Quality (red, bottom) then High Quality (blue, top).
+ proportions.plot(
+ kind="bar",
+ stacked=True,
+ color=[
+ "#d62728",
+ "#1f77b4",
+ ], # Red for low quality (bottom), blue for high quality (top)
+ figsize=(12, 6),
+ )
+
+ # Customize the plot
+ plt.title(
+ f"Proportion of Cells by Nuclear Texture Quality (Threshold: {texture_threshold})",
+ fontsize=14,
+ pad=20,
+ )
+ plt.xlabel(GROUP_BY_COLUMN, fontsize=12)
+ plt.ylabel("Percentage of Cells", fontsize=12)
+
+ # Rotate x-axis labels if needed
+ plt.xticks(rotation=45, ha="right")
+
+ # Add percentage labels on the bars
+ for c in plt.gca().containers:
+ # Add labels
+ plt.gca().bar_label(c, fmt="%.1f%%", label_type="center")
+
+ # Add a grid for better readability
+ plt.grid(True, axis="y", linestyle="--", alpha=0.7)
+
+ # Adjust layout to prevent label cutoff
+ plt.tight_layout()
+
+ # Save figure instead of showing
+ plt.savefig(
+ figures_dir / "nuclear_texture_proportions.png", dpi=300, bbox_inches="tight"
+ )
+ plt.close(fig)
+
+ # Save data as CSV
+ df_texture_proportions = (
+ df_temp.groupby([GROUP_BY_COLUMN, "texture_category"])
+ .size()
+ .reset_index(name="count")
+ )
+ if figures_source_dir is not None:
+ df_texture_proportions.to_csv(
+ figures_source_dir / "nuclear_texture_proportions.csv", index=False
+ )
+
+ # Print summary statistics
+ logging.info("Summary statistics by cluster:")
+ summary_stats = (
+ df_temp.groupby([GROUP_BY_COLUMN, "texture_category"])
+ .size()
+ .unstack(fill_value=0)
+ )
+ logging.info("Count of cells by texture category:")
+ logging.info(summary_stats)
+ logging.info("Percentage of cells by texture category:")
+ logging.info(summary_stats.div(summary_stats.sum(axis=1), axis=0) * 100)
+
+
+def plot_tile_blur_proportions_roi(
+ new_df,
+ figures_dir,
+ figures_source_dir,
+ roi_threshold=-1.0,
+ GROUP_BY_COLUMN="Cluster_kmeans10",
+):
+ """
+ Create a stacked bar plot showing proportion of cells with high and low tile-based blur scores per cluster.
+ Uses GMM 2D classification if available, otherwise falls back to threshold-based method.
+
+ Parameters:
+ -----------
+ new_df : pandas DataFrame
+ DataFrame containing DAPI_RFSnorm_roi, is_blurred_roi (or is_blurred_gmm_2d_roi), and Cluster_kmeans10 columns
+ roi_threshold : float, optional
+ Threshold to classify cells as high/low blur (default: -1.0) - used only if GMM 2D not available
+ """
+ # Check if ROI data is available - prefer GMM 2D, fall back to threshold-based
+ has_gmm_2d = "is_blurred_gmm_2d_roi" in new_df.columns
+ has_threshold = "is_blurred_roi" in new_df.columns
+
+ if not has_gmm_2d and not has_threshold:
+ logging.warning(
+ "Warning: Tile-based blur scores not available. Skipping tile blur score proportions plot."
+ )
+ return
+
+ # Create blur score categories (create temporary column to avoid modifying original)
+ df_temp = new_df.copy()
+
+ # Use GMM 2D if available, otherwise use threshold-based
+ if has_gmm_2d:
+ df_temp["blur_category"] = np.where(
+ df_temp["is_blurred_gmm_2d_roi"], "Blurred", "In Focus"
+ )
+ method_name = "2D GMM"
+ else:
+ df_temp["blur_category"] = np.where(
+ df_temp["is_blurred_roi"], "Blurred", "In Focus"
+ )
+ method_name = "Threshold"
+
+ # Calculate proportions
+ proportions = (
+ df_temp.groupby([GROUP_BY_COLUMN, "blur_category"]).size().unstack(fill_value=0)
+ )
+ proportions = (
+ proportions.div(proportions.sum(axis=1), axis=0) * 100
+ ) # Convert to percentages
+
+ # Set up the plot
+ fig = plt.figure(figsize=(12, 6))
+
+ # Create stacked bar plot with specified colors
+ # Colors are applied in alphabetical order: 'Blurred' (red), 'In Focus' (blue)
+ proportions.plot(
+ kind="bar",
+ stacked=True,
+ color=["#d62728", "#1f77b4"], # Red for blurred, blue for in focus
+ figsize=(12, 6),
+ )
+
+ # Customize the plot
+ if has_gmm_2d:
+ plt.title(
+ "Proportion of Cells by Tile-based Blur Score (2D GMM Classification)",
+ fontsize=14,
+ pad=20,
+ )
+ else:
+ # Note: roi_threshold is in raw score units, classification uses combined approach
+ # Try to get intensity threshold from the threshold config if available
+ intensity_threshold_display = getattr(
+ plot_tile_blur_proportions_roi, "_intensity_threshold", None
+ )
+ if intensity_threshold_display is None:
+ intensity_threshold_display = 20.0 # Default
+ plt.title(
+ f"Proportion of Cells by Tile-based Blur Score\n(Raw score threshold: {roi_threshold:.2f}, Intensity threshold: {intensity_threshold_display})",
+ fontsize=14,
+ pad=20,
+ )
+ plt.xlabel(GROUP_BY_COLUMN, fontsize=12)
+ plt.ylabel("Percentage of Cells", fontsize=12)
+
+ # Rotate x-axis labels if needed
+ plt.xticks(rotation=45, ha="right")
+
+ # Add percentage labels on the bars
+ for c in plt.gca().containers:
+ # Add labels
+ plt.gca().bar_label(c, fmt="%.1f%%", label_type="center")
+
+ # Add a grid for better readability
+ plt.grid(True, axis="y", linestyle="--", alpha=0.7)
+
+ # Adjust layout to prevent label cutoff
+ plt.tight_layout()
+
+ # Save figure instead of showing
+ plt.savefig(
+ figures_dir / "tile_blur_proportions_roi.png", dpi=300, bbox_inches="tight"
+ )
+ plt.close(fig)
+
+ # Save data as CSV
+ df_blur_proportions = (
+ df_temp.groupby([GROUP_BY_COLUMN, "blur_category"])
+ .size()
+ .reset_index(name="count")
+ )
+ if figures_source_dir is not None:
+ df_blur_proportions.to_csv(
+ figures_source_dir / "tile_blur_proportions_roi.csv", index=False
+ )
+
+ # Print summary statistics
+ logging.info(f"Summary statistics by cluster (tile-based, {method_name}):")
+ summary_stats = (
+ df_temp.groupby([GROUP_BY_COLUMN, "blur_category"]).size().unstack(fill_value=0)
+ )
+ logging.info("Count of cells by blur category:")
+ logging.info(summary_stats)
+ logging.info("Percentage of cells by blur category:")
+ logging.info(summary_stats.div(summary_stats.sum(axis=1), axis=0) * 100)
+
+
+def plot_cell_focus_distribution(
+ new_df,
+ figures_dir,
+ figures_source_dir,
+ GROUP_BY_COLUMN="Cluster_kmeans10",
+):
+ """Plot cell-level focus score distributions using tile-propagated metrics.
+
+ Creates a 1×2 panel (v4 r5 trim):
+ - Left: histogram of DAPI_RFSnorm_roi (tile focus score per cell)
+ - Right: histogram coloured by GMM blur classification
+
+ The previous bottom row (focus density per cluster, blur prob density per
+ cluster) was clustering-related and overlapped with figures in the
+ Per-cluster subsection of 8.B. The blur-prob-density-per-cluster view
+ moved to its own dedicated function `plot_blur_prob_density_by_cluster`;
+ the focus-density-per-cluster view was dropped as redundant with the
+ nuclear texture density per cluster figure.
+ """
+ has_focus = "DAPI_RFSnorm_roi" in new_df.columns
+ has_gmm_prob = (
+ "blur_prob_gmm_2d_roi" in new_df.columns
+ ) # used for source-CSV column inclusion below
+ has_gmm_class = "is_blurred_gmm_2d_roi" in new_df.columns
+
+ if not has_focus:
+ logging.warning(
+ "No DAPI_RFSnorm_roi column — skipping cell focus distribution plot"
+ )
+ return
+
+ fig, axes = plt.subplots(1, 2, figsize=(16, 6))
+
+ focus_vals = new_df["DAPI_RFSnorm_roi"].dropna()
+
+ # --- Left: overall focus score histogram ---
+ ax = axes[0]
+ ax.hist(focus_vals, bins=60, alpha=0.7, edgecolor="black", color="steelblue")
+ ax.set_xlabel("Tile focus score (DAPI_RFSnorm_roi)", fontsize=11)
+ ax.set_ylabel("Number of cells", fontsize=11)
+ ax.set_title("Overall histogram of tile focus scores across all cells", fontsize=12)
+ stats_text = f"n={len(focus_vals):,}\nMedian={focus_vals.median():.4f}\nMean={focus_vals.mean():.4f}"
+ ax.text(
+ 0.95,
+ 0.95,
+ stats_text,
+ transform=ax.transAxes,
+ va="top",
+ ha="right",
+ bbox=dict(boxstyle="round", facecolor="wheat", alpha=0.7),
+ fontsize=10,
+ )
+ ax.grid(True, axis="y", linestyle="--", alpha=0.7)
+
+ # --- Right: histogram split by GMM blur class ---
+ ax = axes[1]
+ if has_gmm_class:
+ gmm_col = new_df["is_blurred_gmm_2d_roi"].astype(bool)
+ valid = new_df["DAPI_RFSnorm_roi"].notna()
+ blurred = new_df.loc[valid & gmm_col, "DAPI_RFSnorm_roi"]
+ sharp = new_df.loc[valid & ~gmm_col, "DAPI_RFSnorm_roi"]
+ ax.hist(
+ sharp,
+ bins=60,
+ alpha=0.6,
+ label=f"In focus ({len(sharp):,})",
+ color="#1f77b4",
+ )
+ ax.hist(
+ blurred,
+ bins=60,
+ alpha=0.6,
+ label=f"Blurred ({len(blurred):,})",
+ color="#d62728",
+ )
+ ax.legend(fontsize=10)
+ ax.set_title(
+ "Histogram split by GMM blur classification (blue = in focus, red = blurred)",
+ fontsize=12,
+ )
+ else:
+ ax.hist(focus_vals, bins=60, alpha=0.7, color="steelblue")
+ ax.set_title(
+ "Histogram split by GMM blur classification (no GMM data)", fontsize=12
+ )
+ ax.set_xlabel("Tile focus score (DAPI_RFSnorm_roi)", fontsize=11)
+ ax.set_ylabel("Number of cells", fontsize=11)
+ ax.grid(True, axis="y", linestyle="--", alpha=0.7)
+
+ plt.tight_layout()
+ plt.savefig(
+ figures_dir / "cell_focus_distribution.png", dpi=300, bbox_inches="tight"
+ )
+ plt.close(fig)
+
+ # Save source data
+ src_cols = ["DAPI_RFSnorm_roi"]
+ if has_gmm_class:
+ src_cols.append("is_blurred_gmm_2d_roi")
+ if has_gmm_prob:
+ src_cols.append("blur_prob_gmm_2d_roi")
+ if GROUP_BY_COLUMN in new_df.columns:
+ src_cols.append(GROUP_BY_COLUMN)
+ src = new_df[[c for c in src_cols if c in new_df.columns]].dropna(
+ subset=["DAPI_RFSnorm_roi"]
+ )
+ if figures_source_dir is not None:
+ src.to_csv(figures_source_dir / "cell_focus_distribution.csv", index=False)
+
+
+def plot_tile_focus_gmm_spatial(new_df, figures_dir, figures_source_dir):
+ """Single-panel spatial map of tile-based focus, coloured by GMM 2D blur
+ classification (red = blurred, blue = in-focus). Falls back to RFSnorm
+ viridis colormap when GMM 2D classification is unavailable.
+
+ Replaces the right panel of the now-latent `plot_spatial_comparison` 2-panel
+ figure (v4 r5: spatial concordance dropped per user feedback — known low
+ concordance not informative; tile-blur spatial overview kept on its own).
+ """
+ if "DAPI_RFSnorm_roi" not in new_df.columns:
+ logging.warning("No DAPI_RFSnorm_roi — skipping tile-focus GMM spatial plot")
+ return
+ has_gmm_2d = "is_blurred_gmm_2d_roi" in new_df.columns
+ valid_mask = new_df["DAPI_RFSnorm_roi"].notna()
+ if valid_mask.sum() == 0:
+ logging.warning(
+ "No valid DAPI_RFSnorm_roi values — skipping tile-focus GMM spatial plot"
+ )
+ return
+
+ # Phase 11 (v5): aspect-adaptive figure size matching the slide proportions
+ # rather than a fixed (8, 8) square. Mirrors plot_grid_roi_focus_heatmap and
+ # plot_snr_roi_heatmap so all whole-sample maps render at consistent width.
+ _x = new_df.loc[valid_mask, "x"]
+ _y = new_df.loc[valid_mask, "y"]
+ _x_range = float(_x.max() - _x.min()) if len(_x) else 1.0
+ _y_range = float(_y.max() - _y.min()) if len(_y) else 1.0
+ _img_aspect = _y_range / _x_range if _x_range > 0 else 1.0
+ _panel_width = 6
+ _panel_height = max(_panel_width * 0.5, _panel_width * _img_aspect)
+ fig, ax = plt.subplots(1, 1, figsize=(_panel_width, _panel_height))
+ if has_gmm_2d:
+ gmm_valid = valid_mask & new_df["is_blurred_gmm_2d_roi"].notna()
+ if gmm_valid.sum() > 0:
+ colors = [
+ "red" if blurred else "blue"
+ for blurred in new_df.loc[gmm_valid, "is_blurred_gmm_2d_roi"]
+ ]
+ ax.scatter(
+ new_df.loc[gmm_valid, "x"],
+ -new_df.loc[gmm_valid, "y"],
+ s=0.1,
+ c=colors,
+ rasterized=True,
+ )
+ ax.set_title(
+ "Focus score, coloured by GMM 2D blur classification\n"
+ "(red = blurred, blue = in focus)",
+ fontsize=12,
+ )
+ else:
+ ax.text(
+ 0.5,
+ 0.5,
+ "No valid GMM 2D classification data",
+ transform=ax.transAxes,
+ ha="center",
+ va="center",
+ fontsize=14,
+ )
+ else:
+ roi_values = new_df.loc[valid_mask, "DAPI_RFSnorm_roi"]
+ scatter = ax.scatter(
+ new_df.loc[valid_mask, "x"],
+ -new_df.loc[valid_mask, "y"],
+ s=0.1,
+ c=roi_values,
+ cmap="viridis",
+ rasterized=True,
+ )
+ ax.set_title("Focus score (DAPI_RFSnorm_roi)", fontsize=12)
+ plt.colorbar(scatter, ax=ax, label="DAPI_RFSnorm_roi")
+ ax.set_facecolor("black")
+ ax.set_aspect("equal")
+ plt.tight_layout()
+ plt.savefig(
+ figures_dir / "tile_focus_gmm_spatial.png", dpi=300, bbox_inches="tight"
+ )
+ plt.close(fig)
+ # Save source data
+ src_cols = ["x", "y", "DAPI_RFSnorm_roi"]
+ if has_gmm_2d:
+ src_cols.append("is_blurred_gmm_2d_roi")
+ src = new_df[[c for c in src_cols if c in new_df.columns]].dropna(
+ subset=["DAPI_RFSnorm_roi"]
+ )
+ if figures_source_dir is not None:
+ src.to_csv(figures_source_dir / "tile_focus_gmm_spatial.csv", index=False)
+
+
+def plot_cell_flagged_maps_combined(new_df, figures_dir, figures_source_dir):
+ """Two-panel spatial map combining the nuclear texture flagged and
+ blurriness flagged cell views into one figure for direct visual
+ comparison. Left panel mirrors the standalone ccfs_thresholded.png;
+ right panel mirrors the standalone tile_focus_gmm_spatial.png. Both
+ panels share the sample's coordinate system and use the aspect-adaptive
+ sizing pattern from plot_snr_roi_heatmap so the rendered figure aligns
+ with the §3.4 SNR heatmap panel layout.
+ """
+ needed = ("is_low_nuclear_texture", "is_blurred_gmm_2d_roi")
+ if all(c not in new_df.columns for c in needed):
+ logging.warning(
+ "No nuclear texture or blurry-(GMM) columns — skipping combined cell-flagged maps"
+ )
+ return
+ if "x" not in new_df.columns or "y" not in new_df.columns:
+ logging.warning("No x / y coordinates — skipping combined cell-flagged maps")
+ return
+
+ _x = new_df["x"].dropna()
+ _y = new_df["y"].dropna()
+ if len(_x) == 0 or len(_y) == 0:
+ logging.warning("Empty x / y — skipping combined cell-flagged maps")
+ return
+ _x_range = float(_x.max() - _x.min()) if _x.max() > _x.min() else 1.0
+ _y_range = float(_y.max() - _y.min()) if _y.max() > _y.min() else 1.0
+ _img_aspect = _y_range / _x_range if _x_range > 0 else 1.0
+ panel_width = 6
+ panel_height = max(panel_width * 0.5, panel_width * _img_aspect)
+ fig, axes = plt.subplots(1, 2, figsize=(2 * panel_width + 2, panel_height))
+
+ # ── Left panel: nuclear texture-flagged ────────────────────────────
+ ax = axes[0]
+ if "is_low_nuclear_texture" in new_df.columns:
+ _mask_low = new_df["is_low_nuclear_texture"].fillna(False).astype(bool)
+ _high = new_df[~_mask_low]
+ _low = new_df[_mask_low]
+ if len(_high) > 0:
+ ax.scatter(_high["x"], -_high["y"], s=0.1, color="#64B5F6", rasterized=True)
+ if len(_low) > 0:
+ # Slightly larger red points so flagged cells remain visible
+ # against the dense blue background and on the white figure bg.
+ ax.scatter(_low["x"], -_low["y"], s=0.5, color="red", rasterized=True)
+ ax.set_title(
+ "Spatial distribution of low nuclear texture cells\n(red = low nuclear texture, blue = high)",
+ fontsize=14,
+ )
+ ax.set_aspect("equal")
+ ax.set_xticks([])
+ ax.set_yticks([])
+ for _spine in ax.spines.values():
+ _spine.set_edgecolor("black")
+ _spine.set_linewidth(1.2)
+
+ # ── Right panel: blurriness-flagged (tile-GMM inherited) ───────────
+ ax = axes[1]
+ if "is_blurred_gmm_2d_roi" in new_df.columns:
+ _valid_mask = new_df["is_blurred_gmm_2d_roi"].notna()
+ if _valid_mask.any():
+ _valid_df = new_df.loc[_valid_mask]
+ _blur_mask = _valid_df["is_blurred_gmm_2d_roi"].astype(bool)
+ _in_focus = _valid_df[~_blur_mask]
+ _blurred = _valid_df[_blur_mask]
+ if len(_in_focus) > 0:
+ ax.scatter(
+ _in_focus["x"],
+ -_in_focus["y"],
+ s=0.1,
+ color="#64B5F6",
+ rasterized=True,
+ )
+ if len(_blurred) > 0:
+ # Same size as in-focus: GMM blur flags typically cover contiguous
+ # regions, so enlargement is not needed for visibility and would
+ # swamp the panel.
+ ax.scatter(
+ _blurred["x"], -_blurred["y"], s=0.1, color="red", rasterized=True
+ )
+ ax.set_title(
+ "Spatial distribution of blurry cells\n(red = blurred, blue = in focus)",
+ fontsize=14,
+ )
+ ax.set_aspect("equal")
+ ax.set_xticks([])
+ ax.set_yticks([])
+ for _spine in ax.spines.values():
+ _spine.set_edgecolor("black")
+ _spine.set_linewidth(1.2)
+
+ plt.tight_layout()
+ plt.savefig(figures_dir / "cell_flagged_maps.png", dpi=300, bbox_inches="tight")
+ plt.close(fig)
+
+
+def plot_blur_prob_density_by_cluster(
+ new_df,
+ figures_dir,
+ figures_source_dir,
+ GROUP_BY_COLUMN="Cluster_kmeans10",
+):
+ """Single-panel KDE of `blur_prob_gmm_2d_roi` per expression cluster, with
+ a vertical line at the blur threshold (0.5). Parallels the existing
+ nuclear texture density by cluster figure (CCFS density per cluster) but
+ for tile-blur probability — sits in 8.B's Per-cluster subsection.
+
+ Extracted from the bottom-right panel of the v4-r4 `plot_cell_focus_distribution`
+ (which was a 2×2 grid; trimmed to 1×2 in v4 r5 because the bottom row
+ was clustering-related and belonged in the Per-cluster subsection).
+ """
+ if "blur_prob_gmm_2d_roi" not in new_df.columns:
+ logging.warning(
+ "No blur_prob_gmm_2d_roi column — skipping per-cluster blur probability density"
+ )
+ return
+
+ # Aesthetic harmonization with plot_nuclear_texture_density: same
+ # `palette="husl"` + `fill=True` + `alpha=0.3` so cluster colours match
+ # across the two density figures (same Cluster_kmeans10 hue → same colour
+ # assignment by seaborn) and both have semi-transparent fill.
+ fig = plt.figure(figsize=(12, 6))
+ if GROUP_BY_COLUMN in new_df.columns:
+ df_kde = new_df[[GROUP_BY_COLUMN, "blur_prob_gmm_2d_roi"]].dropna()
+ if len(df_kde) > 0:
+ ax = sns.kdeplot(
+ data=df_kde,
+ x="blur_prob_gmm_2d_roi",
+ hue=GROUP_BY_COLUMN,
+ palette="husl",
+ common_norm=False,
+ fill=True,
+ alpha=0.3,
+ )
+ plt.axvline(
+ x=0.5,
+ color="black",
+ linestyle="--",
+ alpha=0.5,
+ label="Blur threshold (0.5)",
+ )
+ # Reformat legend labels to "Cluster N" prefix and include the
+ # threshold line — matches plot_nuclear_texture_density legend.
+ legend = ax.get_legend()
+ if legend is not None:
+ handles = legend.legend_handles
+ labels = [f"Cluster {label.get_text()}" for label in legend.get_texts()]
+ threshold_line = plt.Line2D(
+ [0], [0], color="black", linestyle="--", alpha=0.5
+ )
+ handles = [threshold_line] + handles
+ labels = ["Blur threshold (0.5)"] + labels
+ plt.legend(
+ handles,
+ labels,
+ title=GROUP_BY_COLUMN,
+ bbox_to_anchor=(1.05, 1),
+ loc="upper left",
+ )
+ plt.title(
+ f"Distribution of GMM blur probability by {GROUP_BY_COLUMN}",
+ fontsize=14,
+ pad=20,
+ )
+ else:
+ ax = plt.gca()
+ new_df["blur_prob_gmm_2d_roi"].dropna().plot.kde(ax=ax, color="steelblue")
+ plt.axvline(x=0.5, color="black", linestyle="--", alpha=0.5)
+ plt.title(
+ "GMM blur probability density (threshold at 0.5)", fontsize=14, pad=20
+ )
+ plt.xlabel("P(blur component)", fontsize=12)
+ plt.ylabel("Density", fontsize=12)
+ plt.grid(True, linestyle="--", alpha=0.7)
+ plt.tight_layout()
+ plt.savefig(
+ figures_dir / "blur_prob_density_by_cluster.png",
+ dpi=300,
+ bbox_inches="tight",
+ )
+ plt.close(fig)
+ src_cols = ["blur_prob_gmm_2d_roi"]
+ if GROUP_BY_COLUMN in new_df.columns:
+ src_cols.append(GROUP_BY_COLUMN)
+ src = new_df[[c for c in src_cols if c in new_df.columns]].dropna(
+ subset=["blur_prob_gmm_2d_roi"]
+ )
+ if figures_source_dir is not None:
+ src.to_csv(figures_source_dir / "blur_prob_density_by_cluster.csv", index=False)
+
+
+def plot_intensity_transcript_correlation(new_df, figures_dir, figures_source_dir):
+ """Spearman correlation heatmap between per-cell quality metrics and
+ transcript counts.
+
+ Includes six per-cell variables: nuclear texture (CCFS_DAPI), focus
+ score (tile-level, propagated to cell), three channel intensities
+ (DAPI / Boundary / IntRNA), and transcript counts.
+
+ Diagnostic for "which quality axis is critical for transcript yield
+ in THIS sample's tissue type?". Per-channel intensity_critical cutoffs
+ (DAPI=500, Boundary=100, IntRNA=300) are calibrated on lung/liver and
+ may not be appropriate for tissues like brain (where DAPI is
+ unreliable due to large neurons with naturally lower nuclear
+ contrast). The heatmap shows which axis actually correlates with
+ transcript yield, helping the analyst weight per-channel verdicts in
+ §4.1 appropriately.
+
+ Saves `figures/intensity_transcript_correlation.png` + source CSV.
+ Skips silently if fewer than 2 of the input columns are available.
+ """
+ # All columns are read from `new_df`, the post-merge cell-aligned
+ # frame produced at line ~5502 by `pd.merge(df_spatial, myData, on="CellID")`.
+ # `mean_intensity` (DAPI) comes in via the merge from `myData`. Reading
+ # all columns from a single index-aligned DataFrame guarantees that
+ # row N is the SAME cell across all series — critical for the
+ # correlation to be scientifically meaningful (reading half from `myData`
+ # and half from `new_df` would pair values across DIFFERENT cells).
+ cols = {}
+ # Image-quality metrics first (nuclear texture + focus) — these are the
+ # axes a reader is checking "is X the operative quality signal for my
+ # tissue?". Stain intensities follow; transcript counts last.
+ if "CCFS_DAPI" in new_df.columns:
+ cols["Nuclear texture"] = new_df["CCFS_DAPI"]
+ if "DAPI_RFSnorm_roi" in new_df.columns:
+ cols["Focus score"] = new_df["DAPI_RFSnorm_roi"]
+ if "mean_intensity" in new_df.columns:
+ cols["DAPI"] = new_df["mean_intensity"]
+ if "mean_intensity_Boundary" in new_df.columns:
+ cols["Boundary"] = new_df["mean_intensity_Boundary"]
+ if "mean_intensity_IntRNA" in new_df.columns:
+ cols["IntRNA"] = new_df["mean_intensity_IntRNA"]
+ if "transcript_counts" in new_df.columns:
+ cols["transcripts"] = new_df["transcript_counts"]
+
+ if len(cols) < 2:
+ logging.warning(
+ "Fewer than 2 quality / transcript columns available — skipping "
+ "per-cell correlation heatmap."
+ )
+ return
+
+ df_corr_input = pd.DataFrame(cols).dropna()
+ if len(df_corr_input) < 30:
+ logging.warning(
+ "Fewer than 30 cells with all quality / transcript columns — "
+ "skipping per-cell correlation heatmap (rho unstable)."
+ )
+ return
+
+ # Spearman (rank-based) — robust to non-normal distributions and outliers
+ # which intensity / transcript count data often exhibits.
+ corr_matrix = df_corr_input.corr(method="spearman")
+
+ fig, ax = plt.subplots(1, 1, figsize=(8, 7))
+ sns.heatmap(
+ corr_matrix,
+ annot=True,
+ fmt=".2f",
+ cmap="RdBu_r",
+ vmin=-1.0,
+ vmax=1.0,
+ center=0.0,
+ square=True,
+ cbar_kws={"label": "Spearman ρ", "shrink": 0.8},
+ ax=ax,
+ linewidths=0.5,
+ linecolor="white",
+ )
+ ax.set_title(
+ f"Per-cell quality metrics vs transcript count correlation (Spearman ρ; n={len(df_corr_input):,} cells)",
+ fontsize=12,
+ )
+ plt.tight_layout()
+ plt.savefig(
+ figures_dir / "intensity_transcript_correlation.png",
+ dpi=300,
+ bbox_inches="tight",
+ )
+ plt.close(fig)
+ # Save the input data (long form: cell + 4 columns) for reproducibility.
+ if figures_source_dir is not None:
+ df_corr_input.to_csv(
+ figures_source_dir / "intensity_transcript_correlation.csv", index=False
+ )
+
+
+def plot_nuclear_texture_density(
+ new_df,
+ figures_dir,
+ figures_source_dir,
+ GROUP_BY_COLUMN="Cluster_kmeans10",
+ ccfs_low_texture_threshold=DEFAULT_CCFS_LOW_TEXTURE_THRESHOLD,
+):
+ """
+ Create a density plot of CCFS_DAPI values grouped by Cluster_kmeans10.
+
+ Parameters:
+ -----------
+ new_df : pandas DataFrame
+ DataFrame containing CCFS_DAPI and Cluster_kmeans10 columns
+ """
+ # Set up the plot
+ fig = plt.figure(figsize=(12, 6))
+
+ # Create the density plot
+ ax = sns.kdeplot(
+ data=new_df,
+ x="CCFS_DAPI",
+ hue=GROUP_BY_COLUMN,
+ palette="husl",
+ common_norm=False, # Normalize each cluster separately
+ fill=True,
+ alpha=0.3,
+ ) # Make the fill semi-transparent
+
+ # Add a vertical line for the blur threshold
+ plt.axvline(
+ x=ccfs_low_texture_threshold,
+ color="black",
+ linestyle="--",
+ alpha=0.5,
+ label=f"Low Texture Threshold ({ccfs_low_texture_threshold})",
+ )
+
+ # Customize the plot
+ plt.title(
+ f"Distribution of Nuclear Texture Scores (CCFS_DAPI) by {GROUP_BY_COLUMN}",
+ fontsize=14,
+ pad=20,
+ )
+ plt.xlabel("CCFS_DAPI (Nuclear Texture Score)", fontsize=12)
+ plt.ylabel("Density", fontsize=12)
+
+ # Add a grid for better readability
+ plt.grid(True, linestyle="--", alpha=0.7)
+
+ # Get the current legend
+ legend = ax.get_legend()
+
+ # Get the handles and labels
+ handles = legend.legend_handles
+ labels = [f"Cluster {label.get_text()}" for label in legend.get_texts()]
+
+ # Add the threshold line to the legend
+ threshold_line = plt.Line2D([0], [0], color="black", linestyle="--", alpha=0.5)
+ handles = [threshold_line] + handles
+ labels = ["Default Threshold"] + labels
+
+ # Create new legend
+ plt.legend(
+ handles,
+ labels,
+ title=GROUP_BY_COLUMN,
+ bbox_to_anchor=(1.05, 1),
+ loc="upper left",
+ )
+
+ # Adjust layout to prevent label cutoff
+ plt.tight_layout()
+
+ # Save figure instead of showing
+ plt.savefig(
+ figures_dir / "nuclear_texture_density.png", dpi=300, bbox_inches="tight"
+ )
+ plt.close(fig)
+
+ # Save data as CSV
+ if figures_source_dir is not None:
+ df_density_data = new_df[["CCFS_DAPI", GROUP_BY_COLUMN]].copy()
+ df_density_data.to_csv(
+ figures_source_dir / "nuclear_texture_density.csv", index=False
+ )
+
+
+def plot_nuclear_texture_vs_transcripts(
+ new_df,
+ figures_dir,
+ figures_source_dir,
+ log_scale=False,
+ ccfs_low_texture_threshold=DEFAULT_CCFS_LOW_TEXTURE_THRESHOLD,
+):
+ """
+ Create a scatter plot of nuclear texture scores (CCFS_DAPI) vs transcript counts.
+
+ Parameters:
+ -----------
+ new_df : pandas DataFrame
+ DataFrame containing CCFS_DAPI and transcript_counts columns
+ log_scale : bool, optional
+ Whether to use log scale for x-axis (default: False)
+ """
+ # Set up the plot
+ fig = plt.figure(figsize=(12, 6))
+ ax = plt.gca()
+
+ _valid = new_df[["transcript_counts", "CCFS_DAPI"]].dropna()
+ x_vals = _valid["transcript_counts"].values.astype(np.float64)
+ y_vals = _valid["CCFS_DAPI"].values.astype(np.float64)
+
+ # Density-coloured scatter (same idiom as §4.4.b and the CCFS rank
+ # comparison plot at line ~6692). High-overlap regions paint at the
+ # high-density end of viridis, revealing trend / correlation that
+ # pure alpha-blending obscures at large N.
+ try:
+ from scipy.stats import gaussian_kde
+
+ # KDE in display coordinates: log space when axes are log-scaled
+ # so density reflects what the reader sees.
+ if log_scale:
+ x_kde = np.log10(np.clip(x_vals, 1e-9, None))
+ y_kde = np.log10(np.clip(y_vals, 1e-9, None))
+ else:
+ x_kde, y_kde = x_vals, y_vals
+
+ max_kde_pts = 8000
+ n_pts = len(x_kde)
+ if n_pts > max_kde_pts:
+ rng = np.random.default_rng(42)
+ sidx = rng.choice(n_pts, max_kde_pts, replace=False)
+ kde = gaussian_kde(np.vstack([x_kde[sidx], y_kde[sidx]]))
+ else:
+ kde = gaussian_kde(np.vstack([x_kde, y_kde]))
+ # Subsample the PLOTTED points to the standard cap (see
+ # plot_per_cell_intensity_vs_transcripts): ~15M cells overplot into a solid
+ # cloud, so ~10k render identically in ~1s vs ~35s. Density colour is still
+ # from the full-data KDE fit.
+ pidx = _subsample_idx(np.ones(n_pts, dtype=bool))
+ zc = _kde_density_grid(kde, x_kde[pidx], y_kde[pidx])
+ order = zc.argsort() # draw high-density points last (on top)
+ scatter = ax.scatter(
+ x_vals[pidx][order],
+ y_vals[pidx][order],
+ c=zc[order],
+ s=8,
+ cmap="viridis",
+ alpha=0.6,
+ rasterized=True,
+ )
+ plt.colorbar(scatter, ax=ax, label="Cell density (relative)")
+ except Exception as _kde_err:
+ logging.warning(
+ f"Density coloring failed for nuclear texture scatter ({_kde_err}); "
+ "falling back to alpha-blended scatter."
+ )
+ fidx = _subsample_idx(np.ones(len(x_vals), dtype=bool))
+ ax.scatter(
+ x_vals[fidx],
+ y_vals[fidx],
+ alpha=0.3,
+ s=8,
+ color="C0",
+ rasterized=True,
+ )
+
+ # Add a horizontal line for the blur threshold
+ plt.axhline(
+ y=ccfs_low_texture_threshold,
+ color="red",
+ linestyle="--",
+ alpha=0.5,
+ label=f"Low Texture Threshold ({ccfs_low_texture_threshold})",
+ )
+
+ # Customize the plot
+ title = "Per-cell nuclear texture score vs transcript count"
+ if log_scale:
+ title += " (log scale)"
+ plt.title(title, fontsize=14, pad=20)
+ plt.ylabel("CCFS_DAPI (Nuclear Texture Score)", fontsize=12)
+ plt.xlabel("Transcript Counts", fontsize=12)
+
+ # Set log scale for both axes if requested
+ if log_scale:
+ plt.xscale("log")
+ plt.xlabel("Transcript Counts (log scale)", fontsize=12)
+ plt.yscale("log")
+ plt.ylabel("CCFS_DAPI (Nuclear Texture Score) (log scale)", fontsize=12)
+
+ # Add a grid for better readability
+ plt.grid(True, linestyle="--", alpha=0.7)
+
+ # Add legend
+ plt.legend()
+
+ # Adjust layout to prevent label cutoff
+ plt.tight_layout()
+
+ # Save figure instead of showing
+ suffix = "_log" if log_scale else ""
+ plt.savefig(
+ figures_dir / f"nuclear_texture_vs_transcripts{suffix}.png",
+ dpi=300,
+ bbox_inches="tight",
+ )
+ plt.close(fig)
+
+ # Save data as CSV
+ if figures_source_dir is not None:
+ df_scatter_data = new_df[["transcript_counts", "CCFS_DAPI"]].copy()
+ df_scatter_data.to_csv(
+ figures_source_dir / f"nuclear_texture_vs_transcripts{suffix}.csv",
+ index=False,
+ )
+
+ # Print summary statistics
+ logging.info("Summary statistics:")
+ logging.info("Nuclear Texture Score (CCFS_DAPI):")
+ logging.info(new_df["CCFS_DAPI"].describe().round(4))
+ logging.info("Transcript Counts:")
+ logging.info(new_df["transcript_counts"].describe())
+
+
+def plot_per_cell_intensity_vs_transcripts(
+ new_df,
+ figures_dir,
+ figures_source_dir,
+ log_scale=False,
+):
+ """Per-cell mean intensity (DAPI / Boundary / IntRNA) vs transcript counts.
+
+ Three separate scatter plots, same format as
+ `plot_nuclear_texture_vs_transcripts` so readers learn the axes once.
+ Used in §4.4.b "Stain intensity vs transcript count" alongside the
+ higher-level `intensity_transcript_correlation.png` heatmap.
+
+ Skips silently when a channel intensity column is missing (e.g. DAPI-only
+ bundles produce no Boundary / IntRNA columns).
+
+ Parameters
+ ----------
+ new_df : pandas.DataFrame
+ Per-cell DataFrame; expected columns: `transcript_counts`,
+ `mean_intensity`, `mean_intensity_Boundary`, `mean_intensity_IntRNA`.
+ log_scale : bool, optional
+ If True, render both axes on a log scale (default False).
+ """
+ channels = [
+ ("DAPI", "mean_intensity"),
+ ("Boundary", "mean_intensity_Boundary"),
+ ("IntRNA", "mean_intensity_IntRNA"),
+ ]
+
+ for ch_label, col in channels:
+ if col not in new_df.columns:
+ logging.warning(
+ f"Column {col!r} not in per-cell DataFrame — "
+ f"skipping {ch_label} intensity vs transcripts scatter."
+ )
+ continue
+
+ _valid = new_df[[col, "transcript_counts"]].dropna()
+ if len(_valid) == 0:
+ logging.warning(
+ f"No valid (non-NaN) cells for {ch_label} intensity vs transcripts — skipping."
+ )
+ continue
+
+ fig = plt.figure(figsize=(12, 6))
+ ax = plt.gca()
+ x_vals = _valid["transcript_counts"].values.astype(np.float64)
+ y_vals = _valid[col].values.astype(np.float64)
+
+ # Density-colored scatter — high-overlap regions paint in the
+ # high-density end of the colormap, making trend / correlation
+ # readable even at large N where simple alpha blending saturates.
+ # Pattern matches plot_ccfs_vs_roi_comparison (line ~6692).
+ try:
+ from scipy.stats import gaussian_kde
+
+ # Compute KDE in display coordinates: log space when the axes
+ # are log-scaled, so the density gradient reflects what the
+ # reader sees rather than the raw coordinate distance.
+ if log_scale:
+ x_kde = np.log10(np.clip(x_vals, 1e-9, None))
+ y_kde = np.log10(np.clip(y_vals, 1e-9, None))
+ else:
+ x_kde, y_kde = x_vals, y_vals
+
+ max_kde_pts = 8000
+ n_pts = len(x_kde)
+ if n_pts > max_kde_pts:
+ rng = np.random.default_rng(42)
+ sidx = rng.choice(n_pts, max_kde_pts, replace=False)
+ kde = gaussian_kde(np.vstack([x_kde[sidx], y_kde[sidx]]))
+ else:
+ kde = gaussian_kde(np.vstack([x_kde, y_kde]))
+ # Subsample the PLOTTED points to the standard scatter cap. A
+ # density-colored scatter of ~15M cells overplots into a solid cloud at
+ # display resolution, so ~10k points render the identical cloud in ~1s
+ # vs ~32s/channel (measured 96s total). Density colour is still computed
+ # per plotted point from the full-data KDE fit; matches _subsample_idx as
+ # used by roi_focus_vs_intensity and the other density scatters.
+ pidx = _subsample_idx(np.ones(n_pts, dtype=bool))
+ zc = _kde_density_grid(kde, x_kde[pidx], y_kde[pidx])
+ order = zc.argsort() # draw high-density points last (on top)
+ scatter = ax.scatter(
+ x_vals[pidx][order],
+ y_vals[pidx][order],
+ c=zc[order],
+ s=8,
+ cmap="viridis",
+ alpha=0.6,
+ rasterized=True,
+ )
+ plt.colorbar(scatter, ax=ax, label="Cell density (relative)")
+ except Exception as _kde_err:
+ logging.warning(
+ f"Density coloring failed for {ch_label} ({_kde_err}); "
+ "falling back to alpha-blended scatter."
+ )
+ fidx = _subsample_idx(np.ones(len(x_vals), dtype=bool))
+ ax.scatter(
+ x_vals[fidx],
+ y_vals[fidx],
+ alpha=0.3,
+ s=8,
+ color="C0",
+ rasterized=True,
+ )
+
+ title = f"Per-cell {ch_label} mean intensity vs transcript count"
+ if log_scale:
+ title += " (log scale)"
+ plt.title(title, fontsize=14, pad=20)
+ plt.ylabel(f"{ch_label} mean intensity (16-bit counts)", fontsize=12)
+ plt.xlabel("Transcript counts", fontsize=12)
+
+ if log_scale:
+ plt.xscale("log")
+ plt.yscale("log")
+ plt.xlabel("Transcript counts (log scale)", fontsize=12)
+ plt.ylabel(
+ f"{ch_label} mean intensity (16-bit counts, log scale)", fontsize=12
+ )
+
+ plt.grid(True, linestyle="--", alpha=0.7)
+ plt.tight_layout()
+
+ suffix = "_log" if log_scale else ""
+ fname = f"{ch_label.lower()}_intensity_vs_transcripts{suffix}"
+ plt.savefig(figures_dir / f"{fname}.png", dpi=300, bbox_inches="tight")
+ plt.close(fig)
+
+ if figures_source_dir is not None:
+ _valid.to_csv(figures_source_dir / f"{fname}.csv", index=False)
+
+ logging.info(
+ f"Saved {fname}.png (n={len(_valid):,}, "
+ f"median intensity={_valid[col].median():.0f})"
+ )
+
+
+def plot_gmm_focus_vs_transcripts(
+ new_df,
+ figures_dir,
+ figures_source_dir,
+):
+ """Scatter plot of tile-level GMM focus score vs transcript counts, coloured by blur class.
+
+ Mirrors plot_nuclear_texture_vs_transcripts but uses the tile-level Laplacian
+ GMM focus score (DAPI_RFSnorm_roi) propagated to cells, with points coloured
+ by GMM blur classification.
+ """
+ focus_col = "DAPI_RFSnorm_roi"
+ tx_col = "transcript_counts"
+ gmm_col = "is_blurred_gmm_2d_roi"
+
+ if focus_col not in new_df.columns or tx_col not in new_df.columns:
+ logging.warning(
+ "Missing %s or %s — skipping GMM focus vs transcripts plot.",
+ focus_col,
+ tx_col,
+ )
+ return
+
+ df = new_df[
+ [tx_col, focus_col] + ([gmm_col] if gmm_col in new_df.columns else [])
+ ].dropna()
+ if df.empty:
+ return
+
+ has_gmm = gmm_col in df.columns
+ fig, ax = plt.subplots(figsize=(12, 6))
+
+ if has_gmm:
+ sharp = df[~df[gmm_col].astype(bool)]
+ blurred = df[df[gmm_col].astype(bool)]
+ ax.scatter(
+ sharp[tx_col],
+ sharp[focus_col],
+ alpha=0.3,
+ s=8,
+ color="#1f77b4",
+ label=f"In Focus ({len(sharp):,})",
+ rasterized=True,
+ )
+ ax.scatter(
+ blurred[tx_col],
+ blurred[focus_col],
+ alpha=0.3,
+ s=8,
+ color="#d62728",
+ label=f"Blurred ({len(blurred):,})",
+ rasterized=True,
+ )
+ ax.legend(fontsize=10)
+ else:
+ ax.scatter(df[tx_col], df[focus_col], alpha=0.3, s=8, rasterized=True)
+
+ ax.set_xscale("log")
+ ax.set_yscale("log")
+ ax.set_xlabel("Transcript Counts (log scale)", fontsize=12)
+ ax.set_ylabel("Tile Focus Score (DAPI_RFSnorm_roi, log scale)", fontsize=12)
+ ax.set_title("GMM Tile Focus Score vs Transcript Counts", fontsize=14, pad=20)
+ ax.grid(True, linestyle="--", alpha=0.7)
+
+ plt.tight_layout()
+ plt.savefig(
+ figures_dir / "gmm_focus_vs_transcripts.png", dpi=300, bbox_inches="tight"
+ )
+ plt.close(fig)
+
+ # Save source data
+ if figures_source_dir is not None:
+ src_cols = [tx_col, focus_col]
+ if has_gmm:
+ src_cols.append(gmm_col)
+ df[src_cols].to_csv(
+ figures_source_dir / "gmm_focus_vs_transcripts.csv", index=False
+ )
+
+ logging.info("GMM focus vs transcripts: %d cells plotted.", len(df))
+
+
+def plot_ccfs_vs_roi_comparison(new_df, figures_dir, figures_source_dir):
+ """
+ Create comparison plots between CCFS (nuclei-based) and cell-independent tile-based focus scores.
+
+ Parameters:
+ -----------
+ new_df : pandas DataFrame
+ DataFrame containing both CCFS_DAPI and DAPI_RFSnorm_roi columns
+ (DAPI_RFSnorm_roi contains cell-independent grid ROI focus scores transferred to cells)
+ figures_dir : Path
+ Directory to save figures
+ figures_source_dir : Path
+ Directory to save source data
+ """
+ # Create figure with subplots
+ fig, axes = plt.subplots(2, 2, figsize=(15, 15))
+
+ # Plot 1: Ranked scatter plot with density coloring
+ ax = axes[0, 0]
+ # Calculate ranks (higher rank = higher focus score)
+ ccfs_ranks = new_df["CCFS_DAPI"].rank(method="average")
+ roi_ranks = new_df["DAPI_RFSnorm_roi"].rank(method="average")
+
+ # Create scatter plot with density coloring (subsample for KDE performance)
+ try:
+ from scipy.stats import gaussian_kde
+
+ valid = ccfs_ranks.dropna().index.intersection(roi_ranks.dropna().index)
+ x_vals, y_vals = ccfs_ranks[valid].values, roi_ranks[valid].values
+ max_kde_pts = 8000
+ if len(x_vals) > max_kde_pts:
+ rng = np.random.default_rng(42)
+ sidx = rng.choice(len(x_vals), max_kde_pts, replace=False)
+ xy_sub = np.vstack([x_vals[sidx], y_vals[sidx]])
+ kde = gaussian_kde(xy_sub)
+ z = _kde_density_grid(kde, x_vals, y_vals)
+ else:
+ z = gaussian_kde(np.vstack([x_vals, y_vals]))(np.vstack([x_vals, y_vals]))
+ idx = z.argsort()
+ scatter = ax.scatter(
+ x_vals[idx],
+ y_vals[idx],
+ c=z[idx],
+ s=1,
+ cmap="viridis",
+ alpha=0.6,
+ rasterized=True,
+ )
+ plt.colorbar(scatter, ax=ax, label="Density")
+ except (ImportError, Exception):
+ # Fallback if scipy not available or density calculation fails
+ ax.scatter(ccfs_ranks, roi_ranks, alpha=0.3, s=1, marker="o", rasterized=True)
+
+ ax.set_xlabel("CCFS_DAPI Rank (Nuclei-based)", fontsize=12)
+ ax.set_ylabel("DAPI_RFSnorm_roi Rank (Cell-Independent Tile-based)", fontsize=12)
+ ax.set_title("CCFS vs Cell-Independent Tile Focus Score (Ranked)", fontsize=14)
+ ax.grid(True, linestyle="--", alpha=0.7)
+
+ # Add correlation coefficient (Spearman rank-based correlation)
+ correlation = new_df["CCFS_DAPI"].corr(
+ new_df["DAPI_RFSnorm_roi"], method="spearman"
+ )
+ ax.text(
+ 0.05,
+ 0.95,
+ f"Spearman Correlation: {correlation:.4f}",
+ transform=ax.transAxes,
+ fontsize=12,
+ verticalalignment="top",
+ bbox=dict(boxstyle="round", facecolor="wheat", alpha=0.5),
+ )
+
+ # Plot 2: Density plot overlay
+ ax = axes[0, 1]
+ ax.scatter(
+ new_df["CCFS_DAPI"],
+ new_df["DAPI_RFSnorm_roi"],
+ alpha=0.1,
+ s=0.5,
+ marker="o",
+ c="blue",
+ label="Data points",
+ rasterized=True,
+ )
+ # Add 2D density contour (subsample for KDE performance)
+ try:
+ from scipy.stats import gaussian_kde
+
+ valid_mask = new_df["CCFS_DAPI"].notna() & new_df["DAPI_RFSnorm_roi"].notna()
+ x_vals = new_df.loc[valid_mask, "CCFS_DAPI"].values
+ y_vals = new_df.loc[valid_mask, "DAPI_RFSnorm_roi"].values
+ max_kde_pts = 8000
+ if len(x_vals) > max_kde_pts:
+ rng = np.random.default_rng(42)
+ sidx = rng.choice(len(x_vals), max_kde_pts, replace=False)
+ kde = gaussian_kde(np.vstack([x_vals[sidx], y_vals[sidx]]))
+ z = _kde_density_grid(kde, x_vals, y_vals)
+ else:
+ z = gaussian_kde(np.vstack([x_vals, y_vals]))(np.vstack([x_vals, y_vals]))
+ idx = z.argsort()
+ ax.scatter(
+ x_vals[idx],
+ y_vals[idx],
+ c=z[idx],
+ s=1,
+ cmap="viridis",
+ alpha=0.6,
+ rasterized=True,
+ )
+ except (ImportError, Exception):
+ pass # Skip density overlay if unavailable
+ ax.set_xlabel("CCFS_DAPI (Nuclei-based)", fontsize=12)
+ ax.set_ylabel("DAPI_RFSnorm_roi (Cell-Independent Tile-based)", fontsize=12)
+ ax.set_title("CCFS vs Cell-Independent Tile Focus Score (Density)", fontsize=14)
+ ax.grid(True, linestyle="--", alpha=0.7)
+
+ # Plot 3: Agreement/disagreement classification
+ ax = axes[1, 0]
+ # Prefer GMM 2D if available, otherwise use threshold-based
+ has_gmm_2d = "is_blurred_gmm_2d_roi" in new_df.columns
+ has_threshold = "is_blurred_roi" in new_df.columns
+
+ if "is_low_nuclear_texture" in new_df.columns and (has_gmm_2d or has_threshold):
+ # Use GMM 2D if available, otherwise threshold-based
+ if has_gmm_2d:
+ roi_blurred_col = "is_blurred_gmm_2d_roi"
+ roi_high_focus = ~new_df["is_blurred_gmm_2d_roi"]
+ method_label = "2D GMM"
+ else:
+ roi_blurred_col = "is_blurred_roi"
+ roi_high_focus = new_df["is_high_focus_roi"]
+ method_label = "Threshold"
+
+ # Create agreement categories
+ agreement = new_df["is_low_nuclear_texture"] == new_df[roi_blurred_col]
+ agree_high = new_df["is_high_nuclear_texture"] & roi_high_focus
+ agree_low = new_df["is_low_nuclear_texture"] & new_df[roi_blurred_col]
+ disagree = ~agreement
+
+ # Plot
+ ax.scatter(
+ new_df.loc[agree_high, "CCFS_DAPI"],
+ new_df.loc[agree_high, "DAPI_RFSnorm_roi"],
+ alpha=0.3,
+ s=1,
+ c="green",
+ label="Both high focus",
+ marker="o",
+ rasterized=True,
+ )
+ ax.scatter(
+ new_df.loc[agree_low, "CCFS_DAPI"],
+ new_df.loc[agree_low, "DAPI_RFSnorm_roi"],
+ alpha=0.3,
+ s=1,
+ c="red",
+ label="Both blurred",
+ marker="o",
+ rasterized=True,
+ )
+ ax.scatter(
+ new_df.loc[disagree, "CCFS_DAPI"],
+ new_df.loc[disagree, "DAPI_RFSnorm_roi"],
+ alpha=0.5,
+ s=2,
+ c="orange",
+ label="Disagree",
+ marker="x",
+ rasterized=True,
+ )
+ ax.set_xlabel("CCFS_DAPI (Nuclei-based)", fontsize=12)
+ ax.set_ylabel("DAPI_RFSnorm_roi (Cell-Independent Tile-based)", fontsize=12)
+ ax.set_title(
+ f"Classification Agreement (CCFS vs Tile-based {method_label})", fontsize=14
+ )
+ ax.legend(fontsize=10)
+ ax.grid(True, linestyle="--", alpha=0.7)
+
+ # Add agreement percentage
+ agreement_rate = agreement.sum() / len(new_df) * 100
+ ax.text(
+ 0.05,
+ 0.95,
+ f"Agreement: {agreement_rate:.2f}%",
+ transform=ax.transAxes,
+ fontsize=12,
+ verticalalignment="top",
+ bbox=dict(boxstyle="round", facecolor="wheat", alpha=0.5),
+ )
+
+ # Plot 4: Distribution comparison
+ ax = axes[1, 1]
+ ax.hist(
+ new_df["CCFS_DAPI"].dropna(),
+ bins=50,
+ alpha=0.5,
+ label="CCFS_DAPI (Nuclei-based)",
+ color="blue",
+ density=True,
+ )
+ ax.hist(
+ new_df["DAPI_RFSnorm_roi"].dropna(),
+ bins=50,
+ alpha=0.5,
+ label="DAPI_RFSnorm_roi (Cell-Independent Tile)",
+ color="red",
+ density=True,
+ )
+ ax.set_xlabel("Focus Score", fontsize=12)
+ ax.set_ylabel("Density", fontsize=12)
+ ax.set_title("Distribution Comparison (CCFS vs Cell-Independent Tile)", fontsize=14)
+ ax.legend(fontsize=10)
+ ax.grid(True, linestyle="--", alpha=0.7)
+
+ plt.tight_layout()
+ plt.savefig(
+ figures_dir / "ccfs_vs_roi_comparison.pdf", dpi=300, bbox_inches="tight"
+ )
+ plt.savefig(
+ figures_dir / "ccfs_vs_roi_comparison.png", dpi=300, bbox_inches="tight"
+ )
+ plt.close(fig)
+
+ # Save data as CSV (including ranks)
+ df_comparison = new_df[["CCFS_DAPI", "DAPI_RFS_roi", "DAPI_RFSnorm_roi"]].copy()
+ # Add ranks for analysis
+ df_comparison["CCFS_DAPI_rank"] = new_df["CCFS_DAPI"].rank(method="average")
+ df_comparison["DAPI_RFSnorm_roi_rank"] = new_df["DAPI_RFSnorm_roi"].rank(
+ method="average"
+ )
+
+ # Add classification columns - prefer GMM 2D, include threshold-based if available
+ if "is_low_nuclear_texture" in new_df.columns:
+ df_comparison["is_low_nuclear_texture"] = new_df["is_low_nuclear_texture"]
+
+ # GMM 2D classification (preferred)
+ if "is_blurred_gmm_2d_roi" in new_df.columns:
+ df_comparison["is_blurred_gmm_2d_roi"] = new_df["is_blurred_gmm_2d_roi"]
+ df_comparison["classification_agreement_gmm_2d"] = (
+ new_df["is_low_nuclear_texture"] == new_df["is_blurred_gmm_2d_roi"]
+ )
+ if "blur_prob_gmm_2d_roi" in new_df.columns:
+ df_comparison["blur_prob_gmm_2d_roi"] = new_df["blur_prob_gmm_2d_roi"]
+
+ # Threshold-based classification (for comparison)
+ if "is_blurred_roi" in new_df.columns:
+ df_comparison["is_blurred_roi"] = new_df["is_blurred_roi"]
+ df_comparison["classification_agreement"] = (
+ new_df["is_low_nuclear_texture"] == new_df["is_blurred_roi"]
+ )
+
+ if figures_source_dir is not None:
+ df_comparison.to_csv(
+ figures_source_dir / "ccfs_vs_roi_comparison.csv", index=False
+ )
+
+
+def plot_spatial_comparison(new_df, myData, figures_dir, figures_source_dir):
+ """
+ Create spatial comparison plots showing both nuclei-based and tile-based focus scores.
+
+ Parameters:
+ -----------
+ new_df : pandas DataFrame
+ DataFrame containing spatial and focus score data
+ myData : pandas DataFrame
+ DataFrame with nuclei-based measurements (for spatial coordinates)
+ figures_dir : Path
+ Directory to save figures
+ figures_source_dir : Path
+ Directory to save source data
+ """
+ # Compute figure size from data extents to avoid empty white space
+ _x_range = (
+ myData["centroid-1"].max() - myData["centroid-1"].min()
+ if "centroid-1" in myData.columns
+ else 1
+ )
+ _y_range = (
+ myData["centroid-0"].max() - myData["centroid-0"].min()
+ if "centroid-0" in myData.columns
+ else 1
+ )
+ _data_aspect = _y_range / _x_range if _x_range > 0 else 1.0
+ _pw = 7
+ _ph = max(4, _pw * _data_aspect)
+ fig, axes = plt.subplots(1, 2, figsize=(2 * _pw + 2, _ph))
+
+ # Plot 1: Nuclei-based CCFS spatial
+ ax = axes[0]
+ if "centroid-1" in myData.columns and "centroid-0" in myData.columns:
+ # Filter out NaN values for plotting
+ valid_mask = myData["CCFS_DAPI"].notna()
+ if valid_mask.sum() > 0:
+ # Use CCFS_DAPI directly (not negated) with auto-scaling
+ ccfs_values = myData.loc[valid_mask, "CCFS_DAPI"]
+ scatter = ax.scatter(
+ myData.loc[valid_mask, "centroid-1"],
+ -myData.loc[valid_mask, "centroid-0"],
+ s=0.1,
+ c=ccfs_values,
+ cmap="viridis",
+ rasterized=True,
+ )
+ ax.set_title("Nuclei-based Focus Score (CCFS_DAPI)", fontsize=14)
+ ax.set_facecolor("black")
+ ax.set_aspect("equal")
+ plt.colorbar(scatter, ax=ax, label="CCFS_DAPI")
+ else:
+ ax.text(
+ 0.5,
+ 0.5,
+ "No valid CCFS_DAPI data",
+ transform=ax.transAxes,
+ ha="center",
+ va="center",
+ fontsize=14,
+ bbox=dict(boxstyle="round", facecolor="lightgray", alpha=0.5),
+ )
+ ax.set_facecolor("black")
+ ax.set_aspect("equal")
+
+ # Plot 2: Tile-based RFS spatial (colored by GMM 2D classification if available)
+ ax = axes[1]
+ if "DAPI_RFSnorm_roi" in new_df.columns:
+ # Prefer GMM 2D classification for coloring, otherwise use raw focus scores
+ has_gmm_2d = "is_blurred_gmm_2d_roi" in new_df.columns
+ valid_mask = new_df["DAPI_RFSnorm_roi"].notna()
+
+ if has_gmm_2d:
+ # Color by GMM 2D classification (blurred vs in-focus)
+ gmm_valid_mask = valid_mask & new_df["is_blurred_gmm_2d_roi"].notna()
+ if gmm_valid_mask.sum() > 0:
+ # Color: red for blurred, blue for in-focus
+ colors = [
+ "red" if blurred else "blue"
+ for blurred in new_df.loc[gmm_valid_mask, "is_blurred_gmm_2d_roi"]
+ ]
+ ax.scatter(
+ new_df.loc[gmm_valid_mask, "x"],
+ -new_df.loc[gmm_valid_mask, "y"],
+ s=0.1,
+ c=colors,
+ alpha=0.5,
+ rasterized=True,
+ )
+ ax.set_title(
+ "Tile-based Focus Score (GMM 2D Classification)", fontsize=14
+ )
+ # Add legend
+ from matplotlib.patches import Patch
+
+ legend_elements = [
+ Patch(facecolor="blue", label="In-Focus (2D GMM)"),
+ Patch(facecolor="red", label="Blurred (2D GMM)"),
+ ]
+ ax.legend(handles=legend_elements, fontsize=10, loc="upper right")
+ else:
+ ax.text(
+ 0.5,
+ 0.5,
+ "No valid GMM 2D classification data",
+ transform=ax.transAxes,
+ ha="center",
+ va="center",
+ fontsize=14,
+ bbox=dict(boxstyle="round", facecolor="lightgray", alpha=0.5),
+ )
+ else:
+ # Fallback: use raw focus scores with color scale
+ if valid_mask.sum() > 0:
+ roi_values = new_df.loc[valid_mask, "DAPI_RFSnorm_roi"]
+ scatter = ax.scatter(
+ new_df.loc[valid_mask, "x"],
+ -new_df.loc[valid_mask, "y"],
+ s=0.1,
+ c=roi_values,
+ cmap="viridis",
+ rasterized=True,
+ )
+ ax.set_title("Tile-based Focus Score (DAPI_RFSnorm_roi)", fontsize=14)
+ plt.colorbar(scatter, ax=ax, label="DAPI_RFSnorm_roi")
+ else:
+ ax.text(
+ 0.5,
+ 0.5,
+ "No valid DAPI_RFSnorm_roi data",
+ transform=ax.transAxes,
+ ha="center",
+ va="center",
+ fontsize=14,
+ bbox=dict(boxstyle="round", facecolor="lightgray", alpha=0.5),
+ )
+
+ ax.set_facecolor("black")
+ ax.set_aspect("equal")
+
+ plt.tight_layout()
+ plt.savefig(
+ figures_dir / "spatial_comparison_nuclei_vs_roi.pdf",
+ dpi=300,
+ bbox_inches="tight",
+ )
+ plt.savefig(
+ figures_dir / "spatial_comparison_nuclei_vs_roi.png",
+ dpi=300,
+ bbox_inches="tight",
+ )
+ plt.close(fig)
+
+ # Save data as CSV
+ if "centroid-1" in myData.columns:
+ df_spatial_comparison = pd.DataFrame(
+ {
+ "x_nuclei": myData["centroid-1"],
+ "y_nuclei": -myData["centroid-0"],
+ "CCFS_DAPI": myData["CCFS_DAPI"],
+ }
+ )
+ if "DAPI_RFSnorm_roi" in new_df.columns:
+ # Try to merge on coordinates
+ df_spatial_comparison = df_spatial_comparison.merge(
+ new_df[["x", "y", "DAPI_RFSnorm_roi"]],
+ left_on=["x_nuclei", "y_nuclei"],
+ right_on=["x", "y"],
+ how="left",
+ )
+ if figures_source_dir is not None:
+ df_spatial_comparison.to_csv(
+ figures_source_dir / "spatial_comparison_nuclei_vs_roi.csv",
+ index=False,
+ )
+
+
+def save_cell_qc_metrics(new_df, outdir, roi_size=None):
+ """Save QC metrics to CSV and JSON"""
+
+ # Move cell_id to first position
+ cell_id_col = new_df.pop("cell_id")
+ new_df.insert(0, "cell_id", cell_id_col)
+ # Save full dataset
+ new_df.to_csv(outdir / "image_qc_cell_metrics.csv", index=False)
+
+ # Generate QC summary statistics
+ qc_metrics = {
+ "total_cells": len(new_df),
+ "cells_with_transcripts": len(new_df[new_df["transcript_counts"] > 0]),
+ "mean_transcript_count": float(new_df["transcript_counts"].mean()),
+ "median_transcript_count": float(new_df["transcript_counts"].median()),
+ "std_transcript_count": float(new_df["transcript_counts"].std()),
+ "mean_ccfs_dapi": float(new_df["CCFS_DAPI"].mean()),
+ "cells_high_nuclear_texture": len(new_df[new_df["is_high_nuclear_texture"]]),
+ "cells_low_nuclear_texture": len(new_df[new_df["is_low_nuclear_texture"]]),
+ "cells_near_edge": len(new_df[new_df["is_near_edge"]]),
+ "cells_near_holes": len(new_df[new_df["is_near_hole"]]),
+ "cells_with_dense_intensity_regions": len(
+ new_df[new_df["has_dense_intensity_regions"]]
+ ),
+ "unique_dense_intensity_regions": sorted(
+ new_df["Dense-Intensity-Region-ID"].unique().tolist()
+ )
+ if "Dense-Intensity-Region-ID" in new_df.columns
+ else [],
+ "total_dense_intensity_regions": len(
+ new_df[new_df["Dense-Intensity-Region-ID"] > 0][
+ "Dense-Intensity-Region-ID"
+ ].unique()
+ )
+ if "Dense-Intensity-Region-ID" in new_df.columns
+ else 0,
+ "clusters_present": sorted(new_df["Cluster_kmeans10"].unique().tolist()),
+ "segmentation_methods": sorted(new_df["segmentation_method"].unique().tolist()),
+ }
+
+ # Add tile-based metrics if available
+ if "DAPI_RFS_roi" in new_df.columns:
+ # Save ROI size used for this analysis
+ if roi_size is not None:
+ qc_metrics["roi_size_used"] = int(roi_size)
+ qc_metrics["mean_dapi_rfs_roi"] = float(new_df["DAPI_RFS_roi"].mean())
+ qc_metrics["median_dapi_rfs_roi"] = float(new_df["DAPI_RFS_roi"].median())
+ qc_metrics["mean_dapi_rfsnorm_roi"] = float(new_df["DAPI_RFSnorm_roi"].mean())
+ qc_metrics["median_dapi_rfsnorm_roi"] = float(
+ new_df["DAPI_RFSnorm_roi"].median()
+ )
+
+ # GMM 2D metrics (preferred method)
+ if "is_blurred_gmm_2d_roi" in new_df.columns:
+ qc_metrics["cells_high_focus_gmm_2d_roi"] = len(
+ new_df[~new_df["is_blurred_gmm_2d_roi"]]
+ )
+ qc_metrics["cells_blurred_gmm_2d_roi"] = len(
+ new_df[new_df["is_blurred_gmm_2d_roi"]]
+ )
+ if "blur_prob_gmm_2d_roi" in new_df.columns:
+ valid_probs = new_df["blur_prob_gmm_2d_roi"].dropna()
+ if len(valid_probs) > 0:
+ qc_metrics["mean_blur_prob_gmm_2d_roi"] = float(valid_probs.mean())
+ qc_metrics["median_blur_prob_gmm_2d_roi"] = float(
+ valid_probs.median()
+ )
+
+ # Threshold-based metrics (for comparison/backward compatibility)
+ if "is_blurred_roi" in new_df.columns:
+ qc_metrics["cells_high_focus_roi"] = len(
+ new_df[new_df["is_high_focus_roi"]]
+ )
+ qc_metrics["cells_blurred_roi"] = len(new_df[new_df["is_blurred_roi"]])
+
+ # Comparison metrics between nuclei-based and tile-based methods
+ # Calculate Spearman rank-based correlation between CCFS_DAPI and DAPI_RFSnorm_roi
+ # Spearman is better suited for different scales and non-linear relationships
+ if "CCFS_DAPI" in new_df.columns and "DAPI_RFSnorm_roi" in new_df.columns:
+ correlation = new_df["CCFS_DAPI"].corr(
+ new_df["DAPI_RFSnorm_roi"], method="spearman"
+ )
+ qc_metrics["ccfs_vs_rfs_correlation"] = (
+ float(correlation) if not np.isnan(correlation) else None
+ )
+ qc_metrics["ccfs_vs_rfs_correlation_method"] = "spearman"
+
+ # Calculate agreement rate - prefer GMM 2D, include threshold-based for comparison
+ if "is_low_nuclear_texture" in new_df.columns:
+ # GMM 2D agreement (preferred)
+ if "is_blurred_gmm_2d_roi" in new_df.columns:
+ agreement_gmm_2d = (
+ new_df["is_low_nuclear_texture"]
+ == new_df["is_blurred_gmm_2d_roi"]
+ ).sum()
+ agreement_rate_gmm_2d = agreement_gmm_2d / len(new_df)
+ qc_metrics["classification_agreement_gmm_2d"] = float(
+ agreement_rate_gmm_2d
+ )
+ qc_metrics["classification_agreement_gmm_2d_count"] = int(
+ agreement_gmm_2d
+ )
+ qc_metrics["classification_disagreement_gmm_2d_count"] = int(
+ len(new_df) - agreement_gmm_2d
+ )
+
+ # Threshold-based agreement (for comparison)
+ if "is_blurred_roi" in new_df.columns:
+ agreement = (
+ new_df["is_low_nuclear_texture"] == new_df["is_blurred_roi"]
+ ).sum()
+ agreement_rate = agreement / len(new_df)
+ qc_metrics["classification_agreement"] = float(agreement_rate)
+ qc_metrics["classification_agreement_count"] = int(agreement)
+ qc_metrics["classification_disagreement_count"] = int(
+ len(new_df) - agreement
+ )
+
+ # Save metrics
+ with open(outdir / "image_qc_cell_metrics.json", "w") as f:
+ json.dump(qc_metrics, f, indent=2)
+
+ # Save ROI size to simple text file for easy retrieval
+ if roi_size is not None:
+ with open(outdir / "roi_size.txt", "w") as f:
+ f.write(f"{roi_size}\n")
+
+ logging.info("\n=== IMAGE QC SUMMARY ===")
+ logging.info(f"Total cells analyzed: {qc_metrics['total_cells']:,}")
+ logging.info(f"Cells with transcripts: {qc_metrics['cells_with_transcripts']:,}")
+ logging.info(f"Mean transcript count: {qc_metrics['mean_transcript_count']:.1f}")
+ logging.info(f"Mean CCFS DAPI: {qc_metrics['mean_ccfs_dapi']:.6f}")
+ logging.info(
+ f"Cells high nuclear texture: {qc_metrics['cells_high_nuclear_texture']:,}"
+ )
+ logging.info(
+ f"Cells low nuclear texture: {qc_metrics['cells_low_nuclear_texture']:,}"
+ )
+ logging.info(f"Cells near edge: {qc_metrics['cells_near_edge']:,}")
+ logging.info(f"Cells near holes: {qc_metrics['cells_near_holes']:,}")
+ logging.info(
+ f"Cells with dense intensity regions: {qc_metrics['cells_with_dense_intensity_regions']:,}"
+ )
+ if (
+ "total_dense_intensity_regions" in qc_metrics
+ and qc_metrics["total_dense_intensity_regions"] > 0
+ ):
+ logging.info(
+ f"Total unique dense intensity regions detected: {qc_metrics['total_dense_intensity_regions']}"
+ )
+ logging.info(
+ f" (Region IDs: {qc_metrics['unique_dense_intensity_regions'][:10]}{'...' if len(qc_metrics['unique_dense_intensity_regions']) > 10 else ''})"
+ )
+ logging.info(f"Clusters present: {qc_metrics['clusters_present']}")
+ logging.info(f"Segmentation methods: {qc_metrics['segmentation_methods']}")
+
+ # Print tile-based metrics if available
+ if "mean_dapi_rfs_roi" in qc_metrics:
+ logging.info("\n=== TILE-BASED FOCUS SCORE SUMMARY ===")
+ logging.info(f"Mean DAPI RFS (tile): {qc_metrics['mean_dapi_rfs_roi']:.6f}")
+ logging.info(
+ f"Mean DAPI RFS normalized (tile): {qc_metrics['mean_dapi_rfsnorm_roi']:.6f}"
+ )
+
+ # GMM 2D metrics (preferred)
+ if "cells_blurred_gmm_2d_roi" in qc_metrics:
+ logging.info("\n--- 2D GMM Classification (Preferred) ---")
+ logging.info(
+ f"Cells with high focus (2D GMM): {qc_metrics['cells_high_focus_gmm_2d_roi']:,}"
+ )
+ logging.info(
+ f"Cells blurred (2D GMM): {qc_metrics['cells_blurred_gmm_2d_roi']:,}"
+ )
+ if "mean_blur_prob_gmm_2d_roi" in qc_metrics:
+ logging.info(
+ f"Mean blur probability (2D GMM): {qc_metrics['mean_blur_prob_gmm_2d_roi']:.4f}"
+ )
+
+ # Threshold-based metrics (for comparison)
+ if "cells_blurred_roi" in qc_metrics:
+ logging.info("\n--- Threshold-based Classification (Comparison) ---")
+ logging.info(
+ f"Cells with high focus (Threshold): {qc_metrics['cells_high_focus_roi']:,}"
+ )
+ logging.info(
+ f"Cells blurred (Threshold): {qc_metrics['cells_blurred_roi']:,}"
+ )
+
+ if (
+ "ccfs_vs_rfs_correlation" in qc_metrics
+ and qc_metrics["ccfs_vs_rfs_correlation"] is not None
+ ):
+ logging.info("\n=== METHOD COMPARISON ===")
+ logging.info(
+ f"Spearman Correlation (CCFS vs RFS): {qc_metrics['ccfs_vs_rfs_correlation']:.4f}"
+ )
+
+ # GMM 2D agreement (preferred)
+ if "classification_agreement_gmm_2d" in qc_metrics:
+ logging.info("\n--- CCFS vs 2D GMM Agreement (Preferred) ---")
+ logging.info(
+ f"Classification agreement: {qc_metrics['classification_agreement_gmm_2d'] * 100:.2f}%"
+ )
+ logging.info(
+ f" - Agreeing cells: {qc_metrics['classification_agreement_gmm_2d_count']:,}"
+ )
+ logging.info(
+ f" - Disagreeing cells: {qc_metrics['classification_disagreement_gmm_2d_count']:,}"
+ )
+
+ # Threshold-based agreement (for comparison)
+ if "classification_agreement" in qc_metrics:
+ logging.info("\n--- CCFS vs Threshold Agreement (Comparison) ---")
+ logging.info(
+ f"Classification agreement: {qc_metrics['classification_agreement'] * 100:.2f}%"
+ )
+ logging.info(
+ f" - Agreeing cells: {qc_metrics['classification_agreement_count']:,}"
+ )
+ logging.info(
+ f" - Disagreeing cells: {qc_metrics['classification_disagreement_count']:,}"
+ )
+
+
+def generate_cell_figures(
+ data,
+ new_df,
+ myData,
+ figures_dir,
+ figures_source_dir,
+ roi_threshold=None,
+ roi_intensity_threshold=None,
+ ccfs_low_texture_threshold=DEFAULT_CCFS_LOW_TEXTURE_THRESHOLD,
+):
+ """
+ Generate cell-based figures from merged data using multithreading.
+
+ Parameters:
+ -----------
+ data : dict
+ Dictionary with paths and directories
+ new_df : pandas DataFrame
+ Merged DataFrame with cell data and ROI mappings
+ myData : pandas DataFrame
+ Nuclei-based measurements DataFrame
+ figures_dir : Path
+ Directory to save figures
+ figures_source_dir : Path
+ Directory to save source data
+ roi_threshold : float, optional
+ Tile-based blur threshold for display
+ roi_intensity_threshold : float, optional
+ Tile intensity threshold for display
+ """
+ if figures_source_dir is not None:
+ figures_source_dir.mkdir(parents=True, exist_ok=True)
+
+ logging.info("Generating cell-based figures...")
+
+ def _plot1_nuclear_texture_proportions():
+ logging.info(" - Nuclear texture proportions (CCFS)...")
+ plot_nuclear_texture_proportions(
+ new_df,
+ figures_dir,
+ figures_source_dir,
+ texture_threshold=ccfs_low_texture_threshold,
+ )
+
+ def _plot2_blur_proportions_roi():
+ if (
+ "is_blurred_gmm_2d_roi" in new_df.columns
+ or "is_blurred_roi" in new_df.columns
+ ):
+ logging.info(" - Blur score proportions (tile-based)...")
+ roi_threshold_display = roi_threshold if roi_threshold is not None else -1.0
+ plot_tile_blur_proportions_roi._intensity_threshold = (
+ roi_intensity_threshold if roi_intensity_threshold is not None else 20.0
+ )
+ plot_tile_blur_proportions_roi(
+ new_df,
+ figures_dir,
+ figures_source_dir,
+ roi_threshold=roi_threshold_display,
+ )
+
+ def _plot3_nuclear_texture_density():
+ logging.info(" - Nuclear texture density...")
+ plot_nuclear_texture_density(
+ new_df,
+ figures_dir,
+ figures_source_dir,
+ ccfs_low_texture_threshold=ccfs_low_texture_threshold,
+ )
+
+ def _plot5_ccfs_vs_roi():
+ if "DAPI_RFSnorm_roi" in new_df.columns:
+ logging.info(" - CCFS vs tile comparison...")
+ plot_ccfs_vs_roi_comparison(new_df, figures_dir, figures_source_dir)
+
+ def _plot6_spatial_comparison():
+ logging.info(" - Spatial comparison...")
+ plot_spatial_comparison(new_df, myData, figures_dir, figures_source_dir)
+
+ def _plot7_cell_focus_distribution():
+ logging.info(" - Cell-level focus score distribution...")
+ plot_cell_focus_distribution(new_df, figures_dir, figures_source_dir)
+
+ def _plot8_gmm_focus_vs_transcripts():
+ logging.info(" - GMM focus vs transcripts...")
+ plot_gmm_focus_vs_transcripts(new_df, figures_dir, figures_source_dir)
+
+ def _plot9_tile_focus_gmm_spatial():
+ logging.info(" - Tile-focus GMM spatial...")
+ plot_tile_focus_gmm_spatial(new_df, figures_dir, figures_source_dir)
+
+ def _plot9b_cell_flagged_maps_combined():
+ logging.info(" - Cell-flagged maps (combined)...")
+ plot_cell_flagged_maps_combined(new_df, figures_dir, figures_source_dir)
+
+ def _plot10_blur_prob_density_by_cluster():
+ logging.info(" - Blur probability density by cluster...")
+ plot_blur_prob_density_by_cluster(new_df, figures_dir, figures_source_dir)
+
+ def _plot11_intensity_transcript_correlation():
+ logging.info(" - Intensity-transcript correlation heatmap...")
+ plot_intensity_transcript_correlation(new_df, figures_dir, figures_source_dir)
+
+ def _plot12_per_cell_intensity_vs_transcripts():
+ logging.info(
+ " - Per-cell intensity vs transcripts (DAPI / Boundary / IntRNA)..."
+ )
+ plot_per_cell_intensity_vs_transcripts(
+ new_df, figures_dir, figures_source_dir, log_scale=True
+ )
+
+ tasks = [
+ _plot1_nuclear_texture_proportions,
+ _plot2_blur_proportions_roi,
+ # Latent figures disabled 2026-05-20 (user request) — function defs
+ # retained above as latent code: _plot3_nuclear_texture_density,
+ # _plot5_ccfs_vs_roi, _plot6_spatial_comparison,
+ # _plot7_cell_focus_distribution, _plot10_blur_prob_density_by_cluster.
+ _plot8_gmm_focus_vs_transcripts,
+ _plot9_tile_focus_gmm_spatial,
+ _plot9b_cell_flagged_maps_combined,
+ _plot11_intensity_transcript_correlation,
+ _plot12_per_cell_intensity_vs_transcripts,
+ ]
+
+ _run_figure_pool(tasks, phase="cell-based figures")
+
+ logging.info("All cell-based figures generated successfully!")
+
+
+# ===== QUARTO FIGURE GENERATION (from image_qc_processing.py) =====
+
+
+def generate_all_figures(
+ data,
+ df_spatial,
+ new_df,
+ myData,
+ small0,
+ small1,
+ small2,
+ distance_map,
+ distance_map2,
+ whole_sample,
+ holes,
+ artefacts,
+ ccfs_low_texture_threshold=DEFAULT_CCFS_LOW_TEXTURE_THRESHOLD,
+ multistain_whole_sample=None,
+ multistain_distance_map=None,
+ multistain_distance_map2=None,
+ figure_source_tables=False,
+):
+ """Generate ALL 13 Quarto-required figures using multithreading, and save the data used for each plot as CSV."""
+ figures_dir = data["figures_dir"]
+ figures_source_dir = (
+ (figures_dir / "figures_source") if figure_source_tables else None
+ )
+ if figures_source_dir is not None:
+ figures_source_dir.mkdir(parents=True, exist_ok=True)
+ # if df_spatial['In-Area-with-Artefact'] have a single unique value, use that value, otherwise use 0.5
+ if len(df_spatial["In-Area-with-Artefact"].unique()) == 1:
+ iawa_vmax = df_spatial["In-Area-with-Artefact"].unique()[0]
+ else:
+ iawa_vmax = 0.5
+
+ # §2.4 distance figures use the multi-stain extent mask + its distance maps when present,
+ # matching the edge/hole burden metrics in save_roi_qc_metrics. DAPI fallback (None on
+ # DAPI-only bundles) keeps those bundles byte-identical. distance_map/distance_map2 are
+ # referenced only by the distance closures, so rebinding them here is safe; the mask is
+ # aliased as _dm_mask because whole_sample is also used by the masks figure below.
+ if multistain_distance_map is not None:
+ distance_map = multistain_distance_map
+ if multistain_distance_map2 is not None:
+ distance_map2 = multistain_distance_map2
+ _dm_mask = (
+ whole_sample if multistain_whole_sample is None else multistain_whole_sample
+ )
+
+ def _fig1_distance_edge():
+ logging.info("Generating Figure 1: Distance map (edge)...")
+ _h, _w = distance_map.shape
+ _aspect = (_h / _w) if _w > 0 else 1.0
+ # Floor the height at 50% of width so very wide slides (e.g. brain) don't get squashed
+ fig, ax = plt.subplots(1, 1, figsize=(6, max(3, 6 * _aspect)))
+ # Phase v5: distance-to-edge readability fix.
+ # - viridis so near-edge tissue (low |distance|) renders as bright
+ # yellow against the white background — visible diagnostic region.
+ # - Absolute distance in µm: signed maurer is negative inside;
+ # |·| × 8 × 0.2125 → 0-at-boundary → max-deep-inside gradient.
+ # - NaN outside tissue mask → white via cmap.set_bad.
+ # - Thin black tissue outline as unambiguous boundary marker.
+ # TODO: 8 (downsample factor) and 0.2125 (Xenium native µm/px) are
+ # hardcoded here and in three other sites. See task #15 — future
+ # plumbing reads pixel_size from the bundle's experiment.xenium.
+ from copy import copy as _copy_cmap
+
+ _cmap_edge = _copy_cmap(plt.cm.viridis)
+ _cmap_edge.set_bad(color="white")
+ # Cap the colorscale at 300 µm so the near-edge band uses most of the
+ # spectrum; tiles further than 300 µm from the boundary saturate at
+ # yellow and the colorbar shows an "extend max" arrow. Beyond 300 µm
+ # the tile is unambiguously deep-tissue and not edge-affected.
+ _EDGE_VMAX_UM = 300.0
+ if _dm_mask.shape == distance_map.shape:
+ _dist_um = np.abs(distance_map) * 8 * 0.2125
+ _dm_edge = np.where(_dm_mask > 0, _dist_um, np.nan)
+ im = _imshow_thumb(
+ ax, _dm_edge, cmap=_cmap_edge, vmin=0.0, vmax=_EDGE_VMAX_UM
+ )
+ _edge_cbar_extend = "max"
+ else:
+ _dm_edge = (
+ distance_map # fallback: shapes mismatch, preserve prior behaviour
+ )
+ im = _imshow_thumb(ax, _dm_edge, cmap=_cmap_edge)
+ _edge_cbar_extend = "neither"
+ if _dm_mask.shape == distance_map.shape:
+ _contour_thumb(
+ ax,
+ (_dm_mask > 0).astype(np.uint8),
+ levels=[0.5],
+ colors="black",
+ linewidths=0.5,
+ )
+ ax.set_title("Distance to Edge")
+ ax.set_aspect("equal")
+ cbar = fig.colorbar(
+ im, ax=ax, fraction=0.035, pad=0.04, shrink=0.45, extend=_edge_cbar_extend
+ )
+ cbar.set_label("Distance from edge (µm)")
+ # Explicit "300+" label at the cap when extend="max" is active
+ if _edge_cbar_extend == "max":
+ cbar.set_ticks([0, 50, 100, 150, 200, 250, _EDGE_VMAX_UM])
+ cbar.set_ticklabels(
+ ["0", "50", "100", "150", "200", "250", f"{int(_EDGE_VMAX_UM)}+"]
+ )
+ plt.tight_layout()
+ plt.savefig(figures_dir / "distance_map_edge.pdf", dpi=300, bbox_inches="tight")
+ plt.savefig(figures_dir / "distance_map_edge.png", dpi=300, bbox_inches="tight")
+ plt.close(fig)
+ # Full-resolution export. These raster figures_source CSVs are gated behind
+ # --figure-source-tables (off by default), so the ~86-Mpx / ~1 GB dump only
+ # happens when a user explicitly asks for the raw plotting data — and then it
+ # must match the analysis exactly, not a thumbnail. The rendered PNG still uses
+ # the _imshow_thumb / _thumb display-resolution path for speed regardless.
+ if figures_source_dir is not None:
+ pd.DataFrame(distance_map).to_csv(
+ figures_source_dir / "distance_map_edge.csv",
+ index=False,
+ header=False,
+ )
+
+ def _fig2_distance_holes():
+ logging.info("Generating Figure 2: Distance map (holes)...")
+ _h, _w = distance_map2.shape
+ _aspect = (_h / _w) if _w > 0 else 1.0
+ # Floor the height at 50% of width so very wide slides (e.g. brain) don't get squashed
+ fig, ax = plt.subplots(1, 1, figsize=(6, max(3, 6 * _aspect)))
+ # Phase v5: distance-to-holes readability fix.
+ # - viridis colormap (same as edge map) for visual consistency.
+ # - Absolute distance in µm: |signed_maurer_distance| × 8 × 0.2125.
+ # - Linear vmin=0 / vmax=300 µm — matches the edge map for visual
+ # consistency across both panels of §2.4 (edge + holes). Tiles beyond 300 µm saturate
+ # at yellow with an "extend max" arrow on the colorbar.
+ # - NaN outside tissue mask → white via cmap.set_bad.
+ # - Thin black tissue outline as boundary marker.
+ # TODO: pixel-size conversion hardcoded — see task #15.
+ from copy import copy as _copy_cmap
+
+ _cmap_holes = _copy_cmap(plt.cm.viridis)
+ _cmap_holes.set_bad(color="white")
+ _HOLES_VMAX_UM = 300.0
+ if _dm_mask.shape == distance_map2.shape:
+ _dist_um_h = np.abs(distance_map2) * 8 * 0.2125
+ _dm_holes = np.where(_dm_mask > 0, _dist_um_h, np.nan)
+ im = _imshow_thumb(
+ ax, _dm_holes, cmap=_cmap_holes, vmin=0.0, vmax=_HOLES_VMAX_UM
+ )
+ _holes_cbar_extend = "max"
+ else:
+ _dm_holes = distance_map2 # fallback: shapes mismatch
+ im = _imshow_thumb(ax, _dm_holes, cmap=_cmap_holes)
+ _holes_cbar_extend = "neither"
+ if _dm_mask.shape == distance_map2.shape:
+ _contour_thumb(
+ ax,
+ (_dm_mask > 0).astype(np.uint8),
+ levels=[0.5],
+ colors="black",
+ linewidths=0.5,
+ )
+ ax.set_title("Distance to Nearest Hole")
+ ax.set_aspect("equal")
+ cbar = fig.colorbar(
+ im, ax=ax, fraction=0.035, pad=0.04, shrink=0.45, extend=_holes_cbar_extend
+ )
+ cbar.set_label("Distance from nearest hole (µm)")
+ # Explicit "300+" label at the cap when extend="max" is active.
+ if _holes_cbar_extend == "max":
+ cbar.set_ticks([0, 50, 100, 150, 200, 250, _HOLES_VMAX_UM])
+ cbar.set_ticklabels(
+ ["0", "50", "100", "150", "200", "250", f"{int(_HOLES_VMAX_UM)}+"]
+ )
+ plt.tight_layout()
+ plt.savefig(
+ figures_dir / "distance_map_holes.pdf", dpi=300, bbox_inches="tight"
+ )
+ plt.savefig(
+ figures_dir / "distance_map_holes.png", dpi=300, bbox_inches="tight"
+ )
+ plt.close(fig)
+ # Full-resolution export (see distance_map_edge.csv note): gated behind
+ # --figure-source-tables (off by default); rendering uses the display-res thumbnail.
+ if figures_source_dir is not None:
+ pd.DataFrame(distance_map2).to_csv(
+ figures_source_dir / "distance_map_holes.csv",
+ index=False,
+ header=False,
+ )
+
+ def _fig2b_distance_combined():
+ """Two-panel combined figure: distance to edge (left) + distance to
+ holes (right), sharing coordinate system and colorbar conventions.
+ Standalone _fig1_distance_edge / _fig2_distance_holes continue to
+ render the individual PNGs as latent artefacts; the combined figure
+ is what the QMD §2.4 embeds.
+ """
+ logging.info("Generating Figure 2b: Distance combined (edge + holes)...")
+ _h, _w = distance_map.shape
+ _aspect = (_h / _w) if _w > 0 else 1.0
+ _panel_width = 6
+ _panel_height = max(3, _panel_width * _aspect)
+ fig, axes = plt.subplots(1, 2, figsize=(2 * _panel_width + 2, _panel_height))
+
+ from copy import copy as _copy_cmap
+
+ _VMAX_UM = 300.0
+ _cmap = _copy_cmap(plt.cm.viridis)
+ _cmap.set_bad(color="white")
+
+ # ── Left panel: distance to edge ───────────────────────────────
+ ax = axes[0]
+ if _dm_mask.shape == distance_map.shape:
+ _dist_um = np.abs(distance_map) * 8 * 0.2125
+ _dm_edge = np.where(_dm_mask > 0, _dist_um, np.nan)
+ im_e = _imshow_thumb(ax, _dm_edge, cmap=_cmap, vmin=0.0, vmax=_VMAX_UM)
+ _contour_thumb(
+ ax,
+ (_dm_mask > 0).astype(np.uint8),
+ levels=[0.5],
+ colors="black",
+ linewidths=0.5,
+ )
+ _edge_extend = "max"
+ else:
+ im_e = _imshow_thumb(ax, distance_map, cmap=_cmap)
+ _edge_extend = "neither"
+ ax.set_title("Distance to Edge", fontsize=14)
+ ax.set_aspect("equal")
+ cbar_e = fig.colorbar(
+ im_e, ax=ax, fraction=0.035, pad=0.04, shrink=0.45, extend=_edge_extend
+ )
+ cbar_e.set_label("Distance from edge (µm)")
+ if _edge_extend == "max":
+ cbar_e.set_ticks([0, 50, 100, 150, 200, 250, _VMAX_UM])
+ cbar_e.set_ticklabels(
+ ["0", "50", "100", "150", "200", "250", f"{int(_VMAX_UM)}+"]
+ )
+
+ # ── Right panel: distance to holes ─────────────────────────────
+ ax = axes[1]
+ if _dm_mask.shape == distance_map2.shape:
+ _dist_um_h = np.abs(distance_map2) * 8 * 0.2125
+ _dm_holes = np.where(_dm_mask > 0, _dist_um_h, np.nan)
+ im_h = _imshow_thumb(ax, _dm_holes, cmap=_cmap, vmin=0.0, vmax=_VMAX_UM)
+ _contour_thumb(
+ ax,
+ (_dm_mask > 0).astype(np.uint8),
+ levels=[0.5],
+ colors="black",
+ linewidths=0.5,
+ )
+ _holes_extend = "max"
+ else:
+ im_h = _imshow_thumb(ax, distance_map2, cmap=_cmap)
+ _holes_extend = "neither"
+ ax.set_title("Distance to Nearest Hole", fontsize=14)
+ ax.set_aspect("equal")
+ cbar_h = fig.colorbar(
+ im_h, ax=ax, fraction=0.035, pad=0.04, shrink=0.45, extend=_holes_extend
+ )
+ cbar_h.set_label("Distance from nearest hole (µm)")
+ if _holes_extend == "max":
+ cbar_h.set_ticks([0, 50, 100, 150, 200, 250, _VMAX_UM])
+ cbar_h.set_ticklabels(
+ ["0", "50", "100", "150", "200", "250", f"{int(_VMAX_UM)}+"]
+ )
+
+ plt.tight_layout()
+ plt.savefig(figures_dir / "distance_maps.png", dpi=300, bbox_inches="tight")
+ plt.savefig(figures_dir / "distance_maps.pdf", dpi=300, bbox_inches="tight")
+ plt.close(fig)
+
+ def _fig3_morphology_overview():
+ logging.info("Generating Figure 3: Morphology overview...")
+ # Phase v5 TODO #12: 1×3 layout (DAPI + Boundary + Interior only). The
+ # artefacts panel that used to live in this figure's bottom-right is
+ # also rendered in imageqc_masks.png (§2.3 Masks), so showing it here
+ # duplicates the same plot — dropped here.
+ _img_aspect = small0.shape[0] / small0.shape[1] if small0.shape[1] > 0 else 1.0
+ _panel_width = 6
+ _panel_height = max(_panel_width * 0.5, _panel_width * _img_aspect)
+ fig, ax = plt.subplots(1, 3, figsize=(3 * _panel_width, _panel_height))
+ _imshow_thumb(ax[0], small0, cmap="Greys_r", vmax=np.percentile(small0, 99))
+ ax[0].set_title("DAPI", fontsize=14)
+ ax[0].set_aspect("equal")
+ ax[0].axis("off")
+ _imshow_thumb(ax[1], small1, cmap="Greys_r", vmax=np.percentile(small1, 99))
+ ax[1].set_title("Boundary", fontsize=14)
+ ax[1].set_aspect("equal")
+ ax[1].axis("off")
+ _imshow_thumb(ax[2], small2, cmap="Greys_r", vmax=np.percentile(small2, 99))
+ ax[2].set_title("Interior", fontsize=14)
+ ax[2].set_aspect("equal")
+ ax[2].axis("off")
+ plt.tight_layout()
+ plt.savefig(
+ figures_dir / "morphology_overview.pdf", dpi=300, bbox_inches="tight"
+ )
+ plt.savefig(
+ figures_dir / "morphology_overview.png", dpi=300, bbox_inches="tight"
+ )
+ plt.close(fig)
+ # Full-resolution export (see distance_map_edge.csv note): four ~86-Mpx channel
+ # grids, gated behind --figure-source-tables (off by default); rendering uses
+ # the display-res thumbnail so the figure wall time is unaffected.
+ if figures_source_dir is not None:
+ pd.DataFrame(small0).to_csv(
+ figures_source_dir / "morphology_overview_DAPI.csv",
+ index=False,
+ header=False,
+ )
+ pd.DataFrame(small1).to_csv(
+ figures_source_dir / "morphology_overview_Boundary.csv",
+ index=False,
+ header=False,
+ )
+ pd.DataFrame(small2).to_csv(
+ figures_source_dir / "morphology_overview_Interior.csv",
+ index=False,
+ header=False,
+ )
+ pd.DataFrame(artefacts).to_csv(
+ figures_source_dir / "morphology_overview_Artefacts.csv",
+ index=False,
+ header=False,
+ )
+
+ def _fig4_sample_qc_metrics():
+ logging.info("Generating Figure 4: Sample QC metrics...")
+ idx4 = _subsample_idx(np.ones(len(df_spatial), dtype=bool))
+ fig, ax = plt.subplots(2, 2, figsize=(12, 10))
+ _imshow_thumb(ax[0, 0], small0, cmap="Greys_r", vmax=np.percentile(small0, 99))
+ ax[0, 0].set_title("DAPI", fontsize=14)
+ ax[0, 0].set_aspect("equal")
+ ax[0, 1].scatter(
+ df_spatial["x"].iloc[idx4],
+ -df_spatial["y"].iloc[idx4],
+ c=df_spatial["Distance-to-edge"].iloc[idx4],
+ s=0.1,
+ marker="o",
+ rasterized=True,
+ )
+ ax[0, 1].set_title("Distance to sample Edge")
+ ax[0, 1].set_aspect("equal")
+ ax[0, 1].set_facecolor("black")
+ ax[1, 0].scatter(
+ df_spatial["x"].iloc[idx4],
+ -df_spatial["y"].iloc[idx4],
+ c=df_spatial["Distance-to-nearest-hole"].iloc[idx4],
+ s=0.1,
+ marker="o",
+ rasterized=True,
+ )
+ ax[1, 0].set_title("Distance to nearest hole")
+ ax[1, 0].set_aspect("equal")
+ ax[1, 0].set_facecolor("black")
+ ax[1, 1].scatter(
+ df_spatial["x"].iloc[idx4],
+ -df_spatial["y"].iloc[idx4],
+ c=df_spatial["In-Area-with-Artefact"].iloc[idx4],
+ cmap="Blues_r",
+ s=0.1,
+ marker="o",
+ vmax=iawa_vmax,
+ rasterized=True,
+ )
+ ax[1, 1].set_title("Overlap with sample artefact")
+ ax[1, 1].set_aspect("equal")
+ ax[1, 1].set_facecolor("black")
+ plt.tight_layout()
+ plt.savefig(figures_dir / "sample_qc_metrics.png", dpi=300, bbox_inches="tight")
+ plt.close(fig)
+ df_sample_qc_metrics = pd.DataFrame(
+ {
+ "x": df_spatial["x"],
+ "y": df_spatial["y"],
+ "Distance-to-edge": df_spatial["Distance-to-edge"],
+ "Distance-to-nearest-hole": df_spatial["Distance-to-nearest-hole"],
+ "In-Area-with-Artefact": df_spatial["In-Area-with-Artefact"],
+ }
+ )
+ if figures_source_dir is not None:
+ df_sample_qc_metrics.to_csv(
+ figures_source_dir / "sample_qc_metrics.csv", index=False
+ )
+
+ def _fig5_imageqc_masks():
+ logging.info("Generating Figure 5: ImageQC masks...")
+ # Phase v5: 1x3 triplet — three mask panels only. DAPI morphology
+ # image was dropped (already shown under §2.4 Stainings).
+ _img_aspect = small0.shape[0] / small0.shape[1] if small0.shape[1] > 0 else 1.0
+ _panel_width = 6
+ _panel_height = max(_panel_width * 0.5, _panel_width * _img_aspect)
+ # §2.3 tissue mask shows the EXTENT mask (all available stains) so it matches the
+ # reported tissue coverage; falls back to the DAPI mask on single-stain slides.
+ _extent_ws = (
+ multistain_whole_sample
+ if multistain_whole_sample is not None
+ else whole_sample
+ )
+ fig, ax = plt.subplots(1, 3, figsize=(3 * _panel_width, _panel_height))
+ ax[0].set_title("Tissue mask (all stains)", fontsize=14)
+ _imshow_thumb(ax[0], _extent_ws, rgb=lambda d: color.label2rgb(d, bg_label=0))
+ ax[0].set_aspect("equal")
+ ax[0].axis("off")
+ ax[1].set_title("Holes in sample", fontsize=14)
+ _imshow_thumb(ax[1], holes, rgb=lambda d: color.label2rgb(d, bg_label=0))
+ ax[1].set_aspect("equal")
+ ax[1].axis("off")
+ ax[2].set_title("Optically dense regions", fontsize=14)
+ _imshow_thumb(ax[2], artefacts, rgb=lambda d: color.label2rgb(d, bg_label=0))
+ ax[2].set_aspect("equal")
+ ax[2].axis("off")
+ plt.tight_layout()
+ plt.savefig(figures_dir / "imageqc_masks.pdf", dpi=300, bbox_inches="tight")
+ plt.savefig(figures_dir / "imageqc_masks.png", dpi=300, bbox_inches="tight")
+ plt.close(fig)
+ # DAPI csv export dropped — same data exposed by §2.4 Stainings.
+ # Full-resolution export (see distance_map_edge.csv note): three ~86-Mpx mask
+ # grids, gated behind --figure-source-tables (off by default); rendering uses
+ # the display-res thumbnail so the figure wall time is unaffected.
+ if figures_source_dir is not None:
+ pd.DataFrame(_extent_ws).to_csv(
+ figures_source_dir / "imageqc_masks_WholeSample.csv",
+ index=False,
+ header=False,
+ )
+ pd.DataFrame(holes).to_csv(
+ figures_source_dir / "imageqc_masks_Holes.csv",
+ index=False,
+ header=False,
+ )
+ pd.DataFrame(artefacts).to_csv(
+ figures_source_dir / "imageqc_masks_Artefacts.csv",
+ index=False,
+ header=False,
+ )
+
+ def _fig6_ccfs_spatial():
+ logging.info("Generating Figure 6: CCFS Spatial...")
+ idx6 = _subsample_idx(np.ones(len(myData), dtype=bool))
+ # Phase 11 (v5): aspect-adaptive figure size matching slide proportions
+ # (was a fixed 8x8 square). Mirrors plot_grid_roi_focus_heatmap.
+ _x = myData["centroid-1"]
+ _y = myData["centroid-0"]
+ _x_range = float(_x.max() - _x.min()) if len(_x) else 1.0
+ _y_range = float(_y.max() - _y.min()) if len(_y) else 1.0
+ _img_aspect = _y_range / _x_range if _x_range > 0 else 1.0
+ _panel_width = 6
+ _panel_height = max(_panel_width * 0.5, _panel_width * _img_aspect)
+ fig, ax = plt.subplots(1, 1, figsize=(_panel_width, _panel_height))
+ ax.scatter(
+ myData["centroid-1"].iloc[idx6],
+ -myData["centroid-0"].iloc[idx6],
+ s=0.1,
+ c=-myData["CCFS_DAPI"].iloc[idx6],
+ cmap="viridis",
+ vmin=-0.012,
+ rasterized=True,
+ )
+ ax.set_title("Calculated Cell Focus Score")
+ ax.set_facecolor("black")
+ ax.set_aspect("equal")
+ plt.tight_layout()
+ plt.savefig(figures_dir / "ccfs_spatial.png", dpi=300, bbox_inches="tight")
+ plt.close(fig)
+ df_ccfs_spatial = pd.DataFrame(
+ {
+ "centroid-1": myData["centroid-1"],
+ "centroid-0": myData["centroid-0"],
+ "CCFS_DAPI": myData["CCFS_DAPI"],
+ }
+ )
+ if figures_source_dir is not None:
+ df_ccfs_spatial.to_csv(figures_source_dir / "ccfs_spatial.csv", index=False)
+
+ def _fig7_ccfs_thresholded():
+ logging.info("Generating Figure 7: CCFS Thresholded...")
+ highC = myData[myData["is_high_nuclear_texture"]]
+ lowC = myData[myData["is_low_nuclear_texture"]]
+ idx7h = _subsample_idx(np.ones(len(highC), dtype=bool))
+ idx7l = _subsample_idx(np.ones(len(lowC), dtype=bool))
+ # Phase 11 (v5): aspect-adaptive figure size matching slide proportions
+ # (was a fixed 8x8 square). Uses full myData coordinate range so both
+ # high/low subsets share the same axis layout.
+ _x = myData["centroid-1"]
+ _y = myData["centroid-0"]
+ _x_range = float(_x.max() - _x.min()) if len(_x) else 1.0
+ _y_range = float(_y.max() - _y.min()) if len(_y) else 1.0
+ _img_aspect = _y_range / _x_range if _x_range > 0 else 1.0
+ _panel_width = 6
+ _panel_height = max(_panel_width * 0.5, _panel_width * _img_aspect)
+ fig, ax = plt.subplots(1, 1, figsize=(_panel_width, _panel_height))
+ # High-texture cells were `#1B2631` (near-black) on black background
+ # — invisible. Lightened to `#999999` (mid-grey) for clear contrast
+ # against the black facecolor while keeping the red low-texture cells
+ # visually dominant.
+ ax.scatter(
+ highC["centroid-1"].iloc[idx7h],
+ -highC["centroid-0"].iloc[idx7h],
+ s=0.1,
+ color="#999999",
+ rasterized=True,
+ )
+ ax.scatter(
+ lowC["centroid-1"].iloc[idx7l],
+ -lowC["centroid-0"].iloc[idx7l],
+ s=0.1,
+ color="red",
+ rasterized=True,
+ )
+ ax.set_title("Thresholded CCFS (red = low nuclear texture cells)")
+ ax.set_facecolor("black")
+ ax.set_aspect("equal")
+ plt.tight_layout()
+ plt.savefig(figures_dir / "ccfs_thresholded.png", dpi=300, bbox_inches="tight")
+ plt.close(fig)
+ df_ccfs_thresholded = pd.DataFrame(
+ {
+ "centroid-1": myData["centroid-1"],
+ "centroid-0": myData["centroid-0"],
+ "CCFS_DAPI": myData["CCFS_DAPI"],
+ "is_low_nuclear_texture": myData["is_low_nuclear_texture"],
+ }
+ )
+ if figures_source_dir is not None:
+ df_ccfs_thresholded.to_csv(
+ figures_source_dir / "ccfs_thresholded.csv", index=False
+ )
+
+ def _fig8_umap_multiple_metrics():
+ logging.info("Generating Figure 8: UMAP by multiple metrics...")
+ fig, ax = plt.subplots(2, 2, figsize=(15, 15))
+ ax[0, 0].scatter(
+ new_df["UMAP-1"],
+ new_df["UMAP-2"],
+ s=0.1,
+ c=-new_df["CCFS_DAPI"],
+ marker="o",
+ cmap="viridis",
+ vmin=-0.012,
+ rasterized=True,
+ )
+ ax[0, 0].set_title("UMAP by Nuclear Texture Score")
+ ax[0, 0].set_facecolor("black")
+ ax[0, 0].set_aspect("equal")
+ ax[0, 1].scatter(
+ new_df["UMAP-1"],
+ new_df["UMAP-2"],
+ s=0.1,
+ c=new_df["Cluster_kmeans10"],
+ marker="o",
+ rasterized=True,
+ )
+ ax[0, 1].set_title("UMAP by Cluster Allocation")
+ ax[0, 1].set_facecolor("black")
+ ax[0, 1].set_aspect("equal")
+ ax[1, 0].scatter(
+ new_df["UMAP-1"],
+ new_df["UMAP-2"],
+ s=0.1,
+ c=new_df["segPal"],
+ marker="o",
+ rasterized=True,
+ )
+ ax[1, 0].set_title("UMAP by Segmentation method")
+ ax[1, 0].set_facecolor("black")
+ ax[1, 0].set_aspect("equal")
+ ax[1, 1].scatter(
+ new_df["UMAP-1"],
+ new_df["UMAP-2"],
+ s=0.1,
+ c=new_df["transcript_counts"],
+ marker="o",
+ rasterized=True,
+ )
+ ax[1, 1].set_title("UMAP by Transcript counts")
+ ax[1, 1].set_facecolor("black")
+ ax[1, 1].set_aspect("equal")
+ plt.tight_layout()
+ plt.savefig(
+ figures_dir / "umap_multiple_metrics.png", dpi=300, bbox_inches="tight"
+ )
+ plt.close(fig)
+ df_umap_multiple_metrics = new_df[
+ [
+ "UMAP-1",
+ "UMAP-2",
+ "CCFS_DAPI",
+ "Cluster_kmeans10",
+ "segPal",
+ "transcript_counts",
+ ]
+ ]
+ if figures_source_dir is not None:
+ df_umap_multiple_metrics.to_csv(
+ figures_source_dir / "umap_multiple_metrics.csv", index=False
+ )
+
+ def _fig9_umap_distance_metrics():
+ logging.info("Generating Figure 9: UMAP by distance metrics...")
+ highC_new = new_df[new_df["is_high_nuclear_texture"]]
+ lowC_new = new_df[new_df["is_low_nuclear_texture"]]
+ fig, ax = plt.subplots(2, 2, figsize=(15, 15))
+ ax[0, 0].scatter(
+ new_df["UMAP-1"],
+ new_df["UMAP-2"],
+ s=0.1,
+ c=-new_df["Distance-to-edge"],
+ marker="o",
+ rasterized=True,
+ )
+ ax[0, 0].set_title("UMAP by Distance to edge")
+ ax[0, 0].set_facecolor("black")
+ ax[0, 0].set_aspect("equal")
+ ax[0, 1].scatter(
+ new_df["UMAP-1"],
+ new_df["UMAP-2"],
+ s=0.1,
+ c=-new_df["Distance-to-nearest-hole"],
+ marker="o",
+ rasterized=True,
+ )
+ ax[0, 1].set_title("UMAP by Distance to nearest hole")
+ ax[0, 1].set_facecolor("black")
+ ax[0, 1].set_aspect("equal")
+ ax[1, 0].scatter(
+ new_df["UMAP-1"],
+ new_df["UMAP-2"],
+ s=0.1,
+ c=new_df["In-Area-with-Artefact"],
+ cmap="Blues_r",
+ marker="o",
+ vmax=iawa_vmax,
+ rasterized=True,
+ )
+ ax[1, 0].set_title("UMAP by Overlap with artefact")
+ ax[1, 0].set_facecolor("black")
+ ax[1, 0].set_aspect("equal")
+ ax[1, 1].scatter(
+ highC_new["UMAP-1"],
+ highC_new["UMAP-2"],
+ s=0.1,
+ color="#1B2631",
+ rasterized=True,
+ )
+ ax[1, 1].scatter(
+ lowC_new["UMAP-1"], lowC_new["UMAP-2"], s=0.1, color="red", rasterized=True
+ )
+ ax[1, 1].set_title("UMAP by Blurred cells (in red)")
+ ax[1, 1].set_facecolor("black")
+ ax[1, 1].set_aspect("equal")
+ plt.tight_layout()
+ plt.savefig(
+ figures_dir / "umap_distance_metrics.png", dpi=300, bbox_inches="tight"
+ )
+ plt.close(fig)
+ df_umap_distance_metrics = new_df[
+ [
+ "UMAP-1",
+ "UMAP-2",
+ "Distance-to-edge",
+ "Distance-to-nearest-hole",
+ "In-Area-with-Artefact",
+ "CCFS_DAPI",
+ "is_low_nuclear_texture",
+ ]
+ ]
+ if figures_source_dir is not None:
+ df_umap_distance_metrics.to_csv(
+ figures_source_dir / "umap_distance_metrics.csv", index=False
+ )
+
+ def _fig10_umap_thresholded_metrics():
+ logging.info("Generating Figure 10: UMAP thresholded metrics...")
+ highE = new_df[new_df["is_far_from_edge"]]
+ lowE = new_df[new_df["is_near_edge"]]
+ highH = new_df[new_df["is_far_from_hole"]]
+ lowH = new_df[new_df["is_near_hole"]]
+ fig, ax = plt.subplots(2, 2, figsize=(15, 15))
+ ax[0, 0].scatter(
+ highE["UMAP-1"], highE["UMAP-2"], s=0.1, color="#1B2631", rasterized=True
+ )
+ ax[0, 0].scatter(
+ lowE["UMAP-1"], lowE["UMAP-2"], s=0.1, color="red", rasterized=True
+ )
+ ax[0, 0].set_title("UMAP by Distance to edge")
+ ax[0, 0].set_facecolor("black")
+ ax[0, 0].set_aspect("equal")
+ ax[0, 1].scatter(
+ highH["UMAP-1"], highH["UMAP-2"], s=0.1, color="#1B2631", rasterized=True
+ )
+ ax[0, 1].scatter(
+ lowH["UMAP-1"], lowH["UMAP-2"], s=0.1, color="red", rasterized=True
+ )
+ ax[0, 1].set_title("UMAP by Distance to nearest hole")
+ ax[0, 1].set_facecolor("black")
+ ax[0, 1].set_aspect("equal")
+ ax[1, 0].scatter(
+ new_df["UMAP-1"],
+ new_df["UMAP-2"],
+ s=0.1,
+ c=new_df["In-Area-with-Artefact"],
+ cmap="Blues_r",
+ marker="o",
+ vmax=iawa_vmax,
+ rasterized=True,
+ )
+ ax[1, 0].set_title("UMAP by Overlap with artefact")
+ ax[1, 0].set_facecolor("black")
+ ax[1, 0].set_aspect("equal")
+ ax[1, 1].scatter(
+ new_df[new_df["is_high_nuclear_texture"]]["UMAP-1"],
+ new_df[new_df["is_high_nuclear_texture"]]["UMAP-2"],
+ s=0.1,
+ color="#1B2631",
+ rasterized=True,
+ )
+ ax[1, 1].scatter(
+ new_df[new_df["is_low_nuclear_texture"]]["UMAP-1"],
+ new_df[new_df["is_low_nuclear_texture"]]["UMAP-2"],
+ s=0.1,
+ color="red",
+ rasterized=True,
+ )
+ ax[1, 1].set_title("UMAP by Blurred cells (in red)")
+ ax[1, 1].set_facecolor("black")
+ ax[1, 1].set_aspect("equal")
+ plt.tight_layout()
+ plt.savefig(
+ figures_dir / "umap_thresholded_metrics.png", dpi=300, bbox_inches="tight"
+ )
+ plt.close(fig)
+ # Save data as CSV
+ df_umap_thresholded_metrics = new_df[
+ [
+ "UMAP-1",
+ "UMAP-2",
+ "Distance-to-edge",
+ "Distance-to-nearest-hole",
+ "In-Area-with-Artefact",
+ "CCFS_DAPI",
+ "is_near_edge",
+ "is_near_hole",
+ "is_low_nuclear_texture",
+ ]
+ ]
+ if figures_source_dir is not None:
+ df_umap_thresholded_metrics.to_csv(
+ figures_source_dir / "umap_thresholded_metrics.csv", index=False
+ )
+
+ def _fig11_nuclear_texture_proportions():
+ logging.info("Generating Figure 11: Nuclear Texture Proportions by Cluster...")
+ plot_nuclear_texture_proportions(
+ new_df,
+ figures_dir,
+ figures_source_dir,
+ texture_threshold=ccfs_low_texture_threshold,
+ GROUP_BY_COLUMN="Cluster_kmeans10",
+ )
+
+ def _fig12_nuclear_texture_density():
+ logging.info("Generating Figure 12: Nuclear Texture Density by Cluster...")
+ plot_nuclear_texture_density(
+ new_df,
+ figures_dir,
+ figures_source_dir,
+ GROUP_BY_COLUMN="Cluster_kmeans10",
+ ccfs_low_texture_threshold=ccfs_low_texture_threshold,
+ )
+
+ def _fig13_nuclear_texture_vs_transcripts():
+ logging.info(
+ "Generating Figure 13: Nuclear Texture vs Transcripts (Log Scale)..."
+ )
+ plot_nuclear_texture_vs_transcripts(
+ new_df,
+ figures_dir,
+ figures_source_dir,
+ log_scale=True,
+ ccfs_low_texture_threshold=ccfs_low_texture_threshold,
+ )
+
+ def _fig14_cell_focus_distribution():
+ logging.info("Generating Figure 14: Cell-level focus score distribution...")
+ plot_cell_focus_distribution(new_df, figures_dir, figures_source_dir)
+
+ def _fig15_gmm_blur_proportions_by_cluster():
+ if (
+ "is_blurred_gmm_2d_roi" in new_df.columns
+ or "is_blurred_roi" in new_df.columns
+ ):
+ logging.info("Generating Figure 15: GMM Blur Proportions by Cluster...")
+ plot_tile_blur_proportions_roi(
+ new_df,
+ figures_dir,
+ figures_source_dir,
+ GROUP_BY_COLUMN="Cluster_kmeans10",
+ )
+
+ def _fig16_gmm_focus_vs_transcripts():
+ logging.info("Generating Figure 16: GMM Focus vs Transcripts...")
+ plot_gmm_focus_vs_transcripts(new_df, figures_dir, figures_source_dir)
+
+ def _fig17_tile_focus_gmm_spatial():
+ logging.info("Generating Figure 17: Tile-focus GMM spatial...")
+ plot_tile_focus_gmm_spatial(new_df, figures_dir, figures_source_dir)
+
+ def _fig17b_cell_flagged_maps_combined():
+ logging.info("Generating Figure 17b: Cell-flagged maps (combined)...")
+ plot_cell_flagged_maps_combined(new_df, figures_dir, figures_source_dir)
+
+ def _fig18_blur_prob_density_by_cluster():
+ logging.info("Generating Figure 18: Blur probability density by cluster...")
+ plot_blur_prob_density_by_cluster(new_df, figures_dir, figures_source_dir)
+
+ def _fig19_intensity_transcript_correlation():
+ logging.info(
+ "Generating Figure 19: Intensity-transcript correlation heatmap..."
+ )
+ plot_intensity_transcript_correlation(new_df, figures_dir, figures_source_dir)
+
+ def _fig20_per_cell_intensity_vs_transcripts():
+ logging.info(
+ "Generating Figure 20: Per-cell intensity vs transcripts (DAPI / Boundary / IntRNA)..."
+ )
+ plot_per_cell_intensity_vs_transcripts(
+ new_df, figures_dir, figures_source_dir, log_scale=True
+ )
+
+ tasks = [
+ # Deduped 2026-08-01: distance_map_edge/holes, distance_maps,
+ # morphology_overview and imageqc_masks are already rendered by the
+ # always-run tile path (generate_roi_figures); render once, not twice.
+ _fig7_ccfs_thresholded,
+ # Latent figures disabled 2026-05-20 (user request) — function defs
+ # retained above as latent code: _fig4_sample_qc_metrics,
+ # _fig6_ccfs_spatial, _fig8/_fig9/_fig10 UMAP, _fig12_nuclear_texture_density,
+ # _fig14_cell_focus_distribution, _fig18_blur_prob_density_by_cluster.
+ _fig11_nuclear_texture_proportions,
+ _fig13_nuclear_texture_vs_transcripts,
+ _fig15_gmm_blur_proportions_by_cluster,
+ _fig16_gmm_focus_vs_transcripts,
+ _fig17_tile_focus_gmm_spatial,
+ _fig17b_cell_flagged_maps_combined,
+ _fig19_intensity_transcript_correlation,
+ _fig20_per_cell_intensity_vs_transcripts,
+ ]
+
+ _run_figure_pool(tasks, phase="figures")
+
+ logging.info("All figures generated successfully!")
+ logging.info(f"[OK] Saved figures to {figures_dir}")
+ if figures_source_dir is not None:
+ logging.info(f"[OK] Saved source data to {figures_source_dir}")
+
+
+# ===== SIMPLE QC METRICS (from image_qc_processing.py) =====
+
+
+def save_simple_qc_metrics(new_df, outdir):
+ """Save QC metrics to CSV and JSON"""
+
+ # Move cell_id to first position
+ cell_id_col = new_df.pop("cell_id")
+ new_df.insert(0, "cell_id", cell_id_col)
+ # Save full dataset
+ new_df.to_csv(outdir / "image_qc_metrics.csv", index=False)
+
+ # Generate QC summary statistics
+ qc_metrics = {
+ "total_cells": len(new_df),
+ "cells_with_transcripts": len(new_df[new_df["transcript_counts"] > 0]),
+ "mean_transcript_count": float(new_df["transcript_counts"].mean()),
+ "median_transcript_count": float(new_df["transcript_counts"].median()),
+ "std_transcript_count": float(new_df["transcript_counts"].std()),
+ "mean_ccfs_dapi": float(new_df["CCFS_DAPI"].mean()),
+ "cells_high_nuclear_texture": len(new_df[new_df["is_high_nuclear_texture"]]),
+ "cells_low_nuclear_texture": len(new_df[new_df["is_low_nuclear_texture"]]),
+ "cells_near_edge": len(new_df[new_df["is_near_edge"]]),
+ "cells_near_holes": len(new_df[new_df["is_near_hole"]]),
+ "cells_with_artifacts": len(new_df[new_df["has_artifacts"]]),
+ "clusters_present": sorted(new_df["Cluster_kmeans10"].unique().tolist()),
+ "segmentation_methods": sorted(new_df["segmentation_method"].unique().tolist()),
+ }
+
+ # Save metrics
+ with open(outdir / "image_qc_metrics.json", "w") as f:
+ json.dump(qc_metrics, f, indent=2)
+
+ logging.info("\n=== IMAGE QC SUMMARY ===")
+ logging.info(f"Total cells analyzed: {qc_metrics['total_cells']:,}")
+ logging.info(f"Cells with transcripts: {qc_metrics['cells_with_transcripts']:,}")
+ logging.info(f"Mean transcript count: {qc_metrics['mean_transcript_count']:.1f}")
+ logging.info(f"Mean CCFS DAPI: {qc_metrics['mean_ccfs_dapi']:.6f}")
+ logging.info(
+ f"Cells high nuclear texture: {qc_metrics['cells_high_nuclear_texture']:,}"
+ )
+ logging.info(
+ f"Cells low nuclear texture: {qc_metrics['cells_low_nuclear_texture']:,}"
+ )
+ logging.info(f"Cells near edge: {qc_metrics['cells_near_edge']:,}")
+ logging.info(f"Cells near holes: {qc_metrics['cells_near_holes']:,}")
+ logging.info(f"Cells with artifacts: {qc_metrics['cells_with_artifacts']:,}")
+ logging.info(f"Clusters present: {qc_metrics['clusters_present']}")
+ logging.info(f"Segmentation methods: {qc_metrics['segmentation_methods']}")
+
+
+# ===== NEW: PIXEL-MAP AGGREGATION FOR CCFS =====
+
+
+# Pixels per row block for labeled reductions. 32 M px keeps each transient
+# float64/intp cast at ~256 MB, so a full pass costs ~1.3 GB regardless of how
+# large the image is.
+_LABEL_CHUNK_PIXELS = 32 * 1024 * 1024
+
+
+def _grow_to(arr: NDArray[Any], n: int) -> NDArray[Any]:
+ """Zero-extend *arr* to length *n*; no-op when it is already long enough."""
+ if arr.size >= n:
+ return arr
+ out = np.zeros(n, dtype=arr.dtype)
+ out[: arr.size] = arr
+ return out
+
+
+def _labeled_sums_chunked(
+ labels_img,
+ value_planes: dict[str, Any],
+ include_coords: bool = False,
+ rows_per_chunk: int | None = None,
+) -> tuple[NDArray[np.int64], dict[str, NDArray[np.float64]]]:
+ """Per-label pixel counts and value sums, accumulated in row blocks.
+
+ Equivalent to ``scipy.ndimage.sum`` over every label, but it never casts a
+ full-resolution plane. ``scipy.ndimage.mean`` reduces via ``np.bincount``,
+ which requires ``intp`` labels and ``float64`` weights, so a whole-image
+ call materialises an int64 copy of the label plane *and* a float64 copy of
+ the value plane — 88 GB combined on a 5.5 gigapixel sample, invisible in the
+ source. Blocking by rows bounds both casts to one block.
+
+ Counts and sums are additive, so a label straddling a block boundary
+ accumulates partial contributions from each block and its final
+ ``sum / count`` is exact — not an approximation.
+
+ Args:
+ labels_img: 2-D integer label plane. May be a lazily-sliced handle
+ (zarr array, memmap); only one row block is materialised at a time.
+ value_planes: Named 2-D value planes aligned with *labels_img*. May be
+ empty to collect counts only.
+ include_coords: Also accumulate coordinate sums, returned under the
+ ``centroid_y_sum`` / ``centroid_x_sum`` keys. ``sum / count`` of
+ these is exactly ``skimage.measure.regionprops`` ``centroid``.
+ rows_per_chunk: Rows per block. Defaults to ~32 M pixels per block.
+
+ Returns:
+ ``(counts, sums)`` indexed by raw label value, index 0 = background.
+ ``counts[i]`` is the pixel count for label ``i``; ``sums[name][i]`` the
+ summed value. Both are sized to the largest label actually seen.
+ """
+ height, width = labels_img.shape
+ if rows_per_chunk is None:
+ rows_per_chunk = max(1, _LABEL_CHUNK_PIXELS // max(width, 1))
+
+ counts = np.zeros(1, dtype=np.int64)
+ sums: dict[str, NDArray[np.float64]] = {
+ name: np.zeros(1, dtype=np.float64) for name in value_planes
+ }
+ if include_coords:
+ sums["centroid_y_sum"] = np.zeros(1, dtype=np.float64)
+ sums["centroid_x_sum"] = np.zeros(1, dtype=np.float64)
+
+ def _accumulate(key: str, block_sums: NDArray[np.float64]) -> None:
+ sums[key] = _grow_to(sums[key], block_sums.size)
+ sums[key][: block_sums.size] += block_sums
+
+ for y0 in range(0, height, rows_per_chunk):
+ y1 = min(y0 + rows_per_chunk, height)
+ lab = np.asarray(labels_img[y0:y1]).ravel()
+
+ block_counts = np.bincount(lab)
+ counts = _grow_to(counts, block_counts.size)
+ counts[: block_counts.size] += block_counts
+ n = block_counts.size
+
+ for name, plane in value_planes.items():
+ vals = np.asarray(plane[y0:y1], dtype=np.float64).ravel()
+ _accumulate(name, np.bincount(lab, weights=vals, minlength=n))
+ del vals
+
+ if include_coords:
+ n_rows = y1 - y0
+ rows = np.repeat(np.arange(y0, y1, dtype=np.float64), width)
+ _accumulate("centroid_y_sum", np.bincount(lab, weights=rows, minlength=n))
+ del rows
+ cols = np.tile(np.arange(width, dtype=np.float64), n_rows)
+ _accumulate("centroid_x_sum", np.bincount(lab, weights=cols, minlength=n))
+ del cols
+
+ del lab, block_counts
+
+ return counts, sums
+
+
+def calculate_ccfs_from_focus_maps(
+ focus_maps, cell_masks_zarr, cellseg_mask, xoa_morphology_files, streamed=None
+):
+ """
+ Derive per-cell CCFS from pixel focus maps using scipy.ndimage.mean.
+
+ This replaces the old regionprops-based calculate_ccfs_measurements() for
+ DAPI focus scores, but still uses regionprops for boundary/RNA per-cell
+ intensities (different channels).
+
+ Args:
+ focus_maps: dict from compute_all_focus_maps() with at least 'dapi_focus_map' and 'dapi_mean_map'
+ cell_masks_zarr: zarr group with 'masks/0' (nuclear mask) and 'masks/1' (cell mask)
+ cellseg_mask: numpy array of cell segmentation mask (from masks/1)
+ xoa_morphology_files: list of morphology file paths (for boundary/RNA channels)
+
+ Returns:
+ pandas DataFrame with columns: CellID, centroid-0, centroid-1,
+ CCFS_DAPI, mean_intensity, area_nucleus, area_cell,
+ mean_intensity_Boundary, mean_intensity_IntRNA
+ """
+ if streamed is None:
+ focus_map = focus_maps.get("dapi_focus_map")
+ mean_map = focus_maps.get("dapi_mean_map")
+ if focus_map is None or mean_map is None:
+ raise ValueError(
+ "focus_maps must contain 'dapi_focus_map' and 'dapi_mean_map'"
+ )
+ elif not streamed.has_per_cell:
+ raise ValueError(
+ "streamed results carry no per-cell reduction; "
+ "cell_masks_path was not passed to calculate_roi_focusscore"
+ )
+
+ # The nuclear label plane stays lazy: _labeled_sums_chunked slices row
+ # blocks straight out of zarr, so the full-res uint32 plane is never
+ # materialised (22 GB on a 5.5 GP sample). One blocked pass replaces
+ # np.unique() (a 22 GB flatten copy), two whole-image ndimage.mean() calls
+ # (88 GB of intp/float64 casts each), and regionprops() (a whole-plane
+ # pass) — centroids and areas fall out of the same accumulators.
+ if streamed is not None:
+ # The tile pass already folded every tile into these, so there is no plane
+ # left to read. Keys are the accumulator's, mapped onto this function's.
+ logging.info(" Per-nucleus sums reduced during the tile pass (streamed)")
+ nuc_counts = streamed.nuclear_counts
+ nuc_sums = {
+ "focus": streamed.nuclear_sums["focus_map"],
+ "intensity": streamed.nuclear_sums["mean_map"],
+ "centroid_y_sum": streamed.nuclear_sums["centroid_y_sum"],
+ "centroid_x_sum": streamed.nuclear_sums["centroid_x_sum"],
+ }
+ else:
+ nuclear_mask = cell_masks_zarr.get("masks").get("0")
+ logging.info(
+ " Aggregating focus/intensity over nuclear masks (row-blocked)..."
+ )
+ t0 = time.time()
+ nuc_counts, nuc_sums = _labeled_sums_chunked(
+ nuclear_mask,
+ {"focus": focus_map, "intensity": mean_map},
+ include_coords=True,
+ )
+ logging.info(f" [TIMING] labeled nuclear aggregation: {time.time() - t0:.1f}s")
+
+ # Labels present, ascending, background excluded — same set and order as
+ # regionprops(), which yields one region per distinct label value.
+ labels = np.nonzero(nuc_counts)[0]
+ labels = labels[labels > 0]
+ if labels.size == 0:
+ raise ValueError("nuclear mask contains no labelled pixels")
+ counts = nuc_counts[labels].astype(np.float64)
+
+ cell_focus = nuc_sums["focus"][labels] / counts
+ cell_intensity = nuc_sums["intensity"][labels] / counts
+ # sum(coord)/count is exactly regionprops' centroid definition.
+ centroid_y = nuc_sums["centroid_y_sum"][labels] / counts
+ centroid_x = nuc_sums["centroid_x_sum"][labels] / counts
+
+ # Normalize: 99th percentile of mean intensity
+ dapi_norm = np.percentile(cell_intensity, 99)
+ if dapi_norm == 0:
+ dapi_norm = 1.0
+ ccfs_dapi = cell_focus / dapi_norm
+
+ # Map each nucleus centroid to a cell ID (point lookups only)
+ iy = np.minimum(centroid_y.astype(np.int64), cellseg_mask.shape[0] - 1)
+ ix = np.minimum(centroid_x.astype(np.int64), cellseg_mask.shape[1] - 1)
+ cell_ids = np.asarray(cellseg_mask[iy, ix]).astype(np.int64)
+
+ nucleus_props = pd.DataFrame(
+ {
+ "label": labels,
+ "centroid-0": centroid_y,
+ "centroid-1": centroid_x,
+ "area_nucleus": nuc_counts[labels],
+ "CCFS_DAPI": ccfs_dapi,
+ "mean_intensity": cell_intensity,
+ "CellID": cell_ids,
+ }
+ )
+ del nuc_counts, nuc_sums
+
+ # One blocked pass over the cell mask yields cell areas *and* the per-cell
+ # boundary/IntRNA means, replacing np.bincount() on the full-res uint32
+ # plane (a 44 GB intp cast) plus one ndimage.mean() per channel.
+ if streamed is not None:
+ logging.info(" Per-cell sums reduced during the tile pass (streamed)")
+ cell_counts = streamed.cell_counts
+ cell_sums = streamed.cell_sums or {}
+ else:
+ cell_value_planes: dict[str, Any] = {}
+ boundary_mean = focus_maps.get("boundary_mean_map")
+ intrna_mean = focus_maps.get("intrna_mean_map")
+ if boundary_mean is not None:
+ cell_value_planes["boundary"] = boundary_mean
+ if intrna_mean is not None:
+ cell_value_planes["intrna"] = intrna_mean
+
+ logging.info(
+ " Aggregating cell areas/intensities over cell masks (row-blocked)..."
+ )
+ t0 = time.time()
+ cell_counts, cell_sums = _labeled_sums_chunked(cellseg_mask, cell_value_planes)
+ logging.info(f" [TIMING] labeled cell aggregation: {time.time() - t0:.1f}s")
+
+ cids = nucleus_props["CellID"].to_numpy()
+ in_range = cids < cell_counts.size
+
+ # area_cell: index by raw CellID, 0 when out of range. CellID 0 keeps
+ # resolving to the background pixel count, matching the previous mapping.
+ area_cell = np.zeros(cids.size, dtype=np.int64)
+ area_cell[in_range] = cell_counts[cids[in_range]]
+ nucleus_props["area_cell"] = area_cell
+
+ # Per-cell channel means. CellID 0 stays NaN: the old code built its
+ # lookup from the >0 labels only, so background mapped to NaN.
+ labelled = in_range & (cids > 0)
+ for name, column in (
+ ("boundary", "mean_intensity_Boundary"),
+ ("intrna", "mean_intensity_IntRNA"),
+ ):
+ if name not in cell_sums:
+ nucleus_props[column] = np.nan
+ continue
+ with np.errstate(invalid="ignore", divide="ignore"):
+ per_cell = cell_sums[name] / cell_counts
+ values = np.full(cids.size, np.nan, dtype=np.float64)
+ values[labelled] = per_cell[cids[labelled]]
+ nucleus_props[column] = values
+
+ return nucleus_props
+
+
+# ===== FALLBACK: REGIONPROPS-BASED CCFS (for legacy mode) =====
+
+
+def calculate_ccfs_measurements(xoa_morphology_files, cellseg_mask, cell_masks_zarr):
+ """
+ Calculate CCFS measurements with improved variable naming and memory management.
+ Loads data just before use and deletes it immediately after.
+
+ Legacy path (--legacy-focus). regionprops_table needs a materialised label
+ array, so a LazyLabelPlane is realised here rather than in the caller -- the
+ streaming path never does this.
+ """
+ if isinstance(cellseg_mask, LazyLabelPlane):
+ logging.info(
+ " Materialising the cell mask for regionprops (legacy focus path)..."
+ )
+ # A full-height slice; LazyLabelPlane already returns numpy. Not
+ # np.asarray(...): this function has its own local `import numpy as np`
+ # further down, so `np` is a local name here and referencing it before
+ # that import raises UnboundLocalError.
+ cellseg_mask = cellseg_mask[0 : cellseg_mask.shape[0]]
+ import numpy as np
+ import pandas as pd
+
+ # Load full resolution image channels only when needed
+ fullres_channels = imread(
+ xoa_morphology_files[0], is_ome=False, level=0, aszarr=False
+ )
+
+ # Check number of channels (could be 2D array for single channel, or 3D array for multi-channel)
+ if len(fullres_channels.shape) == 2:
+ # Single channel (2D array)
+ dapi_image = fullres_channels
+ boundary_image = None
+ rna_image = None
+ else:
+ # Multi-channel (3D array: channels, height, width)
+ dapi_image = fullres_channels[0]
+ boundary_image = fullres_channels[1] if fullres_channels.shape[0] > 1 else None
+ rna_image = fullres_channels[2] if fullres_channels.shape[0] > 2 else None
+ del fullres_channels
+
+ # Load nuclear segmentation mask
+ nuclear_mask = np.array(cell_masks_zarr.get("masks").get("0"))
+
+ # Per-nucleus measurements on DAPI image
+ nucleus_props = pd.DataFrame(
+ regionprops_table(dapi_image, nuclear_mask, position=True)
+ )
+ # Calculate CCFS_DAPI
+ mean_intensity = nucleus_props["mean_intensity"]
+ std_intensity = nucleus_props["standard_deviation_intensity"]
+ dapi_norm = np.percentile(mean_intensity, 99)
+ ccfs_dapi = (std_intensity * std_intensity) / (mean_intensity * dapi_norm)
+ nucleus_props["CCFS_DAPI"] = ccfs_dapi
+ del dapi_image
+
+ # Get CellID for each nucleus by measuring mean intensity in cellseg_mask
+ cellid_props = pd.DataFrame(
+ regionprops_table(cellseg_mask, nuclear_mask, position=True)
+ )
+ nucleus_props["CellID"] = cellid_props["mean_intensity"].astype(int)
+ del nuclear_mask, cellid_props
+
+ # Measure boundary (red channel) intensity per cell
+ if boundary_image is not None:
+ boundary_props = pd.DataFrame(regionprops_table(boundary_image, cellseg_mask))
+ boundary_props["CellID"] = boundary_props["label"].astype(int)
+ del boundary_image
+ else:
+ # Create empty boundary_props if boundary channel not available
+ boundary_props = pd.DataFrame(
+ {"CellID": nucleus_props["CellID"], "mean_intensity": np.nan}
+ )
+
+ # Measure RNA (interior) intensity per cell
+ if rna_image is not None:
+ rna_props = pd.DataFrame(regionprops_table(rna_image, cellseg_mask))
+ rna_props["CellID"] = rna_props["label"].astype(int)
+ rna_props["mean_intensity_IntRNA"] = rna_props["mean_intensity"]
+ del rna_image
+ else:
+ # Create empty rna_props if RNA channel not available
+ rna_props = pd.DataFrame(
+ {
+ "CellID": nucleus_props["CellID"],
+ "area": np.nan,
+ "mean_intensity_IntRNA": np.nan,
+ }
+ )
+
+ # Merge all measurements into a single DataFrame
+ merged_data = pd.merge(
+ nucleus_props,
+ rna_props[["CellID", "area"]],
+ how="inner",
+ on="CellID",
+ suffixes=("_nucleus", "_cell"),
+ )
+ merged_data = pd.merge(
+ merged_data,
+ boundary_props[["CellID", "mean_intensity"]],
+ how="inner",
+ on="CellID",
+ suffixes=("_DAPI", "_Boundary"),
+ )
+ merged_data = pd.merge(
+ merged_data,
+ rna_props[["CellID", "mean_intensity_IntRNA"]],
+ how="inner",
+ on="CellID",
+ )
+
+ return merged_data
+
+
+# ===== COMBINED MAIN =====
+
+
+def _check_cell_data_exists(xenium_bundle_dir):
+ """
+ Check if cell data files exist in the Xenium bundle directory.
+
+ Args:
+ xenium_bundle_dir: Path to Xenium bundle directory
+
+ Returns:
+ bool: True if required cell data files exist
+ """
+ xenium_bundle_dir = Path(xenium_bundle_dir)
+ required_files = [
+ xenium_bundle_dir / "cells.parquet",
+ xenium_bundle_dir / "cells.zarr.zip",
+ ]
+ # Check clustering path
+ clusters_path = (
+ xenium_bundle_dir
+ / "analysis"
+ / "clustering"
+ / "gene_expression_kmeans_10_clusters"
+ / "clusters.csv"
+ )
+
+ # Also check for analysis.tar.gz (test data)
+ has_analysis = (
+ clusters_path.exists() or (xenium_bundle_dir / "analysis.tar.gz").is_file()
+ )
+
+ for f in required_files:
+ if not f.exists():
+ logging.info(f" Cell data check: {f.name} not found")
+ return False
+
+ if not has_analysis:
+ logging.info(" Cell data check: clustering/UMAP data not found")
+ return False
+
+ return True
+
+
+@click.command()
+@click.option(
+ "--xenium-bundle-dir", required=True, help="Path to Xenium bundle directory"
+)
+@click.option("--outdir", required=True, help="Output directory for results")
+@click.option(
+ "--stain-names",
+ default=None,
+ help="Semicolon-separated list of stain names. If not provided, uses defaults.",
+)
+@click.option(
+ "--roi-size", default=35, type=int, show_default=True, help="Tile size in pixels"
+)
+@click.option(
+ "--max-scatter-points",
+ default=10000,
+ type=int,
+ show_default=True,
+ help="Maximum number of points to plot in scatter figures. Set to 0 to plot all points.",
+)
+@click.option(
+ "--legacy-focus",
+ is_flag=True,
+ default=False,
+ help="Use legacy per-tile loop focus scoring instead of the default convolution-based GPU-accelerated method.",
+)
+@click.option(
+ "--sample-id",
+ default=None,
+ help="Sample identifier for logging and metrics output.",
+)
+@click.option(
+ "--no-snr",
+ is_flag=True,
+ default=False,
+ help="Disable SNR metrics (image Otsu / quartiles, transcripts, slide matrix, neg spatial).",
+)
+@click.option(
+ "--snr-no-roi-tx-table",
+ is_flag=True,
+ default=False,
+ help="Do not write SNR_roi_tx.parquet (or .csv.gz) alongside roi_qc_metrics.",
+)
+@click.option(
+ "--snr-otsu-max-rois",
+ type=int,
+ default=None,
+ help="Cap tiles for per-tile Otsu image SNR (default: all tiles).",
+)
+@click.option(
+ "--snr-with-moran",
+ is_flag=True,
+ default=False,
+ help="SNR only: enable Moran's I for neg spatial (needs PySAL/esda; default off).",
+)
+@click.option(
+ "--stream-tiles/--no-stream-tiles",
+ "stream_tiles",
+ default=True,
+ help=(
+ "Reduce each tile as it is computed instead of assembling full-resolution "
+ "pixel planes (default: stream). The planes cost ~154 GB of scratch on a "
+ "5.5 gigapixel sample and mmap over a FUSE/S3 work directory is "
+ "pathological; no downstream metric needs a whole plane. "
+ "--no-stream-tiles restores the plane-based path, and "
+ "--save-dapi-maps-tiff implies it."
+ ),
+)
+@click.option(
+ "--save-dapi-maps-tiff",
+ "save_dapi_maps_tiff",
+ is_flag=True,
+ default=False,
+ help=(
+ "Write full-resolution per-pixel maps as tiled float32 TIFF "
+ "(dapi_focus/mean/lap_var and boundary/intrna focus/mean when present). "
+ "Default off — QC and SNR use in-memory arrays only; enable for archival "
+ "or external tools (large files)."
+ ),
+)
+@click.option(
+ "--max-gpus",
+ "max_gpus",
+ default=0,
+ type=int,
+ help=(
+ "Cap the number of CUDA devices used (0 = use every device detected). "
+ "Nextflow's `accelerator` directive only sizes the Batch request; it does "
+ "not restrict CUDA visibility, so a task that asked for one GPU but landed "
+ "on a multi-GPU instance would otherwise use all of them."
+ ),
+)
+@click.option(
+ "--roi-thresholds-yaml",
+ default=None,
+ type=click.Path(exists=True),
+ help="Path to tile image QC thresholds YAML. Overrides hardcoded defaults.",
+)
+@click.option(
+ "--lap-sigma",
+ default=1.0,
+ type=float,
+ show_default=True,
+ help="Gaussian sigma for Laplacian of Gaussian (LoG) pre-smoothing.",
+)
+@click.option(
+ "--pipeline-segmentation",
+ default="skip",
+ show_default=True,
+ help="Pipeline segmentation method (params.segmentation); 'skip' for none.",
+)
+@click.option(
+ "--is-resegmented",
+ is_flag=True,
+ default=False,
+ help="Set when this run analyses a pipeline-resegmented bundle (post-seg).",
+)
+@click.option(
+ "--figure-source-tables/--no-figure-source-tables",
+ "figure_source_tables",
+ default=False,
+ help=(
+ "Write the per-figure figures_source/*.csv source-data exports "
+ "(unused downstream; default off; ~80s on a 5.5 GP sample)."
+ ),
+)
+@click.option(
+ "--figures/--no-figures",
+ "figures",
+ default=True,
+ help=(
+ "Generate QC figures (default true; --no-figures skips all figure "
+ "rendering for a metrics-only fast run). Metric/JSON/parquet outputs "
+ "are always computed regardless of this flag."
+ ),
+)
+def main(
+ xenium_bundle_dir,
+ outdir,
+ stain_names,
+ roi_size,
+ max_scatter_points,
+ legacy_focus,
+ sample_id,
+ no_snr,
+ snr_no_roi_tx_table,
+ snr_otsu_max_rois,
+ snr_with_moran,
+ save_dapi_maps_tiff,
+ stream_tiles,
+ max_gpus,
+ roi_thresholds_yaml,
+ lap_sigma,
+ pipeline_segmentation,
+ is_resegmented,
+ figure_source_tables,
+ figures,
+):
+ """
+ Combined Xenium Image QC pipeline.
+
+ Performs pixel-level focus analysis (GPU-accelerated), tile-level analysis,
+ and optionally cell-level analysis when cell data is available.
+
+ Produces: tile figures, cell figures, image_qc_metrics.json, image_qc_metrics.csv,
+ 13 Quarto-required PNGs, and versions.yml.
+ """
+ logging.basicConfig(
+ level=logging.INFO,
+ format="%(asctime)s [%(levelname)s] %(message)s",
+ datefmt="%Y-%m-%d %H:%M:%S",
+ )
+
+ global _MAX_SCATTER_POINTS
+ if max_scatter_points > 0:
+ _MAX_SCATTER_POINTS = max_scatter_points
+
+ t_total_start = time.time()
+ logging.info("=" * 60)
+ logging.info("Starting Combined Xenium Image QC Pipeline")
+ if sample_id:
+ logging.info("Sample ID: %s", sample_id)
+ logging.info("=" * 60)
+ logging.info(f"Input directory: {xenium_bundle_dir}")
+ logging.info(f"Output directory: {outdir}")
+
+ # Load QC thresholds from YAML (or use hardcoded defaults)
+ qc_thresholds = _load_qc_thresholds(roi_thresholds_yaml)
+ if roi_thresholds_yaml:
+ logging.info(f"Loaded QC thresholds from: {roi_thresholds_yaml}")
+ else:
+ logging.info("Using built-in default QC thresholds")
+ # Resolve lap_sigma from YAML (CLI value is the fallback)
+ _yaml_sigma = qc_thresholds.get("lap_sigma", lap_sigma)
+ try:
+ _lap_sigma = float(_yaml_sigma)
+ except (TypeError, ValueError):
+ logging.warning(
+ "Invalid lap_sigma value %r in YAML, falling back to CLI=%s",
+ _yaml_sigma,
+ lap_sigma,
+ )
+ _lap_sigma = float(lap_sigma)
+ logging.info(
+ f"LoG sigma: {_lap_sigma} (CLI={lap_sigma}, YAML={qc_thresholds.get('lap_sigma', 'not set')})"
+ )
+
+ # Resolve top-level operational thresholds from YAML (with hardcoded fallbacks)
+ _roi_intensity_threshold = float(
+ qc_thresholds.get("roi_intensity_threshold", ROI_INTENSITY_THRESHOLD)
+ )
+ _min_tissue_cov = float(
+ qc_thresholds.get(
+ "min_tissue_coverage_for_intensity_qc",
+ ROI_MIN_TISSUE_COVERAGE_FOR_INTENSITY_QC,
+ )
+ )
+ _roi_focus_pct = float(
+ qc_thresholds.get("roi_focus_score_percentile", ROI_FOCUS_SCORE_PERCENTILE)
+ )
+ _blur_prob_thresh = float(qc_thresholds.get("blur_prob_threshold", 0.5))
+ _focus_cfg = qc_thresholds.get("focus") or {}
+ _ccfs_low_texture_threshold = float(
+ _focus_cfg.get("ccfs_low_texture_threshold", DEFAULT_CCFS_LOW_TEXTURE_THRESHOLD)
+ )
+ logging.info(
+ f" roi_intensity_threshold={_roi_intensity_threshold}, "
+ f"min_tissue_cov={_min_tissue_cov}, "
+ f"roi_focus_pct={_roi_focus_pct}, "
+ f"blur_prob_thresh={_blur_prob_thresh}, "
+ f"ccfs_low_texture_threshold={_ccfs_low_texture_threshold}"
+ )
+
+ # ===== PHASE 1: LOAD =====
+ logging.info("\n--- PHASE 1: LOAD ---")
+
+ # Handle stain_names (default if None)
+ if stain_names is None:
+ stain_names_list = [
+ "DAPI",
+ "Boundary (ATP1A1/E-Cadherin/CD45)",
+ "Interior - RNA (18S)",
+ "Protein (alphaSMA/Vimentin)",
+ ]
+ logging.info(f"Using default stain names: {stain_names_list}")
+ else:
+ logging.info(f"Stain names: {stain_names}")
+ if ";" in stain_names:
+ stain_names_list = stain_names.split(";")
+ else:
+ stain_names_list = [stain_names]
+
+ # Load downsampled images for sample QC
+ morphology_focus_dir = Path(xenium_bundle_dir) / "morphology_focus"
+ if (morphology_focus_dir / "morphology_focus_0000.ome.tif").exists():
+ xoa_morphology_files = [
+ morphology_focus_dir / "morphology_focus_0000.ome.tif",
+ morphology_focus_dir / "morphology_focus_0001.ome.tif",
+ morphology_focus_dir / "morphology_focus_0002.ome.tif",
+ morphology_focus_dir / "morphology_focus_0003.ome.tif",
+ ]
+ else:
+ xoa_morphology_files = sorted(
+ list(morphology_focus_dir.glob("ch000*.ome.tif")),
+ key=lambda x: x.stem.split("_")[0],
+ )
+
+ # Boundary / IntRNA / Protein channels are optional — DAPI-only morphology
+ # bundles are valid (e.g. samples staining nuclei only). The downstream
+ # loader `_load_morphology_channels` returns `None` for missing channels;
+ # `generate_tissue_mask` and per-cell intensity emissions handle None.
+ if len(xoa_morphology_files) < 1:
+ raise ValueError("No morphology files found in morphology_focus/ directory")
+ if not xoa_morphology_files[0].exists():
+ raise ValueError(
+ f"Morphology focus file does not exist: {xoa_morphology_files[0]}"
+ )
+
+ # Load and prepare data (creates output directories)
+ data = load_and_prepare_data(xenium_bundle_dir, outdir)
+
+ # Load morphology images (level 3, downsampled)
+ logging.info("Loading morphology images...")
+ t0 = time.time()
+ # Use the robust loader that already handles 1/2/3-channel inputs
+ # (returns None for missing Boundary / IntRNA channels). 5 other call
+ # sites already exercise this loader (lines ~1366/1389/1659/1677/3456),
+ # so the None-tolerant code path is well-tested for DAPI-only bundles.
+ small0, small1, small2 = _load_morphology_channels(xoa_morphology_files, level=3)
+ _n_channels_loaded = sum(x is not None for x in (small0, small1, small2))
+ _shape_str = f"DAPI={small0.shape}"
+ if small1 is not None:
+ _shape_str += f", Boundary={small1.shape}"
+ if small2 is not None:
+ _shape_str += f", IntRNA={small2.shape}"
+ logging.info(
+ f"Loaded morphology images ({_n_channels_loaded} channel(s)): {_shape_str}"
+ )
+ logging.info(f"[TIMING] Loading morphology images: {time.time() - t0:.1f}s")
+ _log_mem("morphology loaded")
+
+ # Check if cell data exists
+ has_cell_data = _check_cell_data_exists(xenium_bundle_dir)
+ if has_cell_data:
+ logging.info("Cell data detected - will perform cell-level analysis")
+ else:
+ logging.info("No cell data detected - will skip cell-level analysis")
+
+ # ===== PHASE 2: PIXEL-LEVEL FOCUS MAPS =====
+ logging.info("\n--- PHASE 2: PIXEL-LEVEL FOCUS MAPS ---")
+
+ # Generate tissue masks and distance maps (cell-independent)
+ logging.info("Generating tissue masks and distance maps...")
+ t0 = time.time()
+ (
+ whole_sample,
+ holes,
+ dense_intensity_regions,
+ distance_map,
+ distance_map2,
+ ms_whole_sample,
+ ms_distance_map,
+ ms_distance_map2,
+ ) = generate_tissue_mask(xoa_morphology_files, small0, small1, small2)
+ logging.info("Generated tissue masks and distance maps")
+ logging.info(f"[TIMING] Tissue mask generation: {time.time() - t0:.1f}s")
+ _log_mem("tissue mask")
+
+ # Validate ROI size
+ if roi_size <= 0:
+ roi_size = 35
+ logging.info(f"Using default tile size: {roi_size}px")
+ else:
+ logging.info(f"Using tile size: {roi_size}px")
+
+ # Auto-detect available GPUs
+ available_gpus = detect_gpu_ids()
+ if available_gpus and max_gpus and len(available_gpus) > max_gpus:
+ logging.info(
+ f"Detected {len(available_gpus)} GPU(s) {available_gpus} but --max-gpus "
+ f"={max_gpus}; using {available_gpus[:max_gpus]}. The accelerator "
+ "directive sizes the Batch request only, it does not limit CUDA "
+ "visibility."
+ )
+ available_gpus = available_gpus[:max_gpus]
+ if available_gpus:
+ logging.info(f"Detected {len(available_gpus)} GPU(s): {available_gpus}")
+ else:
+ logging.info("No GPUs detected, using CPU backend")
+
+ # Calculate cell-independent grid ROI focus scores
+ logging.info("Calculating cell-independent grid tile focus scores...")
+ t0 = time.time()
+ if legacy_focus:
+ logging.info(" Using LEGACY per-tile loop method (--legacy-focus)")
+ df_grid_roi = calculate_roi_focusscore_without_laplace(
+ xoa_morphology_files,
+ roi_size=roi_size,
+ stride=None,
+ tissue_filter=True,
+ min_tissue_coverage=0.0,
+ )
+ focus_maps = None
+ streamed = None
+ else:
+ logging.info(" Using convolution-based GPU-accelerated method")
+ # Streaming folds each tile into small reductions and never assembles a
+ # pixel plane. --save-dapi-maps-tiff is the one output that genuinely needs
+ # the planes, so asking for it selects the plane-based path.
+ _stream = stream_tiles and not save_dapi_maps_tiff
+ if stream_tiles and save_dapi_maps_tiff:
+ logging.info(
+ " --save-dapi-maps-tiff requires full pixel planes; "
+ "streaming disabled for this run"
+ )
+ logging.info(
+ f" Tile reduction mode: {'streamed' if _stream else 'pixel planes'}"
+ )
+ df_grid_roi, focus_maps, streamed = calculate_roi_focusscore(
+ xoa_morphology_files,
+ roi_size=roi_size,
+ stride=None,
+ tissue_filter=True,
+ min_tissue_coverage=0.0,
+ gpu_ids=available_gpus if available_gpus else None,
+ return_pixel_maps=True,
+ lap_sigma=_lap_sigma,
+ stream_tiles=_stream,
+ # Gated on has_cell_data for the same reason the CCFS block below is:
+ # a bundle without cells.zarr.zip has no masks to reduce over, and
+ # opening it during the tile pass would fail where the plane path
+ # simply skipped the whole cell section.
+ cell_masks_path=(
+ data["cell_masks_path"] if _stream and has_cell_data else None
+ ),
+ # Reuse the already-loaded level-3 DAPI plane and the tissue mask
+ # generate_tissue_mask just computed from it, instead of re-decoding
+ # level 3 and recomputing the mask inside calculate_roi_focusscore.
+ # `whole_sample` == compute_tissue_mask(small0)[0] (same small0, same
+ # min_size_hole=1500), so the result is bit-identical.
+ small0_ds=small0,
+ tissue_mask=whole_sample,
+ )
+ logging.info(f"Calculated grid tile focus scores for {len(df_grid_roi):,} tiles")
+ logging.info(f"[TIMING] calculate_roi_focusscore(): {time.time() - t0:.1f}s")
+ _log_mem("focus maps built")
+
+ # ===== PHASE 3: ROI-LEVEL ANALYSIS =====
+ logging.info("\n--- PHASE 3: TILE-LEVEL ANALYSIS ---")
+
+ # Calculate ROI blur threshold
+ logging.info("Calculating tile blur threshold...")
+ roi_threshold = calculate_roi_blur_threshold(
+ df_grid_roi,
+ intensity_threshold=_roi_intensity_threshold,
+ focus_percentile=_roi_focus_pct,
+ )
+ logging.info(f" Calculated threshold: {roi_threshold:.2f} (raw score units)")
+ logging.info(f" Intensity threshold: {_roi_intensity_threshold}")
+
+ # Fit 1D GMM model. On a very dim sample the GMM can fail to fit (e.g. no
+ # tissue tiles clear the intensity gate); fall back to the percentile
+ # threshold so the run still completes and produces a report, mirroring the
+ # try/except used by the 2D GMM below.
+ logging.info("Fitting 1D GMM model for tile focus scores...")
+ try:
+ gmm, blur_component_idx = fit_focus_gmm(
+ df_grid_roi,
+ intensity_threshold=_roi_intensity_threshold,
+ focus_col_name="dapi_focus_score",
+ )
+ logging.info("Classifying tiles (1D GMM)...")
+ df_grid_roi = classify_roi_blur(
+ df_grid_roi,
+ gmm=gmm,
+ blur_component_idx=blur_component_idx,
+ blur_prob_threshold=_blur_prob_thresh,
+ intensity_threshold=_roi_intensity_threshold,
+ focus_col_name="dapi_focus_score",
+ )
+ except Exception as e:
+ logging.warning(f" Warning: 1D GMM failed: {e}")
+ logging.info(" Using percentile-threshold fallback for 1D blur classification")
+ gmm = None
+ blur_component_idx = None
+ df_grid_roi = classify_roi_blur_by_threshold(
+ df_grid_roi,
+ roi_threshold=roi_threshold,
+ intensity_threshold=_roi_intensity_threshold,
+ focus_col_name="dapi_focus_score",
+ )
+ n_blurred = int(df_grid_roi["is_blurred_gmm"].sum())
+ n_in_focus = int((~df_grid_roi["is_blurred_gmm"]).sum())
+ logging.info(f" Blurred: {n_blurred:,}, In-focus: {n_in_focus:,} (1D GMM)")
+
+ # Fit 2D GMM if Laplacian variance available
+ gmm_2d = None
+ blur_component_idx_2d = None
+ if "dapi_lap_var" in df_grid_roi.columns:
+ logging.info("\nFitting 2D GMM model (focus_score + Laplacian variance)...")
+ try:
+ gmm_2d, blur_component_idx_2d = fit_focus_gmm_2d(
+ df_grid_roi,
+ intensity_threshold=_roi_intensity_threshold,
+ focus_col_name="dapi_focus_score",
+ )
+ logging.info("Classifying tiles (2D GMM)...")
+ df_grid_roi = classify_roi_blur_2d(
+ df_grid_roi,
+ gmm=gmm_2d,
+ blur_component_idx=blur_component_idx_2d,
+ blur_prob_threshold=_blur_prob_thresh,
+ intensity_threshold=_roi_intensity_threshold,
+ focus_col_name="dapi_focus_score",
+ )
+ n_blurred_2d = int(df_grid_roi["is_blurred_gmm_2d"].sum())
+ n_in_focus_2d = int((~df_grid_roi["is_blurred_gmm_2d"]).sum())
+ logging.info(
+ f" Blurred: {n_blurred_2d:,}, In-focus: {n_in_focus_2d:,} (2D GMM)"
+ )
+ except Exception as e:
+ logging.warning(f" Warning: 2D GMM failed: {e}")
+ logging.info(" Continuing with 1D GMM results only")
+ gmm_2d = None
+ blur_component_idx_2d = None
+ else:
+ logging.info("\nSkipping 2D GMM: dapi_lap_var column not found")
+
+ # Save grid ROI data to CSV
+ grid_roi_csv = data["outdir"] / "grid_roi_focus_scores.csv"
+ df_grid_roi.to_csv(grid_roi_csv, index=False)
+ logging.info(f"Saved grid tile data to {grid_roi_csv}")
+
+ # Optional: write full-res per-pixel maps (very large); not needed for QC/SNR in-process
+ if focus_maps is not None and save_dapi_maps_tiff:
+ logging.info("Saving pixel-level focus maps as TIFF (--save-dapi-maps-tiff)...")
+ t0 = time.time()
+ save_pixel_focus_maps(focus_maps, data["outdir"])
+ logging.info(f"[TIMING] save_pixel_focus_maps(): {time.time() - t0:.1f}s")
+ elif focus_maps is not None:
+ logging.info(
+ "Skipping pixel-map TIFF export (default). "
+ "Pass --save-dapi-maps-tiff to write dapi_*/boundary_*/intrna_* map TIFFs."
+ )
+ elif streamed is not None:
+ logging.info(
+ "No pixel-map TIFF export: tiles were streamed, so no full-resolution "
+ "planes exist. Pass --save-dapi-maps-tiff to build them."
+ )
+
+ # Free Laplacian map — no longer needed after TIFF export / GMM fitting.
+ # Remaining consumers (SNR, figures, CCFS) only use focus/mean maps.
+ if focus_maps is not None:
+ for _k in ("dapi_lap_var_map",):
+ focus_maps.pop(_k, None)
+
+ # Save threshold configuration
+ total_rois_gmm = int(len(df_grid_roi))
+ n_blurred_gmm = int(df_grid_roi["is_blurred_gmm"].sum())
+ pct_blurred_gmm = (
+ (n_blurred_gmm / total_rois_gmm * 100.0) if total_rois_gmm > 0 else 0.0
+ )
+
+ threshold_config = {
+ "roi_focus_score_threshold": float(roi_threshold),
+ "roi_intensity_threshold": float(_roi_intensity_threshold),
+ "roi_focus_score_percentile": float(_roi_focus_pct),
+ "threshold_method": "Option B: Percentile from tissue tiles (intensity >= threshold) + fixed intensity threshold",
+ "units": {
+ "roi_focus_score_threshold_percentile": "raw score units",
+ "roi_intensity_threshold": "raw pixel intensity (16-bit, 0-65535)",
+ "component_means_log1p_focus": "log1p(raw focus score)",
+ },
+ }
+
+ # The 1D GMM may have failed on a dim sample (gmm is None), in which case
+ # the percentile-threshold fallback was used. Guard the dereference the same
+ # way the gmm_2d stanza below is guarded, and emit a fallback marker instead.
+ if gmm is not None:
+ threshold_config["gmm_1d"] = {
+ "n_components": int(gmm.n_components),
+ "blur_component_index": int(blur_component_idx),
+ "component_means_log1p_focus": [float(m) for m in gmm.means_.flatten()],
+ "component_weights": [float(w) for w in gmm.weights_.flatten()],
+ "blur_prob_threshold": float(_blur_prob_thresh),
+ "fraction_rois_blurred_gmm": float(pct_blurred_gmm),
+ "features": ["log1p(dapi_focus_score)"],
+ }
+ else:
+ threshold_config["gmm_1d"] = {
+ "status": "fallback",
+ "method": "percentile_threshold",
+ "roi_focus_score_threshold": float(roi_threshold),
+ "blur_prob_threshold": float(_blur_prob_thresh),
+ "fraction_rois_blurred_gmm": float(pct_blurred_gmm),
+ "features": ["dapi_focus_score <= roi_focus_score_threshold"],
+ }
+
+ if "is_blurred_gmm_2d" in df_grid_roi.columns and gmm_2d is not None:
+ n_blurred_gmm_2d = int(df_grid_roi["is_blurred_gmm_2d"].sum())
+ pct_blurred_gmm_2d = (
+ (n_blurred_gmm_2d / total_rois_gmm * 100.0) if total_rois_gmm > 0 else 0.0
+ )
+ threshold_config["gmm_2d"] = {
+ "n_components": int(gmm_2d.n_components),
+ "blur_component_index": int(blur_component_idx_2d),
+ "component_means": [[float(m[0]), float(m[1])] for m in gmm_2d.means_],
+ "component_weights": [float(w) for w in gmm_2d.weights_.flatten()],
+ "blur_prob_threshold": float(_blur_prob_thresh),
+ "fraction_rois_blurred_gmm": float(pct_blurred_gmm_2d),
+ "features": ["log1p(dapi_focus_score)", "log1p(dapi_lap_var)"],
+ }
+ threshold_config["units"]["component_means_2d"] = (
+ "[log1p(focus_score), log1p(lap_var)]"
+ )
+
+ threshold_json = data["outdir"] / "roi_blur_threshold.json"
+ with open(threshold_json, "w") as f:
+ json.dump(threshold_config, f, indent=2)
+ logging.info(f"Saved tile blur threshold configuration to {threshold_json}")
+
+ # Save ROI count
+ roi_count_file = data["outdir"] / "roi_count.txt"
+ with open(roi_count_file, "w") as f:
+ f.write(str(len(df_grid_roi)))
+
+ # Calculate ROI intensities
+ logging.info("Calculating tile intensities...")
+ t0 = time.time()
+ df_roi_intensities = calculate_roi_intensities(xoa_morphology_files, df_grid_roi)
+ logging.info(f"[TIMING] Tile intensity calculation: {time.time() - t0:.1f}s")
+
+ snr_summary = None
+ if not no_snr:
+ logging.info("Computing SNR metrics (roi_qc / snr_metrics)...")
+ t_snr = time.time()
+ pix_um = snr_metrics.read_xenium_pixel_size_um(Path(xenium_bundle_dir))
+ if pix_um is None:
+ logging.info(
+ " experiment.xenium missing or invalid pixel_size — transcript SNR may skip"
+ )
+ try:
+ df_roi_intensities, snr_summary = snr_metrics.compute_snr_summary(
+ df_roi_intensities,
+ bundle_dir=Path(xenium_bundle_dir),
+ outdir=data["outdir"],
+ focus_maps=focus_maps,
+ # Positional order, not a join: roi_snr_db is indexed by the
+ # compute_roi_grid() order that built df_grid_roi, and
+ # calculate_roi_intensities() returns df_grid_roi.copy() without
+ # filtering or reordering, so row i is the same ROI in both. If that
+ # function ever starts filtering, this silently mislabels every tile
+ # -- the grid/DataFrame half of the invariant is pinned by
+ # test_agrees_with_the_dataframe_coordinates.
+ roi_snr_db=streamed.roi_snr_db if streamed else None,
+ pixel_size_um=pix_um,
+ otsu_max_rois=snr_otsu_max_rois,
+ save_roi_tx_table=not snr_no_roi_tx_table,
+ snr_include_moran=snr_with_moran,
+ # Same stride as grid in calculate_roi_focusscore (stride=None → roi_size).
+ roi_grid_stride=(roi_size, roi_size),
+ snr_thresholds=qc_thresholds.get("snr") or {},
+ )
+ except Exception as e:
+ logging.warning("SNR metrics failed (continuing image QC): %s", e)
+ snr_summary = {
+ "status": "error",
+ "error": str(e),
+ "components": {},
+ "verdict": {"overall_snr_verdict": "NOT_COMPUTED"},
+ }
+ logging.info(f"[TIMING] SNR metrics: {time.time() - t_snr:.1f}s")
+ _log_mem("SNR metrics")
+ else:
+ logging.info("SNR metrics disabled (--no-snr)")
+ df_grid_roi = df_roi_intensities
+
+ # Assess raw intensity quality (thresholds from YAML or defaults)
+ _ch_cfg = qc_thresholds.get("channels") or {}
+
+ # XOA-version-specific intensity floors (2026-06-22, qc_drift_analysis).
+ # XOA 4.0 images are ~14x dimmer than 3.x, so a single floor can't serve both.
+ # Pick `intensity_critical_v{major}` when present, else the legacy
+ # `intensity_critical`, else the module default. Version unknown → legacy/default.
+ _xoa_major = read_xenium_major_version(Path(xenium_bundle_dir))
+ logging.info(f" XOA major version for intensity floors: {_xoa_major}")
+
+ def _pick_critical(ch_key, default):
+ ch = _ch_cfg.get(ch_key) or {}
+ if _xoa_major is not None:
+ v = ch.get(f"intensity_critical_v{_xoa_major}")
+ if isinstance(v, (int, float)):
+ return v
+ return ch.get("intensity_critical", default)
+
+ _ic_dapi = _pick_critical("DAPI", _INTENSITY_CRITICAL_DEFAULTS["dapi"])
+ _ic_boundary = _pick_critical("boundary", _INTENSITY_CRITICAL_DEFAULTS["boundary"])
+ _ic_intrna = _pick_critical("intRNA", _INTENSITY_CRITICAL_DEFAULTS["intrna"])
+ logging.info("Assessing raw intensity quality...")
+ intensity_stats = assess_raw_intensity_quality(
+ df_roi_intensities,
+ dapi_threshold_critical=_ic_dapi,
+ boundary_threshold_critical=_ic_boundary,
+ intrna_threshold_critical=_ic_intrna,
+ min_tissue_coverage=_min_tissue_cov,
+ channel_pct_thresholds=_ch_cfg,
+ )
+
+ # Save intensity statistics
+ intensity_json = data["outdir"] / "intensity_assessment.json"
+ with open(intensity_json, "w") as f:
+ json.dump(intensity_stats, f, indent=2)
+
+ # Generate ROI figures (cell-independent)
+ logging.info("Generating tile-level figures...")
+ t0 = time.time()
+ if figures:
+ generate_roi_figures(
+ data,
+ small0,
+ small1,
+ small2,
+ distance_map,
+ distance_map2,
+ whole_sample,
+ holes,
+ dense_intensity_regions,
+ df_grid_roi,
+ df_roi_intensities,
+ intensity_stats,
+ xoa_morphology_files,
+ focus_maps=focus_maps,
+ focus_heatmap=streamed.focus_heatmap if streamed else None,
+ snr_thresholds=(qc_thresholds.get("snr") or {}).get("roi_tx") or {},
+ multistain_whole_sample=ms_whole_sample,
+ multistain_distance_map=ms_distance_map,
+ multistain_distance_map2=ms_distance_map2,
+ figure_source_tables=figure_source_tables,
+ )
+ logging.info(f"[TIMING] generate_roi_figures(): {time.time() - t0:.1f}s")
+ _log_mem("tile figures")
+
+ # Free focus-score maps no longer needed. CCFS only requires
+ # dapi_focus_map, dapi_mean_map, boundary_mean_map, intrna_mean_map.
+ if focus_maps is not None:
+ for _k in ("boundary_focus_map", "intrna_focus_map"):
+ focus_maps.pop(_k, None)
+
+ # Save ROI QC metrics
+ logging.info("Saving tile QC metrics...")
+ save_roi_qc_metrics(
+ df_grid_roi,
+ intensity_stats,
+ data["outdir"],
+ roi_size=roi_size,
+ snr_summary=snr_summary,
+ distance_map=distance_map,
+ distance_map2=distance_map2,
+ multistain_whole_sample=ms_whole_sample,
+ multistain_distance_map=ms_distance_map,
+ multistain_distance_map2=ms_distance_map2,
+ edge_distance_threshold=-25.0,
+ hole_distance_threshold=-25.0,
+ min_tissue_coverage_for_qc=_min_tissue_cov,
+ qc_thresholds=qc_thresholds,
+ lap_sigma=_lap_sigma,
+ segmentation_software=resolve_segmentation_software(
+ xenium_bundle_dir, pipeline_segmentation, is_resegmented
+ ),
+ xoa_version=read_xenium_analysis_sw_version(xenium_bundle_dir),
+ )
+
+ # ===== PHASE 4: CELL-LEVEL ANALYSIS (conditional on cell data) =====
+ if has_cell_data:
+ logging.info("\n--- PHASE 4: CELL-LEVEL ANALYSIS ---")
+
+ # Prepare cell-centred output directory
+ figures_dir_cell = data["outdir"] / "figures"
+ figures_dir_cell.mkdir(parents=True, exist_ok=True)
+ if figure_source_tables:
+ figures_source_dir_cell = figures_dir_cell / "figures_source"
+ figures_source_dir_cell.mkdir(parents=True, exist_ok=True)
+
+ # Prepare data dict for cell-level functions
+ cell_data = dict(data)
+ cell_data["figures_dir"] = figures_dir_cell
+
+ # Unpack analysis.tar.gz if needed (test data)
+ xenium_bundle_dir_path = Path(xenium_bundle_dir)
+ if (xenium_bundle_dir_path / "analysis.tar.gz").is_file():
+ import shutil
+
+ shutil.unpack_archive(
+ xenium_bundle_dir_path / "analysis.tar.gz", extract_dir="."
+ )
+ cell_data["clusters_csv_path"] = (
+ Path("analysis")
+ / "clustering"
+ / "gene_expression_kmeans_10_clusters"
+ / "clusters.csv"
+ )
+ cell_data["umap_path"] = (
+ Path("analysis")
+ / "umap"
+ / "gene_expression_2_components"
+ / "projection.csv"
+ )
+ else:
+ cell_data["clusters_csv_path"] = (
+ xenium_bundle_dir_path
+ / "analysis"
+ / "clustering"
+ / "gene_expression_kmeans_10_clusters"
+ / "clusters.csv"
+ )
+ cell_data["umap_path"] = (
+ xenium_bundle_dir_path
+ / "analysis"
+ / "umap"
+ / "gene_expression_2_components"
+ / "projection.csv"
+ )
+ cell_data["cells_parquet_path"] = xenium_bundle_dir_path / "cells.parquet"
+ cell_data["cell_masks_path"] = xenium_bundle_dir_path / "cells.zarr.zip"
+
+ # Load spatial data
+ logging.info("Loading spatial data and masks...")
+ t0 = time.time()
+ try:
+ df_spatial, cellseg_mask, cell_masks_zarr = load_spatial_data(cell_data)
+ logging.info(f"Loaded {len(df_spatial):,} cells")
+ logging.info(f"[TIMING] Loading spatial data: {time.time() - t0:.1f}s")
+ except (FileNotFoundError, KeyError, AttributeError) as e:
+ logging.warning(f"Warning: Could not load spatial data: {e}")
+ logging.info("Skipping cell-level analysis")
+ has_cell_data = False
+
+ if has_cell_data:
+ # Map distances to cells
+ logging.info("Mapping distances to cells...")
+ t0 = time.time()
+ df_spatial = map_distances_to_cells(
+ df_spatial, distance_map, distance_map2, dense_intensity_regions
+ )
+ logging.info(f"[TIMING] map_distances_to_cells(): {time.time() - t0:.1f}s")
+
+ # Calculate CCFS measurements
+ logging.info("Calculating CCFS measurements...")
+ t0 = time.time()
+ if streamed is not None and streamed.has_per_cell:
+ logging.info(" Using per-cell sums reduced during the tile pass")
+ myData = calculate_ccfs_from_focus_maps(
+ None,
+ cell_masks_zarr,
+ cellseg_mask,
+ xoa_morphology_files,
+ streamed=streamed,
+ )
+ elif (
+ focus_maps is not None
+ and "dapi_focus_map" in focus_maps
+ and "dapi_mean_map" in focus_maps
+ ):
+ # NEW: Use pixel-map aggregation
+ logging.info(" Using pixel-map aggregation (scipy.ndimage.mean)")
+ myData = calculate_ccfs_from_focus_maps(
+ focus_maps, cell_masks_zarr, cellseg_mask, xoa_morphology_files
+ )
+ else:
+ # Fallback: use regionprops-based method
+ logging.info(" Using regionprops-based method (legacy)")
+ myData = calculate_ccfs_measurements(
+ xoa_morphology_files, cellseg_mask, cell_masks_zarr
+ )
+ logging.info(f"Calculated CCFS for {len(myData):,} cells")
+ logging.info(f"[TIMING] CCFS calculation: {time.time() - t0:.1f}s")
+ _log_mem("per-cell CCFS")
+
+ # All pixel-level focus maps consumed — free remaining memory. The streamed
+ # reductions are kept: nothing downstream re-reads them, but they are tens
+ # of MB, and dropping the name here would break the `streamed` references
+ # in the figure calls that follow.
+ del focus_maps
+ focus_maps = None
+
+ # Add boolean columns to myData for thresholding (needed by generate_all_figures)
+ ccfs_threshold = _ccfs_low_texture_threshold
+ myData["is_low_nuclear_texture"] = myData["CCFS_DAPI"] <= ccfs_threshold
+ myData["is_high_nuclear_texture"] = myData["CCFS_DAPI"] > ccfs_threshold
+
+ # Map ROI focus scores to cells
+ logging.info("Mapping tile focus scores to cells...")
+ t0 = time.time()
+ roi_mapped = map_grid_roi_to_cells(df_grid_roi, df_spatial, overlapping=False)
+ logging.info(
+ f"Mapped tile focus scores to {len(roi_mapped[roi_mapped['DAPI_RFSnorm_roi'].notna()]):,} cells"
+ )
+ logging.info(f"[TIMING] map_grid_roi_to_cells(): {time.time() - t0:.1f}s")
+
+ # Load ROI blur threshold
+ roi_threshold_cell, roi_intensity_threshold = load_roi_blur_threshold(
+ data["outdir"]
+ )
+ if roi_threshold_cell is None or roi_intensity_threshold is None:
+ roi_threshold_cell = roi_threshold
+ roi_intensity_threshold = _roi_intensity_threshold
+
+ # Create merged dataset (superset version with ROI data)
+ logging.info("Creating final merged dataset...")
+ t0 = time.time()
+ new_df, calculated_roi_threshold = create_final_merged_data(
+ df_spatial,
+ myData,
+ roi_data=roi_mapped,
+ ccfs_threshold=_ccfs_low_texture_threshold,
+ roi_threshold=roi_threshold_cell,
+ roi_intensity_threshold=roi_intensity_threshold,
+ )
+ logging.info(f"Created merged dataset with {len(new_df):,} cells")
+ logging.info(f"[TIMING] create_final_merged_data(): {time.time() - t0:.1f}s")
+
+ # Alias dense_intensity_regions as artefacts for generate_all_figures compatibility
+ artefacts = dense_intensity_regions
+
+ # Ensure In-Area-with-Artefact column exists (needed by generate_all_figures)
+ if "In-Area-with-Artefact" not in df_spatial.columns:
+ # Map dense_intensity_regions to the artefact column name
+ if "Dense-Intensity-Region-ID" in df_spatial.columns:
+ df_spatial["In-Area-with-Artefact"] = df_spatial[
+ "Dense-Intensity-Region-ID"
+ ]
+ else:
+ df_spatial["In-Area-with-Artefact"] = 0
+
+ if "In-Area-with-Artefact" not in new_df.columns:
+ if "Dense-Intensity-Region-ID" in new_df.columns:
+ new_df["In-Area-with-Artefact"] = new_df["Dense-Intensity-Region-ID"]
+ else:
+ new_df["In-Area-with-Artefact"] = 0
+
+ # Ensure has_artifacts column exists
+ if "has_artifacts" not in new_df.columns:
+ new_df["has_artifacts"] = new_df.get(
+ "has_dense_intensity_regions",
+ new_df.get("In-Area-with-Artefact", 0) > 0,
+ )
+
+ # Generate 13 Quarto-required figures
+ logging.info("Generating Quarto-required figures (13 PNGs)...")
+ t0 = time.time()
+ if figures:
+ generate_all_figures(
+ cell_data,
+ df_spatial,
+ new_df,
+ myData,
+ small0,
+ small1,
+ small2,
+ distance_map,
+ distance_map2,
+ whole_sample,
+ holes,
+ artefacts,
+ ccfs_low_texture_threshold=_ccfs_low_texture_threshold,
+ multistain_whole_sample=ms_whole_sample,
+ multistain_distance_map=ms_distance_map,
+ multistain_distance_map2=ms_distance_map2,
+ figure_source_tables=figure_source_tables,
+ )
+ logging.info(f"[TIMING] generate_all_figures(): {time.time() - t0:.1f}s")
+
+ # Generate cell-centred comparison figures (ROI vs CCFS)
+ if figures:
+ figures_cell_centred_dir = data["outdir"] / "figures_cell_centred"
+ figures_cell_centred_dir.mkdir(parents=True, exist_ok=True)
+ figures_cell_centred_source = (
+ (figures_cell_centred_dir / "figures_source")
+ if figure_source_tables
+ else None
+ )
+ if figures_cell_centred_source is not None:
+ figures_cell_centred_source.mkdir(parents=True, exist_ok=True)
+ t0 = time.time()
+ generate_cell_figures(
+ cell_data,
+ new_df,
+ myData,
+ figures_cell_centred_dir,
+ figures_cell_centred_source,
+ roi_threshold=calculated_roi_threshold,
+ roi_intensity_threshold=roi_intensity_threshold,
+ ccfs_low_texture_threshold=_ccfs_low_texture_threshold,
+ )
+ logging.info(f"[TIMING] generate_cell_figures(): {time.time() - t0:.1f}s")
+ _log_mem("cell figures")
+
+ # Save cell QC metrics (superset version with ROI metrics)
+ logging.info("Saving cell QC metrics...")
+ t0 = time.time()
+ save_cell_qc_metrics(new_df, data["outdir"], roi_size=roi_size)
+ logging.info(f"[TIMING] save_cell_qc_metrics(): {time.time() - t0:.1f}s")
+
+ # Save simple image_qc_metrics.json for Quarto compatibility
+ logging.info("Saving Quarto-compatible image_qc_metrics.json...")
+ n_total = len(new_df)
+ n_ccfs_low_texture = int(new_df["is_low_nuclear_texture"].sum())
+ qc_metrics = {
+ "total_cells": n_total,
+ "cells_with_transcripts": len(new_df[new_df["transcript_counts"] > 0]),
+ "mean_transcript_count": float(new_df["transcript_counts"].mean()),
+ "median_transcript_count": float(new_df["transcript_counts"].median()),
+ "mean_ccfs_dapi": float(new_df["CCFS_DAPI"].mean()),
+ "median_ccfs_dapi": float(new_df["CCFS_DAPI"].median()),
+ "ccfs_low_texture_threshold": float(_ccfs_low_texture_threshold),
+ "cells_high_nuclear_texture": len(
+ new_df[new_df["is_high_nuclear_texture"]]
+ ),
+ "cells_low_nuclear_texture": n_ccfs_low_texture,
+ "pct_low_nuclear_texture": round(100.0 * n_ccfs_low_texture / n_total, 4)
+ if n_total > 0
+ else 0.0,
+ "cells_near_edge": len(new_df[new_df["is_near_edge"]]),
+ "cells_near_holes": len(new_df[new_df["is_near_hole"]]),
+ "cells_with_artifacts": int(new_df["has_artifacts"].sum()),
+ "clusters_present": sorted(new_df["Cluster_kmeans10"].unique().tolist()),
+ "segmentation_methods": sorted(
+ new_df["segmentation_method"].unique().tolist()
+ ),
+ }
+ # Phase 2a-revised: per-cell aggregate emissions for Section 9.A's
+ # sample-level aggregates table + 9.B's roi_tissue_coverage row.
+ # Naming asymmetry note: per-cell DAPI intensity column is `mean_intensity`
+ # (no suffix); aggregate JSON key adds `_DAPI` for sibling-channel
+ # consistency with mean_intensity_Boundary / mean_intensity_IntRNA. See
+ # v3 plan §9 assumption #9.
+ for _src_col, _agg_key_base, _decimals in (
+ ("mean_intensity", "intensity_DAPI", 4),
+ ("mean_intensity_Boundary", "intensity_Boundary", 4),
+ ("mean_intensity_IntRNA", "intensity_IntRNA", 4),
+ ("area_nucleus", "area_nucleus", 2),
+ ("area_cell", "area_cell", 2),
+ ):
+ if _src_col not in new_df.columns:
+ continue
+ _series = new_df[_src_col]
+ _mean = _series.mean()
+ _median = _series.median()
+ if pd.notna(_mean):
+ qc_metrics[f"mean_{_agg_key_base}"] = round(float(_mean), _decimals)
+ if pd.notna(_median):
+ qc_metrics[f"median_{_agg_key_base}"] = round(float(_median), _decimals)
+
+ # Phase 13 (v4): % cells below per-channel intensity floor for the
+ # 9.A Tier 1 verdict rows. 2026-06-23: use the SAME XOA-version-specific
+ # floors as the tile-level intensity QC (via _pick_critical), so a dim
+ # XOA-4.0 sample isn't flagged against the bright-era 500/100/300 floors.
+ # NaN-intensity cells are excluded from the numerator (NaN < cutoff →
+ # False) but stay in the n_total denominator — same convention as
+ # `pct_cells_in_low_coverage_tiles`.
+ if n_total > 0:
+ _cell_intensity_emissions = (
+ (
+ "mean_intensity",
+ _pick_critical("DAPI", _INTENSITY_CRITICAL_DEFAULTS["dapi"]),
+ "pct_cells_below_intensity_DAPI",
+ ),
+ (
+ "mean_intensity_Boundary",
+ _pick_critical(
+ "boundary", _INTENSITY_CRITICAL_DEFAULTS["boundary"]
+ ),
+ "pct_cells_below_intensity_Boundary",
+ ),
+ (
+ "mean_intensity_IntRNA",
+ _pick_critical("intRNA", _INTENSITY_CRITICAL_DEFAULTS["intrna"]),
+ "pct_cells_below_intensity_IntRNA",
+ ),
+ )
+ for _src_col, _cutoff, _emit_key in _cell_intensity_emissions:
+ if _src_col not in new_df.columns:
+ continue
+ _n_below = int((new_df[_src_col] < _cutoff).sum())
+ qc_metrics[_emit_key] = round(100.0 * _n_below / n_total, 4)
+
+ # 9.B (Phase 2a-revised, folds Phase 11): % cells in low-coverage tiles
+ # (roi_tissue_coverage < 0.5). Informational; no PASS/WARN/FAIL pill until
+ # calibration. NaN coverage cells are excluded (NaN < 0.5 → False), so the
+ # count reflects only cells with assigned ROI tile coverage data.
+ # 2026-06-26 (multi-stain): roi_tissue_coverage here is DAPI-based (it is also the
+ # denominator of pct_blurred_gmm_2d_roi below, which MUST stay DAPI to match the
+ # tile-level DAPI blur figure). A multi-stain cell-coverage view for this
+ # informational count is a deferred follow-up; tile-level extent already uses all
+ # stains. Report prose marks this as DAPI-derived.
+ if "roi_tissue_coverage" in new_df.columns and n_total > 0:
+ _n_low_cov = int((new_df["roi_tissue_coverage"] < 0.5).sum())
+ qc_metrics["cells_in_low_coverage_tiles"] = _n_low_cov
+ qc_metrics["pct_cells_in_low_coverage_tiles"] = round(
+ 100.0 * _n_low_cov / n_total, 4
+ )
+
+ # GMM-ROI blur metrics (if cell-to-ROI mapping was performed).
+ # 2026-06-24: pct_blurred_gmm_2d_roi is reported over SOLID-tissue cells
+ # (roi_tissue_coverage >= 0.5) so it matches the tile-level "tiles in
+ # focus" metric (also tissue-filtered). Cells in low-coverage / edge tiles
+ # are force-labelled blurred regardless of optical focus and otherwise
+ # inflate this far above the tile figure (e.g. 50% cells vs 9% tiles).
+ # The all-cells value is kept for context / the report's "why it differs"
+ # note, alongside the existing pct_cells_in_low_coverage_tiles.
+ if "is_blurred_gmm_2d_roi" in new_df.columns:
+ _blur = new_df["is_blurred_gmm_2d_roi"]
+ _n_blur_all = int(_blur.sum())
+ qc_metrics["pct_blurred_gmm_2d_roi_all_cells"] = (
+ round(100.0 * _n_blur_all / n_total, 4) if n_total > 0 else 0.0
+ )
+ if "roi_tissue_coverage" in new_df.columns:
+ _solid = (
+ new_df["roi_tissue_coverage"]
+ >= ROI_MIN_TISSUE_COVERAGE_FOR_INTENSITY_QC
+ )
+ else:
+ _solid = _blur.notna()
+ _n_solid = int(_solid.sum())
+ n_gmm = int(_blur[_solid].sum())
+ qc_metrics["cells_blurred_gmm_2d_roi"] = n_gmm
+ qc_metrics["cells_evaluated_for_blur"] = _n_solid
+ qc_metrics["pct_blurred_gmm_2d_roi"] = (
+ round(100.0 * n_gmm / _n_solid, 4) if _n_solid > 0 else 0.0
+ )
+ qc_metrics["pct_blurred_gmm_2d_roi_denominator"] = (
+ "solid_tissue_cells_coverage_ge_0.5"
+ if "roi_tissue_coverage" in new_df.columns
+ else "all_cells"
+ )
+ agreement = int(
+ (
+ new_df["is_low_nuclear_texture"] == new_df["is_blurred_gmm_2d_roi"]
+ ).sum()
+ )
+ qc_metrics["ccfs_gmm_agreement_pct"] = (
+ round(100.0 * agreement / n_total, 4) if n_total > 0 else 0.0
+ )
+ # Per-cluster blur stats for cluster-outlier detection
+ if "Cluster_kmeans10" in new_df.columns:
+ blur_col = (
+ "is_blurred_gmm_2d_roi"
+ if "is_blurred_gmm_2d_roi" in new_df.columns
+ else "is_low_nuclear_texture"
+ )
+ cluster_blur = {}
+ for cl, grp in new_df.groupby("Cluster_kmeans10"):
+ n_cl = len(grp)
+ n_blur_cl = int(grp[blur_col].sum())
+ pct_blur_cl = round(100.0 * n_blur_cl / n_cl, 2) if n_cl > 0 else 0.0
+ cluster_blur[int(cl)] = {
+ "n_cells": n_cl,
+ "n_blurred": n_blur_cl,
+ "pct_blurred": pct_blur_cl,
+ "median_ccfs_dapi": round(float(grp["CCFS_DAPI"].median()), 4)
+ if not pd.isna(grp["CCFS_DAPI"].median())
+ else 0.0,
+ }
+ qc_metrics["cluster_blur"] = cluster_blur
+ qc_metrics["cluster_blur_method"] = blur_col
+ # Flag clusters that stand apart from the rest (robust MAD z-score +
+ # absolute floor). See detect_cluster_outliers.
+ outlier_clusters = detect_cluster_outliers(cluster_blur, "pct_blurred")
+ if outlier_clusters:
+ qc_metrics["cluster_blur_outliers"] = outlier_clusters
+
+ # Per-cluster CCFS-low-texture stats for cluster-outlier detection.
+ # Parallel to cluster_blur above, but always derived from
+ # is_low_nuclear_texture (independent of whether tile-mapped GMM blur is
+ # available — distinct from cluster_blur, which falls back to
+ # is_low_nuclear_texture only when is_blurred_gmm_2d_roi is absent).
+ # Consumed by the §4.3 Tier-3 row and the §4.6 per-cluster breakdown in
+ # notebooks/xenium_image_qc_report.qmd. Hidden by the qmd until this
+ # field is present in the JSON.
+ if (
+ "Cluster_kmeans10" in new_df.columns
+ and "is_low_nuclear_texture" in new_df.columns
+ ):
+ cluster_ccfs = {}
+ for cl, grp in new_df.groupby("Cluster_kmeans10"):
+ n_cl = len(grp)
+ n_low_cl = int(grp["is_low_nuclear_texture"].sum())
+ pct_low_cl = round(100.0 * n_low_cl / n_cl, 2) if n_cl > 0 else 0.0
+ cluster_ccfs[int(cl)] = {
+ "n_cells": n_cl,
+ "n_low_texture": n_low_cl,
+ "pct_low_texture": pct_low_cl,
+ "median_ccfs_dapi": round(float(grp["CCFS_DAPI"].median()), 4)
+ if not pd.isna(grp["CCFS_DAPI"].median())
+ else 0.0,
+ }
+ qc_metrics["cluster_ccfs"] = cluster_ccfs
+ # Same robust rule as cluster_blur. The absolute floor backstops the
+ # MAD=0 case — most clusters sit near 0% low-texture, so a real spike
+ # is caught by the floor even when the spread is degenerate.
+ ccfs_outlier_clusters = detect_cluster_outliers(
+ cluster_ccfs, "pct_low_texture"
+ )
+ if ccfs_outlier_clusters:
+ qc_metrics["cluster_ccfs_outliers"] = ccfs_outlier_clusters
+
+ with open(data["outdir"] / "image_qc_metrics.json", "w") as f:
+ json.dump(qc_metrics, f, indent=2)
+
+ # Save image_qc_metrics.csv
+ cell_id_col = new_df.pop("cell_id")
+ new_df.insert(0, "cell_id", cell_id_col)
+ new_df.to_csv(data["outdir"] / "image_qc_metrics.csv", index=False)
+
+ # Save dense intensity region summary
+ if "Dense-Intensity-Region-ID" in new_df.columns:
+ region_summary = (
+ new_df[new_df["Dense-Intensity-Region-ID"] > 0]
+ .groupby("Dense-Intensity-Region-ID")
+ .agg({"cell_id": "count", "x": "mean", "y": "mean"})
+ .reset_index()
+ )
+ region_summary.columns = [
+ "Dense-Intensity-Region-ID",
+ "cell_count",
+ "mean_x",
+ "mean_y",
+ ]
+ region_summary = region_summary.sort_values("Dense-Intensity-Region-ID")
+ region_summary["annotation"] = ""
+ region_summary.to_csv(
+ data["outdir"] / "dense_intensity_regions_summary.csv", index=False
+ )
+ else:
+ logging.info("\n--- PHASE 4: SKIPPED (no cell data) ---")
+
+ # ===== PHASE 5: FINALIZE =====
+ logging.info("\n--- PHASE 5: FINALIZE ---")
+
+ # Save versions file
+ logging.info("Saving versions file...")
+ save_versions_file(data["outdir"])
+
+ logging.info(f"\n[TIMING] Total pipeline time: {time.time() - t_total_start:.1f}s")
+ _log_mem_summary()
+ logging.info("=" * 60)
+ logging.info("Combined Image QC Pipeline completed successfully!")
+ logging.info(f" Tile figures: {data['figures_dir']}")
+ if has_cell_data:
+ logging.info(f" Cell figures: {data['outdir'] / 'figures'}")
+ logging.info(f" Cell metrics: {data['outdir'] / 'image_qc_metrics.json'}")
+ logging.info(f" Cell CSV: {data['outdir'] / 'image_qc_metrics.csv'}")
+ logging.info(f" Tile metrics: {data['outdir'] / 'roi_qc_metrics.json'}")
+ logging.info(f" Versions: {data['outdir'] / 'versions.yml'}")
+ logging.info("=" * 60)
+
+
+if __name__ == "__main__":
+ main()
diff --git a/bin/snr_metrics.py b/bin/snr_metrics.py
new file mode 100755
index 00000000..c9dd4c17
--- /dev/null
+++ b/bin/snr_metrics.py
@@ -0,0 +1,1729 @@
+#!/usr/bin/env python3
+"""
+SNR metrics for Xenium image QC (ROI image, transcripts, slide matrix, neg spatial).
+
+Used by ``bin/image_qc.py``. See ``plans/image_qc_report/SNR_plan.md``.
+
+Dependencies: numpy, pandas; optional h5py, pyarrow, scipy, libpysal/esda, scikit-image.
+"""
+
+from __future__ import annotations
+
+import json
+import logging
+import math
+import os
+import re
+import threading
+from collections import deque
+from concurrent.futures import ThreadPoolExecutor
+from pathlib import Path
+from typing import Any, Dict, List, Optional, Tuple
+
+import numpy as np
+import pandas as pd
+import pyarrow.parquet as pq
+
+logger = logging.getLogger(__name__)
+
+# JSON ``summary["components"]`` keys — ``SNR_`` prefix keeps this module separate from
+# generic ``roi_qc_metrics`` / image_qc fields when integrated into ``bin/image_qc.py``.
+SNR_CKEY_IMAGE_ROI_QUARTILE_DB = "SNR_image_roi_quartile_db"
+SNR_CKEY_IMAGE_OTSU = "SNR_image_otsu"
+SNR_CKEY_ROI_TX = "SNR_roi_tx"
+SNR_CKEY_SLIDE_PLUMMER = "SNR_slide_plummer"
+SNR_CKEY_SLIDE_SPATIALQM = "SNR_slide_spatialqm"
+SNR_CKEY_ROI_NEG_SPATIAL = "SNR_roi_neg_spatial"
+
+# Per-ROI transcript SNR table (same rows as input grid + count/ratio columns) for histograms / maps.
+SNR_ROI_TX_TABLE_BASENAME = "SNR_roi_tx"
+
+# ---------------------------------------------------------------------------
+# SNR_plan.md: snr_verdict_to_quality_status
+# ---------------------------------------------------------------------------
+
+
+def snr_verdict_to_quality_status(snr_summary: dict) -> str:
+ """
+ Convert SNR module verdict to the 'pass'/'warn'/'fail' scale used by
+ assess_raw_intensity_quality() so the HTML report scorecard can treat
+ SNR like any other channel quality metric.
+
+ PASS → 'pass'
+ WARN → 'warn'
+ FAIL → 'fail'
+ NOT_COMPUTED / ERROR → 'not_available'
+ """
+ mapping = {
+ "PASS": "pass",
+ "WARN": "warn",
+ "FAIL": "fail",
+ }
+ overall = snr_summary.get("verdict", {}).get("overall_snr_verdict", "NOT_COMPUTED")
+ return mapping.get(overall, "not_available")
+
+
+# ---------------------------------------------------------------------------
+# Neg-control classification (Xenium-style)
+# ---------------------------------------------------------------------------
+
+_NEG_PATTERNS = (
+ re.compile(r"^NegControlProbe_", re.I),
+ re.compile(r"^NegControlCodeword_", re.I),
+ re.compile(r"^Blank[-_]", re.I),
+ re.compile(r"^BLANK[-_]", re.I),
+ re.compile(r"^antisense_", re.I),
+)
+
+
+def is_neg_probe_feature(name: str) -> bool:
+ if not isinstance(name, str) or not name:
+ return False
+ return any(p.search(name) for p in _NEG_PATTERNS)
+
+
+# ---------------------------------------------------------------------------
+# Image SNR — lightweight path (ROI DataFrame columns)
+# SNR_plan: top ~75th pct tissue ROIs = signal; bottom quartile = background;
+# SNR_dB = 20 * log10(mean_fg / std_bg)
+# ---------------------------------------------------------------------------
+
+
+def compute_image_snr_from_roi_df(
+ df_grid_roi: pd.DataFrame,
+ intensity_threshold: float = 0.0,
+ intensity_col: Optional[str] = None,
+ snr_thresholds: Optional[Dict[str, Any]] = None,
+) -> Dict[str, Any]:
+ """
+ Slide-level image SNR (dB) from pre-aggregated ROI intensities; same value
+ conceptually applies to all tissue ROIs (SNR_plan lightweight path).
+ """
+ if intensity_col is None:
+ intensity_col = (
+ "dapi_intensity"
+ if "dapi_intensity" in df_grid_roi.columns
+ else "raw_intensity"
+ )
+ if intensity_col not in df_grid_roi.columns:
+ return {"status": "skipped", "reason": f"missing column {intensity_col!r}"}
+
+ df = df_grid_roi.copy()
+ if "overlaps_tissue" in df.columns:
+ tissue = df[df["overlaps_tissue"]].copy()
+ elif "tissue_coverage" in df.columns:
+ tissue = df[df["tissue_coverage"] > 0].copy()
+ else:
+ tissue = df
+
+ tissue = tissue[tissue[intensity_col] >= intensity_threshold]
+ if len(tissue) < 8:
+ return {
+ "status": "skipped",
+ "reason": "too_few_tissue_rois",
+ "n": int(len(tissue)),
+ }
+
+ vals = tissue[intensity_col].astype(np.float64).values
+ q25, q75 = np.percentile(vals, [25, 75])
+ bg = vals[vals <= q25]
+ fg = vals[vals >= q75]
+ if len(bg) < 2 or len(fg) < 2:
+ return {"status": "skipped", "reason": "empty_quartile_split"}
+
+ mean_fg = float(np.mean(fg))
+ std_bg = float(np.std(bg, ddof=1)) if len(bg) > 1 else float(np.std(bg))
+ eps = 1e-12
+ if std_bg < eps:
+ return {"status": "skipped", "reason": "zero_background_std"}
+
+ ratio = mean_fg / std_bg
+ snr_db = 20.0 * math.log10(max(ratio, eps))
+ _t = snr_thresholds or {}
+ _img = _t.get("image_snr_db") or {}
+ warn_db = float(_img.get("warn", 15.0))
+ fail_db = float(_img.get("fail", 10.0))
+ verdict = "PASS" if snr_db >= warn_db else ("WARN" if snr_db >= fail_db else "FAIL")
+
+ return {
+ "status": "ok",
+ "method": "roi_df_quartiles",
+ "intensity_col": intensity_col,
+ "snr_db": snr_db,
+ "mean_foreground": mean_fg,
+ "std_background": std_bg,
+ "n_tissue_rois": int(len(tissue)),
+ "verdict": verdict,
+ }
+
+
+# ---------------------------------------------------------------------------
+# Otsu (numpy-only fallback; optional skimage)
+# ---------------------------------------------------------------------------
+
+
+def _otsu_threshold_uint16(arr: np.ndarray) -> float:
+ """Histogram Otsu for non-negative array (2D)."""
+ a = np.asarray(arr, dtype=np.float64).ravel()
+ a = a[np.isfinite(a)]
+ if a.size < 4:
+ return float(np.median(a)) if a.size else 0.0
+ # uint16-style bins for Xenium-ish data; cap bins for speed
+ a_min, a_max = float(np.min(a)), float(np.max(a))
+ if a_max <= a_min:
+ return a_min
+ nb = min(256, int(a_max - a_min) + 1)
+ hist, bin_edges = np.histogram(a, bins=nb, range=(a_min, a_max))
+ bin_centers = (bin_edges[:-1] + bin_edges[1:]) / 2.0
+ w = hist.astype(np.float64)
+ total = w.sum()
+ if total <= 0:
+ return float(np.median(a))
+ w = w / total
+ mu = (w * bin_centers).sum()
+ omega = np.cumsum(w)
+ mu_t = np.cumsum(w * bin_centers)
+ sigma_b2 = (mu_t - mu * omega) ** 2 / (omega * (1 - omega) + 1e-12)
+ idx = int(np.nanargmax(sigma_b2))
+ return float(bin_centers[idx])
+
+
+def roi_snr_db(tile: np.ndarray) -> float:
+ """SNR in dB for one ROI window: Otsu foreground mean over background stdev.
+
+ Returns NaN when the window cannot yield a value — too few pixels, an empty
+ foreground or background after thresholding, or a degenerate background with
+ zero spread. NaN rather than an exception because a slide legitimately
+ contains such windows (empty edge tissue) and they are reported as N/A.
+
+ Factored out so the whole-map path and the streaming tile consumer
+ (``image_qc.RoiOtsuSnrAccumulator``) share one implementation: equivalence is
+ then structural rather than something a test has to keep rediscovering.
+
+ The window is reduced in **float64**. Production maps are float32, but the
+ batched (``roi_snr_db_batch``) and numba/GPU paths promote to float64, and a
+ float32 masked reduction is order-dependent so it cannot bit-match a batched
+ one. Computing every path in float64 makes CPU and GPU results identical to
+ float64 rounding instead of differing by ~2e-6 dB with the instance type.
+ """
+ if tile.size < 4:
+ return float("nan")
+ tile = np.asarray(tile, dtype=np.float64)
+
+ try:
+ from skimage.filters import threshold_otsu as _skimage_threshold_otsu # type: ignore
+ except Exception:
+ _skimage_threshold_otsu = None
+
+ if _skimage_threshold_otsu is not None:
+ try:
+ threshold = float(_skimage_threshold_otsu(tile))
+ except Exception:
+ threshold = _otsu_threshold_uint16(tile)
+ else:
+ threshold = _otsu_threshold_uint16(tile)
+
+ foreground = tile[tile >= threshold]
+ background = tile[tile < threshold]
+ if foreground.size < 2 or background.size < 2:
+ return float("nan")
+
+ mean_fg = float(np.mean(foreground))
+ std_bg = float(np.std(background, ddof=1))
+ eps = 1e-12
+ if std_bg < eps:
+ return float("nan")
+ return 20.0 * math.log10(max(mean_fg / std_bg, eps))
+
+
+def roi_snr_db_batch(tiles: Any, xp: Any = np) -> Any:
+ """Vectorized ``roi_snr_db`` over a stack of equal-size ROI windows.
+
+ ``tiles`` is ``(K, h, w)`` or ``(K, N)``; returns a ``(K,)`` float64 array of
+ dB values, one per window, computed as a single batched reduction with **no
+ Python per-ROI loop**. Pass ``xp=cupy`` to run entirely on the GPU — every op
+ below (min/max, bincount, cumsum, argmax, masked reductions) exists in both
+ numpy and cupy, so the same code runs on either device. This is the form the
+ streaming tile pass uses when the tile's ``mean_map`` is already resident in
+ VRAM (``image_qc._compute_channel_maps_on_gpu``): the Otsu SNR is folded on
+ device and only the small ``(K,)`` dB vector returns to host.
+
+ Reproduces ``skimage.filters.threshold_otsu`` (256 bins over each window's own
+ min..max, the float-corrected uniform-bin assignment numpy uses) and the
+ ``foreground-mean / background-std(ddof=1)`` dB. Equivalence vs the per-ROI
+ ``roi_snr_db`` is exact to float64 rounding (max ~1e-14 dB, verified in
+ ``tests/test_roi_snr_db_batch.py``).
+
+ Assumes finite, full-size windows. Constant windows (zero span) return NaN,
+ matching the scalar path (an all-foreground split leaves the background empty).
+ Callers route edge-clipped or non-finite windows to the scalar ``roi_snr_db``.
+ """
+ nbins = 256
+ eps = 1e-12
+ flat = xp.asarray(tiles).reshape(xp.asarray(tiles).shape[0], -1).astype(xp.float64)
+ K, N = flat.shape[0], flat.shape[1]
+ out = xp.full(K, xp.nan, dtype=xp.float64)
+ if N < 4:
+ return out
+
+ mn = flat.min(axis=1)
+ mx = flat.max(axis=1)
+ ok = mx > mn # constant windows -> NaN, as in roi_snr_db
+ if not bool(ok.any()):
+ return out
+ sub = flat[ok]
+ mn_s = mn[ok]
+ step = (mx[ok] - mn_s) / nbins
+
+ # numpy's uniform-bin assignment for np.histogram(row, nbins, (min, max)),
+ # including its float corrections against the arithmetic edge mn + idx*step.
+ idx = ((sub - mn_s[:, None]) / step[:, None]).astype(xp.intp)
+ idx = xp.where(idx == nbins, nbins - 1, idx)
+ idx = xp.clip(idx, 0, nbins - 1)
+ left = mn_s[:, None] + idx * step[:, None]
+ idx = xp.where(sub < left, idx - 1, idx)
+ idx = xp.clip(idx, 0, nbins - 1)
+ right = mn_s[:, None] + (idx + 1) * step[:, None]
+ idx = xp.where((sub >= right) & (idx != nbins - 1), idx + 1, idx)
+ idx = xp.clip(idx, 0, nbins - 1)
+
+ Ks = sub.shape[0]
+ flat_idx = (idx + xp.arange(Ks)[:, None] * nbins).reshape(-1)
+ counts = (
+ xp.bincount(flat_idx, minlength=Ks * nbins)
+ .reshape(Ks, nbins)
+ .astype(xp.float64)
+ )
+ centers = mn_s[:, None] + (xp.arange(nbins) + 0.5) * step[:, None]
+
+ # skimage.filters.threshold_otsu's histogram math, per row.
+ with np.errstate(invalid="ignore", divide="ignore"):
+ w1 = xp.cumsum(counts, axis=1)
+ w2 = xp.cumsum(counts[:, ::-1], axis=1)[:, ::-1]
+ cb = counts * centers
+ mean1 = xp.cumsum(cb, axis=1) / w1
+ mean2 = (xp.cumsum(cb[:, ::-1], axis=1) / w2[:, ::-1])[:, ::-1]
+ var12 = w1[:, :-1] * w2[:, 1:] * (mean1[:, :-1] - mean2[:, 1:]) ** 2
+ thr = centers[xp.arange(Ks), xp.argmax(var12, axis=1)]
+
+ fg = sub >= thr[:, None]
+ nfg = fg.sum(axis=1)
+ nbg = N - nfg
+ with np.errstate(invalid="ignore", divide="ignore"):
+ mean_fg = xp.where(fg, sub, 0.0).sum(axis=1) / nfg
+ mean_bg = xp.where(~fg, sub, 0.0).sum(axis=1) / nbg
+ ss_bg = xp.where(~fg, (sub - mean_bg[:, None]) ** 2, 0.0).sum(axis=1)
+ std_bg = xp.sqrt(ss_bg / (nbg - 1))
+ good = (nfg >= 2) & (nbg >= 2) & (std_bg >= eps)
+ db = xp.full(Ks, xp.nan, dtype=xp.float64)
+ ratio = xp.maximum(mean_fg / std_bg, eps)
+ db = xp.where(good, 20.0 * xp.log10(ratio), db)
+ out[ok] = db
+ return out
+
+
+try:
+ import numba as _numba
+
+ _HAS_NUMBA = True
+except Exception: # numba is an optional accelerator; callers fall back otherwise
+ _HAS_NUMBA = False
+
+if _HAS_NUMBA:
+ # cache=False deliberately: each Nextflow task is a fresh process with an
+ # ephemeral work dir, so an on-disk cache never survives to be reused, and
+ # numba's cache key for a path-loaded module is "", which poisons a
+ # recompile with `ModuleNotFoundError: No module named ''`. Compiling
+ # once per task (~seconds) is negligible against the fold.
+ @_numba.njit(cache=False, fastmath=False)
+ def _roi_snr_db_one_numba(tile_flat, nbins): # noqa: D401
+ """One ROI: skimage-Otsu threshold then fg-mean/bg-std dB, all float64.
+
+ Reproduces ``roi_snr_db`` for a finite, non-constant window; returns NaN
+ for the same degenerate cases (too small, constant, empty fg/bg, zero bg
+ spread). Compiled + released-GIL so ``prange`` scales across cores.
+ """
+ n = tile_flat.size
+ if n < 4:
+ return np.nan
+ mn = tile_flat[0]
+ mx = tile_flat[0]
+ for v in tile_flat:
+ if v < mn:
+ mn = v
+ if v > mx:
+ mx = v
+ if mx <= mn:
+ return np.nan
+ step = (mx - mn) / nbins
+ counts = np.zeros(nbins, dtype=np.float64)
+ for v in tile_flat:
+ b = int((v - mn) / step)
+ if b == nbins:
+ b = nbins - 1
+ if v < mn + b * step:
+ b -= 1
+ elif b != nbins - 1 and v >= mn + (b + 1) * step:
+ b += 1
+ if b < 0:
+ b = 0
+ elif b >= nbins:
+ b = nbins - 1
+ counts[b] += 1.0
+ w1 = np.cumsum(counts)
+ centers = np.empty(nbins, dtype=np.float64)
+ for j in range(nbins):
+ centers[j] = mn + (j + 0.5) * step
+ m1cum = np.cumsum(counts * centers)
+ total = w1[nbins - 1]
+ cbtot = m1cum[nbins - 1]
+ best_var = -1.0
+ best_idx = 0
+ for t in range(nbins - 1):
+ wa = w1[t]
+ wb = total - wa
+ if wa == 0.0 or wb == 0.0:
+ continue
+ ma = m1cum[t] / wa
+ mb = (cbtot - m1cum[t]) / wb
+ var = wa * wb * (ma - mb) ** 2
+ if var > best_var:
+ best_var = var
+ best_idx = t
+ thr = centers[best_idx]
+ sfg = 0.0
+ nfg = 0
+ sbg = 0.0
+ nbg = 0
+ for v in tile_flat:
+ if v >= thr:
+ sfg += v
+ nfg += 1
+ else:
+ sbg += v
+ nbg += 1
+ if nfg < 2 or nbg < 2:
+ return np.nan
+ mean_fg = sfg / nfg
+ mean_bg = sbg / nbg
+ ss = 0.0
+ for v in tile_flat:
+ if v < thr:
+ d = v - mean_bg
+ ss += d * d
+ std_bg = np.sqrt(ss / (nbg - 1))
+ if std_bg < 1e-12:
+ return np.nan
+ ratio = mean_fg / std_bg
+ if ratio < 1e-12:
+ ratio = 1e-12
+ return 20.0 * np.log10(ratio)
+
+ @_numba.njit(cache=False, parallel=True, fastmath=False) # see cache note above
+ def _roi_snr_db_numba_kernel(stack2d, nbins):
+ K = stack2d.shape[0]
+ out = np.empty(K, dtype=np.float64)
+ for k in _numba.prange(K):
+ out[k] = _roi_snr_db_one_numba(stack2d[k], nbins)
+ return out
+
+
+def roi_snr_db_numba(tiles: Any) -> np.ndarray:
+ """CPU-parallel batched ``roi_snr_db`` (numba ``prange``).
+
+ The efficient path for **CPU** instances: ~50x the Python per-ROI loop on 8
+ cores, versus the pure-numpy ``roi_snr_db_batch(xp=np)`` which is memory-bound
+ and no faster than the loop. On **GPU** instances use ``roi_snr_db_batch(xp=cp)``
+ instead. Equivalent to ``roi_snr_db`` to float64 rounding.
+
+ Thread count follows ``numba.set_num_threads`` / ``NUMBA_NUM_THREADS``; the
+ caller must cap it so it does not oversubscribe against the tile worker pool.
+ """
+ if not _HAS_NUMBA:
+ raise RuntimeError("roi_snr_db_numba requires numba, which is not installed")
+ arr = np.asarray(tiles)
+ flat = np.ascontiguousarray(arr.reshape(arr.shape[0], -1), dtype=np.float64)
+ if flat.shape[1] < 4:
+ return np.full(flat.shape[0], np.nan, dtype=np.float64)
+ return _roi_snr_db_numba_kernel(flat, 256)
+
+
+def compute_image_snr_from_pixel_maps(
+ focus_maps: Optional[Dict[str, Any]],
+ df_grid_roi: pd.DataFrame,
+ map_key: str = "dapi_mean_map",
+ max_rois: Optional[int] = None,
+ snr_thresholds: Optional[Dict[str, Any]] = None,
+ precomputed_db: Optional[Any] = None,
+) -> Dict[str, Any]:
+ """
+ Per-ROI SNR_dB from pixel tiles: Otsu foreground vs background std (SNR_plan accurate path).
+ Uses ``dapi_mean_map`` (or ``map_key``) slices [y1:y2, x1:x2] per ROI row.
+
+ ``precomputed_db`` supplies one dB value per ROI when the caller already
+ reduced them during a tiled pass, in which case no pixel map is needed. Both
+ paths get their dB from :func:`roi_snr_db`, so the aggregate statistics below
+ are computed identically either way.
+ """
+ if precomputed_db is None:
+ arr = focus_maps.get(map_key) if focus_maps else None
+ if arr is None:
+ return {"status": "skipped", "reason": f"missing focus_maps[{map_key!r}]"}
+
+ # Keep the map lazy (do NOT force a whole-array float64 copy — for a
+ # full-res / memmap-backed map that materialises tens of GB). Each tile is
+ # sliced then converted per-tile below.
+ img = np.asarray(arr)
+ h, w = img.shape[:2]
+
+ required = {"x1", "x2", "y1", "y2"}
+ if not required.issubset(df_grid_roi.columns):
+ return {"status": "skipped", "reason": f"df_grid_roi needs columns {required}"}
+
+ df = df_grid_roi.copy()
+ if max_rois is not None:
+ df = df.iloc[: int(max_rois)]
+
+ x1a = df["x1"].to_numpy(dtype=np.int64)
+ x2a = df["x2"].to_numpy(dtype=np.int64)
+ y1a = df["y1"].to_numpy(dtype=np.int64)
+ y2a = df["y2"].to_numpy(dtype=np.int64)
+
+ # Per-tile dB array aligned with df rows (NaN for skipped tiles).
+ # Used both for aggregate stats and to expose per-tile values to df_grid_roi
+ # downstream — the §3.4 cross-section-concordance scatter consumes this.
+ per_tile_db = np.full(len(df), np.nan, dtype=np.float64)
+ if precomputed_db is not None:
+ supplied = np.asarray(precomputed_db, dtype=np.float64)
+ if supplied.size < len(df):
+ return {
+ "status": "skipped",
+ "reason": f"precomputed_db has {supplied.size} values for {len(df)} tiles",
+ }
+ per_tile_db = supplied[: len(df)].copy()
+ dbs = [float(v) for v in per_tile_db[~np.isnan(per_tile_db)]]
+ else:
+ dbs = []
+ for k in range(len(df)):
+ x1, x2, y1, y2 = int(x1a[k]), int(x2a[k]), int(y1a[k]), int(y2a[k])
+ x1, y1 = max(0, x1), max(0, y1)
+ x2, y2 = min(w, x2), min(h, y2)
+ if x2 <= x1 or y2 <= y1:
+ continue
+ db_val = roi_snr_db(img[y1:y2, x1:x2])
+ if math.isnan(db_val):
+ continue
+ per_tile_db[k] = db_val
+ dbs.append(db_val)
+
+ # Write per-tile values back to the caller's df_grid_roi (in-place,
+ # parallel to how compute_roi_transcript_snr writes roi_tx_snr_ratio).
+ # df.index is a subset of df_grid_roi.index when max_rois is set; rows
+ # outside that subset retain NaN.
+ df_grid_roi.loc[df.index, "snr_image_otsu_db"] = per_tile_db
+
+ if not dbs:
+ return {"status": "skipped", "reason": "no_valid_roi_tiles"}
+
+ _t = snr_thresholds or {}
+ _img = _t.get("image_snr_db") or {}
+ warn_db = float(_img.get("warn", 15.0))
+ fail_db = float(_img.get("fail", 10.0))
+ med_db = float(np.median(dbs))
+ return {
+ "status": "ok",
+ "method": "per_roi_otsu",
+ "map_key": map_key,
+ "n_rois_computed": len(dbs),
+ "snr_db_median": med_db,
+ "snr_db_mean": float(np.mean(dbs)),
+ "snr_db_p25": float(np.percentile(dbs, 25)),
+ "snr_db_p75": float(np.percentile(dbs, 75)),
+ "verdict": "PASS"
+ if med_db >= warn_db
+ else ("WARN" if med_db >= fail_db else "FAIL"),
+ }
+
+
+# ---------------------------------------------------------------------------
+# Transcripts
+# ---------------------------------------------------------------------------
+
+
+#: Transcript rows per batch for the per-ROI SNR reduction. Bounded so the
+#: per-batch temporaries never approach the frame this replaces.
+ROI_TX_BATCH_ROWS = 2_000_000
+
+
+def load_transcripts(path: Path) -> pd.DataFrame:
+ """Load minimal columns: feature_name, x_location, y_location, cell_id (optional).
+
+ Projects the columns at read time rather than after. A Xenium transcript
+ table has ~20 columns and can run to several hundred million rows, and this
+ load happens while the full-resolution focus planes are still live — reading
+ everything and then slicing cost three overlapping copies of the full frame.
+ """
+ path = Path(path)
+ cols_want = ["feature_name", "x_location", "y_location"]
+
+ if path.suffix.lower() in (".parquet", ".pq"):
+ available = set(pq.ParquetFile(path).schema_arrow.names)
+ missing = [c for c in cols_want if c not in available]
+ if missing:
+ raise ValueError(f"transcripts missing column(s) {missing!r}")
+ keep = cols_want + (["cell_id"] if "cell_id" in available else [])
+ return pd.read_parquet(path, columns=keep)
+
+ if path.name.endswith(".csv.gz") or path.suffix.lower() == ".csv":
+ available = set(pd.read_csv(path, nrows=0).columns)
+ missing = [c for c in cols_want if c not in available]
+ if missing:
+ raise ValueError(f"transcripts missing column(s) {missing!r}")
+ keep = cols_want + (["cell_id"] if "cell_id" in available else [])
+ return pd.read_csv(path, usecols=keep)
+
+ raise ValueError(f"Unsupported transcript format: {path}")
+
+
+def transcripts_um_to_px(df_tx: pd.DataFrame, pixel_size_um: float) -> pd.DataFrame:
+ """Convert x_location / y_location from microns to pixels (divide by pixel size)."""
+ out = df_tx.copy()
+ ps = float(pixel_size_um)
+ if ps <= 0:
+ raise ValueError("pixel_size_um must be positive")
+ out["x_px"] = out["x_location"].astype(np.float64) / ps
+ out["y_px"] = out["y_location"].astype(np.float64) / ps
+ return out
+
+
+# ---------------------------------------------------------------------------
+# ROI transcript SNR + neg_pct
+# ---------------------------------------------------------------------------
+
+
+def _infer_uniform_grid_strides(
+ df_grid_roi: pd.DataFrame,
+) -> Optional[Tuple[int, int]]:
+ """
+ Infer (stride_x, stride_y) for a regular grid like ``image_qc`` (arange(0, W, stride)).
+
+ Returns None if x1/y1 spacing is irregular (fall back to slow geometric assign).
+ """
+ x1 = df_grid_roi["x1"].to_numpy(dtype=np.int64)
+ y1 = df_grid_roi["y1"].to_numpy(dtype=np.int64)
+ ux = np.sort(np.unique(x1))
+ uy = np.sort(np.unique(y1))
+ if len(ux) >= 2:
+ dx = np.diff(ux)
+ if dx.size == 0 or int(dx[0]) <= 0 or not np.all(dx == dx[0]):
+ return None
+ sx = int(dx[0])
+ else:
+ sx = int(df_grid_roi["x2"].iloc[0] - df_grid_roi["x1"].iloc[0])
+ if sx <= 0:
+ return None
+ if len(uy) >= 2:
+ dy = np.diff(uy)
+ if dy.size == 0 or int(dy[0]) <= 0 or not np.all(dy == dy[0]):
+ return None
+ sy = int(dy[0])
+ else:
+ sy = int(df_grid_roi["y2"].iloc[0] - df_grid_roi["y1"].iloc[0])
+ if sy <= 0:
+ return None
+ if not bool(np.all((x1 % sx) == 0)) or not bool(np.all((y1 % sy) == 0)):
+ return None
+ return sx, sy
+
+
+class _UniformRoiLookup:
+ """Prebuilt (iy, ix) -> roi_id lattice, so the streaming path builds it once.
+
+ `_roi_grid_assign_fast_uniform` derives the lattice from the ROI table on every
+ call, and its collision check is an `np.unique` over one entry per ROI -- 424 ms
+ for 4.49 M ROIs. Calling it per transcript batch would spend that repeatedly on a
+ lookup that cannot change: at 1,325,798,498 transcripts the reference sample is
+ 663 batches, so ~291 s of pure rebuild.
+
+ Returns None from :meth:`build` when the ROI layout is not a simple lattice, so
+ the caller can fall back to the general path.
+ """
+
+ __slots__ = ("_grid", "_stride_x", "_stride_y", "_max_ix", "_max_iy")
+
+ def __init__(self, grid, stride_x, stride_y, max_ix, max_iy):
+ self._grid = grid
+ self._stride_x = stride_x
+ self._stride_y = stride_y
+ self._max_ix = max_ix
+ self._max_iy = max_iy
+
+ @classmethod
+ def build(
+ cls, df_grid_roi: pd.DataFrame, stride_x: int, stride_y: int
+ ) -> Optional["_UniformRoiLookup"]:
+ x1 = df_grid_roi["x1"].to_numpy(dtype=np.int64)
+ y1 = df_grid_roi["y1"].to_numpy(dtype=np.int64)
+ rids = df_grid_roi["roi_id"].to_numpy(dtype=np.int32)
+ ix = x1 // stride_x
+ iy = y1 // stride_y
+ max_ix, max_iy = int(ix.max()) + 1, int(iy.max()) + 1
+ flat_idx = iy.astype(np.int64) * max_ix + ix.astype(np.int64)
+ if len(np.unique(flat_idx)) != len(flat_idx):
+ return None
+ grid = np.full(max_iy * max_ix, -1, dtype=np.int32)
+ grid[flat_idx] = rids
+ return cls(grid.reshape(max_iy, max_ix), stride_x, stride_y, max_ix, max_iy)
+
+ def assign(self, x_px: np.ndarray, y_px: np.ndarray) -> np.ndarray:
+ """roi_id per point. Out-of-lattice coordinates clip to the edge ROI, which
+ is what `_roi_grid_assign_fast_uniform` has always done."""
+ xi = (x_px.astype(np.int64, copy=False) // self._stride_x).clip(
+ 0, self._max_ix - 1
+ )
+ yi = (y_px.astype(np.int64, copy=False) // self._stride_y).clip(
+ 0, self._max_iy - 1
+ )
+ return self._grid[yi, xi]
+
+
+def _roi_grid_assign_fast_uniform(
+ df_grid_roi: pd.DataFrame,
+ x_px: np.ndarray,
+ y_px: np.ndarray,
+ stride_x: int,
+ stride_y: int,
+) -> Optional[np.ndarray]:
+ """
+ O(n_tx) assignment: tile index from pixel coords, lookup pre-filled roi_id grid.
+
+ Returns None if ROI layout does not match a simple (iy, ix) lattice (collisions).
+ """
+ lookup = _UniformRoiLookup.build(df_grid_roi, stride_x, stride_y)
+ if lookup is None:
+ return None
+ return lookup.assign(x_px, y_px)
+
+
+def _roi_grid_assign(
+ df_grid_roi: pd.DataFrame,
+ x_px: np.ndarray,
+ y_px: np.ndarray,
+ stride_xy: Optional[Tuple[int, int]] = None,
+) -> np.ndarray:
+ """
+ Assign each transcript pixel to ``roi_id``, or -1.
+
+ **Fast path (typical):** uniform stride grid from ``image_qc`` → O(n_tx) index + lookup.
+ Pass ``stride_xy=(stride_x, stride_y)`` from the same ``roi_size`` / stride used to build
+ the grid (see ``bin/image_qc.py``) to skip inference on large ROI tables.
+
+ **Slow path:** O(n_roi × n_tx) rectangle tests (irregular ROIs / spikes).
+ """
+ if stride_xy is not None:
+ sx, sy = int(stride_xy[0]), int(stride_xy[1])
+ if sx > 0 and sy > 0:
+ fast = _roi_grid_assign_fast_uniform(df_grid_roi, x_px, y_px, sx, sy)
+ if fast is not None:
+ logger.debug(
+ "SNR ROI assignment: fast path (caller stride %d×%d), %d points",
+ sx,
+ sy,
+ int(x_px.shape[0]),
+ )
+ return fast
+ logger.info(
+ "SNR ROI assignment: caller stride (%d, %d) incompatible with ROI layout; "
+ "inferring or slow path",
+ sx,
+ sy,
+ )
+
+ inferred = _infer_uniform_grid_strides(df_grid_roi)
+ if inferred is not None:
+ sx, sy = inferred
+ fast = _roi_grid_assign_fast_uniform(df_grid_roi, x_px, y_px, sx, sy)
+ if fast is not None:
+ logger.info(
+ "SNR ROI assignment: fast uniform grid (inferred stride %d×%d), %d transcripts",
+ sx,
+ sy,
+ int(x_px.shape[0]),
+ )
+ return fast
+ logger.info(
+ "SNR ROI assignment: inferred stride rejected (collision); using slow path"
+ )
+
+ rid_out = np.full(x_px.shape[0], -1, dtype=np.int32)
+ rids = df_grid_roi["roi_id"].to_numpy(dtype=np.int32)
+ x1a = df_grid_roi["x1"].to_numpy(dtype=np.float64)
+ x2a = df_grid_roi["x2"].to_numpy(dtype=np.float64)
+ y1a = df_grid_roi["y1"].to_numpy(dtype=np.float64)
+ y2a = df_grid_roi["y2"].to_numpy(dtype=np.float64)
+ n_tx = int(x_px.shape[0])
+ n_roi = len(df_grid_roi)
+ if n_tx > 2_000_000 and n_roi * n_tx > 5e9:
+ logger.warning(
+ "Large transcript count (%d) × ROIs (%d): slow ROI assignment O(n_roi×n_tx).",
+ n_tx,
+ n_roi,
+ )
+ for i in range(n_roi):
+ m = (x_px >= x1a[i]) & (x_px < x2a[i]) & (y_px >= y1a[i]) & (y_px < y2a[i])
+ rid_out[m] = int(rids[i])
+ return rid_out
+
+
+def _neg_mask_vectorized(feats: np.ndarray) -> np.ndarray:
+ """Same logic as ``is_neg_probe_feature``, vectorised for large transcript tables."""
+ s = pd.Series(feats, dtype="string")
+ pat = r"^(?:NegControlProbe_|NegControlCodeword_|Blank[-_]|antisense_)"
+ return s.str.match(pat, case=False).fillna(False).to_numpy(dtype=bool)
+
+
+def _neg_mask_for_column(values: pd.Series) -> np.ndarray:
+ """Negative-control mask for a transcript ``feature_name`` column.
+
+ When the column is dictionary-encoded -- which is how Xenium writes it, and what
+ ``ParquetFile.iter_batches`` hands back -- the regex runs over the ~13 k distinct
+ categories and the result is indexed by the codes, instead of materialising one
+ Python string per transcript. The mask is identical either way.
+ """
+ if isinstance(values.dtype, pd.CategoricalDtype):
+ cats = _neg_mask_vectorized(values.cat.categories.to_numpy())
+ codes = values.cat.codes.to_numpy()
+ out = np.zeros(codes.shape[0], dtype=bool)
+ known = codes >= 0
+ out[known] = cats[codes[known]]
+ return out
+ return _neg_mask_vectorized(values.astype(str).to_numpy())
+
+
+def _accumulate_roi_tx_counts(
+ df_grid_roi: pd.DataFrame,
+ row_ix: pd.Series,
+ x_px: np.ndarray,
+ y_px: np.ndarray,
+ is_neg: np.ndarray,
+ counters: Tuple[np.ndarray, np.ndarray, np.ndarray],
+ stride_xy: Optional[Tuple[int, int]],
+ lookup: Optional["_UniformRoiLookup"] = None,
+ roi_id_is_arange: bool = False,
+) -> None:
+ """Fold one batch of transcripts into the per-ROI counters.
+
+ ``np.bincount`` rather than ``np.add.at``: both count occurrences exactly, but
+ ``add.at`` is an order of magnitude slower, and this runs over every transcript.
+
+ Two count-preserving micro-opts:
+
+ * When ``roi_id_is_arange`` (``roi_id == arange(n_rois)``, how ``image_qc`` builds
+ the grid), the ``roi_id -> row index`` map is the identity: a valid id maps to
+ itself and an out-of-grid ``-1`` stays ``-1``. The pandas ``reindex`` is then a
+ no-op, so ``j`` is ``assign`` directly.
+ * ``real = total - neg`` instead of a third ``bincount``. Every counted transcript
+ is exactly one of neg / real, so this is the same integer per ROI -- two
+ ``bincount`` passes rather than three.
+ """
+ if lookup is not None:
+ assign = lookup.assign(x_px, y_px)
+ else:
+ assign = _roi_grid_assign(df_grid_roi, x_px, y_px, stride_xy=stride_xy)
+ assign = np.asarray(assign, dtype=np.int64)
+ if roi_id_is_arange:
+ j = assign
+ else:
+ j = row_ix.reindex(assign, fill_value=-1).to_numpy()
+ ok = (j >= 0) & (assign >= 0)
+ real_c, neg_c, total_c = counters
+ n = real_c.shape[0]
+ total_b = np.bincount(j[ok], minlength=n)
+ neg_b = np.bincount(j[ok & is_neg], minlength=n)
+ total_c += total_b
+ neg_c += neg_b
+ real_c += total_b - neg_b
+
+
+def _stream_roi_tx_counts(
+ df_grid_roi: pd.DataFrame,
+ row_ix: pd.Series,
+ path: Path,
+ pixel_size_um: float,
+ stride_xy: Optional[Tuple[int, int]],
+ batch_rows: int = ROI_TX_BATCH_ROWS,
+ roi_id_is_arange: bool = False,
+) -> Tuple[Tuple[np.ndarray, np.ndarray, np.ndarray], int]:
+ """Per-ROI real/neg/total transcript counts, read in batches, folded in parallel.
+
+ The whole-frame version materialised ``feature_name``, ``x_location`` and
+ ``y_location`` for every transcript, then ``transcripts_um_to_px`` copied the
+ frame and added two more float64 columns. On run 1ZyVIlaKBYxJrQ that OOM-killed
+ IMAGE_QC (exit 137) at the 180 GB tier *after* the tile pass had completed in
+ 19.4 minutes. The outputs are three per-ROI int64 arrays, so nothing about this
+ needs the transcripts resident.
+
+ Batches are decoded (``batch.to_pandas``, which releases the GIL) and folded
+ (``np.bincount``, also GIL-releasing) on a thread pool sized to ``os.cpu_count()``.
+ Each worker thread owns a private ``(real, neg, total)`` counter triple and folds
+ every batch it draws into it via ``_accumulate_roi_tx_counts``; the per-thread
+ partials are summed on the main thread once the pool drains. Counts are additive,
+ so the sum is bit-identical to the serial fold no matter how the batches interleave
+ across threads -- there is no order dependence in integer addition. In-flight
+ batches are capped at the worker count, so peak memory is bounded exactly as the
+ serial loop's was (a handful of ``batch_rows`` batches, not the whole table).
+ """
+ n_rois = len(df_grid_roi)
+ ps = float(pixel_size_um)
+ if ps <= 0:
+ raise ValueError("pixel_size_um must be positive")
+
+ # Built once, not per batch: see _UniformRoiLookup. None means the ROI layout is
+ # not a lattice, and each batch falls back to the general assignment.
+ lookup = None
+ if stride_xy is not None and stride_xy[0] > 0 and stride_xy[1] > 0:
+ lookup = _UniformRoiLookup.build(
+ df_grid_roi, int(stride_xy[0]), int(stride_xy[1])
+ )
+ if lookup is None:
+ logger.info(
+ "SNR ROI assignment: no uniform lattice; assigning per batch (slower)"
+ )
+
+ handle = pq.ParquetFile(str(path))
+ columns = ["feature_name", "x_location", "y_location"]
+ max_workers = max(1, os.cpu_count() or 1)
+ logger.info(
+ "SNR ROI transcript counts: streaming %s rows in %s-row batches across %d threads",
+ f"{handle.metadata.num_rows:,}",
+ f"{batch_rows:,}",
+ max_workers,
+ )
+
+ # One counter triple per worker thread. Registered under a lock the first time a
+ # thread runs, so the main thread can sum them after the pool drains.
+ tls = threading.local()
+ partials: List[Tuple[np.ndarray, np.ndarray, np.ndarray]] = []
+ partials_lock = threading.Lock()
+
+ def _fold_one(batch) -> int:
+ thread_counters = getattr(tls, "counters", None)
+ if thread_counters is None:
+ thread_counters = (
+ np.zeros(n_rois, dtype=np.int64),
+ np.zeros(n_rois, dtype=np.int64),
+ np.zeros(n_rois, dtype=np.int64),
+ )
+ tls.counters = thread_counters
+ with partials_lock:
+ partials.append(thread_counters)
+ frame = batch.to_pandas()
+ _accumulate_roi_tx_counts(
+ df_grid_roi,
+ row_ix,
+ frame["x_location"].to_numpy(dtype=np.float64) / ps,
+ frame["y_location"].to_numpy(dtype=np.float64) / ps,
+ _neg_mask_for_column(frame["feature_name"]),
+ thread_counters,
+ stride_xy,
+ lookup=lookup,
+ roi_id_is_arange=roi_id_is_arange,
+ )
+ return len(frame)
+
+ n_used = 0
+ n_batches = 0
+ batch_iter = handle.iter_batches(batch_size=batch_rows, columns=columns)
+ with ThreadPoolExecutor(max_workers=max_workers) as pool:
+ # Cap outstanding futures at the worker count so at most ~max_workers batches
+ # are resident at once (the same memory envelope as the serial loop).
+ inflight: deque = deque()
+ for batch in batch_iter:
+ inflight.append(pool.submit(_fold_one, batch))
+ del batch
+ if len(inflight) >= max_workers:
+ n_used += inflight.popleft().result()
+ n_batches += 1
+ while inflight:
+ n_used += inflight.popleft().result()
+ n_batches += 1
+
+ real_c = np.zeros(n_rois, dtype=np.int64)
+ neg_c = np.zeros(n_rois, dtype=np.int64)
+ total_c = np.zeros(n_rois, dtype=np.int64)
+ for p_real, p_neg, p_total in partials:
+ real_c += p_real
+ neg_c += p_neg
+ total_c += p_total
+ logger.info(
+ "SNR ROI transcript counts: %s rows in %d batches across %d threads",
+ f"{n_used:,}",
+ n_batches,
+ len(partials),
+ )
+ return (real_c, neg_c, total_c), n_used
+
+
+def compute_roi_snr(
+ df_grid_roi: pd.DataFrame,
+ df_tx: Optional[pd.DataFrame] = None,
+ x_col: str = "x_px",
+ y_col: str = "y_px",
+ roi_grid_stride: Optional[Tuple[int, int]] = None,
+ snr_thresholds: Optional[Dict[str, Any]] = None,
+ *,
+ transcripts_path: Optional[Path] = None,
+ pixel_size_um: Optional[float] = None,
+ batch_rows: int = ROI_TX_BATCH_ROWS,
+) -> Tuple[pd.DataFrame, Dict[str, Any]]:
+ """
+ Per-ROI real vs neg transcript counts, ratio, neg_pct; **mutates** *df_grid_roi* in place
+ (caller should pass a copy if the original must stay unchanged — ``run_snr_module`` does).
+
+ Give either *df_tx* -- a frame with feature_name and pixel columns x_col, y_col,
+ see ``transcripts_um_to_px`` -- or *transcripts_path* plus *pixel_size_um*, in
+ which case the table is read in batches and never held whole. The counters are
+ additive, so both give identical results; the streaming form exists because the
+ whole-frame one OOM-killed IMAGE_QC on a production sample.
+
+ ``roi_grid_stride`` should match the grid used in ``image_qc.py`` (typically
+ ``(roi_size, roi_size)`` when stride defaults to roi_size).
+ """
+ if (df_tx is None) == (transcripts_path is None):
+ raise ValueError("pass exactly one of df_tx or transcripts_path")
+
+ df = df_grid_roi
+ if "roi_id" not in df.columns:
+ df = df.copy()
+ df["roi_id"] = np.arange(len(df), dtype=np.int32)
+
+ n_rois = len(df)
+ # Map roi_id → row index (avoids O(max(roi_id)) array if ids are sparse)
+ row_ix = pd.Series(np.arange(n_rois, dtype=np.int32), index=df["roi_id"].values)
+
+ # image_qc builds the grid with roi_id == arange(n_rois); then row_ix is the
+ # identity and the per-batch reindex can be skipped (see _accumulate_roi_tx_counts).
+ roi_ids = df["roi_id"].to_numpy()
+ roi_id_is_arange = bool(
+ roi_ids.dtype.kind in ("i", "u")
+ and np.array_equal(roi_ids, np.arange(n_rois, dtype=roi_ids.dtype))
+ )
+
+ if transcripts_path is not None:
+ if pixel_size_um is None:
+ raise ValueError("pixel_size_um is required with transcripts_path")
+ counters, n_used = _stream_roi_tx_counts(
+ df,
+ row_ix,
+ Path(transcripts_path),
+ pixel_size_um,
+ roi_grid_stride,
+ batch_rows=batch_rows,
+ roi_id_is_arange=roi_id_is_arange,
+ )
+ else:
+ counters = (
+ np.zeros(n_rois, dtype=np.int64),
+ np.zeros(n_rois, dtype=np.int64),
+ np.zeros(n_rois, dtype=np.int64),
+ )
+ _accumulate_roi_tx_counts(
+ df,
+ row_ix,
+ df_tx[x_col].to_numpy(dtype=np.float64),
+ df_tx[y_col].to_numpy(dtype=np.float64),
+ _neg_mask_for_column(df_tx["feature_name"]),
+ counters,
+ roi_grid_stride,
+ roi_id_is_arange=roi_id_is_arange,
+ )
+ n_used = int(len(df_tx))
+ real_c, neg_c, total_c = counters
+
+ df["snr_real_tx"] = real_c
+ df["snr_neg_tx"] = neg_c
+ df["snr_total_tx"] = total_c
+ df["neg_pct"] = np.where(
+ total_c > 0, neg_c.astype(np.float64) / total_c.astype(np.float64), np.nan
+ )
+
+ ratio = np.divide(
+ real_c.astype(np.float64),
+ neg_c.astype(np.float64),
+ out=np.full(n_rois, np.nan, dtype=np.float64),
+ where=neg_c > 0,
+ )
+ df["roi_tx_snr_ratio"] = ratio
+ df["roi_tx_snr_log"] = np.log10(np.maximum(ratio, 1e-12))
+
+ med_ratio = float(np.nanmedian(ratio[total_c > 0]))
+ med_neg_pct = float(np.nanmedian(df.loc[total_c > 0, "neg_pct"]))
+ _t = snr_thresholds or {}
+ _rt = _t.get("roi_tx") or {}
+ ratio_warn = float(_rt.get("ratio_warn", 3.0))
+ ratio_fail = float(_rt.get("ratio_fail", 1.5))
+ neg_pct_warn = float(_rt.get("neg_pct_warn", 0.15))
+ neg_pct_fail = float(_rt.get("neg_pct_fail", 0.30))
+ verdict = "PASS"
+ if med_ratio < ratio_warn or med_neg_pct > neg_pct_warn:
+ verdict = "WARN"
+ if med_ratio < ratio_fail or med_neg_pct > neg_pct_fail:
+ verdict = "FAIL"
+
+ summary = {
+ "status": "ok",
+ "method": "roi_tx_target_vs_neg",
+ "n_transcripts_used": int(n_used),
+ "n_rois_with_tx": int(np.sum(total_c > 0)),
+ "median_roi_tx_snr_ratio": med_ratio,
+ "median_neg_pct": med_neg_pct,
+ "verdict": verdict,
+ }
+ return df, summary
+
+
+# ---------------------------------------------------------------------------
+# Spatial clustering of neg_pct — Moran (optional) + quadrant fallback
+# ---------------------------------------------------------------------------
+
+
+def compute_neg_spatial_autocorrelation(
+ df_grid_roi: pd.DataFrame,
+ moran_max_rois: int = 25_000,
+ moran_permutations: int = 99,
+ moran_subsample_seed: int = 42,
+ *,
+ include_moran: bool = False,
+ snr_thresholds: Optional[Dict[str, Any]] = None,
+) -> Dict[str, Any]:
+ """
+ SNR_plan: Moran's I with 4-neighbour weights if libpysal/esda available;
+ else quadrant variance proxy on neg_pct.
+
+ Moran + permutations on hundreds of thousands of ROIs is impractical; when
+ ``len(df) > moran_max_rois``, a fixed-seed random subsample is used **only**
+ for the Moran branch. Quadrant summary still uses all finite-``neg_pct`` ROIs.
+
+ If ``include_moran`` is False, the Moran branch is not run (quadrant summary only;
+ avoids PySAL/esda and saves time). If True but packages are missing, Moran is
+ skipped inside a try/except and ``moran_note`` is set — the run still succeeds.
+ """
+ if "neg_pct" not in df_grid_roi.columns:
+ return {"status": "skipped", "reason": "neg_pct not computed"}
+
+ df = df_grid_roi[np.isfinite(df_grid_roi["neg_pct"])].copy()
+ if len(df) < 8:
+ return {"status": "skipped", "reason": "too_few_rois"}
+
+ cx = (df["x1"].astype(np.float64) + df["x2"].astype(np.float64)) / 2.0
+ cy = (df["y1"].astype(np.float64) + df["y2"].astype(np.float64)) / 2.0
+ z = df["neg_pct"].astype(np.float64).values
+
+ # Quadrant proxy
+ mx, my = float(np.median(cx)), float(np.median(cy))
+ quad = np.zeros(4, dtype=np.float64)
+ nq = np.zeros(4, dtype=np.int64)
+ for k, mask in enumerate(
+ [
+ (cx < mx) & (cy < my),
+ (cx >= mx) & (cy < my),
+ (cx < mx) & (cy >= my),
+ (cx >= mx) & (cy >= my),
+ ]
+ ):
+ quad[k] = float(np.mean(z[mask])) if np.any(mask) else 0.0
+ nq[k] = int(np.sum(mask))
+ spread = float((np.max(quad) - np.min(quad)) / (np.mean(z) + 1e-9))
+ _t = snr_thresholds or {}
+ _ns = _t.get("neg_spatial") or {}
+ qs_warn = float(_ns.get("quadrant_spread_warn", 0.5))
+ qs_fail = float(_ns.get("quadrant_spread_fail", 1.0))
+ q_verdict = "PASS"
+ if spread > qs_warn:
+ q_verdict = "WARN"
+ if spread > qs_fail:
+ q_verdict = "FAIL"
+
+ out: Dict[str, Any] = {
+ "status": "ok",
+ "method_primary": "quadrant_spread",
+ "quadrant_neg_pct_means": quad.tolist(),
+ "quadrant_counts": nq.tolist(),
+ "quadrant_spread_index": spread,
+ "quadrant_verdict": q_verdict,
+ }
+
+ # Optional Moran (subsampled when n is large — see docstring)
+ if include_moran:
+ try:
+ from esda.moran import Moran # type: ignore
+ from libpysal.weights import W # type: ignore
+ from scipy.spatial import cKDTree # type: ignore
+
+ n_all = len(df)
+ if n_all > moran_max_rois:
+ rng = np.random.default_rng(moran_subsample_seed)
+ pick = rng.choice(n_all, size=moran_max_rois, replace=False)
+ cx_m = cx.iloc[pick].to_numpy(dtype=np.float64)
+ cy_m = cy.iloc[pick].to_numpy(dtype=np.float64)
+ z_m = z[pick]
+ out["moran_subsample"] = {
+ "n_used": int(moran_max_rois),
+ "n_available": int(n_all),
+ "seed": int(moran_subsample_seed),
+ }
+ else:
+ cx_m = cx.to_numpy(dtype=np.float64)
+ cy_m = cy.to_numpy(dtype=np.float64)
+ z_m = z
+
+ xy = np.column_stack([cx_m, cy_m])
+ # kNN weights k=4 as SNR_plan neighbour count
+ tree = cKDTree(xy)
+ _d, idx = tree.query(xy, k=min(5, len(xy)))
+ neighbors = {i: [int(j) for j in idx[i][1:]] for i in range(len(xy))}
+
+ w = W(neighbors, silence_warnings=True)
+ mi = Moran(z_m, w, permutations=int(moran_permutations))
+ moran_i = float(mi.I)
+ p_sim = float(mi.p_sim) if mi.p_sim is not None else float("nan")
+ m_warn = float(_ns.get("moran_warn", 0.5))
+ m_fail = float(_ns.get("moran_fail", 1.0))
+ m_verdict = "PASS"
+ if moran_i > m_warn:
+ m_verdict = "WARN"
+ if moran_i > m_fail:
+ m_verdict = "FAIL"
+ if not math.isnan(p_sim) and p_sim >= 0.05:
+ m_verdict = "PASS"
+ out["method_secondary"] = "moran_knn4"
+ out["moran_i"] = moran_i
+ out["moran_p_sim"] = p_sim
+ out["moran_verdict"] = m_verdict
+ except Exception as e:
+ out["moran_note"] = f"skipped ({type(e).__name__}: {e})"
+ else:
+ out["moran_note"] = "skipped (Moran disabled; quadrant summary only)"
+
+ # Combined: escalate per SNR_plan (clustered noise)
+ prim = out.get("moran_verdict", q_verdict)
+ if prim == "FAIL" or q_verdict == "FAIL":
+ out["verdict"] = "FAIL"
+ elif prim == "WARN" or q_verdict == "WARN":
+ out["verdict"] = "WARN"
+ else:
+ out["verdict"] = "PASS"
+
+ if out["verdict"] in ("WARN", "FAIL"):
+ out["report_note"] = (
+ "Elevated spatial clustering of negative-control probes can reflect "
+ "genuine biological heterogeneity (e.g. necrosis, adipose tissue, or "
+ "varying cell density) rather than a technical artefact. Consider "
+ "reviewing the spatial distribution map before concluding a quality issue."
+ )
+ return out
+
+
+# ---------------------------------------------------------------------------
+# Slide-level matrix SNR (h5) — Plummer-style vs SpatialQM-style
+# ---------------------------------------------------------------------------
+
+
+def load_expression_matrix_h5(
+ path: Path,
+) -> Tuple[Any, List[str], List[str], Dict[str, Any]]:
+ """
+ Load 10x-style Xenium h5 sparse matrix [features x cells].
+
+ 10x / Xenium may store **CSR** (``len(indptr) == n_features + 1``) or **CSC**
+ (``len(indptr) == n_cells + 1``). Returns a **scipy.sparse.csr_matrix** without
+ densifying (full matrix can be tens of GB).
+ """
+ import h5py
+
+ path = Path(path)
+ meta: Dict[str, Any] = {"path": str(path)}
+ with h5py.File(path, "r") as f:
+ if "matrix" in f:
+ g = f["matrix"]
+ shape_t = tuple(g["shape"][:]) if "shape" in g else None
+ if shape_t is None or len(shape_t) != 2:
+ raise ValueError("h5 matrix missing valid shape")
+ n_feat, n_cell = int(shape_t[0]), int(shape_t[1])
+ data = g["data"][:]
+ indices = g["indices"][:]
+ indptr = g["indptr"][:]
+ from scipy.sparse import csc_matrix, csr_matrix # type: ignore
+
+ if len(indptr) == n_feat + 1:
+ mat = csr_matrix(
+ (data, indices, indptr), shape=shape_t, dtype=np.float64
+ )
+ meta["sparse_layout"] = "csr"
+ elif len(indptr) == n_cell + 1:
+ mat = csc_matrix(
+ (data, indices, indptr), shape=shape_t, dtype=np.float64
+ ).tocsr()
+ meta["sparse_layout"] = "csc_assembled_csr"
+ else:
+ raise ValueError(
+ f"Unrecognized sparse indptr length {len(indptr)} for shape {shape_t}"
+ )
+ fg = g["features"]
+ if "name" in fg:
+ raw = fg["name"][:]
+ elif "id" in fg:
+ raw = fg["id"][:]
+ else:
+ raise ValueError("h5 matrix/features has neither 'name' nor 'id'")
+ names = [x.decode() if isinstance(x, bytes) else str(x) for x in raw]
+ if "feature_type" in fg:
+ ftype = [
+ x.decode() if isinstance(x, bytes) else str(x)
+ for x in fg["feature_type"][:]
+ ]
+ else:
+ ftype = ["Gene Expression"] * len(names)
+ else:
+ raise ValueError("Unrecognized h5 layout (expected 'matrix' group)")
+
+ meta.update({"n_features": mat.shape[0], "n_cells": mat.shape[1]})
+ return mat, names, ftype, meta
+
+
+PSEUDOCOUNT = 0.1
+EPS = 1e-8
+
+# Fallback defaults for slide-level SNR (used when YAML thresholds not provided).
+SLIDE_SNR_THRESHOLDS = {
+ "plummer_pass_min": 0.12,
+ "plummer_fail_below": 0.10,
+}
+
+
+def _mean_all(x: Any) -> float:
+ """Mean across all elements; works for scipy sparse and dense arrays."""
+ m = x.mean()
+ if hasattr(m, "A1"):
+ return float(m.A1[0])
+ return float(np.asarray(m).reshape(-1)[0])
+
+
+def _mean_per_feature(feature_by_cell: Any) -> np.ndarray:
+ """Row means (features across cells); works for sparse and dense."""
+ out = feature_by_cell.mean(axis=1)
+ return np.asarray(out, dtype=np.float64).ravel()
+
+
+def _build_feature_masks(feature_names: List[str]) -> tuple[np.ndarray, np.ndarray]:
+ is_neg = np.array([is_neg_probe_feature(n) for n in feature_names], dtype=bool)
+ is_real = ~is_neg
+ return is_real, is_neg
+
+
+def compute_slide_snr_plummer_corrected(
+ mat: Any,
+ feature_names: List[str],
+ snr_thresholds: Optional[Dict[str, Any]] = None,
+) -> Dict[str, Any]:
+ """
+ Slide SNR: log10(mean_real + 0.1) - log10(mean_neg + 0.1) over matrix elements
+ (features × cells), real vs neg probe subsets.
+ """
+ is_real, is_neg = _build_feature_masks(feature_names)
+ if not np.any(is_real) or not np.any(is_neg):
+ return {"status": "skipped", "reason": "missing real or neg features"}
+
+ real_mat = mat[is_real]
+ neg_mat = mat[is_neg]
+
+ mean_real = _mean_all(real_mat)
+ mean_neg = _mean_all(neg_mat)
+ log_real = math.log10(mean_real + PSEUDOCOUNT)
+ log_neg = math.log10(mean_neg + PSEUDOCOUNT)
+ snr = float(log_real - log_neg)
+
+ real_feature_means = _mean_per_feature(real_mat)
+ dynamic_range = float(
+ math.log10(float(real_feature_means.max()) + PSEUDOCOUNT) - log_neg
+ )
+ noise_floor_pct = float(mean_neg / (mean_real + EPS) * 100.0)
+
+ _t = snr_thresholds or {}
+ _pl = _t.get("slide_plummer") or {}
+ pmin = float(_pl.get("pass_min", SLIDE_SNR_THRESHOLDS["plummer_pass_min"]))
+ fbelow = float(_pl.get("fail_below", SLIDE_SNR_THRESHOLDS["plummer_fail_below"]))
+ if snr < fbelow:
+ verdict = "FAIL"
+ elif snr < pmin:
+ verdict = "WARN"
+ else:
+ verdict = "PASS"
+
+ return {
+ "status": "ok",
+ "method": "plummer_corrected",
+ "formula": "log10(mean_real+0.1)-log10(mean_neg+0.1)",
+ "snr": snr,
+ "dynamic_range": dynamic_range,
+ "mean_real": mean_real,
+ "mean_neg": mean_neg,
+ "n_real_genes": int(np.sum(is_real)),
+ "n_neg_probes": int(np.sum(is_neg)),
+ "noise_floor_pct": noise_floor_pct,
+ "verdict": verdict,
+ "thresholds_used": {
+ "pass_min": pmin,
+ "fail_below": fbelow,
+ "note": "PASS if snr >= pass_min; WARN if fail_below <= snr < pass_min; FAIL if snr < fail_below",
+ },
+ }
+
+
+def compute_slide_snr_spatialqm_corrected(
+ mat: Any,
+ feature_names: List[str],
+ snr_thresholds: Optional[Dict[str, Any]] = None,
+) -> Dict[str, Any]:
+ """
+ Per-gene variant: snr_g = log10(mean_gene_g + 0.1) - log10(mean_neg + 0.1); slide_snr = mean(snr_g).
+ """
+ is_real, is_neg = _build_feature_masks(feature_names)
+ if not np.any(is_real) or not np.any(is_neg):
+ return {"status": "skipped", "reason": "missing real or neg features"}
+
+ real_mat = mat[is_real]
+ neg_mat = mat[is_neg]
+ mean_neg = _mean_all(neg_mat)
+ log_neg = math.log10(mean_neg + PSEUDOCOUNT)
+
+ real_feature_means = _mean_per_feature(real_mat)
+ per_gene_snr = np.log10(real_feature_means + PSEUDOCOUNT) - log_neg
+ snr_mean = float(np.mean(per_gene_snr))
+ snr_median = float(np.median(per_gene_snr))
+ snr_p10 = float(np.percentile(per_gene_snr, 10))
+ pct_genes_above_neg = float(np.mean(per_gene_snr > 0.0) * 100.0)
+ dynamic_range = float(
+ math.log10(float(real_feature_means.max()) + PSEUDOCOUNT) - log_neg
+ )
+
+ _t = snr_thresholds or {}
+ _sq = _t.get("slide_spatialqm") or {}
+ verdict_metric = _sq.get("metric", "pct_genes_above_neg")
+ sq_warn = float(_sq.get("warn", 45))
+ sq_fail = float(_sq.get("fail", 35))
+
+ if verdict_metric == "pct_genes_above_neg":
+ verdict_value = pct_genes_above_neg
+ else:
+ # Legacy: snr_mean
+ verdict_value = snr_mean
+ verdict = (
+ "PASS"
+ if verdict_value >= sq_warn
+ else ("WARN" if verdict_value >= sq_fail else "FAIL")
+ )
+
+ return {
+ "status": "ok",
+ "method": "spatialqm_per_gene_corrected",
+ "snr_mean": snr_mean,
+ "snr_median": snr_median,
+ "snr_p10": snr_p10,
+ "pct_genes_above_neg": pct_genes_above_neg,
+ "dynamic_range": dynamic_range,
+ "mean_neg": mean_neg,
+ "n_real_genes": int(np.sum(is_real)),
+ "n_neg_probes": int(np.sum(is_neg)),
+ "verdict": verdict,
+ "thresholds_used": {
+ "metric": verdict_metric,
+ "warn": sq_warn,
+ "fail": sq_fail,
+ },
+ }
+
+
+# ---------------------------------------------------------------------------
+# aggregate_snr_verdict + run_snr_module
+# ---------------------------------------------------------------------------
+
+
+def save_snr_roi_tx_table(
+ df: pd.DataFrame,
+ outdir: Path,
+ *,
+ basename: str = SNR_ROI_TX_TABLE_BASENAME,
+) -> Optional[Path]:
+ """
+ Persist the grid with ``compute_roi_snr`` columns (``snr_real_tx``, ``snr_neg_tx``,
+ ``snr_total_tx``, ``neg_pct``, ``roi_tx_snr_ratio``, …) for downstream plots.
+
+ Writes ``{basename}.parquet`` when Parquet is available; otherwise ``{basename}.csv.gz``.
+ """
+ need = {"snr_real_tx", "snr_neg_tx", "snr_total_tx"}
+ if not need.issubset(df.columns):
+ return None
+ outdir = Path(outdir)
+ outdir.mkdir(parents=True, exist_ok=True)
+ path_pq = outdir / f"{basename}.parquet"
+ try:
+ df.to_parquet(path_pq, index=False)
+ logger.info("Wrote %s", path_pq)
+ return path_pq
+ except Exception as e:
+ logger.warning("SNR_roi_tx Parquet write failed (%s), trying CSV.gz", e)
+ path_gz = outdir / f"{basename}.csv.gz"
+ try:
+ df.to_csv(path_gz, index=False, compression="gzip")
+ logger.info("Wrote %s", path_gz)
+ return path_gz
+ except Exception as e2:
+ logger.warning("Could not write SNR_roi_tx table: %s", e2)
+ return None
+
+
+def aggregate_snr_verdict(parts: Dict[str, Dict[str, Any]]) -> Dict[str, Any]:
+ """Any FAIL → overall FAIL; clustered neg (FAIL) escalates WARN → FAIL."""
+ overall = "PASS"
+ for sub in parts.values():
+ if not isinstance(sub, dict):
+ continue
+ v = sub.get("verdict")
+ if v == "FAIL":
+ overall = "FAIL"
+ elif v == "WARN" and overall != "FAIL":
+ overall = "WARN"
+ ns = parts.get(SNR_CKEY_ROI_NEG_SPATIAL) or {}
+ if ns.get("verdict") == "FAIL" and overall == "WARN":
+ overall = "FAIL"
+ return {"overall_snr_verdict": overall}
+
+
+def run_snr_module(
+ xenium_bundle_dir: Optional[Path],
+ df_grid_roi: pd.DataFrame,
+ outdir: Path,
+ focus_maps: Optional[Dict[str, Any]] = None,
+ roi_snr_db: Optional[Any] = None,
+ intensity_threshold: float = 0.0,
+ transcripts_path: Optional[Path] = None,
+ cell_matrix_h5: Optional[Path] = None,
+ pixel_size_um: Optional[float] = None,
+ otsu_max_rois: Optional[int] = None,
+ save_roi_tx_table: bool = True,
+ write_snr_json: bool = True,
+ snr_include_moran: bool = False,
+ roi_grid_stride: Optional[Tuple[int, int]] = None,
+ snr_thresholds: Optional[Dict[str, Any]] = None,
+) -> Tuple[pd.DataFrame, Dict[str, Any]]:
+ """
+ Run all SNR sub-components; returns grid (with transcript columns when computed) and summary dict.
+
+ Writes ``snr_metrics.json`` under outdir when ``write_snr_json`` is True. When ROI transcript
+ SNR succeeds and ``save_roi_tx_table`` is True, also writes ``SNR_roi_tx.parquet`` (or
+ ``.csv.gz`` fallback) and sets ``components[SNR_roi_tx]["per_roi_table_file"]`` to the basename.
+
+ If ``snr_include_moran`` is False (default), neg-control spatial uses quadrant spread only (no PySAL Moran).
+
+ ``roi_grid_stride``: optional ``(stride_x, stride_y)`` matching ``image_qc`` grid construction
+ (same as ``roi_size`` when stride is unset). Enables O(n_tx) transcript-to-ROI assignment
+ without scanning unique x1/y1 on huge grids.
+ """
+ outdir = Path(outdir)
+ outdir.mkdir(parents=True, exist_ok=True)
+ bundle = Path(xenium_bundle_dir) if xenium_bundle_dir else None
+
+ df = df_grid_roi.copy()
+ parts: Dict[str, Dict[str, Any]] = {}
+
+ thresholds = snr_thresholds or {}
+
+ # 1) Image SNR (ROI df quartiles → dB)
+ parts[SNR_CKEY_IMAGE_ROI_QUARTILE_DB] = compute_image_snr_from_roi_df(
+ df, intensity_threshold=intensity_threshold, snr_thresholds=thresholds
+ )
+
+ # 2) Image SNR (Otsu)
+ if focus_maps or roi_snr_db is not None:
+ parts[SNR_CKEY_IMAGE_OTSU] = compute_image_snr_from_pixel_maps(
+ focus_maps,
+ df,
+ max_rois=otsu_max_rois,
+ snr_thresholds=thresholds,
+ precomputed_db=roi_snr_db,
+ )
+ else:
+ parts[SNR_CKEY_IMAGE_OTSU] = {"status": "skipped", "reason": "no focus_maps"}
+
+ # 3) ROI transcript SNR
+ tx_path = transcripts_path
+ if tx_path is None and bundle:
+ cand = bundle / "transcripts.parquet"
+ if cand.exists():
+ tx_path = cand
+ else:
+ cg = bundle / "transcripts.csv.gz"
+ if cg.exists():
+ tx_path = cg
+ if tx_path and tx_path.exists():
+ try:
+ if pixel_size_um is None:
+ parts[SNR_CKEY_ROI_TX] = {
+ "status": "skipped",
+ "reason": "pixel_size_um required to convert transcript coordinates to pixels",
+ }
+ else:
+ # Streamed: see _stream_roi_tx_counts. Reading the table whole and
+ # then copying it in transcripts_um_to_px OOM-killed this step.
+ df, rtx = compute_roi_snr(
+ df,
+ transcripts_path=Path(tx_path),
+ pixel_size_um=pixel_size_um,
+ roi_grid_stride=roi_grid_stride,
+ snr_thresholds=thresholds,
+ )
+ if save_roi_tx_table and rtx.get("status") == "ok":
+ pth = save_snr_roi_tx_table(df, outdir)
+ if pth is not None:
+ rtx["per_roi_table_file"] = pth.name
+ parts[SNR_CKEY_ROI_TX] = rtx
+ except Exception as e:
+ logger.exception("ROI transcript SNR failed")
+ parts[SNR_CKEY_ROI_TX] = {"status": "error", "error": str(e)}
+ else:
+ parts[SNR_CKEY_ROI_TX] = {"status": "skipped", "reason": "no transcripts file"}
+
+ # 4) Slide SNR
+ h5_path = cell_matrix_h5
+ if h5_path is None and bundle:
+ hp = bundle / "cell_feature_matrix.h5"
+ if hp.exists():
+ h5_path = hp
+ if h5_path and Path(h5_path).exists():
+ try:
+ mat, names, _ftype, _meta = load_expression_matrix_h5(Path(h5_path))
+ parts[SNR_CKEY_SLIDE_PLUMMER] = compute_slide_snr_plummer_corrected(
+ mat, names, snr_thresholds=thresholds
+ )
+ parts[SNR_CKEY_SLIDE_SPATIALQM] = compute_slide_snr_spatialqm_corrected(
+ mat, names, snr_thresholds=thresholds
+ )
+ except Exception as e:
+ logger.exception("Slide SNR failed")
+ parts[SNR_CKEY_SLIDE_PLUMMER] = {"status": "error", "error": str(e)}
+ parts[SNR_CKEY_SLIDE_SPATIALQM] = {"status": "error", "error": str(e)}
+ else:
+ parts[SNR_CKEY_SLIDE_PLUMMER] = {
+ "status": "skipped",
+ "reason": "no cell_feature_matrix.h5",
+ }
+ parts[SNR_CKEY_SLIDE_SPATIALQM] = {
+ "status": "skipped",
+ "reason": "no cell_feature_matrix.h5",
+ }
+
+ # 5) Neg spatial autocorrelation (needs neg_pct)
+ if "neg_pct" in df.columns:
+ parts[SNR_CKEY_ROI_NEG_SPATIAL] = compute_neg_spatial_autocorrelation(
+ df, include_moran=snr_include_moran, snr_thresholds=thresholds
+ )
+ else:
+ parts[SNR_CKEY_ROI_NEG_SPATIAL] = {
+ "status": "skipped",
+ "reason": "neg_pct not available",
+ }
+
+ verdict = aggregate_snr_verdict(parts)
+ summary = {
+ "components": parts,
+ "verdict": verdict,
+ }
+
+ if write_snr_json:
+ try:
+ out_json = outdir / "snr_metrics.json"
+ with open(out_json, "w") as f:
+ json.dump(summary, f, indent=2, default=str)
+ logger.info("Wrote %s", out_json)
+ except Exception as e:
+ logger.warning("Could not write snr_metrics.json: %s", e)
+
+ return df, summary
+
+
+def read_xenium_pixel_size_um(bundle_dir: Path) -> Optional[float]:
+ """Read ``pixel_size`` (or ``pixel_size_um``) from ``experiment.xenium``."""
+ exp = Path(bundle_dir) / "experiment.xenium"
+ if not exp.is_file():
+ return None
+ try:
+ with open(exp, encoding="utf-8") as f:
+ meta = json.load(f)
+ ps = float(meta.get("pixel_size", meta.get("pixel_size_um", 0.0)))
+ except (OSError, ValueError, TypeError, json.JSONDecodeError):
+ return None
+ if ps <= 0 or not math.isfinite(ps):
+ return None
+ return ps
+
+
+def compute_snr_summary(
+ df_grid_roi: pd.DataFrame,
+ *,
+ bundle_dir: Path,
+ outdir: Path,
+ focus_maps: Optional[Dict[str, Any]] = None,
+ roi_snr_db: Optional[Any] = None,
+ pixel_size_um: Optional[float] = None,
+ intensity_threshold: float = 0.0,
+ otsu_max_rois: Optional[int] = None,
+ save_roi_tx_table: bool = True,
+ write_snr_json: bool = True,
+ snr_include_moran: bool = False,
+ roi_grid_stride: Optional[Tuple[int, int]] = None,
+ snr_thresholds: Optional[Dict[str, Any]] = None,
+) -> Tuple[pd.DataFrame, Dict[str, Any]]:
+ """High-level API for :mod:`image_qc` — same as :func:`run_snr_module` with keyword-only opts."""
+ return run_snr_module(
+ bundle_dir,
+ df_grid_roi,
+ outdir,
+ focus_maps=focus_maps,
+ roi_snr_db=roi_snr_db,
+ intensity_threshold=intensity_threshold,
+ pixel_size_um=pixel_size_um,
+ otsu_max_rois=otsu_max_rois,
+ save_roi_tx_table=save_roi_tx_table,
+ write_snr_json=write_snr_json,
+ snr_include_moran=snr_include_moran,
+ roi_grid_stride=roi_grid_stride,
+ snr_thresholds=snr_thresholds,
+ )
+
+
+__all__ = [
+ "SNR_CKEY_IMAGE_OTSU",
+ "SNR_CKEY_IMAGE_ROI_QUARTILE_DB",
+ "SNR_CKEY_ROI_NEG_SPATIAL",
+ "SNR_CKEY_ROI_TX",
+ "SNR_CKEY_SLIDE_PLUMMER",
+ "SNR_CKEY_SLIDE_SPATIALQM",
+ "SNR_ROI_TX_TABLE_BASENAME",
+ "aggregate_snr_verdict",
+ "compute_image_snr_from_pixel_maps",
+ "compute_image_snr_from_roi_df",
+ "compute_neg_spatial_autocorrelation",
+ "compute_roi_snr",
+ "compute_snr_summary",
+ "compute_slide_snr_plummer_corrected",
+ "compute_slide_snr_spatialqm_corrected",
+ "is_neg_probe_feature",
+ "load_expression_matrix_h5",
+ "load_transcripts",
+ "read_xenium_pixel_size_um",
+ "run_snr_module",
+ "save_snr_roi_tx_table",
+ "snr_verdict_to_quality_status",
+ "transcripts_um_to_px",
+]
diff --git a/bin/transcript_qc_processing.py b/bin/transcript_qc_processing.py
new file mode 100755
index 00000000..27eaab5f
--- /dev/null
+++ b/bin/transcript_qc_processing.py
@@ -0,0 +1,695 @@
+#!/usr/bin/env python3
+
+"""
+Transcript QC Processing Module
+Performs all analysis from notebooks/1_qc_molecule.ipynb and generates figures and metrics.
+Authors: Malwina Prater, mprater@altoslabs.com; Dongze He, dhe@altoslabs.com; Felix Krueger, fkrueger@altoslabs.com
+"""
+
+import argparse
+import json
+import os
+import sys
+from pathlib import Path
+import numpy as np
+import pandas as pd
+import matplotlib.pyplot as plt
+import seaborn as sns
+import scanpy as sc
+import pyarrow.parquet as pq
+
+
+def calculate_noise_bound(n_molecules_non_gene_prefix, quant: float = 0.99):
+ """Calculate the noise bounds based on the non-gene molecules."""
+ from scipy.stats import median_abs_deviation, norm
+
+ if n_molecules_non_gene_prefix.empty:
+ return 0, 0
+
+ quant_val = norm.ppf(quant)
+ n_mols_log = np.log10(n_molecules_non_gene_prefix.values)
+ std = median_abs_deviation(n_mols_log, scale="normal")
+ noise_lb = np.mean(n_mols_log) - quant_val * std
+ noise_ub = np.mean(n_mols_log) + quant_val * std
+ return 10**noise_lb, 10**noise_ub
+
+
+def estimate_min_mols_per_cell(n_mols_per_cell, min_value: int = 10):
+ n_mols_per_cell = np.log10(np.asarray(n_mols_per_cell) + 1)
+ nm_hist = np.histogram(n_mols_per_cell, bins=100)
+ mode = nm_hist[1][nm_hist[0].argmax()]
+ ci = np.quantile(n_mols_per_cell[n_mols_per_cell > mode], 0.99) - mode
+ return max(min_value, int(round(10 ** (mode - ci))))
+
+
+# Set plotting style
+sns.set_theme(style="whitegrid")
+plt.rcParams["figure.dpi"] = 300
+
+
+def write_versions(outdir: Path, task_process: str) -> None:
+ """Write versions.yml into the output dir for the transcript QC report to display.
+
+ This is separate from the module's topic-channel version reporting: it records
+ the analysis package versions for the rendered HTML report. A missing package
+ raises (importlib.metadata.version), so a broken environment surfaces loudly.
+ """
+ from importlib.metadata import version
+ import platform
+
+ packages = [
+ "matplotlib",
+ "numpy",
+ "pandas",
+ "scipy",
+ "seaborn",
+ "scanpy",
+ "pyarrow",
+ ]
+ lines = [f'"{task_process}":']
+ for pkg in packages:
+ lines.append(f" {pkg}: {version(pkg)}")
+ lines.append(f" python: {platform.python_version()}")
+ (outdir / "versions.yml").write_text("\n".join(lines) + "\n")
+
+
+def read_random_parquet_row_groups(parquet_file, num_row_groups=4, random_seed=42):
+ """
+ Read a random subset of row groups from a parquet file. Each row group is a set of rows that are contiguous in the file. For 10x transcripts.parquet file, each row group has about 262,000 rows.
+ """
+ np.random.seed(random_seed)
+ selected_row_groups = np.random.choice(
+ parquet_file.metadata.num_row_groups, size=num_row_groups, replace=False
+ )
+ return parquet_file.read_row_groups(selected_row_groups).to_pandas()
+
+
+def main():
+ parser = argparse.ArgumentParser(description="Transcript QC Processing")
+ parser.add_argument(
+ "--xenium-bundle-dir", required=True, help="Path to Xenium bundle directory"
+ )
+ parser.add_argument("--outdir", required=True, help="Output directory")
+ parser.add_argument(
+ "--non-gene-prefix",
+ default="NegControlProbe",
+ help="Prefix for non-gene features",
+ )
+ parser.add_argument(
+ "--stain-names", help="Stain names (unused but kept for compatibility)"
+ )
+ parser.add_argument(
+ "--task-process", default="TRANSCRIPT_QC", help="Task process name"
+ )
+ parser.add_argument(
+ "--num-row-groups",
+ type=int,
+ default=None,
+ help="Number of row groups to process",
+ )
+ parser.add_argument(
+ "--threads", type=int, default=1, help="Number of threads (for compatibility)"
+ )
+
+ args = parser.parse_args()
+
+ # Validate parameters
+ if args.xenium_bundle_dir is None or not os.path.exists(args.xenium_bundle_dir):
+ raise FileNotFoundError(
+ f'The given XENIUM_BUNDLE_DIR, "{args.xenium_bundle_dir}" doesn\'t exist'
+ )
+
+ XENIUM_BUNDLE_DIR = Path(args.xenium_bundle_dir)
+ # Create output directories
+ outdir = Path(args.outdir)
+ outdir.mkdir(parents=True, exist_ok=True)
+ output_fig_dir = outdir / "figures"
+ output_fig_dir.mkdir(parents=True, exist_ok=True)
+ figures_source_dir = outdir / "figures_source"
+ figures_source_dir.mkdir(parents=True, exist_ok=True)
+ output_metrics_path = outdir / "transcript_qc_metrics.json"
+
+ NUM_ROW_GROUPS = args.num_row_groups
+
+ transcripts_parquet_path = XENIUM_BUNDLE_DIR / "transcripts.parquet"
+ morphology_focus_dir = XENIUM_BUNDLE_DIR / "morphology_focus"
+ cells_parquet_path = XENIUM_BUNDLE_DIR / "cells.parquet"
+ cell_feature_matrix_h5_path = XENIUM_BUNDLE_DIR / "cell_feature_matrix.h5"
+
+ # EXACT CODE FROM ORIGINAL NOTEBOOK - check if all required files are present
+ required_files = [
+ transcripts_parquet_path,
+ morphology_focus_dir,
+ cell_feature_matrix_h5_path,
+ cells_parquet_path,
+ ]
+ for file in required_files:
+ if not os.path.exists(file):
+ print(f"Required file not found: {file}")
+ sys.exit(1)
+
+ print("=== Transcript QC Processing ===")
+ print(f"Input directory: {XENIUM_BUNDLE_DIR}")
+ print(f"Output directory: {outdir}")
+
+ # EXACT CODE FROM ORIGINAL NOTEBOOK - check if all required columns are present in the transcripts parquet file
+ transcripts_parquet = pq.ParquetFile(transcripts_parquet_path)
+ transcripts_parquet_columns = transcripts_parquet.schema.names
+ num_molecules = transcripts_parquet.metadata.num_rows
+ print("Total number of molecules: {:,}".format(num_molecules))
+
+ # EXACT CODE FROM ORIGINAL NOTEBOOK - required columns
+ required_columns = [
+ "cell_id",
+ "qv",
+ "fov_name",
+ "codeword_category",
+ "is_gene",
+ "feature_name",
+ ]
+ missing_columns = [
+ col for col in required_columns if col not in transcripts_parquet_columns
+ ]
+ if missing_columns:
+ print(f"Missing required columns in transcripts.parquet: {missing_columns}")
+ sys.exit(1)
+
+ # EXACT CODE FROM ORIGINAL NOTEBOOK - load data
+ if NUM_ROW_GROUPS is None:
+ df_spatial = pd.read_parquet(
+ transcripts_parquet_path,
+ columns=required_columns,
+ )
+ elif NUM_ROW_GROUPS > transcripts_parquet.metadata.num_row_groups:
+ print(
+ f"NUM_ROW_GROUPS ({NUM_ROW_GROUPS}) is greater than the number of molecules in the parquet file ({transcripts_parquet.metadata.num_row_groups}). Read the entire file."
+ )
+ NUM_ROW_GROUPS = None
+ df_spatial = pd.read_parquet(transcripts_parquet_path, columns=required_columns)
+ else:
+ df_spatial = read_random_parquet_row_groups(transcripts_parquet, NUM_ROW_GROUPS)
+
+ num_selected_molecules = df_spatial.shape[0]
+
+ if num_selected_molecules != num_molecules:
+ print(
+ f"Number of random molecules selected for analysis: {num_selected_molecules:,} (out of {num_molecules:,})"
+ )
+
+ codeword_category_counts = df_spatial["codeword_category"].value_counts()
+
+ print(f"Features: {df_spatial.feature_name.unique().size:,}")
+ print("\nFeature categories:")
+ for cc in codeword_category_counts.sort_values(ascending=False).keys():
+ count = codeword_category_counts[cc]
+ percentage = count / num_selected_molecules * 100
+ print(f" {cc:<26} - {count:>12,} molecules ({percentage:>6.3f}%)")
+
+ # EXACT CODE FROM ORIGINAL NOTEBOOK - quality distribution plots
+ print("\nGenerating quality distribution plots...")
+
+ # Filter data for quality analysis
+ num_gene_molecules = df_spatial["is_gene"].sum()
+ df_spatial_nongene = df_spatial.query("is_gene == False").sample(
+ min(1000000, num_selected_molecules - num_gene_molecules), random_state=42
+ )
+ df_spatial_gene = df_spatial.query("is_gene == True").sample(
+ min(1000000, num_gene_molecules), random_state=42
+ )
+
+ # define hue order: genes first, then non-genes
+ codeword_categories_order = codeword_category_counts.index
+ codeword_categories_order = (
+ codeword_categories_order[codeword_categories_order.str.endswith("gene")]
+ .sort_values(ascending=False)
+ .tolist()
+ + codeword_categories_order[~codeword_categories_order.str.endswith("gene")]
+ .sort_values()
+ .tolist()
+ )
+
+ df_spatial_quality = pd.concat([df_spatial_nongene, df_spatial_gene])
+
+ print(
+ f"Using {len(df_spatial_nongene):,} non-gene molecules and {len(df_spatial_gene):,} gene molecules for quality values (qv) distribution"
+ )
+
+ # Print total number of rows and rows per category in df_spatial_quality
+ print("\nmolecule per category:")
+ print(df_spatial_quality["codeword_category"].value_counts())
+
+ # EXACT CODE FROM ORIGINAL NOTEBOOK - Figure 1: Quality distribution density plot with overlapping curves
+ fig = plt.figure(figsize=(15, 8))
+
+ # Create density plots with all curves in the same plot
+ for i, category in enumerate(codeword_categories_order):
+ # Filter data for this category
+ category_data = df_spatial_quality[
+ df_spatial_quality["codeword_category"] == category
+ ]["qv"]
+
+ # Create density plot for this category
+ sns.kdeplot(x=category_data, fill=True, alpha=0.5, linewidth=2, label=category)
+
+ plt.xlabel("Quality Value (qv)", fontsize=12)
+ plt.ylabel("Density", fontsize=12)
+ plt.title(
+ "Distribution of Transcript Quality by Codeword Category", fontsize=14, pad=20
+ )
+
+ # Add vertical line at qv=20
+ plt.axvline(
+ x=20, color="darkred", linestyle="--", linewidth=2, label="QV threshold = 20"
+ )
+
+ # Add legend
+ plt.legend(title="Codeword Category", bbox_to_anchor=(1.05, 1), loc="upper left")
+ plt.tight_layout()
+ plt.savefig(
+ output_fig_dir / "quality_distribution_density.pdf",
+ dpi=300,
+ bbox_inches="tight",
+ )
+ plt.savefig(
+ output_fig_dir / "quality_distribution_density.png",
+ dpi=300,
+ bbox_inches="tight",
+ )
+ plt.close(fig)
+
+ # Save data for quality distribution density plot
+ df_spatial_quality.to_csv(
+ figures_source_dir / "quality_distribution_density.csv", index=False
+ )
+
+ # EXACT CODE FROM ORIGINAL NOTEBOOK - Figure 2: Quality distribution by codeword category (violin plot)
+ fig = plt.figure(figsize=(15, 8))
+ sns.violinplot(
+ data=df_spatial_quality,
+ x="codeword_category",
+ y="qv",
+ hue="codeword_category",
+ split=False,
+ inner="box",
+ palette="husl",
+ density_norm="width",
+ legend=False,
+ order=codeword_categories_order,
+ )
+ plt.xlabel("Codeword Category", fontsize=12)
+ plt.ylabel("Quality Value (qv)", fontsize=12)
+ plt.title(
+ "Distribution of Transcript Quality by Codeword Category", fontsize=14, pad=20
+ )
+ plt.xticks(rotation=45, ha="right")
+ plt.tight_layout()
+ # add a horizontal line at qv=20
+ plt.axhline(y=20, color="grey", linestyle="--")
+ plt.savefig(
+ output_fig_dir / "quality_distributions_comprehensive.pdf",
+ dpi=300,
+ bbox_inches="tight",
+ )
+ plt.savefig(
+ output_fig_dir / "quality_distributions_comprehensive.png",
+ dpi=300,
+ bbox_inches="tight",
+ )
+ plt.close(fig)
+
+ # Save data for quality distributions comprehensive
+ df_spatial_quality.to_csv(
+ figures_source_dir / "quality_distributions_comprehensive.csv", index=False
+ )
+
+ # Save df_spatial_quality to CSV
+ output_file = outdir / "df_spatial_quality.csv"
+ df_spatial_quality.to_csv(output_file, index=False)
+ print(f"Saved df_spatial_quality to: {output_file}")
+
+ # EXACT CODE FROM ORIGINAL NOTEBOOK - Figure 3: Quality distribution by Field of View
+ fig = plt.figure(figsize=(15, 8))
+ sns.violinplot(
+ data=df_spatial_gene,
+ x="fov_name",
+ y="qv",
+ hue="codeword_category",
+ split=False,
+ inner="box",
+ palette="husl",
+ density_norm="width",
+ legend=False,
+ )
+ plt.xlabel("Field of View", fontsize=12)
+ plt.ylabel("Quality Value (qv)", fontsize=12)
+ plt.title(
+ "Distribution of Transcript Quality by Field of View", fontsize=14, pad=20
+ )
+ plt.xticks(rotation=45, ha="right")
+ plt.tight_layout()
+ # add a horizontal line at qv=20
+ plt.axhline(y=20, color="grey", linestyle="--")
+ plt.savefig(output_fig_dir / "quality_by_fov.pdf", dpi=300, bbox_inches="tight")
+ plt.savefig(output_fig_dir / "quality_by_fov.png", dpi=300, bbox_inches="tight")
+ plt.close(fig)
+
+ # Save data for quality by fov
+ df_spatial_gene.to_csv(figures_source_dir / "quality_by_fov.csv", index=False)
+
+ del df_spatial_quality, df_spatial_gene
+
+ # EXACT CODE FROM ORIGINAL NOTEBOOK - Calculate noise threshold
+ _, n_mols_threshold = calculate_noise_bound(
+ df_spatial_nongene["feature_name"][
+ df_spatial_nongene["feature_name"].str.startswith("NegControl")
+ ].value_counts()
+ )
+ n_mols_threshold = n_mols_threshold * (num_molecules / num_selected_molecules)
+ print(
+ f"Noise threshold for genes' molecule count: {n_mols_threshold:.0f} molecules"
+ )
+
+ # EXACT CODE FROM ORIGINAL NOTEBOOK - group by feature_name
+ n_mols_per_gene_df = df_spatial.groupby("feature_name").agg(
+ # count of molecules per gene
+ n_molecules=("feature_name", "count"),
+ is_gene=("is_gene", "first"),
+ )
+ # rescale the number of molecules to account for subsampling
+ n_mols_per_gene_df["n_molecules"] = n_mols_per_gene_df["n_molecules"] * (
+ num_molecules / num_selected_molecules
+ )
+
+ # EXACT CODE FROM ORIGINAL NOTEBOOK - Figure 4: Distribution of molecules per feature
+ fig = plt.figure(figsize=(8, 4))
+ sns.histplot(
+ n_mols_per_gene_df,
+ x="n_molecules",
+ multiple="layer", # Overlap instead of stack
+ hue="is_gene",
+ log_scale=True,
+ bins=50,
+ ax=plt.gca(),
+ element="step", # Outlined bars
+ fill=False, # No fill, just lines
+ )
+ plt.xlabel("Num. molecules")
+ plt.ylabel("Num. features")
+ plt.axvline(x=n_mols_threshold, color="grey", linestyle="--")
+ plt.title("Distribution of molecules per feature", fontsize=14, pad=20)
+ plt.tight_layout()
+ plt.savefig(
+ f"{output_fig_dir}/num_transcripts_per_feature.pdf",
+ dpi=300,
+ bbox_inches="tight",
+ )
+ plt.savefig(
+ f"{output_fig_dir}/num_transcripts_per_feature.png",
+ dpi=300,
+ bbox_inches="tight",
+ )
+ plt.close(fig)
+
+ # Save data for molecules per feature
+ df_molecules_per_feature = n_mols_per_gene_df.copy()
+ df_molecules_per_feature["n_mols_threshold"] = n_mols_threshold
+ df_molecules_per_feature.to_csv(
+ figures_source_dir / "num_transcripts_per_feature.csv", index=True
+ )
+
+ retained_genes = n_mols_per_gene_df.query(
+ "n_molecules > @n_mols_threshold and is_gene == True"
+ )
+ print(
+ f"Number of genes with a total molecule count higher than the threshold : {len(retained_genes):,}"
+ )
+ retained_genes.to_csv(outdir / "retained_genes.csv", index=True)
+
+ # EXACT CODE FROM ORIGINAL NOTEBOOK - Cell size distribution
+ print("\nGenerating cell statistics plots...")
+
+ # Check available columns and use appropriate cell area column
+ available_columns = pq.ParquetFile(cells_parquet_path).schema.names
+ cell_area_column = "cell_area" if "cell_area" in available_columns else "volume"
+ cells_parquet = pd.read_parquet(
+ cells_parquet_path, columns=["cell_id", cell_area_column]
+ )
+ # Rename the column for consistency
+ cells_parquet.rename(columns={cell_area_column: "cell_size"}, inplace=True)
+
+ # EXACT CODE FROM ORIGINAL NOTEBOOK - Figure 5: Cell size distribution
+ fig = plt.figure(figsize=(8, 6))
+ sns.kdeplot(
+ data=cells_parquet, x="cell_size", fill=True, color="skyblue", alpha=0.5
+ )
+ plt.xlabel("Genes per cell")
+ plt.ylabel("Density")
+ plt.title("Cell size distribution")
+
+ # Add vertical lines for mean and median
+ plt.axvline(
+ cells_parquet["cell_size"].mean(), color="red", linestyle="--", label="Mean"
+ )
+ plt.axvline(
+ cells_parquet["cell_size"].median(),
+ color="green",
+ linestyle="--",
+ label="Median",
+ )
+ plt.legend()
+
+ # Save the plot
+ plt.savefig(
+ os.path.join(outdir, "figures", "genes_per_cell_distribution.pdf"),
+ dpi=300,
+ bbox_inches="tight",
+ )
+ plt.savefig(
+ os.path.join(outdir, "figures", "genes_per_cell_distribution.png"),
+ dpi=300,
+ bbox_inches="tight",
+ )
+ plt.close(fig)
+
+ # Save data for cell size distribution
+ df_cell_size = cells_parquet[["cell_id", "cell_size"]].copy()
+ df_cell_size["mean_cell_size"] = cells_parquet["cell_size"].mean()
+ df_cell_size["median_cell_size"] = cells_parquet["cell_size"].median()
+ df_cell_size.to_csv(
+ figures_source_dir / "genes_per_cell_distribution.csv", index=False
+ )
+ del cells_parquet
+
+ # Figure 6: Nucleus RNA fraction per cell
+ cells_parquet = pd.read_parquet(
+ cells_parquet_path,
+ columns=["nucleus_count", "total_counts"],
+ filters=[("total_counts", ">", 0)],
+ )
+ nucleus_count_fraction = cells_parquet["nucleus_count"] / (
+ cells_parquet["total_counts"] + 1
+ )
+
+ # Create density plot if requested
+ fig = plt.figure(figsize=(8, 6))
+ sns.kdeplot(x=nucleus_count_fraction, fill=True, color="skyblue", alpha=0.5)
+ plt.xlabel("Nucleus molecule fraction per cell")
+ plt.ylabel("Density")
+ # if nucleus_count_fraction is all 0, set the title to "nucleus molecule fraction distribution"
+ plt.title(
+ "no nucleus molecule detected; Empty plot shown"
+ if nucleus_count_fraction.all() == 0
+ else "nucleus molecule fraction distribution"
+ )
+
+ # Save the plot
+ plt.savefig(
+ os.path.join(
+ outdir, "figures", "nucleus_transcript_fraction_per_cell_distribution.pdf"
+ ),
+ dpi=300,
+ bbox_inches="tight",
+ )
+ plt.savefig(
+ os.path.join(
+ outdir, "figures", "nucleus_transcript_fraction_per_cell_distribution.png"
+ ),
+ dpi=300,
+ bbox_inches="tight",
+ )
+ plt.close(fig)
+
+ # Save data for nucleus molecule fraction
+ df_nucleus_fraction = pd.DataFrame(
+ {
+ "nucleus_count": cells_parquet["nucleus_count"],
+ "total_counts": cells_parquet["total_counts"],
+ "nucleus_fraction": nucleus_count_fraction,
+ }
+ )
+ df_nucleus_fraction.to_csv(
+ figures_source_dir / "nucleus_transcript_fraction_per_cell_distribution.csv",
+ index=False,
+ )
+ del cells_parquet
+
+ # EXACT CODE FROM ORIGINAL NOTEBOOK - Figure 7: Nucleus-to-cell area fraction
+ cells_parquet = pd.read_parquet(
+ cells_parquet_path, columns=["nucleus_area", "cell_area"]
+ )
+ nucleus_size_fraction = cells_parquet["nucleus_area"] / (
+ cells_parquet["cell_area"] + 1
+ )
+
+ # Create density plot if requested
+ fig = plt.figure(figsize=(8, 6))
+ sns.kdeplot(x=nucleus_size_fraction, fill=True, color="skyblue", alpha=0.5)
+ plt.xlabel("Nucleus to cell fraction")
+ plt.ylabel("Density")
+ plt.title(
+ "nucleus to cell fraction distribution"
+ if nucleus_size_fraction.all() != 0
+ else "no nucleus molecule detected; Empty plot shown"
+ )
+
+ # Save the plot
+ plt.savefig(
+ os.path.join(
+ outdir, "figures", "nucleus_to_cell_size_fraction_per_cell_distribution.pdf"
+ ),
+ dpi=300,
+ bbox_inches="tight",
+ )
+ plt.savefig(
+ os.path.join(
+ outdir, "figures", "nucleus_to_cell_size_fraction_per_cell_distribution.png"
+ ),
+ dpi=300,
+ bbox_inches="tight",
+ )
+ plt.close(fig)
+
+ # Save data for nucleus to cell size fraction
+ df_nucleus_size_fraction = pd.DataFrame(
+ {
+ "nucleus_area": cells_parquet["nucleus_area"],
+ "cell_area": cells_parquet["cell_area"],
+ "nucleus_size_fraction": nucleus_size_fraction,
+ }
+ )
+ df_nucleus_size_fraction.to_csv(
+ figures_source_dir / "nucleus_to_cell_size_fraction_per_cell_distribution.csv",
+ index=False,
+ )
+ del cells_parquet
+
+ # EXACT CODE FROM ORIGINAL NOTEBOOK - Load cell feature matrix
+ ad = sc.read_10x_h5(cell_feature_matrix_h5_path)
+ # filter for retained genes
+ ad = ad[:, ad.var_names.isin(retained_genes.index)]
+ print(f"AnnData object with n_obs × n_vars = {ad.shape[0]} × {ad.shape[1]}")
+
+ # EXACT CODE FROM ORIGINAL NOTEBOOK - Figure 8: Distribution of molecules per cell
+ n_mols_per_cell = ad.X.sum(axis=1).A1
+ n_mols_threshold_cell = estimate_min_mols_per_cell(n_mols_per_cell)
+
+ fig = plt.figure(figsize=(8, 4))
+ sns.histplot(n_mols_per_cell, log_scale=True, bins=50, ax=plt.gca())
+ plt.xlabel("Num. molecules")
+ plt.ylabel("Num. cells")
+ plt.axvline(x=n_mols_threshold_cell, color="grey", linestyle="--")
+ plt.title("Distribution of molecules per Cell", fontsize=14, pad=20)
+ plt.tight_layout()
+ plt.savefig(
+ f"{output_fig_dir}/num_transcripts_per_cell.pdf", dpi=300, bbox_inches="tight"
+ )
+ plt.savefig(
+ f"{output_fig_dir}/num_transcripts_per_cell.png", dpi=300, bbox_inches="tight"
+ )
+ plt.close(fig)
+ print(f"Threshold for molecules per cell: {n_mols_threshold_cell}")
+
+ # Save data for molecules per cell
+ df_molecules_per_cell = pd.DataFrame(
+ {
+ "n_molecules_per_cell": n_mols_per_cell,
+ "n_mols_threshold_cell": n_mols_threshold_cell,
+ }
+ )
+ df_molecules_per_cell.to_csv(
+ figures_source_dir / "num_transcripts_per_cell.csv", index=False
+ )
+
+ # Convert numpy array to pandas DataFrame
+ n_mols_per_cell_df = pd.DataFrame(n_mols_per_cell, columns=["num_of_molecules"])
+ # Save to CSV using os.path.join for path handling
+ output_file = os.path.join(outdir, "num_transcripts_per_cell.csv")
+ n_mols_per_cell_df.to_csv(output_file, index=False)
+ print(f"Saved transcript distribution to: {output_file}")
+
+ # EXACT CODE FROM ORIGINAL NOTEBOOK - Figure 9: Distribution of genes per cell
+ n_genes_per_cell = (ad.X != 0).sum(axis=1).A1
+ n_genes_threshold = estimate_min_mols_per_cell(n_genes_per_cell)
+
+ fig = plt.figure(figsize=(8, 4))
+ sns.histplot(n_genes_per_cell, log_scale=True, bins=50, ax=plt.gca())
+ plt.xlabel("Num. genes")
+ plt.ylabel("Num. cells")
+ plt.axvline(x=n_genes_threshold, color="grey", linestyle="--")
+ plt.title("Distribution of number of detected genes per Cell", fontsize=14, pad=20)
+ plt.tight_layout()
+ plt.savefig(
+ f"{output_fig_dir}/num_genes_per_cell.pdf", dpi=300, bbox_inches="tight"
+ )
+ plt.savefig(
+ f"{output_fig_dir}/num_genes_per_cell.png", dpi=300, bbox_inches="tight"
+ )
+ plt.close(fig)
+
+ # Save data for genes per cell
+ df_genes_per_cell = pd.DataFrame(
+ {"n_genes_per_cell": n_genes_per_cell, "n_genes_threshold": n_genes_threshold}
+ )
+ df_genes_per_cell.to_csv(figures_source_dir / "num_genes_per_cell.csv", index=False)
+
+ # Convert numpy array to pandas DataFrame
+ n_genes_per_cell_df = pd.DataFrame(n_genes_per_cell, columns=["num_of_genes"])
+ # Save to CSV using os.path.join for path handling
+ output_file = os.path.join(outdir, "num_genes_per_cell.csv")
+ n_genes_per_cell_df.to_csv(output_file, index=False)
+ print(f"Saved gene distribution to: {output_file}")
+
+ # EXACT CODE FROM ORIGINAL NOTEBOOK - Save metrics
+ metrics = {
+ "total_transcripts": int(num_molecules),
+ "selected_transcripts": int(num_selected_molecules),
+ "total_features": int(df_spatial.feature_name.nunique()),
+ "codeword_category_counts": {
+ str(k): int(v) for k, v in codeword_category_counts.items()
+ },
+ "neg_control_quantile": int(n_mols_threshold),
+ "min_transcripts_per_cell": int(n_mols_threshold_cell),
+ "min_genes_per_cell": int(n_genes_threshold),
+ "retained_genes_count": len(retained_genes),
+ "total_cells": int(ad.shape[0]),
+ "analyzed_genes": int(ad.shape[1]),
+ }
+
+ # Save metrics
+ with open(output_metrics_path, "w") as f:
+ json.dump(metrics, f, indent=2)
+
+ # Save versions.yml — read and displayed by the transcript QC report
+ write_versions(outdir, args.task_process)
+
+ print("\n=== Processing Complete ===")
+ print(f"Generated {len(list(output_fig_dir.glob('*.pdf')))} figures")
+ print(f"Generated {len(list(figures_source_dir.glob('*.csv')))} CSV data files")
+ print(f"Saved metrics to: {output_metrics_path}")
+ print(f"Output directory: {outdir}")
+ print(f"Figures directory: {output_fig_dir}")
+ print(f"Source data directory: {figures_source_dir}")
+
+
+if __name__ == "__main__":
+ main()
diff --git a/bin/utility_downscale_morphology.py b/bin/utility_downscale_morphology.py
index 8544ecf3..f1f41568 100755
--- a/bin/utility_downscale_morphology.py
+++ b/bin/utility_downscale_morphology.py
@@ -42,6 +42,7 @@ def downscale_image(
print(f"Original: {img.shape}, dtype={img.dtype}, ndim={img.ndim}")
# Handle multichannel OME-TIFFs: shape can be (H, W), (C, H, W), or (Z, C, H, W)
+ output_shape: tuple[int, ...]
if img.ndim == 2:
orig_h, orig_w = img.shape
new_h = max(int(orig_h * scale), MIN_DIM)
@@ -87,8 +88,12 @@ def parse_args() -> argparse.Namespace:
description="Pre-downscale a morphology image for Cellpose."
)
parser.add_argument("--image", required=True, help="Morphology TIFF input")
- parser.add_argument("--diameter", type=float, required=True, help="Target object diameter")
- parser.add_argument("--diam-mean", type=float, required=True, help="Cellpose model diam_mean")
+ parser.add_argument(
+ "--diameter", type=float, required=True, help="Target object diameter"
+ )
+ parser.add_argument(
+ "--diam-mean", type=float, required=True, help="Cellpose model diam_mean"
+ )
parser.add_argument("--prefix", required=True, help="Output directory")
return parser.parse_args()
diff --git a/bin/utility_upscale_mask.py b/bin/utility_upscale_mask.py
index 6cc1694e..e38ed2a3 100755
--- a/bin/utility_upscale_mask.py
+++ b/bin/utility_upscale_mask.py
@@ -39,7 +39,7 @@ def upscale_mask(mask_path: str, scale_info_path: str, prefix: str) -> None:
print(f"Upscaling to ({orig_h}, {orig_w})")
pil_mask = Image.fromarray(mask)
- pil_mask = pil_mask.resize((orig_w, orig_h), Image.NEAREST)
+ pil_mask = pil_mask.resize((orig_w, orig_h), Image.Resampling.NEAREST)
mask_up = np.array(pil_mask, dtype=mask.dtype)
out_dir = Path(prefix)
@@ -47,9 +47,7 @@ def upscale_mask(mask_path: str, scale_info_path: str, prefix: str) -> None:
base = Path(mask_path).stem
out_name = out_dir / f"upscaled_{base}.tif"
tifffile.imwrite(str(out_name), mask_up, compression="zlib")
- print(
- f"Done: {out_name}, unique cells: {len(np.unique(mask_up)) - 1}"
- )
+ print(f"Done: {out_name}, unique cells: {len(np.unique(mask_up)) - 1}")
def parse_args() -> argparse.Namespace:
@@ -58,7 +56,9 @@ def parse_args() -> argparse.Namespace:
description="Upscale a Cellpose mask back to original resolution."
)
parser.add_argument("--mask", required=True, help="Downscaled mask TIFF")
- parser.add_argument("--scale-info", required=True, help="scale_info.json from downscale step")
+ parser.add_argument(
+ "--scale-info", required=True, help="scale_info.json from downscale step"
+ )
parser.add_argument("--prefix", required=True, help="Output directory")
return parser.parse_args()
diff --git a/bin/xenium_patch_stitch_postprocess.py b/bin/xenium_patch_stitch_postprocess.py
index 7144b1ac..5fde995a 100755
--- a/bin/xenium_patch_stitch_postprocess.py
+++ b/bin/xenium_patch_stitch_postprocess.py
@@ -65,7 +65,7 @@ def reassign_dropped(csv_path: str, dropped_cells: set) -> None:
with open(csv_path) as f:
reader = csv.DictReader(f)
- fieldnames = reader.fieldnames
+ fieldnames = reader.fieldnames or []
rows = list(reader)
reassigned = 0
@@ -87,8 +87,12 @@ def parse_args() -> argparse.Namespace:
parser = argparse.ArgumentParser(
description="Clean stitched GeoJSON polygons and reconcile transcript CSV."
)
- parser.add_argument("--geojson", required=True, help="Path to xr-cell-polygons.geojson")
- parser.add_argument("--csv", required=True, help="Path to xr-transcript-metadata.csv")
+ parser.add_argument(
+ "--geojson", required=True, help="Path to xr-cell-polygons.geojson"
+ )
+ parser.add_argument(
+ "--csv", required=True, help="Path to xr-transcript-metadata.csv"
+ )
return parser.parse_args()
diff --git a/conf/base.config b/conf/base.config
index 476c4dbe..97961829 100644
--- a/conf/base.config
+++ b/conf/base.config
@@ -100,4 +100,15 @@ process {
time = { 16.h * task.attempt }
}
+ // Image QC analysis: GPU-optional. `params.image_qc_gpus` sets how many GPUs
+ // each task requests; the accelerator is gated on `params.use_gpu` so non-GPU
+ // runs request none (image_qc.py then runs its CPU path). The script's own
+ // --max-gpus (from params.image_qc_gpus) caps the devices CUDA actually uses.
+ withLabel:process_gpu_qc {
+ accelerator = { params.use_gpu ? (params.image_qc_gpus as int) : null }
+ cpus = { 30 * task.attempt }
+ memory = { 180.GB * task.attempt }
+ time = { 8.h * task.attempt }
+ }
+
}
diff --git a/conf/modules.config b/conf/modules.config
index 81cf6a25..beb32487 100644
--- a/conf/modules.config
+++ b/conf/modules.config
@@ -378,4 +378,61 @@ process {
mode: params.publish_dir_mode,
]
}
+
+ // ---------------------------- image + transcript QC -----------------------
+
+ withName: '.*IMAGE_QC:ANALYSIS' {
+ ext.prefix = 'image_qc'
+ // The re-ported module reads params.* directly (faithful to upstream), so no
+ // ext.args flag-building here — that would duplicate the flags the module builds.
+ // output dir is already named 'image_qc' (ext.prefix), so publish into qc/ (not qc/image_qc) to avoid qc/image_qc/image_qc
+ publishDir = [
+ path: { "${params.outdir}/${params.mode}/qc" },
+ mode: params.publish_dir_mode,
+ ]
+ }
+
+ withName: '.*IMAGE_QC:REPORT' {
+ // The stock QUARTONOTEBOOK container lacks the report's Python deps
+ // (pandas); render in the image QC container, which carries them.
+ container = 'quay.io/dongzehe/image_qc:1.0.0'
+ ext.prefix = 'image_qc'
+ // QUARTONOTEBOOK's process_low (12 GB) OOM-kills quarto's second pandoc
+ // pass (--embed-resources re-reads the whole rendered HTML) on
+ // production-size reports (~60 MB with full-res figures). The child is
+ // SIGKILLed and quarto exits 1 with an empty stderr. Size for the
+ // report, not the input.
+ cpus = { 2 * task.attempt }
+ memory = { 42.GB * task.attempt }
+ publishDir = [
+ path: { "${params.outdir}/${params.mode}/qc/image_qc" },
+ mode: params.publish_dir_mode,
+ pattern: '*.html',
+ ]
+ }
+
+ withName: '.*TRANSCRIPT_QC:ANALYSIS' {
+ ext.prefix = 'transcript_qc'
+ // output dir is already named 'transcript_qc' (ext.prefix), so publish into qc/ (not qc/transcript_qc)
+ publishDir = [
+ path: { "${params.outdir}/${params.mode}/qc" },
+ mode: params.publish_dir_mode,
+ ]
+ }
+
+ withName: '.*TRANSCRIPT_QC:REPORT' {
+ // The stock QUARTONOTEBOOK container lacks the report's Python deps
+ // (pandas); render in the transcript QC container, which carries them.
+ container = 'quay.io/dongzehe/transcript_qc:1.0.0'
+ ext.prefix = 'transcript_qc'
+ // Same OOM headroom as IMAGE_QC:REPORT — quarto's --embed-resources
+ // pass scales memory with report size, not input size.
+ cpus = { 2 * task.attempt }
+ memory = { 42.GB * task.attempt }
+ publishDir = [
+ path: { "${params.outdir}/${params.mode}/qc/transcript_qc" },
+ mode: params.publish_dir_mode,
+ pattern: '*.html',
+ ]
+ }
}
diff --git a/conf/roi_image_qc_thresholds.yaml b/conf/roi_image_qc_thresholds.yaml
new file mode 100644
index 00000000..107326d8
--- /dev/null
+++ b/conf/roi_image_qc_thresholds.yaml
@@ -0,0 +1,188 @@
+image_qc:
+ tile_size_um: 100
+ # Gaussian sigma for Laplacian of Gaussian (LoG) pre-smoothing.
+ # Controls the spatial scale of edges detected; suppresses noise below that scale.
+ # Higher values detect coarser edges and are more tolerant of noise; lower values
+ # are more sensitive to fine detail but noisier. Stored in roi_qc_metrics.json
+ # for calibration tracking.
+ lap_sigma: 1.0
+ # --- Tile pre-filtering thresholds (these three are NOT redundant) ---
+ # 1) TISSUE GATE: Minimum mean DAPI intensity (16-bit, 0-65535) for a tile to
+ # be treated as tissue in blur/focus classification. Tiles below this are
+ # auto-marked as background and excluded from GMM fitting, intensity QC, and
+ # SNR calculations. Prevents empty-slide / coverslip tiles from contaminating
+ # focus statistics.
+ roi_intensity_threshold: 100.0
+ # 2) COVERAGE GATE: Minimum fraction of a tile that must overlap the binary
+ # tissue mask to be included in per-channel intensity assessments (DAPI,
+ # boundary, intRNA warn/fail percentages). Partial-tissue tiles (e.g. at
+ # tissue edges) are excluded so they do not artificially inflate the "dim
+ # tile" fraction. Does NOT affect blur/focus classification (which uses the
+ # intensity gate above).
+ min_tissue_coverage_for_intensity_qc: 0.5
+ # 3) GMM FALLBACK: Percentile of raw focus scores (after excluding
+ # low-intensity tiles) used as the blur/focus separation threshold ONLY when
+ # GMM fitting fails (too few tiles, singular covariance, convergence failure).
+ # In normal operation the GMM posterior probability (blur_prob_threshold
+ # below) is used instead.
+ roi_focus_score_percentile: 5.0
+ # GMM posterior probability threshold: tile classified as blurred when
+ # P(blur_component) > this value. This is the primary blur classification
+ # gate used during normal GMM operation.
+ blur_prob_threshold: 0.5
+ channels:
+ DAPI:
+ # Absolute minimum tile intensity (16-bit scale) below which the channel is
+ # considered too dim for reliable QC. Used by assess_intensity_quality().
+ intensity_critical: 500
+ # Fraction of tissue tiles (coverage >= 0.5) below intensity_critical that
+ # triggers WARN / FAIL. Calibrated on 11 samples (2026-04-02):
+ # good (lung_high, liver_303): 32-33% → PASS
+ # moderate (liver_37, pancreas_254): 44-53% → WARN
+ # bad (pancreas_257, lung_low, brain): 55-84% → FAIL/CRITICAL
+ # Brain/neural tissue caveat: large diffuse nuclei produce inherently dim
+ # DAPI tiles (both brain samples score 71-75% regardless of actual quality).
+ # DAPI intensity WARN/FAIL should be interpreted with caution for neural
+ # tissue — cross-check with Laplacian floor and boundary/intRNA channels.
+ intensity_warn: 0.35
+ intensity_fail: 0.60
+ # Slightly relaxed for FFPE heterogeneity and field-to-field variation.
+ spatial_cv_warn: 0.40
+ spatial_cv_fail: 0.65
+ # Tile blur fraction (GMM-derived): relaxed warn/fail for practical runs.
+ focus_warn: 0.15
+ focus_fail: 0.30
+ # Agreement between focus_score and lap_var across tissue tiles.
+ # Low correlation can indicate one metric is unstable for that sample.
+ lap_focus_corr_warn: 0.50
+ lap_focus_corr_fail: 0.25
+ boundary:
+ intensity_critical: 100
+ # Boundary channel is biologically patchy across tissues (neural/stromal/immune).
+ intensity_warn: 0.15
+ intensity_fail: 0.35
+ # Tissue-type note (not parameterized yet):
+ # - neural/stromal tissues may need further relaxation to 0.25 / 0.50
+ # - epithelial tissues often tolerate stricter defaults
+ spatial_cv_warn: 0.45
+ spatial_cv_fail: 0.65
+ # Keep spatial_cv defaults unless tissue-type-specific config is introduced.
+ # Correlation with DAPI is tissue-dependent; use permissive thresholds.
+ dapi_corr_warn: 0.20
+ dapi_corr_fail: 0.05
+ intRNA:
+ intensity_critical: 300
+ # Most biology-dependent channel (necrosis/adipose/white matter strongly affect signal).
+ intensity_warn: 0.25
+ intensity_fail: 0.50
+ # Keep relatively permissive: high spatial variance is expected for intRNA in heterogeneous tissues.
+ spatial_cv_warn: 0.55
+ spatial_cv_fail: 0.75
+ # Very relaxed by design: intRNA↔DAPI coupling is weak in many valid mixed tissues.
+ dapi_corr_warn: 0.10
+ dapi_corr_fail: 0.02
+ slide_level:
+ # Definition note:
+ # "multi_channel_fail_*" depends on how channel failure is counted.
+ # - Any-one-channel critical fail per tile -> more sensitive (more WARN/FAIL calls).
+ # - Two-or-more channels critical fail per tile -> more specific but stricter.
+ # Keep this explicit in code/report outputs when these thresholds are applied.
+ multi_channel_fail_warn: 0.10
+ multi_channel_fail_fail: 0.30
+ # NOTE: semantics matter.
+ # Recommended definition:
+ # usable = tile is not blurred AND not below intensity threshold
+ # usable_tissue_* should be computed as "% of tissue-overlapping tiles that are usable"
+ # (not "% of all slide tiles"), so small inputs (TMA/biopsy) are not unfairly penalized.
+ usable_tissue_warn: 0.70
+ usable_tissue_fail: 0.50
+ # Contiguous bad-tile area >15% usually indicates a localized physical artefact (fold/chip/shadow).
+ cluster_zone_fail: 0.15
+ morphology:
+ # Coverage thresholds relaxed to avoid treating tiny/partially missing sections as acceptable.
+ tissue_coverage_warn: 0.30
+ tissue_coverage_fail: 0.10
+ # Sensible defaults; consider relaxing for small inputs (TMA/biopsy/needle cores) where edge-dominance is expected.
+ edge_zone_warn: 0.30
+ edge_zone_fail: 0.50
+ # Relaxed until hole mask better separates anatomical lumens from artefactual tears.
+ hole_area_warn: 0.25
+ hole_area_fail: 0.50
+ focus:
+ # Cell-level CCFS nuclear texture threshold: cells with CCFS_DAPI below
+ # this are classified as low nuclear texture quality. CCFS measures per-cell
+ # nuclear contrast (local_var/local_mean), NOT optical blur — it is distinct
+ # from the tile-level GMM blur metric.
+ # Calibrated on 5 samples (2026-04-02): cells below 0.02 lose >50%
+ # transcripts vs top-50% CCFS cells in lung/liver.
+ # Brain/neural tissues: CCFS is confounded by nucleus size (large neurons
+ # score lower); interpret with caution.
+ ccfs_low_texture_threshold: 0.02
+ # Cell-level only: applies to per-cell CCFS nuclear texture calls
+ # (not tile-level GMM blur fractions).
+ # 10%/25% is a practical default for many tissues; review for very
+ # small-nucleus contexts (lymphoid / some brain regions) where cell-level
+ # CCFS can overcall low texture.
+ low_texture_cell_warn: 0.10
+ low_texture_cell_fail: 0.25
+ # Median raw DAPI focus score (std²/mean) across tissue tiles.
+ # Highly tissue-dependent: brain/small-nucleus tissues score lower than
+ # lung/liver. Calibrated on 8 samples (2026-04-01):
+ # bad (panc/brain): 0.95 - 126, moderate: 137 - 797, good (lung): 1079.
+ # WARN-only; no FAIL — insufficient cross-tissue data for a hard cutoff.
+ focus_median_warn: 100
+ morans_i_warn: 0.30
+ morans_i_fail: 0.60
+ # Absolute Laplacian variance floor (raw units, not normalised).
+ # Catches uniformly blurry slides where relative z-scores are all near 0.
+ # Calibrated on 6 samples (2026-04-01): critical samples 45-71, good 1123-3458.
+ lap_var_absolute_floor_warn: 100
+ lap_var_absolute_floor_fail: 50
+ # 2D GMM component separation (Cohen's d between blur and focus populations
+ # in log1p(lap_var) space). Low values indicate the GMM cannot reliably
+ # distinguish blurred from focused tiles.
+ # Calibrated range across 6 samples: 2.98-4.64 (all well-separated).
+ gmm_2d_laplacian_component_separation_warn: 1.0
+ gmm_2d_laplacian_component_separation_fail: 0.5
+ snr:
+ # Per-channel image SNR dB thresholds.
+ # RESERVED: not yet consumed by code (quartile/Otsu currently use
+ # image_snr_db fallback below). Will be wired in when per-channel
+ # SNR assessment is implemented.
+ dapi:
+ warn_db: 15
+ fail_db: 8
+ boundary:
+ warn_db: 10
+ fail_db: 5
+ intrna:
+ warn_db: 8
+ fail_db: 3
+ # Fallback image SNR dB thresholds when channel is not specified.
+ image_snr_db:
+ warn: 12.0
+ fail: 8.0
+ # Per-tile transcript SNR (target-to-negative ratio and negative fraction).
+ roi_tx:
+ ratio_warn: 3.0
+ ratio_fail: 1.5
+ neg_pct_warn: 0.15
+ neg_pct_fail: 0.30
+ # Negative-probe spatial autocorrelation (quadrant spread + Moran's I).
+ # Loosened from 0.3/0.6 — heterogeneous tissues (pancreas, liver) can
+ # produce biologically-driven spatial variation in negative probe rates.
+ neg_spatial:
+ quadrant_spread_warn: 0.5
+ quadrant_spread_fail: 1.0
+ moran_warn: 0.5
+ moran_fail: 1.0
+ # Slide-level Plummer SNR (log-ratio of real vs negative gene means).
+ slide_plummer:
+ pass_min: 0.12
+ fail_below: 0.10
+ # Slide-level SpatialQM — verdict based on pct_genes_above_neg (% of genes
+ # whose mean expression exceeds the mean negative-control level).
+ slide_spatialqm:
+ metric: pct_genes_above_neg
+ warn: 45
+ fail: 35
diff --git a/docs/usage.md b/docs/usage.md
index 406f538a..c99f9e3d 100644
--- a/docs/usage.md
+++ b/docs/usage.md
@@ -71,6 +71,22 @@ nextflow run nf-core/spatialaxe \
--mode segfree
```
+### QC mode
+
+`IMAGE_QC ➔ TRANSCRIPT_QC`
+
+Runs the quality-control layer only — image QC (focus / SNR / morphology) and transcript QC (per-transcript and per-cell metrics), each producing an HTML report. The QC layer also runs as part of the other modes (`run_qc = true` by default).
+
+```bash
+nextflow run nf-core/spatialaxe \
+ -profile \
+ --input samplesheet.csv \
+ --outdir \
+ --mode qc
+```
+
+Image QC is GPU-optional: leave `--num_gpus` unset for CPU, or set `--num_gpus N` to request and use N GPUs (requires a GPU-enabled profile/queue).
+
### Preview mode
`BAYSOR_PREVIEW`
diff --git a/modules.json b/modules.json
index 494ffa18..1f0ac6fb 100644
--- a/modules.json
+++ b/modules.json
@@ -35,6 +35,11 @@
"installed_by": ["modules"],
"patch": "modules/nf-core/opt/track/opt-track.diff"
},
+ "quartonotebook": {
+ "branch": "master",
+ "git_sha": "6d46786420b4d7bc88eba026eb389c0c5535d120",
+ "installed_by": ["modules"]
+ },
"stardist": {
"branch": "master",
"git_sha": "4e783502ab661bed13f15189401b73c93966831f",
diff --git a/modules/local/image_qc/Dockerfile b/modules/local/image_qc/Dockerfile
new file mode 100644
index 00000000..6a54ef30
--- /dev/null
+++ b/modules/local/image_qc/Dockerfile
@@ -0,0 +1,24 @@
+# Container for the image QC module (analysis + Quarto report).
+# Built from environment.yml in this directory. Tagged and pushed as
+# quay.io/dongzehe/image_qc:1.0.0 (to be migrated to the nf-core org for release).
+#
+# docker build -t quay.io//image_qc:1.0.0 modules/local/image_qc
+#
+# Note: behind a corporate proxy / private conda mirror you may need to provide a
+# CA bundle and channel config at build time (e.g. via BuildKit --mount=type=secret);
+# a standard build resolves the pinned packages directly from conda-forge/bioconda.
+FROM mambaorg/micromamba:2.0.5
+COPY --chown=$MAMBA_USER:$MAMBA_USER environment.yml /tmp/environment.yml
+RUN micromamba install -y -n base -f /tmp/environment.yml \
+ && micromamba clean -a -f -y \
+ && rm -f /tmp/environment.yml
+ENV PATH="/opt/conda/bin:$PATH" \
+ QUARTO_DENO=/opt/conda/bin/deno \
+ QUARTO_DENO_DOM=/opt/conda/lib/deno_dom.so \
+ QUARTO_PANDOC=/opt/conda/bin/pandoc \
+ QUARTO_ESBUILD=/opt/conda/bin/esbuild \
+ QUARTO_TYPST=/opt/conda/bin/typst \
+ QUARTO_DART_SASS=/opt/conda/bin/sass \
+ QUARTO_SHARE_PATH=/opt/conda/share/quarto \
+ QUARTO_CONDA_PREFIX=/opt/conda \
+ ES2_LIBRARY=/opt/conda/lib/libGLESv2.so.2
diff --git a/modules/local/image_qc/environment.yml b/modules/local/image_qc/environment.yml
new file mode 100644
index 00000000..00ab029b
--- /dev/null
+++ b/modules/local/image_qc/environment.yml
@@ -0,0 +1,51 @@
+---
+# yaml-language-server: $schema=https://raw.githubusercontent.com/nf-core/modules/master/modules/environment-schema.json
+channels:
+ - conda-forge
+ - bioconda
+dependencies:
+ # procps-ng provides `ps`, required by Nextflow for task-metric collection
+ - conda-forge::procps-ng=4.0.4
+ - conda-forge::python=3.11.0
+ - conda-forge::numpy=2.3.3
+ - conda-forge::pandas=2.3.2
+ - conda-forge::scipy=1.16.2
+ - conda-forge::matplotlib=3.10.6
+ - conda-forge::seaborn=0.13.2
+ - conda-forge::scikit-image=0.25.2
+ - conda-forge::scikit-learn=1.7.2
+ - conda-forge::numba=0.62.0
+ - conda-forge::tifffile=2025.9.20
+ - conda-forge::zarr=2.18.7
+ - conda-forge::imagecodecs=2025.8.2
+ - conda-forge::pillow=12.0.0
+ - conda-forge::pyyaml=6.0.3
+ - conda-forge::h5py=3.14.0
+ - conda-forge::pyarrow=21.0.0
+ - conda-forge::click=8.3.0
+ - conda-forge::tqdm=4.67.1
+ - conda-forge::tbb=2023.0.0
+ # Report rendering (Quarto) — this container also renders the image QC HTML report
+ - conda-forge::quarto=1.6.42
+ - conda-forge::jupyter=1.1.1
+ - conda-forge::ipython=9.5.0
+ - conda-forge::ipykernel=6.30.1
+ - conda-forge::papermill=2.6.0
+ - conda-forge::napari-simpleitk-image-processing=0.4.9
+ - conda-forge::napari-skimage-regionprops=0.10.1
+ # GL ES runtime for the napari->vispy import chain pulled in by the two napari
+ # plugins above. Never used for actual rendering (headless run); vispy merely
+ # dlopens libGLESv2 at import. The container sets ES2_LIBRARY to this library
+ # so vispy finds it without ldconfig visibility of /opt/conda/lib.
+ - conda-forge::libgles=1.7.0
+ # Spatial autocorrelation (Moran's I) for the ROI / SNR QC metrics
+ - conda-forge::libpysal=4.14.1
+ - conda-forge::esda=2.9.0
+ - conda-forge::pysal=26.1
+ - conda-forge::pip=26.2.1
+ # GPU acceleration (optional at runtime). cupy imports are guarded by try/except
+ # in image_qc.py, so this environment also runs correctly on CPU-only hosts.
+ - pip:
+ - cupy-cuda12x==14.0.1
+ - nvidia-cuda-nvrtc-cu12==12.9.86
+ - nvidia-cuda-runtime-cu12==12.9.79
diff --git a/modules/local/image_qc/main.nf b/modules/local/image_qc/main.nf
new file mode 100644
index 00000000..bd642c7e
--- /dev/null
+++ b/modules/local/image_qc/main.nf
@@ -0,0 +1,163 @@
+process IMAGE_QC_ANALYSIS {
+ tag "${meta.id}"
+ label 'process_gpu_qc'
+
+ conda "${moduleDir}/environment.yml"
+ // Built from environment.yml (see the pipeline Dockerfile assets). Hosted on the
+ // author's quay.io namespace for now; to be migrated to the nf-core org before release.
+ container "quay.io/dongzehe/image_qc:1.0.0"
+
+ input:
+ tuple val(meta), val(parameters), path(input_files)
+ path(roi_thresholds_yaml)
+
+ output:
+ tuple val(meta), path(outdir), emit: outdir
+ tuple val("${task.process}"), val('python'), eval("python3 --version | sed 's/Python //'"), topic: versions, emit: versions_python
+ tuple val("${task.process}"), val('numpy'), eval("python3 -c 'import numpy; print(numpy.__version__)'"), topic: versions, emit: versions_numpy
+ tuple val("${task.process}"), val('scikit-image'), eval("python3 -c 'import skimage; print(skimage.__version__)'"), topic: versions, emit: versions_skimage
+
+ when:
+ task.ext.when == null || task.ext.when
+
+ script:
+ def prefix = task.ext.prefix ?: "${meta.id}"
+ outdir = prefix
+
+ // Convert parameters to script arguments
+ def args = []
+
+ // Required parameters
+ args << "--xenium-bundle-dir '${input_files[0]}'"
+ args << "--outdir '${outdir}'"
+ args << "--sample-id '${meta.id}'"
+
+ // ROI thresholds YAML (staged file)
+ if (roi_thresholds_yaml.name != 'NO_FILE') {
+ args << "--roi-thresholds-yaml '${roi_thresholds_yaml}'"
+ }
+ // Parse optional parameters from the parameters list
+ def param_map = [:]
+ if (parameters) {
+ parameters.collate(2).each { key, value ->
+ if (value != null && value != '') {
+ param_map[key] = value
+ }
+ }
+ }
+
+ // Pass stain-names as a single semicolon-separated string
+ if (param_map.containsKey('STAIN_NAMES') && param_map['STAIN_NAMES']) {
+ def stains_str = param_map['STAIN_NAMES'].toString()
+ args << "--stain-names '${stains_str}'"
+ } else {
+ // Use params.stain_names from config if available, otherwise fall back to defaults
+ def default_stains_str = params.stain_names ?: "DAPI;Boundary (ATP1A1/E-Cadherin/CD45);Interior - RNA (18S);Protein (alphaSMA/Vimentin)"
+ args << "--stain-names '${default_stains_str}'"
+ }
+
+ // Tile size parameter (a.k.a. ROI size internally). Per-sample
+ // `parameters` map can override via ROI_SIZE; otherwise the top-level
+ // `params.tile_size` (Seqera-visible) is used. Default 35 px.
+ if (param_map.containsKey('ROI_SIZE') && param_map['ROI_SIZE']) {
+ args << "--roi-size ${param_map['ROI_SIZE']}"
+ } else {
+ args << "--roi-size ${params.tile_size ?: 35}"
+ }
+
+ // Legacy focus score mode (CPU for-loop instead of GPU convolution)
+ if (params.legacy_focus) {
+ args << "--legacy-focus"
+ }
+
+ if (params.image_qc_no_snr) {
+ args << '--no-snr'
+ }
+ if (params.image_qc_snr_no_roi_tx_table) {
+ args << '--snr-no-roi-tx-table'
+ }
+ if (params.image_qc_snr_otsu_max_rois != null) {
+ args << "--snr-otsu-max-rois ${params.image_qc_snr_otsu_max_rois}"
+ }
+ // image_qc_snr_no_moran true (default): Moran off — no flag. false: opt in to Moran.
+ if (!params.image_qc_snr_no_moran) {
+ args << '--snr-with-moran'
+ }
+ if (params.image_qc_save_dapi_maps_tiff) {
+ args << '--save-dapi-maps-tiff'
+ }
+ // Streaming is the script default: each tile is reduced and dropped, so no
+ // full-resolution pixel plane is written. The planes cost ~154 GB of scratch on
+ // a 5.5 gigapixel sample and their writeback to S3 is what pushed run
+ // 3nkeHOEV1ONlbK past the 4 h wall. Opt out only to reproduce the old path.
+ if (!params.image_qc_stream_tiles) {
+ args << '--no-stream-tiles'
+ }
+ // Per-figure figures_source/*.csv source-data exports are unused downstream
+ // (the QMD embeds only the PNGs) and the big ROI-table dumps cost ~80 s each.
+ // Off by default; opt in to write them.
+ if (params.image_qc_figure_source_tables) {
+ args << '--figure-source-tables'
+ }
+ // Figure generation master toggle. Off skips ALL figure rendering for a
+ // metrics-only fast run; metrics/JSON/parquet are always produced.
+ if (!params.image_qc_figures) {
+ args << '--no-figures'
+ }
+ // Cap the devices the script uses. `accelerator` only tells AWS Batch how many
+ // GPUs to request -- it does not restrict what CUDA can see, so a task asking
+ // for 1 GPU that lands on a 4-GPU instance would otherwise detect and use all
+ // four. Observed on run 3nkeHOEV1ONlbK: image_qc_gpus=1 placed on a
+ // g6e.12xlarge and the script reported "4 GPU(s)".
+ args << "--max-gpus ${params.image_qc_gpus}"
+ if (params.image_qc_lap_sigma != null) {
+ args << "--lap-sigma ${params.image_qc_lap_sigma}"
+ }
+
+ if (param_map.containsKey('PIPELINE_SEGMENTATION') && param_map['PIPELINE_SEGMENTATION']) {
+ args << "--pipeline-segmentation '${param_map['PIPELINE_SEGMENTATION']}'"
+ }
+
+ // Boolean flag — emit only when truthy (never '--is-resegmented false')
+ if (param_map.containsKey('IS_RESEGMENTED') && param_map['IS_RESEGMENTED'].toString() == 'true') {
+ args << "--is-resegmented"
+ }
+
+ """
+ export MKL_NUM_THREADS="$task.cpus"
+ export OPENBLAS_NUM_THREADS="$task.cpus"
+ export OMP_NUM_THREADS="$task.cpus"
+ export NUMBA_NUM_THREADS="$task.cpus"
+
+ # Capture the analysis exit code without aborting (Nextflow runs with set -e).
+ # A very dim sample can make image_qc.py exit 1 (e.g. no tissue tiles clear the
+ # intensity gate); we still want a report, so on exit 1 we record a failed
+ # status and exit 0. The Quarto report reads image_qc_status.json and renders a
+ # QC-FAILED banner. Signal / OOM / preemption codes (104, 130-145) are re-raised
+ # so Nextflow's retry errorStrategy still fires.
+ rc=0
+ image_qc.py \\
+ ${args.join(' \\\n ')} || rc=\$?
+
+ mkdir -p "${outdir}"
+
+ if [ "\$rc" -eq 0 ]; then
+ echo '{"status": "ok", "sample_id": "${meta.id}"}' > "${outdir}/image_qc_status.json"
+ elif [ "\$rc" -eq 1 ]; then
+ echo "WARNING: image_qc.py exited 1 (Python error); writing failed status for report" >&2
+ echo '{"status": "failed", "exit_code": 1, "sample_id": "${meta.id}"}' > "${outdir}/image_qc_status.json"
+ else
+ echo "image_qc.py exited \$rc (signal/OOM); propagating for errorStrategy" >&2
+ exit \$rc
+ fi
+ """
+
+ stub:
+ def prefix = task.ext.prefix ?: "${meta.id}"
+ outdir = prefix
+ """
+ mkdir -p "${outdir}/figures"
+ touch "${outdir}/image_qc_metrics.json"
+ touch "${outdir}/image_qc_metrics.csv"
+ """
+}
diff --git a/modules/local/image_qc/meta.yml b/modules/local/image_qc/meta.yml
new file mode 100644
index 00000000..57ce75d9
--- /dev/null
+++ b/modules/local/image_qc/meta.yml
@@ -0,0 +1,84 @@
+name: "image_qc_analysis"
+description: Image-based quality control for a Xenium bundle (focus / SNR / morphology metrics and figures)
+keywords:
+ - spatial
+ - xenium
+ - quality control
+ - image
+ - QC
+tools:
+ - "image_qc":
+ description: "Custom Python image-QC analysis for 10x Genomics Xenium morphology images (focus score, signal-to-noise, ROI metrics and figures)"
+ homepage: "https://github.com/nf-core/spatialaxe"
+ documentation: "https://github.com/nf-core/spatialaxe"
+ tool_dev_url: "https://github.com/nf-core/spatialaxe"
+ licence: ["MIT"]
+ identifier: ""
+input:
+ - - meta:
+ type: map
+ description: |
+ Groovy Map containing sample information
+ (sample id)
+ - parameters:
+ type: list
+ description: |
+ Flat key/value list of analysis parameters (collated by the module),
+ e.g. XENIUM_BUNDLE_DIR, STAIN_NAMES, ROI_SIZE
+ - input_files:
+ type: file
+ description: |
+ Input file(s) for the analysis; the first element is the Xenium bundle directory
+ pattern: "*"
+ ontologies: []
+ - roi_thresholds_yaml:
+ type: file
+ description: |
+ ROI threshold config YAML for image QC. Pass an empty placeholder (NO_FILE)
+ to use the bundled default.
+ pattern: "*.{yml,yaml}"
+ ontologies: []
+output:
+ outdir:
+ - - meta:
+ type: map
+ description: |
+ Groovy Map containing sample information
+ [sample id]
+ - "${outdir}":
+ type: directory
+ description: Output directory with image QC metrics (JSON/CSV) and figures
+ pattern: "${prefix}"
+topics:
+ versions:
+ - - ${task.process}:
+ type: string
+ description: The name of the process
+ - python:
+ type: string
+ description: The name of the tool
+ - "python3 --version | sed 's/Python //'":
+ type: eval
+ description: The expression to obtain the version of the tool
+ - - ${task.process}:
+ type: string
+ description: The name of the process
+ - numpy:
+ type: string
+ description: The name of the tool
+ - "python3 -c 'import numpy; print(numpy.__version__)'":
+ type: eval
+ description: The expression to obtain the version of the tool
+ - - ${task.process}:
+ type: string
+ description: The name of the process
+ - scikit-image:
+ type: string
+ description: The name of the tool
+ - "python3 -c 'import skimage; print(skimage.__version__)'":
+ type: eval
+ description: The expression to obtain the version of the tool
+authors:
+ - "@an-altosian"
+maintainers:
+ - "@an-altosian"
diff --git a/modules/local/image_qc/tests/fixtures/bundle/placeholder.txt b/modules/local/image_qc/tests/fixtures/bundle/placeholder.txt
new file mode 100644
index 00000000..36b04cfa
--- /dev/null
+++ b/modules/local/image_qc/tests/fixtures/bundle/placeholder.txt
@@ -0,0 +1 @@
+placeholder xenium bundle file for -stub tests
diff --git a/modules/local/image_qc/tests/fixtures/roi_thresholds.yaml b/modules/local/image_qc/tests/fixtures/roi_thresholds.yaml
new file mode 100644
index 00000000..52ba65ca
--- /dev/null
+++ b/modules/local/image_qc/tests/fixtures/roi_thresholds.yaml
@@ -0,0 +1,2 @@
+# placeholder ROI thresholds YAML for -stub tests
+thresholds: {}
diff --git a/modules/local/image_qc/tests/main.nf.test b/modules/local/image_qc/tests/main.nf.test
new file mode 100644
index 00000000..76feca97
--- /dev/null
+++ b/modules/local/image_qc/tests/main.nf.test
@@ -0,0 +1,37 @@
+nextflow_process {
+
+ name "Test Process IMAGE_QC_ANALYSIS"
+ script "../main.nf"
+ process "IMAGE_QC_ANALYSIS"
+ config "./nextflow.config"
+
+ tag "modules"
+ tag "modules_local"
+ tag "image_qc"
+
+ test("image qc analysis stub") {
+
+ options "-stub"
+
+ when {
+ process {
+ """
+ input[0] = [
+ [id: "test"],
+ [],
+ file("${moduleTestDir}/fixtures/bundle", checkIfExists: true)
+ ]
+ input[1] = file("${moduleTestDir}/fixtures/roi_thresholds.yaml", checkIfExists: true)
+ """
+ }
+ }
+
+ then {
+ assertAll(
+ { assert process.success },
+ { assert file(process.out.outdir[0][1]).exists() },
+ { assert file(process.out.outdir[0][1] + "/image_qc_metrics.json").exists() }
+ )
+ }
+ }
+}
diff --git a/modules/local/image_qc/tests/nextflow.config b/modules/local/image_qc/tests/nextflow.config
new file mode 100644
index 00000000..f8b3a30a
--- /dev/null
+++ b/modules/local/image_qc/tests/nextflow.config
@@ -0,0 +1,9 @@
+process {
+
+ resourceLimits = [
+ cpus: 4,
+ memory: '8.GB',
+ time: '2.h',
+ ]
+
+}
diff --git a/modules/local/transcript_qc/Dockerfile b/modules/local/transcript_qc/Dockerfile
new file mode 100644
index 00000000..691bf3a4
--- /dev/null
+++ b/modules/local/transcript_qc/Dockerfile
@@ -0,0 +1,23 @@
+# Container for the transcript QC module (analysis + Quarto report).
+# Built from environment.yml in this directory. Tagged and pushed as
+# quay.io/dongzehe/transcript_qc:1.0.0 (to be migrated to the nf-core org for release).
+#
+# docker build -t quay.io//transcript_qc:1.0.0 modules/local/transcript_qc
+#
+# Note: behind a corporate proxy / private conda mirror you may need to provide a
+# CA bundle and channel config at build time (e.g. via BuildKit --mount=type=secret);
+# a standard build resolves the pinned packages directly from conda-forge/bioconda.
+FROM mambaorg/micromamba:2.0.5
+COPY --chown=$MAMBA_USER:$MAMBA_USER environment.yml /tmp/environment.yml
+RUN micromamba install -y -n base -f /tmp/environment.yml \
+ && micromamba clean -a -f -y \
+ && rm -f /tmp/environment.yml
+ENV PATH="/opt/conda/bin:$PATH" \
+ QUARTO_DENO=/opt/conda/bin/deno \
+ QUARTO_DENO_DOM=/opt/conda/lib/deno_dom.so \
+ QUARTO_PANDOC=/opt/conda/bin/pandoc \
+ QUARTO_ESBUILD=/opt/conda/bin/esbuild \
+ QUARTO_TYPST=/opt/conda/bin/typst \
+ QUARTO_DART_SASS=/opt/conda/bin/sass \
+ QUARTO_SHARE_PATH=/opt/conda/share/quarto \
+ QUARTO_CONDA_PREFIX=/opt/conda
diff --git a/modules/local/transcript_qc/environment.yml b/modules/local/transcript_qc/environment.yml
new file mode 100644
index 00000000..b7f46a97
--- /dev/null
+++ b/modules/local/transcript_qc/environment.yml
@@ -0,0 +1,33 @@
+---
+# yaml-language-server: $schema=https://raw.githubusercontent.com/nf-core/modules/master/modules/environment-schema.json
+channels:
+ - conda-forge
+ - bioconda
+dependencies:
+ # procps-ng provides `ps`, required by Nextflow for task-metric collection
+ - conda-forge::procps-ng=4.0.4
+ - conda-forge::python=3.11.0
+ - conda-forge::numpy=2.3.3
+ - conda-forge::pandas=2.3.2
+ - conda-forge::scipy=1.16.2
+ - conda-forge::matplotlib=3.10.6
+ - conda-forge::seaborn=0.13.2
+ - conda-forge::pyarrow=21.0.0
+ # Report rendering (Quarto) — this container also renders the transcript QC HTML report
+ - conda-forge::pyyaml=6.0.3
+ - conda-forge::quarto=1.6.42
+ - conda-forge::jupyter=1.1.1
+ - conda-forge::ipython=9.5.0
+ - conda-forge::ipykernel=6.30.1
+ - conda-forge::papermill=2.6.0
+ - conda-forge::pip=26.2.1
+ - pip:
+ - scanpy==1.11.5
+ - anndata==0.12.2
+ - leidenalg==0.10.2
+ - python-igraph==0.11.9
+ - pynndescent==0.5.13
+ - scikit-learn==1.7.2
+ - statsmodels==0.14.5
+ - h5py==3.14.0
+ - natsort==8.4.0
diff --git a/modules/local/transcript_qc/main.nf b/modules/local/transcript_qc/main.nf
new file mode 100644
index 00000000..a24187c2
--- /dev/null
+++ b/modules/local/transcript_qc/main.nf
@@ -0,0 +1,78 @@
+process TRANSCRIPT_QC_PROCESSING {
+ tag "${meta.id}"
+ label 'process_high'
+
+ conda "${moduleDir}/environment.yml"
+ // Built from environment.yml (see the pipeline Dockerfile assets). Hosted on the
+ // author's quay.io namespace for now; to be migrated to the nf-core org before release.
+ container "quay.io/dongzehe/transcript_qc:1.0.0"
+
+ input:
+ tuple val(meta), val(parameters), path(input_files)
+
+ output:
+ tuple val(meta), path(outdir), emit: outdir
+ tuple val("${task.process}"), val('python'), eval("python3 --version | sed 's/Python //'"), topic: versions, emit: versions_python
+ tuple val("${task.process}"), val('scanpy'), eval("python3 -c 'import scanpy; print(scanpy.__version__)'"), topic: versions, emit: versions_scanpy
+ tuple val("${task.process}"), val('anndata'), eval("python3 -c 'import anndata; print(anndata.__version__)'"), topic: versions, emit: versions_anndata
+
+ when:
+ task.ext.when == null || task.ext.when
+
+ script:
+ def prefix = task.ext.prefix ?: "${meta.id}"
+ outdir = prefix
+
+ // Convert parameters to script arguments
+ def args = []
+
+ // Required parameters
+ args << "--xenium-bundle-dir '${input_files[0]}'"
+ args << "--outdir '${outdir}'"
+ args << "--threads ${task.cpus}"
+ args << "--task-process '${task.process}'"
+
+ // Parse optional parameters from the parameters list
+ def param_map = [:]
+ if (parameters) {
+ parameters.collate(2).each { key, value ->
+ if (value != null && value != '') {
+ param_map[key] = value
+ }
+ }
+ }
+
+ // Add optional parameters
+ if (param_map.containsKey('NON_GENE_PREFIX') && param_map['NON_GENE_PREFIX']) {
+ def prefixes = param_map['NON_GENE_PREFIX'].toString().split(';').collect { "'${it.trim()}'" }.join(' ')
+ args << "--non-gene-prefix ${prefixes}"
+ }
+
+ if (param_map.containsKey('STAIN_NAMES') && param_map['STAIN_NAMES']) {
+ def stains = param_map['STAIN_NAMES'].toString().split(';').collect { "'${it.trim()}'" }.join(' ')
+ args << "--stain-names ${stains}"
+ }
+
+ if (param_map.containsKey('NUM_ROW_GROUPS') && param_map['NUM_ROW_GROUPS']) {
+ args << "--num-row-groups ${param_map['NUM_ROW_GROUPS']}"
+ }
+
+ """
+ export MKL_NUM_THREADS="${task.cpus}"
+ export OPENBLAS_NUM_THREADS="${task.cpus}"
+ export OMP_NUM_THREADS="${task.cpus}"
+ export NUMBA_NUM_THREADS="${task.cpus}"
+
+ transcript_qc_processing.py \\
+ ${args.join(' \\\n ')}
+ """
+
+ stub:
+ def prefix = task.ext.prefix ?: "${meta.id}"
+ outdir = prefix
+ """
+ mkdir -p "${outdir}/figures"
+ touch "${outdir}/transcript_qc_metrics.json"
+ touch "${outdir}/versions.yml"
+ """
+}
diff --git a/modules/local/transcript_qc/meta.yml b/modules/local/transcript_qc/meta.yml
new file mode 100644
index 00000000..824775e9
--- /dev/null
+++ b/modules/local/transcript_qc/meta.yml
@@ -0,0 +1,78 @@
+name: "transcript_qc_processing"
+description: Transcript / molecule-level quality control for a Xenium bundle (per-transcript and per-cell QC metrics and figures)
+keywords:
+ - spatial
+ - xenium
+ - quality control
+ - transcript
+ - molecule
+ - QC
+tools:
+ - "transcript_qc":
+ description: "Custom Python transcript-QC analysis for 10x Genomics Xenium data (per-transcript and per-cell QC metrics and figures)"
+ homepage: "https://github.com/nf-core/spatialaxe"
+ documentation: "https://github.com/nf-core/spatialaxe"
+ tool_dev_url: "https://github.com/nf-core/spatialaxe"
+ licence: ["MIT"]
+ identifier: ""
+input:
+ - - meta:
+ type: map
+ description: |
+ Groovy Map containing sample information
+ (sample id)
+ - parameters:
+ type: list
+ description: |
+ Flat key/value list of analysis parameters (collated by the module),
+ e.g. XENIUM_BUNDLE_DIR, NON_GENE_PREFIX, STAIN_NAMES, NUM_ROW_GROUPS
+ - input_files:
+ type: file
+ description: |
+ Input file(s) for the analysis; the first element is the Xenium bundle directory
+ pattern: "*"
+ ontologies: []
+output:
+ outdir:
+ - - meta:
+ type: map
+ description: |
+ Groovy Map containing sample information
+ [sample id]
+ - "${outdir}":
+ type: directory
+ description: Output directory with transcript QC metrics (JSON) and figures
+ pattern: "${prefix}"
+topics:
+ versions:
+ - - ${task.process}:
+ type: string
+ description: The name of the process
+ - python:
+ type: string
+ description: The name of the tool
+ - "python3 --version | sed 's/Python //'":
+ type: eval
+ description: The expression to obtain the version of the tool
+ - - ${task.process}:
+ type: string
+ description: The name of the process
+ - scanpy:
+ type: string
+ description: The name of the tool
+ - "python3 -c 'import scanpy; print(scanpy.__version__)'":
+ type: eval
+ description: The expression to obtain the version of the tool
+ - - ${task.process}:
+ type: string
+ description: The name of the process
+ - anndata:
+ type: string
+ description: The name of the tool
+ - "python3 -c 'import anndata; print(anndata.__version__)'":
+ type: eval
+ description: The expression to obtain the version of the tool
+authors:
+ - "@an-altosian"
+maintainers:
+ - "@an-altosian"
diff --git a/modules/local/transcript_qc/tests/fixtures/bundle/placeholder.txt b/modules/local/transcript_qc/tests/fixtures/bundle/placeholder.txt
new file mode 100644
index 00000000..36b04cfa
--- /dev/null
+++ b/modules/local/transcript_qc/tests/fixtures/bundle/placeholder.txt
@@ -0,0 +1 @@
+placeholder xenium bundle file for -stub tests
diff --git a/modules/local/transcript_qc/tests/main.nf.test b/modules/local/transcript_qc/tests/main.nf.test
new file mode 100644
index 00000000..f74a6bcf
--- /dev/null
+++ b/modules/local/transcript_qc/tests/main.nf.test
@@ -0,0 +1,36 @@
+nextflow_process {
+
+ name "Test Process TRANSCRIPT_QC_PROCESSING"
+ script "../main.nf"
+ process "TRANSCRIPT_QC_PROCESSING"
+ config "./nextflow.config"
+
+ tag "modules"
+ tag "modules_local"
+ tag "transcript_qc"
+
+ test("transcript qc processing stub") {
+
+ options "-stub"
+
+ when {
+ process {
+ """
+ input[0] = [
+ [id: "test"],
+ [],
+ file("${moduleTestDir}/fixtures/bundle", checkIfExists: true)
+ ]
+ """
+ }
+ }
+
+ then {
+ assertAll(
+ { assert process.success },
+ { assert file(process.out.outdir[0][1]).exists() },
+ { assert file(process.out.outdir[0][1] + "/transcript_qc_metrics.json").exists() }
+ )
+ }
+ }
+}
diff --git a/modules/local/transcript_qc/tests/nextflow.config b/modules/local/transcript_qc/tests/nextflow.config
new file mode 100644
index 00000000..f8b3a30a
--- /dev/null
+++ b/modules/local/transcript_qc/tests/nextflow.config
@@ -0,0 +1,9 @@
+process {
+
+ resourceLimits = [
+ cpus: 4,
+ memory: '8.GB',
+ time: '2.h',
+ ]
+
+}
diff --git a/modules/nf-core/quartonotebook/environment.yml b/modules/nf-core/quartonotebook/environment.yml
new file mode 100644
index 00000000..baffa162
--- /dev/null
+++ b/modules/nf-core/quartonotebook/environment.yml
@@ -0,0 +1,12 @@
+---
+# yaml-language-server: $schema=https://raw.githubusercontent.com/nf-core/modules/master/modules/environment-schema.json
+channels:
+ - conda-forge
+ - bioconda
+dependencies:
+ # renovate: datasource=conda depName=conda-forge/quarto
+ - conda-forge::jupyter=1.1.1
+ - conda-forge::matplotlib=3.10.3
+ - conda-forge::papermill=2.6.0
+ - conda-forge::quarto=1.7.31
+ - conda-forge::r-rmarkdown=2.29
diff --git a/modules/nf-core/quartonotebook/main.nf b/modules/nf-core/quartonotebook/main.nf
new file mode 100644
index 00000000..c96429e4
--- /dev/null
+++ b/modules/nf-core/quartonotebook/main.nf
@@ -0,0 +1,104 @@
+// NB: You'll likely want to override this with a container containing all
+// required dependencies for your analyses. Or use wave to build the container
+// for you from the environment.yml You'll at least need Quarto itself,
+// Papermill and whatever language you are running your analyses on; you can see
+// an example in this module's environment file.
+process QUARTONOTEBOOK {
+ tag "${meta.id}"
+ label 'process_low'
+ conda "${moduleDir}/environment.yml"
+ container "${workflow.containerEngine in ['singularity', 'apptainer'] && !task.ext.singularity_pull_docker_container
+ ? 'https://community-cr-prod.seqera.io/docker/registry/v2/blobs/sha256/28/28717ccd9ce22dbfc219f3db088d5a1fc2ca1f575b5c65621218596dcdbaac95/data'
+ : 'community.wave.seqera.io/library/jupyter_matplotlib_papermill_quarto_r-rmarkdown:6d15193ce3dfc665'}"
+
+ input:
+ tuple val(meta), path(notebook)
+ val(parameters)
+ path input_files
+ path extensions
+
+ output:
+ tuple val(meta), path("*.html") , emit: html
+ tuple val(meta), path(notebook) , emit: notebook
+ tuple val(meta), path("params.yml") , emit: params_yaml
+ tuple val(meta), path("${notebook_parameters.artifact_dir}/*") , emit: artifacts , optional: true
+ tuple val(meta), path("_extensions") , emit: extensions , optional: true
+ tuple val("${task.process}"), val('quarto'), eval('quarto -v'), emit: versions_quarto, topic: versions
+ tuple val("${task.process}"), val('papermill'), eval('papermill --version | cut -f1 -d" "'), emit: versions_papermill, topic: versions
+
+ when:
+ task.ext.when == null || task.ext.when
+
+ script:
+ def args = task.ext.args ?: ''
+ def prefix = task.ext.prefix ?: "${meta.id}"
+ // Implicit parameters can be overwritten by supplying a value with parameters
+ notebook_parameters = [
+ meta: meta,
+ cpus: task.cpus,
+ artifact_dir: "artifacts",
+ ] + (parameters ?: [:])
+ // Parse parameters through a YAML file, which is better than CLI because:
+ // - No issue with escaping
+ // - Allows passing nested maps instead of just single values
+ // - Allows running with the language-agnostic `--execute-params`
+ def yamlBuilder = new groovy.yaml.YamlBuilder()
+ yamlBuilder.call(notebook_parameters)
+ def yaml_content = yamlBuilder.toString().tokenize('\n').join("\n ")
+ """
+ # Dump parameters to yaml file
+ cat <<- END_YAML_PARAMS > params.yml
+ ${yaml_content}
+ END_YAML_PARAMS
+
+ # Create output directory
+ mkdir "${notebook_parameters.artifact_dir}"
+
+ # Set environment variables needed for Quarto rendering
+ export XDG_CACHE_HOME="./.xdg_cache_home"
+ export XDG_DATA_HOME="./.xdg_data_home"
+
+ # Fix Quarto for Apptainer (see https://community.seqera.io/t/confusion-over-why-a-tool-works-in-docker-but-fails-in-singularity-when-the-installation-doesnt-differ-i-e-using-wave-micromamba/1244)
+ ENV_QUARTO=/opt/conda/etc/conda/activate.d/quarto.sh
+ set +u
+ if [ -z "\${QUARTO_DENO}" ] && [ -f "\${ENV_QUARTO}" ]; then
+ source "\${ENV_QUARTO}"
+ fi
+ set -u
+
+ # Set parallelism for BLAS/MKL etc. to avoid over-booking of resources
+ export MKL_NUM_THREADS="${task.cpus}"
+ export OPENBLAS_NUM_THREADS="${task.cpus}"
+ export OMP_NUM_THREADS="${task.cpus}"
+ export NUMBA_NUM_THREADS="${task.cpus}"
+
+ # Render notebook
+ quarto render \\
+ ${notebook} \\
+ ${args} \\
+ --execute-params params.yml \\
+ --output ${prefix}.html
+ """
+
+ stub:
+ def prefix = task.ext.prefix ?: "${meta.id}"
+ // Implicit parameters can be overwritten by supplying a value with parameters
+ notebook_parameters = [
+ meta: meta,
+ cpus: task.cpus,
+ artifact_dir: "artifacts",
+ ] + (parameters ?: [:])
+ """
+ # Fix Quarto for Apptainer (see https://community.seqera.io/t/confusion-over-why-a-tool-works-in-docker-but-fails-in-singularity-when-the-installation-doesnt-differ-i-e-using-wave-micromamba/1244)
+ # Note: This is needed in the stub for `quarto -v` to work.
+ ENV_QUARTO=/opt/conda/etc/conda/activate.d/quarto.sh
+ set +u
+ if [ -z "\${QUARTO_DENO}" ] && [ -f "\${ENV_QUARTO}" ]; then
+ source "\${ENV_QUARTO}"
+ fi
+ set -u
+
+ touch ${prefix}.html
+ touch params.yml
+ """
+}
diff --git a/modules/nf-core/quartonotebook/meta.yml b/modules/nf-core/quartonotebook/meta.yml
new file mode 100644
index 00000000..11ebcf8d
--- /dev/null
+++ b/modules/nf-core/quartonotebook/meta.yml
@@ -0,0 +1,155 @@
+name: "quartonotebook"
+description: Render a Quarto notebook, including parametrization.
+keywords:
+ - quarto
+ - notebook
+ - reports
+ - python
+ - r
+tools:
+ - quartonotebook:
+ description: An open-source scientific and technical publishing system.
+ homepage: https://quarto.org/
+ documentation: https://quarto.org/docs/reference/
+ tool_dev_url: https://github.com/quarto-dev/quarto-cli
+ licence: ["MIT"]
+ identifier: ""
+ - papermill:
+ description: Parameterize, execute, and analyze notebooks
+ homepage: https://github.com/nteract/papermill
+ documentation: http://papermill.readthedocs.io/en/latest/
+ tool_dev_url: https://github.com/nteract/papermill
+ licence: ["BSD 3-clause"]
+ identifier: ""
+
+input:
+ - - meta:
+ type: map
+ description: |
+ Groovy Map containing sample information
+ e.g. `[ id:'sample1', single_end:false ]`.
+ - notebook:
+ type: file
+ description: The Quarto notebook to be rendered.
+ pattern: "*.{qmd}"
+ ontologies: []
+ - parameters:
+ type: map
+ description: |
+ Groovy map with notebook parameters which will be passed to Quarto to
+ generate parametrized reports.
+ - input_files:
+ type: file
+ description: One or multiple files serving as input data for the notebook.
+ pattern: "*"
+ ontologies: []
+ - extensions:
+ type: file
+ description: |
+ A quarto `_extensions` directory with custom template(s) to be
+ available for rendering.
+ pattern: "*"
+ ontologies: []
+output:
+ html:
+ - - meta:
+ type: map
+ description: |
+ Groovy Map containing sample information
+ e.g. `[ id:'sample1', single_end:false ]`.
+ - "*.html":
+ type: file
+ description: HTML report generated by Quarto.
+ pattern: "*.html"
+ ontologies: []
+ notebook:
+ - - meta:
+ type: map
+ description: |
+ Groovy Map containing sample information
+ e.g. `[ id:'sample1', single_end:false ]`.
+ - notebook:
+ type: file
+ description: The Quarto notebook that was rendered. Allows user to
+ continue working on the notebook.
+ pattern: "*.{qmd}"
+ ontologies: []
+ params_yaml:
+ - - meta:
+ type: map
+ description: |
+ Groovy Map containing sample information
+ e.g. `[ id:'sample1', single_end:false ]`.
+ - params.yml:
+ type: file
+ description: Parameters used during report rendering.
+ pattern: "*"
+ ontologies: []
+ artifacts:
+ - - meta:
+ type: map
+ description: |
+ Groovy Map containing sample information
+ e.g. `[ id:'sample1', single_end:false ]`.
+ - ${notebook_parameters.artifact_dir}/*:
+ type: file
+ description: Artifacts generated during report rendering.
+ pattern: "*"
+ ontologies: []
+ extensions:
+ - - meta:
+ type: map
+ description: |
+ Groovy Map containing sample information
+ e.g. `[ id:'sample1', single_end:false ]`.
+ - _extensions:
+ type: file
+ description: Quarto extensions used during report rendering.
+ pattern: "*"
+ ontologies: []
+ versions_quarto:
+ - - ${task.process}:
+ type: string
+ description: The name of the process
+ - quarto:
+ type: string
+ description: The name of the tool
+ - quarto -v:
+ type: eval
+ description: The expression to obtain the version of the tool
+ versions_papermill:
+ - - ${task.process}:
+ type: string
+ description: The name of the process
+ - papermill:
+ type: string
+ description: The name of the tool
+ - papermill --version | cut -f1 -d" ":
+ type: eval
+ description: The expression to obtain the version of the tool
+
+topics:
+ versions:
+ - - ${task.process}:
+ type: string
+ description: The name of the process
+ - quarto:
+ type: string
+ description: The name of the tool
+ - quarto -v:
+ type: eval
+ description: The expression to obtain the version of the tool
+ - - ${task.process}:
+ type: string
+ description: The name of the process
+ - papermill:
+ type: string
+ description: The name of the tool
+ - papermill --version | cut -f1 -d" ":
+ type: eval
+ description: The expression to obtain the version of the tool
+
+authors:
+ - "@fasterius"
+maintainers:
+ - "@fasterius"
diff --git a/modules/nf-core/quartonotebook/tests/main.nf.test b/modules/nf-core/quartonotebook/tests/main.nf.test
new file mode 100644
index 00000000..72b45ba0
--- /dev/null
+++ b/modules/nf-core/quartonotebook/tests/main.nf.test
@@ -0,0 +1,222 @@
+nextflow_process {
+
+ name "Test Process QUARTONOTEBOOK"
+ script "../main.nf"
+ process "QUARTONOTEBOOK"
+
+ tag "modules"
+ tag "modules_nfcore"
+ tag "quartonotebook"
+
+ test("test notebook - [qmd:r]") {
+
+ when {
+ process {
+ """
+ input[0] = [
+ [ id:'test' ], // meta map
+ file(params.modules_testdata_base_path + 'generic/notebooks/quarto/quarto_r.qmd', checkIfExists: true) // Notebook
+ ]
+ input[1] = [:] // Parameters
+ input[2] = [] // Input files
+ input[3] = [] // Extensions
+ """
+ }
+ }
+
+ then {
+ assertAll(
+ { assert process.success },
+ { assert snapshot(
+ process.out.findAll { key, val -> key.startsWith('versions') },
+ process.out.artifacts,
+ process.out.params_yaml
+ ).match() },
+ { assert path(process.out.html[0][1]).readLines().any { it.contains('Hello World 1') } },
+ )
+ }
+
+ }
+
+ test("test notebook - [qmd:python]") {
+
+ when {
+ process {
+ """
+ input[0] = [
+ [ id:'test' ], // meta map
+ file(params.modules_testdata_base_path + 'generic/notebooks/quarto/quarto_python.qmd', checkIfExists: true) // Notebook
+ ]
+ input[1] = [:] // Parameters
+ input[2] = [] // Input files
+ input[3] = [] // Extensions
+ """
+ }
+ }
+
+ then {
+ assertAll(
+ { assert process.success },
+ { assert snapshot(
+ process.out.findAll { key, val -> key.startsWith('versions') },
+ process.out.artifacts,
+ process.out.params_yaml
+ ).match() },
+ { assert path(process.out.html[0][1]).readLines().any { it.contains('Hello World 1') } },
+ )
+ }
+
+ }
+
+ test("test notebook - parametrized - [qmd:r]") {
+
+ when {
+ process {
+ """
+ input[0] = [
+ [ id:'test' ], // meta map
+ file(params.modules_testdata_base_path + 'generic/notebooks/quarto/quarto_r.qmd', checkIfExists: true) // Notebook
+ ]
+ input[1] = [input_filename: "hello.txt", n_iter: 12] // Parameters
+ input[2] = file(params.modules_testdata_base_path + 'generic/txt/hello.txt', checkIfExists: true) // Input files
+ input[3] = [] // Extensions
+ """
+ }
+ }
+
+ then {
+ assertAll(
+ { assert process.success },
+ { assert snapshot(
+ process.out.findAll { key, val -> key.startsWith('versions') },
+ process.out.artifacts,
+ process.out.params_yaml
+ ).match() },
+ { assert path(process.out.html[0][1]).readLines().any { it.contains('Hello World 1') } },
+ { assert path(process.out.params_yaml[0][1]).readLines().any { it.contains('meta:') } },
+ )
+ }
+
+ }
+
+ test("test notebook - parametrized - [qmd:python]") {
+
+ when {
+ process {
+ """
+ input[0] = [
+ [ id:'test' ], // meta map
+ file(params.modules_testdata_base_path + 'generic/notebooks/quarto/quarto_python.qmd', checkIfExists: true) // Notebook
+ ]
+ input[1] = [input_filename: "hello.txt", n_iter: 12] // Parameters
+ input[2] = file(params.modules_testdata_base_path + 'generic/txt/hello.txt', checkIfExists: true) // Input files
+ input[3] = [] // Extensions
+ """
+ }
+ }
+
+ then {
+ assertAll(
+ { assert process.success },
+ { assert snapshot(
+ process.out.findAll { key, val -> key.startsWith('versions') },
+ process.out.artifacts,
+ process.out.params_yaml
+ ).match() },
+ { assert path(process.out.html[0][1]).readLines().any { it.contains('Hello World 1') } },
+ { assert path(process.out.params_yaml[0][1]).readLines().any { it.contains('meta:') } },
+ )
+ }
+
+ }
+
+ test("test notebook - parametrized - [rmd]") {
+
+ when {
+ process {
+ """
+ input[0] = [
+ [ id:'test' ], // meta map
+ file(params.modules_testdata_base_path + 'generic/notebooks/rmarkdown/rmarkdown_notebook.Rmd', checkIfExists: true) // notebook
+ ]
+ input[1] = [input_filename: "hello.txt", n_iter: 12] // Parameters
+ input[2] = file(params.modules_testdata_base_path + 'generic/txt/hello.txt', checkIfExists: true) // Input files
+ input[3] = [] // Extensions
+ """
+ }
+ }
+
+ then {
+ assertAll(
+ { assert process.success },
+ { assert snapshot(
+ process.out.findAll { key, val -> key.startsWith('versions') },
+ process.out.artifacts,
+ process.out.params_yaml
+ ).match() },
+ { assert path(process.out.html[0][1]).readLines().any { it.contains('Hello World 1') } },
+ { assert path(process.out.params_yaml[0][1]).readLines().any { it.contains('meta:') } },
+ )
+ }
+
+ }
+
+ test("test notebook - parametrized - [ipynb]") {
+
+ when {
+ process {
+ """
+ input[0] = [
+ [ id:'test' ], // meta map
+ file(params.modules_testdata_base_path + 'generic/notebooks/jupyter/ipython_notebook.ipynb', checkIfExists: true) // notebook
+ ]
+ input[1] = [input_filename: "hello.txt", n_iter: 12] // Parameters
+ input[2] = file(params.modules_testdata_base_path + 'generic/txt/hello.txt', checkIfExists: true) // Input files
+ input[3] = [] // Extensions
+ """
+ }
+ }
+
+ then {
+ assertAll(
+ { assert process.success },
+ { assert snapshot(
+ process.out.findAll { key, val -> key.startsWith('versions') },
+ process.out.artifacts,
+ process.out.params_yaml
+ ).match() },
+ { assert path(process.out.html[0][1]).readLines().any { it.contains('Hello World') } },
+ { assert path(process.out.params_yaml[0][1]).readLines().any { it.contains('meta:') } },
+ )
+ }
+
+ }
+
+ test("test notebook - stub - [qmd:r]") {
+
+ options "-stub"
+
+ when {
+ process {
+ """
+ input[0] = [
+ [ id:'test' ], // meta map
+ file(params.modules_testdata_base_path + 'generic/notebooks/quarto/quarto_r.qmd', checkIfExists: true) // Notebook
+ ]
+ input[1] = [:] // Parameters
+ input[2] = [] // Input files
+ input[3] = [] // Extensions
+ """
+ }
+ }
+
+ then {
+ assertAll(
+ { assert process.success },
+ { assert snapshot(process.out).match() },
+ )
+ }
+
+ }
+
+}
diff --git a/modules/nf-core/quartonotebook/tests/main.nf.test.snap b/modules/nf-core/quartonotebook/tests/main.nf.test.snap
new file mode 100644
index 00000000..66b7676f
--- /dev/null
+++ b/modules/nf-core/quartonotebook/tests/main.nf.test.snap
@@ -0,0 +1,342 @@
+{
+ "test notebook - stub - [qmd:r]": {
+ "content": [
+ {
+ "0": [
+ [
+ {
+ "id": "test"
+ },
+ "test.html:md5,d41d8cd98f00b204e9800998ecf8427e"
+ ]
+ ],
+ "1": [
+ [
+ {
+ "id": "test"
+ },
+ "quarto_r.qmd:md5,b3fa8b456efae62495c0b278a4f7694c"
+ ]
+ ],
+ "2": [
+ [
+ {
+ "id": "test"
+ },
+ "params.yml:md5,d41d8cd98f00b204e9800998ecf8427e"
+ ]
+ ],
+ "3": [
+
+ ],
+ "4": [
+
+ ],
+ "5": [
+ [
+ "QUARTONOTEBOOK",
+ "quarto",
+ "1.7.31"
+ ]
+ ],
+ "6": [
+ [
+ "QUARTONOTEBOOK",
+ "papermill",
+ "2.6.0"
+ ]
+ ],
+ "artifacts": [
+
+ ],
+ "extensions": [
+
+ ],
+ "html": [
+ [
+ {
+ "id": "test"
+ },
+ "test.html:md5,d41d8cd98f00b204e9800998ecf8427e"
+ ]
+ ],
+ "notebook": [
+ [
+ {
+ "id": "test"
+ },
+ "quarto_r.qmd:md5,b3fa8b456efae62495c0b278a4f7694c"
+ ]
+ ],
+ "params_yaml": [
+ [
+ {
+ "id": "test"
+ },
+ "params.yml:md5,d41d8cd98f00b204e9800998ecf8427e"
+ ]
+ ],
+ "versions_papermill": [
+ [
+ "QUARTONOTEBOOK",
+ "papermill",
+ "2.6.0"
+ ]
+ ],
+ "versions_quarto": [
+ [
+ "QUARTONOTEBOOK",
+ "quarto",
+ "1.7.31"
+ ]
+ ]
+ }
+ ],
+ "meta": {
+ "nf-test": "0.9.2",
+ "nextflow": "25.10.3"
+ },
+ "timestamp": "2026-01-27T09:38:07.013617"
+ },
+ "test notebook - [qmd:r]": {
+ "content": [
+ {
+ "versions_papermill": [
+ [
+ "QUARTONOTEBOOK",
+ "papermill",
+ "2.6.0"
+ ]
+ ],
+ "versions_quarto": [
+ [
+ "QUARTONOTEBOOK",
+ "quarto",
+ "1.7.31"
+ ]
+ ]
+ },
+ [
+ [
+ {
+ "id": "test"
+ },
+ "artifact.txt:md5,b10a8db164e0754105b7a99be72e3fe5"
+ ]
+ ],
+ [
+ [
+ {
+ "id": "test"
+ },
+ "params.yml:md5,8b9e438c0eb4850e20035a1222cd151e"
+ ]
+ ]
+ ],
+ "meta": {
+ "nf-test": "0.9.2",
+ "nextflow": "25.10.3"
+ },
+ "timestamp": "2026-01-27T09:36:34.251547"
+ },
+ "test notebook - parametrized - [qmd:python]": {
+ "content": [
+ {
+ "versions_papermill": [
+ [
+ "QUARTONOTEBOOK",
+ "papermill",
+ "2.6.0"
+ ]
+ ],
+ "versions_quarto": [
+ [
+ "QUARTONOTEBOOK",
+ "quarto",
+ "1.7.31"
+ ]
+ ]
+ },
+ [
+ [
+ {
+ "id": "test"
+ },
+ "artifact.txt:md5,8ddd8be4b179a529afa5f2ffae4b9858"
+ ]
+ ],
+ [
+ [
+ {
+ "id": "test"
+ },
+ "params.yml:md5,0dbcd58276179f0700e144e45d8d3c76"
+ ]
+ ]
+ ],
+ "meta": {
+ "nf-test": "0.9.2",
+ "nextflow": "25.10.3"
+ },
+ "timestamp": "2026-01-27T09:42:32.224154"
+ },
+ "test notebook - parametrized - [rmd]": {
+ "content": [
+ {
+ "versions_papermill": [
+ [
+ "QUARTONOTEBOOK",
+ "papermill",
+ "2.6.0"
+ ]
+ ],
+ "versions_quarto": [
+ [
+ "QUARTONOTEBOOK",
+ "quarto",
+ "1.7.31"
+ ]
+ ]
+ },
+ [
+ [
+ {
+ "id": "test"
+ },
+ "artifact.txt:md5,b10a8db164e0754105b7a99be72e3fe5"
+ ]
+ ],
+ [
+ [
+ {
+ "id": "test"
+ },
+ "params.yml:md5,0dbcd58276179f0700e144e45d8d3c76"
+ ]
+ ]
+ ],
+ "meta": {
+ "nf-test": "0.9.2",
+ "nextflow": "25.10.3"
+ },
+ "timestamp": "2026-01-27T09:37:47.267654"
+ },
+ "test notebook - parametrized - [ipynb]": {
+ "content": [
+ {
+ "versions_papermill": [
+ [
+ "QUARTONOTEBOOK",
+ "papermill",
+ "2.6.0"
+ ]
+ ],
+ "versions_quarto": [
+ [
+ "QUARTONOTEBOOK",
+ "quarto",
+ "1.7.31"
+ ]
+ ]
+ },
+ [
+
+ ],
+ [
+ [
+ {
+ "id": "test"
+ },
+ "params.yml:md5,0dbcd58276179f0700e144e45d8d3c76"
+ ]
+ ]
+ ],
+ "meta": {
+ "nf-test": "0.9.2",
+ "nextflow": "25.10.3"
+ },
+ "timestamp": "2026-01-27T09:38:02.01495"
+ },
+ "test notebook - [qmd:python]": {
+ "content": [
+ {
+ "versions_papermill": [
+ [
+ "QUARTONOTEBOOK",
+ "papermill",
+ "2.6.0"
+ ]
+ ],
+ "versions_quarto": [
+ [
+ "QUARTONOTEBOOK",
+ "quarto",
+ "1.7.31"
+ ]
+ ]
+ },
+ [
+ [
+ {
+ "id": "test"
+ },
+ "artifact.txt:md5,8ddd8be4b179a529afa5f2ffae4b9858"
+ ]
+ ],
+ [
+ [
+ {
+ "id": "test"
+ },
+ "params.yml:md5,8b9e438c0eb4850e20035a1222cd151e"
+ ]
+ ]
+ ],
+ "meta": {
+ "nf-test": "0.9.2",
+ "nextflow": "25.10.3"
+ },
+ "timestamp": "2026-01-27T09:36:54.331697"
+ },
+ "test notebook - parametrized - [qmd:r]": {
+ "content": [
+ {
+ "versions_papermill": [
+ [
+ "QUARTONOTEBOOK",
+ "papermill",
+ "2.6.0"
+ ]
+ ],
+ "versions_quarto": [
+ [
+ "QUARTONOTEBOOK",
+ "quarto",
+ "1.7.31"
+ ]
+ ]
+ },
+ [
+ [
+ {
+ "id": "test"
+ },
+ "artifact.txt:md5,b10a8db164e0754105b7a99be72e3fe5"
+ ]
+ ],
+ [
+ [
+ {
+ "id": "test"
+ },
+ "params.yml:md5,0dbcd58276179f0700e144e45d8d3c76"
+ ]
+ ]
+ ],
+ "meta": {
+ "nf-test": "0.9.2",
+ "nextflow": "25.10.3"
+ },
+ "timestamp": "2026-01-27T09:37:12.792978"
+ }
+}
\ No newline at end of file
diff --git a/mypy.ini b/mypy.ini
new file mode 100644
index 00000000..371b36b2
--- /dev/null
+++ b/mypy.ini
@@ -0,0 +1,14 @@
+[mypy]
+# The QC analysis scripts are vendored from the upstream nf-xenium-processing
+# repo (image_qc.py, snr_metrics.py, transcript_qc_processing.py). They are
+# maintained and validated upstream and re-synced wholesale (image_qc.py carries
+# one documented adaptation: the xenium_helpers.utils label helpers are inlined
+# so the script is self-contained). They are not type-annotated for mypy here, so
+# exclude them from type-checking — fixing mypy findings locally would fork them
+# and break future re-syncs. All other bin/ scripts are still type-checked by
+# `make typecheck`.
+exclude = (?x)(
+ ^bin/image_qc\.py$
+ | ^bin/snr_metrics\.py$
+ | ^bin/transcript_qc_processing\.py$
+ )
diff --git a/nextflow.config b/nextflow.config
index 3f91e590..a5c3bc59 100644
--- a/nextflow.config
+++ b/nextflow.config
@@ -113,6 +113,23 @@ params {
run_qc = true // whether to run the qc layer of pipeline
offtarget_probe_tracking = false // whether to run off-target probe tracking (provide probe_fasta, reference sequences, gene synonyms )
+ // image + transcript QC
+ tile_size = 35 // image QC: tile edge length in native pixels for tile-level focus/SNR analysis (passed as --roi-size)
+ stain_names = null // semicolon-separated stain/channel names for the morphology image (null = image QC defaults)
+ neg_control_prefix = 'NegControl' // prefix for Xenium negative controls (matches both NegControlProbe_* and NegControlCodeword_*); passed to transcript QC as --non-gene-prefix
+ roi_image_qc_thresholds_yaml = null // path to ROI threshold config for image QC (null = bundled conf/roi_image_qc_thresholds.yaml)
+ image_qc_gpus = 1 // image QC: cap CUDA devices used (--max-gpus); the accelerator request is set in conf/base.config
+ legacy_focus = false // image QC: use the CPU for-loop focus score instead of GPU convolution
+ image_qc_no_snr = false // image QC: skip signal-to-noise (SNR) metrics
+ image_qc_snr_no_roi_tx_table = false // image QC: do not write the per-ROI transcript SNR table
+ image_qc_snr_otsu_max_rois = null // image QC: cap the number of Otsu-selected ROIs for SNR (null = no cap)
+ image_qc_snr_no_moran = true // image QC: skip Moran's I spatial autocorrelation (default off; set false to opt in)
+ image_qc_save_dapi_maps_tiff = false // image QC: also save DAPI focus maps as TIFF
+ image_qc_stream_tiles = true // image QC: reduce each tile as computed instead of assembling full-resolution planes (--no-stream-tiles to opt out)
+ image_qc_figure_source_tables = false // image QC: write per-figure figures_source/*.csv exports (unused downstream; default off)
+ image_qc_figures = true // image QC: generate QC figures (set false / --no-figures for a metrics-only fast run)
+ image_qc_lap_sigma = 1.0 // image QC: Gaussian sigma for the Laplacian-of-Gaussian focus score
+
// utility modules
csplit_x_bins = 2 // number of tiles along the x axis (total number of bins is product of x_bins * y_bins)
csplit_y_bins = 2 // number of tiles along the y axis
@@ -283,6 +300,16 @@ profiles {
containerOptions = { "--shm-size ${task.memory.toGiga()}g" }
queue = { params.cellpose_queue ?: params.gpu_queue ?: null }
}
+ withLabel:process_gpu_qc {
+ // Must repeat base.config label properties — profile withLabel replaces, not merges.
+ // Route image QC to the GPU queue whenever an accelerator is requested (params.image_qc_gpus).
+ accelerator = { params.use_gpu ? (params.image_qc_gpus as int) : null }
+ cpus = { 30 * task.attempt }
+ memory = { 180.GB * task.attempt }
+ time = { 8.h * task.attempt }
+ containerOptions = { "--shm-size ${task.memory.toGiga()}g" }
+ queue = { params.gpu_queue ?: null }
+ }
}
}
test { includeConfig 'conf/test.config' }
@@ -378,6 +405,25 @@ manifest {
github: '@dongzehe',
contribution: ['contributor'],
orcid: '0000-0001-8259-7434'
+ ],
+ // NOTE: keep a scalar field (affiliation) LAST in each entry. nf-core
+ // tools 4.0.3 converts this block to JSON with naive bracket
+ // replacements, and an entry whose last field is a list (e.g.
+ // contribution) corrupts the closing brackets and fails lint.
+ [
+ name: 'Malwina Prater',
+ contribution: ['contributor'],
+ affiliation: 'Altos Labs'
+ ],
+ [
+ name: 'Nell Nie',
+ contribution: ['contributor'],
+ affiliation: 'Altos Labs'
+ ],
+ [
+ name: 'Christel Krueger',
+ contribution: ['contributor'],
+ affiliation: 'Altos Labs'
]
]
homePage = 'https://github.com/nf-core/spatialaxe'
diff --git a/nextflow_schema.json b/nextflow_schema.json
index 58a47276..2dc0492b 100644
--- a/nextflow_schema.json
+++ b/nextflow_schema.json
@@ -122,16 +122,6 @@
"description": "Options for the segmentation layer of the spatialaxe pipeline",
"default": "",
"properties": {
- "run_qc": {
- "type": "boolean",
- "description": "Whether to run the qc layer in the pipeline.",
- "default": true
- },
- "offtarget_probe_tracking": {
- "type": "boolean",
- "description": "Whether to run the off-target probe tracking.",
- "default": false
- },
"segmentation_refinement": {
"type": "boolean",
"description": "Whether to run refinement on the image-based segmentation methods. Runs coordinate-based methods after the initial image-based segmentation run."
@@ -155,7 +145,7 @@
"expansion_distance": {
"type": "integer",
"default": 5,
- "description": "Nuclei boundary expansion distance in µm. Default: 5 (Min: 0, Max: 15 if either boundary-stain or interior-stain are enabled and 100 if nucleus-expansion only)"
+ "description": "Nuclei boundary expansion distance in \u00b5m. Default: 5 (Min: 0, Max: 15 if either boundary-stain or interior-stain are enabled and 100 if nucleus-expansion only)"
},
"dapi_filter": {
"type": "integer",
@@ -422,6 +412,98 @@
}
}
},
+ "qc_options": {
+ "title": "QC options",
+ "type": "object",
+ "description": "Options for the image QC and transcript QC layer of the spatialaxe pipeline.",
+ "properties": {
+ "run_qc": {
+ "type": "boolean",
+ "description": "Whether to run the qc layer in the pipeline.",
+ "default": true
+ },
+ "offtarget_probe_tracking": {
+ "type": "boolean",
+ "description": "Whether to run the off-target probe tracking.",
+ "default": false
+ },
+ "tile_size": {
+ "type": "integer",
+ "minimum": 1,
+ "default": 35,
+ "description": "Image QC: tile edge length in native pixels for the tile-level focus / SNR analysis (passed as --roi-size)."
+ },
+ "stain_names": {
+ "type": "string",
+ "description": "Semicolon-separated stain/channel names for the morphology image. null = use image QC defaults."
+ },
+ "neg_control_prefix": {
+ "type": "string",
+ "description": "Prefix for Xenium negative-control features, matching both NegControlProbe_* and NegControlCodeword_* (passed to transcript QC as --non-gene-prefix; use ';' to separate multiple).",
+ "default": "NegControl"
+ },
+ "roi_image_qc_thresholds_yaml": {
+ "type": "string",
+ "format": "file-path",
+ "description": "Path to the ROI threshold config YAML for image QC. null = use the bundled conf/roi_image_qc_thresholds.yaml."
+ },
+ "image_qc_gpus": {
+ "type": "integer",
+ "minimum": 0,
+ "default": 1,
+ "description": "Image QC: number of GPUs each task requests and the cap CUDA actually uses (--max-gpus). The accelerator request is gated on use_gpu in conf/base.config."
+ },
+ "legacy_focus": {
+ "type": "boolean",
+ "description": "Image QC: use the CPU for-loop focus score instead of the GPU convolution.",
+ "default": false
+ },
+ "image_qc_no_snr": {
+ "type": "boolean",
+ "description": "Image QC: skip signal-to-noise (SNR) metrics.",
+ "default": false
+ },
+ "image_qc_snr_no_roi_tx_table": {
+ "type": "boolean",
+ "description": "Image QC: do not write the per-ROI transcript SNR table.",
+ "default": false
+ },
+ "image_qc_snr_otsu_max_rois": {
+ "type": "integer",
+ "description": "Image QC: cap the number of Otsu-selected ROIs for SNR. null = no cap."
+ },
+ "image_qc_snr_no_moran": {
+ "type": "boolean",
+ "description": "Image QC: skip Moran's I spatial autocorrelation (default on; set false to opt in to Moran).",
+ "default": true
+ },
+ "image_qc_save_dapi_maps_tiff": {
+ "type": "boolean",
+ "description": "Image QC: also save DAPI focus maps as TIFF.",
+ "default": false
+ },
+ "image_qc_stream_tiles": {
+ "type": "boolean",
+ "default": true,
+ "description": "Image QC: reduce each tile as it is computed instead of assembling full-resolution pixel planes. Set false (--no-stream-tiles) to restore the plane-based path."
+ },
+ "image_qc_figure_source_tables": {
+ "type": "boolean",
+ "default": false,
+ "description": "Image QC: write the per-figure figures_source/*.csv source-data exports (unused downstream; default off)."
+ },
+ "image_qc_figures": {
+ "type": "boolean",
+ "default": true,
+ "description": "Image QC: generate QC figures. Set false (--no-figures) for a metrics-only fast run; metric/JSON/parquet outputs are always produced."
+ },
+ "image_qc_lap_sigma": {
+ "type": "number",
+ "description": "Image QC: Gaussian sigma for the Laplacian-of-Gaussian focus score.",
+ "default": 1.0
+ }
+ }
+ },
"institutional_config_options": {
"title": "Institutional config options",
"type": "object",
@@ -598,6 +680,9 @@
{
"$ref": "#/$defs/segmentation_options"
},
+ {
+ "$ref": "#/$defs/qc_options"
+ },
{
"$ref": "#/$defs/institutional_config_options"
},
diff --git a/subworkflows/local/image_qc/main.nf b/subworkflows/local/image_qc/main.nf
new file mode 100644
index 00000000..182ebd1c
--- /dev/null
+++ b/subworkflows/local/image_qc/main.nf
@@ -0,0 +1,54 @@
+//
+// IMAGE_QC: image-based quality control for a Xenium bundle.
+//
+// Runs the image QC analysis (focus / SNR / morphology metrics + figures) and
+// renders an HTML report from the analysis outputs with the nf-core
+// QUARTONOTEBOOK module. Versions are reported via the `versions` topic channel
+// by each module (collected centrally by the pipeline), so this subworkflow
+// does not thread versions through emit.
+//
+
+include { IMAGE_QC_ANALYSIS as ANALYSIS } from '../../../modules/local/image_qc/main'
+include { QUARTONOTEBOOK as REPORT } from '../../../modules/nf-core/quartonotebook/main'
+
+workflow IMAGE_QC {
+ take:
+ ch_input // channel: [ val(meta), val(parameters), path(input_files) ]
+ notebook // path: the image QC report .qmd
+
+ main:
+ // ROI threshold config: user-provided path, else the bundled default.
+ ch_roi_yaml = channel.fromPath(
+ params.roi_image_qc_thresholds_yaml ?: "${projectDir}/conf/roi_image_qc_thresholds.yaml",
+ checkIfExists: true,
+ )
+
+ // Single process handles both ROI-based and cell-level QC
+ ANALYSIS(ch_input, ch_roi_yaml.first())
+
+ // Render the report with QUARTONOTEBOOK. Its four inputs are separate
+ // channels paired by emission order, so all per-sample channels are derived
+ // from the same upstream channel to guarantee alignment. The analysis
+ // output directory and the ROI YAML are staged as input files; the
+ // notebook's parameters cell receives their staged names via params.yml.
+ ch_report = ANALYSIS.out.outdir.combine(ch_roi_yaml.first()).map { meta, outdir, roi_yaml ->
+ def parameters = [
+ INDIR : outdir.name,
+ SAMPLE_NAME : meta.id,
+ XENIUM_BUNDLE : meta.samplesheet_xenium_bundle ?: '',
+ SAMPLE_PUBLISHED_OUTDIR: meta.samplesheet_xenium_bundle ? "${params.outdir}/${params.mode}/qc/image_qc" : '',
+ ROI_THRESHOLDS_YAML : roi_yaml.name,
+ ]
+ [meta, parameters, [outdir, roi_yaml]]
+ }
+ REPORT(
+ ch_report.map { meta, parameters, input_files -> [meta, notebook] },
+ ch_report.map { meta, parameters, input_files -> parameters },
+ ch_report.map { meta, parameters, input_files -> input_files },
+ [],
+ )
+
+ emit:
+ outdir = ANALYSIS.out.outdir // channel: [ val(meta), path(outdir) ]
+ report = REPORT.out.html // channel: [ val(meta), path(html) ]
+}
diff --git a/subworkflows/local/image_qc/meta.yml b/subworkflows/local/image_qc/meta.yml
new file mode 100644
index 00000000..f4fd4bc6
--- /dev/null
+++ b/subworkflows/local/image_qc/meta.yml
@@ -0,0 +1,35 @@
+# yaml-language-server: $schema=https://raw.githubusercontent.com/nf-core/modules/master/subworkflows/yaml-schema.json
+name: "image_qc"
+description: Run image-based quality control on a Xenium bundle and render an HTML report
+keywords:
+ - spatial
+ - xenium
+ - quality control
+ - image
+ - QC
+ - quarto
+components:
+ - image_qc
+ - quartonotebook
+input:
+ - ch_input:
+ description: |
+ input channel for the image QC analysis
+ Structure: [ val(meta), val(parameters), path(input_files) ]
+ - notebook:
+ description: |
+ Quarto notebook (.qmd) used to render the image QC report
+ Structure: path(qmd_file)
+output:
+ - outdir:
+ description: |
+ the image QC analysis output directory (metrics and figures)
+ Structure: [ val(meta), path(outdir) ]
+ - report:
+ description: |
+ the rendered image QC HTML report
+ Structure: [ val(meta), path(html) ]
+authors:
+ - "@an-altosian"
+maintainers:
+ - "@an-altosian"
diff --git a/subworkflows/local/qc/main.nf b/subworkflows/local/qc/main.nf
new file mode 100644
index 00000000..2acf0c0d
--- /dev/null
+++ b/subworkflows/local/qc/main.nf
@@ -0,0 +1,68 @@
+//
+// QC: quality-control layer for a Xenium bundle.
+//
+// Thin wrapper that runs the two QC subworkflows — image QC and transcript QC —
+// on a reported bundle and emits their analysis directories and HTML reports.
+// It deliberately holds no pre/post-segmentation logic and no MultiQC: when and
+// on which bundle QC runs is decided by the caller (workflows/spatialaxe.nf).
+//
+
+include { IMAGE_QC } from '../image_qc/main'
+include { TRANSCRIPT_QC } from '../transcript_qc/main'
+
+workflow QC {
+ take:
+ ch_bundle // channel: [ val(meta), path(xenium_bundle) ]
+
+ main:
+ // Shared per-sample meta for both QC reports.
+ ch_qc_meta = ch_bundle.map { meta, bundle ->
+ [
+ [
+ id: meta.id,
+ samplesheet_xenium_bundle: bundle.toString(),
+ xenium_bundle_source: meta.xenium_bundle_source ?: '',
+ cropped: meta.cropped ?: false,
+ ],
+ bundle,
+ ]
+ }
+
+ // ---------------------------- Image QC ------------------------------
+ imageqc_input_ch = ch_qc_meta.map { meta, bundle ->
+ tuple(
+ meta,
+ // parameters (flat key/value list, collated by the module)
+ ["STAIN_NAMES", params.stain_names],
+ // input files
+ [bundle],
+ )
+ }
+ image_qc_notebook = file("${projectDir}/assets/notebooks/xenium_image_qc_report.qmd", checkIfExists: true)
+ IMAGE_QC(
+ imageqc_input_ch,
+ image_qc_notebook,
+ )
+
+ // -------------------------- Transcript QC ---------------------------
+ transcriptqc_input_ch = ch_qc_meta.map { meta, bundle ->
+ tuple(
+ meta,
+ // parameters (NON_GENE_PREFIX is the key the module's arg builder reads)
+ ["NON_GENE_PREFIX", params.neg_control_prefix],
+ // input files
+ [bundle],
+ )
+ }
+ transcript_qc_notebook = file("${projectDir}/assets/notebooks/transcript_qc.qmd", checkIfExists: true)
+ TRANSCRIPT_QC(
+ transcriptqc_input_ch,
+ transcript_qc_notebook,
+ )
+
+ emit:
+ image_qc_outdir = IMAGE_QC.out.outdir // channel: [ val(meta), path(outdir) ]
+ image_qc_report = IMAGE_QC.out.report // channel: [ val(meta), path(html) ]
+ transcript_qc_outdir = TRANSCRIPT_QC.out.outdir // channel: [ val(meta), path(outdir) ]
+ transcript_qc_report = TRANSCRIPT_QC.out.report // channel: [ val(meta), path(html) ]
+}
diff --git a/subworkflows/local/qc/meta.yml b/subworkflows/local/qc/meta.yml
new file mode 100644
index 00000000..c50f374b
--- /dev/null
+++ b/subworkflows/local/qc/meta.yml
@@ -0,0 +1,39 @@
+# yaml-language-server: $schema=https://raw.githubusercontent.com/nf-core/modules/master/subworkflows/yaml-schema.json
+name: "qc"
+description: Quality-control layer for a Xenium bundle; runs image QC and transcript QC and emits their outputs
+keywords:
+ - spatial
+ - xenium
+ - quality control
+ - image
+ - transcript
+ - QC
+components:
+ - image_qc
+ - transcript_qc
+input:
+ - ch_bundle:
+ description: |
+ input channel with the Xenium bundle to run QC on
+ Structure: [ val(meta), path(xenium_bundle) ]
+output:
+ - image_qc_outdir:
+ description: |
+ the image QC analysis output directory (metrics and figures)
+ Structure: [ val(meta), path(outdir) ]
+ - image_qc_report:
+ description: |
+ the rendered image QC HTML report
+ Structure: [ val(meta), path(html) ]
+ - transcript_qc_outdir:
+ description: |
+ the transcript QC analysis output directory (metrics and figures)
+ Structure: [ val(meta), path(outdir) ]
+ - transcript_qc_report:
+ description: |
+ the rendered transcript QC HTML report
+ Structure: [ val(meta), path(html) ]
+authors:
+ - "@an-altosian"
+maintainers:
+ - "@an-altosian"
diff --git a/subworkflows/local/transcript_qc/main.nf b/subworkflows/local/transcript_qc/main.nf
new file mode 100644
index 00000000..3003847a
--- /dev/null
+++ b/subworkflows/local/transcript_qc/main.nf
@@ -0,0 +1,46 @@
+//
+// TRANSCRIPT_QC: transcript / molecule-level quality control for a Xenium bundle.
+//
+// Runs the transcript QC analysis (per-transcript and per-cell QC metrics +
+// figures) and renders an HTML report from the analysis outputs with the
+// nf-core QUARTONOTEBOOK module. Versions are reported via the `versions`
+// topic channel by each module.
+//
+
+include { TRANSCRIPT_QC_PROCESSING as ANALYSIS } from '../../../modules/local/transcript_qc/main'
+include { QUARTONOTEBOOK as REPORT } from '../../../modules/nf-core/quartonotebook/main'
+
+workflow TRANSCRIPT_QC {
+ take:
+ ch_input // channel: [ val(meta), val(parameters), path(input_files) ]
+ notebook // path: the transcript QC report .qmd
+
+ main:
+ // Run computational processing
+ ANALYSIS(ch_input)
+
+ // Render the report with QUARTONOTEBOOK. Its four inputs are separate
+ // channels paired by emission order, so all per-sample channels are derived
+ // from the same upstream channel to guarantee alignment. The analysis
+ // output directory is staged as an input file; the notebook's parameters
+ // cell receives its staged name via params.yml.
+ ch_report = ANALYSIS.out.outdir.map { meta, outdir ->
+ def parameters = [
+ INDIR : outdir.name,
+ SAMPLE_NAME : meta.id,
+ XENIUM_BUNDLE : meta.samplesheet_xenium_bundle ?: '',
+ SAMPLE_PUBLISHED_OUTDIR: meta.samplesheet_xenium_bundle ? "${params.outdir}/${params.mode}/qc/transcript_qc" : '',
+ ]
+ [meta, parameters, outdir]
+ }
+ REPORT(
+ ch_report.map { meta, parameters, outdir -> [meta, notebook] },
+ ch_report.map { meta, parameters, outdir -> parameters },
+ ch_report.map { meta, parameters, outdir -> outdir },
+ [],
+ )
+
+ emit:
+ outdir = ANALYSIS.out.outdir // channel: [ val(meta), path(outdir) ]
+ report = REPORT.out.html // channel: [ val(meta), path(html) ]
+}
diff --git a/subworkflows/local/transcript_qc/meta.yml b/subworkflows/local/transcript_qc/meta.yml
new file mode 100644
index 00000000..eb416511
--- /dev/null
+++ b/subworkflows/local/transcript_qc/meta.yml
@@ -0,0 +1,36 @@
+# yaml-language-server: $schema=https://raw.githubusercontent.com/nf-core/modules/master/subworkflows/yaml-schema.json
+name: "transcript_qc"
+description: Run transcript / molecule-level quality control on a Xenium bundle and render an HTML report
+keywords:
+ - spatial
+ - xenium
+ - quality control
+ - transcript
+ - molecule
+ - QC
+ - quarto
+components:
+ - transcript_qc
+ - quartonotebook
+input:
+ - ch_input:
+ description: |
+ input channel for the transcript QC analysis
+ Structure: [ val(meta), val(parameters), path(input_files) ]
+ - notebook:
+ description: |
+ Quarto notebook (.qmd) used to render the transcript QC report
+ Structure: path(qmd_file)
+output:
+ - outdir:
+ description: |
+ the transcript QC analysis output directory (metrics and figures)
+ Structure: [ val(meta), path(outdir) ]
+ - report:
+ description: |
+ the rendered transcript QC HTML report
+ Structure: [ val(meta), path(html) ]
+authors:
+ - "@an-altosian"
+maintainers:
+ - "@an-altosian"
diff --git a/tests/test_transcript_qc_math.py b/tests/test_transcript_qc_math.py
new file mode 100644
index 00000000..b40e1151
--- /dev/null
+++ b/tests/test_transcript_qc_math.py
@@ -0,0 +1,124 @@
+"""Unit tests for the inlined pure-math helpers in bin/transcript_qc_processing.py.
+
+The script imports scanpy (and other heavy deps) at module top level, which is
+not available in the test environment, so we cannot ``import`` it directly.
+Instead we parse the source with ``ast``, extract the two pure functions by
+name, and ``exec`` only those function definitions into a controlled namespace
+that provides ``np`` (numpy) and ``pd`` (pandas). This exercises the REAL
+inlined code without triggering the module-level heavy imports.
+"""
+
+import ast
+from pathlib import Path
+
+import numpy as np
+import pandas as pd
+import pytest
+
+SCRIPT_PATH = (
+ Path(__file__).resolve().parent.parent / "bin" / "transcript_qc_processing.py"
+)
+
+FUNCTIONS_TO_EXTRACT = ("calculate_noise_bound", "estimate_min_mols_per_cell")
+
+
+def _load_inlined_functions():
+ """Extract the two target functions from the script source and exec them.
+
+ Returns a namespace dict containing the compiled functions plus ``np``/``pd``.
+ """
+ source = SCRIPT_PATH.read_text()
+ tree = ast.parse(source)
+
+ namespace = {"np": np, "pd": pd}
+ found = {}
+ for node in tree.body:
+ if isinstance(node, ast.FunctionDef) and node.name in FUNCTIONS_TO_EXTRACT:
+ segment = ast.get_source_segment(source, node)
+ assert segment is not None, f"could not extract source for {node.name}"
+ exec(segment, namespace)
+ found[node.name] = namespace[node.name]
+
+ missing = set(FUNCTIONS_TO_EXTRACT) - set(found)
+ assert not missing, f"functions not found in script: {missing}"
+ return namespace
+
+
+@pytest.fixture(scope="module")
+def funcs():
+ ns = _load_inlined_functions()
+ return {name: ns[name] for name in FUNCTIONS_TO_EXTRACT}
+
+
+# ---------------------------------------------------------------------------
+# estimate_min_mols_per_cell
+# ---------------------------------------------------------------------------
+
+
+def test_estimate_min_mols_returns_int(funcs):
+ estimate_min_mols_per_cell = funcs["estimate_min_mols_per_cell"]
+ rng = np.random.default_rng(0)
+ # An obviously-high distribution centred around ~500 molecules/cell.
+ data = rng.normal(loc=500, scale=50, size=5000).clip(min=1)
+ result = estimate_min_mols_per_cell(data)
+ assert isinstance(result, int)
+ assert result >= 10
+
+
+def test_estimate_min_mols_respects_min_value_floor(funcs):
+ estimate_min_mols_per_cell = funcs["estimate_min_mols_per_cell"]
+ # A tiny, low-count distribution: the computed threshold would be well
+ # below the floor, so the floor must win.
+ data = np.array([0, 0, 1, 1, 2])
+ result = estimate_min_mols_per_cell(data, min_value=10)
+ assert result == 10
+
+
+def test_estimate_min_mols_custom_min_value(funcs):
+ estimate_min_mols_per_cell = funcs["estimate_min_mols_per_cell"]
+ data = np.array([0, 1, 2, 3])
+ # A higher floor is respected on a distribution whose estimate is tiny.
+ assert estimate_min_mols_per_cell(data, min_value=42) == 42
+
+
+def test_estimate_min_mols_high_distribution_exceeds_floor(funcs):
+ estimate_min_mols_per_cell = funcs["estimate_min_mols_per_cell"]
+ rng = np.random.default_rng(1)
+ # Bimodal: a big population of high-count cells makes the mode high enough
+ # that the returned value comfortably exceeds the min_value floor.
+ high = rng.normal(loc=1000, scale=30, size=10000).clip(min=1)
+ result = estimate_min_mols_per_cell(high, min_value=10)
+ assert result > 10
+
+
+# ---------------------------------------------------------------------------
+# calculate_noise_bound
+# ---------------------------------------------------------------------------
+
+
+def test_calculate_noise_bound_empty_series(funcs):
+ calculate_noise_bound = funcs["calculate_noise_bound"]
+ result = calculate_noise_bound(pd.Series([], dtype=float))
+ assert result == (0, 0)
+
+
+def test_calculate_noise_bound_nonempty_series(funcs):
+ calculate_noise_bound = funcs["calculate_noise_bound"]
+ rng = np.random.default_rng(2)
+ # Per-feature molecule counts for negative-control probes.
+ counts = pd.Series(rng.integers(low=5, high=500, size=200))
+ lb, ub = calculate_noise_bound(counts)
+ assert lb > 0
+ assert ub > 0
+ assert lb < ub
+
+
+def test_calculate_noise_bound_quantile_widens_bounds(funcs):
+ calculate_noise_bound = funcs["calculate_noise_bound"]
+ rng = np.random.default_rng(3)
+ counts = pd.Series(rng.integers(low=10, high=1000, size=300))
+ lb_narrow, ub_narrow = calculate_noise_bound(counts, quant=0.90)
+ lb_wide, ub_wide = calculate_noise_bound(counts, quant=0.999)
+ # A higher quantile pushes the bounds further apart.
+ assert ub_wide > ub_narrow
+ assert lb_wide < lb_narrow
diff --git a/workflows/spatialaxe.nf b/workflows/spatialaxe.nf
index 3ff05db9..e3689ee5 100644
--- a/workflows/spatialaxe.nf
+++ b/workflows/spatialaxe.nf
@@ -45,6 +45,7 @@ include { SPATIALDATA_WRITE_META_MERGE } from '../subworkflo
// qc layer subworkflows
include { OPT_FLIP_TRACK_STAT } from '../subworkflows/local/opt_flip_track_stat/main'
+include { QC } from '../subworkflows/local/qc/main'
/*
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
@@ -610,6 +611,12 @@ workflow SPATIALAXE {
// check to run the qc layer
if (mode == 'qc' || run_qc) {
+ // image QC + transcript QC on the validated Xenium bundle
+ QC(ch_bundle_path)
+
+ // collect the rendered QC reports so the MultiQC layer can pick them up
+ ch_qc_reports = QC.out.image_qc_report.mix(QC.out.transcript_qc_report)
+
if (offtarget_probe_tracking) {
// run off-target probe tracking