diff --git a/.github/workflows/nf-test.yml b/.github/workflows/nf-test.yml index 8abe0187..9d8217d8 100644 --- a/.github/workflows/nf-test.yml +++ b/.github/workflows/nf-test.yml @@ -64,7 +64,10 @@ jobs: runs-on: # use self-hosted runners - runs-on=${{ github.run_id }}-nf-test - runner=4cpu-linux-x64 - - volume=80gb + # 80 GB is no longer enough: a shard that pulls xeniumranger (which bundles + # cellranger) alongside the image QC and transcript QC containers ran out of + # disk with `docker: failed to register layer: no space left on device`. + - volume=120gb strategy: fail-fast: false matrix: diff --git a/.github/workflows/python-tests.yml b/.github/workflows/python-tests.yml new file mode 100644 index 00000000..60f0cf0d --- /dev/null +++ b/.github/workflows/python-tests.yml @@ -0,0 +1,56 @@ +name: Python tests +# Runs the Python unit tests for the QC analysis scripts. +# +# The nf-test suites exercise the Nextflow layer, and the pipeline-level QC +# tests run with `-stub`, so without this workflow none of the QC Python code is +# ever executed in CI. +# +# The scripts import their heavy dependencies at module scope, so the tests need +# a real runtime environment (matplotlib, tifffile, scanpy, ...), not a linting +# environment. The env is therefore built from the QC modules' own +# environment.yml files plus the test-only extras, which keeps every runtime pin +# declared in exactly one place. +on: + push: + branches: + - dev + - master + pull_request: + release: + types: [published] + +env: + NXF_ANSI_LOG: false + +concurrency: + group: "${{ github.workflow }}-${{ github.event.pull_request.number || github.ref }}" + cancel-in-progress: true + +jobs: + pytest: + name: Run Python tests + runs-on: ubuntu-latest + steps: + - name: Check out pipeline code + uses: actions/checkout@11bd71901bbe5b1630ceea73d27597364c9af683 # v4 + + - name: Set up the QC runtime environment + uses: mamba-org/setup-micromamba@0dea6379afdaffa5d528b3d1dabc45da37f443fc # v2 + with: + environment-name: qc-tests + # One self-contained environment, not a merge of the two QC module + # environments: those cannot be co-installed (image QC pins + # pyarrow 21, while transcript QC's anndata requires pyarrow <21), so + # the test environment declares its own coherent set. + environment-file: bin/tests/environment.yml + cache-environment: true + + - name: Show environment + shell: bash -el {0} + run: | + python --version + python -c "import numpy, pandas, matplotlib, tifffile, scanpy; print('runtime imports OK')" + + - name: Run pytest + shell: bash -el {0} + run: pytest tests/ bin/tests/ -v -n auto diff --git a/.gitignore b/.gitignore index 2ef7dde1..66a3ab68 100644 --- a/.gitignore +++ b/.gitignore @@ -10,3 +10,5 @@ null/ .nf-test/ .nf-test.log .nf-test-* +.mypy_cache/ +.vscode/ diff --git a/.nf-core.yml b/.nf-core.yml index 352e6ed3..4f513fb3 100644 --- a/.nf-core.yml +++ b/.nf-core.yml @@ -10,6 +10,11 @@ lint: - docs/images/nf-core-spatialaxe_logo_dark.png - docs/images/nf-core-spatialaxe_logo_light.png - .github/PULL_REQUEST_TEMPLATE.md + # The QC report notebooks use doubled braces as Python f-string escapes (they + # emit pandoc callout-note divs), which this check reads as leftover Jinja. + template_strings: + - bin/xenium_image_qc_report.qmd + - bin/transcript_qc.qmd nf_core_version: 4.0.3 repository_type: pipeline template: diff --git a/.vscode/settings.json b/.vscode/settings.json deleted file mode 100644 index a33b527c..00000000 --- a/.vscode/settings.json +++ /dev/null @@ -1,3 +0,0 @@ -{ - "markdown.styles": ["public/vscode_markdown.css"] -} diff --git a/CHANGELOG.md b/CHANGELOG.md index 04de5b1c..b0c8e793 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -19,6 +19,10 @@ Initial release of nf-core/spatialaxe, created with the [nf-core](https://nf-co. - samplesheet redefinition: `sample,bundle,image,annotation,stainings`, samplesheet allows for two additional optional columns `annotation,stainings` that are useful for the QC subworkflow. - `spatialdata_write_meta_merge/main.nf`: Change to subworkflow to account for proper `qc` mode. - Change to `bin/spatialdata_write.py`: Adding an `all` mode to set all available features to `True`, which is important for QC. +- **Image QC and transcript QC**, ported from the internal nf-xenium-processing repository. Adds the `QC` subworkflow (`subworkflows/local/qc/`) wrapping `IMAGE_QC` and `TRANSCRIPT_QC` (`subworkflows/local/image_qc/`, `subworkflows/local/transcript_qc/`), the `image_qc` and `transcript_qc` local modules, and their analysis scripts, report notebooks and threshold configs in `bin/`. Image QC computes focus, SNR and morphology metrics; transcript QC computes per-transcript, per-cell and per-FoV metrics with a bounded-memory streaming reader. Each renders an HTML report with the nf-core `quarto/notebook` module. +- 17 new QC parameters under the `qc_options` schema group, including `image_qc_gpus` (image QC is GPU-optional; the value is both the accelerator request and the device cap passed to the script), `tile_size`, `neg_control_prefix`, and figure/SNR/streaming toggles. +- `conf/base.config`: new `process_gpu_qc` label for the optionally-GPU image QC step, routed to the GPU queue in the `aws` profile. +- `.github/workflows/python-tests.yml` plus `bin/tests/`: the QC analysis scripts' Python unit tests, which were previously never executed in CI. Run in a runtime environment (`bin/tests/environment.yml`) because the scripts import their heavy dependencies at module scope. ### `Fixed` diff --git a/bin/baysor_create_dataset.py b/bin/baysor_create_dataset.py index 4e5a263a..c3feaa29 100755 --- a/bin/baysor_create_dataset.py +++ b/bin/baysor_create_dataset.py @@ -13,18 +13,19 @@ from pathlib import Path -class BaysorPreview(): +class BaysorPreview: """ Utility class to generate baysor preview dataset """ + @staticmethod def generate_dataset( - transcripts: Path, - sampled_transcripts: Path, - sample_fraction: float = 0.3, - random_state: int = 42, - prefix: str = "" - ) -> None: + transcripts: Path, + sampled_transcripts: Path, + sample_fraction: float = 0.3, + random_state: int = 42, + prefix: str = "", + ) -> None: """ Reads a csv file & randomly samples a fraction of rows, and writes the result to a .csv file. @@ -40,9 +41,10 @@ def generate_dataset( random.seed(random_state) output_path = f"{prefix}/{sampled_transcripts}" os.makedirs(os.path.dirname(output_path), exist_ok=True) - with open(transcripts, mode='rt', newline='') as infile, \ - open(output_path, mode='wt', newline='') as outfile: - + with ( + open(transcripts, mode="rt", newline="") as infile, + open(output_path, mode="wt", newline="") as outfile, + ): reader = csv.reader(infile) writer = csv.writer(outfile) @@ -66,27 +68,25 @@ def main() -> None: description="Create sampled dataset for Baysor preview" ) parser.add_argument( - "--transcripts", required=True, - help="Path to transcripts CSV file" - ) - parser.add_argument( - "--sample-fraction", required=True, type=float, - help="Fraction of rows to sample" + "--transcripts", required=True, help="Path to transcripts CSV file" ) parser.add_argument( - "--prefix", required=True, - help="Output directory prefix" + "--sample-fraction", + required=True, + type=float, + help="Fraction of rows to sample", ) + parser.add_argument("--prefix", required=True, help="Output directory prefix") args = parser.parse_args() - sampled_transcripts = "sampled_transcripts.csv" + sampled_transcripts = Path("sampled_transcripts.csv") # generate dataset BaysorPreview.generate_dataset( - transcripts=args.transcripts, + transcripts=Path(args.transcripts), sampled_transcripts=sampled_transcripts, sample_fraction=args.sample_fraction, - prefix=args.prefix + prefix=args.prefix, ) return None diff --git a/bin/image_qc.py b/bin/image_qc.py new file mode 100755 index 00000000..26ce983c --- /dev/null +++ b/bin/image_qc.py @@ -0,0 +1,14331 @@ +#!/usr/bin/env python3 +""" +Combined Image QC Script + +This script combines three image QC scripts into one: +- image_qc_roi_processing.py: GPU-accelerated tile/pixel-level focus maps +- image_qc_processing.py: Cell-based QC with Quarto figures +- image_qc_mapping_to_cells.py: Maps tile results to cells + +Author: Hanneke Okkenhaug, Malwina Prater +""" + +from __future__ import annotations + +import logging +from dataclasses import dataclass +import math +import os +import queue +import sys +import threading +import time +import traceback +import warnings +import click +import json +import numpy as np +import pandas as pd +import tifffile +import zarr +from pathlib import Path +import matplotlib.pyplot as plt +from napari_skimage_regionprops import regionprops_table +from skimage.segmentation import clear_border +from skimage import measure, color, morphology +from skimage.filters import apply_hysteresis_threshold, threshold_otsu +import napari_simpleitk_image_processing as nsitk +import seaborn as sns +from sklearn.preprocessing import RobustScaler +from tifffile import imread +import multiprocessing +import multiprocessing.connection +import shutil +from concurrent.futures import ThreadPoolExecutor, as_completed +from typing import Any, Dict, Optional +from numpy.typing import NDArray +from scipy.ndimage import gaussian_laplace as scipy_gaussian_laplace +from scipy.ndimage import laplace as scipy_laplace +from scipy.ndimage import uniform_filter as scipy_uniform_filter +from scipy import ndimage +from sklearn.mixture import GaussianMixture + +import snr_metrics + +# Set matplotlib to use a non-interactive backend +import matplotlib + +matplotlib.use("Agg") + +# GPU backend detection (CuPy) +try: + import cupy as cp # type: ignore[import-untyped] + import cupyx.scipy.ndimage # type: ignore[import-untyped] # noqa: F401 (binds `cupyx` for warmup) + from cupyx.scipy.ndimage import laplace as cupy_laplace # type: ignore[import-untyped] + from cupyx.scipy.ndimage import uniform_filter as cupy_uniform_filter # type: ignore[import-untyped] + + def cupy_gaussian_laplace(image, sigma): # type: ignore[misc] + """CuPy Laplacian of Gaussian: Gaussian smooth then Laplacian.""" + from cupyx.scipy.ndimage import gaussian_filter as _gf # type: ignore[import-untyped] + + return cupy_laplace(_gf(image, sigma=sigma)) + + HAS_CUPY = True +# Broad guard is deliberate: cupy-cuda12x is installed in the container, so a +# host with no NVIDIA driver fails at import with RuntimeError / +# CUDARuntimeError, not ImportError — catching ImportError alone would abort +# the script instead of taking the intended CPU-only fallback. +except Exception: + HAS_CUPY = False + +# --------------------------------------------------------------------------- +# Segmentation-software label helpers, vendored VERBATIM from upstream +# xenium_helpers.utils (bin/xenium_helpers/src/xenium_helpers/utils.py at +# nf-xenium-processing dev HEAD 5e35cae). This pipeline does not ship the +# xenium_helpers package, so the transitive closure needed by image_qc.py is +# inlined here to keep the script self-contained. +# Re-sync note: if upstream xenium_helpers.utils changes any of these symbols, +# re-copy this whole block verbatim rather than hand-patching it. +# --------------------------------------------------------------------------- + +# Display names for pipeline resegmentation tools (raw params.segmentation +# value -> human-readable). Used as the name-only fallback when no parsed tool +# version is available. +SEGMENTATION_PRETTY = { + "cellpose": "Cellpose", + "cellpose_baysor": "Cellpose + Baysor", + "proseg": "Proseg", + "segger": "Segger", +} + +# Per-method component tools as (display name, versions.yml key) pairs. The key +# is the tool name as it appears inside the segmentation modules' versions.yml +# (e.g. ``cellpose: 3.0.6``). Order defines how multi-tool labels read. +SEGMENTATION_TOOL_KEYS = { + "cellpose": [("Cellpose", "cellpose")], + "cellpose_baysor": [("Cellpose", "cellpose"), ("Baysor", "baysor")], + "proseg": [("Proseg", "proseg")], + "segger": [("Segger", "segger")], +} + + +def _tool_label(display: str, key: str, tool_versions: Optional[Dict[str, str]]) -> str: + """``"Cellpose"`` + version -> ``"Cellpose v3.0.6"`` (name only if absent).""" + version = (tool_versions or {}).get(key) + return f"{display} v{version}" if version else display + + +def read_xenium_analysis_sw_version(bundle_dir) -> Optional[str]: + """Read ``analysis_sw_version`` from ``experiment.xenium`` (e.g. + ``"xenium-4.0.1.0"``). Returns ``None`` on missing file, missing key, or + malformed JSON. Mirrors ``read_xenium_pixel_size_um`` in bin/snr_metrics.py. + """ + exp = Path(bundle_dir) / "experiment.xenium" + if not exp.is_file(): + return None + try: + with open(exp, encoding="utf-8") as f: + meta = json.load(f) + version = meta.get("analysis_sw_version") + except (OSError, ValueError, TypeError, json.JSONDecodeError): + return None + if not version or not isinstance(version, str): + return None + return version + + +def _parse_xenium_version(analysis_sw_version: Optional[str]) -> Optional[str]: + """``"xenium-4.0.1.0"`` -> ``"4.0.1"`` (major.minor.patch). Returns ``None`` + if no leading numeric components can be parsed.""" + if not analysis_sw_version: + return None + tail = analysis_sw_version.split("-", 1)[-1] # drop a 'xenium-' style prefix + nums = [] + for part in tail.split("."): + if part.isdigit(): + nums.append(part) + else: + break + if not nums: + return None + return ".".join(nums[:3]) + + +def read_xenium_major_version(bundle_dir) -> Optional[int]: + """Major XOA version for a bundle, read from ``experiment.xenium`` + (``"xenium-4.0.1.0"`` -> ``4``). Returns ``None`` when the file is absent or + the version cannot be parsed. Used to pick XOA-version-specific QC floors + (e.g. intensity gates differ sharply between XOA 3.x and 4.0).""" + parsed = _parse_xenium_version(read_xenium_analysis_sw_version(bundle_dir)) + if not parsed: + return None + first = parsed.split(".", 1)[0] + return int(first) if first.isdigit() else None + + +def resolve_segmentation_software( + bundle_dir, + pipeline_segmentation: str = "skip", + is_resegmented: bool = False, + tool_versions: Optional[Dict[str, str]] = None, +) -> str: + """Human-readable label for the segmentation software that produced the + bundle a QC report describes. + + - Un-resegmented / pre-seg / ``skip``: the onboard analysis version from the + bundle's ``experiment.xenium`` -> ``"Xenium Onboard Analysis v4.0.1"``. + - Pipeline ``xr`` resegmentation: the reseg bundle's own + ``analysis_sw_version`` -> ``"Xenium Ranger v4.0.1 (resegmentation)"``. + - Other pipeline tools (cellpose / cellpose_baysor / proseg / segger): the + tool name plus its version from ``tool_versions`` (parsed from the + segmentation ``versions.yml``), e.g. ``"Cellpose v3.0.6"`` or + ``"Cellpose v3.0.6 + Baysor v0.6.2"``. Falls back to name-only when the + version is unavailable. Their reseg bundle is packaged via ``xeniumranger + import-segmentation``, so its ``experiment.xenium`` would mislabel them as + Xenium Ranger; the pipeline tool name is authoritative here. + """ + seg = (pipeline_segmentation or "skip").strip() + parsed = _parse_xenium_version(read_xenium_analysis_sw_version(bundle_dir)) + + if not is_resegmented or seg == "skip": + if parsed: + return f"Xenium Onboard Analysis v{parsed}" + return "Xenium Onboard Analysis (version unknown)" + + if seg == "xr": + if parsed: + return f"Xenium Ranger v{parsed} (resegmentation)" + return "Xenium Ranger (resegmentation)" + + components = SEGMENTATION_TOOL_KEYS.get(seg) + if components: + return " + ".join( + _tool_label(display, key, tool_versions) for display, key in components + ) + return SEGMENTATION_PRETTY.get(seg, seg) + + +# --------------------------- end vendored block ---------------------------- + +# Xenium pixel size in micrometers (used for coordinate conversions) +XENIUM_PIXEL_SIZE_UM = 0.2125 + +# Default CCFS threshold for classifying cells as low nuclear texture quality. +# CCFS measures per-cell nuclear contrast (local_var/local_mean), NOT optical blur. +# Calibrated on 5 samples (2026-04-02): cells below 0.02 lose >50% transcripts +# relative to top-50% CCFS cells in lung/liver. Brain tissue is confounded by +# nucleus size (large neurons score lower) — interpret with caution. +DEFAULT_CCFS_LOW_TEXTURE_THRESHOLD = 0.02 + +# sensitivity for blurred cell detection (catches more background regions) +ROI_INTENSITY_THRESHOLD = 100.0 +# 2026-06-26: tissue-tile gate lowered 0.5 -> 0.2. The background-from-nonzero mask is +# tighter (un-flooded), so real tissue tiles are only thinly covered; 0.5 discarded them. +# See plans/2026-06-26_PLAN_tissue-mask-recalibration.md. +ROI_MIN_TISSUE_COVERAGE_FOR_INTENSITY_QC = 0.2 +# Per-tile DAPI floor for the usable_tissue low-intensity term ONLY (decoupled from the +# is_low_intensity column, which stays at ROI_INTENSITY_THRESHOLD for the 1D-GMM tissue +# scope). Lowered to 50 to match the relaxed DAPI intensity QC (channels.DAPI. +# intensity_critical_v3/v4 = 50); a dim-but-real tissue tile should not count as unusable +# on absolute brightness alone. +DAPI_LOW_INTENSITY_FLOOR = 50.0 + + +# Fallback intensity critical thresholds (used when YAML not provided) +_INTENSITY_CRITICAL_DEFAULTS = {"dapi": 500, "boundary": 100, "intrna": 300} + +# Fixed per-channel colorbar caps for the §3.3 intensity spatial heatmaps. +# Calibrated against 9 tissues (lung / pancreas / liver / brain) — pooled p99 +# was ~6300 (DAPI), ~7000 (Boundary), ~3600 (IntRNA). Caps were rounded down +# below pooled p99 so bright tissues (brain, pancreas) saturate to the +# extend="max" triangle and dim tissues (low-signal lung) genuinely render +# dim — cross-sample comparability over per-sample auto-scaling. +_INTENSITY_DISPLAY_CAP = {"dapi": 4000, "boundary": 4000, "intrna": 2000} + + +def _yaml_channel_cfg(channels_cfg: dict | None, channel: str) -> dict: + """Resolve one channel's section of the thresholds YAML, case-insensitively. + + The YAML spells the same three channels two ways — ``DAPI``/``boundary``/ + ``intRNA`` under ``channels:`` but ``dapi``/``boundary``/``intrna`` under + ``snr:`` — so callers used to bridge the casings with a hand-written map. + Matching on the lower-cased key removes that map: casing can no longer + drift out of sync and silently resolve to a default while the YAML says + otherwise. + + An absent/empty ``channels:`` section is the legitimate "no YAML supplied" + path and yields ``{}`` so the callers' documented defaults apply. A + populated section that is missing this channel is real config/code drift + and raises instead of falling back silently. + """ + if not channels_cfg: + return {} + lowered = {str(k).lower(): v for k, v in channels_cfg.items()} + if channel.lower() not in lowered: + raise KeyError( + f"channel {channel!r} is missing from the 'channels:' section of " + f"roi_image_qc_thresholds.yaml (present: {sorted(channels_cfg)})" + ) + return lowered[channel.lower()] or {} + + +def _load_qc_thresholds(yaml_path: str | None) -> dict: + """Load thresholds from roi_image_qc_thresholds.yaml. + + Returns the ``image_qc`` sub-dict, or an empty dict if the file + is missing / unreadable. + """ + if yaml_path is None: + return {} + try: + import yaml + + p = Path(yaml_path) + if not p.exists(): + logging.warning("Thresholds YAML not found: %s", yaml_path) + return {} + with open(p, encoding="utf-8") as f: + d = yaml.safe_load(f) or {} + return d.get("image_qc", {}) if isinstance(d, dict) else {} + except Exception as exc: + logging.warning("Failed to load thresholds YAML: %s", exc) + return {} + + +# Focus score percentile: Percentile of raw focus scores to use as threshold +# Applied to: tiles only after exclusion of low-intensity tiles (intensity >= ROI_INTENSITY_THRESHOLD) +# Purpose: Separate blurred vs in-focus tiles + +ROI_FOCUS_SCORE_PERCENTILE = 5.0 + +# --- Per-cluster outlier detection (cluster_blur_outliers / cluster_ccfs_outliers) --- +# A cluster is flagged when its blur (or low-texture) rate is both a robust +# statistical outlier among the sample's clusters AND above an absolute floor. +CLUSTER_OUTLIER_MIN_CELLS = 50 # ignore tiny clusters +CLUSTER_OUTLIER_MIN_CLUSTERS = 5 # need enough clusters for robust stats +CLUSTER_OUTLIER_FLOOR_PCT = 15.0 # absolute minimum % to ever flag +CLUSTER_OUTLIER_MAD_Z = 3.5 # modified (MAD-based) z-score cutoff + + +def detect_cluster_outliers(cluster_stats, pct_key): + """Flag clusters that stand apart from the rest of the sample. + + A cluster is flagged when its percentage (``pct_key``) is at least + ``CLUSTER_OUTLIER_FLOOR_PCT`` AND it is a robust statistical outlier vs the + other clusters, measured by a modified z-score ``(x - median) / (1.4826 * + MAD) > CLUSTER_OUTLIER_MAD_Z``. When the MAD degenerates to 0 (common for + low-texture, where most clusters sit near 0%), the floor alone decides, so a + genuine spike is still caught. Needs at least ``CLUSTER_OUTLIER_MIN_CLUSTERS`` + clusters with ``>= CLUSTER_OUTLIER_MIN_CELLS`` cells; below that, robust + stats are unreliable and nothing is flagged. + + Parameters + ---------- + cluster_stats : dict[int, dict] + Per-cluster stats, e.g. ``{cluster_id: {"n_cells": int, pct_key: float}}``. + pct_key : str + Key holding the percentage to test ("pct_blurred" or "pct_low_texture"). + + Returns + ------- + dict[int, dict] + Subset of ``cluster_stats`` for flagged clusters, full stats retained + (downstream report consumers read ``n_cells`` and the percentage). + """ + eligible = { + cl: s + for cl, s in cluster_stats.items() + if s.get("n_cells", 0) >= CLUSTER_OUTLIER_MIN_CELLS + } + if len(eligible) < CLUSTER_OUTLIER_MIN_CLUSTERS: + return {} + vals = np.array([eligible[cl][pct_key] for cl in eligible], dtype=float) + med = float(np.median(vals)) + mad = 1.4826 * float(np.median(np.abs(vals - med))) + flagged = {} + for cl, s in eligible.items(): + x = s[pct_key] + if x < CLUSTER_OUTLIER_FLOOR_PCT: + continue + if mad <= 0 or (x - med) / mad > CLUSTER_OUTLIER_MAD_Z: + flagged[cl] = s + return flagged + + +def _run_figure_task(fn): + """Wrapper for multiprocessing figure tasks that logs exceptions before exit. + + When using fork-based multiprocessing, child process tracebacks are lost + if the child crashes. This wrapper catches and logs the full traceback + in the child process before re-raising, so errors are visible in logs. + """ + try: + fn() + except Exception: + logging.error(f"Figure task {fn.__name__} failed:\n{traceback.format_exc()}") + raise + + +def open_zarr(path: Path, zarr3: bool = False) -> zarr.Group: + if zarr3: + store = ( + zarr.storage.ZipStore(path) + if path.suffix == ".zip" + else zarr.storage.LocalStore(path) + ) + return zarr.open_group(store=store, mode="r") + else: + """Open a Zarr file (compatible with zarr < 3)""" + store = ( + zarr.ZipStore(path, mode="r") + if path.suffix == ".zip" + else zarr.DirectoryStore(path) + ) + return zarr.group(store=store) + + +def load_and_prepare_data(xenium_bundle_dir, outdir): + """Load all required data files and prepare paths""" + + xenium_bundle_dir = Path(xenium_bundle_dir) + # Resolve outdir to absolute path to avoid nested directory issues + # If outdir is already absolute, resolve() returns it as-is + # If outdir is relative, resolve() makes it absolute relative to current working directory + outdir = Path(outdir).resolve() + + # Required paths + cells_parquet_path = xenium_bundle_dir / "cells.parquet" + clusters_csv_path = ( + xenium_bundle_dir + / "analysis" + / "clustering" + / "gene_expression_kmeans_10_clusters" + / "clusters.csv" + ) + umap_path = ( + xenium_bundle_dir + / "analysis" + / "umap" + / "gene_expression_2_components" + / "projection.csv" + ) + cell_masks_path = xenium_bundle_dir / "cells.zarr.zip" + morphology_focus_dir = xenium_bundle_dir / "morphology_focus" + + # Create output directories + figures_dir = outdir / "figures" + figures_dir.mkdir(parents=True, exist_ok=True) + + return { + "xenium_bundle_dir": xenium_bundle_dir, + "outdir": outdir, + "figures_dir": figures_dir, + "cells_parquet_path": cells_parquet_path, + "clusters_csv_path": clusters_csv_path, + "umap_path": umap_path, + "cell_masks_path": cell_masks_path, + "morphology_focus_dir": morphology_focus_dir, + } + + +def _load_morphology_channels( + xoa_morphology_files, + *, + level: int = 0, +): + """ + Load DAPI/Boundary/IntRNA channels robustly from either: + - a single multi-channel OME-TIFF, or + - separate single-channel files (e.g. *_0000, *_0001, *_0002). + """ + + def _split_channels(arr): + # Support both channel-first (C,Y,X) and channel-last (Y,X,C). + # .copy() on each slice so the parent 3D array can be GC'd. + if arr.ndim == 2: + return arr, None, None + if arr.ndim != 3: + raise ValueError(f"Unexpected morphology array shape: {arr.shape}") + if arr.shape[0] <= 4 and arr.shape[1] > 16 and arr.shape[2] > 16: + d = arr[0].copy() + b = arr[1].copy() if arr.shape[0] > 1 else None + r = arr[2].copy() if arr.shape[0] > 2 else None + return d, b, r + if arr.shape[2] <= 4 and arr.shape[0] > 16 and arr.shape[1] > 16: + d = arr[:, :, 0].copy() + b = arr[:, :, 1].copy() if arr.shape[2] > 1 else None + r = arr[:, :, 2].copy() if arr.shape[2] > 2 else None + return d, b, r + # Conservative fallback: prefer first axis as channels. + d = arr[0].copy() + b = arr[1].copy() if arr.shape[0] > 1 else None + r = arr[2].copy() if arr.shape[0] > 2 else None + return d, b, r + + primary = tifffile.imread( + xoa_morphology_files[0], is_ome=False, level=level, aszarr=False + ) + + if primary.ndim == 3: + dapi, boundary, intrna = _split_channels(primary) + return dapi, boundary, intrna + + dapi = primary + boundary = None + intrna = None + + if len(xoa_morphology_files) > 1 and Path(xoa_morphology_files[1]).exists(): + try: + b = tifffile.imread( + xoa_morphology_files[1], is_ome=False, level=level, aszarr=False + ) + boundary = _split_channels(b)[0] if getattr(b, "ndim", 0) == 3 else b + except Exception as e: + logging.warning( + "Boundary channel load failed for %s (continuing with DAPI): %s", + xoa_morphology_files[1], + e, + ) + boundary = None + if len(xoa_morphology_files) > 2 and Path(xoa_morphology_files[2]).exists(): + try: + r = tifffile.imread( + xoa_morphology_files[2], is_ome=False, level=level, aszarr=False + ) + intrna = _split_channels(r)[0] if getattr(r, "ndim", 0) == 3 else r + except Exception as e: + logging.warning( + "IntRNA channel load failed for %s (continuing with DAPI): %s", + xoa_morphology_files[2], + e, + ) + intrna = None + + return dapi, boundary, intrna + + +# Tissue-mask threshold guard (2026-06-23, fixes the generate_tissue_mask bug; +# see plans/2026-06-23_PLAN_fix-tissue-mask-bug.md and the Otsu spike). The old +# fixed 60th-percentile threshold assumed ~40% of the field is tissue and +# degenerated on sparse/dim slides. Otsu is background-aware and adapts to the +# real tissue fraction, but on a UNIMODAL field (all background or all tissue) +# Otsu still returns a split, fabricating a mask — so a guard rejects those. +# Calibrated on the spike's synthetic Gaussian fields: real bimodal class +# separation 8.9-24.4 sd vs unimodal 2.6-2.8 sd (clean margin around 3.0). +# CAVEAT: the 3.0 sd cutoff is synthetic-calibrated; confirm on real small0. +TISSUE_OTSU_GUARD_MIN_FG = 0.02 # foreground < 2% of field -> nothing detected +TISSUE_OTSU_GUARD_MAX_FG = 0.90 # foreground > 90% of field -> no background +TISSUE_OTSU_GUARD_MIN_SEP_SD = 3.0 # tissue mean must exceed bg mean by >= 3 bg-sd + + +def otsu_tissue_threshold_with_guard(small0): + """Background-aware tissue threshold (Otsu) with a degeneracy guard. + + Returns ``(threshold, ok)``. ``ok=False`` means the field is unimodal / + has no separable tissue, so no mask should be formed (the caller returns an + empty mask, which downstream becomes ``tissue_mask_qc.status == FAIL``). + + The load-bearing test is class separation, not foreground fraction: on an + all-background field Otsu splits the noise at ~40% foreground (inside any + fraction band), so only the separation floor catches it. + """ + arr = np.asarray(small0, dtype=np.float64) + finite = arr[np.isfinite(arr)] + if finite.size == 0 or float(finite.max()) == float(finite.min()): + return None, False # empty or uniform field + t = float(threshold_otsu(finite)) + fg = arr >= t + fg_frac = float(np.mean(fg)) + bg_vals = arr[(arr < t) & np.isfinite(arr)] + fg_vals = arr[fg & np.isfinite(arr)] + if bg_vals.size == 0 or fg_vals.size == 0: + return t, False + bg_std = float(bg_vals.std()) + sep_sd = ( + (float(fg_vals.mean()) - float(bg_vals.mean())) / bg_std + if bg_std > 1e-9 + else np.inf + ) + ok = ( + TISSUE_OTSU_GUARD_MIN_FG <= fg_frac <= TISSUE_OTSU_GUARD_MAX_FG + and sep_sd >= TISSUE_OTSU_GUARD_MIN_SEP_SD + ) + return t, ok + + +# Hysteresis tissue mask (2026-06-25): replaces the global-Otsu threshold, which +# under-captured dim/sparse tissue (validated non-circularly against decoded +# transcripts across 14 tissues — Otsu keeps ~20% of transcript tiles, hysteresis +# ~60%; see plans/2026-06-26_PLAN_tissue-mask-recalibration.md). Background mean + robust +# SD are estimated from NON-ZERO pixels (the level-3 DAPI overview is 57-80% exact zeros +# outside the imaged area; including them collapses the robust SD to 0 -> rsd falls back +# to 1 -> the hysteresis cutoffs become tiny and the mask floods). Estimating from +# non-zero pixels makes the spread genuinely per-sample. Tissue = pixels connected to a +# confident-bright seed (bg + SEED*robSD) grown down to a floor (bg + GROW*robSD), so the +# mask follows dim/uneven tissue without flooding background. The degeneracy guard is +# STRUCTURAL only (foreground fraction + class separation): verified empty-field-safe in +# plans/spikes/spike_empty_field_guard.py (an empty/noise field fails on fraction < 2% or +# separation < 3). No absolute brightness floor on the mask — absolute DAPI quality lives +# in usable_tissue via is_low_intensity (per-tile means, flood-immune). +TISSUE_HYST_SEED_SD = 3.0 # seed: confident tissue at bg + 3*robSD +TISSUE_HYST_GROW_SD = 1.0 # grow connected tissue down to bg + 1*robSD + + +def _robust_background_stats(small0): + """``(bg_median, robust_sd)`` of the background, estimated from NON-ZERO pixels + (Otsu split on the non-zero values, below-threshold = background). ``(None, None)`` + on a degenerate (empty / uniform) field. Estimating from non-zero pixels avoids the + hard-zero collapse that flooded the mask (see module comment above).""" + arr = np.asarray(small0, dtype=np.float64) + nz = arr[np.isfinite(arr) & (arr > 0)] + if nz.size == 0 or float(nz.max()) == float(nz.min()): + return None, None + t = float(threshold_otsu(nz)) + bg = nz[nz < t] + if bg.size == 0: + bg = nz + med = float(np.median(bg)) + mad = float(np.median(np.abs(bg - med))) + rsd = 1.4826 * mad if mad > 0 else 1.0 + return med, rsd + + +def hysteresis_tissue_mask_with_guard(small0): + """Background-relative hysteresis tissue mask with a degeneracy guard. + + Returns ``(mask, low, ok)``: + - mask: boolean tissue mask — pixels connected to a ``bg + 3*robSD`` seed, + grown down to ``bg + 1*robSD``. Anchored to the field's own (non-zero) + background, so it follows dim/uneven tissue instead of flooding. + - low: the grow threshold (``bg + 1*robSD``); callers build the background + components (edge/hole distance maps) from ``small0 < low``. + - ok: ``False`` when the field is degenerate / has no separable tissue, so the + caller returns an empty mask -> ``tissue_mask_qc.status == FAIL``. + + Guard (STRUCTURAL, both required): foreground fraction in ``[MIN_FG, MAX_FG]`` AND + class separation ``>= MIN_SEP_SD`` background-SD. No absolute brightness floor: + verified empty-field-safe in plans/spikes/spike_empty_field_guard.py (empty/noise + fields fail on fraction < 2% or separation < 3; real tissue passes). Absolute DAPI + quality is judged separately by usable_tissue (is_low_intensity), not by the mask. + """ + bg_med, rsd = _robust_background_stats(small0) + if bg_med is None: + return np.zeros_like(small0, dtype=bool), None, False + low = bg_med + TISSUE_HYST_GROW_SD * rsd + high = bg_med + TISSUE_HYST_SEED_SD * rsd + arr = np.asarray(small0, dtype=np.float64) + mask = apply_hysteresis_threshold(arr, low, high) + fg_vals = arr[mask & np.isfinite(arr)] + bg_vals = arr[(~mask) & np.isfinite(arr)] + if fg_vals.size == 0 or bg_vals.size == 0: + return mask, low, False + fg_frac = float(np.mean(mask)) + bg_std = float(bg_vals.std()) + sep_sd = ( + (float(fg_vals.mean()) - float(bg_vals.mean())) / bg_std + if bg_std > 1e-9 + else np.inf + ) + ok = ( + TISSUE_OTSU_GUARD_MIN_FG <= fg_frac <= TISSUE_OTSU_GUARD_MAX_FG + and sep_sd >= TISSUE_OTSU_GUARD_MIN_SEP_SD + ) + return mask, low, ok + + +def compute_tissue_mask(small0, min_size_hole=1500): + """Shared tissue-mask logic — the single source of truth for the + threshold + labelling, used by generate_tissue_mask AND the + calculate_roi_focusscore* tissue filters so the three call sites cannot + drift (2026-06-23 bug fix; previously the logic was copy-pasted three times, + each with the `percentile 60` + `test_mask > 1` defects). + + 2026-06-25: tissue threshold is now background-relative HYSTERESIS (see + hysteresis_tissue_mask_with_guard) instead of global Otsu, which under-captured + dim/sparse tissue. `test_mask > 0` keeps all foreground components. + + Returns ``(whole_sample, objects, holes)``: + - whole_sample: labelled tissue mask. Empty when the guard rejects a + degenerate / faint field (no separable tissue) -> downstream + ``tissue_mask_qc.status == FAIL`` rather than a fabricated mask. + - objects: labelled background components (``label(small0 < low)``, the + hysteresis grow threshold) — callers that need the edge/distance map reuse + this. + - holes: labelled holes (border-cleared background components, small ones + removed) — callers that need the hole distance map reuse this. + """ + mask, low, ok = hysteresis_tissue_mask_with_guard(small0) + if ok: + thresh1 = small0 < low # background (below the hysteresis grow threshold) + thresh2 = mask # tissue + else: + thresh1 = np.ones_like(small0, dtype=bool) + thresh2 = np.zeros_like(small0, dtype=bool) + objects = measure.label(thresh1) + noborder = clear_border(objects) + holes = morphology.remove_small_objects(noborder, min_size=min_size_hole) + small_objects = noborder ^ holes + test_mask = measure.label(thresh2) + small_objects + whole_sample = measure.label(test_mask > 0) + return whole_sample, objects, holes + + +def compute_multistain_tissue_mask(small0, small1, small2, min_size_hole=1500): + """Tissue EXTENT mask combining all available morphology stains. + + DAPI nuclei are sparse in some tissues (muscle fibres, brain neuropil), so a + DAPI-only mask under-captures them; the Boundary (membrane) and Interior + (cytoplasm/rRNA) stains cover those regions. This builds a per-channel + background-relative hysteresis mask (reusing hysteresis_tissue_mask_with_guard, so + each channel is guarded against its OWN background) and ORs the channels that pass + their guard. The union gets a foreground-fraction sanity check only (the SD-relative + separation test is per-channel and undefined on a boolean union). + + Used for tissue EXTENT / coverage ONLY. The DAPI-only mask (compute_tissue_mask) + still drives the focus/blur QC, because feeding nuclei-poor tiles into the DAPI + focus GMM falsely reads as blurry. See plans/2026-06-26_PLAN_multistain-mask.md. + + DAPI-only bundles (small1 and small2 both None) return EXACTLY + compute_tissue_mask(small0) — byte-identical to the DAPI-only behaviour. + + Returns ``(whole_sample, objects, holes)`` like compute_tissue_mask. The union's + own background components (``objects``/``holes``) are rebuilt from ``~union`` (a + union has no single grow threshold), so edge/hole geometry reflects the combined + tissue extent. + + NOTE: dense-intensity-artefact subtraction is intentionally NOT applied — the + dense-intensity mask is the brightest p97 of each channel, which includes real + bright tissue, so subtracting it would remove real tissue. Artefact handling is a + deferred follow-up (the spike measured ~1% false tissue without it). + """ + if small1 is None and small2 is None: + return compute_tissue_mask(small0, min_size_hole=min_size_hole) + + union = None + for ch in (small0, small1, small2): + if ch is None: + continue + mask, _low, ok = hysteresis_tissue_mask_with_guard(ch) + if ok: + union = mask if union is None else (union | mask) + if union is None: + union = np.zeros_like(small0, dtype=bool) + + fg_frac = float(union.mean()) + if TISSUE_OTSU_GUARD_MIN_FG <= fg_frac <= TISSUE_OTSU_GUARD_MAX_FG: + thresh1 = ~union # background = everything outside the combined tissue + thresh2 = union + else: + thresh1 = np.ones_like(small0, dtype=bool) + thresh2 = np.zeros_like(small0, dtype=bool) + objects = measure.label(thresh1) + noborder = clear_border(objects) + holes = morphology.remove_small_objects(noborder, min_size=min_size_hole) + small_objects = noborder ^ holes + test_mask = measure.label(thresh2) + small_objects + whole_sample = measure.label(test_mask > 0) + return whole_sample, objects, holes + + +def _per_tile_coverage(whole_sample, x1, x2, y1, y2, downsample_factor=8): + """Per-tile tissue-coverage fraction for ROI tiles given in full-resolution coords, + computed from a labelled `whole_sample` mask at `downsample_factor` resolution via a + summed-area table (same logic as the grid builders). Returns a float array aligned to + the x1/x2/y1/y2 arrays.""" + mask = np.asarray(whole_sample) > 0 + h, w = mask.shape + ii = np.zeros((h + 1, w + 1), dtype=np.int64) + ii[1:, 1:] = np.cumsum(np.cumsum(mask.astype(np.int64), axis=0), axis=1) + r1 = np.clip(np.asarray(y1) // downsample_factor, 0, h) + r2 = np.clip((np.asarray(y2) - 1) // downsample_factor + 1, 0, h) + c1 = np.clip(np.asarray(x1) // downsample_factor, 0, w) + c2 = np.clip((np.asarray(x2) - 1) // downsample_factor + 1, 0, w) + area = np.maximum((r2 - r1) * (c2 - c1), 1) + s = ii[r2, c2] - ii[r1, c2] - ii[r2, c1] + ii[r1, c1] + return s.astype(np.float64) / area + + +def generate_tissue_mask( + xoa_morphology_files, + small0, + small1, + small2, + threshold_percentile=60, + min_size_edge=500000, + min_size_hole=1500, + dense_intensity_region_percentile=97, + min_size_dense_intensity_region=500, + downsample_factor=8, +): + """ + Generate tissue masks and distance maps from morphology images (cell/segmentation independent). + + Parameters: + ----------- + xoa_morphology_files : list + List of paths to morphology image files (for compatibility, not used if small0/1/2 provided) + small0, small1, small2 : numpy.ndarray + Downsampled morphology images (DAPI, Boundary, Interior) + threshold_percentile : float, optional + Percentile for thresholding (default: 60) + min_size_edge : int, optional + Minimum size for edge objects in downsampled space (default: 500000) + min_size_hole : int, optional + Minimum size for holes in downsampled space (default: 1500) + dense_intensity_region_percentile : float, optional + Percentile for dense intensity region detection (default: 97) + min_size_dense_intensity_region : int, optional + Minimum size for dense intensity regions (default: 500) + downsample_factor : int, optional + Downsampling factor (default: 8, for level 3) + + Returns: + -------- + tuple + (whole_sample, holes, dense_intensity_regions, distance_map, distance_map2, + multistain_whole_sample, multistain_distance_map, multistain_distance_map2) + - whole_sample: Labeled DAPI tissue mask (drives focus/blur QC) + - holes: Labeled holes mask (DAPI) + - dense_intensity_regions: Labeled dense intensity regions mask + - distance_map: Distance to edge map (DAPI) + - distance_map2: Distance to nearest hole map (DAPI) + - multistain_whole_sample: Labeled multi-stain tissue-EXTENT mask (DAPI OR Boundary + OR Interior), or None on DAPI-only bundles. Drives the reported coverage / extent. + - multistain_distance_map / multistain_distance_map2: edge / hole distance maps for + the multi-stain mask (None on DAPI-only bundles) + """ + # Tissue mask via the shared helper (hysteresis + degeneracy guard + `> 0`). + # 2026-06-23 bug fix: the old `np.percentile(small0, threshold_percentile)` + # (60) assumed ~40% of the field is tissue and degenerated on sparse/dim + # slides, and `test_mask > 1` emptied the mask on a single whole-field blob. + # `threshold_percentile` is retained in the signature but no longer used. + # See compute_tissue_mask and plans/2026-06-23_PLAN_fix-tissue-mask-bug.md. + whole_sample, objects, holes = compute_tissue_mask( + small0, min_size_hole=min_size_hole + ) + + # Edge + distance maps (generate_tissue_mask-specific; reuse `objects`/`holes`). + mask = morphology.remove_small_objects(objects, min_size=min_size_edge) + edge_sample = measure.label(mask) + distance_map = nsitk.signed_maurer_distance_map(edge_sample) + distance_map2 = nsitk.signed_maurer_distance_map(holes) + + # Detecting dense intensity regions — a pixel counts as artefact when it is + # bright in ANY available channel. `small1` / `small2` are None for DAPI-only + # bundles (returned from `_load_morphology_channels`); blindly indexing them + # would crash. + # + # The `> 0` below is load-bearing: from 2025-06-06 to 2026-05-06 this block + # added three *bool* arrays, and NumPy `+` on bool dtype is logical OR + # returning bool, so binary_fill_holes received a genuine union. 119cff9 added + # .astype(np.int_) while making the block None-tolerant, turning the OR into a + # 0..3 count -- and BinaryFillhole treats only value 1 as foreground, so pixels + # bright in 2 or 3 channels (the strongest artefact evidence) were dropped. The + # count is kept because it is informative, but the reduction is now written down + # rather than riding on a dtype. `>= 2` was rejected on measured data in + # plans/2026-06-26_SPIKE_multistain-mask.md and is impossible on DAPI-only + # bundles. See tests/test_dense_intensity_mask.py. + t0 = np.percentile(small0, dense_intensity_region_percentile) + thresh_sum = (small0 >= t0).astype(np.int_) + if small1 is not None: + t1 = np.percentile(small1, dense_intensity_region_percentile) + thresh_sum = thresh_sum + (small1 >= t1).astype(np.int_) + if small2 is not None: + t2 = np.percentile(small2, dense_intensity_region_percentile) + thresh_sum = thresh_sum + (small2 >= t2).astype(np.int_) + thresh_fill = nsitk.binary_fill_holes(thresh_sum > 0) + objects_art = measure.label(thresh_fill) + dense_intensity_regions = morphology.remove_small_objects( + objects_art, min_size=min_size_dense_intensity_region + ) + + # Multi-stain tissue-EXTENT mask (DAPI OR Boundary OR Interior) + its edge/hole distance + # maps, for the extent metrics and the §2.3 figure. None on DAPI-only bundles so the + # extent path falls back to the DAPI coverage and stays byte-identical. + if small1 is None and small2 is None: + ms_whole_sample = ms_distance_map = ms_distance_map2 = None + else: + ms_whole_sample, ms_objects, ms_holes = compute_multistain_tissue_mask( + small0, small1, small2, min_size_hole=min_size_hole + ) + ms_edge = morphology.remove_small_objects(ms_objects, min_size=min_size_edge) + ms_distance_map = nsitk.signed_maurer_distance_map(measure.label(ms_edge)) + ms_distance_map2 = nsitk.signed_maurer_distance_map(ms_holes) + + return ( + whole_sample, + holes, + dense_intensity_regions, + distance_map, + distance_map2, + ms_whole_sample, + ms_distance_map, + ms_distance_map2, + ) + + +# --------------------------------------------------------------------------- +# Internal helpers (from focus_score_maps) +# --------------------------------------------------------------------------- + +_EPSILON: float = 1e-8 + + +def _get_backend( + use_gpu: bool, +) -> tuple[Any, Any, Any]: + """Return the array module, uniform_filter, and laplace for the chosen backend. + + Args: + use_gpu: If ``True`` and CuPy is available, return GPU primitives. + + Returns: + Tuple of ``(array_module, uniform_filter_fn, laplace_fn)``. + + Raises: + RuntimeError: If ``use_gpu`` is ``True`` but CuPy is not installed. + """ + if use_gpu: + if not HAS_CUPY: + raise RuntimeError( + "use_gpu=True but CuPy is not installed. " + "Install CuPy or set use_gpu=False." + ) + return cp, cupy_uniform_filter, cupy_laplace + return np, scipy_uniform_filter, scipy_laplace + + +def _to_device(image: NDArray[np.generic], xp: Any) -> Any: + """Move a numpy array to the target device (no-op for NumPy). + + Args: + image: Input numpy array. + xp: Array module (``numpy`` or ``cupy``). + + Returns: + Array on the target device. + """ + if xp is np: + return image + return cp.asarray(image) + + +def _to_numpy(arr: Any, xp: Any) -> NDArray[np.float32]: + """Move an array back to host memory as float32 numpy (no-op for NumPy). + + Args: + arr: Array on device. + xp: Array module that created *arr*. + + Returns: + numpy float32 array on the host. + """ + if xp is np: + return np.asarray(arr, dtype=np.float32) + return cp.asnumpy(arr).astype(np.float32) + + +def _sanitize(arr: NDArray[np.float32]) -> NDArray[np.float32]: + """Replace NaN and Inf values with zero. + + Args: + arr: Input array (modified in-place). + + Returns: + The same array with non-finite values set to zero. + """ + arr[~np.isfinite(arr)] = 0.0 + return arr + + +#: Longest-axis pixel budget for arrays sent to imshow/contour/label2rgb. A panel +#: is only ~1600 px at dpi 300, so imshow of an 86-megapixel map (12755x6738) +#: resamples all of it and discards ~97%: measured 15.9 s vs 0.5 s downsampled +#: first (32x). 2000 keeps every pixel a >=300-dpi panel can resolve. +_FIG_DISPLAY_MAX_PX = 2000 + + +def _thumb(arr: Any, max_long: int = _FIG_DISPLAY_MAX_PX) -> Any: + """Downsample a 2-D array to <= ``max_long`` on its long axis, for DISPLAY only. + + Float arrays are area-averaged (block mean, NaN-safe) to match matplotlib's own + antialiased downscale, so the rendered panel is visually identical to + ``imshow(arr)``. Integer/bool arrays (labels, masks) are block-MAX reduced (a mean + of labels is meaningless): a block stays non-zero if any pixel in it is, so a + feature thinner than ``step`` still renders instead of falling between strided + rows. No-op on already-small / non-2-D input. Never use for exported data -- only + at the draw call. Pair with ``extent`` of the *original* shape (see + ``_imshow_thumb``) so axes are unchanged. + """ + a = np.asarray(arr) + if a.ndim != 2: + return a + # Ceil, not floor: floor overshoots the cap (12755 // 2000 = 6 leaves 2125 px; + # 3000 // 2000 = 1 would not downsample a 3000 px axis at all). + step = max(1, -(-max(a.shape) // max_long)) + if step == 1: + return a + ny, nx = (a.shape[0] // step) * step, (a.shape[1] // step) * step + blocks = a[:ny, :nx].reshape(ny // step, step, nx // step, step) + if np.issubdtype(a.dtype, np.floating): + with warnings.catch_warnings(): # all-NaN blocks -> NaN, as intended + warnings.simplefilter("ignore", category=RuntimeWarning) + return np.nanmean(blocks, axis=(1, 3)) + return blocks.max(axis=(1, 3)) + + +#: Target number of display bins along the long axis for Figure 5's binned focus +#: and classification heatmaps. ~180 keeps regional signal readable while dropping +#: the per-pixel speckle of an ~86-megapixel field, and it draws in ~1 s instead of +#: the ~126 s an imshow of the full field cost. +_FOCUS_HEATMAP_BINS_LONG = 180 + + +def _bin_nanmean(arr: Any, step: int) -> Any: + """Block-reduce a 2-D float array by ``step``x``step``, averaging over the finite + (non-NaN) pixels of each block. Blocks with no finite pixel return NaN. + + NaN encodes "no tissue here" for the caller: pre-set non-tissue pixels to NaN and + this returns the per-bin mean over tissue only. Uses the same NaN-safe block-mean as + :func:`_thumb`; the all-NaN ``RuntimeWarning`` is suppressed because an empty + (non-tissue) bin returning NaN is the intended result, not an error. + """ + a = np.asarray(arr) + ny = (a.shape[0] // step) * step + nx = (a.shape[1] // step) * step + blocks = a[:ny, :nx].reshape(ny // step, step, nx // step, step) + with warnings.catch_warnings(): # all-NaN blocks -> NaN, as intended + warnings.simplefilter("ignore", category=RuntimeWarning) + return np.nanmean(blocks, axis=(1, 3)) + + +def _imshow_thumb(ax: Any, arr: Any, *, rgb: Any = None, **kwargs: Any) -> Any: + """``ax.imshow`` of ``arr`` downsampled for speed but with the extent/axes of the + FULL ``arr``, so the panel is visually identical to ``ax.imshow(arr)``. ``rgb`` is + an optional callable (e.g. ``label2rgb``) applied to the downsampled array.""" + a = np.asarray(arr) + disp = _thumb(a) + if rgb is not None: + disp = rgb(disp) + if a.ndim == 2: + kwargs.setdefault("extent", (-0.5, a.shape[1] - 0.5, a.shape[0] - 0.5, -0.5)) + return ax.imshow(disp, **kwargs) + + +def _contour_thumb(ax: Any, mask: Any, **kwargs: Any) -> Any: + """``ax.contour`` of a boolean/label ``mask`` downsampled to display resolution, + with X/Y mapped to the FULL-res pixel coordinate space so the boundary overlays an + ``_imshow_thumb`` panel exactly. Runs marching-squares on the ~2000-px thumbnail + instead of the full ~86-megapixel field (block-MAX keeps every thin boundary), so + the traced contour is visually identical to ``ax.contour(mask)`` at >= 300 dpi but + costs ~0.1 s instead of several seconds of contouring + tens of thousands of vector + segments.""" + a = np.asarray(mask) + m = _thumb(a) + ny, nx = m.shape + h, w = a.shape + xs = np.linspace(0, w - 1, nx) + ys = np.linspace(0, h - 1, ny) + return ax.contour(xs, ys, m, **kwargs) + + +def _kde_density_grid(kde: Any, xs: Any, ys: Any, gridsize: int = 128) -> Any: + """Evaluate a fitted ``gaussian_kde`` at every ``(xs, ys)`` via a coarse grid + + bilinear interpolation, instead of ``kde(all points)`` which is O(N * n_fit) and + dominates the large density-scatter figures (>100 s at N=531k, ~16 s here). Keeps + ALL points -- no outlier dropped -- and the density coloring is visually identical + (Spearman 1.000, max normalized-colour diff 6e-4 vs the exact eval).""" + from scipy.interpolate import RegularGridInterpolator + + xs = np.asarray(xs) + ys = np.asarray(ys) + gx = np.linspace(float(xs.min()), float(xs.max()), gridsize) + gy = np.linspace(float(ys.min()), float(ys.max()), gridsize) + grid_x, grid_y = np.meshgrid(gx, gy) + gz = kde(np.vstack([grid_x.ravel(), grid_y.ravel()])).reshape(gridsize, gridsize) + interp = RegularGridInterpolator((gy, gx), gz, bounds_error=False, fill_value=None) + return interp(np.column_stack([ys, xs])) + + +def detect_gpu_ids() -> list[int]: + """Detect available CUDA GPUs. + + Returns: + List of GPU device IDs. Empty list if CuPy is not available or no GPUs + are detected. + """ + if not HAS_CUPY: + logging.info("CuPy is not importable; GPU backend unavailable, using CPU.") + return [] + try: + n_devices = cp.cuda.runtime.getDeviceCount() + return list(range(n_devices)) + except Exception as exc: + # CuPy is installed but the CUDA runtime could not be queried (driver + # missing, GPU not attached to the container, init failure, ...). Surface + # it: otherwise this is indistinguishable from "no GPU present" and a + # silent CPU fallback on a GPU node looks like correct behaviour. + logging.warning( + "CuPy is installed but GPU detection failed (%s: %s); " + "falling back to CPU backend.", + type(exc).__name__, + exc, + ) + return [] + + +# --------------------------------------------------------------------------- +# Focus-map computation +# --------------------------------------------------------------------------- + + +def compute_ccfs_map( + image: NDArray[np.generic], + window_size: int = 35, + use_gpu: bool = False, + gpu_id: int = 0, +) -> tuple[NDArray[np.float32], NDArray[np.float32]]: + """Compute a per-pixel CCFS (Coefficient of Contrast Focus Score) map. + + The CCFS focus score at each pixel is defined as:: + + focus = local_var / local_mean + + which is equivalent to ``std**2 / mean`` computed over a square window + centred on that pixel. + + Args: + image: 2-D input image (any numeric dtype). + window_size: Side length of the square averaging window. + use_gpu: Use CuPy GPU backend when ``True``. + gpu_id: CUDA device ID to use when ``use_gpu=True``. + + Returns: + Tuple of ``(focus_map, mean_map)`` — both float32 arrays with the + same shape as *image*. + + Raises: + ValueError: If *image* is not 2-D or is smaller than the window in + either dimension. + RuntimeError: If *use_gpu* is ``True`` but CuPy is unavailable. + """ + if image.ndim != 2: + raise ValueError(f"Expected a 2-D image, got shape {image.shape}.") + if image.shape[0] < window_size or image.shape[1] < window_size: + raise ValueError( + f"Image shape {image.shape} is smaller than window_size " + f"{window_size} in at least one dimension." + ) + + xp, uniform_filter, _ = _get_backend(use_gpu) + + if use_gpu: + with cp.cuda.Device(gpu_id): + image_f = cp.asarray(image.astype(np.float32, copy=False)) + local_mean = uniform_filter(image_f, size=window_size) + local_sq_mean = uniform_filter(image_f**2, size=window_size) + # Promote to float64 for subtraction to avoid catastrophic cancellation + local_var = ( + local_sq_mean.astype(cp.float64) - local_mean.astype(cp.float64) ** 2 + ) + local_var = xp.maximum(local_var, 0.0) + focus_map_dev = ( + local_var / (local_mean.astype(cp.float64) + _EPSILON) + ).astype(cp.float32) + focus_map = _to_numpy(focus_map_dev, xp) + mean_map = _to_numpy(local_mean, xp) + else: + image_f = image.astype(np.float32, copy=False) + local_mean = uniform_filter(image_f, size=window_size) + sq = image_f**2 + del image_f + local_sq_mean = uniform_filter(sq, size=window_size) + del sq + # Promote to float64 for the subtraction to avoid catastrophic + # cancellation on high-intensity uint16 images. + # Explicit steps to limit peak memory (avoid 3+ simultaneous f64 temps). + local_var = local_sq_mean.astype(np.float64) + del local_sq_mean + temp_mean_f64 = local_mean.astype(np.float64) + local_var -= temp_mean_f64**2 + del temp_mean_f64 + local_var = np.maximum(local_var, 0.0) + # Compute focus_map = local_var / (local_mean + eps). + # Save mean_map first, then free local_mean before creating f64 temp. + mean_map = local_mean.astype(np.float32) + temp_denom = local_mean.astype(np.float64) + del local_mean + temp_denom += _EPSILON + local_var /= temp_denom + del temp_denom + focus_map = local_var.astype(np.float32) + del local_var + + return _sanitize(focus_map), _sanitize(mean_map) + + +def compute_laplacian_variance_map( + image: NDArray[np.generic], + window_size: int = 35, + use_gpu: bool = False, + gpu_id: int = 0, + lap_sigma: float = 1.0, +) -> NDArray[np.float32]: + """Compute a per-pixel windowed Laplacian-variance focus map. + + This is a *local* variant of the standard Laplacian variance focus metric + (Pech-Pacheco et al., 2000). Instead of computing a single global + variance, it produces a spatial map where each pixel holds the variance + of the Laplacian of Gaussian (LoG) response inside the surrounding + *window_size* × *window_size* neighbourhood:: + + lap_var = uniform_filter(lap**2) - uniform_filter(lap)**2 + + Using LoG (``gaussian_laplace`` with *lap_sigma*) rather than the bare + 3×3 Laplacian suppresses pixel-level noise and makes the metric more + specific to genuine edge content (Sun et al., 2004; Pertuz et al., 2013). + + Variance is computed in float64 to avoid catastrophic cancellation in the + ``E[X²] − E[X]²`` formula, then cast back to float32. + + Args: + image: 2-D input image (any numeric dtype). + window_size: Side length of the square averaging window. + use_gpu: Use CuPy GPU backend when ``True``. + gpu_id: CUDA device ID to use when ``use_gpu=True``. + lap_sigma: Gaussian sigma for LoG pre-smoothing (pixels). + Set to 0 to revert to the bare Laplacian. + + Returns: + Float32 array with the same shape as *image*. + + Raises: + ValueError: If *image* is not 2-D or is smaller than the window in + either dimension. + RuntimeError: If *use_gpu* is ``True`` but CuPy is unavailable. + """ + if image.ndim != 2: + raise ValueError(f"Expected a 2-D image, got shape {image.shape}.") + if image.shape[0] < window_size or image.shape[1] < window_size: + raise ValueError( + f"Image shape {image.shape} is smaller than window_size " + f"{window_size} in at least one dimension." + ) + + xp, uniform_filter, laplace_fn = _get_backend(use_gpu) + + if use_gpu: + with cp.cuda.Device(gpu_id): + image_f = cp.asarray(image.astype(np.float32, copy=False)) + if lap_sigma > 0: + lap = cupy_gaussian_laplace(image_f, sigma=lap_sigma) + else: + lap = laplace_fn(image_f) + lap = lap.astype(cp.float64) + lap_mean = uniform_filter(lap, size=window_size) + lap_sq_mean = uniform_filter(lap * lap, size=window_size) + lap_var = lap_sq_mean - lap_mean**2 + lap_var = xp.maximum(lap_var, 0.0) + result = _to_numpy(lap_var.astype(cp.float32), xp) + else: + image_f = image.astype(np.float32, copy=False) + if lap_sigma > 0: + lap = scipy_gaussian_laplace(image_f, sigma=lap_sigma).astype(np.float64) + else: + lap = laplace_fn(image_f).astype(np.float64) + del image_f + lap_sq = lap * lap + lap_mean = uniform_filter(lap, size=window_size) + del lap + lap_sq_mean = uniform_filter(lap_sq, size=window_size) + del lap_sq + np.square(lap_mean, out=lap_mean) # in-place to avoid temporary + lap_var = lap_sq_mean - lap_mean + del lap_sq_mean, lap_mean + lap_var = np.maximum(lap_var, 0.0) + result = lap_var.astype(np.float32) + del lap_var + + return _sanitize(result) + + +def _compute_channel_maps_on_gpu( + channel: NDArray[np.generic], + window_size: int, + gpu_id: int, + include_laplacian: bool = False, + lap_sigma: float = 1.0, + keep_mean_device: bool = False, + keep_focus_device: bool = False, + drop_mean_host: bool = False, +) -> dict[str, NDArray[np.float32]]: + """Compute focus + mean maps for a single channel on a specific GPU. + + This fused implementation opens a single device context and transfers the + image to the GPU once, computing CCFS (focus_map, mean_map) and optionally + the Laplacian variance map within the same context. This avoids redundant + CPU-to-GPU transfers of the full image. + + Args: + channel: 2-D image array. + window_size: Convolution window size. + gpu_id: CUDA device ID. + include_laplacian: Also compute Laplacian variance map. + lap_sigma: Gaussian sigma for LoG pre-smoothing (0 = bare Laplacian). + keep_mean_device: Also return the device (CuPy) mean map under + ``mean_map_device`` without copying it to host, so a consumer can fold + a reduction (the per-ROI Otsu SNR) on the GPU. Kept alive through the + Laplacian phase and freed by the caller once folded. + keep_focus_device: Same as *keep_mean_device* for the focus map, returned + under ``focus_map_device``. The per-nucleus CCFS (``LabeledSumAccumulator``) + and centre-pixel sampling fold on-device from it, so the full focus map + never has to be read back for those reductions. + drop_mean_host: Skip the ``mean_map`` host copy entirely. Valid only when no + consumer reads the host mean map (all mean reductions run on-device from + ``mean_map_device``); this is the transfer the device-resident fold + eliminates. The host ``focus_map`` copy is always produced -- the Figure 5 + heatmap reduces it in float32 on the host, which a GPU reduction cannot + reproduce to float rounding, so that copy stays. + + Returns: + Dict with ``focus_map``, ``mean_map`` (unless *drop_mean_host*), optionally + ``lap_var_map``, and -- when the respective flag is set -- + ``mean_map_device`` / ``focus_map_device`` (CuPy arrays on *gpu_id*). + """ + if not HAS_CUPY: + raise RuntimeError( + "_compute_channel_maps_on_gpu requires CuPy but it is not installed." + ) + + pool = cp.get_default_memory_pool() + + with cp.cuda.Device(gpu_id): + # Upload image to GPU + image_f = cp.asarray(channel.astype(np.float32, copy=False)) + + # --- CCFS (focus_map + mean_map) --- + # Peak memory: max 3 arrays alive at once (~10.5GB for 34K×26K images) + local_mean = cupy_uniform_filter(image_f, size=window_size) + sq = image_f**2 + # Free image_f before allocating local_sq_mean to cap at 3 arrays + del image_f + pool.free_all_blocks() + local_sq_mean = cupy_uniform_filter(sq, size=window_size) + del sq + # Promote to float64 for subtraction to avoid catastrophic cancellation + local_var = ( + local_sq_mean.astype(cp.float64) - local_mean.astype(cp.float64) ** 2 + ) + del local_sq_mean + cp.maximum(local_var, 0.0, out=local_var) + focus_map_dev = (local_var / (local_mean.astype(cp.float64) + _EPSILON)).astype( + cp.float32 + ) + del local_var + + # Transfer CCFS results to CPU. The host focus map is always produced -- + # the Figure 5 heatmap (BlockMeanAccumulator) reduces it in float32 on the + # host, and a GPU reduction of float32 does not reproduce numpy's float32 + # summation order (measured ~2e-7 relative, enough to move Figure 5 pixels; + # see docs/failures/2026-07-26_heatmap-16px-dtype-not-summation-order.md), so + # that copy stays. The host mean map is skipped when every mean reduction + # folds on-device (drop_mean_host) -- that is the D2H transfer this path drops. + focus_map = cp.asnumpy(focus_map_dev).astype(np.float32) + result: dict[str, NDArray[np.float32]] = {"focus_map": _sanitize(focus_map)} + if not drop_mean_host: + mean_map = cp.asnumpy(local_mean).astype(np.float32) + result["mean_map"] = _sanitize(mean_map) + + # --- Device-resident maps handed to the on-GPU consumers --- + # _sanitize each device array in place so it matches the host copy above + # bit-for-bit: local_mean/focus are uniform_filter derivatives of a finite + # image so this is a no-op in practice, but it keeps the on-device and host + # paths structurally identical. Each kept array adds one float32 map of VRAM, + # held past this block (through the Laplacian phase); the caller drops it + # after folding. + if keep_focus_device: + focus_map_dev[~cp.isfinite(focus_map_dev)] = 0.0 + result["focus_map_device"] = focus_map_dev + else: + del focus_map_dev + if keep_mean_device: + # Hand RoiOtsuSnrAccumulator / LabeledSumAccumulator the device mean map + # so their per-ROI / per-label reductions fold on the GPU (no full-map + # D2H copy for those reductions). + local_mean[~cp.isfinite(local_mean)] = 0.0 + result["mean_map_device"] = local_mean + else: + del local_mean + pool.free_all_blocks() + if include_laplacian: + image_f = cp.asarray(channel.astype(np.float32, copy=False)) + if lap_sigma > 0: + lap = cupy_gaussian_laplace(image_f, sigma=lap_sigma) + else: + lap = cupy_laplace(image_f) + del image_f + pool.free_all_blocks() + lap = lap.astype(cp.float64) + lap_mean = cupy_uniform_filter(lap, size=window_size) + sq = lap * lap + del lap + pool.free_all_blocks() + lap_sq_mean = cupy_uniform_filter(sq, size=window_size) + del sq + lap_var = lap_sq_mean - lap_mean**2 + del lap_sq_mean, lap_mean + cp.maximum(lap_var, 0.0, out=lap_var) + lap_var_np = cp.asnumpy(lap_var).astype(np.float32) + del lap_var + pool.free_all_blocks() + result["lap_var_map"] = _sanitize(lap_var_np) + + return result + + +# Shared TIFF-tile decode pool. page.decode (imagecodecs) releases the GIL, so +# decoding the compressed blobs across a pool sized to the CPU count saturates the +# cores that would otherwise idle while the 4 GPU workers wait on a single-threaded +# decoder -- decode was the tile-pass wall (measured). Sized to the CPU count so the +# total decode concurrency is bounded by cores regardless of GPU count (no +# oversubscription). Lazily created; lives for the process (cleaned up at exit). +_DECODE_POOL: ThreadPoolExecutor | None = None +_DECODE_POOL_LOCK = threading.Lock() + + +def _get_decode_pool() -> ThreadPoolExecutor: + global _DECODE_POOL + if _DECODE_POOL is None: + with _DECODE_POOL_LOCK: + if _DECODE_POOL is None: + _DECODE_POOL = ThreadPoolExecutor( + max_workers=max(2, os.cpu_count() or 4), + thread_name_prefix="tiff-decode", + ) + return _DECODE_POOL + + +class _LazyTiffChannel: + """Lazy 2-D view into one channel page of a TIFF file. + + Supports ``[y0:y1, x0:x1]`` slicing. For tiled TIFFs (production + Xenium images) only the overlapping tiles are decoded, keeping I/O + minimal. For non-tiled TIFFs (e.g. test images written by + ``tifffile.imwrite``) the full page is cached on first access. + + Thread-safe: a lock serialises file-handle reads so that + ``_process_tile_on_gpu`` can call ``[slice]`` from multiple threads. + """ + + #: Class-level default. ``_open_morphology_lazy`` builds instances via + #: ``__new__`` for single-3D-page TIFFs, which skips ``__init__`` and set five + #: attributes by hand -- missing this one, while ``__getitem__`` reads it + #: unguarded on every region request. That raised AttributeError on the first + #: tile read, inside a thread pool. Those instances are always backed by a + #: cached full read, so False (the fallback path) is the correct default, and + #: keeping it here means a future ``__new__`` site cannot reintroduce the bug. + _is_tiled: bool = False + + def __init__(self, page, source: tuple[str, int] | None = None): + """Accept a ``TiffPage`` or ``TiffFrame``. + + ``TiffFrame`` (non-first pages in a multi-page TIFF) lacks + ``imagelength`` / ``is_tiled`` — we fall back to ``.shape`` and + always use the cached-full-read path for frames. + + Args: + page: The ``TiffPage`` / ``TiffFrame`` to wrap. + source: ``(path, page_index)`` this channel can be reopened from in + another process. A live ``TiffPage`` holds an OS file handle + and a ``threading.Lock``, so it cannot be pickled; the process + pool re-opens the channel from this descriptor instead. + ``None`` means the channel is not independently reopenable + (e.g. it is backed by a cached full read), which disqualifies + process-mode tiling. + """ + import threading + + self._page = page + # .shape works on both TiffPage and TiffFrame + self.shape: tuple[int, int] = (page.shape[0], page.shape[1]) + self._lock = threading.Lock() + self._cached_data: NDArray | None = None + # TiffFrame has no is_tiled; treat as non-tiled (fallback path) + self._is_tiled: bool = getattr(page, "is_tiled", False) + # Profiling accumulators (guarded by _lock). Split the tile pass so we can + # tell IO-bound from decode-bound from GPU-convolve-bound: convolve+upload is + # then (compute_seconds - read_seconds - decode_seconds) at the caller. + self._read_seconds: float = 0.0 + self._decode_seconds: float = 0.0 + self._read_bytes: int = 0 + # tifffile may lazily initialise its decoder on first use; warm it once under + # the read lock (see _read_region_tiled) before any off-lock parallel decode. + self._decoder_warmed: bool = False + self._source: tuple[str, int] | None = source + + # ------------------------------------------------------------------ + # public API + # ------------------------------------------------------------------ + + def __getitem__(self, key): + if not isinstance(key, tuple) or len(key) != 2: + raise ValueError("_LazyTiffChannel supports only [y0:y1, x0:x1] slicing") + yslice, xslice = key + y0, y1, _ = yslice.indices(self.shape[0]) + x0, x1, _ = xslice.indices(self.shape[1]) + height = y1 - y0 + width = x1 - x0 + if height <= 0 or width <= 0: + return np.empty((0, 0), dtype=self._page.dtype) + if self._is_tiled: + return self._read_region_tiled(y0, x0, height, width) + return self._read_region_fallback(y0, x0, height, width) + + # ------------------------------------------------------------------ + # tiled path — decode only the tiles that overlap the request + # ------------------------------------------------------------------ + + def _read_region_tiled(self, y0: int, x0: int, height: int, width: int) -> NDArray: + page = self._page + tw, th = page.tilewidth, page.tilelength + im_w, im_h = page.imagewidth, page.imagelength + + y1 = min(y0 + height, im_h) + x1 = min(x0 + width, im_w) + + tile_y0 = y0 // th + tile_x0 = x0 // tw + tile_y1 = int(np.ceil(y1 / th)) + tile_x1 = int(np.ceil(x1 / tw)) + tiles_per_row = int(np.ceil(im_w / tw)) + + buf_h = (tile_y1 - tile_y0) * th + buf_w = (tile_x1 - tile_x0) * tw + out = np.empty((buf_h, buf_w), dtype=page.dtype) + + fh = page.parent.filehandle + jpegtables = page.jpegtables + + def _decode_blob(blob: tuple[int, int, int, bytes]) -> None: + index, oi, oj, data = blob + tile_arr, _indices, _shape = page.decode(data, index, jpegtables=jpegtables) + out[oi : oi + th, oj : oj + tw] = tile_arr.squeeze() + + # Phase 1: read the compressed tile blobs under the lock. Only seek+read touch + # the shared file handle, so the critical section is just the IO (measured at + # ~1-3 s/channel -- negligible). Decoding is pulled OUT of the lock below. + blobs: list[tuple[int, int, int, bytes]] = [] # (index, oi, oj, data) + read_s = 0.0 + read_bytes = 0 + warm_decode_s = 0.0 + with self._lock: + for ti in range(tile_y0, tile_y1): + for tj in range(tile_x0, tile_x1): + index = ti * tiles_per_row + tj + offset = page.dataoffsets[index] + bytecount = page.databytecounts[index] + _t_io = time.perf_counter() + fh.seek(offset) + data = fh.read(bytecount) + read_s += time.perf_counter() - _t_io + read_bytes += bytecount + blobs.append( + (index, (ti - tile_y0) * th, (tj - tile_x0) * tw, data) + ) + self._read_seconds += read_s + self._read_bytes += read_bytes + # Warm tifffile's (possibly lazily-initialised) decoder ONCE, single- + # threaded under the lock, so the concurrent off-lock decodes below cannot + # race on a first-use init. Assembles the first tile of the first call. + if not self._decoder_warmed and blobs: + _t_warm = time.perf_counter() + _decode_blob(blobs[0]) + warm_decode_s = time.perf_counter() - _t_warm + blobs = blobs[1:] + self._decoder_warmed = True + + # Phase 2: decode + assemble OFF the lock. page.decode is a pure function of the + # read bytes and releases the GIL (imagecodecs), so across the GPU-worker + # threads whole strips decode concurrently instead of single-file behind the + # reader lock -- the serialized decode was ~91% of the tile-pass wall clock on a + # 5.5 GP sample (run bVC3fObZxHK6p). Each tile writes a disjoint region of + # `out`, so the assembly is race-free. + _t_dec = time.perf_counter() + if len(blobs) > 1: + # Decode across the shared pool: page.decode releases the GIL, so the + # strip's tiles decode on the otherwise-idle cores instead of single-file. + # Exhaust the map generator so writes complete and errors propagate. + for _ in _get_decode_pool().map(_decode_blob, blobs): + pass + else: + for blob in blobs: + _decode_blob(blob) + decode_s = warm_decode_s + (time.perf_counter() - _t_dec) + with self._lock: + self._decode_seconds += decode_s + + ry0 = y0 - tile_y0 * th + rx0 = x0 - tile_x0 * tw + return out[ry0 : ry0 + (y1 - y0), rx0 : rx0 + (x1 - x0)] + + # ------------------------------------------------------------------ + # fallback path — cache full page (non-tiled / test images) + # ------------------------------------------------------------------ + + def _read_region_fallback( + self, y0: int, x0: int, height: int, width: int + ) -> NDArray: + with self._lock: + if self._cached_data is None: + self._cached_data = self._page.asarray() + return self._cached_data[y0 : y0 + height, x0 : x0 + width] + + +def _open_morphology_lazy( + xoa_morphology_files, + *, + level: int = 0, +) -> tuple[list, tuple[int, int]]: + """Open morphology channels as lazy TIFF page wrappers (no pixel data loaded). + + Uses ``tifffile.TiffFile`` directly instead of the zarr store interface, + avoiding the zarr v3 dependency introduced by ``tifffile.imread(aszarr=True)``. + + Args: + xoa_morphology_files: List of OME-TIFF file paths. + level: Resolution level to open (0 = full resolution). + + Returns: + (channels, image_shape) where channels is [dapi, boundary, intrna] + (None for missing channels) and image_shape is (height, width). + """ + tiff_handles: list[tifffile.TiffFile] = [] # prevent GC + + primary_tif = tifffile.TiffFile(str(xoa_morphology_files[0])) + tiff_handles.append(primary_tif) + + # Use PHYSICAL pages in the file, not OME series pages. + # Xenium multi-file OME-TIFFs report series.shape=(4,H,W) referencing + # all files, but each file has only 1 physical page. series.pages would + # include stubs for data in other files that decode to zeros. + n_physical = len(primary_tif.pages) + if level > 0: + # For pyramid levels, use the series API to navigate levels + series = primary_tif.series[0] + if level < len(series.levels): + pages = list(series.levels[level].pages) + else: + pages = list(primary_tif.pages) + else: + pages = list(primary_tif.pages) + + n_pages = len(pages) + + # Multi-channel single file: truly multi-page TIFF (each page = channel) + # Only enter this path if the file physically contains multiple pages + if n_pages >= 2 and n_physical >= 2: + # Each page is one channel (C, H, W layout across pages). + # At level 0 `pages` is the physical page list, so the list index is the + # physical page index and each channel can be reopened elsewhere by + # (path, page_index). Pyramid levels come from the series API, where + # that identity does not hold — leave those non-reopenable. + primary_path = str(xoa_morphology_files[0]) + + def _page_source(idx: int) -> tuple[str, int] | None: + return (primary_path, idx) if level == 0 else None + + dapi = _LazyTiffChannel(pages[0], source=_page_source(0)) + shape = dapi.shape + boundary = ( + _LazyTiffChannel(pages[1], source=_page_source(1)) if n_pages > 1 else None + ) + intrna = ( + _LazyTiffChannel(pages[2], source=_page_source(2)) if n_pages > 2 else None + ) + # Attach TiffFile handles to prevent garbage collection + dapi._tiff_handles = tiff_handles # type: ignore[attr-defined] + return [dapi, boundary, intrna], shape + + # Single-channel (or single-page) primary + # Check if the single page is actually multi-channel (C, H, W) in one page + page0 = pages[0] + arr_shape = page0.shape # could be (H, W) or (C, H, W) + if len(arr_shape) == 3: + # Multi-channel packed into one page — fall back to cached full read + full = page0.asarray() + if arr_shape[0] <= 4 and arr_shape[1] > 16 and arr_shape[2] > 16: + # Channel-first (C, H, W) + shape = (arr_shape[1], arr_shape[2]) + ch_axis = 0 + elif arr_shape[2] <= 4 and arr_shape[0] > 16 and arr_shape[1] > 16: + # Channel-last (H, W, C) + shape = (arr_shape[0], arr_shape[1]) + ch_axis = 2 + else: + shape = (arr_shape[1], arr_shape[2]) + ch_axis = 0 + n_ch = arr_shape[ch_axis] + dapi = _LazyTiffChannel.__new__(_LazyTiffChannel) + dapi._page = page0 + dapi.shape = shape + import threading + + dapi._lock = threading.Lock() + dapi._source = None # cached full read: not reopenable in another process + dapi._cached_data = np.take(full, 0, axis=ch_axis) + boundary_ch = _LazyTiffChannel.__new__(_LazyTiffChannel) if n_ch > 1 else None + if boundary_ch is not None: + boundary_ch._page = page0 + boundary_ch.shape = shape + boundary_ch._lock = threading.Lock() + # cached full read: not reopenable in another process + boundary_ch._source = None + boundary_ch._cached_data = np.take(full, 1, axis=ch_axis) + intrna_ch = _LazyTiffChannel.__new__(_LazyTiffChannel) if n_ch > 2 else None + if intrna_ch is not None: + intrna_ch._page = page0 + intrna_ch.shape = shape + intrna_ch._lock = threading.Lock() + # cached full read: not reopenable in another process + intrna_ch._source = None + intrna_ch._cached_data = np.take(full, 2, axis=ch_axis) + dapi._tiff_handles = tiff_handles # type: ignore[attr-defined] + return [dapi, boundary_ch, intrna_ch], shape + + # Truly single-channel 2-D page + dapi = _LazyTiffChannel( + page0, source=(str(xoa_morphology_files[0]), 0) if level == 0 else None + ) + shape = dapi.shape + boundary = None + intrna = None + + if len(xoa_morphology_files) > 1 and Path(xoa_morphology_files[1]).exists(): + try: + t = tifffile.TiffFile(str(xoa_morphology_files[1])) + tiff_handles.append(t) + boundary = _LazyTiffChannel( + t.pages[0], + source=(str(xoa_morphology_files[1]), 0) if level == 0 else None, + ) + except Exception as e: + logging.warning( + "Boundary TIFF open failed for %s: %s", xoa_morphology_files[1], e + ) + + if len(xoa_morphology_files) > 2 and Path(xoa_morphology_files[2]).exists(): + try: + t = tifffile.TiffFile(str(xoa_morphology_files[2])) + tiff_handles.append(t) + intrna = _LazyTiffChannel( + t.pages[0], + source=(str(xoa_morphology_files[2]), 0) if level == 0 else None, + ) + except Exception as e: + logging.warning( + "IntRNA TIFF open failed for %s: %s", xoa_morphology_files[2], e + ) + + # Attach TiffFile handles to prevent garbage collection + dapi._tiff_handles = tiff_handles # type: ignore[attr-defined] + return [dapi, boundary, intrna], shape + + +def _compute_adaptive_tile_size( + height: int, + width: int, + n_gpus: int, + gpu_mem_bytes: int | None = None, + target_utilization: float = 0.65, + min_tiles_per_gpu: int = 2, +) -> int: + """Compute tile size that maximizes GPU memory utilization. + + Balances two constraints: + 1. Each tile's peak GPU memory stays within target_utilization of VRAM. + 2. Total tiles >= n_gpus * min_tiles_per_gpu for load balancing. + + Args: + height: Image height in pixels. + width: Image width in pixels. + n_gpus: Number of available GPUs. + gpu_mem_bytes: Total GPU VRAM in bytes. Auto-detected if None. + target_utilization: Fraction of VRAM to target per tile (default 0.65). + min_tiles_per_gpu: Minimum tiles per GPU for load balancing. + + Returns: + Tile size in pixels (square tiles). + """ + if gpu_mem_bytes is None: + gpu_mem_bytes = cp.cuda.Device(0).mem_info[1] + + # Peak memory per pixel: ~6 concurrent float32 arrays + BYTES_PER_PIXEL = 6 * 4 # 24 bytes + + # Max tile dim from GPU memory constraint + max_pixels = int(gpu_mem_bytes * target_utilization / BYTES_PER_PIXEL) + max_tile_dim = int(math.sqrt(max_pixels)) + + # Min tiles needed for load balancing + min_tiles = max(n_gpus * min_tiles_per_gpu, 1) + + # Minimum useful tile dimension (must fit the convolution window) + min_dim = 256 + + # Compute tile_size that produces at least min_tiles + # Start from max_tile_dim and shrink until we have enough tiles + tile_size = max_tile_dim + while tile_size > min_dim: + n_y = math.ceil(height / tile_size) + n_x = math.ceil(width / tile_size) + if n_y * n_x >= min_tiles: + break + tile_size = int(tile_size * 0.7) # shrink by 30% + + # Clamp to minimum useful dimension + tile_size = max(min_dim, tile_size) + + # If image is smaller than tile_size, just use image dims + if height <= tile_size and width <= tile_size: + tile_size = max(height, width) + + return tile_size + + +def _compute_adaptive_strip_height( + height: int, + width: int, + n_gpus: int, + gpu_mem_bytes: int | None = None, + target_utilization: float = 0.65, + min_strips_per_gpu: int = 2, +) -> int: + """Rows per full-width strip that fit the VRAM budget. + + Full-width strips beat square tiles here for two measured reasons: + + * **Contiguous writes.** A square tile's write region is a rectangle, so a + 25,238-wide tile inside a 102,045-wide plane is 25,238 separate + non-contiguous row segments. A strip's write region is one byte range. On + the Fusion (FUSE/S3) work directory the scattered pattern was pathological + -- 6 tiles took ~3 h against a ~46 min whole-run baseline, see + docs/failures/2026-07-24_imageqc-mmap-over-fusion.md. + * **Half the halo.** A strip needs overlap on top and bottom only, not all + four edges, so less redundant convolution. + + At 48 GB VRAM and a 102,045 px width this gives ~12,700 rows, about 5 strips + for the 5.5 gigapixel reference image. + """ + if gpu_mem_bytes is None: + gpu_mem_bytes = cp.cuda.Device(0).mem_info[1] + + # Same per-pixel budget as the square-tile sizing: ~6 concurrent float32 + # arrays live at once inside _compute_channel_maps_on_gpu. + bytes_per_pixel = 6 * 4 + max_pixels = int(gpu_mem_bytes * target_utilization / bytes_per_pixel) + rows = max(1, max_pixels // max(width, 1)) + + # Enough strips to keep every GPU busy, when the image is tall enough. + wanted = max(n_gpus * min_strips_per_gpu, 1) + if rows * wanted > height: + rows = max(1, height // wanted) + + # Round the strip *count* up to a multiple of the GPU count, then re-derive the + # height from it. Two problems this fixes, both measured on run 3nkeHOEV1ONlbK + # (102045 rows, 4 GPUs): height // wanted gave 12755 rows, so + # ceil(102045/12755) = 9 strips -- eight full ones plus a 5-row sliver -- and 9 + # strips across 4 GPUs left occupancy at 74% / 88% / 100% / 100%. Deriving the + # height from a count of 8 gives 12756 rows, no sliver, and two strips per GPU. + # Increasing the count only ever shrinks strips, so the VRAM bound still holds. + n_strips = max(1, -(-height // rows)) + n_strips = min(((n_strips + n_gpus - 1) // n_gpus) * n_gpus, height) + rows = max(1, -(-height // n_strips)) + + return int(min(rows, height)) + + +def _compute_tile_grid( + height: int, + width: int, + tile_size: int = 8192, + overlap: int = 17, + tile_width: int | None = None, +) -> list[dict[str, int]]: + """Compute overlapping tile coordinates for tiled convolution. + + Each tile has a "write" region (non-overlapping, covers full image) and + a "read" region (expanded by overlap, clipped to image bounds). + + Args: + height: Image height in pixels. + width: Image width in pixels. + tile_size: Core tile height (before overlap). + overlap: Border pixels to add for convolution safety. + tile_width: Core tile width; defaults to *tile_size* (square tiles). Pass + the image width for full-width row strips, whose write region is one + contiguous byte range. + + Returns: + List of tile spec dicts with keys: read_y0, read_y1, read_x0, read_x1, + write_y0, write_y1, write_x0, write_x1, trim_top, trim_bottom, + trim_left, trim_right. + """ + tiles = [] + step_x = int(tile_width) if tile_width else tile_size + for y in range(0, height, tile_size): + for x in range(0, width, step_x): + wy0 = y + wy1 = min(y + tile_size, height) + wx0 = x + wx1 = min(x + step_x, width) + + ry0 = max(0, wy0 - overlap) + ry1 = min(height, wy1 + overlap) + rx0 = max(0, wx0 - overlap) + rx1 = min(width, wx1 + overlap) + + tiles.append( + { + "read_y0": ry0, + "read_y1": ry1, + "read_x0": rx0, + "read_x1": rx1, + "write_y0": wy0, + "write_y1": wy1, + "write_x0": wx0, + "write_x1": wx1, + "trim_top": wy0 - ry0, + "trim_bottom": ry1 - wy1, + "trim_left": wx0 - rx0, + "trim_right": rx1 - wx1, + } + ) + return tiles + + +# --------------------------------------------------------------------------- +# Host memory instrumentation +# --------------------------------------------------------------------------- + +_GIB = float(1024**3) + +# High-water marks across the run, reported in the final summary. +_MEM_PEAK: dict[str, float] = {"working_set": 0.0, "rss": 0.0} + + +# (usage file, stat file, inactive_file key, active_file key) for cgroup v2 then +# v1. AWS Batch nodes run either depending on the ECS AMI, so try both. +_CGROUP_SOURCES = ( + ( + "/sys/fs/cgroup/memory.current", + "/sys/fs/cgroup/memory.stat", + "inactive_file", + "active_file", + ), + ( + "/sys/fs/cgroup/memory/memory.usage_in_bytes", + "/sys/fs/cgroup/memory/memory.stat", + "total_inactive_file", + "total_active_file", + ), +) + + +#: Kernel-maintained high-water usage for the whole cgroup, v2 then v1. Unlike the +#: sampled figures this is continuous and covers every process, so it cannot miss a +#: peak that falls between samples. It does include page cache, which makes it an +#: upper bound rather than an OOM-relevant working set. +_CGROUP_PEAK_PATHS = ( + "/sys/fs/cgroup/memory.peak", + "/sys/fs/cgroup/memory/memory.max_usage_in_bytes", +) + + +def _cgroup_peak() -> float | None: + """Peak cgroup usage in bytes, or None where the counter is absent.""" + for path in _CGROUP_PEAK_PATHS: + try: + with open(path) as fh: + return float(fh.read().strip()) + except (OSError, ValueError): + continue + return None + + +def _cgroup_memory() -> tuple[float, float] | None: + """``(working_set_bytes, page_cache_bytes)`` for this cgroup, or None. + + The usage counter includes reclaimable page cache, and the disk-backed focus + planes deliberately generate a lot of it — reading usage alone would show a + large number and wrongly suggest nothing improved. The quantity that + actually drives an OOM kill is the working set, ``usage - inactive_file``. + Both are reported so the two are never confused when sizing the + process_gpu_qc memory ladder. + """ + for usage_path, stat_path, inactive_key, active_key in _CGROUP_SOURCES: + try: + with open(usage_path) as fh: + usage = float(fh.read().strip()) + inactive_file = 0.0 + active_file = 0.0 + with open(stat_path) as fh: + for line in fh: + key, _, value = line.partition(" ") + if key == inactive_key: + inactive_file = float(value) + elif key == active_key: + active_file = float(value) + except (OSError, ValueError): + continue + return max(usage - inactive_file, 0.0), inactive_file + active_file + return None + + +# Cgroup memory hard limit, v2 then v1. Paired with `_cgroup_memory`'s working-set +# read to size the figure pool by available RAM headroom, not cores alone. +_CGROUP_LIMIT_PATHS = ( + "/sys/fs/cgroup/memory.max", # v2 + "/sys/fs/cgroup/memory/memory.limit_in_bytes", # v1 +) + + +def _cgroup_memory_limit() -> float | None: + """Cgroup memory hard limit in bytes, or None if unlimited/unreadable. + + v2 reports the literal string ``max`` when unlimited; v1 reports a sentinel + close to ``2**63``, so an implausibly large limit is also treated as + unlimited (a real Batch node is <=1 TB, well under ``2**60``). + """ + for path in _CGROUP_LIMIT_PATHS: + try: + with open(path) as fh: + raw = fh.read().strip() + except OSError: + continue + if raw == "max": + return None + try: + value = float(raw) + except ValueError: + continue + if value >= float(1 << 60): + return None + return value + return None + + +def _tree_rss() -> float | None: + """RSS summed over every process in this PID namespace, in bytes. + + ``VmHWM`` from /proc/self/status is the main process only, and the figure phase + forks up to ``_FIGURE_WORKERS_MAX`` children whose matplotlib buffers land on top + of the parent's resident set. On run 3V53J4ewZt1vsU that made the difference + between the 33.4 GB this module reported and the 100.6 GB Tower measured for the + same task -- and the summary line told the reader to size the memory request from + the smaller number, which would under-provision by 3x. + + Restricted to *our own* descendants rather than all of /proc. Summing everything + is right in a container, whose PID namespace holds only our processes, but these + modules also run directly on shared servers where it would silently add other + users' processes to the total. + + Cgroup ``memory.peak`` would also cover children, but it counts reclaimable page + cache — the confusion ``_cgroup_memory`` exists to avoid. + """ + ppid_of: dict[int, int] = {} + rss_of: dict[int, float] = {} + try: + entries = os.listdir("/proc") + except OSError: + return None + for entry in entries: + if not entry.isdigit(): + continue + pid = int(entry) + try: + with open(f"/proc/{pid}/status") as fh: + ppid = None + rss = None + for line in fh: + if line.startswith("PPid:"): + ppid = int(line.split()[1]) + elif line.startswith("VmRSS:"): + rss = float(line.split()[1]) * 1024.0 + if ppid is not None and rss is not None: + break + except (OSError, ValueError, IndexError): + continue # exited between listdir and read, or not readable + if ppid is None: + continue + ppid_of[pid] = ppid + rss_of[pid] = rss or 0.0 + + me = os.getpid() + if me not in rss_of: + return None + # Walk parents to decide membership; the tree is shallow (pool workers are + # direct children), so this stays cheap. + total = 0.0 + for pid in rss_of: + walker = pid + for _ in range(64): # bounded: never loop on a malformed parent chain + if walker == me: + total += rss_of[pid] + break + nxt = ppid_of.get(walker) + if nxt is None or nxt == walker or nxt <= 1: + break + walker = nxt + return total + + +def _process_rss() -> float | None: + """Peak RSS of this process in bytes (VmHWM), excluding children.""" + try: + with open("/proc/self/status") as fh: + for line in fh: + if line.startswith("VmHWM:"): + return float(line.split()[1]) * 1024.0 + except (OSError, ValueError, IndexError): + return None + return None + + +def _log_mem(stage: str) -> None: + """Log host memory at a stage boundary and update the high-water marks. + + Instrumenting in-code rather than diagnosing after the fact: a Tower run + only reports the peak for the whole task, which cannot say *which* stage + drove it. + """ + parts = [] + tree = _tree_rss() + if tree is not None and tree > _MEM_PEAK.get("tree_rss", 0.0): + _MEM_PEAK["tree_rss"] = tree + cgroup = _cgroup_memory() + if cgroup is not None: + working_set, page_cache = cgroup + _MEM_PEAK["working_set"] = max(_MEM_PEAK["working_set"], working_set) + parts.append( + f"working_set={working_set / _GIB:.1f}GB " + f"(+{page_cache / _GIB:.1f}GB reclaimable page cache)" + ) + rss = _process_rss() + if rss is not None: + _MEM_PEAK["rss"] = max(_MEM_PEAK["rss"], rss) + parts.append(f"peak_rss={rss / _GIB:.1f}GB") + if parts: + logging.info(f" [MEM] {stage}: {' '.join(parts)}") + + +def _log_mem_summary() -> None: + """Final high-water summary — the number that sizes the memory request.""" + tree = _MEM_PEAK.get("tree_rss", 0.0) + cgroup_peak = _cgroup_peak() + parts = [ + f"cgroup_peak={cgroup_peak / _GIB:.1f}GB" + if cgroup_peak is not None + else "cgroup_peak=n/a", + f"tree_rss={tree / _GIB:.1f}GB", + f"working_set={_MEM_PEAK['working_set'] / _GIB:.1f}GB", + f"main_process_rss={_MEM_PEAK['rss'] / _GIB:.1f}GB", + ] + logging.info("[MEM] PEAK " + " ".join(parts)) + # Each of these measures something different, and three of the four can understate + # the task's real high-water mark. Spelling out which is which, because an earlier + # version of this summary pointed at one number and was wrong twice over: + # cgroup_peak continuous, all processes, but includes page cache + # tree_rss all processes, but SAMPLED at stage boundaries -- it can miss + # a peak between samples, and has come in *below* + # main_process_rss for exactly that reason + # working_set usage - inactive_file, the OOM-relevant quantity, also sampled + # main_process_rss VmHWM: continuous, but this process only, so it excludes the + # forked figure workers + # The authoritative figure for sizing is the peakRss the Nextflow trace reports for + # the task, which polls the whole process tree; these are for attributing cost to a + # stage, which the trace cannot do. + logging.info( + "[MEM] For sizing use the Tower/trace peakRss for the task; the figures above " + "attribute cost to stages and each understates the total in a different way " + "(see the comment in _log_mem_summary)." + ) + + +# Absolute upper bound on concurrent forked figure workers. Each child inherits +# the parent's address space copy-on-write, so raising the count re-holds only +# each child's OWN matplotlib render buffers (~0.5-1 GB for a full-res 86 Mpx +# imshow), not the shared planes -- but N of those still land on the same cgroup, +# so `_figure_worker_limit` gates the pool on memory headroom, not cores alone. +_FIGURE_WORKERS_MAX = 8 + +#: Peak extra RAM a single figure child adds on top of the shared copy-on-write +#: planes, dominated by matplotlib's render buffers for a full-resolution imshow. +#: Deliberately generous (measured ~0.5-1 GB) so the memory gate errs toward +#: fewer workers rather than an OOM kill. +_FIGURE_PER_WORKER_GB = 2.0 + + +def _cgroup_cpu_quota() -> int | None: + """Cores the cgroup actually permits, or None if unlimited/unreadable. + + ``os.cpu_count()`` reports the host: a container given 30 of a 48-vCPU + instance's cores still sees 48, so sizing a pool from it oversubscribes. + """ + try: + with open("/sys/fs/cgroup/cpu.max") as fh: + raw_quota, raw_period = fh.read().split() + except (OSError, ValueError): + return None + if raw_quota == "max": + return None + try: + return max(1, int(int(raw_quota) / int(raw_period))) + except (ValueError, ZeroDivisionError): + return None + + +def _figure_worker_limit(n_tasks: int) -> int: + """Concurrent figure-rendering processes to allow for ``n_tasks`` figures. + + Gated on BOTH the cgroup CPU quota and the cgroup memory headroom, because a + high-core node can still OOM if every core forks a full-res-imshow child. The + width is:: + + min(n_tasks, cpu_quota, floor(available_gb / per_figure_gb), _FIGURE_WORKERS_MAX) + + where ``available_gb`` is the cgroup memory limit minus the working set + (``usage - inactive_file``, the OOM-relevant quantity; see ``_cgroup_memory``). + When the memory files cannot be read, falls back to the CPU-and-cap width and + logs that memory gating was skipped. The cgroup CPU quota is preferred over + ``os.cpu_count()`` (which reports the host, not the container's share). + Override with IMAGE_QC_FIGURE_WORKERS. + """ + if n_tasks <= 0: + return 1 + + override = os.environ.get("IMAGE_QC_FIGURE_WORKERS") + if override: + width = max(1, min(int(override), n_tasks)) + logging.info(f"[FIGPOOL] width={width} (IMAGE_QC_FIGURE_WORKERS override)") + return width + + cpu = _cgroup_cpu_quota() or os.cpu_count() or 4 + width = min(n_tasks, cpu, _FIGURE_WORKERS_MAX) + + limit = _cgroup_memory_limit() + mem = _cgroup_memory() + if limit is not None and mem is not None: + working_set, _page_cache = mem + available_gb = max(0.0, limit - working_set) / _GIB + mem_width = max(1, int(available_gb // _FIGURE_PER_WORKER_GB)) + width = max(1, min(width, mem_width)) + logging.info( + f"[FIGPOOL] width={width} (tasks={n_tasks}, cpu_quota={cpu}, " + f"mem_avail={available_gb:.1f}GB / {_FIGURE_PER_WORKER_GB:.0f}GB " + f"-> {mem_width}, cap={_FIGURE_WORKERS_MAX})" + ) + else: + width = max(1, width) + logging.info( + f"[FIGPOOL] width={width} (tasks={n_tasks}, cpu_quota={cpu}, " + f"cap={_FIGURE_WORKERS_MAX}, memory gate skipped: cgroup mem unreadable)" + ) + return width + + +def _run_figure_pool(tasks, *, phase: str = "") -> None: + """Render independent figure tasks concurrently in forked child processes. + + Rolling pool: keeps up to ``width`` children alive at once and starts the + next task the instant a slot frees (submit + as-completed semantics), instead + of the old batch barrier that started ``width`` children, joined ALL of them, + then started the next batch -- so the slowest figure in a batch stalled every + idle core until the whole batch drained (measured ~255 s for 11 ROI figures). + + NOT a ``concurrent.futures.ProcessPoolExecutor``: the tasks are closures over + large numpy planes, and the executor's work queue pickles every submitted + item even under a fork context (verified: "Can't pickle local object"). A raw + fork ``Process`` never pickles its target, so children inherit the planes + copy-on-write for free -- the whole reason this phase is fork-based. matplotlib + is not thread-safe, so a thread pool is not an option either. + + Figure-failure semantics match the pre-pool code exactly: a task that raises + logs its full traceback in the child (``_run_figure_task``) and the step + CONTINUES, producing every other figure plus all metrics -- a single broken + figure never nukes the QC step. The pool additionally logs any non-zero child + exit (e.g. a hard crash / OOM-kill the old barrier ignored silently) naming the + task, then carries on. This is a performance change only, not a change to what + a figure failure does. + """ + tasks = list(tasks) + if not tasks: + return + + width = _figure_worker_limit(len(tasks)) + mp_ctx = multiprocessing.get_context("fork") + logging.info(f"[FIGPOOL] {phase or 'figures'}: dispatching {len(tasks)} task(s)") + + pending = list(reversed(tasks)) # pop() from the end preserves task order + running: dict[Any, tuple[Any, str]] = {} # sentinel fd -> (process, name) + failures: list[tuple[str, int]] = [] + + def _launch() -> None: + fn = pending.pop() + p = mp_ctx.Process(target=_run_figure_task, args=(fn,)) + p.start() + running[p.sentinel] = (p, fn.__name__, time.perf_counter()) + + while pending and len(running) < width: + _launch() + + while running: + # Block until at least one child exits; no busy-wait. A process sentinel + # is a file descriptor that becomes ready when the process terminates. + for sentinel in multiprocessing.connection.wait(list(running)): + p, name, started = running.pop(sentinel) + p.join() + logging.info( + f" [TIMING] figure {name} ({phase or 'figures'}): " + f"{time.perf_counter() - started:.1f}s" + ) + if p.exitcode != 0: + failures.append((name, p.exitcode)) + if pending: + _launch() + + if failures: + detail = ", ".join(f"{name} (exit {code})" for name, code in failures) + # Match the pre-pool barrier: log loudly, do not abort the step. The child + # already logged the Python traceback via _run_figure_task; this covers + # non-zero exits (hard crash / OOM-kill) the old p.join() ignored silently. + logging.error( + f"[FIGPOOL] {phase or 'figures'}: figure task(s) failed: {detail}" + ) + + +# --------------------------------------------------------------------------- +# Tile consumers — fold a tile into small state, then drop it +# +# The disk-backed planes below bound host RAM, but they are not the right final +# design: they need ~154 GB of scratch on a 5.5 GP sample, mmap over a FUSE/S3 +# work directory is pathological (see +# docs/failures/2026-07-24_imageqc-mmap-over-fusion.md), and these modules must +# run local / server / cloud on their way to nf-core/spatialaxe. +# +# No consumer of the full-resolution maps needs a whole map: ROI sampling reads +# one pixel per tile, SNR already loops per ROI window, the per-cell means are +# additive (sum, count) reductions, and the focus heatmap needs an 8x block mean. +# A consumer receives each finished tile and folds it into state that scales with +# the number of ROIs or cells, never with image size. +# --------------------------------------------------------------------------- + + +def _is_device_array(array: Any) -> bool: + """True when *array* is a CuPy device array (so its reduction folds on-GPU).""" + return HAS_CUPY and isinstance(array, cp.ndarray) + + +def _device_or_host(maps: dict[str, Any], key: str) -> Any: + """The device variant (``key + "_device"``) of a map if the tile kept one on the + GPU, else the host map under *key*. Lets a consumer fold on-device transparently: + the streaming GPU path supplies ``focus_map_device`` / ``mean_map_device`` while + the CPU / plane path supplies only the host ``focus_map`` / ``mean_map``. + """ + device = maps.get(key + "_device") + return maps.get(key) if device is None else device + + +class CentrePixelSampler: + """Sample one pixel per ROI from tiles as they are produced. + + Replaces the centre-pixel sampling in ``downsample_maps_to_roi_dataframe``, + which indexes ``map[cy, cx]`` on the assembled full-resolution planes. That + pinned ~154 GB in order to read one pixel per ROI — 0.02 % of it. + + A tile's *trimmed* arrays cover exactly its write region, so an ROI whose + centre lies in ``[write_y0, write_y1) x [write_x0, write_x1)`` is sampled at + local offset ``(cy - write_y0, cx - write_x0)``. Write regions are disjoint + and cover the image (``_compute_tile_grid``; asserted by + ``test_no_write_overlap``), so every ROI is sampled exactly once. + """ + + wants_untrimmed = False + #: This consumer does NOT read the host-side mean map, so the tile pass + #: may drop that D2H copy. A consumer that needs it must set this True. + reads_host_mean = False + #: Row/col multiple this consumer needs its tile write origins to fall on. + #: 1 means any split is fine. See BlockMeanAccumulator for why it needs more. + write_alignment = 1 + #: Take the focus / mean maps from the GPU (``*_device``) when the tile pass + #: kept them resident, so the one-pixel-per-ROI gather runs on-device and the + #: full maps are not read back for it. The gather is a pure index, so the + #: sampled float32 pixel is bit-identical to the host read. + wants_device_focus = True + wants_device_mean = True + + def __init__(self, cy: NDArray[np.integer], cx: NDArray[np.integer], keys): + self.cy = np.asarray(cy) + self.cx = np.asarray(cx) + self.values: dict[str, NDArray[np.float64]] = { + key: np.full(self.cy.size, np.nan, dtype=np.float64) for key in keys + } + # Sanity: every ROI must be claimed by exactly one tile. + self._claimed = np.zeros(self.cy.size, dtype=np.int8) + + def consume(self, tile_spec: dict[str, int], maps: dict[str, Any]) -> None: + """Fold one tile's trimmed maps into the per-ROI arrays. + + Each value map is taken device-first (``_device_or_host``): on the streaming + GPU path the gather runs on the resident ``focus_map_device`` / + ``mean_map_device`` and only the sampled pixels come back to host; on the CPU + / plane path the host maps are gathered exactly as before. A gather is a pure + index with no arithmetic, so the device and host reads return the identical + float32 pixel. + """ + wy0, wy1 = tile_spec["write_y0"], tile_spec["write_y1"] + wx0, wx1 = tile_spec["write_x0"], tile_spec["write_x1"] + inside = (self.cy >= wy0) & (self.cy < wy1) & (self.cx >= wx0) & (self.cx < wx1) + if not inside.any(): + return + local_y = self.cy[inside] - wy0 + local_x = self.cx[inside] - wx0 + self._claimed[inside] += 1 + for key in self.values: + array = _device_or_host(maps, key) + if array is None: + continue + if _is_device_array(array): + # Gather on the array's own device (the fold runs after the GPU was + # returned to the pool, so the thread's current device may differ). + with array.device: + ly = cp.asarray(local_y) + lx = cp.asarray(local_x) + sampled = cp.asnumpy(array[ly, lx]).astype(np.float64) + else: + sampled = np.asarray(array)[local_y, local_x].astype(np.float64) + self.values[key][inside] = sampled + + def finalize(self) -> dict[str, NDArray[np.float64]]: + """Return the per-ROI arrays, checking every ROI was covered once.""" + unclaimed = int((self._claimed == 0).sum()) + duplicated = int((self._claimed > 1).sum()) + if unclaimed or duplicated: + raise RuntimeError( + f"ROI coverage is wrong: {unclaimed} ROI(s) claimed by no tile, " + f"{duplicated} by more than one. Tile write regions must be " + "disjoint and cover the image." + ) + return dict(self.values) + + def spawn(self) -> "CentrePixelSampler": + """A fresh, empty sampler sharing this one's ROI grid and keys. + + Used by the tile pass to give each worker thread its own accumulator so + folds run without a lock; the partials are then reduced with ``merge``. + """ + return CentrePixelSampler(self.cy, self.cx, list(self.values)) + + def merge(self, other: "CentrePixelSampler") -> None: + """Fold a per-worker partial into this one. + + Each ROI is claimed by exactly one tile, hence by exactly one worker, so the + partials hold DISJOINT per-ROI entries: the merge is a pure overlay of the + ROIs *other* claimed, with no arithmetic and so bit-identical to the serial + fold regardless of order. The claim counts add so ``finalize`` still verifies + global coverage. + """ + overlap = (self._claimed > 0) & (other._claimed > 0) + assert not overlap.any(), ( + "CentrePixelSampler partials claim overlapping ROIs; tile write regions " + "must be disjoint (each ROI centre lands in exactly one write region)." + ) + claimed = other._claimed > 0 + self._claimed += other._claimed + for key in self.values: + self.values[key][claimed] = other.values[key][claimed] + + +class LazyLabelPlane: + """Lazily-sliced view over a full-resolution label plane in zarr. + + ``load_spatial_data`` does ``np.array(cell_masks_zarr["masks"]["1"])``, which + materialises 22 GB of uint32 on a 5.5 gigapixel sample, and + ``calculate_ccfs_from_focus_maps`` did the same for ``masks/0``. Neither needs + the whole plane: the consumers of these arrays are row-block reductions and + scattered point lookups, both of which this serves from zarr on demand. + + Supports the three access patterns the callers actually use: + + * ``plane[y0:y1]`` and ``plane[y0:y1, x0:x1]`` — row/rect slices, passed + straight through to zarr. + * ``plane[iy, ix]`` with integer arrays — coordinate lookup, served by reading + only the row blocks the points fall in. ``masks[y, x]`` on a zarr array does + not do numpy-style pair indexing, so this is why the wrapper exists. + """ + + def __init__(self, source, rows_per_chunk: int | None = None): + self._source = source + self.shape: tuple[int, int] = (int(source.shape[0]), int(source.shape[1])) + self.dtype = getattr(source, "dtype", None) + width = max(self.shape[1], 1) + self.rows_per_chunk = rows_per_chunk or max(1, _LABEL_CHUNK_PIXELS // width) + + def __getitem__(self, key): + # Coordinate lookup: two integer arrays. + if ( + isinstance(key, tuple) + and len(key) == 2 + and all( + isinstance(part, (np.ndarray, list)) or np.isscalar(part) + for part in key + ) + and not any(isinstance(part, slice) for part in key) + ): + return self._gather(np.asarray(key[0]), np.asarray(key[1])) + return np.asarray(self._source[key]) + + def _gather( + self, rows: NDArray[np.integer], cols: NDArray[np.integer] + ) -> NDArray[Any]: + """Point lookup, reading only the row blocks the points land in.""" + rows = np.asarray(rows, dtype=np.int64).ravel() + cols = np.asarray(cols, dtype=np.int64).ravel() + if rows.size != cols.size: + raise ValueError("row and column index arrays must be the same length") + out = None + for start in range(0, self.shape[0], self.rows_per_chunk): + stop = min(start + self.rows_per_chunk, self.shape[0]) + inside = (rows >= start) & (rows < stop) + if not inside.any(): + continue + block = np.asarray(self._source[start:stop]) + if out is None: + out = np.zeros(rows.size, dtype=block.dtype) + out[inside] = block[rows[inside] - start, cols[inside]] + del block + if out is None: + out = np.zeros(rows.size, dtype=self.dtype or np.int64) + return out + + def max(self) -> int: + """Largest label value, read in row blocks.""" + highest = 0 + for start in range(0, self.shape[0], self.rows_per_chunk): + stop = min(start + self.rows_per_chunk, self.shape[0]) + block = np.asarray(self._source[start:stop]) + if block.size: + highest = max(highest, int(block.max())) + del block + return highest + + +class LabeledSumAccumulator: + """Accumulate per-label ``(count, sum)`` from tiles as they are produced. + + The same additive reduction as ``_labeled_sums_chunked``, keyed on tile write + regions instead of row blocks. Counts and sums are additive, so a cell split + across tiles contributes partial sums to each and its final ``sum / count`` is + exact — this is not an approximation. + + State is ``O(n_labels)``: two arrays per value plane, tens of MB for ~530 k + cells, against the 22 GB per plane the whole-map path needed. The label plane + is sliced per tile, so it is never materialised either — which also removes + the ``cellseg_mask`` array the current path still holds. + + The per-label reduction (``xp.bincount``) is ``xp``-generic. On the streaming GPU + path the value maps arrive resident on the device (``focus_map_device`` / + ``mean_map_device``); this uploads the int label block — cheaper than reading the + float value maps back — and runs every ``bincount`` on the GPU, returning only the + per-label ``O(n_labels)`` vectors to host. That moves the largest term of the tile + fold (~112 s/channel of host ``bincount``) onto the otherwise-idle GPU. The counts + and float64 sums are the same additive reduction either way, and the GPU float64 + ``bincount`` matches the host result to float64 rounding (tests/test_tile_consumers). + On the CPU / plane path the maps are host numpy and the fold is byte-identical to + before. + """ + + wants_untrimmed = False + #: This consumer does NOT read the host-side mean map, so the tile pass + #: may drop that D2H copy. A consumer that needs it must set this True. + reads_host_mean = False + #: Row/col multiple this consumer needs its tile write origins to fall on. + #: 1 means any split is fine. See BlockMeanAccumulator for why it needs more. + write_alignment = 1 + #: Fold from the GPU-resident maps when the tile pass kept them there, so the + #: per-label bincounts run on-device instead of a serial host fold. + wants_device_focus = True + wants_device_mean = True + + def __init__( + self, + label_plane, + value_keys, + include_coords: bool = False, + skip_background: bool = False, + ): + self.label_plane = label_plane + self.include_coords = include_coords + # Drop label-0 pixels before reducing. Only valid where the caller never reads + # index 0: the nuclear reduction does `labels[labels > 0]`, but the *cell* + # reduction deliberately reads `cell_counts[0]` as the background pixel count + # for CellID 0, so it must keep them. + self.skip_background = skip_background + self.counts = np.zeros(1, dtype=np.int64) + self.sums: dict[str, NDArray[np.float64]] = { + key: np.zeros(1, dtype=np.float64) for key in value_keys + } + if include_coords: + self.sums["centroid_y_sum"] = np.zeros(1, dtype=np.float64) + self.sums["centroid_x_sum"] = np.zeros(1, dtype=np.float64) + + def _add(self, key: str, block: NDArray[np.float64]) -> None: + self.sums[key] = _grow_to(self.sums[key], block.size) + self.sums[key][: block.size] += block + + def consume(self, tile_spec: dict[str, int], maps: dict[str, Any]) -> None: + """Fold one tile's trimmed maps into the per-label accumulators. + + Runs on the GPU when the tile kept its maps device-resident (``xp=cp``), + else on host (``xp=np``); ``_fold_blocks`` is written once against ``xp``. + Blocked by rows within the tile, for the same reason + ``_labeled_sums_chunked`` blocks: ``bincount`` needs ``intp`` labels and + ``float64`` weights, so a whole-tile call casts both. A production tile is + a full-width row strip, so that is 2.7 GB of labels plus 5.5 GB per + coordinate array — larger than the tile's own maps. One block is ~32 M + pixels regardless of tile size, which also keeps the on-device label upload + and the ``K*256``-free bincount well under the tile's VRAM budget. + """ + wx0, wx1 = tile_spec["write_x0"], tile_spec["write_x1"] + tile_width = wx1 - wx0 + if tile_width <= 0 or tile_spec["write_y1"] <= tile_spec["write_y0"]: + return + + # Take each value map device-first, then decide the fold backend from what the + # tile actually handed over. A value-less counts-only accumulator (cells) has no + # value map to check, so fall back to any resident device array in the tile. + value_maps = { + key: _device_or_host(maps, key) + for key in self.sums + if not key.startswith("centroid_") + } + device_arr = next((a for a in value_maps.values() if _is_device_array(a)), None) + if device_arr is None: + device_arr = next((a for a in maps.values() if _is_device_array(a)), None) + + if device_arr is not None: + # Run every bincount on the map's own device (the fold happens after the + # GPU is returned to the pool, so the thread's current device may differ). + with device_arr.device: + self._fold_blocks(tile_spec, value_maps, cp, on_gpu=True) + else: + self._fold_blocks(tile_spec, value_maps, np, on_gpu=False) + + def _fold_blocks( + self, + tile_spec: dict[str, int], + value_maps: dict[str, Any], + xp: Any, + on_gpu: bool, + ) -> None: + """Row-blocked per-label reduction over one tile, on ``xp`` (numpy or cupy). + + Reduces on-device when ``xp`` is cupy, pulling only the ``O(n_labels)`` result + vectors back to the host accumulators; the intermediate labels/values/coords + never leave the device. + """ + wy0, wy1 = tile_spec["write_y0"], tile_spec["write_y1"] + wx0, wx1 = tile_spec["write_x0"], tile_spec["write_x1"] + tile_width = wx1 - wx0 + rows_per_chunk = max(1, _LABEL_CHUNK_PIXELS // tile_width) + + def _host(arr: Any) -> Any: + return cp.asnumpy(arr) if on_gpu else arr + + for y0 in range(wy0, wy1, rows_per_chunk): + y1 = min(y0 + rows_per_chunk, wy1) + labels_host = np.asarray(self.label_plane[y0:y1, wx0:wx1]).ravel() + if labels_host.size == 0: + continue + # Upload the int label block to the device (cheaper than reading the float + # value maps back), or keep it on host for the numpy path. + labels = xp.asarray(labels_host) if on_gpu else labels_host + del labels_host + + # Restrict to labelled pixels where index 0 is never read. On the reference + # sample the nuclear mask is 9.8 % non-zero (505 k nuclei x ~1066 px of + # 5.50 G), so this drops ~90 % of the work from every pass below -- + # measured at ~169 s for this accumulator, the largest term in the fold + # after the Otsu SNR. `flatnonzero` also lets the coordinates be built at + # the compressed size instead of materialising a full-chunk repeat/tile. + selected = None + if self.skip_background: + selected = xp.flatnonzero(labels) + labels = labels[selected] + if labels.size == 0: + continue + + block_counts = xp.bincount(labels) + n = int(block_counts.size) + self.counts = _grow_to(self.counts, n) + self.counts[:n] += _host(block_counts) + + # value_maps rows are trimmed to the write region, so map row `y0 - wy0` + # is global row `y0`. + for key, array in value_maps.items(): + if array is None: + continue + values = xp.asarray( + array[y0 - wy0 : y1 - wy0], dtype=xp.float64 + ).ravel() + if selected is not None: + values = values[selected] + self._add(key, _host(xp.bincount(labels, weights=values, minlength=n))) + del values + + if self.include_coords: + # Global coordinates, so sum / count is regionprops' centroid in + # image space rather than tile-local space. + if selected is not None: + rows = (y0 + selected // tile_width).astype(xp.float64) + cols = (wx0 + selected % tile_width).astype(xp.float64) + else: + rows = xp.repeat(xp.arange(y0, y1, dtype=xp.float64), tile_width) + cols = xp.tile(xp.arange(wx0, wx1, dtype=xp.float64), y1 - y0) + self._add( + "centroid_y_sum", + _host(xp.bincount(labels, weights=rows, minlength=n)), + ) + del rows + self._add( + "centroid_x_sum", + _host(xp.bincount(labels, weights=cols, minlength=n)), + ) + del cols + del labels, block_counts, selected + + def finalize(self) -> tuple[NDArray[np.int64], dict[str, NDArray[np.float64]]]: + """Return ``(counts, sums)`` indexed by raw label value, 0 = background.""" + return self.counts, dict(self.sums) + + def spawn(self) -> "LabeledSumAccumulator": + """A fresh, empty accumulator sharing this one's label plane and keys. + + Used by the tile pass to give each worker thread its own accumulator so + folds run without a lock; the partials are then reduced with ``merge``. The + label plane is shared by reference (read-only, per-tile slices), not copied. + """ + value_keys = [k for k in self.sums if not k.startswith("centroid_")] + return LabeledSumAccumulator( + self.label_plane, + value_keys, + include_coords=self.include_coords, + skip_background=self.skip_background, + ) + + def merge(self, other: "LabeledSumAccumulator") -> None: + """Fold a per-worker partial into this one by element-wise addition. + + Counts and per-label float64 sums are additive, so adding a worker's + partial totals reduces the same values as the serial per-tile fold -- but + it regroups the float64 additions (per-worker subtotals then combined, + rather than one running sum in tile order). Reassociation of float64 sums + can differ by ULPs; that this reproduces the serial fold byte-for-byte is + empirical and is pinned by tests/test_fold_parallel_equivalence, not + assumed. Arrays are grown to the longer length before adding. + """ + n = max(self.counts.size, other.counts.size) + self.counts = _grow_to(self.counts, n) + self.counts[: other.counts.size] += other.counts + for key, total in other.sums.items(): + self.sums[key] = _grow_to(self.sums[key], total.size) + self.sums[key][: total.size] += total + + def means(self, labels: NDArray[np.integer]) -> dict[str, NDArray[np.float64]]: + """Per-label means for *labels*, NaN where a label has no pixels.""" + counts = self.counts[labels].astype(np.float64) + out = {} + with np.errstate(invalid="ignore", divide="ignore"): + for key, total in self.sums.items(): + out[key] = total[labels] / counts + return out + + +def _block_sum( + array: NDArray[np.generic], y_offset: int, x_offset: int, factor: int +) -> tuple[NDArray[np.float64], NDArray[np.intp], NDArray[np.intp]]: + """Sum *array* into the global factor-grid blocks it overlaps. + + ``array[0, 0]`` sits at global ``(y_offset, x_offset)``. Returns the block + sums plus the global block row/column indices they belong to. + + Two ``np.add.reduceat`` passes rather than a flat block-index array: for a + 36k x 36k tile the index array alone would be 1.3e9 int64 = 10 GB, whereas + the row-reduced intermediate is (n_block_rows, width). ``reduceat`` also + handles the ragged first/last groups natively when the tile does not start on + a block boundary. + """ + values = np.asarray(array, dtype=np.float64) + rows = np.arange(values.shape[0]) + y_offset + row_starts = np.flatnonzero( + np.r_[True, (rows[1:] // factor) != (rows[:-1] // factor)] + ) + partial = np.add.reduceat(values, row_starts, axis=0) + + cols = np.arange(values.shape[1]) + x_offset + col_starts = np.flatnonzero( + np.r_[True, (cols[1:] // factor) != (cols[:-1] // factor)] + ) + block = np.add.reduceat(partial, col_starts, axis=1) + return block, rows[row_starts] // factor, cols[col_starts] // factor + + +class BlockMeanAccumulator: + """Build a factor-``f`` down-sampled canvas from tiles, matching skimage. + + ``_fig5_focus_heatmap`` calls ``downscale_local_mean(dapi_focus, (8, 8))`` on + the assembled plane — the only use of that plane, and it pads the array before + reducing, so on a 5.5 GP sample it costs another ~22 GB on top of the 22 GB + plane it reads. + + ``downscale_local_mean`` zero-pads incomplete blocks and divides by the + *full* block size, so accumulating per-block sums and dividing by + ``factor**2`` reproduces it **bit-exactly**, including the partial final block, and + that is verified against skimage rather than argued (``TestBlockMeanEpsilon``). + + Exactness needs two things, and the *dtype* one is by far the larger: + + First, the same precision. Production focus maps are float32, so the plane-based path + reduces in float32; this class reduces in whatever dtype it is handed and must not + widen it. An earlier version upcast to float64 and disagreed with skimage in 200 of + 200 measured random planes, by up to 1.86e-07 relative. + + Second, *grouping*: every block reduced by one addition chain, never + as two partials summed across a tile seam. The dispatcher arranges that by aligning + write boundaries to ``factor`` -- it reads ``write_alignment`` off each consumer, so + a caller cannot forget -- and ``consume`` refuses an unaligned write rather than + silently accumulating one. + + The axis order within a block turns out not to matter -- ``sum(axis=(1, 3))`` and + ``sum(axis=3).sum(axis=1)`` agree with skimage bit-for-bit over 300 random + shapes/magnitudes, and mutating one to the other is an equivalent mutant. Only the + grouping is load-bearing. + + An earlier version summed with two ``np.add.reduceat`` passes over an unaligned grid, + so a block straddling a seam *was* accumulated as two partials. That is real, but it + contributes only ~3.5e-16 and was **not** the cause of the 16 pixels of Figure 5 that + differed from the plane-based path: the float64 upcast above was, at ~1e-7. Fixing + the ordering alone left the output byte-for-byte unchanged (16 px, delta 1) -- + see ``docs/failures/2026-07-26_heatmap-16px-dtype-not-summation-order.md``. + + State is the ``(H/f, W/f)`` canvas: 0.34 GB at f=8 on a 5.5 GP sample, + against 44 GB for the plane plus skimage's pad. + """ + + #: Every consumer takes ``consume(tile_spec, maps)`` where *maps* is the dict + #: of this tile's channel maps. A uniform signature lets the tile dispatcher + #: feed them all without knowing which is which. + wants_untrimmed = False + #: This consumer does NOT read the host-side mean map, so the tile pass + #: may drop that D2H copy. A consumer that needs it must set this True. + reads_host_mean = False + #: Row/col multiple this consumer needs its tile write origins to fall on. + #: 1 means any split is fine. See BlockMeanAccumulator for why it needs more. + write_alignment = 1 + + def __init__( + self, + shape: tuple[int, int], + factor: int = 8, + map_key: str = "focus_map", + ): + self.map_key = map_key + self.factor = int(factor) + # Blocks must not straddle tiles, so the dispatcher must land write origins on + # multiples of the factor. Declared here rather than passed in by the caller: + # a caller that forgot would only find out via consume()'s RuntimeError. + self.write_alignment = self.factor + #: dtype of the first array consumed; the canvas is returned in it. + self._out_dtype: np.dtype | None = None + self.shape = (int(shape[0]), int(shape[1])) + rows = -(-self.shape[0] // self.factor) # ceil + cols = -(-self.shape[1] // self.factor) + self.sums = np.zeros((rows, cols), dtype=np.float64) + + def consume(self, tile_spec: dict[str, int], maps: dict[str, Any]) -> None: + """Fold this tile's ``map_key`` plane into the block canvas. + + Reduces each block with a single reshape, matching ``downscale_local_mean``'s own + grouping and summation order, which makes the canvas **bit-exact** rather than + equal to within a float64 epsilon. That requires every block to lie wholly inside + one tile, which the dispatcher guarantees by aligning write boundaries to + ``factor``; the check below is that contract, and it raises rather than silently + degrading. + + Previously this used two ``np.add.reduceat`` passes, so a block straddling a seam + was summed as two partials in a different order from skimage. The resulting + ~3.5e-16 discrepancy reached the rendered figure: 16 of Figure 5's 5,553,835 + pixels landed 1/255 apart from the plane-based path. + + The image's own trailing rows and columns may still form a partial block; those + are zero-padded exactly as skimage pads, which is bit-exact for dimensions that + are not multiples of the factor. + """ + array = maps.get(self.map_key) + if array is None: + return + # Reduce in the INPUT's dtype, never a wider one. Production focus maps are + # float32 and the plane-based path calls downscale_local_mean on them, so it + # reduces in float32; accumulating in float64 here disagreed with it by up to + # 1.86e-07 relative in 200 of 200 measured cases -- enough to move 16 of Figure + # 5's 5.5M pixels across a uint8 boundary. That dtype term is ~5e8 times larger + # than the summation-order term this class used to blame. + # No integer special-case: numpy already promotes e.g. uint16 sums to uint64, + # which lands on skimage's float64 mean exactly. Verified, not assumed -- + # test_integer_input_follows_skimage_to_float64, and a mutation removing an + # explicit promotion survived because it was doing nothing. + values = np.asarray(array) + if self._out_dtype is None: + # The canvas must come out in the dtype downscale_local_mean would have + # produced, not merely with the same values. Figure 5 takes its colour scale + # from np.percentile of this array, and percentile on a float64 container + # returns 2082.678369140625 where float32 returns 2082.6785 -- a different + # vmin/vmax, which flips pixels sitting on a colormap boundary. That was the + # third and last cause of the differing heatmap pixels. + self._out_dtype = ( + values.dtype + if np.issubdtype(values.dtype, np.floating) + else np.dtype(np.float64) # skimage's np.mean promotes integers + ) + wy0, wx0 = tile_spec["write_y0"], tile_spec["write_x0"] + f = self.factor + if wy0 % f or wx0 % f: + raise RuntimeError( + f"tile write origin ({wy0}, {wx0}) is not aligned to the heatmap block " + f"size {f}: blocks would straddle tiles and the canvas would stop " + "matching downscale_local_mean exactly. _compute_channel_maps_tiled " + "derives this from each consumer's write_alignment, so reaching here " + "means the accumulator was driven by something else." + ) + pad_y = (-values.shape[0]) % f + pad_x = (-values.shape[1]) % f + if pad_y or pad_x: + values = np.pad(values, ((0, pad_y), (0, pad_x)), mode="constant") + blocks = values.reshape(values.shape[0] // f, f, values.shape[1] // f, f).sum( + axis=(1, 3) + ) + r0, c0 = wy0 // f, wx0 // f + self.sums[r0 : r0 + blocks.shape[0], c0 : c0 + blocks.shape[1]] += blocks + + def finalize(self) -> NDArray[np.floating]: + """Return the block means, identical to ``downscale_local_mean`` in value *and* + dtype. The dtype is part of the contract: Figure 5 derives its colour scale from + ``np.percentile`` of this array, which answers differently in float32 and float64. + """ + means = self.sums / float(self.factor * self.factor) + if self._out_dtype is not None and means.dtype != self._out_dtype: + means = means.astype(self._out_dtype) + return means + + def spawn(self) -> "BlockMeanAccumulator": + """A fresh, empty canvas of the same shape / factor / map key. + + Used by the tile pass to give each worker thread its own accumulator so + folds run without a lock; the partials are then reduced with ``merge``. + """ + return BlockMeanAccumulator( + self.shape, factor=self.factor, map_key=self.map_key + ) + + def merge(self, other: "BlockMeanAccumulator") -> None: + """Fold a per-worker partial canvas into this one. + + Write origins are aligned to ``factor`` (the dispatcher enforces it via + ``write_alignment``), so every block lies wholly inside one tile and hence + one worker. The per-worker canvases are therefore DISJOINT -- each block is + non-zero in exactly one partial -- so the element-wise add is a pure overlay + (``0.0 + x == x``) with no reassociation, bit-identical to the serial fold. + """ + self.sums += other.sums + if other._out_dtype is not None: + if self._out_dtype is not None and self._out_dtype != other._out_dtype: + raise RuntimeError( + f"BlockMeanAccumulator partials disagree on output dtype " + f"({self._out_dtype} vs {other._out_dtype}); every tile's map " + "must have the same dtype." + ) + self._out_dtype = other._out_dtype + + +#: ROIs folded per batched-Otsu call. Two constraints set it: +#: +#: 1. **cupy bincount ceiling (hard).** ``roi_snr_db_batch`` builds a per-row +#: histogram as one flattened ``cp.bincount`` over ``K*1225`` indices into +#: ``K*256`` bins. In cupy 14.0.1 that bincount raises +#: ``cudaErrorIllegalAddress`` once K is large: measured OK at K=40_000 +#: (10.2M bins / 49M elems), FAIL at K=50_000 (12.8M / 61.25M). This is what +#: crashed Tower run 4bBYe2TAP5QEqA on the first fold. Staying an order of +#: magnitude under that boundary makes the fault structurally impossible and +#: leaves headroom for other GPUs/cupy builds. +#: 2. VRAM: bounds the ``(K, N)`` / ``(K, 256)`` float64 temporaries so a tall +#: strip's owned ROIs stay well under the tile's ~10.5 GB budget. +#: +#: 8192 gives ~5x margin on (1) and a small device peak; the extra kernel launches +#: over a larger chunk are negligible against the fold. See +#: docs/failures/2026-07-27_gpu-otsu-cuda-illegal-address.md. +_OTSU_ROI_CHUNK = 8_192 + + +class RoiOtsuSnrAccumulator: + """Per-ROI Otsu SNR in dB, computed from tiles as they are produced. + + Replaces ``snr_metrics.compute_image_snr_from_pixel_maps``'s loop over + ``dapi_mean_map[y1:y2, x1:x2]``, which needed the assembled 22 GB plane. + + The reduction is ``xp``-generic. In the streaming GPU path the tile pass hands + over the tile's **device** ``mean_map`` (a CuPy array still resident in VRAM, + under key ``"mean_map_device"``) and the batched Otsu SNR + (``snr_metrics.roi_snr_db_batch``, ``xp=cp``) folds every owned interior ROI in + one vectorized GPU pass, returning only the small per-ROI dB vector to host -- + instead of the 4.49 M-iteration Python loop over host windows this class used + to run (~439 s of CPU). On the CPU / no-GPU path the map is a numpy array and + the identical batched kernel runs with ``xp=np``. Either way the result matches + the per-ROI scalar ``snr_metrics.roi_snr_db`` to float64 rounding (~1e-14 dB; + see ``tests/test_roi_snr_db_batch.py``). + + Windows clipped at the image edge (variable size) or holding a non-finite + pixel fall back to the scalar ``roi_snr_db``, which the batched kernel -- it + assumes finite, full-size windows -- does not model. Only the full-size + interior windows, which share one shape, are batched. + + Unlike the other consumers this one needs the **untrimmed** tile: an ROI is + owned by the tile whose write region contains its top-left corner, and its + window then extends up to ``roi_size - 1`` px beyond that write region. The + read region extends by ``overlap``, so the window is covered exactly when + ``overlap >= roi_size``. That is currently true (both are ``window_size``, + 35) but it is a real constraint, so a violation raises rather than silently + truncating a window and reporting a wrong dB. + """ + + #: Tells the tile dispatcher to hand this consumer the haloed tile. + wants_untrimmed = True + #: This consumer does NOT read the host-side mean map, so the tile pass + #: may drop that D2H copy. A consumer that needs it must set this True. + reads_host_mean = False + #: Tells the tile dispatcher to keep the tile's mean map on the GPU and pass + #: the device array (``"mean_map_device"``) so the Otsu SNR folds on-device. + wants_device_mean = True + write_alignment = 1 + + def __init__( + self, + y1: NDArray[np.integer], + y2: NDArray[np.integer], + x1: NDArray[np.integer], + x2: NDArray[np.integer], + image_shape: tuple[int, int], + map_key: str = "mean_map", + ): + self.map_key = map_key + self.y1 = np.asarray(y1, dtype=np.int64) + self.y2 = np.asarray(y2, dtype=np.int64) + self.x1 = np.asarray(x1, dtype=np.int64) + self.x2 = np.asarray(x2, dtype=np.int64) + self.height, self.width = int(image_shape[0]), int(image_shape[1]) + self.db = np.full(self.y1.size, np.nan, dtype=np.float64) + self._claimed = np.zeros(self.y1.size, dtype=np.int8) + # The full ROI side; interior windows equal it, edge-clipped windows are + # smaller. Windows of exactly this shape are the ones the batched kernel + # folds together (it needs one uniform shape); everything else is scalar. + self._full_h = int((self.y2 - self.y1).max()) if self.y1.size else 0 + self._full_w = int((self.x2 - self.x1).max()) if self.x1.size else 0 + + def consume(self, tile_spec: dict[str, int], maps: dict[str, Any]) -> None: + """Fold this tile's *untrimmed* (haloed) mean map into the per-ROI dB array. + + Reads the device mean map (``"mean_map_device"``) when the tile pass kept + one resident, else the host ``map_key``; ``xp`` follows the array type so + the same code batches on GPU (``cp``) or CPU (``np``). Full-size interior + windows go through ``snr_metrics.roi_snr_db_batch`` in bounded chunks; + edge-clipped or non-finite windows fall back to the scalar + ``roi_snr_db``. + """ + # The device map the tile pass keeps resident is the mean map specifically + # (``mean_map_device``). Only reach for it when this accumulator is actually + # configured for the mean map; otherwise fall through to the configured + # ``map_key``. Without this guard a focus-configured instance would silently + # reduce the mean map on a GPU tile and return plausible but wrong numbers. + array = maps.get("mean_map_device") if self.map_key == "mean_map" else None + if array is None: + array = maps.get(self.map_key) + if array is None: + return + + on_gpu = HAS_CUPY and isinstance(array, cp.ndarray) + xp = cp if on_gpu else np + + ry0, ry1 = tile_spec["read_y0"], tile_spec["read_y1"] + rx0, rx1 = tile_spec["read_x0"], tile_spec["read_x1"] + wy0, wy1 = tile_spec["write_y0"], tile_spec["write_y1"] + wx0, wx1 = tile_spec["write_x0"], tile_spec["write_x1"] + + owned = (self.y1 >= wy0) & (self.y1 < wy1) & (self.x1 >= wx0) & (self.x1 < wx1) + owned_idx = np.flatnonzero(owned) + if owned_idx.size == 0: + return + + # Clip each owned window to the image exactly as the whole-map path does. + y_start = np.maximum(0, self.y1[owned_idx]) + x_start = np.maximum(0, self.x1[owned_idx]) + y_stop = np.minimum(self.height, self.y2[owned_idx]) + x_stop = np.minimum(self.width, self.x2[owned_idx]) + self._claimed[owned_idx] += 1 + + valid = (y_stop > y_start) & (x_stop > x_start) + # The read region must cover every non-empty window, or a window would be + # silently truncated and its dB wrong. Raise, as the per-ROI path did. + covered = ( + (ry0 <= y_start) & (y_stop <= ry1) & (rx0 <= x_start) & (x_stop <= rx1) + ) + bad = np.flatnonzero(valid & ~covered) + if bad.size: + b = int(bad[0]) + raise RuntimeError( + f"ROI window [{int(y_start[b])}:{int(y_stop[b])}, " + f"{int(x_start[b])}:{int(x_stop[b])}] is not covered by its tile's " + f"read region [{ry0}:{ry1}, {rx0}:{rx1}]. The tile overlap must be " + "at least the ROI size." + ) + + # Window bounds relative to the read region. + y0r = (y_start - ry0).astype(np.int64) + x0r = (x_start - rx0).astype(np.int64) + y1r = (y_stop - ry0).astype(np.int64) + x1r = (x_stop - rx0).astype(np.int64) + # Full-size interior windows share one shape and are batched together; + # partial (edge-clipped) windows keep the scalar path. + full = ( + valid + & (y_stop - y_start == self._full_h) + & (x_stop - x_start == self._full_w) + ) + + def _scalar(positions: NDArray[np.integer]) -> None: + for pos in positions: + win = array[y0r[pos] : y1r[pos], x0r[pos] : x1r[pos]] + if on_gpu: + win = cp.asnumpy(win) + self.db[owned_idx[pos]] = snr_metrics.roi_snr_db(win) + + def _batch_reduce(st: Any) -> NDArray[np.float64]: + # GPU: batched Otsu on-device (CuPy), no map transfer. CPU: numba prange + # kernel (~50x the Python loop). The pure-numpy batch is memory-bound and + # no faster than the loop, so it is only the fallback when numba is absent. + # All three match the scalar roi_snr_db to float64 rounding. + if on_gpu: + return cp.asnumpy(snr_metrics.roi_snr_db_batch(st, xp=cp)) + if snr_metrics._HAS_NUMBA: + return snr_metrics.roi_snr_db_numba(st) + return snr_metrics.roi_snr_db_batch(st, xp=np) + + def _reduce() -> None: + full_pos = np.flatnonzero(full) + for start in range(0, full_pos.size, _OTSU_ROI_CHUNK): + sel = full_pos[start : start + _OTSU_ROI_CHUNK] + # xp.stack, not xp.asarray(list): the windows are on-device CuPy + # slices, and cp.asarray of a Python list of CuPy arrays is + # version-fragile, whereas stack of same-shape arrays is not. + stack = xp.stack([array[y0r[p] : y1r[p], x0r[p] : x1r[p]] for p in sel]) + # The batched/numba kernels assume finite windows; route any + # non-finite one to the scalar path (which filters them itself). + finite = xp.isfinite(stack.reshape(stack.shape[0], -1)).all(axis=1) + finite_host = cp.asnumpy(finite) if on_gpu else finite + if bool(finite_host.all()): + self.db[owned_idx[sel]] = _batch_reduce(stack) + else: + self.db[owned_idx[sel[finite_host]]] = _batch_reduce(stack[finite]) + _scalar(sel[~finite_host]) + _scalar(np.flatnonzero(valid & ~full)) + + if on_gpu: + # Run every CuPy op under the array's own device -- the fold happens + # after the GPU is returned to the pool, so the thread's current + # device may be another one. + with array.device: + _reduce() + else: + _reduce() + + def finalize(self) -> NDArray[np.float64]: + """Return the per-ROI dB array, checking each ROI was claimed once.""" + unclaimed = int((self._claimed == 0).sum()) + duplicated = int((self._claimed > 1).sum()) + if unclaimed or duplicated: + raise RuntimeError( + f"ROI coverage is wrong: {unclaimed} claimed by no tile, " + f"{duplicated} by more than one." + ) + return self.db + + def spawn(self) -> "RoiOtsuSnrAccumulator": + """A fresh, empty accumulator sharing this one's ROI geometry. + + Used by the tile pass to give each worker thread its own accumulator so + folds run without a lock; the partials are then reduced with ``merge``. + """ + return RoiOtsuSnrAccumulator( + self.y1, + self.y2, + self.x1, + self.x2, + (self.height, self.width), + map_key=self.map_key, + ) + + def merge(self, other: "RoiOtsuSnrAccumulator") -> None: + """Fold a per-worker partial into this one. + + An ROI is owned by the single tile whose write region holds its top-left + corner, so partials hold DISJOINT per-ROI dB entries: the merge overlays the + ROIs *other* computed, with no arithmetic and so bit-identical to the serial + fold. The claim counts add so ``finalize`` still verifies global coverage. + """ + overlap = (self._claimed > 0) & (other._claimed > 0) + assert not overlap.any(), ( + "RoiOtsuSnrAccumulator partials claim overlapping ROIs; each ROI must be " + "owned by exactly one tile (its top-left corner lands in one write region)." + ) + claimed = other._claimed > 0 + self._claimed += other._claimed + self.db[claimed] = other.db[claimed] + + +# --------------------------------------------------------------------------- +# Full-resolution plane storage (RAM or disk-backed) +# --------------------------------------------------------------------------- + +# Planes at or above this size are backed by a file rather than anonymous RAM. +_PLANE_SPILL_BYTES = 2 * 1024**3 + + +class _PlaneStore: + """Owns the full-resolution output planes for one channel. + + A plane that reaches ``_PLANE_SPILL_BYTES`` is backed by an ``np.memmap`` + file in *plane_dir* (the task working directory by default) instead of + anonymous RAM. Nothing declares these files as process outputs, so Nextflow + discards them with the work directory; use the ``scratch`` directive to put + that directory on node-local NVMe. Two + things follow from that, and together they are why this class exists: + + * **Bounded resident set.** Every consumer of these planes reads row blocks + or ROI windows, never the whole array (see ``_labeled_sums_chunked`` and + the SNR per-tile loop). File backing turns 22 GB of unreclaimable + anonymous pages per plane into page cache the kernel can evict under + pressure. On a 5.5 gigapixel sample the seven planes come to ~154 GB + resident, which is what forced the 180 -> 720 GB retry ladder. + * **Real multi-GPU parallelism.** A memmap is shared between processes by + *filename*, so tile workers can be separate processes — one per GPU, each + with its own interpreter, its own GIL and its own CUDA context — all + writing into the same planes with no pixel data ever pickled. Thread + workers cannot scale much past a single GPU's throughput no matter how + many devices are present, because the host-side numpy in every tile (the + input dtype cast, ``_sanitize``, the post-D2H cast) holds the GIL. NVLink + is irrelevant to this workload: tiles are independent and nothing is + exchanged between devices. + + Planes are always handed out as ordinary ndarrays, so consumers never need + to know which backing is in use. + """ + + def __init__( + self, + shape: tuple[int, int], + keys: list[str], + plane_dir: Path | None = None, + prefix: str = "plane", + dtype: Any = np.float32, + ) -> None: + self.shape = (int(shape[0]), int(shape[1])) + self.dtype = np.dtype(dtype) + self.keys = list(keys) + plane_bytes = self.shape[0] * self.shape[1] * self.dtype.itemsize + # Small planes are not worth a file; large ones always get one. + self.on_disk = plane_bytes >= _PLANE_SPILL_BYTES + self._dir = Path(plane_dir or Path.cwd()) if self.on_disk else None + self.paths: dict[str, str] = {} + self._arrays: dict[str, NDArray[Any]] = {} + + if self._dir is not None: + self._dir.mkdir(parents=True, exist_ok=True) + required = plane_bytes * len(self.keys) + free = shutil.disk_usage(self._dir).free + if free < required: + raise RuntimeError( + f"scratch dir {self._dir} has {free / 1024**3:.1f} GB free, but " + f"{len(self.keys)} plane(s) of {plane_bytes / 1024**3:.1f} GB " + f"need {required / 1024**3:.1f} GB. Point --scratch-dir at a " + "larger filesystem, or drop the flag to keep planes in RAM." + ) + logging.info( + f" Plane store: {len(self.keys)} x " + f"{plane_bytes / 1024**3:.1f} GB on disk at {self._dir}" + ) + + for key in self.keys: + if self._dir is not None: + path = self._dir / f"{prefix}_{key}.dat" + self._arrays[key] = np.memmap( + path, dtype=self.dtype, mode="w+", shape=self.shape + ) + self.paths[key] = str(path) + else: + self._arrays[key] = np.empty(self.shape, dtype=self.dtype) + + def arrays(self) -> dict[str, Any]: + """Plane arrays keyed by name, for in-process use.""" + return dict(self._arrays) + + def descriptors(self) -> dict[str, str] | None: + """``{key: path}`` for reopening in another process, None when in RAM.""" + return dict(self.paths) if self.on_disk else None + + def flush(self) -> None: + """Push dirty memmap pages to the backing files.""" + for arr in self._arrays.values(): + if isinstance(arr, np.memmap): + arr.flush() + + def release(self) -> None: + """Drop references and delete the backing files.""" + self.flush() + self._arrays.clear() + for path in self.paths.values(): + Path(path).unlink(missing_ok=True) + self.paths.clear() + + +# --------------------------------------------------------------------------- +# Process-per-GPU tile workers +# --------------------------------------------------------------------------- + +# Per-process state, populated by _tile_worker_init in each pool worker. +_WORKER: dict[str, Any] = {} + + +def _tile_worker_init( + slot_counter, + gpu_ids: tuple[int, ...], + channel_source: tuple[str, int], + plane_paths: dict[str, str], + shape: tuple[int, int], + dtype_str: str, +) -> None: + """Pool initializer: claim one GPU, open our own TIFF handle and plane views. + + Each worker claims a distinct device by taking the next slot from a shared + counter, so with ``processes == len(gpu_ids)`` every GPU gets exactly one + process. Tiles are then pulled from the pool's task queue, which + load-balances naturally — unlike static ``gpu_ids[i % n_gpus]`` round-robin, + which leaves fast devices idle whenever tile costs differ (the DAPI channel + also computes the Laplacian, so its tiles cost roughly 2.5x the others'). + """ + # A spawned child never runs main(), so it never runs logging.basicConfig and + # the root logger would sit at WARNING -- silencing every worker-side message + # exactly where multi-GPU problems would show up. + logging.basicConfig( + level=logging.INFO, + format="%(asctime)s [%(levelname)s] %(message)s", + datefmt="%Y-%m-%d %H:%M:%S", + force=True, + ) + + with slot_counter.get_lock(): + slot = slot_counter.value + slot_counter.value += 1 + gpu_id = gpu_ids[slot % len(gpu_ids)] + + path, page_index = channel_source + tif = tifffile.TiffFile(path) + channel = _LazyTiffChannel(tif.pages[page_index], source=channel_source) + channel._tiff_handles = [tif] # type: ignore[attr-defined] + + dtype = np.dtype(dtype_str) + _WORKER.clear() + _WORKER.update( + gpu_id=gpu_id, + channel=channel, + planes={ + key: np.memmap(p, dtype=dtype, mode="r+", shape=tuple(shape)) + for key, p in plane_paths.items() + }, + ) + logging.info(f" Tile worker pid={os.getpid()} bound to GPU {gpu_id}") + + +def _tile_worker_run( + task: tuple[dict[str, int], int, bool, float], +) -> tuple[int, float]: + """Pool task: compute one tile on this process's GPU. Returns (gpu_id, seconds).""" + tile_spec, window_size, include_laplacian, lap_sigma = task + t0 = time.perf_counter() + _process_tile_on_gpu( + _WORKER["channel"], + tile_spec, + window_size, + _WORKER["gpu_id"], + _WORKER["planes"], + include_laplacian, + lap_sigma, + ) + for plane in _WORKER["planes"].values(): + if isinstance(plane, np.memmap): + plane.flush() + return int(_WORKER["gpu_id"]), time.perf_counter() - t0 + + +def _log_gpu_balance(timings: list[tuple[int, float]], label: str) -> None: + """Log per-GPU occupancy so multi-GPU scaling is measurable, not assumed. + + An aggregate wall-clock number cannot reveal load imbalance: it reports + max(per-device time), so an idle device looks identical to a saturated one. + """ + if not timings: + return + per_gpu: dict[int, list[float]] = {} + for gpu_id, seconds in timings: + per_gpu.setdefault(gpu_id, []).append(seconds) + busiest = max(sum(v) for v in per_gpu.values()) or 1.0 + for gpu_id in sorted(per_gpu): + secs = per_gpu[gpu_id] + total = sum(secs) + logging.info( + f" [TIMING] {label} GPU {gpu_id}: {len(secs)} tiles, " + f"{total:.1f}s busy ({100.0 * total / busiest:.0f}% of busiest)" + ) + + +# --------------------------------------------------------------------------- +# Focus-map computation +# --------------------------------------------------------------------------- + + +def _process_tile_on_gpu( + channel_data, + tile_spec: dict[str, int], + window_size: int, + gpu_id: int, + out: dict[str, NDArray[np.float32] | None], + include_laplacian: bool = False, + lap_sigma: float = 1.0, +) -> None: + """Read a tile, compute focus maps on GPU, trim overlap, write into *out*. + + Writes its trimmed results straight into the caller's preallocated + full-resolution planes rather than returning them. This is what bounds + host RAM: a returned dict would be pinned by ``Future._result`` until the + assembly loop finished, and ``as_completed()`` holds every ``Future`` for + the duration of iteration — so returning arrays retains *every* tile's + output (roughly one extra full copy of each plane, plus halo). Writing in + place means the ``Future`` carries only ``None``. + + Each tile's write region is disjoint from every other tile's (guaranteed by + ``_compute_tile_grid``; asserted by ``test_no_write_overlap``), so + concurrent in-place writes from the worker threads are safe. + + Args: + channel_data: Array-like supporting slicing (_LazyTiffChannel or numpy array). + tile_spec: Dict from _compute_tile_grid() with read/write/trim keys. + window_size: Convolution window size. + gpu_id: CUDA device ID. + out: Dict of preallocated full-res planes keyed ``focus_map`` / + ``mean_map`` / ``lap_var_map``. A ``None`` value skips that plane. + include_laplacian: Also compute Laplacian variance map. + lap_sigma: Gaussian sigma for LoG. + """ + # Read tile (triggers actual I/O for lazy TIFF-backed arrays) + tile = np.asarray( + channel_data[ + tile_spec["read_y0"] : tile_spec["read_y1"], + tile_spec["read_x0"] : tile_spec["read_x1"], + ] + ) + + # Compute on GPU using existing function + result = _compute_channel_maps_on_gpu( + tile, window_size, gpu_id, include_laplacian, lap_sigma + ) + del tile + + # Trim overlap and write each output straight into the caller's plane + tt = tile_spec["trim_top"] + tb = tile_spec["trim_bottom"] + tl = tile_spec["trim_left"] + tr = tile_spec["trim_right"] + wy0, wy1 = tile_spec["write_y0"], tile_spec["write_y1"] + wx0, wx1 = tile_spec["write_x0"], tile_spec["write_x1"] + + for key in list(result): + plane = out.get(key) + arr = result.pop(key) + if plane is None: + continue + h, w = arr.shape + y1 = h - tb if tb > 0 else h + x1 = w - tr if tr > 0 else w + plane[wy0:wy1, wx0:wx1] = arr[tt:y1, tl:x1] + + +def _process_tile_for_consumers( + channel_data, + tile_spec: dict[str, int], + window_size: int, + gpu_id: int, + include_laplacian: bool = False, + lap_sigma: float = 1.0, + keep_mean_device: bool = False, + keep_focus_device: bool = False, + drop_mean_host: bool = False, +) -> tuple[dict[str, NDArray[np.float32]], dict[str, NDArray[np.float32]]]: + """Compute one tile and return ``(trimmed, untrimmed)`` maps for consumers. + + Returns rather than writes: with no plane to write into, the parent folds each + tile into the consumers and drops it. Both forms are needed — + ``RoiOtsuSnrAccumulator`` requires the haloed (untrimmed) tile because an ROI + it owns can extend up to ``roi_size - 1`` px past the write region, while the + other consumers want the trimmed tile that maps exactly onto the write region. + + Device maps (``focus_map_device`` / ``mean_map_device``) are trimmed too — the + write-region view for the trimmed consumers (``CentrePixelSampler``, + ``LabeledSumAccumulator``) while the untrimmed dict keeps the full device array + for ``RoiOtsuSnrAccumulator``. Slicing a CuPy array is a view, so this adds no + device allocation. + + The untrimmed arrays are *views* into the same buffers, so returning both adds + no allocation. + """ + tile = np.asarray( + channel_data[ + tile_spec["read_y0"] : tile_spec["read_y1"], + tile_spec["read_x0"] : tile_spec["read_x1"], + ] + ) + untrimmed = _compute_channel_maps_on_gpu( + tile, + window_size, + gpu_id, + include_laplacian, + lap_sigma, + keep_mean_device=keep_mean_device, + keep_focus_device=keep_focus_device, + drop_mean_host=drop_mean_host, + ) + del tile + + top = tile_spec["trim_top"] + bottom = tile_spec["trim_bottom"] + left = tile_spec["trim_left"] + right = tile_spec["trim_right"] + trimmed = {} + for key, arr in untrimmed.items(): + # Trim every map, host and device alike, to the write region. The trimmed + # device slices are what CentrePixelSampler / LabeledSumAccumulator fold; + # the untrimmed dict keeps the full device arrays for RoiOtsuSnrAccumulator, + # whose owned ROI windows reach past the write region into the halo. + height, width = arr.shape + y_stop = height - bottom if bottom > 0 else height + x_stop = width - right if right > 0 else width + trimmed[key] = arr[top:y_stop, left:x_stop] + return trimmed, untrimmed + + +def _compute_channel_maps_tiled( + channel_data, + image_shape: tuple[int, int], + window_size: int, + gpu_ids: list[int], + include_laplacian: bool = False, + lap_sigma: float = 1.0, + gpu_mem_bytes: int | None = None, + plane_dir: Path | None = None, + plane_prefix: str = "plane", + consumers: list[Any] | None = None, +) -> dict[str, Any] | None: + """Compute focus maps for one channel using tiled multi-GPU processing. + + Tiles the image with convolution-safe overlap, distributes tiles across the + available GPUs, and assembles the results into full-resolution planes. Tile + size is chosen adaptively to fit GPU memory. + + With more than one GPU and disk-backed planes, tiles are dispatched to one + worker **process** per GPU (see ``_tile_worker_init``); otherwise a thread + pool is used, which keeps single-GPU behaviour unchanged. + + Args: + channel_data: Array-like (_LazyTiffChannel or numpy) with shape matching image_shape. + image_shape: (height, width). + window_size: Convolution window size. + gpu_ids: List of CUDA device IDs. + include_laplacian: Also compute Laplacian variance. + lap_sigma: Gaussian sigma for LoG. + gpu_mem_bytes: Total GPU VRAM in bytes. Auto-detected if None. + plane_dir: Directory for the disk-backed output planes. Defaults to + the task working directory, which is what Nextflow's ``scratch`` + directive relocates onto node-local storage. + plane_prefix: Filename prefix / log label for this channel's planes. + + Returns: + Dict with 'focus_map', 'mean_map', and optionally 'lap_var_map' + (full-resolution float32 arrays; memmaps when disk-backed). + """ + H, W = image_shape + n_gpus = len(gpu_ids) + # Overlap must cover both uniform_filter radius (window_size // 2) and the + # Gaussian pre-smoothing in the Laplacian path (~3 * lap_sigma). Using the + # full window_size is safe for all kernels and adds negligible I/O overhead. + overlap = window_size + # Streaming mode needs MORE than the convolution radius. RoiOtsuSnrAccumulator + # reads ROI windows out of the *untrimmed* tile, and it claims an ROI by its top + # edge y1, so the window reaches `roi_size - 1` px past write_y1 -- while + # uniform_filter's reflect padding corrupts the last `window_size // 2` rows of the + # read region. With overlap == window_size == roi_size == 35 the padded band starts + # at write_y1 + 18 and the ROI reaches write_y1 + 33, so up to 16 of its 35 rows + # came from padding. + # + # Measured before this fix: 2,855 of 335,241 ROIs (0.85 %) had a different + # snr_image_otsu_db from the plane-based path, by up to 19.9 % relative, and that + # single column was the ONLY difference across 66 output files + # (runs 5ZL9TWmJ701ham vs 1AjX0aBBfdbAFQ). + # + # roi_size is the ROI side, which equals window_size on the production path but is + # kept separate here because the requirement is genuinely about the ROI, not the + # kernel. + if consumers: + roi_reach = window_size # ROI side length; grid stride defaults to roi_size + needed = roi_reach - 1 + window_size // 2 + if needed > overlap: + overlap = needed + # VRAM cost of the larger halo, checked rather than assumed: the strip height from + # _compute_adaptive_strip_height does NOT include the overlap, so the read region + # always exceeds that budget by 2*overlap rows. At the production geometry (20409-row + # strips, 53908 px wide, ~24 B/px of concurrent buffers) going 35 -> 51 adds 32 rows, + # i.e. 0.039 GB, taking a tile from 24.68 to 24.71 GB of the 44.5 GB device. The + # budget was already optimistic by 70 rows; it is now optimistic by 102. + + # Fail loud if lap_sigma is raised past what the halo can cover: the LoG + # needs window_size//2 (uniform_filter) + ~4*lap_sigma+1 (Gaussian+Laplace) + # of halo; beyond `overlap` the DAPI lap_var map would differ from the + # whole-image result at tile seams (silently). Default lap_sigma=1.0 is safe. + if include_laplacian: + lap_halo = window_size // 2 + int(math.ceil(4 * lap_sigma)) + 1 + if lap_halo > overlap: + logging.warning( + f" lap_sigma={lap_sigma} needs {lap_halo}px halo > overlap " + f"{overlap}px; Laplacian tile seams may differ from whole-image. " + f"Raise the tile overlap or lower lap_sigma." + ) + + # Full-width row strips rather than square tiles: a strip's write region is + # one contiguous byte range, and it needs halo on top/bottom only. See + # _compute_adaptive_strip_height for the measurements that motivated this. + strip_height = _compute_adaptive_strip_height(H, W, n_gpus, gpu_mem_bytes) + # Derived from the consumers, not supplied by the caller: BlockMeanAccumulator needs + # write origins on multiples of its block size, and a caller who had to remember to + # say so would discover the omission only as a RuntimeError mid-fold. + align_writes_to = 1 + for consumer in consumers or (): + align_writes_to = math.lcm( + align_writes_to, int(getattr(consumer, "write_alignment", 1) or 1) + ) + if align_writes_to > 1: + # Align write boundaries to the heatmap block size so every block lies wholly + # inside one tile; BlockMeanAccumulator then reduces it with one reshape and the + # canvas is bit-exact. See that class for the 16-pixel figure difference. + # max(), not a conditional skip: a strip shorter than one block would otherwise + # be left unaligned and trip the accumulator's contract check at runtime. Rounding + # up to one block costs at most align_writes_to rows of halo. + strip_height = max( + align_writes_to, (strip_height // align_writes_to) * align_writes_to + ) + tiles = _compute_tile_grid(H, W, strip_height, overlap, tile_width=W) + + logging.info( + f" Tiled processing: {len(tiles)} row strip(s) " + f"({strip_height}x{W}px), {n_gpus} GPU(s), overlap={overlap}px" + ) + + # Ensure CUDA_PATH is set for CuPy NVRTC kernel compilation in worker + # threads. CuPy auto-detects CUDA in the main thread but worker threads + # can fail with "Failed to auto-detect CUDA root directory". The pip + # package nvidia-cuda-runtime-cu12 installs headers under + # site-packages/nvidia/cuda_runtime/include/. + if "CUDA_PATH" not in os.environ: + try: + import nvidia.cuda_runtime as _cr + + _cr_dir = _cr.__path__[0] # .../site-packages/nvidia/cuda_runtime + if os.path.isdir(os.path.join(_cr_dir, "include")): + os.environ["CUDA_PATH"] = _cr_dir + logging.info(f" Set CUDA_PATH={_cr_dir}") + except (ImportError, IndexError, AttributeError): + pass # CUDA_PATH remains unset; CuPy will try its own detection + + # Warm up CuPy kernel cache in the main thread. NVRTC compiles kernels on + # first use; running a tiny operation here populates the disk cache so that + # worker threads hit the cache instead of compiling in parallel. + try: + _warmup = cp.array([1.0], dtype=cp.float32) + cupyx.scipy.ndimage.uniform_filter(_warmup.reshape(1, 1), size=1) + del _warmup + except Exception: + pass + + # ------------------------------------------------------------------ + # Consumer mode: no planes at all. Each tile is folded into the consumers and + # dropped, so peak host memory is O(max_inflight * tile) + O(n_rois, n_cells) -- + # NOT O(tile). Results sit in their futures until the parent pops and folds them, + # and folding is serialised here, so up to max_inflight tiles can be complete and + # resident at once. Measured on a 102045x53908 sample: 12 strips of 8504 rows is + # 1.71 GB per map, so DAPI's 3 maps x 8 in flight is ~41 GB worst case against + # ~154 GB of planes for the whole image. + # ------------------------------------------------------------------ + if consumers: + # Each worker THREAD folds its tiles into its OWN set of consumer + # accumulators, then the per-thread partials are reduced into the caller's + # consumers after the executor drains. There is no fold lock, so the n_gpus + # workers fold CONCURRENTLY -- the fold is the wall clock here (dominated by + # RoiOtsuSnrAccumulator's per-ROI Otsu, LabeledSumAccumulator's bincounts), + # so a single fold_lock serialised the folds and threw away the multi-GPU + # parallelism (the DAPI fold measured ~240 s of the run behind that lock). + # + # Correctness of the reduce, per consumer: + # * RoiOtsuSnrAccumulator / CentrePixelSampler: every ROI is owned by + # exactly one tile (the halo guarantees it), so partials hold DISJOINT + # per-ROI entries -- merge is an overlay, bit-identical to the serial + # fold regardless of order. + # * BlockMeanAccumulator: write origins are aligned to the block size, so + # every block lies wholly in one tile -> one partial; canvases are + # disjoint and add exactly. + # * LabeledSumAccumulator: per-label float64 count/sum arrays are added + # element-wise. This regroups the additions relative to the serial + # tile-order fold; that it still reproduces them byte-for-byte is + # empirical, pinned by tests/test_fold_parallel_equivalence. + # + # Folding in the worker (not the dispatch loop) is also what bounds memory: a + # tile is released as soon as it is folded, instead of staying alive in its + # Future until the parent catches up. + gpu_pool: queue.Queue[int] = queue.Queue() + for _gpu in gpu_ids: + gpu_pool.put(_gpu) + + # Per-worker-thread consumer sets, created lazily on the thread's first tile + # and registered under a lock with a stable creation index so the reduce + # order is deterministic (run-to-run reproducible). threading.local() keys + # the set to the OS thread the ThreadPoolExecutor reuses, so a thread folds + # all its tiles into the one set it created. + _thread_state = threading.local() + partials: list[tuple[int, list[Any]]] = [] + partials_lock = threading.Lock() + per_consumer_lock = threading.Lock() + + def _thread_consumers() -> list[Any]: + local = getattr(_thread_state, "consumers", None) + if local is not None: + return local + local = [c.spawn() for c in consumers] + with partials_lock: + index = len(partials) + partials.append((index, local)) + _thread_state.consumers = local + return local + + # Keep the maps resident on the GPU when a consumer folds a reduction there + # (RoiOtsuSnrAccumulator, LabeledSumAccumulator, CentrePixelSampler do their + # per-ROI / per-label / centre-pixel reductions on-device). The host mean map + # is dropped when every consumer that reads the mean map takes the device one + # -- then nothing reads the host copy and the D2H transfer is eliminated. The + # host focus map is always produced: BlockMeanAccumulator (Figure 5) reduces + # it in float32 on the host, which a GPU reduction cannot match to float + # rounding, so it never sets wants_device_focus. + keep_mean_device = any( + getattr(c, "wants_device_mean", False) for c in consumers + ) + keep_focus_device = any( + getattr(c, "wants_device_focus", False) for c in consumers + ) + drop_mean_host = keep_mean_device and not any( + getattr(c, "reads_host_mean", False) for c in consumers + ) + + timings: list[tuple[int, float]] = [] + fold_seconds = 0.0 + per_consumer: dict[str, float] = {} + + def _run_tile(spec) -> tuple[int, float, float]: + gpu_id = gpu_pool.get() + try: + t0 = time.perf_counter() + trimmed, untrimmed = _process_tile_for_consumers( + channel_data, + spec, + window_size, + gpu_id, + include_laplacian, + lap_sigma, + keep_mean_device=keep_mean_device, + keep_focus_device=keep_focus_device, + drop_mean_host=drop_mean_host, + ) + compute_seconds = time.perf_counter() - t0 + + # Fold into THIS thread's own consumer set -- no lock, so the n_gpus + # workers fold concurrently. + local_consumers = _thread_consumers() + t1 = time.perf_counter() + local_per: dict[str, float] = {} + for consumer in local_consumers: + t_c = time.perf_counter() + if getattr(consumer, "wants_untrimmed", False): + consumer.consume(spec, untrimmed) + else: + consumer.consume(spec, trimmed) + name = type(consumer).__name__ + local_per[name] = local_per.get(name, 0.0) + ( + time.perf_counter() - t_c + ) + fold = time.perf_counter() - t1 + # Aggregate the per-consumer fold time across threads (these overlap + # in wall clock now, so this is summed CPU, not wall time). + with per_consumer_lock: + for name, seconds in local_per.items(): + per_consumer[name] = per_consumer.get(name, 0.0) + seconds + del trimmed, untrimmed + return gpu_id, compute_seconds, fold + finally: + # Release the GPU slot AFTER the fold, not before it. With + # keep_mean_device/keep_focus_device this tile's mean/focus maps are + # device arrays that the fold reduces on THIS card. Releasing the slot + # before the fold (the previous behaviour) let another worker grab this + # card and start a second tile's compute while these maps were still + # resident -- an unbounded cross-worker pile-up (up to one compute + + # n_gpus-1 held maps on one card) that breaks the per-device VRAM bound + # #55 established, since _compute_adaptive_strip_height sizes for one + # tile's compute arrays only. Holding the slot through the fold bounds + # each card to a single tile at a time (its compute arrays, then its + # held maps), so the strip-height budget is the real per-card ceiling. + # The trade is small: the fold runs on this card regardless (on-device + # reduction) and is short since the GPU Otsu change, so the only lost + # overlap is a peer tile's compute starting on this card mid-fold. + gpu_pool.put(gpu_id) + + t_dispatch = time.perf_counter() + logging.info( + f" Consumer mode: {len(tiles)} tiles, {n_gpus} GPU(s), " + f"per-worker parallel fold, no planes materialised" + ) + with ThreadPoolExecutor(max_workers=n_gpus) as executor: + for gpu_id, compute_seconds, fold in executor.map(_run_tile, tiles): + timings.append((gpu_id, compute_seconds)) + fold_seconds += fold + + # Reduce the per-thread partials into the caller's consumers, in a + # deterministic order (stable creation index). The caller's consumers never + # consumed a tile, so they start empty and this fills them for the + # downstream finalize(). + for _index, local_consumers in sorted(partials, key=lambda item: item[0]): + for target, part in zip(consumers, local_consumers): + target.merge(part) + + elapsed = time.perf_counter() - t_dispatch + logging.info( + f" [TIMING] {plane_prefix} tiled compute (consumers, {len(tiles)} " + f"tiles, {n_gpus} GPU): {elapsed:.1f}s" + ) + logging.info( + f" [TIMING] {plane_prefix} consumer folding: {fold_seconds:.1f}s " + f"({100.0 * fold_seconds / max(elapsed, 1e-9):.0f}% of wall clock)" + ) + for name, seconds in sorted(per_consumer.items(), key=lambda kv: -kv[1]): + logging.info( + f" [TIMING] {plane_prefix} fold {name}: {seconds:.1f}s " + f"({100.0 * seconds / max(fold_seconds, 1e-9):.0f}% of folding)" + ) + # Read/decode/convolve split. read+decode are serialized under the reader's + # lock, so compare their sum to the wall clock (elapsed): sum ~ elapsed means + # the reader is the wall (IO/decode-bound, GPUs starve); sum << elapsed means + # the GPU convolution dominates. total_compute is the summed per-tile + # read+upload+convolve across all workers, so convolve+upload ~ total_compute + # - read - decode. + read_s = float(getattr(channel_data, "_read_seconds", 0.0)) + decode_s = float(getattr(channel_data, "_decode_seconds", 0.0)) + read_gb = float(getattr(channel_data, "_read_bytes", 0)) / 1024**3 + total_compute = sum(secs for _gid, secs in timings) + logging.info( + f" [TIMING] {plane_prefix} read (S3/fh): {read_s:.1f}s " + f"({read_gb:.2f} GB, {read_gb / max(read_s, 1e-9):.2f} GB/s), " + f"decode: {decode_s:.1f}s, read+decode = " + f"{100.0 * (read_s + decode_s) / max(elapsed, 1e-9):.0f}% of wall clock" + ) + logging.info( + f" [TIMING] {plane_prefix} convolve+upload ~= " + f"{max(total_compute - read_s - decode_s, 0.0):.1f}s " + f"(sum tile-compute {total_compute:.1f}s - read {read_s:.1f}s - " + f"decode {decode_s:.1f}s)" + ) + _log_gpu_balance(timings, f"{plane_prefix} streamed") + return None + + # Allocate the output planes. With a scratch dir configured these are + # disk-backed memmaps rather than anonymous RAM, which both bounds the + # resident set and lets the tile workers be separate processes — see + # _PlaneStore. Workers write their trimmed tiles straight into these, so no + # per-tile result is ever retained (`as_completed()` holds every Future for + # the duration of iteration, so a worker that *returned* its arrays would + # keep all of them alive until the executor block exited). + keys = ["focus_map", "mean_map"] + (["lap_var_map"] if include_laplacian else []) + store = _PlaneStore( + (H, W), keys, plane_dir=plane_dir, prefix=plane_prefix, dtype=np.float32 + ) + out_planes: dict[str, Any] = store.arrays() + out_planes.setdefault("lap_var_map", None) + + channel_source = getattr(channel_data, "_source", None) + plane_paths = store.descriptors() + # Process mode needs: more than one GPU to be worth it, a reopenable channel + # (a live TiffPage holds an OS handle and a lock, so it cannot be pickled), + # and file-backed planes to write into. Single-GPU runs keep the thread + # path, so their behaviour is unchanged. + use_processes = ( + n_gpus > 1 and channel_source is not None and plane_paths is not None + ) + tasks = [(spec, window_size, include_laplacian, lap_sigma) for spec in tiles] + t_dispatch = time.perf_counter() + + if use_processes: + assert channel_source is not None and plane_paths is not None + # One process per GPU. Separate interpreters mean the host-side numpy + # in each tile (input cast, _sanitize, post-D2H cast) no longer + # serialises on a shared GIL, and each process builds its own CUDA + # context. Must be "spawn": CUDA does not survive fork(). + logging.info( + f" Dispatching {len(tasks)} tiles to {n_gpus} worker process(es), " + f"one per GPU {gpu_ids} (spawn)" + ) + ctx = multiprocessing.get_context("spawn") + slot_counter = ctx.Value("i", 0) + with ctx.Pool( + processes=n_gpus, + initializer=_tile_worker_init, + initargs=( + slot_counter, + tuple(gpu_ids), + channel_source, + plane_paths, + (H, W), + np.dtype(np.float32).str, + ), + ) as pool: + timings = list(pool.imap_unordered(_tile_worker_run, tasks)) + else: + # Pipelined I/O + GPU: each worker reads its tile from TIFF then + # computes on its assigned GPU. The TIFF file handle lock serializes + # reads, but I/O for tile N+1 overlaps with GPU compute for tile N. + def _run_tile(index: int, spec: dict[str, int]) -> tuple[int, float]: + gpu_id = gpu_ids[index % n_gpus] + t0 = time.perf_counter() + _process_tile_on_gpu( + channel_data, + spec, + window_size, + gpu_id, + out_planes, + include_laplacian, + lap_sigma, + ) + return gpu_id, time.perf_counter() - t0 + + with ThreadPoolExecutor(max_workers=n_gpus) as executor: + futures = [ + executor.submit(_run_tile, i, spec) for i, spec in enumerate(tiles) + ] + timings = [f.result() for f in as_completed(futures)] + + store.flush() + mode = "process" if use_processes else "thread" + logging.info( + f" [TIMING] {plane_prefix} tiled compute ({mode}, {len(tiles)} tiles, " + f"{n_gpus} GPU): {time.perf_counter() - t_dispatch:.1f}s" + ) + _log_gpu_balance(timings, f"{plane_prefix} tiled") + + # Per-tile results are already sanitized in _compute_channel_maps_on_gpu, + # so no redundant final full-array _sanitize here. + return { + "focus_map": out_planes["focus_map"], + "mean_map": out_planes["mean_map"], + "lap_var_map": out_planes.get("lap_var_map"), + } + + +def compute_all_focus_maps( + channels: NDArray[np.generic], + window_size: int = 35, + use_gpu: bool = False, + gpu_ids: list[int] | None = None, + lap_sigma: float = 1.0, +) -> dict[str, NDArray[np.float32] | None]: + """Compute focus maps for all available image channels. + + When ``gpu_ids`` contains multiple IDs, channel computations are + distributed across GPUs in parallel using a thread pool. Each channel's + convolution work runs entirely on one GPU; with 3 channels and 3+ GPUs + all channels are processed concurrently. + + Args: + channels: Either a 2-D array (single DAPI channel, shape ``(H, W)``) + or a 3-D array with channels first (shape ``(C, H, W)``). + window_size: Side length of the square averaging window. + use_gpu: Use CuPy GPU backend when ``True``. Ignored when *gpu_ids* + is provided (GPU is assumed). + gpu_ids: List of CUDA device IDs for multi-GPU parallelism. If + ``None`` and ``use_gpu=True``, uses device 0 only. If ``None`` + and ``use_gpu=False``, runs on CPU. + + Returns: + Dictionary with the following keys (values are ``None`` when the + corresponding channel is not present in *channels*): + + - ``dapi_focus_map`` — CCFS focus map for DAPI (channel 0). + - ``dapi_mean_map`` — Local mean map for DAPI. + - ``dapi_lap_var_map`` — Laplacian variance map for DAPI. + - ``boundary_focus_map`` — CCFS focus map for Boundary (channel 1). + - ``boundary_mean_map`` — Local mean map for Boundary. + - ``intrna_focus_map`` — CCFS focus map for IntRNA (channel 2). + - ``intrna_mean_map`` — Local mean map for IntRNA. + + Raises: + ValueError: If *channels* has fewer than 2 or more than 3 dimensions. + """ + # Resolve GPU configuration + if gpu_ids is not None and len(gpu_ids) > 0: + use_gpu = True + elif use_gpu and gpu_ids is None: + gpu_ids = [0] + + # Split channels — .copy() so the parent 3-D array can be freed. + if channels.ndim == 2: + dapi = channels.copy() + boundary = None + intrna = None + elif channels.ndim == 3: + n_ch = channels.shape[0] + dapi = channels[0].copy() + boundary = channels[1].copy() if n_ch > 1 else None + intrna = channels[2].copy() if n_ch > 2 else None + else: + raise ValueError( + f"Expected 2-D or 3-D array, got {channels.ndim}-D " + f"(shape {channels.shape})." + ) + del channels + + # ---- Multi-GPU path: distribute channels across GPUs ---- + if use_gpu and gpu_ids is not None and len(gpu_ids) > 0: + # Build work items: (channel_name, channel_data, include_laplacian) + work_items: list[tuple[str, NDArray[np.generic], bool]] = [ + ("dapi", dapi, True), # DAPI always gets Laplacian + ] + if boundary is not None: + work_items.append(("boundary", boundary, False)) + if intrna is not None: + work_items.append(("intrna", intrna, False)) + + # Assign GPUs round-robin + results: dict[str, dict[str, NDArray[np.float32]]] = {} + n_gpus = len(gpu_ids) + logging.info( + f" Distributing {len(work_items)} channel(s) across {n_gpus} GPU(s): {gpu_ids}" + ) + + t_gpu = time.perf_counter() + gpu_failed = False + try: + with ThreadPoolExecutor( + max_workers=min(len(work_items), n_gpus) + ) as executor: + futures = {} + for idx, (name, ch_data, inc_lap) in enumerate(work_items): + assigned_gpu = gpu_ids[idx % n_gpus] + future = executor.submit( + _compute_channel_maps_on_gpu, + ch_data, + window_size, + assigned_gpu, + inc_lap, + lap_sigma, + ) + futures[future] = name + + for future in as_completed(futures): + ch_name = futures[future] + results[ch_name] = future.result() + logging.info( + f" [TIMING] GPU compute (multi-GPU, {len(work_items)} channels): {time.perf_counter() - t_gpu:.1f}s" + ) + except Exception as gpu_err: + logging.warning( + f" WARNING: GPU computation failed ({gpu_err}), falling back to CPU..." + ) + gpu_failed = True + + if not gpu_failed: + # Assemble output dict from GPU results + dapi_res = results["dapi"] + return { + "dapi_focus_map": dapi_res["focus_map"], + "dapi_mean_map": dapi_res["mean_map"], + "dapi_lap_var_map": dapi_res.get("lap_var_map"), + "boundary_focus_map": results["boundary"]["focus_map"] + if "boundary" in results + else None, + "boundary_mean_map": results["boundary"]["mean_map"] + if "boundary" in results + else None, + "intrna_focus_map": results["intrna"]["focus_map"] + if "intrna" in results + else None, + "intrna_mean_map": results["intrna"]["mean_map"] + if "intrna" in results + else None, + } + # else: fall through to CPU path below + + # ---- Single-GPU or CPU fallback path ---- + # Reached when: no multi-GPU available, use_gpu=False, or GPU OOM fallback + use_gpu = ( + False # Force CPU to avoid repeated OOM if we fell through from GPU failure + ) + gpu_id = gpu_ids[0] if gpu_ids else 0 + + t_gpu = time.perf_counter() + # Process channels sequentially, freeing each input before the next + # to keep peak memory at ~1 channel + its output maps. + + # DAPI — always present + dapi_focus_map, dapi_mean_map = compute_ccfs_map( + dapi, window_size=window_size, use_gpu=use_gpu, gpu_id=gpu_id + ) + dapi_lap_var_map = compute_laplacian_variance_map( + dapi, + window_size=window_size, + use_gpu=use_gpu, + gpu_id=gpu_id, + lap_sigma=lap_sigma, + ) + del dapi + + # Boundary (channel 1) + boundary_focus_map: NDArray[np.float32] | None = None + boundary_mean_map: NDArray[np.float32] | None = None + if boundary is not None: + boundary_focus_map, boundary_mean_map = compute_ccfs_map( + boundary, window_size=window_size, use_gpu=use_gpu, gpu_id=gpu_id + ) + del boundary + + # IntRNA (channel 2) + intrna_focus_map: NDArray[np.float32] | None = None + intrna_mean_map: NDArray[np.float32] | None = None + if intrna is not None: + intrna_focus_map, intrna_mean_map = compute_ccfs_map( + intrna, window_size=window_size, use_gpu=use_gpu, gpu_id=gpu_id + ) + del intrna + + backend_label = "GPU" if use_gpu else "CPU" + logging.info( + f" [TIMING] {backend_label} compute (single device, all channels): {time.perf_counter() - t_gpu:.1f}s" + ) + + return { + "dapi_focus_map": dapi_focus_map, + "dapi_mean_map": dapi_mean_map, + "dapi_lap_var_map": dapi_lap_var_map, + "boundary_focus_map": boundary_focus_map, + "boundary_mean_map": boundary_mean_map, + "intrna_focus_map": intrna_focus_map, + "intrna_mean_map": intrna_mean_map, + } + + +# --------------------------------------------------------------------------- +# Down-sampling to per-tile DataFrame +# --------------------------------------------------------------------------- + + +def _build_roi_grid( + image_shape: tuple[int, int], + roi_size: int, + stride: int, + tissue_mask: NDArray[np.generic] | None, + downsample_factor: int = 8, +) -> dict[str, Any]: + """Build the tile grid and compute tissue coverages (one-time setup). + + Returns a dict with keys: ``x1_arr``, ``x2_arr``, ``y1_arr``, ``y2_arr``, + ``cx``, ``cy``, ``n_rois``, ``height``, ``width``, ``tissue_coverages``, + ``is_boundary_roi``, ``roi_coords_arr``. + """ + height, width = image_shape + + # Build ROI grid — vectorized + x1_arr, x2_arr, y1_arr, y2_arr, cx, cy = compute_roi_grid( + height, width, roi_size, stride + ) + n_rois = len(x1_arr) + roi_coords_arr = np.stack([x1_arr, x2_arr, y1_arr, y2_arr], axis=1) + + # Tissue coverage + if tissue_mask is not None: + binary_mask = (tissue_mask > 0).astype(np.float64) + mask_h, mask_w = binary_mask.shape + x1_ds = np.clip(x1_arr // downsample_factor, 0, mask_w) + x2_ds = np.clip((x2_arr - 1) // downsample_factor + 1, 0, mask_w) + y1_ds = np.clip(y1_arr // downsample_factor, 0, mask_h) + y2_ds = np.clip((y2_arr - 1) // downsample_factor + 1, 0, mask_h) + + integral = np.zeros((mask_h + 1, mask_w + 1), dtype=np.float64) + integral[1:, 1:] = np.cumsum(np.cumsum(binary_mask, axis=0), axis=1) + block_sums = ( + integral[y2_ds, x2_ds] + - integral[y1_ds, x2_ds] + - integral[y2_ds, x1_ds] + + integral[y1_ds, x1_ds] + ) + block_w = x2_ds - x1_ds + block_h = y2_ds - y1_ds + total_pixels_arr = (block_w * block_h).astype(np.float64) + total_pixels_arr[total_pixels_arr == 0] = 1.0 + tissue_coverages = block_sums / total_pixels_arr + else: + tissue_coverages = np.ones(n_rois, dtype=np.float64) + + half_win = roi_size // 2 + is_boundary_roi = ( + (cx < half_win) + | (cy < half_win) + | (cx >= width - half_win) + | (cy >= height - half_win) + ) + + return { + "x1_arr": x1_arr, + "x2_arr": x2_arr, + "y1_arr": y1_arr, + "y2_arr": y2_arr, + "cx": cx, + "cy": cy, + "n_rois": n_rois, + "height": height, + "width": width, + "tissue_coverages": tissue_coverages, + "is_boundary_roi": is_boundary_roi, + "roi_coords_arr": roi_coords_arr, + } + + +def compute_roi_grid( + height: int, + width: int, + roi_size: int = 35, + stride: int | None = None, +) -> tuple[ + NDArray[np.int64], + NDArray[np.int64], + NDArray[np.int64], + NDArray[np.int64], + NDArray[np.int64], + NDArray[np.int64], +]: + """The ROI grid and its centre pixels: ``(x1, x2, y1, y2, cx, cy)``. + + Shared by :func:`downsample_maps_to_roi_dataframe` and the streaming path, + which needs the centres up front to build :class:`CentrePixelSampler`. The two + must agree exactly — the sampler fills a positional array that the DataFrame + then labels with these coordinates, so any drift mislabels every ROI silently + rather than raising. + + Partial tiles at the right/bottom edge are kept when at least half a tile + remains, and their centre is the midpoint of the *clipped* extent, not + ``x1 + roi_size // 2``. + """ + if stride is None: + stride = roi_size + yy, xx = np.meshgrid( + np.arange(0, height, stride), np.arange(0, width, stride), indexing="ij" + ) + x1_all, y1_all = xx.ravel(), yy.ravel() + x2_all = np.minimum(x1_all + roi_size, width) + y2_all = np.minimum(y1_all + roi_size, height) + + valid = (x2_all - x1_all >= roi_size // 2) & (y2_all - y1_all >= roi_size // 2) + x1_arr, x2_arr = x1_all[valid], x2_all[valid] + y1_arr, y2_arr = y1_all[valid], y2_all[valid] + if len(x1_arr) == 0: + raise ValueError( + f"No tiles generated. Image: {height}x{width}, " + f"roi_size={roi_size}, stride={stride}." + ) + return ( + x1_arr, + x2_arr, + y1_arr, + y2_arr, + (x1_arr + x2_arr) // 2, + (y1_arr + y2_arr) // 2, + ) + + +def downsample_maps_to_roi_dataframe( + focus_maps: dict[str, NDArray[np.float32] | None], + tissue_mask: NDArray[np.generic] | None, + roi_size: int = 35, + stride: int | None = None, + downsample_factor: int = 8, + min_tissue_coverage: float = 0.0, + image_shape: tuple[int, int] | None = None, + presampled: dict[str, NDArray[np.float64]] | None = None, +) -> pd.DataFrame: + """Down-sample pixel-level focus maps to per-tile scalars. + + When *presampled* is given, the centre-pixel sampling step is skipped and those + per-ROI arrays are used instead. That is how the streaming path reuses every + column, threshold and normalisation below without ever assembling a + full-resolution plane: ``CentrePixelSampler`` produces exactly the arrays this + function would have read out of ``map[cy, cx]``. Keys are the map names + (``dapi_focus_map``, ``boundary_mean_map``, ...); *focus_maps* may then be an + empty dict. + + The output DataFrame has **exactly** the same columns as the legacy + ``calculate_roi_focusscore()`` function, ensuring full backward + compatibility. + + For each tile the focus score is obtained by sampling the **centre pixel** + of the corresponding pixel-level map. Because ``uniform_filter`` at pixel + ``(cy, cx)`` computes statistics over the surrounding ``window_size x + window_size`` window, the centre pixel of a ``roi_size x roi_size`` tile + yields the exact block-level statistic (for non-edge tiles). Edge tiles may + differ slightly because ``uniform_filter`` uses ``mode='reflect'`` padding + while the legacy code clips to the image boundary. + + Args: + focus_maps: Dictionary returned by :func:`compute_all_focus_maps`. + Required key: ``dapi_focus_map``. All other keys may be ``None``. + tissue_mask: Labelled tissue mask at down-sampled resolution (e.g. + level 3). Pass ``None`` to skip tissue filtering (all tiles get + ``tissue_coverage=1.0``). + roi_size: Side length of each square tile in pixels. + stride: Grid spacing in pixels. Defaults to *roi_size* (non- + overlapping grid). + downsample_factor: Scale factor between full-resolution coordinates + and *tissue_mask* coordinates (default 8 for level 3). + min_tissue_coverage: Minimum tissue fraction for a tile to be + included. ``0.0`` means any tissue overlap is sufficient. This + parameter is recorded but **not** used for filtering — all tiles + are returned so the caller can filter as needed. + image_shape: ``(height, width)`` of the full-resolution image. If + ``None`` the shape is inferred from ``dapi_focus_map``. + + Returns: + :class:`~pandas.DataFrame` with columns identical to the legacy + ``calculate_roi_focusscore()`` output. + + Raises: + ValueError: If ``dapi_focus_map`` is missing from *focus_maps* or the + Tile grid is empty. + """ + # ------------------------------------------------------------------ + # Unpack maps + # ------------------------------------------------------------------ + # In streaming mode there are no planes to inspect, so channel presence — which + # decides both the sampling below and the column set — comes from the same dict + # the samples do. + source: dict[str, object] = presampled if presampled is not None else focus_maps # type: ignore[assignment] + if source.get("dapi_focus_map") is None: + which = "presampled" if presampled is not None else "focus_maps" + raise ValueError(f"{which} must contain 'dapi_focus_map'.") + has_boundary = source.get("boundary_focus_map") is not None + has_intrna = source.get("intrna_focus_map") is not None + + dapi_focus_map = focus_maps.get("dapi_focus_map") + dapi_mean_map = focus_maps.get("dapi_mean_map") + dapi_lap_var_map = focus_maps.get("dapi_lap_var_map") + boundary_focus_map = focus_maps.get("boundary_focus_map") + boundary_mean_map = focus_maps.get("boundary_mean_map") + intrna_focus_map = focus_maps.get("intrna_focus_map") + intrna_mean_map = focus_maps.get("intrna_mean_map") + + # ------------------------------------------------------------------ + # Image shape + # ------------------------------------------------------------------ + if image_shape is not None: + height, width = image_shape + elif dapi_focus_map is not None: + height, width = dapi_focus_map.shape + else: + raise ValueError("image_shape is required when passing presampled arrays.") + + if stride is None: + stride = roi_size + + # ------------------------------------------------------------------ + # Build ROI grid — vectorized (identical logic to legacy code) + # ------------------------------------------------------------------ + t_grid = time.perf_counter() + + x_starts = np.arange(0, width, stride) + y_starts = np.arange(0, height, stride) + yy, xx = np.meshgrid(y_starts, x_starts, indexing="ij") + x1_all = xx.ravel() + y1_all = yy.ravel() + x2_all = np.minimum(x1_all + roi_size, width) + y2_all = np.minimum(y1_all + roi_size, height) + + # Filter out small edge ROIs (same threshold as legacy code) + valid = (x2_all - x1_all >= roi_size // 2) & (y2_all - y1_all >= roi_size // 2) + x1_arr = x1_all[valid] + x2_arr = x2_all[valid] + y1_arr = y1_all[valid] + y2_arr = y2_all[valid] + + n_rois = len(x1_arr) + if n_rois == 0: + raise ValueError( + f"No tiles generated from grid. " + f"Image: {height}x{width}, roi_size={roi_size}, stride={stride}." + ) + + # Keep a structured array for backward-compatible DataFrame columns + # Columns: x1, x2, y1, y2 + roi_coords_arr = np.stack([x1_arr, x2_arr, y1_arr, y2_arr], axis=1) + + logging.info( + f" [TIMING] Tile grid generation ({n_rois} tiles): {time.perf_counter() - t_grid:.1f}s" + ) + + # ------------------------------------------------------------------ + # Tissue coverage — vectorized via block sums + # ------------------------------------------------------------------ + t_tissue = time.perf_counter() + + tissue_coverages_arr: NDArray[np.float64] + if tissue_mask is not None: + # Convert tissue mask to binary + binary_mask = (tissue_mask > 0).astype(np.float64) + mask_h, mask_w = binary_mask.shape + + # Compute downsampled ROI coordinates (vectorized) + x1_ds = x1_arr // downsample_factor + x2_ds = (x2_arr - 1) // downsample_factor + 1 + y1_ds = y1_arr // downsample_factor + y2_ds = (y2_arr - 1) // downsample_factor + 1 + + # Clip to mask boundaries + x1_ds = np.clip(x1_ds, 0, mask_w) + x2_ds = np.clip(x2_ds, 0, mask_w) + y1_ds = np.clip(y1_ds, 0, mask_h) + y2_ds = np.clip(y2_ds, 0, mask_h) + + # Check if all ROI blocks have uniform downsampled size + block_w = x2_ds - x1_ds + block_h = y2_ds - y1_ds + uniform_w = int(block_w[0]) if len(block_w) > 0 else 0 + uniform_h = int(block_h[0]) if len(block_h) > 0 else 0 + all_uniform = bool( + np.all(block_w == uniform_w) and np.all(block_h == uniform_h) + ) + + if all_uniform and uniform_w > 0 and uniform_h > 0: + # Fast path: use a 2D integral image (summed-area table) + # to compute block sums in O(1) per tile + integral = np.zeros((mask_h + 1, mask_w + 1), dtype=np.float64) + integral[1:, 1:] = np.cumsum(np.cumsum(binary_mask, axis=0), axis=1) + # Block sum = integral[y2,x2] - integral[y1,x2] - integral[y2,x1] + integral[y1,x1] + block_sums = ( + integral[y2_ds, x2_ds] + - integral[y1_ds, x2_ds] + - integral[y2_ds, x1_ds] + + integral[y1_ds, x1_ds] + ) + total_pixels = uniform_w * uniform_h + tissue_coverages_arr = block_sums / total_pixels + else: + # Fallback: still use integral image but handle variable block sizes + integral = np.zeros((mask_h + 1, mask_w + 1), dtype=np.float64) + integral[1:, 1:] = np.cumsum(np.cumsum(binary_mask, axis=0), axis=1) + block_sums = ( + integral[y2_ds, x2_ds] + - integral[y1_ds, x2_ds] + - integral[y2_ds, x1_ds] + + integral[y1_ds, x1_ds] + ) + total_pixels_arr = (block_w * block_h).astype(np.float64) + # Avoid division by zero + total_pixels_arr[total_pixels_arr == 0] = 1.0 + tissue_coverages_arr = block_sums / total_pixels_arr + else: + tissue_coverages_arr = np.ones(n_rois, dtype=np.float64) + + logging.info( + f" [TIMING] Tissue coverage computation: {time.perf_counter() - t_tissue:.1f}s" + ) + + # ------------------------------------------------------------------ + # Sample centre pixel for each ROI — vectorized fancy indexing + # ------------------------------------------------------------------ + t_sample = time.perf_counter() + + # Compute centre coordinates for all ROIs at once + cx = (x1_arr + x2_arr) // 2 + cy = (y1_arr + y2_arr) // 2 + + if presampled is not None: + # Streaming path: the tile pass already read one pixel per ROI. + def _sampled(name: str) -> NDArray[np.float64] | None: + arr = presampled.get(name) + return None if arr is None else np.asarray(arr, dtype=np.float64) + + dapi_focus_scores = _sampled("dapi_focus_map") + if dapi_focus_scores is None: + raise ValueError("presampled must contain 'dapi_focus_map'") + _zeros = np.zeros(n_rois, dtype=np.float64) + dapi_intensities = _sampled("dapi_mean_map") + if dapi_intensities is None: + dapi_intensities = _zeros + dapi_lap_vars = _sampled("dapi_lap_var_map") + if dapi_lap_vars is None: + dapi_lap_vars = _zeros + boundary_focus_scores = _sampled("boundary_focus_map") if has_boundary else None + boundary_intensities = _sampled("boundary_mean_map") if has_boundary else None + intrna_focus_scores = _sampled("intrna_focus_map") if has_intrna else None + intrna_intensities = _sampled("intrna_mean_map") if has_intrna else None + logging.info( + f" [TIMING] Centre-pixel sampling ({n_rois} tiles): from the tile pass" + ) + else: + # DAPI channels (always present) + dapi_focus_scores = dapi_focus_map[cy, cx].astype(np.float64) + dapi_intensities = ( + dapi_mean_map[cy, cx].astype(np.float64) + if dapi_mean_map is not None + else np.zeros(n_rois, dtype=np.float64) + ) + dapi_lap_vars = ( + dapi_lap_var_map[cy, cx].astype(np.float64) + if dapi_lap_var_map is not None + else np.zeros(n_rois, dtype=np.float64) + ) + + # Boundary channel + if has_boundary: + boundary_focus_scores = boundary_focus_map[cy, cx].astype(np.float64) # type: ignore[index] + boundary_intensities = boundary_mean_map[cy, cx].astype(np.float64) # type: ignore[index] + else: + boundary_focus_scores = None + boundary_intensities = None + + # IntRNA channel + if has_intrna: + intrna_focus_scores = intrna_focus_map[cy, cx].astype(np.float64) # type: ignore[index] + intrna_intensities = intrna_mean_map[cy, cx].astype(np.float64) # type: ignore[index] + else: + intrna_focus_scores = None + intrna_intensities = None + + logging.info( + f" [TIMING] Centre-pixel sampling ({n_rois} tiles): " + f"{time.perf_counter() - t_sample:.1f}s" + ) + + # ------------------------------------------------------------------ + # Normalize focus scores with RobustScaler + # ------------------------------------------------------------------ + t_scaler = time.perf_counter() + scaler_dapi = RobustScaler() + dapi_focus_scores_norm = scaler_dapi.fit_transform( + dapi_focus_scores.reshape(-1, 1) + ).flatten() + + if has_boundary: + scaler_boundary = RobustScaler() + boundary_focus_scores_norm: NDArray[np.float64] | None = ( + scaler_boundary.fit_transform( + boundary_focus_scores.reshape(-1, 1) # type: ignore[union-attr] + ).flatten() + ) + else: + boundary_focus_scores_norm = None + + if has_intrna: + scaler_intrna = RobustScaler() + intrna_focus_scores_norm: NDArray[np.float64] | None = ( + scaler_intrna.fit_transform( + intrna_focus_scores.reshape(-1, 1) # type: ignore[union-attr] + ).flatten() + ) + else: + intrna_focus_scores_norm = None + + logging.info( + f" [TIMING] RobustScaler normalization: {time.perf_counter() - t_scaler:.1f}s" + ) + + # ------------------------------------------------------------------ + # Mark boundary tiles (centre within roi_size//2 of image edge) + # ------------------------------------------------------------------ + half_win = roi_size // 2 + is_boundary_roi = ( + (cx < half_win) + | (cy < half_win) + | (cx >= width - half_win) + | (cy >= height - half_win) + ) + + # ------------------------------------------------------------------ + # Build output DataFrame (column order matches legacy code) + # ------------------------------------------------------------------ + df_data: dict[str, Any] = { + "roi_id": np.arange(n_rois), + "x1": roi_coords_arr[:, 0], + "x2": roi_coords_arr[:, 1], + "y1": roi_coords_arr[:, 2], + "y2": roi_coords_arr[:, 3], + # DAPI focus scores (duplicated for backward compatibility) + "focus_score": dapi_focus_scores, + "focus_score_norm": dapi_focus_scores_norm, + "dapi_focus_score": dapi_focus_scores, + "dapi_focus_score_norm": dapi_focus_scores_norm, + # Laplacian variance (DAPI) + "dapi_lap_var": dapi_lap_vars, + # Intensities + "dapi_intensity": dapi_intensities, + "raw_intensity": dapi_intensities.copy(), # backward compatibility + # Boundary channel + "boundary_focus_score": (boundary_focus_scores if has_boundary else np.nan), + "boundary_focus_score_norm": ( + boundary_focus_scores_norm if has_boundary else np.nan + ), + "boundary_intensity": (boundary_intensities if has_boundary else np.nan), + # IntRNA channel + "intrna_focus_score": (intrna_focus_scores if has_intrna else np.nan), + "intrna_focus_score_norm": (intrna_focus_scores_norm if has_intrna else np.nan), + "intrna_intensity": (intrna_intensities if has_intrna else np.nan), + # Tissue coverage + "tissue_coverage": tissue_coverages_arr, + "overlaps_tissue": tissue_coverages_arr > 0.0, + # Boundary flag: True if ROI centre is near image edge + "is_boundary_roi": is_boundary_roi, + } + + return pd.DataFrame(df_data) + + +def calculate_roi_focusscore_without_laplace( + xoa_morphology_files, + roi_size=35, + stride=None, + tissue_filter=True, + min_tissue_coverage=0.0, + downsample_factor=8, +): + """ + Calculate cell-independent tile-based focus scores using a regular grid. + + Creates a regular lattice/grid of tiles across the entire slide and calculates + focus scores for each tile independently of cell locations. Calculates focus scores + for all available channels (DAPI, Boundary, IntRNA). + + Parameters: + ----------- + xoa_morphology_files : list + List of paths to morphology image files + roi_size : int, optional + Size of square tile in pixels (default: 35) + stride : int, optional + Grid spacing in pixels. If None, uses non-overlapping grid (stride = roi_size) + tissue_filter : bool, optional + Enable tissue region filtering (default: True) + min_tissue_coverage : float, optional + Minimum fraction of tile that must be tissue (default: 0.0, i.e., any tissue overlap) + downsample_factor : int, optional + Downsampling factor for tissue mask (default: 8, for level 3) + + Returns: + -------- + pandas DataFrame + DataFrame with columns: + - roi_id: Unique identifier + - x1, x2, y1, y2: Tile boundaries (full resolution) + - focus_score: Raw DAPI focus score (std² / mean) [backward compatibility] + - focus_score_norm: Normalized DAPI focus score [backward compatibility] + - dapi_focus_score, dapi_focus_score_norm: DAPI focus scores + - boundary_focus_score, boundary_focus_score_norm: Boundary focus scores (if available) + - intrna_focus_score, intrna_focus_score_norm: IntRNA focus scores (if available) + - dapi_intensity: Mean DAPI intensity per tile + - boundary_intensity: Mean Boundary intensity per tile (if available) + - intrna_intensity: Mean IntRNA intensity per tile (if available) + - raw_intensity: Mean DAPI intensity per tile [backward compatibility] + - tissue_coverage: Fraction of tile that is tissue (0.0-1.0) + """ + + # Set stride (non-overlapping if not specified) + if stride is None: + stride = roi_size + + # Load channels from either multi-channel stack or split single-channel files. + dapi_image, boundary_image, intrna_image = _load_morphology_channels( + xoa_morphology_files, level=0 + ) + n_channels = 1 + int(boundary_image is not None) + int(intrna_image is not None) + + height, width = dapi_image.shape + + # Check which channels are available + has_boundary = n_channels > 1 + has_intrna = n_channels > 2 + + if has_boundary: + logging.info( + f" Found {n_channels} channels: DAPI, Boundary" + + (", IntRNA" if has_intrna else "") + ) + else: + logging.info(f" Found {n_channels} channel(s): DAPI only") + + # Generate tissue mask if filtering enabled + tissue_mask = None + if tissue_filter: + # Load downsampled DAPI for tissue mask generation + small0_ds, _, _ = _load_morphology_channels(xoa_morphology_files, level=3) + # Generate tissue mask (simplified version - just need whole_sample) + # Tissue mask via the shared helper (hysteresis + degeneracy guard + `> 0`) — + # single source of truth, see compute_tissue_mask. Fixed 2026-06-23 (was + # the percentile-60 + `test_mask > 1` defect). This path builds + # tissue_coverage, so the fix here is what corrects the §5.5 mask gate + # and the GMM tissue selection on sparse/dim slides. + tissue_mask = compute_tissue_mask(small0_ds)[0] + del small0_ds + + # Create grid of ROI coordinates + n_x = (width // stride) + (1 if width % stride > 0 else 0) + n_y = (height // stride) + (1 if height % stride > 0 else 0) + + roi_coords = [] + for y_idx in range(n_y): + for x_idx in range(n_x): + x1 = x_idx * stride + x2 = min(x1 + roi_size, width) + y1 = y_idx * stride + y2 = min(y1 + roi_size, height) + + # Skip if ROI is too small (at edges) + if (x2 - x1) < roi_size // 2 or (y2 - y1) < roi_size // 2: + continue + + roi_coords.append((x1, x2, y1, y2)) + + # Calculate tissue coverage for ALL tiles (no filtering) + # This ensures we always have tiles to process, even if tissue detection fails + tissue_coverages = [] + if tissue_filter and tissue_mask is not None: + for x1, x2, y1, y2 in roi_coords: + # Scale coordinates to downsampled space + x1_ds = x1 // downsample_factor + x2_ds = (x2 - 1) // downsample_factor + 1 + y1_ds = y1 // downsample_factor + y2_ds = (y2 - 1) // downsample_factor + 1 + + # Clip to mask boundaries + x1_ds = max(0, x1_ds) + x2_ds = min(tissue_mask.shape[1], x2_ds) + y1_ds = max(0, y1_ds) + y2_ds = min(tissue_mask.shape[0], y2_ds) + + # Calculate tissue coverage + roi_mask = tissue_mask[y1_ds:y2_ds, x1_ds:x2_ds] + tissue_pixels = np.sum(roi_mask > 0) + total_pixels = roi_mask.size + coverage = tissue_pixels / total_pixels if total_pixels > 0 else 0.0 + tissue_coverages.append(coverage) + else: + # If tissue filtering is disabled, assume all tiles have full tissue coverage + tissue_coverages = [1.0] * len(roi_coords) + + # Calculate focus scores for all ROIs and all available channels + n_rois = len(roi_coords) + + # Basic safety check (should never trigger now, but defensive programming) + if n_rois == 0: + raise ValueError( + f"ERROR: No tiles generated from grid!\n" + f" - Image dimensions: {height}x{width} (full resolution)\n" + f" - Tile size: {roi_size}px, stride: {stride}px\n" + f" - This should not happen. Please check image dimensions and tile parameters.\n" + ) + + # Initialize arrays for all channels + dapi_focus_scores = np.empty(n_rois, dtype=np.float64) + dapi_intensities = np.empty(n_rois, dtype=np.float64) + + boundary_focus_scores = np.empty(n_rois, dtype=np.float64) if has_boundary else None + boundary_intensities = np.empty(n_rois, dtype=np.float64) if has_boundary else None + + intrna_focus_scores = np.empty(n_rois, dtype=np.float64) if has_intrna else None + intrna_intensities = np.empty(n_rois, dtype=np.float64) if has_intrna else None + + # Calculate focus scores for each ROI + for i, (x1, x2, y1, y2) in enumerate(roi_coords): + # DAPI channel + roi_dapi = dapi_image[y1:y2, x1:x2] + mean_dapi = np.mean(roi_dapi) + std_dapi = np.std(roi_dapi) + dapi_focus_scores[i] = ( + (std_dapi * std_dapi) / mean_dapi if mean_dapi > 0 else 0.0 + ) + dapi_intensities[i] = mean_dapi + + # Boundary channel (if available) + if has_boundary: + roi_boundary = boundary_image[y1:y2, x1:x2] + mean_boundary = np.mean(roi_boundary) + std_boundary = np.std(roi_boundary) + boundary_focus_scores[i] = ( + (std_boundary * std_boundary) / mean_boundary + if mean_boundary > 0 + else 0.0 + ) + boundary_intensities[i] = mean_boundary + + # IntRNA channel (if available) + if has_intrna: + roi_intrna = intrna_image[y1:y2, x1:x2] + mean_intrna = np.mean(roi_intrna) + std_intrna = np.std(roi_intrna) + intrna_focus_scores[i] = ( + (std_intrna * std_intrna) / mean_intrna if mean_intrna > 0 else 0.0 + ) + intrna_intensities[i] = mean_intrna + + # Normalize focus scores for each channel separately + # Safety check: Ensure we have data before normalizing + if len(dapi_focus_scores) == 0: + raise ValueError( + f"ERROR: Cannot normalize focus scores - empty array detected!\n" + f" - Number of tiles: {n_rois}\n" + f" - This should have been caught earlier. Please report this issue.\n" + ) + + scaler_dapi = RobustScaler() + dapi_focus_scores_norm = scaler_dapi.fit_transform( + dapi_focus_scores.reshape(-1, 1) + ).flatten() + + if has_boundary: + if len(boundary_focus_scores) == 0: + raise ValueError( + "ERROR: Cannot normalize boundary focus scores - empty array detected!" + ) + scaler_boundary = RobustScaler() + boundary_focus_scores_norm = scaler_boundary.fit_transform( + boundary_focus_scores.reshape(-1, 1) + ).flatten() + else: + boundary_focus_scores_norm = None + + if has_intrna: + if len(intrna_focus_scores) == 0: + raise ValueError( + "ERROR: Cannot normalize IntRNA focus scores - empty array detected!" + ) + scaler_intrna = RobustScaler() + intrna_focus_scores_norm = scaler_intrna.fit_transform( + intrna_focus_scores.reshape(-1, 1) + ).flatten() + else: + intrna_focus_scores_norm = None + + # Create output DataFrame + # roi_id: Simple integer IDs (0, 1, 2, ...) for easy indexing + df_data = { + "roi_id": range(n_rois), + "x1": [coords[0] for coords in roi_coords], + "x2": [coords[1] for coords in roi_coords], + "y1": [coords[2] for coords in roi_coords], + "y2": [coords[3] for coords in roi_coords], + # DAPI focus scores (also kept as focus_score/focus_score_norm for backward compatibility) + "focus_score": dapi_focus_scores, + "focus_score_norm": dapi_focus_scores_norm, + "dapi_focus_score": dapi_focus_scores, + "dapi_focus_score_norm": dapi_focus_scores_norm, + "dapi_intensity": dapi_intensities, + "raw_intensity": dapi_intensities, # Backward compatibility + "tissue_coverage": tissue_coverages if tissue_filter else [1.0] * n_rois, + "overlaps_tissue": [ + coverage > 0.0 + for coverage in (tissue_coverages if tissue_filter else [1.0] * n_rois) + ], + } + + # Add Boundary channel data if available + if has_boundary: + df_data["boundary_focus_score"] = boundary_focus_scores + df_data["boundary_focus_score_norm"] = boundary_focus_scores_norm + df_data["boundary_intensity"] = boundary_intensities + else: + df_data["boundary_focus_score"] = np.nan + df_data["boundary_focus_score_norm"] = np.nan + df_data["boundary_intensity"] = np.nan + + # Add IntRNA channel data if available + if has_intrna: + df_data["intrna_focus_score"] = intrna_focus_scores + df_data["intrna_focus_score_norm"] = intrna_focus_scores_norm + df_data["intrna_intensity"] = intrna_intensities + else: + df_data["intrna_focus_score"] = np.nan + df_data["intrna_focus_score_norm"] = np.nan + df_data["intrna_intensity"] = np.nan + + df_grid_roi = pd.DataFrame(df_data) + + # Clean up + del dapi_image + if has_boundary: + del boundary_image + if has_intrna: + del intrna_image + + return df_grid_roi + + +@dataclass +class StreamedTileResults: + """The small reductions that replace the full-resolution pixel planes. + + Assembling the planes cost ~154 GB of scratch on a 5.5 gigapixel sample, and + `mmap` over a FUSE/S3 work directory is pathological (see + `docs/failures/2026-07-24_imageqc-mmap-over-fusion.md`). No downstream consumer + needs a whole plane, so in streaming mode each tile is folded into these arrays + and dropped. Everything here is O(n_ROIs), O(n_cells) or O(pixels / factor^2) — + tens of MB, not tens of GB. + + Attributes: + presampled: One centre pixel per ROI, keyed by map name + (``dapi_focus_map``, ``boundary_mean_map``, ...). Feeds + :func:`downsample_maps_to_roi_dataframe`'s ``presampled`` argument. + roi_snr_db: Per-ROI Otsu SNR in dB, replacing + ``snr_metrics.compute_image_snr_from_pixel_maps``' loop over the DAPI + mean plane. Aligned with the ROI grid. + focus_heatmap: DAPI focus map down-sampled by ``heatmap_factor``, replacing + ``downscale_local_mean`` on the assembled plane in Figure 5. + nuclear_counts / nuclear_sums: Per-nucleus pixel counts and value sums over + ``masks/0``, including the ``centroid_*_sum`` keys. Replaces the + ``_labeled_sums_chunked`` pass in :func:`calculate_ccfs_from_focus_maps`. + cell_counts / cell_sums: The same over ``masks/1``, giving cell areas and + per-cell boundary/IntRNA means. + """ + + presampled: dict[str, NDArray[np.float64]] + roi_snr_db: NDArray[np.float64] | None = None + focus_heatmap: NDArray[np.float64] | None = None + heatmap_factor: int = 8 + nuclear_counts: NDArray[np.int64] | None = None + nuclear_sums: dict[str, NDArray[np.float64]] | None = None + cell_counts: NDArray[np.int64] | None = None + cell_sums: dict[str, NDArray[np.float64]] | None = None + + @property + def has_per_cell(self) -> bool: + return self.nuclear_counts is not None + + +def _stream_channels( + lazy_channels, + image_shape: tuple[int, int], + roi_size: int, + stride: int | None, + gpu_ids: list[int], + *, + lap_sigma: float, + gpu_mem_bytes: int | None, + cell_masks_path: Path | None = None, + heatmap_factor: int = 8, +) -> StreamedTileResults: + """Run the tiled compute per channel, folding each tile into small reductions. + + The plane-based path assembles a full-resolution map per channel and then makes + one pass over it per consumer. Here the consumers are attached to the tile + dispatch instead, so a tile is reduced and dropped as soon as it is computed + and no plane is ever allocated. + + Which consumers attach depends on the channel, because the downstream metrics + do not use every channel the same way: + + * every channel needs one centre pixel per ROI, for the ROI DataFrame; + * only DAPI feeds the Figure 5 heatmap, the Otsu SNR and the per-nucleus CCFS; + * cell areas come from any single channel's pass over ``masks/1``, so they are + taken from DAPI, which is always present, while boundary and IntRNA each + contribute their own per-cell mean. + + The label planes stay lazy: ``LabeledSumAccumulator`` slices row blocks out of + zarr per tile, so neither ``masks/0`` nor ``masks/1`` (22 GB each on a 5.5 + gigapixel sample) is materialised. + """ + x1, x2, y1, y2, cx, cy = compute_roi_grid( + image_shape[0], image_shape[1], roi_size, stride + ) + logging.info(f" Streaming mode: {len(cx):,} ROIs, no pixel planes materialised") + + nuclear_plane = cell_plane = None + if cell_masks_path is not None: + masks = open_zarr(cell_masks_path).get("masks") + nuclear_plane = LazyLabelPlane(masks.get("0")) + cell_plane = LazyLabelPlane(masks.get("1")) + logging.info(" Per-cell CCFS will be reduced during the tile pass") + + presampled: dict[str, NDArray[np.float64]] = {} + nuclear_counts: NDArray[np.int64] | None = None + nuclear_sums: dict[str, NDArray[np.float64]] | None = None + cell_counts: NDArray[np.int64] | None = None + cell_sums: dict[str, NDArray[np.float64]] = {} + roi_snr_db: NDArray[np.float64] | None = None + focus_heatmap: NDArray[np.float64] | None = None + + for name, channel_data in ( + ("dapi", lazy_channels[0]), + ("boundary", lazy_channels[1]), + ("intrna", lazy_channels[2]), + ): + if channel_data is None: + continue + is_dapi = name == "dapi" + + # Built by a factory, because the fold runs concurrently on one set of + # accumulators per slot and the parent merges them. Order must be identical + # across sets -- merging pairs them positionally. + def _make_consumers() -> tuple[list[Any], dict[str, Any]]: + made: dict[str, Any] = { + "sampler": CentrePixelSampler( + cy, + cx, + ["focus_map", "mean_map"] + (["lap_var_map"] if is_dapi else []), + ) + } + ordered: list[Any] = [made["sampler"]] + if is_dapi: + made["heat"] = BlockMeanAccumulator( + image_shape, factor=heatmap_factor, map_key="focus_map" + ) + made["snr"] = RoiOtsuSnrAccumulator( + y1, y2, x1, x2, image_shape, map_key="mean_map" + ) + ordered += [made["heat"], made["snr"]] + if nuclear_plane is not None: + made["nuc"] = LabeledSumAccumulator( + nuclear_plane, + ["focus_map", "mean_map"], + include_coords=True, + # CCFS reads only labels > 0 from this one. + skip_background=True, + ) + made["cells"] = LabeledSumAccumulator(cell_plane, []) + ordered += [made["nuc"], made["cells"]] + elif cell_plane is not None: + made["cells"] = LabeledSumAccumulator(cell_plane, ["mean_map"]) + ordered.append(made["cells"]) + return ordered, made + + consumers, named = _make_consumers() + sampler = named["sampler"] + heat = named.get("heat") + snr = named.get("snr") + nuc = named.get("nuc") + cells = named.get("cells") + + t_ch = time.perf_counter() + _compute_channel_maps_tiled( + channel_data, + image_shape, + roi_size, + gpu_ids, + include_laplacian=is_dapi, + lap_sigma=lap_sigma, + gpu_mem_bytes=gpu_mem_bytes, + plane_prefix=name, + consumers=consumers, + ) + logging.info( + f" [TIMING] {name} channel streamed: {time.perf_counter() - t_ch:.1f}s" + ) + _log_mem(f"{name} channel streamed") + + # CentrePixelSampler keys are per-channel ("focus_map"); the ROI DataFrame + # wants them qualified ("dapi_focus_map"). + for key, values in sampler.finalize().items(): + presampled[f"{name}_{key}"] = values + + if is_dapi: + focus_heatmap = heat.finalize() + roi_snr_db = snr.finalize() + if nuc is not None: + nuclear_counts, nuclear_sums = nuc.finalize() + cell_counts, _ = cells.finalize() + elif cells is not None: + cell_sums[name] = cells.finalize()[1]["mean_map"] + + # No index alignment needed: every channel's pass covers the whole image, so each + # per-label array grows to the same largest label. Asserted in + # test_cell_arrays_share_one_label_index. + + return StreamedTileResults( + presampled=presampled, + roi_snr_db=roi_snr_db, + focus_heatmap=focus_heatmap, + heatmap_factor=heatmap_factor, + nuclear_counts=nuclear_counts, + nuclear_sums=nuclear_sums, + cell_counts=cell_counts, + cell_sums=cell_sums or None, + ) + + +def calculate_roi_focusscore( + xoa_morphology_files, + roi_size=35, + stride=None, + tissue_filter=True, + min_tissue_coverage=0.0, + downsample_factor=8, + use_gpu=False, + gpu_ids=None, + return_pixel_maps=False, + stream_tiles=False, + cell_masks_path=None, + heatmap_factor=8, + lap_sigma: float = 1.0, + small0_ds=None, + tissue_mask=None, +): + """ + Calculate cell-independent tile-based focus scores using a regular grid. + + Uses efficient whole-image convolution operations (uniform_filter + laplace) + to produce per-pixel focus maps, then samples the centre pixel of each tile + for backward-compatible per-tile scalars. GPU acceleration is supported via + CuPy when ``use_gpu=True``. Multi-GPU parallelism is available via + ``gpu_ids``. + + Parameters: + ----------- + xoa_morphology_files : list + List of paths to morphology image files + roi_size : int, optional + Size of square tile in pixels (default: 35) + stride : int, optional + Grid spacing in pixels. If None, uses non-overlapping grid (stride = roi_size) + tissue_filter : bool, optional + Enable tissue region filtering (default: True) + min_tissue_coverage : float, optional + Minimum fraction of tile that must be tissue (default: 0.0) + downsample_factor : int, optional + Downsampling factor for tissue mask (default: 8, for level 3) + use_gpu : bool, optional + Use CuPy GPU backend for convolutions (default: False) + gpu_ids : list[int] or None, optional + List of CUDA device IDs for multi-GPU parallelism. Channels are + distributed across GPUs. If None and use_gpu=True, GPUs are + auto-detected (falling back to device 0). + return_pixel_maps : bool, optional + If True, also return the pixel-level focus map dict (default: False) + small0_ds : numpy.ndarray or None, optional + Pre-loaded level-3 DAPI plane. When provided (with ``tissue_mask``), the + redundant level-3 decode is skipped. ``main`` already holds this array. + tissue_mask : numpy.ndarray or None, optional + Pre-computed labelled tissue mask (``compute_tissue_mask(small0_ds)[0]``). + When provided, the mask recompute is skipped -- ``main`` already produced + the identical array via ``generate_tissue_mask`` (same ``small0``, same + ``compute_tissue_mask`` default ``min_size_hole=1500``), so the result is + bit-identical. Ignored when ``tissue_filter`` is False. + + Returns: + -------- + pandas DataFrame, or ``(DataFrame, focus_maps, streamed)`` when + *return_pixel_maps* is set. Exactly one of the last two is not ``None``: + ``focus_maps`` on the plane-based path, a :class:`StreamedTileResults` when + *stream_tiles* folded each tile into small reductions instead. + DataFrame with columns: + - roi_id: Unique identifier + - x1, x2, y1, y2: Tile boundaries (full resolution) + - focus_score: Raw DAPI focus score (std^2 / mean) [backward compat] + - focus_score_norm: Normalized DAPI focus score [backward compat] + - dapi_focus_score, dapi_focus_score_norm: DAPI focus scores + - dapi_lap_var: Laplacian variance (DAPI) per tile + - boundary_focus_score, boundary_focus_score_norm: Boundary (if avail) + - intrna_focus_score, intrna_focus_score_norm: IntRNA (if available) + - dapi_intensity, boundary_intensity, intrna_intensity: Mean per tile + - raw_intensity: Mean DAPI intensity per tile [backward compat] + - tissue_coverage: Fraction of tile that is tissue (0.0-1.0) + - overlaps_tissue: Boolean, tile overlaps any tissue (>0 coverage) + """ + if stride is None: + stride = roi_size + + # Channel detection + full-resolution loading happen per-path below: the + # GPU path opens lazy TIFF wrappers (no full-res pixels in host RAM) and + # computes in memory-bounded tiles; the CPU path eager-loads numpy arrays. + # This avoids materialising the whole image (tens of GB on large samples) + # before the GPU branch. + + # Generate tissue mask if filtering enabled. `main` already loads the level-3 + # DAPI plane and computes this exact mask (compute_tissue_mask(small0)[0], the + # `whole_sample` output of generate_tissue_mask) — when it threads `small0_ds` + # and `tissue_mask` in we skip a redundant level-3 decode + mask recompute + # (~20-35s on a 5.5 GP sample). Bit-identical: same small0, same + # compute_tissue_mask default min_size_hole=1500. + if tissue_filter and tissue_mask is None: + if small0_ds is None: + small0_ds, _, _ = _load_morphology_channels(xoa_morphology_files, level=3) + # Tissue mask via the shared helper (hysteresis + degeneracy guard + `> 0`) — + # single source of truth, see compute_tissue_mask. Fixed 2026-06-23 (was + # the percentile-60 + `test_mask > 1` defect). This path builds + # tissue_coverage, so the fix here is what corrects the §5.5 mask gate + # and the GMM tissue selection on sparse/dim slides. + tissue_mask = compute_tissue_mask(small0_ds)[0] + del small0_ds + elif not tissue_filter: + tissue_mask = None + + # Resolve GPU configuration + _use_gpu = use_gpu + if gpu_ids is not None and len(gpu_ids) > 0: + _use_gpu = True + + # ------------------------------------------------------------------ + # GPU path: memory-bounded tiled compute with lazy tile reads. Each + # channel is read from disk and computed in tiles so neither a full-res + # channel (tens of GB) nor its float64 intermediates land on the GPU or + # in host RAM at once. Ports the tiled+lazy path from commit 5039b6a. + # ------------------------------------------------------------------ + if _use_gpu: + # Auto-detect GPUs if the caller enabled use_gpu without naming devices + # (mem_info + tiled dispatch below index gpu_ids[0]). + if not gpu_ids: + gpu_ids = detect_gpu_ids() or [0] + + # Lazy TIFF page wrappers — no full-res pixel data loaded into RAM. + lazy_channels, img_shape = _open_morphology_lazy(xoa_morphology_files, level=0) + has_boundary = lazy_channels[1] is not None + has_intrna = lazy_channels[2] is not None + n_channels = 1 + int(has_boundary) + int(has_intrna) + + gpu_label = f"gpu_ids={gpu_ids}" if gpu_ids else f"gpu={_use_gpu}" + logging.info( + f" Found {n_channels} channel(s); computing per-pixel focus maps " + f"(window={roi_size}, {gpu_label}, tiled)..." + ) + logging.info( + f" Image shape: {img_shape[0]}x{img_shape[1]}, {n_channels} channel(s)" + ) + + t_gpu = time.perf_counter() + gpu_mem = cp.cuda.Device(gpu_ids[0]).mem_info[1] # total VRAM per device + logging.info(f" GPU VRAM: {gpu_mem / 1024**3:.1f} GB per device") + + if stream_tiles: + streamed = _stream_channels( + lazy_channels, + img_shape, + roi_size, + stride, + gpu_ids, + lap_sigma=lap_sigma, + gpu_mem_bytes=gpu_mem, + cell_masks_path=cell_masks_path, + heatmap_factor=heatmap_factor, + ) + logging.info( + f" [TIMING] GPU compute (streamed, {n_channels} ch): " + f"{time.perf_counter() - t_gpu:.1f}s" + ) + df_grid_roi = downsample_maps_to_roi_dataframe( + {}, + tissue_mask=tissue_mask, + roi_size=roi_size, + stride=stride, + downsample_factor=downsample_factor, + min_tissue_coverage=min_tissue_coverage, + image_shape=img_shape, + presampled=streamed.presampled, + ) + if return_pixel_maps: + return df_grid_roi, None, streamed + return df_grid_roi + + dapi_result = _compute_channel_maps_tiled( + lazy_channels[0], + img_shape, + roi_size, + gpu_ids, + include_laplacian=True, + lap_sigma=lap_sigma, + gpu_mem_bytes=gpu_mem, + plane_prefix="dapi", + ) + boundary_result = None + if has_boundary and lazy_channels[1] is not None: + boundary_result = _compute_channel_maps_tiled( + lazy_channels[1], + img_shape, + roi_size, + gpu_ids, + include_laplacian=False, + lap_sigma=lap_sigma, + gpu_mem_bytes=gpu_mem, + plane_prefix="boundary", + ) + intrna_result = None + if has_intrna and lazy_channels[2] is not None: + intrna_result = _compute_channel_maps_tiled( + lazy_channels[2], + img_shape, + roi_size, + gpu_ids, + include_laplacian=False, + lap_sigma=lap_sigma, + gpu_mem_bytes=gpu_mem, + plane_prefix="intrna", + ) + logging.info( + f" [TIMING] GPU compute (tiled, {n_channels} ch): " + f"{time.perf_counter() - t_gpu:.1f}s" + ) + + focus_maps = { + "dapi_focus_map": dapi_result["focus_map"], + "dapi_mean_map": dapi_result["mean_map"], + "dapi_lap_var_map": dapi_result.get("lap_var_map"), + "boundary_focus_map": ( + boundary_result["focus_map"] if boundary_result else None + ), + "boundary_mean_map": ( + boundary_result["mean_map"] if boundary_result else None + ), + "intrna_focus_map": intrna_result["focus_map"] if intrna_result else None, + "intrna_mean_map": intrna_result["mean_map"] if intrna_result else None, + } + logging.info(" Per-pixel focus maps computed (tiled).") + + df_grid_roi = downsample_maps_to_roi_dataframe( + focus_maps, + tissue_mask=tissue_mask, + roi_size=roi_size, + stride=stride, + downsample_factor=downsample_factor, + min_tissue_coverage=min_tissue_coverage, + ) + + if return_pixel_maps: + return df_grid_roi, focus_maps, None + return df_grid_roi + + # ------------------------------------------------------------------ + # CPU path: incremental per-channel compute→downsample→free to limit + # peak memory. At most ~3 maps + convolution intermediates at once. + # ------------------------------------------------------------------ + # Eager-load full-resolution channels (CPU path needs numpy arrays). + dapi_image, boundary_image, intrna_image = _load_morphology_channels( + xoa_morphology_files, level=0 + ) + has_boundary = boundary_image is not None + has_intrna = intrna_image is not None + + logging.info(f" Computing per-pixel focus maps (window={roi_size}, gpu=False)...") + + # Build ROI grid once (cheap — just coordinate arrays) + image_shape = dapi_image.shape[:2] + grid = _build_roi_grid( + image_shape, roi_size, stride, tissue_mask, downsample_factor + ) + cx, cy = grid["cx"], grid["cy"] + n_rois = grid["n_rois"] + logging.info(f" Tile grid: {n_rois:,} tiles") + + # -- DAPI (always present) -- + dapi_focus_map, dapi_mean_map = compute_ccfs_map( + dapi_image, window_size=roi_size, use_gpu=False, gpu_id=0 + ) + dapi_lap_var_map = compute_laplacian_variance_map( + dapi_image, + window_size=roi_size, + use_gpu=False, + gpu_id=0, + lap_sigma=lap_sigma, + ) + del dapi_image + # Sample per-tile scalars + dapi_focus_scores = dapi_focus_map[cy, cx].astype(np.float64) + dapi_intensities = dapi_mean_map[cy, cx].astype(np.float64) + dapi_lap_vars = dapi_lap_var_map[cy, cx].astype(np.float64) + del dapi_lap_var_map # not needed downstream + logging.info(" DAPI maps computed and sampled.") + + # -- Boundary -- + boundary_focus_scores = None + boundary_intensities = None + boundary_mean_map = None + if has_boundary: + b_focus, boundary_mean_map = compute_ccfs_map( + boundary_image, window_size=roi_size, use_gpu=False, gpu_id=0 + ) + del boundary_image + boundary_focus_scores = b_focus[cy, cx].astype(np.float64) + boundary_intensities = boundary_mean_map[cy, cx].astype(np.float64) + del b_focus # boundary_mean_map kept for CCFS + logging.info(" Boundary maps computed and sampled.") + else: + del boundary_image + + # -- IntRNA -- + intrna_focus_scores = None + intrna_intensities = None + intrna_mean_map = None + if has_intrna: + i_focus, intrna_mean_map = compute_ccfs_map( + intrna_image, window_size=roi_size, use_gpu=False, gpu_id=0 + ) + del intrna_image + intrna_focus_scores = i_focus[cy, cx].astype(np.float64) + intrna_intensities = intrna_mean_map[cy, cx].astype(np.float64) + del i_focus # intrna_mean_map kept for CCFS + logging.info(" IntRNA maps computed and sampled.") + else: + del intrna_image + + logging.info(" Per-pixel focus maps computed (incremental CPU path).") + + # -- Normalize focus scores with RobustScaler -- + scaler_dapi = RobustScaler() + dapi_focus_scores_norm = scaler_dapi.fit_transform( + dapi_focus_scores.reshape(-1, 1) + ).flatten() + + boundary_focus_scores_norm = None + if boundary_focus_scores is not None: + scaler_b = RobustScaler() + boundary_focus_scores_norm = scaler_b.fit_transform( + boundary_focus_scores.reshape(-1, 1) + ).flatten() + + intrna_focus_scores_norm = None + if intrna_focus_scores is not None: + scaler_i = RobustScaler() + intrna_focus_scores_norm = scaler_i.fit_transform( + intrna_focus_scores.reshape(-1, 1) + ).flatten() + + # -- Assemble DataFrame -- + rc = grid["roi_coords_arr"] + df_data: dict[str, Any] = { + "roi_id": np.arange(n_rois), + "x1": rc[:, 0], + "x2": rc[:, 1], + "y1": rc[:, 2], + "y2": rc[:, 3], + "focus_score": dapi_focus_scores, + "focus_score_norm": dapi_focus_scores_norm, + "dapi_focus_score": dapi_focus_scores, + "dapi_focus_score_norm": dapi_focus_scores_norm, + "dapi_lap_var": dapi_lap_vars, + "dapi_intensity": dapi_intensities, + "raw_intensity": dapi_intensities.copy(), + "boundary_focus_score": boundary_focus_scores if has_boundary else np.nan, + "boundary_focus_score_norm": boundary_focus_scores_norm + if has_boundary + else np.nan, + "boundary_intensity": boundary_intensities if has_boundary else np.nan, + "intrna_focus_score": intrna_focus_scores if has_intrna else np.nan, + "intrna_focus_score_norm": intrna_focus_scores_norm if has_intrna else np.nan, + "intrna_intensity": intrna_intensities if has_intrna else np.nan, + "tissue_coverage": grid["tissue_coverages"], + "overlaps_tissue": grid["tissue_coverages"] > 0.0, + "is_boundary_roi": grid["is_boundary_roi"], + } + df_grid_roi = pd.DataFrame(df_data) + + if return_pixel_maps: + focus_maps = { + "dapi_focus_map": dapi_focus_map, + "dapi_mean_map": dapi_mean_map, + "dapi_lap_var_map": None, # already freed + "boundary_focus_map": None, # already freed + "boundary_mean_map": boundary_mean_map, + "intrna_focus_map": None, # already freed + "intrna_mean_map": intrna_mean_map, + } + return df_grid_roi, focus_maps, None + return df_grid_roi + + +def calculate_roi_blur_threshold( + df_grid_roi, + intensity_threshold=ROI_INTENSITY_THRESHOLD, + focus_percentile=ROI_FOCUS_SCORE_PERCENTILE, +): + """ + Calculate tile blur detection threshold from raw focus scores. + + Option B: Exclude tiles with intensity < intensity_threshold, then calculate + percentile from remaining tissue tiles. This focuses the threshold on tissue regions. + + The threshold is applied to RAW focus scores (not normalized). Classification + is: blurred if (focus_score <= threshold) OR (intensity < intensity_threshold) + + Parameters: + ----------- + df_grid_roi : pandas DataFrame + DataFrame with 'dapi_focus_score' (raw scores) and 'dapi_intensity' + intensity_threshold : float, optional + Minimum intensity to include tiles in percentile calculation (default: ROI_INTENSITY_THRESHOLD) + focus_percentile : float, optional + Percentile of raw scores to use as threshold (default: ROI_FOCUS_SCORE_PERCENTILE) + + Returns: + -------- + float + Threshold value in raw score units + """ + # Get intensity column (handle both naming conventions) + intensity_col = ( + "dapi_intensity" if "dapi_intensity" in df_grid_roi.columns else "raw_intensity" + ) + if intensity_col not in df_grid_roi.columns: + raise ValueError( + "Intensity column not found. Expected 'dapi_intensity' or 'raw_intensity'" + ) + + # Get focus score column (handle both naming conventions) + focus_col = ( + "dapi_focus_score" + if "dapi_focus_score" in df_grid_roi.columns + else "focus_score" + ) + if focus_col not in df_grid_roi.columns: + raise ValueError( + "Focus score column not found. Expected 'dapi_focus_score' or 'focus_score'" + ) + + # Exclude low-intensity tiles (background) before calculating percentile + tissue_rois = df_grid_roi[df_grid_roi[intensity_col] >= intensity_threshold] + + if len(tissue_rois) == 0: + # Fallback: if no tissue tiles, use all tiles + logging.warning( + f" Warning: No tiles with intensity >= {intensity_threshold}, using all tiles for threshold calculation" + ) + tissue_rois = df_grid_roi + + raw_scores = tissue_rois[focus_col].values + + # Calculate threshold on raw scores from tissue ROIs only + threshold_raw = np.percentile(raw_scores, focus_percentile) + + return threshold_raw + + +def fit_focus_gmm( + df_grid_roi, + intensity_threshold: float = ROI_INTENSITY_THRESHOLD, + focus_col_name: str = "dapi_focus_score", + n_components: int = 2, + random_state: int = 0, +): + """ + Fit a Gaussian Mixture Model (GMM) to tile focus scores from tissue tiles + (intensity >= intensity_threshold) in raw focus-score space. + + The model learns 2 components that approximately correspond to: + - Lower-focus (blurred) tiles + - Higher-focus (in-focus) tiles + + This function does NOT classify tiles directly; it only returns the fitted GMM + and identifies which component is the "blur" component (lower mean). + + Parameters + ---------- + df_grid_roi : pandas.DataFrame + DataFrame with at least: + - focus_col_name (e.g. 'dapi_focus_score' or 'focus_score') + - 'dapi_intensity' or 'raw_intensity' + intensity_threshold : float, optional + Minimum intensity to consider a tile as tissue (default: ROI_INTENSITY_THRESHOLD). + Tiles below this are excluded from GMM training. + focus_col_name : str, optional + Name of the focus-score column to use (default: 'dapi_focus_score'). + If not present, 'focus_score' will be used. + n_components : int, optional + Number of Gaussian components for the GMM (default: 2). + random_state : int, optional + Random seed for reproducibility (default: 0). + + Returns + ------- + gmm : sklearn.mixture.GaussianMixture + Fitted GMM model on (log1p(focus_score)) of tissue tiles. + blur_component_idx : int + Index of the GMM component corresponding to blurred tiles (lower mean). + """ + # Determine intensity column + intensity_col = ( + "dapi_intensity" if "dapi_intensity" in df_grid_roi.columns else "raw_intensity" + ) + if intensity_col not in df_grid_roi.columns: + raise ValueError( + "Intensity column not found. Expected 'dapi_intensity' or 'raw_intensity'" + ) + + # Determine focus column + if focus_col_name not in df_grid_roi.columns: + # Fallback to generic 'focus_score' + if "focus_score" not in df_grid_roi.columns: + raise ValueError( + f"Focus score column '{focus_col_name}' not found and 'focus_score' not present either" + ) + focus_col_name = "focus_score" + + # Select tissue tiles for training + tissue_rois = df_grid_roi[df_grid_roi[intensity_col] >= intensity_threshold].copy() + if len(tissue_rois) == 0: + raise ValueError( + f"No tissue tiles found with intensity >= {intensity_threshold}. " + f"Cannot fit GMM. Check intensity thresholds or image quality." + ) + + raw_scores = tissue_rois[focus_col_name].values.astype(np.float64) + + # Log-transform to reduce skewness (handle zeros safely) + x_log = np.log1p(raw_scores).reshape(-1, 1) + + # Fit GMM + gmm = GaussianMixture( + n_components=n_components, covariance_type="full", random_state=random_state + ) + gmm.fit(x_log) + + # Identify which component corresponds to blurred ROIs: + # assume lower mean log-focus = blur + means = gmm.means_.flatten() + blur_component_idx = int(np.argmin(means)) + + logging.info(" GMM focus model fitted on tissue tiles") + logging.info(f" Number of tissue tiles used: {len(tissue_rois)}") + logging.info(f" Component means (log1p focus): {means}") + logging.info(f" Blur component index: {blur_component_idx} (lower mean)") + + return gmm, blur_component_idx + + +def classify_roi_blur_by_threshold( + df_grid_roi, + roi_threshold: float, + intensity_threshold: float = ROI_INTENSITY_THRESHOLD, + focus_col_name: str = "dapi_focus_score", +): + """ + Percentile-threshold fallback blur classification. + + Used when the 1D GMM fit fails (e.g. too few tissue tiles on a very dim + sample). A tile is classified as blurred if its raw focus score is at or + below ``roi_threshold`` OR its intensity is below ``intensity_threshold``. + This mirrors the rule documented in ``calculate_roi_blur_threshold`` and + produces the same columns as ``classify_roi_blur`` so downstream code is + unaffected: + + - 'blur_prob_gmm' : NaN (no posterior probability without a GMM) + - 'is_blurred_gmm' : boolean, final classification + - 'is_low_intensity' : boolean, intensity < intensity_threshold + + Parameters + ---------- + df_grid_roi : pandas.DataFrame + DataFrame with a focus-score column ('dapi_focus_score' or + 'focus_score') and an intensity column ('dapi_intensity' or + 'raw_intensity'). + roi_threshold : float + Raw focus-score threshold (from ``calculate_roi_blur_threshold``). + intensity_threshold : float, optional + Tiles below this are auto-blurred (default: ROI_INTENSITY_THRESHOLD). + focus_col_name : str, optional + Focus-score column to use (default: 'dapi_focus_score'; falls back to + 'focus_score'). + """ + df = df_grid_roi.copy() + + intensity_col = ( + "dapi_intensity" if "dapi_intensity" in df.columns else "raw_intensity" + ) + if intensity_col not in df.columns: + raise ValueError( + "Intensity column not found. Expected 'dapi_intensity' or 'raw_intensity'" + ) + + if focus_col_name not in df.columns: + if "focus_score" not in df.columns: + raise ValueError( + f"Focus score column '{focus_col_name}' not found and 'focus_score' not present either" + ) + focus_col_name = "focus_score" + + is_low_intensity = df[intensity_col] < intensity_threshold + df["is_low_intensity"] = is_low_intensity + df["blur_prob_gmm"] = np.nan + df["is_blurred_gmm"] = (df[focus_col_name] <= roi_threshold) | is_low_intensity + + total_rois = len(df) + n_low_int = int(is_low_intensity.sum()) + n_blur = int(df["is_blurred_gmm"].sum()) + pct_low = (n_low_int / total_rois * 100) if total_rois > 0 else 0.0 + pct_blur = (n_blur / total_rois * 100) if total_rois > 0 else 0.0 + + logging.info(" Percentile-threshold fallback blur classification completed") + logging.info(f" Total tiles: {total_rois}") + logging.info( + f" Low-intensity tiles (auto-blurred): {n_low_int} ({pct_low:.1f}%)" + ) + logging.info( + f" Blurred tiles (threshold + intensity): {n_blur} ({pct_blur:.1f}%)" + ) + + return df + + +def classify_roi_blur( + df_grid_roi, + gmm, + blur_component_idx: int, + blur_prob_threshold: float = 0.5, + intensity_threshold: float = ROI_INTENSITY_THRESHOLD, + focus_col_name: str = "dapi_focus_score", +): + """ + Classify each tile as blurred or in-focus using a fitted GMM and an intensity safeguard. + + Rules: + ------ + - Tiles with intensity < intensity_threshold are always marked as blurred + (low-intensity / background or globally problematic tissue). + - For tissue tiles (intensity >= threshold), use the GMM posterior probability + of belonging to the "blur" component: + - blur_prob = P(component == blur_component_idx | focus_score) + - Tile is blurred if blur_prob > blur_prob_threshold. + + The classification is added to df_grid_roi in new columns: + - 'blur_prob_gmm' : posterior probability of being blurred (NaN for low-intensity tiles) + - 'is_blurred_gmm' : boolean, final classification combining intensity + GMM + - 'is_low_intensity' : boolean, intensity < intensity_threshold + + Parameters + ---------- + df_grid_roi : pandas.DataFrame + DataFrame with at least: + - focus_col_name (e.g. 'dapi_focus_score' or 'focus_score') + - 'dapi_intensity' or 'raw_intensity' + gmm : sklearn.mixture.GaussianMixture + Fitted GMM model from fit_focus_gmm() + blur_component_idx : int + Index of the GMM component corresponding to blurred tiles. + blur_prob_threshold : float, optional + Threshold on posterior blur probability to classify a tile as blurred + (default: 0.5). + intensity_threshold : float, optional + Intensity safeguard: tiles below this are auto-blurred (default: ROI_INTENSITY_THRESHOLD). + focus_col_name : str, optional + Name of focus-score column used (default: 'dapi_focus_score'). + + Returns + ------- + df_grid_roi : pandas.DataFrame + Input DataFrame with new columns: + - 'blur_prob_gmm' + - 'is_blurred_gmm' + - 'is_low_intensity' + """ + df = df_grid_roi.copy() + + # Intensity column + intensity_col = ( + "dapi_intensity" if "dapi_intensity" in df.columns else "raw_intensity" + ) + if intensity_col not in df.columns: + raise ValueError( + "Intensity column not found. Expected 'dapi_intensity' or 'raw_intensity'" + ) + + # Focus column + if focus_col_name not in df.columns: + if "focus_score" not in df.columns: + raise ValueError( + f"Focus score column '{focus_col_name}' not found and 'focus_score' not present either" + ) + focus_col_name = "focus_score" + + # Intensity-based low-intensity flag + is_low_intensity = df[intensity_col] < intensity_threshold + df["is_low_intensity"] = is_low_intensity + + # Initialize columns + df["blur_prob_gmm"] = np.nan + df["is_blurred_gmm"] = False + + # Tissue ROIs: intensity >= threshold + tissue_mask = ~is_low_intensity + if tissue_mask.any(): + raw_scores = df.loc[tissue_mask, focus_col_name].values.astype(np.float64) + x_log = np.log1p(raw_scores).reshape(-1, 1) + + # Posterior probabilities of components + probs = gmm.predict_proba(x_log) + blur_prob = probs[:, blur_component_idx] + + df.loc[tissue_mask, "blur_prob_gmm"] = blur_prob + df.loc[tissue_mask, "is_blurred_gmm"] = blur_prob > blur_prob_threshold + + # Low-intensity tiles are always blurred according to the safeguard + df.loc[is_low_intensity, "is_blurred_gmm"] = True + + # Summary + total_rois = len(df) + n_low_int = int(is_low_intensity.sum()) + n_blur = int(df["is_blurred_gmm"].sum()) + pct_blur = (n_blur / total_rois) * 100 if total_rois > 0 else 0.0 + + logging.info(" GMM-based blur classification completed") + logging.info(f" Total tiles: {total_rois}") + logging.info( + f" Low-intensity tiles (auto-blurred): {n_low_int} ({n_low_int / total_rois * 100:.1f}%)" + ) + logging.info(f" Blurred tiles (GMM + intensity): {n_blur} ({pct_blur:.1f}%)") + + return df + + +def fit_focus_gmm_2d( + df_grid_roi, + intensity_threshold: float = ROI_INTENSITY_THRESHOLD, + focus_col_name: str = "dapi_focus_score", + n_components: int = 2, + random_state: int = 0, +): + """ + Fit a 2D Gaussian Mixture Model (GMM) to tile focus scores using both + focus score (std²/mean) and Laplacian variance as features. + + The model uses log1p-transformed features: + - Feature 1: log1p(dapi_focus_score) + - Feature 2: log1p(dapi_lap_var) + + This allows the model to use both global contrast (focus_score) and + high-frequency content (Laplacian variance) to distinguish blurred vs in-focus tiles. + + Parameters + ---------- + df_grid_roi : pandas.DataFrame + DataFrame with at least: + - focus_col_name (e.g. 'dapi_focus_score' or 'focus_score') + - 'dapi_lap_var' (Laplacian variance) + - 'dapi_intensity' or 'raw_intensity' + intensity_threshold : float, optional + Minimum intensity to consider a tile as tissue (default: ROI_INTENSITY_THRESHOLD). + Tiles below this are excluded from GMM training. + focus_col_name : str, optional + Name of the focus-score column to use (default: 'dapi_focus_score'). + If not present, 'focus_score' will be used. + n_components : int, optional + Number of Gaussian components for the GMM (default: 2). + random_state : int, optional + Random seed for reproducibility (default: 0). + + Returns + ------- + gmm : sklearn.mixture.GaussianMixture + Fitted 2D GMM model on [log1p(focus_score), log1p(lap_var)] of tissue tiles. + blur_component_idx : int + Index of the GMM component corresponding to blurred tiles (lower mean on focus_score dimension). + """ + # Determine intensity column + intensity_col = ( + "dapi_intensity" if "dapi_intensity" in df_grid_roi.columns else "raw_intensity" + ) + if intensity_col not in df_grid_roi.columns: + raise ValueError( + "Intensity column not found. Expected 'dapi_intensity' or 'raw_intensity'" + ) + + # Determine focus column + if focus_col_name not in df_grid_roi.columns: + # Fallback to generic 'focus_score' + if "focus_score" not in df_grid_roi.columns: + raise ValueError( + f"Focus score column '{focus_col_name}' not found and 'focus_score' not present either" + ) + focus_col_name = "focus_score" + + # Check for Laplacian variance column + if "dapi_lap_var" not in df_grid_roi.columns: + raise ValueError( + "dapi_lap_var column not found. 2D GMM requires Laplacian variance. " + "Make sure calculate_roi_focusscore() was used (not calculate_roi_focusscore_without_laplace)." + ) + + # Select tissue tiles for training. + # 2026-06-22 (qc_drift_analysis): select by tissue-mask coverage, not a raw + # intensity floor. The old `intensity >= ROI_INTENSITY_THRESHOLD` gate broke + # on dim XOA-4.0 images (~14x dimmer), dropping most real tissue from the + # training set and biasing the blur/focus split. Coverage is + # brightness-independent. Falls back to the intensity gate only when + # tissue_coverage is unavailable. NB: this still trains the GMM on tissue + # only (it is NOT the reverted "all-tiles" Scope B, commit e8e6731), so the + # within-tissue blur-vs-focus bimodality is preserved. + if "tissue_coverage" in df_grid_roi.columns: + tissue_rois = df_grid_roi[ + df_grid_roi["tissue_coverage"] >= ROI_MIN_TISSUE_COVERAGE_FOR_INTENSITY_QC + ].copy() + _selection_desc = ( + f"tissue_coverage >= {ROI_MIN_TISSUE_COVERAGE_FOR_INTENSITY_QC}" + ) + else: + tissue_rois = df_grid_roi[ + df_grid_roi[intensity_col] >= intensity_threshold + ].copy() + _selection_desc = f"intensity >= {intensity_threshold}" + if len(tissue_rois) == 0: + raise ValueError( + f"No tissue tiles found with {_selection_desc}. " + f"Cannot fit GMM. Check tissue mask / intensity thresholds or image quality." + ) + + # Get both features + focus_scores = tissue_rois[focus_col_name].values.astype(np.float64) + lap_vars = tissue_rois["dapi_lap_var"].values.astype(np.float64) + + # Filter out NaN values + valid_mask = ~(np.isnan(focus_scores) | np.isnan(lap_vars)) + if valid_mask.sum() == 0: + raise ValueError("No valid tile data (all NaN) for 2D GMM training") + + focus_scores_valid = focus_scores[valid_mask] + lap_vars_valid = lap_vars[valid_mask] + + # Log-transform both features (clamp negatives to 0 before log1p) + x_focus_log = np.log1p(np.maximum(focus_scores_valid, 0.0)) + x_lap_log = np.log1p(np.maximum(lap_vars_valid, 0.0)) + + # Combine into 2D feature matrix + x_2d = np.column_stack([x_focus_log, x_lap_log]) + + # Fit 2D GMM + gmm = GaussianMixture( + n_components=n_components, covariance_type="full", random_state=random_state + ) + gmm.fit(x_2d) + + # Identify which component corresponds to blurred ROIs: + # Use the focus_score dimension (first column) - lower mean = blur + means_focus = gmm.means_[:, 0] # First dimension (focus_score) + blur_component_idx = int(np.argmin(means_focus)) + + logging.info(" 2D GMM focus model fitted on tissue tiles") + logging.info(f" Number of tissue tiles used: {valid_mask.sum()}") + logging.info(" Component means (log1p focus_score, log1p lap_var):") + for i, mean in enumerate(gmm.means_): + logging.info(f" Component {i}: [{mean[0]:.4f}, {mean[1]:.4f}]") + logging.info( + f" Blur component index: {blur_component_idx} (lower mean on focus_score dimension)" + ) + + return gmm, blur_component_idx + + +def classify_roi_blur_2d( + df_grid_roi, + gmm, + blur_component_idx: int, + blur_prob_threshold: float = 0.5, + intensity_threshold: float = ROI_INTENSITY_THRESHOLD, + focus_col_name: str = "dapi_focus_score", +): + """ + Classify each tile as blurred or in-focus using a fitted 2D GMM and an intensity safeguard. + + Uses both focus_score and Laplacian variance as features. + + Rules: + ------ + - Tissue is defined by tissue_coverage >= ROI_MIN_TISSUE_COVERAGE_FOR_INTENSITY_QC + when the column is present (brightness-independent); otherwise it falls back + to intensity >= intensity_threshold. + - Non-tissue tiles (low coverage / background) are marked as blurred. They are + excluded from the tissue-filtered blur % regardless. + - Tissue tiles missing dapi_lap_var are marked as blurred (cannot use 2D GMM). + - For tissue tiles with valid lap_var, use the 2D GMM posterior probability + of belonging to the "blur" component: + - blur_prob = P(component == blur_component_idx | focus_score, lap_var) + - Tile is blurred if blur_prob > blur_prob_threshold. + + The classification is added to df_grid_roi in new columns: + - 'blur_prob_gmm_2d' : posterior probability of being blurred (NaN for low-intensity or missing lap_var tiles) + - 'is_blurred_gmm_2d' : boolean, final classification combining intensity + 2D GMM + - 'is_low_intensity' : boolean, intensity < intensity_threshold (reused if exists) + + Parameters + ---------- + df_grid_roi : pandas.DataFrame + DataFrame with at least: + - focus_col_name (e.g. 'dapi_focus_score' or 'focus_score') + - 'dapi_lap_var' (Laplacian variance) + - 'dapi_intensity' or 'raw_intensity' + gmm : sklearn.mixture.GaussianMixture + Fitted 2D GMM model from fit_focus_gmm_2d() + blur_component_idx : int + Index of the GMM component corresponding to blurred tiles. + blur_prob_threshold : float, optional + Threshold on posterior blur probability to classify a tile as blurred + (default: 0.5). + intensity_threshold : float, optional + Intensity safeguard: tiles below this are auto-blurred (default: ROI_INTENSITY_THRESHOLD). + focus_col_name : str, optional + Name of focus-score column used (default: 'dapi_focus_score'). + + Returns + ------- + df_grid_roi : pandas.DataFrame + Input DataFrame with new columns: + - 'blur_prob_gmm_2d' + - 'is_blurred_gmm_2d' + - 'is_low_intensity' (if not already present) + """ + df = df_grid_roi.copy() + + # Intensity column + intensity_col = ( + "dapi_intensity" if "dapi_intensity" in df.columns else "raw_intensity" + ) + if intensity_col not in df.columns: + raise ValueError( + "Intensity column not found. Expected 'dapi_intensity' or 'raw_intensity'" + ) + + # Focus column + if focus_col_name not in df.columns: + if "focus_score" not in df.columns: + raise ValueError( + f"Focus score column '{focus_col_name}' not found and 'focus_score' not present either" + ) + focus_col_name = "focus_score" + + # Check for Laplacian variance + if "dapi_lap_var" not in df.columns: + raise ValueError( + "dapi_lap_var column not found. 2D GMM classification requires Laplacian variance." + ) + + # Intensity-based low-intensity flag (reuse if exists, otherwise create) + if "is_low_intensity" not in df.columns: + is_low_intensity = df[intensity_col] < intensity_threshold + df["is_low_intensity"] = is_low_intensity + else: + is_low_intensity = df["is_low_intensity"] + + # Initialize columns + df["blur_prob_gmm_2d"] = np.nan + df["is_blurred_gmm_2d"] = False + + # Tissue ROIs for blur classification. + # 2026-06-22 (qc_drift_analysis): define tissue by mask coverage, not the + # intensity floor — matches fit_focus_gmm_2d. On dim XOA-4.0 images many real + # tissue tiles fall below ROI_INTENSITY_THRESHOLD; the old rule force-blurred + # them and inflated the blur rate (~40% on v4). Coverage is + # brightness-independent. is_low_intensity is still computed above for + # intensity QC / reporting, just not used to gate blur here. + if "tissue_coverage" in df.columns: + tissue_mask = df["tissue_coverage"] >= ROI_MIN_TISSUE_COVERAGE_FOR_INTENSITY_QC + else: + tissue_mask = ~is_low_intensity + has_lap_var = df["dapi_lap_var"].notna() + valid_mask = tissue_mask & has_lap_var + + if valid_mask.any(): + # Get both features for valid ROIs + focus_scores = df.loc[valid_mask, focus_col_name].values.astype(np.float64) + lap_vars = df.loc[valid_mask, "dapi_lap_var"].values.astype(np.float64) + + # Filter out any remaining NaN (shouldn't happen, but safety check) + valid_data_mask = ~(np.isnan(focus_scores) | np.isnan(lap_vars)) + if valid_data_mask.sum() > 0: + focus_scores_valid = focus_scores[valid_data_mask] + lap_vars_valid = lap_vars[valid_data_mask] + + # Log-transform (clamp negatives to 0, matching fit_focus_gmm_2d) + x_focus_log = np.log1p(np.maximum(focus_scores_valid, 0.0)) + x_lap_log = np.log1p(np.maximum(lap_vars_valid, 0.0)) + + # Combine into 2D feature matrix + x_2d = np.column_stack([x_focus_log, x_lap_log]) + + # Posterior probabilities of components + probs = gmm.predict_proba(x_2d) + blur_prob = probs[:, blur_component_idx] + + # Map back to original valid_mask indices + valid_indices = df.index[valid_mask][valid_data_mask] + df.loc[valid_indices, "blur_prob_gmm_2d"] = blur_prob + df.loc[valid_indices, "is_blurred_gmm_2d"] = blur_prob > blur_prob_threshold + + # Non-tissue tiles (low mask coverage / background) are not treated as + # focused. They are excluded from the tissue-filtered blur % anyway; this + # only affects the unfiltered count, preserving the prior convention. + df.loc[~tissue_mask, "is_blurred_gmm_2d"] = True + + # Tissue tiles without lap_var cannot be classified by the 2D GMM → blurred + missing_lap_var = ~has_lap_var & tissue_mask + if missing_lap_var.any(): + df.loc[missing_lap_var, "is_blurred_gmm_2d"] = True + + # Summary + total_rois = len(df) + n_low_int = int(is_low_intensity.sum()) + n_missing_lap = int(missing_lap_var.sum()) if missing_lap_var.any() else 0 + n_blur = int(df["is_blurred_gmm_2d"].sum()) + pct_blur = (n_blur / total_rois) * 100 if total_rois > 0 else 0.0 + + logging.info(" 2D GMM-based blur classification completed") + logging.info(f" Total tiles: {total_rois}") + logging.info( + f" Low-intensity tiles (auto-blurred): {n_low_int} ({n_low_int / total_rois * 100:.1f}%)" + ) + if n_missing_lap > 0: + logging.info( + f" Tiles missing lap_var (auto-blurred): {n_missing_lap} ({n_missing_lap / total_rois * 100:.1f}%)" + ) + logging.info(f" Blurred tiles (2D GMM + intensity): {n_blur} ({pct_blur:.1f}%)") + + return df + + +_MAX_SCATTER_POINTS = 10_000 + + +def _subsample_idx(mask, max_points=None, rng_seed=42): + """Return indices where *mask* is True, randomly subsampled to *max_points*.""" + if max_points is None: + max_points = _MAX_SCATTER_POINTS + idx = np.where(mask)[0] + if max_points <= 0 or len(idx) <= max_points: + return idx + rng = np.random.default_rng(rng_seed) + return rng.choice(idx, size=max_points, replace=False) + + +def plot_grid_roi_focus_heatmap( + df_grid_roi, + small0, + figures_dir, + figures_source_dir, + threshold=-1.0, + focus_maps=None, + focus_heatmap=None, +): + """ + Create heatmap of grid tile focus scores across the whole tissue. + + When pixel-level ``focus_maps`` are provided, renders smooth per-pixel + heatmaps via imshow (much faster and higher resolution than Rectangle + patches). Falls back to the legacy Rectangle-patch approach otherwise. + + Parameters: + ----------- + df_grid_roi : pandas DataFrame + Grid ROI DataFrame with columns: roi_id, x1, x2, y1, y2, focus_score, + focus_score_norm, raw_intensity, tissue_coverage + small0 : numpy.ndarray + Downsampled DAPI image (level 3, 8x downsampling) + figures_dir : Path + Directory to save figures + figures_source_dir : Path + Directory to save source data + threshold : float, optional + Threshold for normalized focus score (default: -1.0) + focus_maps : dict or None, optional + Pixel-level focus maps from compute_all_focus_maps(). If provided, + uses imshow for smooth rendering instead of Rectangle patches. + """ + from skimage.transform import downscale_local_mean + from mpl_toolkits.axes_grid1 import make_axes_locatable + + downsample_factor = 8 + img_height, img_width = small0.shape + img_aspect = img_height / img_width if img_width > 0 else 1.0 + panel_width = 6 + # Floor at 50% of width so very wide slides (e.g. brain) don't squash titles / colorbars + panel_height = max(panel_width * 0.5, panel_width * img_aspect) + fig, axes = plt.subplots(1, 2, figsize=(2 * panel_width + 2, panel_height)) + fig.suptitle( + "Spatial focus-score map across the Xenium region", + fontsize=15, + fontweight="bold", + y=1.02, + ) + + # DAPI background p99 is identical for both panels -- compute once over the full + # ~86 Mpx small0. Draw the background via _imshow_thumb (block-mean to the ~2000px + # panel) not a full-res imshow: the raw imshow was ~50 s/panel x2 panels x2 saves + # and was the entire residual cost of this figure (the binned panels are ~5 s). + _bg_vmax = np.percentile(small0, 99) + + # Plot 1: Focus score heatmap + ax = axes[0] + _imshow_thumb( + ax, + small0, + cmap="Greys_r", + vmax=_bg_vmax, + alpha=0.5, + aspect="auto", + extent=[0, img_width, img_height, 0], + origin="upper", + ) + + # Streaming mode supplies the down-sampled canvas directly, and it reproduces + # downscale_local_mean bit-exactly: write boundaries are aligned to the block size so + # each block is reduced by a single reshape in skimage's own order. No + # full-resolution plane is read (22 GB, plus another 22 GB for skimage's pad). + heatmap = focus_heatmap + if heatmap is None and focus_maps is not None: + dapi_focus = focus_maps.get("dapi_focus_map") + if dapi_focus is not None: + heatmap = downscale_local_mean( + dapi_focus, (downsample_factor, downsample_factor) + ) + if heatmap is not None: + # Binned redesign: an imshow of the full ~86-megapixel per-pixel field cost + # ~126 s (it measured 4459.8 s on a 102045x53908 sample, runs 1ZyVIlaKBYxJrQ / + # 4c0HFuivKDkWXr -- the largest single cost in the step) and rendered millions + # of unreadable per-pixel dots. Aggregating to a coarse grid (~180 cells on the + # long axis) keeps the regional signal and draws in ~1 s. + t_h = time.perf_counter() + # Clip to small0 dimensions (in case of rounding: the canvas can be a row/column + # larger, ceil vs floor). + focus_ds = np.asarray(heatmap)[:img_height, :img_width] + positive = focus_ds > 0 + has_positive = bool(positive.any()) + # Color range comes from the FULL-resolution positive field, not the binned + # means, so viridis maps exactly as the per-pixel figure did. + vmin = np.percentile(focus_ds[positive], 1) if has_positive else 0 + vmax = np.percentile(focus_ds[positive], 99) if has_positive else 1 + # Per-bin MEAN focus over tissue pixels (focus_ds > 0). Non-tissue pixels are + # NaN'd so they do not drag the mean down; all-non-tissue bins come back NaN and + # are set to vmin so they render as viridis-min -- the same dark background the + # per-pixel field showed where focus_ds == 0. + step = max(1, -(-max(img_height, img_width) // _FOCUS_HEATMAP_BINS_LONG)) + tissue_focus = np.where(positive, focus_ds, np.nan) + del positive + focus_binned = _bin_nanmean(tissue_focus, step) + focus_binned = np.where(np.isnan(focus_binned), vmin, focus_binned) + ax.imshow( + focus_binned, + cmap="viridis", + alpha=0.6, + aspect="auto", + extent=[0, img_width, img_height, 0], + origin="upper", + vmin=vmin, + vmax=vmax, + interpolation="nearest", + ) + logging.info( + f" [TIMING] fig5 left bin+imshow (step={step}, " + f"{focus_binned.shape}): {time.perf_counter() - t_h:.1f}s" + ) + sm = plt.cm.ScalarMappable( + cmap="viridis", norm=plt.Normalize(vmin=vmin, vmax=vmax) + ) + sm.set_array([]) + cax = make_axes_locatable(ax).append_axes("right", size="3%", pad=0.05) + cbar = fig.colorbar(sm, cax=cax) + cbar.set_label("Focus Score (var/mean)", fontsize=12) + ax.set_title("Focus Score (binned)", fontsize=14) + else: + # Legacy Rectangle-patch approach + from matplotlib.patches import Rectangle + + for _, roi in df_grid_roi.iterrows(): + x1_ds = roi["x1"] / downsample_factor + x2_ds = roi["x2"] / downsample_factor + y1_ds = roi["y1"] / downsample_factor + y2_ds = roi["y2"] / downsample_factor + w = x2_ds - x1_ds + h = y2_ds - y1_ds + if w <= 0 or h <= 0: + continue + if x1_ds < 0 or y1_ds < 0 or x2_ds > img_width or y2_ds > img_height: + continue + norm_score = np.clip((roi["focus_score_norm"] + 3) / 6, 0, 1) + color = plt.cm.viridis(norm_score) + rect = Rectangle( + (x1_ds, y1_ds), w, h, facecolor=color, alpha=0.6, edgecolor="none" + ) + ax.add_patch(rect) + sm = plt.cm.ScalarMappable(cmap="viridis", norm=plt.Normalize(vmin=-3, vmax=3)) + sm.set_array([]) + cax = make_axes_locatable(ax).append_axes("right", size="3%", pad=0.05) + cbar = fig.colorbar(sm, cax=cax) + cbar.set_label("Focus Score (Normalized)", fontsize=12) + cbar.ax.axhline(y=float(threshold), color="red", linewidth=1.5, linestyle="--") + ax.set_title("Grid Tile Focus Score Heatmap (Normalized)", fontsize=14) + + ax.set_xlim(0, img_width) + ax.set_ylim(img_height, 0) + ax.set_aspect("equal") + ax.axis("off") + + # Plot 2: GMM 2D classification (blurred vs in-focus) + ax = axes[1] + _imshow_thumb( + ax, + small0, + cmap="Greys_r", + vmax=_bg_vmax, + alpha=0.5, + aspect="auto", + extent=[0, img_width, img_height, 0], + origin="upper", + ) + + has_gmm_2d = "is_blurred_gmm_2d" in df_grid_roi.columns + + # Gated on the GMM column alone, NOT on focus_maps: this branch reads no pixel + # data. It rasterises `is_blurred_gmm_2d` from the ROI table's own x1/x2/y1/y2 + # columns at down-sampled resolution, so the focus_maps check was only ever a + # proxy for "not the legacy path". + # + # It mattered because streaming passes focus_maps=None, which sent this panel to + # the Rectangle fallback below -- one patch per ROI through df.iterrows(). On the + # 102045x53908 sample that is 4,490,640 patches, and it took Figure 5 from 15.9 s + # to 4388-4460 s. Measured on runs 47jub5CwOHb82v (streamed, 436.8 s at 429 k + # ROIs) against 3xiurC3181Zgwc (planes, 15.9 s, same bundle). + # + # `has_gmm_2d` reproduces the old behaviour exactly where it mattered: the 2D GMM + # needs dapi_lap_var, which --legacy-focus does not produce, so that path still + # falls through to the Rectangle branch. + if has_gmm_2d: + from matplotlib.colors import LinearSegmentedColormap + + t_h = time.perf_counter() + # Rasterise the in-focus indicator at downsampled resolution: 1.0 for an + # in-focus ROI pixel, 0.0 for a blurred one, NaN off-tissue (no ROI). ALL ROIs + # are painted -- not just blurred ones -- because a bin needs its tissue + # denominator; the fraction in focus is meaningless without it. + infocus_ds = np.full((img_height, img_width), np.nan, dtype=np.float32) + x1_arr = (df_grid_roi["x1"].values // downsample_factor).astype(int) + x2_arr = np.minimum( + df_grid_roi["x2"].values // downsample_factor, img_width + ).astype(int) + y1_arr = (df_grid_roi["y1"].values // downsample_factor).astype(int) + y2_arr = np.minimum( + df_grid_roi["y2"].values // downsample_factor, img_height + ).astype(int) + infocus_val = np.where( + df_grid_roi["is_blurred_gmm_2d"].values, 0.0, 1.0 + ).astype(np.float32) + for i in range(len(x1_arr)): + infocus_ds[y1_arr[i] : y2_arr[i], x1_arr[i] : x2_arr[i]] = infocus_val[i] + # Per-bin FRACTION in focus: mean of the 0/1 indicator over tissue pixels, on + # the SAME coarse grid as the left panel. A continuous 0..1 field is smooth + # under binning (no per-pixel speckle) -- the whole point of the redesign. + # All-non-tissue bins stay NaN and render transparent (set_bad alpha 0), so the + # DAPI background shows through off-tissue. + step = max(1, -(-max(img_height, img_width) // _FOCUS_HEATMAP_BINS_LONG)) + frac_infocus = _bin_nanmean(infocus_ds, step) + # Continuous red(0 = blurred) -> blue(1 = in focus): the same two colors the + # per-tile ListedColormap used, now as the endpoints of a smooth map. + cmap_focus = LinearSegmentedColormap.from_list("focus_frac", ["red", "blue"]) + cmap_focus.set_bad(alpha=0.0) + ax.imshow( + frac_infocus, + cmap=cmap_focus, + alpha=0.5, + aspect="auto", + extent=[0, img_width, img_height, 0], + origin="upper", + vmin=0, + vmax=1, + interpolation="nearest", + ) + sm = plt.cm.ScalarMappable(cmap=cmap_focus, norm=plt.Normalize(vmin=0, vmax=1)) + sm.set_array([]) + cax = make_axes_locatable(ax).append_axes("right", size="3%", pad=0.05) + cbar = fig.colorbar(sm, cax=cax) + cbar.set_label("In-focus fraction (0 = blurred, 1 = in focus)", fontsize=12) + logging.info( + f" [TIMING] fig5 right bin+imshow (step={step}, " + f"{frac_infocus.shape}): {time.perf_counter() - t_h:.1f}s" + ) + ax.set_title("Focus Classification (2D GMM)", fontsize=14) + else: + # Legacy Rectangle-patch approach + from matplotlib.patches import Rectangle + + for _, roi in df_grid_roi.iterrows(): + x1_ds = roi["x1"] / downsample_factor + x2_ds = roi["x2"] / downsample_factor + y1_ds = roi["y1"] / downsample_factor + y2_ds = roi["y2"] / downsample_factor + w = x2_ds - x1_ds + h = y2_ds - y1_ds + if w <= 0 or h <= 0: + continue + if x1_ds < 0 or y1_ds < 0 or x2_ds > img_width or y2_ds > img_height: + continue + if has_gmm_2d: + color = "blue" if not roi["is_blurred_gmm_2d"] else "red" + else: + color = "blue" if roi["focus_score_norm"] > threshold else "red" + rect = Rectangle( + (x1_ds, y1_ds), w, h, facecolor=color, alpha=0.6, edgecolor="none" + ) + ax.add_patch(rect) + if has_gmm_2d: + ax.set_title( + "Grid Tile Focus Score (2D GMM: Blue=In-Focus, Red=Blurred)", + fontsize=14, + ) + else: + ax.set_title(f"Grid Tile Focus Score (Threshold={threshold})", fontsize=14) + + ax.set_xlim(0, img_width) + ax.set_ylim(img_height, 0) + ax.set_aspect("equal") + ax.axis("off") + + # The 2D-GMM branch gives the right panel its own real colorbar. Only the legacy + # (no-GMM) branch has none, so add a phantom cax there to keep its plotting region + # the same width as the left panel (whose width is shrunk by its colorbar). + if not has_gmm_2d: + cax_r = make_axes_locatable(axes[1]).append_axes("right", size="3%", pad=0.05) + cax_r.axis("off") + + plt.tight_layout() + plt.savefig( + figures_dir / "grid_roi_focus_heatmap.png", dpi=300, bbox_inches="tight" + ) + plt.close(fig) + + # Save data as CSV + if figures_source_dir is not None: + df_grid_roi_scaled = df_grid_roi.copy() + df_grid_roi_scaled["x1_ds"] = df_grid_roi_scaled["x1"] / downsample_factor + df_grid_roi_scaled["x2_ds"] = df_grid_roi_scaled["x2"] / downsample_factor + df_grid_roi_scaled["y1_ds"] = df_grid_roi_scaled["y1"] / downsample_factor + df_grid_roi_scaled["y2_ds"] = df_grid_roi_scaled["y2"] / downsample_factor + df_grid_roi_scaled.to_csv( + figures_source_dir / "grid_roi_focus_heatmap.csv", index=False + ) + + +def plot_snr_roi_heatmap( + df_grid_roi, + small0, + figures_dir, + figures_source_dir, + snr_thresholds=None, +): + """Paint per-tile transcript SNR spatially on the DAPI background. + + Two-panel figure: + Left – neg_pct (fraction of negative-control transcripts per tile) + Right – roi_tx_snr_ratio (real / negative transcript ratio, log scale) + + Colorbars autoscale to the data's p99 (no fixed floor). When provided, + WARN/FAIL threshold lines from `snr_thresholds` are drawn on each + colorbar. Lines outside the autoscaled range are clipped by matplotlib — + on a clean slide where data is far below FAIL, the line simply doesn't + appear, which is the intended visual cue. + + Parameters + ---------- + df_grid_roi : pandas.DataFrame + Must contain columns: x1, x2, y1, y2, neg_pct, roi_tx_snr_ratio, + snr_total_tx. Optionally: snr_real_tx, snr_neg_tx — when present, the + right panel paints (real+1)/(neg+1) (Laplace pseudocount) so tiles + with neg=0 or real=0 (currently NaN/0 under raw ratio) render in + their correct extremes. The canonical roi_tx_snr_ratio column stays + raw — pseudocount applies to *display only*, not to verdicts/JSON. + small0 : numpy.ndarray + Downsampled DAPI image (level 3, 8× downsampling). + figures_dir, figures_source_dir : Path + Output directories. + snr_thresholds : dict, optional + YAML ``snr.roi_tx`` block. Keys: neg_pct_warn, neg_pct_fail, + ratio_warn, ratio_fail. + """ + from matplotlib.colors import LogNorm + + _t = snr_thresholds or {} + + downsample_factor = 8 + img_height, img_width = small0.shape + + # Phase v5: scope to within-tissue tiles with transcripts. Tissue tiles + # WITHOUT transcripts (alveolar / bronchiolar airspaces in lung, etc.) + # stay transparent so the DAPI background still shows through them. + if "tissue_coverage" in df_grid_roi.columns: + df_plot = df_grid_roi[ + (df_grid_roi["snr_total_tx"] > 0) & (df_grid_roi["tissue_coverage"] > 0.5) + ].copy() + else: + df_plot = df_grid_roi[df_grid_roi["snr_total_tx"] > 0].copy() + if df_plot.empty: + logging.warning("No tiles with transcripts — skipping SNR heatmap.") + return + + # Compute figure size from image aspect ratio to avoid empty white space + img_aspect = img_height / img_width if img_width > 0 else 1.0 + panel_width = 6 # width per panel in inches (matches morphology overview) + # Floor at 50% of width so very wide slides (e.g. brain) don't squash titles / colorbars + panel_height = max(panel_width * 0.5, panel_width * img_aspect) + fig, axes = plt.subplots(1, 2, figsize=(2 * panel_width + 2, panel_height)) + + # Pre-compute downsampled tile coordinates (vectorized) + x1_arr = np.clip( + (df_plot["x1"].values / downsample_factor).astype(int), 0, img_width + ) + x2_arr = np.clip( + (df_plot["x2"].values / downsample_factor).astype(int), 0, img_width + ) + y1_arr = np.clip( + (df_plot["y1"].values / downsample_factor).astype(int), 0, img_height + ) + y2_arr = np.clip( + (df_plot["y2"].values / downsample_factor).astype(int), 0, img_height + ) + + def _draw_background(ax): + # _imshow_thumb: downsample the ~86 Mpx DAPI background to the display + # resolution before imshow (explicit extent is preserved, so the panel is + # visually identical). Cuts ~16 s/panel of full-res resampling. + _imshow_thumb( + ax, + small0, + cmap="Greys_r", + vmax=np.percentile(small0, 99), + alpha=0.5, + aspect="auto", + extent=[0, img_width, img_height, 0], + origin="upper", + ) + + def _fill_tile_image(values): + """Build a 2D float32 array with tile values; NaN = transparent.""" + img = np.full((img_height, img_width), np.nan, dtype=np.float32) + for i in range(len(df_plot)): + if x2_arr[i] <= x1_arr[i] or y2_arr[i] <= y1_arr[i]: + continue + val = values[i] + if not np.isfinite(val): + continue + img[y1_arr[i] : y2_arr[i], x1_arr[i] : x2_arr[i]] = val + return img + + def _finish_ax(ax): + ax.set_xlim(0, img_width) + ax.set_ylim(img_height, 0) + ax.set_aspect("equal") + ax.axis("off") + + # ── Panel 1: neg_pct ───────────────────────────────────────────────── + ax = axes[0] + _draw_background(ax) + + # Fixed colorbar (0 → 0.40) — enables cross-sample comparison and ensures + # WARN (0.15) / FAIL (0.30) threshold lines always render in frame. + # extend="max" tags tiles above 0.40 with the deepest red + a triangle marker. + _NEG_VMAX = 0.40 + neg_vals = df_plot["neg_pct"].values + + neg_img = _fill_tile_image(neg_vals) + # _imshow_thumb: NaN-safe block-mean downsample of the full-res tile overlay + # (tiles are large blocks, so the mean is visually identical). ~16 s -> ~0.5 s. + _imshow_thumb( + ax, + neg_img, + cmap="RdYlGn_r", + vmin=0, + vmax=_NEG_VMAX, + alpha=0.6, + aspect="auto", + extent=[0, img_width, img_height, 0], + origin="upper", + interpolation="nearest", + ) + _finish_ax(ax) + ax.set_title("Negative Probe Fraction per Tile", fontsize=14) + + sm = plt.cm.ScalarMappable( + cmap="RdYlGn_r", norm=plt.Normalize(vmin=0, vmax=_NEG_VMAX) + ) + sm.set_array([]) + cbar = plt.colorbar(sm, ax=ax, extend="max") + cbar.set_label("neg_pct (fraction)", fontsize=12) + # Threshold lines (now always in frame thanks to fixed colorbar range) + _neg_warn = _t.get("neg_pct_warn") + _neg_fail = _t.get("neg_pct_fail") + if isinstance(_neg_warn, (int, float)): + cbar.ax.axhline( + y=float(_neg_warn), color="orange", linewidth=1.5, linestyle="--" + ) + if isinstance(_neg_fail, (int, float)): + cbar.ax.axhline(y=float(_neg_fail), color="red", linewidth=1.5) + + # ── Panel 2: roi_tx_snr_ratio (log scale) ─────────────────────────── + ax = axes[1] + _draw_background(ax) + + # Display-only pseudocount: paint (real+1)/(neg+1) instead of raw + # real/neg. The raw ratio NaN's whenever neg=0 (divide-by-zero in + # snr_metrics.py) AND zeros out whenever real=0 — exactly the extreme + # tiles a reader most wants to see. Laplace α=1 regularises both + # extremes onto the colorbar without affecting the canonical + # roi_tx_snr_ratio column used for sample-level verdicts / JSON. + _PSEUDOCOUNT = 1 + if "snr_real_tx" in df_plot.columns and "snr_neg_tx" in df_plot.columns: + ratio_vals = ( + (df_plot["snr_real_tx"].astype(np.float64) + _PSEUDOCOUNT) + / (df_plot["snr_neg_tx"].astype(np.float64) + _PSEUDOCOUNT) + ).values + _ratio_cbar_label = "(real+1) / (neg+1) — pseudocount-smoothed" + _ratio_title_suffix = ", α=1 pseudocount" + else: + ratio_vals = df_plot["roi_tx_snr_ratio"].values + _ratio_cbar_label = "roi_tx_snr_ratio (real / neg)" + _ratio_title_suffix = "" + # Fixed log colorbar (1× → 1000×) — enables cross-sample comparison and + # ensures WARN (30×) / FAIL (10×) threshold lines always render in frame. + # extend="min" tags tiles below 1× (the no-signal regime) with the deepest + # red + a triangle marker. High end stays uncapped visually — very-good + # samples saturate at dark green, which is fine for QC purposes. + _RATIO_VMIN = 1.0 + _RATIO_VMAX = 1000.0 + + # For LogNorm, clamp values to [vmin, vmax] range; ≤0 stays NaN + ratio_img = _fill_tile_image(ratio_vals) + # Replace non-positive finite values with NaN (LogNorm requires > 0) + ratio_img[ratio_img <= 0] = np.nan + # _imshow_thumb: NaN-safe block-mean downsample (visually identical block overlay). + _imshow_thumb( + ax, + ratio_img, + cmap="RdYlGn", + norm=LogNorm(vmin=_RATIO_VMIN, vmax=_RATIO_VMAX), + alpha=0.6, + aspect="auto", + extent=[0, img_width, img_height, 0], + origin="upper", + interpolation="nearest", + ) + _finish_ax(ax) + ax.set_title( + f"Transcript SNR Ratio per Tile (log scale{_ratio_title_suffix})", + fontsize=14, + ) + + sm = plt.cm.ScalarMappable( + cmap="RdYlGn", norm=LogNorm(vmin=_RATIO_VMIN, vmax=_RATIO_VMAX) + ) + sm.set_array([]) + cbar = plt.colorbar(sm, ax=ax, extend="min") + cbar.set_label(_ratio_cbar_label, fontsize=12) + # Threshold lines (always in frame thanks to fixed colorbar range) + _ratio_warn = _t.get("ratio_warn") + _ratio_fail = _t.get("ratio_fail") + if isinstance(_ratio_warn, (int, float)) and float(_ratio_warn) > 0: + cbar.ax.axhline( + y=float(_ratio_warn), color="orange", linewidth=1.5, linestyle="--" + ) + if isinstance(_ratio_fail, (int, float)) and float(_ratio_fail) > 0: + cbar.ax.axhline(y=float(_ratio_fail), color="red", linewidth=1.5) + + plt.tight_layout() + plt.savefig(figures_dir / "snr_heatmap.png", dpi=300, bbox_inches="tight") + plt.close(fig) + + # Save source data + _src_cols = [ + "roi_id", + "x1", + "x2", + "y1", + "y2", + "neg_pct", + "roi_tx_snr_ratio", + "roi_tx_snr_log", + "tissue_coverage", + ] + df_src = df_plot[[c for c in _src_cols if c in df_plot.columns]].copy() + df_src["x1_ds"] = df_src["x1"] / downsample_factor + df_src["x2_ds"] = df_src["x2"] / downsample_factor + df_src["y1_ds"] = df_src["y1"] / downsample_factor + df_src["y2_ds"] = df_src["y2"] / downsample_factor + if figures_source_dir is not None: + df_src.to_csv(figures_source_dir / "snr_heatmap.csv", index=False) + + +def plot_cross_section_concordance( + df_grid_roi, + figures_dir, + figures_source_dir, + snr_thresholds=None, +): + """Cross-section concordance: image quality vs transcript SNR per tile. + + Two-panel scatter showing how different quality lenses relate: + Left – focus_score vs roi_tx_snr_ratio (optical quality → transcript quality) + Right – dapi_intensity vs neg_pct (signal strength → noise contamination) + + Only tissue tiles with transcripts are included. + """ + required = {"focus_score", "roi_tx_snr_ratio", "neg_pct", "dapi_intensity"} + missing = required - set(df_grid_roi.columns) + if missing: + logging.warning( + "Skipping cross-section concordance plot: missing columns %s", missing + ) + return + + # Filter to tissue ROIs with transcripts + df = df_grid_roi.copy() + if "overlaps_tissue" in df.columns: + df = df[df["overlaps_tissue"]] + if "snr_total_tx" in df.columns: + df = df[df["snr_total_tx"] > 0] + mask = ( + df["focus_score"].notna() + & df["roi_tx_snr_ratio"].notna() + & df["neg_pct"].notna() + & df["dapi_intensity"].notna() + ) + df = df[mask] + if len(df) < 10: + logging.warning( + "Skipping cross-section concordance: too few valid tiles (%d)", len(df) + ) + return + + from scipy.stats import spearmanr + + _t = snr_thresholds or {} + + # Three stacked panels, all identical size for visual consistency. + # top = Focus vs TxSNR, middle = Image-SNR (Otsu) vs TxSNR, bottom = + # DAPI vs neg_pct. Each panel 8×4 — matches the §3.2 Focus + # distribution / Focus-vs-DAPI scatter dimensions for visual rhythm + # across §3.x figures. Middle panel falls back to a placeholder when + # per-tile Otsu data is unavailable (legacy samples or upstream SNR + # skipped). + fig, axes = plt.subplots(3, 1, figsize=(6, 11)) + plt.subplots_adjust(hspace=0.35) + + # --- Left panel: focus_score vs roi_tx_snr_ratio --- + ax = axes[0] + focus_vals = df["focus_score"].values + snr_vals = df["roi_tx_snr_ratio"].values + + sidx = _subsample_idx(np.ones(len(df), dtype=bool)) + # Color by GMM 2D classification if available + if "is_blurred_gmm_2d" in df.columns and df["is_blurred_gmm_2d"].notna().any(): + colors = np.where( + df["is_blurred_gmm_2d"].values[sidx].astype(bool), "#E57373", "#64B5F6" + ) + else: + colors = "#64B5F6" + + ax.scatter( + np.log1p(focus_vals[sidx]), + np.log1p(snr_vals[sidx]), + s=8, + alpha=0.4, + c=colors, + edgecolors="none", + rasterized=True, + ) + rho_fs, pval_fs = spearmanr(focus_vals, snr_vals) + ax.set_xlabel("log(1 + Focus Score)", fontsize=12) + ax.set_ylabel("log(1 + Transcript SNR Ratio)", fontsize=12) + ax.set_title("Optical Quality vs Transcript Quality", fontsize=13) + ax.text( + 0.05, + 0.95, + f"Spearman ρ = {rho_fs:.3f}\nn = {len(df):,} tiles", + transform=ax.transAxes, + fontsize=11, + verticalalignment="top", + bbox=dict(boxstyle="round", facecolor="wheat", alpha=0.7), + ) + # Threshold lines + ratio_warn = float(_t.get("ratio_warn", 3.0)) + ratio_fail = float(_t.get("ratio_fail", 1.5)) + ax.axhline( + np.log1p(ratio_warn), + ls="--", + lw=1, + color="orange", + alpha=0.8, + label=f"WARN = {ratio_warn}", + ) + ax.axhline( + np.log1p(ratio_fail), + ls="--", + lw=1, + color="red", + alpha=0.8, + label=f"FAIL = {ratio_fail}", + ) + ax.legend(fontsize=9, loc="lower right") + ax.grid(True, ls="--", alpha=0.3) + + # --- Middle panel: Image SNR (Otsu) per-tile vs roi_tx_snr_ratio (NEW) --- + # Direct test of the section's premise: does per-tile image SNR predict + # per-tile transcript SNR? Otsu-split dB is the per-tile Image-SNR axis + # (also surfaced as a slide-level metric in §3.4 Detailed metrics). + ax = axes[1] + rho_o = float("nan") + n_valid_otsu = 0 + if "snr_image_otsu_db" in df.columns and df["snr_image_otsu_db"].notna().any(): + _otsu_mask = df["snr_image_otsu_db"].notna() & df["roi_tx_snr_ratio"].notna() + n_valid_otsu = int(_otsu_mask.sum()) + if n_valid_otsu >= 10: + otsu_db = df["snr_image_otsu_db"].values + snr_ratio_arr = df["roi_tx_snr_ratio"].values + sidx_o = _subsample_idx(_otsu_mask.values) + if ( + "is_blurred_gmm_2d" in df.columns + and df["is_blurred_gmm_2d"].notna().any() + ): + colors_o = np.where( + df["is_blurred_gmm_2d"].values[sidx_o].astype(bool), + "#E57373", + "#64B5F6", + ) + else: + colors_o = "#64B5F6" + ax.scatter( + otsu_db[sidx_o], + np.log1p(snr_ratio_arr[sidx_o]), + s=8, + alpha=0.4, + c=colors_o, + edgecolors="none", + rasterized=True, + ) + rho_o, _ = spearmanr(otsu_db[_otsu_mask], snr_ratio_arr[_otsu_mask]) + ax.set_xlabel("Image SNR — Otsu split (dB)", fontsize=12) + ax.set_ylabel("log(1 + Transcript SNR Ratio)", fontsize=12) + ax.set_title("Image SNR vs Transcript Quality", fontsize=13) + ax.text( + 0.05, + 0.95, + f"Spearman ρ = {rho_o:.3f}\nn = {n_valid_otsu:,} tiles", + transform=ax.transAxes, + fontsize=11, + verticalalignment="top", + bbox=dict(boxstyle="round", facecolor="wheat", alpha=0.7), + ) + # Reuse Transcript-SNR ratio thresholds from the left panel — + # both panels share the same y-axis metric. + ax.axhline( + np.log1p(ratio_warn), + ls="--", + lw=1, + color="orange", + alpha=0.8, + label=f"WARN = {ratio_warn}", + ) + ax.axhline( + np.log1p(ratio_fail), + ls="--", + lw=1, + color="red", + alpha=0.8, + label=f"FAIL = {ratio_fail}", + ) + ax.legend(fontsize=9, loc="lower right") + ax.grid(True, ls="--", alpha=0.3) + else: + ax.text( + 0.5, + 0.5, + "Image SNR (Otsu) per-tile data\nunavailable (< 10 valid tiles)", + transform=ax.transAxes, + ha="center", + va="center", + fontsize=12, + ) + ax.set_xticks([]) + ax.set_yticks([]) + else: + ax.text( + 0.5, + 0.5, + "Image SNR (Otsu) per-tile data\nunavailable (legacy sample or SNR not computed)", + transform=ax.transAxes, + ha="center", + va="center", + fontsize=12, + ) + ax.set_xticks([]) + ax.set_yticks([]) + + # --- Right panel: dapi_intensity vs neg_pct --- + ax = axes[2] + int_vals = df["dapi_intensity"].values + neg_vals = df["neg_pct"].values + + if "is_blurred_gmm_2d" in df.columns and df["is_blurred_gmm_2d"].notna().any(): + colors_r = np.where( + df["is_blurred_gmm_2d"].values[sidx].astype(bool), "#E57373", "#64B5F6" + ) + else: + colors_r = "#64B5F6" + + ax.scatter( + np.log1p(int_vals[sidx]), + neg_vals[sidx], + s=8, + alpha=0.4, + c=colors_r, + edgecolors="none", + rasterized=True, + ) + rho_in, pval_in = spearmanr(int_vals, neg_vals) + ax.set_xlabel("log(1 + DAPI Intensity)", fontsize=12) + ax.set_ylabel("Negative Probe Fraction", fontsize=12) + ax.set_title("Signal Strength vs Noise Contamination", fontsize=13) + ax.text( + 0.05, + 0.95, + f"Spearman ρ = {rho_in:.3f}\nn = {len(df):,} tiles", + transform=ax.transAxes, + fontsize=11, + verticalalignment="top", + bbox=dict(boxstyle="round", facecolor="wheat", alpha=0.7), + ) + neg_warn = float(_t.get("neg_pct_warn", 0.15)) + neg_fail = float(_t.get("neg_pct_fail", 0.30)) + ax.axhline( + neg_warn, ls="--", lw=1, color="orange", alpha=0.8, label=f"WARN = {neg_warn}" + ) + ax.axhline( + neg_fail, ls="--", lw=1, color="red", alpha=0.8, label=f"FAIL = {neg_fail}" + ) + ax.legend(fontsize=9, loc="lower right") + ax.grid(True, ls="--", alpha=0.3) + + # Add GMM legend if coloured + if "is_blurred_gmm_2d" in df.columns and df["is_blurred_gmm_2d"].notna().any(): + from matplotlib.patches import Patch + + for a in axes: + handles = a.get_legend_handles_labels()[0] + handles.extend( + [ + Patch(facecolor="#64B5F6", label="In-Focus (GMM)"), + Patch(facecolor="#E57373", label="Blurred (GMM)"), + ] + ) + a.legend( + handles=handles, + fontsize=8, + loc="lower right" if a is axes[0] else "upper right", + ) + + plt.tight_layout() + plt.savefig(figures_dir / "cross_section_concordance.png", dpi=200) + plt.savefig( + figures_dir / "cross_section_concordance.pdf", dpi=200, bbox_inches="tight" + ) + plt.close(fig) + + # Source data + src_cols = [ + "roi_id", + "focus_score", + "dapi_intensity", + "roi_tx_snr_ratio", + "neg_pct", + "snr_image_otsu_db", + ] + if "is_blurred_gmm_2d" in df.columns: + src_cols.append("is_blurred_gmm_2d") + if "tissue_coverage" in df.columns: + src_cols.append("tissue_coverage") + if figures_source_dir is not None: + df[[c for c in src_cols if c in df.columns]].to_csv( + figures_source_dir / "cross_section_concordance.csv", index=False + ) + logging.info( + "Cross-section concordance: focus-vs-SNR ρ=%.3f, otsu-vs-SNR ρ=%.3f (n=%d), intensity-vs-neg ρ=%.3f (%d tiles)", + rho_fs, + rho_o, + n_valid_otsu, + rho_in, + len(df), + ) + + +def plot_roi_focus_vs_intensity( + df_grid_roi, figures_dir, figures_source_dir, threshold=-1.0 +): + """ + Create scatter plot showing focus score vs raw DAPI intensity for tile threshold analysis. + Uses log scale for intensity to better visualize the relationship. + + Parameters: + ----------- + df_grid_roi : pandas DataFrame + Grid ROI DataFrame with columns: focus_score, focus_score_norm, raw_intensity (or dapi_intensity) + figures_dir : Path + Directory to save figures + figures_source_dir : Path + Directory to save source data + threshold : float, optional + Threshold for normalized focus score (default: -1.0) + """ + # Use dapi_intensity if available, otherwise raw_intensity + intensity_col = ( + "dapi_intensity" if "dapi_intensity" in df_grid_roi.columns else "raw_intensity" + ) + + # Calculate log10 of intensity (add small epsilon to avoid log(0)) + intensity_values = df_grid_roi[intensity_col].values + log_intensity = np.log10( + intensity_values + 1e-10 + ) # Add small epsilon to handle any zeros + + # Phase v5: robust outlier handling for the rendered scatter. + # (A) Filter to tissue tiles (tissue_coverage > 0.5) — empty / non-tissue + # tiles have intensity ≈ 0 → log10(1e-10) = -10, which dominates the + # auto-scaled xlim. CSV still writes unfiltered data. + # (B) Compute 1st-99th percentile xlim bounds as a robustness floor so a + # single saturated tissue tile (fold, debris) doesn't compress the rest. + if "tissue_coverage" in df_grid_roi.columns: + _tissue_mask = df_grid_roi["tissue_coverage"].values > 0.5 + else: + _tissue_mask = np.ones(len(df_grid_roi), dtype=bool) + + # Single-panel figure: normalized focus score vs log DAPI intensity, + # coloured by GMM 2D blurry/in-focus classification. Figsize matches the + # Focus score distribution histogram (figsize=(8, 4)) below for visual + # rhythm. Previously this was a 2-panel figure where the left panel + # showed raw CCFS vs intensity coloured by *normalised* focus score — + # redundant with the right panel since the y-axis there directly + # encodes the same metric. + fig, ax = plt.subplots(1, 1, figsize=(7, 4)) + # Filter out invalid values + valid_mask_plot2 = ( + np.isfinite(log_intensity) + & np.isfinite(df_grid_roi["focus_score_norm"].values) + & _tissue_mask + ) + + if valid_mask_plot2.sum() == 0: + logging.warning(" Warning: No valid data points for plot 2") + ax.text( + 0.5, + 0.5, + "No valid data", + transform=ax.transAxes, + ha="center", + va="center", + fontsize=14, + ) + else: + # Color by GMM 2D classification if available, otherwise fall back to threshold + has_gmm_2d = "is_blurred_gmm_2d" in df_grid_roi.columns + + if has_gmm_2d: + # Color by GMM 2D classification + valid_gmm_mask = valid_mask_plot2 & df_grid_roi["is_blurred_gmm_2d"].notna() + if valid_gmm_mask.sum() > 0: + sidx = _subsample_idx(valid_gmm_mask) + is_blurred_sub = ( + df_grid_roi["is_blurred_gmm_2d"].values[sidx].astype(bool) + ) + colors = np.where(is_blurred_sub, "red", "blue") + ax.scatter( + log_intensity[sidx], + df_grid_roi["focus_score_norm"].values[sidx], + alpha=0.5, + s=10, + c=colors, + edgecolors="none", + rasterized=True, + ) + + # Phase v5: percentile-clipped xlim (see comment at top of function). + _x_lo, _x_hi = np.nanpercentile(log_intensity[valid_gmm_mask], [1, 99]) + ax.set_xlim(_x_lo - 0.1, _x_hi + 0.1) + ax.set_ylim( + df_grid_roi.loc[valid_gmm_mask, "focus_score_norm"].min() - 0.5, + df_grid_roi.loc[valid_gmm_mask, "focus_score_norm"].max() + 0.5, + ) + + # Add text annotation for GMM 2D (counts from full data, not subsample) + n_blurred = df_grid_roi.loc[valid_gmm_mask, "is_blurred_gmm_2d"].sum() + n_in_focus = ( + ~df_grid_roi.loc[valid_gmm_mask, "is_blurred_gmm_2d"] + ).sum() + pct_blurred = ( + n_blurred / valid_gmm_mask.sum() * 100 + if valid_gmm_mask.sum() > 0 + else 0 + ) + ax.text( + 0.05, + 0.95, + f"2D GMM Classification\nRed: Blurred ({n_blurred:,}, {pct_blurred:.1f}%)\nBlue: In-Focus ({n_in_focus:,}, {100 - pct_blurred:.1f}%)", + transform=ax.transAxes, + fontsize=11, + verticalalignment="top", + bbox=dict(boxstyle="round", facecolor="wheat", alpha=0.7), + ) + ax.set_title( + "Normalized Focus Score vs Log10(Raw DAPI Intensity) (2D GMM Classification)", + fontsize=12, + ) + else: + # Fallback to threshold method + sidx = _subsample_idx(valid_mask_plot2) + scores_sub = df_grid_roi["focus_score_norm"].values[sidx] + colors = np.where(scores_sub <= threshold, "red", "blue") + ax.scatter( + log_intensity[sidx], + scores_sub, + alpha=0.5, + s=10, + c=colors, + edgecolors="none", + rasterized=True, + ) + + # Phase v5: percentile-clipped xlim (see comment at top of function). + _x_lo, _x_hi = np.nanpercentile(log_intensity[valid_mask_plot2], [1, 99]) + ax.set_xlim(_x_lo - 0.1, _x_hi + 0.1) + ax.set_ylim( + df_grid_roi.loc[valid_mask_plot2, "focus_score_norm"].min() - 0.5, + df_grid_roi.loc[valid_mask_plot2, "focus_score_norm"].max() + 0.5, + ) + + # Add threshold line + ax.axhline( + y=threshold, + color="black", + linestyle="--", + linewidth=2, + label=f"Threshold ({threshold})", + ) + + # Add text annotation for threshold + ax.text( + 0.05, + 0.95, + f"Threshold: {threshold}\nRed: Blurred (≤{threshold})\nBlue: In-Focus (>{threshold})", + transform=ax.transAxes, + fontsize=11, + verticalalignment="top", + bbox=dict(boxstyle="round", facecolor="wheat", alpha=0.7), + ) + ax.set_title( + "Normalized Focus Score vs Log10(Raw DAPI Intensity) (Thresholded)", + fontsize=12, + ) + + ax.set_xlabel("log₁₀(Raw DAPI Intensity, 16-bit counts)", fontsize=12) + ax.set_ylabel("Normalised focus score", fontsize=12) + ax.grid(True, axis="y", linestyle="--", alpha=0.7) + ax.legend(fontsize=10) + + # Pin axes-box position so the rendered plot box matches the §3.2 Focus + # score distribution figure exactly. Both figures use figsize=(8, 4) and + # the same subplots_adjust margins; identical (left, right, top, bottom) + # ⇒ identical plot-box position regardless of y-tick label width. + plt.subplots_adjust(left=0.13, right=0.95, top=0.88, bottom=0.18) + plt.savefig( + figures_dir / "roi_focus_vs_intensity.pdf", dpi=300, bbox_inches="tight" + ) + plt.savefig(figures_dir / "roi_focus_vs_intensity.png", dpi=300) + plt.close(fig) + + # Save data as CSV (include both raw and log intensity) + df_scatter = df_grid_roi[ + ["focus_score", "focus_score_norm", intensity_col, "tissue_coverage"] + ].copy() + df_scatter["log10_intensity"] = log_intensity + df_scatter["is_low_nuclear_texture"] = df_scatter["focus_score_norm"] <= threshold + if figures_source_dir is not None: + df_scatter.to_csv( + figures_source_dir / "roi_focus_vs_intensity.csv", index=False + ) + + # Print correlation statistics (using log intensity) + correlation = np.corrcoef(log_intensity, df_grid_roi["focus_score_norm"])[0, 1] + logging.info( + f" Correlation (log10(intensity) vs normalized focus score): {correlation:.4f}" + ) + + +def plot_roi_focus_distribution( + df_grid_roi, figures_dir, figures_source_dir, threshold=-1.0 +): + """Histogram of normalized focus scores on tissue-filtered tiles, with + GMM 2D binary classification overlaid (red = blurred, blue = in-focus). + + Phase v5 (2026-05-15): simplified from the previous dual-panel (raw + + normalized) × dual-variant (all-tiles + tissue-filtered) layout. The + all-tiles view was misleading — background tiles are force-classified + blurry by the intensity-floor rule, so the all-tiles histogram showed + a "blurry mass" that was not actually a GMM decision. The raw-score + panel was redundant with the normalized panel for QC interpretation. + Single panel: tissue-filtered, normalized scores, GMM-colored. + + Emits roi_focus_distribution_tissue.png — only when tissue_coverage + column is present with at least one tile above the 0.5 threshold. + + Parameters: + ----------- + df_grid_roi : pandas DataFrame + Grid ROI DataFrame with columns: focus_score, focus_score_norm, + is_blurred_gmm_2d, tissue_coverage (required). + figures_dir : Path + Directory to save figures. + figures_source_dir : Path + Directory to save source data. + threshold : float, optional + Normalized-score fallback threshold (default: -1.0). Used only when + is_blurred_gmm_2d column is absent. + """ + if "tissue_coverage" not in df_grid_roi.columns: + logging.info(" No tissue_coverage column; skipping focus score distribution.") + return + + df_in = df_grid_roi[df_grid_roi["tissue_coverage"] > 0.5] + if len(df_in) == 0: + logging.info( + " No tissue tiles (tissue_coverage > 0.5); skipping focus score distribution." + ) + return + + save_stem = "roi_focus_distribution_tissue" + # figsize chosen to be more landscape-y than the previous (10, 6) so the + # histogram doesn't dominate the §3.2 visual flow against the adjacent + # spatial focus heatmap. Width reduced ~20%, height reduced ~33%. + fig, ax = plt.subplots(figsize=(7, 4)) + + has_gmm_2d = "is_blurred_gmm_2d" in df_in.columns + if has_gmm_2d: + blurred = df_in[df_in["is_blurred_gmm_2d"]] + in_focus = df_in[~df_in["is_blurred_gmm_2d"]] + title_suffix = "2D GMM Classification" + else: + blurred = df_in[df_in["focus_score_norm"] <= threshold] + in_focus = df_in[df_in["focus_score_norm"] > threshold] + title_suffix = f"Threshold: {threshold}" + + ax.hist( + [blurred["focus_score_norm"], in_focus["focus_score_norm"]], + bins=50, + alpha=0.7, + edgecolor="black", + color=["red", "blue"], + label=["Blurred", "In-Focus"], + stacked=False, + ) + + if not has_gmm_2d: + ax.axvline( + x=threshold, + color="black", + linestyle="--", + linewidth=2, + label=f"Threshold ({threshold})", + ) + + ax.set_xlabel("Focus Score (Normalized)", fontsize=12) + ax.set_ylabel("Number of tiles", fontsize=12) + ax.set_title( + f"Distribution of Normalized Focus Scores ({title_suffix})", + fontsize=12, + ) + ax.legend(fontsize=10) + ax.grid(True, axis="y", linestyle="--", alpha=0.7) + + mean_norm = df_in["focus_score_norm"].mean() + median_norm = df_in["focus_score_norm"].median() + pct_blurred = len(blurred) / len(df_in) * 100 if len(df_in) > 0 else 0 + ax.text( + 0.95, + 0.95, + f"Mean: {mean_norm:.4f}\nMedian: {median_norm:.4f}\nBlurred: {pct_blurred:.1f}%\nn = {len(df_in):,} tiles", + transform=ax.transAxes, + verticalalignment="top", + horizontalalignment="right", + bbox=dict(boxstyle="round", facecolor="wheat", alpha=0.7), + fontsize=10, + ) + + # Pin axes-box position so the rendered plot box matches the §3.2 Focus + # score vs DAPI intensity figure exactly (same figsize, same margins). + plt.subplots_adjust(left=0.13, right=0.95, top=0.88, bottom=0.18) + plt.savefig(figures_dir / f"{save_stem}.pdf", dpi=300, bbox_inches="tight") + plt.savefig(figures_dir / f"{save_stem}.png", dpi=300) + plt.close(fig) + + # Source CSV mirrors the rendered subset. + df_dist = df_in[["focus_score", "focus_score_norm"]].copy() + df_dist["is_low_nuclear_texture"] = df_dist["focus_score_norm"] <= threshold + if figures_source_dir is not None: + df_dist.to_csv(figures_source_dir / f"{save_stem}.csv", index=False) + + logging.info(" Focus score distribution summary (tissue tiles):") + logging.info( + f" Normalized focus score - Mean: {mean_norm:.4f}, Median: {median_norm:.4f}" + ) + logging.info(f" Tiles blurred: {len(blurred)} ({pct_blurred:.1f}%)") + logging.info(f" Tiles in-focus: {len(in_focus)} ({100 - pct_blurred:.1f}%)") + + +def calculate_roi_intensities(xoa_morphology_files, df_grid_roi): + """ + Calculate mean intensity per tile for Boundary and IntRNA channels. + + Note: If intensities are already calculated in calculate_roi_focusscore() (new behavior), + this function will detect that and return the DataFrame unchanged. + + Parameters: + ----------- + xoa_morphology_files : list + List of paths to morphology image files + df_grid_roi : pandas DataFrame + Grid ROI DataFrame with columns: roi_id, x1, x2, y1, y2, raw_intensity (DAPI) + If dapi_intensity, boundary_intensity, intrna_intensity already exist, returns unchanged. + + Returns: + -------- + pandas DataFrame + DataFrame with columns: + - dapi_intensity: Mean DAPI intensity per tile + - boundary_intensity: Mean boundary intensity per tile (if available) + - intrna_intensity: Mean IntRNA intensity per tile (if available) + """ + + # Check if intensities are already calculated (new behavior in calculate_roi_focusscore) + # Note: boundary_intensity may be NaN if only DAPI channel is available, but column should exist + if "dapi_intensity" in df_grid_roi.columns: + # Intensities already calculated, just return the DataFrame + logging.info( + " Intensities already calculated in calculate_roi_focusscore(), skipping recalculation" + ) + return df_grid_roi.copy() + + # Legacy behavior: calculate intensities if not already present + # Load channels robustly from multi-channel stack or split channel files. + dapi_image, boundary_image, intrna_image = _load_morphology_channels( + xoa_morphology_files, level=0 + ) + has_boundary = boundary_image is not None + has_intrna = intrna_image is not None + + # Create output DataFrame + df_roi_intensities = df_grid_roi.copy() + if "dapi_intensity" not in df_roi_intensities.columns: + df_roi_intensities["dapi_intensity"] = df_roi_intensities.get( + "raw_intensity", np.nan + ) # Rename for consistency + + # Calculate intensities for each ROI + n_rois = len(df_grid_roi) + + # Pre-extract ROI coordinate arrays for vectorized access + x1_arr = df_grid_roi["x1"].values.astype(int) + x2_arr = df_grid_roi["x2"].values.astype(int) + y1_arr = df_grid_roi["y1"].values.astype(int) + y2_arr = df_grid_roi["y2"].values.astype(int) + + # Boundary intensity + if has_boundary: + boundary_intensities = np.empty(n_rois, dtype=np.float64) + for i in range(n_rois): + boundary_intensities[i] = np.mean( + boundary_image[y1_arr[i] : y2_arr[i], x1_arr[i] : x2_arr[i]] + ) + df_roi_intensities["boundary_intensity"] = boundary_intensities + del boundary_image + else: + df_roi_intensities["boundary_intensity"] = np.nan + logging.warning( + "Warning: Boundary channel not available, setting boundary_intensity to NaN" + ) + + # IntRNA intensity + if has_intrna: + intrna_intensities = np.empty(n_rois, dtype=np.float64) + for i in range(n_rois): + intrna_intensities[i] = np.mean( + intrna_image[y1_arr[i] : y2_arr[i], x1_arr[i] : x2_arr[i]] + ) + df_roi_intensities["intrna_intensity"] = intrna_intensities + else: + df_roi_intensities["intrna_intensity"] = np.nan + logging.warning( + "Warning: IntRNA channel not available, setting intrna_intensity to NaN" + ) + + # Clean up + del dapi_image + + return df_roi_intensities + + +def assess_raw_intensity_quality( + df_roi_intensities, + dapi_threshold_critical=500, + boundary_threshold_critical=100, + intrna_threshold_critical=300, + min_tissue_coverage=ROI_MIN_TISSUE_COVERAGE_FOR_INTENSITY_QC, + channel_pct_thresholds=None, +): + """ + Assess raw intensity quality from per-tile intensities. + Handles missing channels gracefully (NaN values). + + Note: This function can apply an additional tissue coverage filter before + intensity QC. If ``tissue_coverage`` is present, only tiles with + ``tissue_coverage >= min_tissue_coverage`` are used for threshold-based + percentage calculations. This reduces false FAILs from low-content/background + edge tiles that technically overlap tissue but are mostly non-tissue. + + Threshold Method: + ----------------- + Each channel has a single absolute minimum intensity threshold (``intensity_critical`` + from YAML). The metric is the fraction of tissue tiles whose mean intensity falls + below that threshold: ``pct_tissue_roi_below_critical``. + + Critical thresholds (defaults): + - DAPI: 500 — ~0.76% of 16-bit max; typical well-stained range 2000-8000 + - Boundary: 100 — membrane markers 5-10× lower than DAPI + - IntRNA: 300 — rRNA markers 2-3× lower than DAPI + + Quality Status Determination (WARN-only since 2026-06-22): + ---------------------------------------------------------- + Per-channel verdict is driven by ``pct_tissue_roi_below_critical`` compared + against the YAML ``intensity_warn`` fraction (converted to %). There is no + FAIL tier — intensity does not track quality after XOA 4.0, so a dim sample + is flagged for review, never hard-failed on intensity alone: + - **warn**: pct_tissue_roi_below_critical > warn% OR mean < critical_threshold + - **pass**: otherwise + The ``critical_threshold`` is XOA-version-specific (selected by the caller). + + Parameters: + ----------- + df_roi_intensities : pandas DataFrame + DataFrame with per-tile intensities (dapi_intensity, boundary_intensity, intrna_intensity). + dapi_threshold_critical : float, optional + Critical DAPI threshold (default: 500). + boundary_threshold_critical : float, optional + Critical boundary threshold (default: 100). + intrna_threshold_critical : float, optional + Critical IntRNA threshold (default: 300). + + Returns: + -------- + dict + Per-channel dict with: + - mean, median, p10, p25, p75, p90: Intensity statistics + - critical_threshold: Absolute minimum intensity threshold + - n_tissue_rois_below_critical: Count of tissue tiles below threshold + - pct_tissue_roi_below_critical: Percentage of tissue tiles below threshold + - pct_warn_threshold, pct_fail_threshold: YAML-derived population thresholds (%) + - quality_status: 'pass', 'warn', 'fail', or 'not_available' + """ + stats = {} + _cpct = channel_pct_thresholds or {} + df_work = df_roi_intensities + n_rois_input = int(len(df_roi_intensities)) + if "tissue_coverage" in df_roi_intensities.columns: + df_work = df_roi_intensities[ + df_roi_intensities["tissue_coverage"] >= float(min_tissue_coverage) + ] + n_rois_used = int(len(df_work)) + stats["_intensity_qc_scope"] = { + "n_rois_input": n_rois_input, + "n_rois_used": n_rois_used, + "min_tissue_coverage": float(min_tissue_coverage), + "uses_tissue_coverage_filter": "tissue_coverage" in df_roi_intensities.columns, + } + + for channel, critical_threshold in [ + ("dapi", dapi_threshold_critical), + ("boundary", boundary_threshold_critical), + ("intrna", intrna_threshold_critical), + ]: + # Per-channel WARN fraction from YAML (or default). WARN-only since + # 2026-06-22 (qc_drift_analysis): intensity does not track quality + # post-XOA-4.0, so there is no FAIL tier — `intensity_fail` is no longer + # read or applied. + # Case-insensitive lookup (the YAML uses DAPI/boundary/intRNA here and + # dapi/boundary/intrna under snr:) — the old hand-written casing map + # could drift and silently substitute the 0.15 default for the YAML's + # value. 0.15 now applies only when no YAML was supplied at all. + _ch = _yaml_channel_cfg(_cpct, channel) + pct_warn_frac = float(_ch.get("intensity_warn", 0.15)) + col_name = f"{channel}_intensity" + intensities = df_work[col_name].values + + # Check if channel is available (not all NaN) + if np.all(np.isnan(intensities)): + stats[channel] = { + "mean": np.nan, + "median": np.nan, + "p10": np.nan, + "p25": np.nan, + "p75": np.nan, + "p90": np.nan, + "critical_threshold": float(critical_threshold), + "n_tissue_rois_below_critical": 0, + "pct_tissue_roi_below_critical": 0.0, + "quality_status": "not_available", + } + continue + + # Filter out NaN values for calculations + intensities_valid = intensities[~np.isnan(intensities)] + + if len(intensities_valid) == 0: + stats[channel] = { + "mean": np.nan, + "median": np.nan, + "p10": np.nan, + "p25": np.nan, + "p75": np.nan, + "p90": np.nan, + "critical_threshold": float(critical_threshold), + "n_tissue_rois_below_critical": 0, + "pct_tissue_roi_below_critical": 0.0, + "quality_status": "not_available", + } + continue + + # Calculate statistics (using valid values only) + mean_int = np.mean(intensities_valid) + median_int = np.median(intensities_valid) + p10 = np.percentile(intensities_valid, 10) + p25 = np.percentile(intensities_valid, 25) + p75 = np.percentile(intensities_valid, 75) + p90 = np.percentile(intensities_valid, 90) + + # Count tissue tiles below the single critical intensity threshold + n_below = int(np.sum(intensities_valid < critical_threshold)) + pct_tissue_roi_below_critical = (n_below / len(intensities_valid)) * 100 + + # Determine quality status — WARN-only (no FAIL tier). Both the + # prevalence case (too many tiles below the floor) and the mean-below-floor + # case cap at WARN: a dim sample is flagged for review, never hard-failed + # on intensity alone. + pct_warn = pct_warn_frac * 100.0 + if pct_tissue_roi_below_critical > pct_warn or mean_int < critical_threshold: + quality_status = "warn" + else: + quality_status = "pass" + + stats[channel] = { + "mean": float(mean_int), + "median": float(median_int), + "p10": float(p10), + "p25": float(p25), + "p75": float(p75), + "p90": float(p90), + "critical_threshold": float(critical_threshold), + "pct_warn_threshold": float(pct_warn), + # None signals "advisory / no FAIL tier" to the report renderer. + "pct_fail_threshold": None, + "n_tissue_rois_below_critical": n_below, + "pct_tissue_roi_below_critical": float(pct_tissue_roi_below_critical), + "quality_status": quality_status, + } + + # Overall quality — WARN-only (intensity never fails the sample on its own). + statuses = [stats[ch]["quality_status"] for ch in ["dapi", "boundary", "intrna"]] + if n_rois_used == 0: + overall_quality = "not_available" + elif "warn" in statuses: + overall_quality = "warn" + else: + overall_quality = "pass" + + stats["overall_quality"] = overall_quality + + return stats + + +def plot_focus_score_vs_laplacian(df_grid_roi, figures_dir, figures_source_dir): + """ + Compare Laplacian variance vs original focus score (std²/mean). + + Creates scatter plot and distribution histograms comparing the two focus metrics. + + Parameters: + ----------- + df_grid_roi : pandas DataFrame + Grid ROI DataFrame with columns: dapi_focus_score, dapi_lap_var + figures_dir : Path + Directory to save figures + figures_source_dir : Path + Directory to save source data + """ + # Check if required columns exist + if "dapi_lap_var" not in df_grid_roi.columns: + logging.warning( + " Warning: dapi_lap_var column not found, skipping Laplacian comparison plots" + ) + return + + if "dapi_focus_score" not in df_grid_roi.columns: + # Fallback to generic focus_score + if "focus_score" not in df_grid_roi.columns: + logging.warning( + " Warning: No focus score column found, skipping Laplacian comparison plots" + ) + return + focus_col = "focus_score" + else: + focus_col = "dapi_focus_score" + + # Filter out NaN values + valid_mask = df_grid_roi["dapi_lap_var"].notna() & df_grid_roi[focus_col].notna() + df_valid = df_grid_roi[valid_mask].copy() + + if len(df_valid) == 0: + logging.warning( + " Warning: No valid data for Laplacian comparison, skipping plots" + ) + return + + # Scatter plot: log1p(focus_score) vs log1p(laplacian_variance) + logging.info("Generating Figure: Focus score vs Laplacian variance comparison...") + fig, ax = plt.subplots(1, 1, figsize=(6, 5)) + + # Color by GMM 2D classification if available + has_gmm_2d = "is_blurred_gmm_2d" in df_valid.columns + focus_vals = df_valid[focus_col].values + lap_vals = df_valid["dapi_lap_var"].values + if has_gmm_2d: + # Filter to valid mask for GMM 2D + valid_gmm_mask = df_valid["is_blurred_gmm_2d"].notna().values + if valid_gmm_mask.sum() > 0: + sidx = _subsample_idx(valid_gmm_mask) + is_blurred_sub = df_valid["is_blurred_gmm_2d"].values[sidx].astype(bool) + colors = np.where(is_blurred_sub, "red", "blue") + ax.scatter( + np.log1p(focus_vals[sidx]), + np.log1p(lap_vals[sidx]), + s=5, + alpha=0.3, + c=colors, + rasterized=True, + ) + from matplotlib.patches import Patch + + legend_elements = [ + Patch(facecolor="blue", label="In-Focus (2D GMM)"), + Patch(facecolor="red", label="Blurred (2D GMM)"), + ] + ax.legend(handles=legend_elements, fontsize=10) + else: + sidx = _subsample_idx(np.ones(len(df_valid), dtype=bool)) + ax.scatter( + np.log1p(focus_vals[sidx]), + np.log1p(lap_vals[sidx]), + s=5, + alpha=0.3, + rasterized=True, + ) + else: + sidx = _subsample_idx(np.ones(len(df_valid), dtype=bool)) + ax.scatter( + np.log1p(focus_vals[sidx]), + np.log1p(lap_vals[sidx]), + s=5, + alpha=0.3, + rasterized=True, + ) + + ax.set_xlabel("log(1 + std²/mean) (DAPI)", fontsize=12) + ax.set_ylabel("log(1 + Laplacian variance) (DAPI)", fontsize=12) + if has_gmm_2d: + ax.set_title( + "Focus Score vs Laplacian Variance (2D GMM Classification)", fontsize=14 + ) + else: + ax.set_title("Focus Score vs Laplacian Variance", fontsize=14) + ax.grid(True, alpha=0.3) + + plt.tight_layout() + # Save to methodology assessment folder (subfolder of figures_dir) + methodology_figures_dir = figures_dir / "figures_methodology_assessment" + methodology_figures_dir.mkdir(parents=True, exist_ok=True) + plt.savefig( + methodology_figures_dir / "focus_score_vs_laplacian.pdf", + dpi=300, + bbox_inches="tight", + ) + plt.savefig( + methodology_figures_dir / "focus_score_vs_laplacian.png", + dpi=300, + bbox_inches="tight", + ) + plt.close(fig) + + # Distribution comparison + fig, axes = plt.subplots(1, 2, figsize=(10, 4)) + + axes[0].hist(df_valid[focus_col], bins=50, alpha=0.7, edgecolor="black") + axes[0].set_title("std²/mean (DAPI Focus Score)", fontsize=12) + axes[0].set_xlabel("Focus Score (std²/mean)", fontsize=11) + axes[0].set_ylabel("Number of tiles", fontsize=11) + axes[0].grid(True, alpha=0.3, axis="y") + + axes[1].hist(df_valid["dapi_lap_var"], bins=50, alpha=0.7, edgecolor="black") + axes[1].set_title("Laplacian Variance (DAPI)", fontsize=12) + axes[1].set_xlabel("Laplacian Variance", fontsize=11) + axes[1].set_ylabel("Number of tiles", fontsize=11) + axes[1].grid(True, alpha=0.3, axis="y") + + plt.tight_layout() + # Save to methodology assessment folder + plt.savefig( + methodology_figures_dir / "focus_score_vs_laplacian_distributions.pdf", + dpi=300, + bbox_inches="tight", + ) + plt.savefig( + methodology_figures_dir / "focus_score_vs_laplacian_distributions.png", + dpi=300, + bbox_inches="tight", + ) + plt.close(fig) + + # Save source data + df_comparison = df_valid[[focus_col, "dapi_lap_var"]].copy() + df_comparison["log1p_focus_score"] = np.log1p(df_comparison[focus_col]) + df_comparison["log1p_lap_var"] = np.log1p(df_comparison["dapi_lap_var"]) + if figures_source_dir is not None: + df_comparison.to_csv( + figures_source_dir / "focus_score_vs_laplacian.csv", index=False + ) + + # Calculate correlation + correlation = np.corrcoef( + np.log1p(df_valid[focus_col]), np.log1p(df_valid["dapi_lap_var"]) + )[0, 1] + logging.info(f" Correlation (log1p): {correlation:.4f}") + + +def plot_intensity_assessment( + df_roi_intensities, intensity_stats, small0, figures_dir, figures_source_dir +): + """ + Create visualization of per-tile intensity distributions and spatial heatmaps. + + Note: Intensity distributions are calculated from tissue-filtered tiles only + (tiles with any tissue overlap, i.e., coverage > 0%). Background/empty regions + of the image are excluded from the distributions. + + Parameters: + ----------- + df_roi_intensities : pandas DataFrame + DataFrame with per-tile intensities and coordinates (tissue-filtered tiles only) + intensity_stats : dict + Dictionary from assess_raw_intensity_quality() + small0 : numpy.ndarray + Downsampled DAPI image (for spatial overlay) + figures_dir : Path + Directory to save figures + figures_source_dir : Path + Directory to save source data + """ + downsample_factor = 8 + fig, axes = plt.subplots(2, 3, figsize=(15, 9)) + + # Calculate ROI centers + df_roi_intensities["center_x"] = ( + df_roi_intensities["x1"] + df_roi_intensities["x2"] + ) / 2 + df_roi_intensities["center_y"] = ( + df_roi_intensities["y1"] + df_roi_intensities["y2"] + ) / 2 + + channels = [ + ("dapi", "DAPI", intensity_stats["dapi"]), + ("boundary", "Boundary", intensity_stats["boundary"]), + ("intrna", "IntRNA", intensity_stats["intrna"]), + ] + + # p99 of the DAPI background is the same for all three spatial panels; + # compute it once over the full ~86 Mpx small0 instead of once per channel + # inside the loop. PIXEL-IDENTICAL. + _small0_vmax = np.percentile(small0, 99) + + for col_idx, (channel, channel_name, stats) in enumerate(channels): + intensity_col = f"{channel}_intensity" + intensities = df_roi_intensities[intensity_col].values + + # Check if channel is available + is_available = not np.all(np.isnan(intensities)) + intensities_valid = ( + intensities[~np.isnan(intensities)] if is_available else np.array([]) + ) + + # Row 1: Distribution plots + ax = axes[0, col_idx] + + if is_available and len(intensities_valid) > 0: + # Phase v5: clip the histogram x-range to the 99.9th percentile + # to avoid a single saturated/outlier tile compressing the bulk + # distribution. Threshold lines (intensity_critical) and stats + # text (mean/median/p10/p90) are pulled from the FULL-data + # `stats` dict and are unchanged by this clip. + _p_hi_intensity = np.nanpercentile(intensities_valid, 99.9) + _clipped_intensity = intensities_valid[intensities_valid <= _p_hi_intensity] + ax.hist(_clipped_intensity, bins=50, alpha=0.7, edgecolor="black") + + # Add threshold lines (only if thresholds are valid) + if not np.isnan(stats["critical_threshold"]): + ax.axvline( + stats["critical_threshold"], + color="red", + linestyle="--", + linewidth=2, + label=f"Min intensity ({stats['critical_threshold']:.0f})", + ) + + # No green "optimal range" band — only the red critical threshold + # line (from YAML intensity_critical) is shown to avoid implying an + # uncalibrated optimal window. + + # Add statistics text + if not np.isnan(stats["mean"]): + stats_text = f"Mean: {stats['mean']:.0f}\nMedian: {stats['median']:.0f}\nP10: {stats['p10']:.0f}\nP90: {stats['p90']:.0f}" + ax.text( + 0.98, + 0.98, + stats_text, + transform=ax.transAxes, + verticalalignment="top", + horizontalalignment="right", + bbox=dict(boxstyle="round", facecolor="wheat", alpha=0.5), + fontsize=9, + ) + else: + ax.text( + 0.5, + 0.5, + f"{channel_name} channel\nnot available", + transform=ax.transAxes, + ha="center", + va="center", + fontsize=14, + bbox=dict(boxstyle="round", facecolor="lightgray", alpha=0.5), + ) + + ax.set_xlabel(f"{channel_name} Intensity", fontsize=12) + ax.set_ylabel("Number of tiles", fontsize=12) + status_text = ( + stats["quality_status"].upper() + if stats["quality_status"] != "not_available" + else "NOT AVAILABLE" + ) + ax.set_title( + f"{channel_name} Intensity Distribution\n(Status: {status_text})", + fontsize=14, + ) + if is_available and len(intensities_valid) > 0: + ax.legend(fontsize=10) + ax.grid(True, alpha=0.3) + + # Row 2: Spatial heatmaps + ax = axes[1, col_idx] + + # Get image dimensions for setting axes limits + img_height, img_width = small0.shape + + # Show downsampled DAPI as background (optional, light) + _imshow_thumb( + ax, + small0, + cmap="Greys_r", + vmax=_small0_vmax, + alpha=0.3, + aspect="auto", + origin="upper", + extent=[0, img_width, img_height, 0], + ) + + # Scale coordinates to downsampled space + x_scaled = df_roi_intensities["center_x"] / downsample_factor + y_scaled = df_roi_intensities["center_y"] / downsample_factor + + if is_available and len(intensities_valid) > 0: + # Create scatter plot heatmap (only for valid intensities) + valid_mask = ~np.isnan(intensities) + + # Filter coordinates to be within image bounds + x_vals = x_scaled.values if hasattr(x_scaled, "values") else x_scaled + y_vals = y_scaled.values if hasattr(y_scaled, "values") else y_scaled + in_bounds = ( + (x_vals >= 0) + & (x_vals < img_width) + & (y_vals >= 0) + & (y_vals < img_height) + ) + valid_mask = valid_mask & in_bounds + + if valid_mask.sum() > 0: + # Phase v6: fixed per-channel colorbar caps (cross-sample + # comparable). Calibrated against 9 tissues — see the + # _INTENSITY_DISPLAY_CAP module constant. vmin=0 anchors the + # dark end to absolute zero so dim samples render dim and + # bright samples fill the range; mirrors the fixed-vmin/vmax + # convention already used by plot_snr_roi_heatmap. + _vmax_intensity = _INTENSITY_DISPLAY_CAP[channel] + # Use hexbin for spatial heatmap — bins ALL data, renders O(bins) not O(N) + hb = ax.hexbin( + x_vals[valid_mask], + y_vals[valid_mask], + C=intensities[valid_mask], + reduce_C_function=np.mean, + gridsize=100, + cmap="viridis", + mincnt=1, + vmin=0, + vmax=_vmax_intensity, + # rasterized: the PDF embeds a raster of the hex grid instead of + # ~30-60k vector polygons per panel. Pixel-identical at dpi 300, + # but avoids re-rendering the vector layer for BOTH .pdf and .png. + rasterized=True, + ) + + # extend="max" — upper triangle marks hex cells whose mean + # exceeds the cap (bright tissues will saturate routinely). + cbar = plt.colorbar(hb, ax=ax, extend="max") + cbar.set_label(f"{channel_name} Intensity", fontsize=10) + else: + ax.text( + 0.5, + 0.5, + "No valid data points\nwithin image bounds", + transform=ax.transAxes, + ha="center", + va="center", + fontsize=12, + bbox=dict(boxstyle="round", facecolor="lightgray", alpha=0.5), + ) + else: + ax.text( + 0.5, + 0.5, + f"{channel_name} channel\nnot available", + transform=ax.transAxes, + ha="center", + va="center", + fontsize=14, + bbox=dict(boxstyle="round", facecolor="lightgray", alpha=0.5), + ) + + # Set axes limits explicitly + ax.set_xlim(0, img_width) + ax.set_ylim(img_height, 0) # Reversed because origin='upper' + ax.set_xlabel("X coordinate (downsampled)", fontsize=12) + ax.set_ylabel("Y coordinate (downsampled)", fontsize=12) + ax.set_title(f"{channel_name} Intensity Spatial Heatmap", fontsize=14) + ax.set_aspect("equal") + + plt.tight_layout() + plt.savefig(figures_dir / "intensity_assessment.pdf", dpi=300, bbox_inches="tight") + plt.savefig(figures_dir / "intensity_assessment.png", dpi=300, bbox_inches="tight") + plt.close(fig) + + # Save source data + if figures_source_dir is None: + return + df_roi_intensities.to_csv( + figures_source_dir / "intensity_assessment.csv", index=False + ) + + # Save statistics summary + df_stats = pd.DataFrame( + { + "channel": ["dapi", "boundary", "intrna"], + "mean_intensity": [ + intensity_stats["dapi"]["mean"], + intensity_stats["boundary"]["mean"], + intensity_stats["intrna"]["mean"], + ], + "median_intensity": [ + intensity_stats["dapi"]["median"], + intensity_stats["boundary"]["median"], + intensity_stats["intrna"]["median"], + ], + "p10_intensity": [ + intensity_stats["dapi"]["p10"], + intensity_stats["boundary"]["p10"], + intensity_stats["intrna"]["p10"], + ], + "p90_intensity": [ + intensity_stats["dapi"]["p90"], + intensity_stats["boundary"]["p90"], + intensity_stats["intrna"]["p90"], + ], + "critical_threshold": [ + intensity_stats["dapi"]["critical_threshold"], + intensity_stats["boundary"]["critical_threshold"], + intensity_stats["intrna"]["critical_threshold"], + ], + "pct_tissue_roi_below_critical": [ + intensity_stats["dapi"]["pct_tissue_roi_below_critical"], + intensity_stats["boundary"]["pct_tissue_roi_below_critical"], + intensity_stats["intrna"]["pct_tissue_roi_below_critical"], + ], + "quality_status": [ + intensity_stats["dapi"]["quality_status"], + intensity_stats["boundary"]["quality_status"], + intensity_stats["intrna"]["quality_status"], + ], + } + ) + df_stats.to_csv( + figures_source_dir / "intensity_assessment_statistics.csv", index=False + ) + + +def generate_roi_figures( + data, + small0, + small1, + small2, + distance_map, + distance_map2, + whole_sample, + holes, + dense_intensity_regions, + df_grid_roi, + df_roi_intensities, + intensity_stats, + xoa_morphology_files, + focus_maps=None, + focus_heatmap=None, + snr_thresholds=None, + multistain_whole_sample=None, + multistain_distance_map=None, + multistain_distance_map2=None, + figure_source_tables=False, +): + """ + Generate cell-independent tile-based figures using multithreading. + + Parameters: + ----------- + data : dict + Dictionary with 'figures_dir' key + small0, small1, small2 : numpy.ndarray + Downsampled morphology images + distance_map, distance_map2 : numpy.ndarray + Distance maps (edge and holes) + whole_sample, holes, dense_intensity_regions : numpy.ndarray + Tissue masks + df_grid_roi : pandas DataFrame + Grid ROI focus scores + df_roi_intensities : pandas DataFrame + Tile intensity measurements + intensity_stats : dict + Intensity quality assessment statistics + xoa_morphology_files : list + List of morphology file paths + focus_maps : dict, optional + Focus map arrays for heatmap overlay + snr_thresholds : dict, optional + YAML ``snr.roi_tx`` block for SNR heatmap threshold lines + """ + figures_dir = data["figures_dir"] + figures_source_dir = ( + (figures_dir / "figures_source") if figure_source_tables else None + ) + if figures_source_dir is not None: + figures_source_dir.mkdir(parents=True, exist_ok=True) + # Create methodology assessment folder for comparison figures + methodology_figures_dir = figures_dir / "figures_methodology_assessment" + methodology_figures_dir.mkdir(parents=True, exist_ok=True) + + # §2.4 distance figures use the multi-stain extent mask + its distance maps when present, + # matching the edge/hole burden metrics in save_roi_qc_metrics. DAPI fallback (None on + # DAPI-only bundles) keeps those bundles byte-identical. distance_map/distance_map2 are + # referenced only by the distance closures, so rebinding them here is safe; the mask is + # aliased as _dm_mask because whole_sample is also used by the masks figure below. + if multistain_distance_map is not None: + distance_map = multistain_distance_map + if multistain_distance_map2 is not None: + distance_map2 = multistain_distance_map2 + _dm_mask = ( + whole_sample if multistain_whole_sample is None else multistain_whole_sample + ) + + def _fig1_distance_edge(): + logging.info("Generating Figure 1: Distance map (edge)...") + t_fig = time.time() + _h, _w = distance_map.shape + _aspect = (_h / _w) if _w > 0 else 1.0 + # Floor the height at 50% of width so very wide slides (e.g. brain) don't get squashed + fig, ax = plt.subplots(1, 1, figsize=(6, max(3, 6 * _aspect))) + # Phase v5: distance-to-edge readability fix. + # - viridis so near-edge tissue (low |distance|) renders as bright + # yellow against the white background — visible diagnostic region. + # - Absolute distance in µm: signed maurer is negative inside; + # |·| × 8 × 0.2125 → 0-at-boundary → max-deep-inside gradient. + # - NaN outside tissue mask → white via cmap.set_bad. + # - Thin black tissue outline as unambiguous boundary marker. + # TODO: 8 (downsample factor) and 0.2125 (Xenium native µm/px) are + # hardcoded here and in three other sites. See task #15 — future + # plumbing reads pixel_size from the bundle's experiment.xenium. + from copy import copy as _copy_cmap + + _cmap_edge = _copy_cmap(plt.cm.viridis) + _cmap_edge.set_bad(color="white") + # Cap the colorscale at 300 µm so the near-edge band uses most of the + # spectrum; tiles further than 300 µm from the boundary saturate at + # yellow and the colorbar shows an "extend max" arrow. Beyond 300 µm + # the tile is unambiguously deep-tissue and not edge-affected. + _EDGE_VMAX_UM = 300.0 + if _dm_mask.shape == distance_map.shape: + _dist_um = np.abs(distance_map) * 8 * 0.2125 + _dm_edge = np.where(_dm_mask > 0, _dist_um, np.nan) + im = _imshow_thumb( + ax, _dm_edge, cmap=_cmap_edge, vmin=0.0, vmax=_EDGE_VMAX_UM + ) + _edge_cbar_extend = "max" + else: + _dm_edge = ( + distance_map # fallback: shapes mismatch, preserve prior behaviour + ) + im = _imshow_thumb(ax, _dm_edge, cmap=_cmap_edge) + _edge_cbar_extend = "neither" + if _dm_mask.shape == distance_map.shape: + _contour_thumb( + ax, + (_dm_mask > 0).astype(np.uint8), + levels=[0.5], + colors="black", + linewidths=0.5, + ) + ax.set_title("Distance to Edge") + ax.set_aspect("equal") + cbar = fig.colorbar( + im, ax=ax, fraction=0.035, pad=0.04, shrink=0.45, extend=_edge_cbar_extend + ) + cbar.set_label("Distance from edge (µm)") + # Explicit "300+" label at the cap when extend="max" is active + if _edge_cbar_extend == "max": + cbar.set_ticks([0, 50, 100, 150, 200, 250, _EDGE_VMAX_UM]) + cbar.set_ticklabels( + ["0", "50", "100", "150", "200", "250", f"{int(_EDGE_VMAX_UM)}+"] + ) + plt.tight_layout() + plt.savefig(figures_dir / "distance_map_edge.pdf", dpi=300, bbox_inches="tight") + plt.savefig(figures_dir / "distance_map_edge.png", dpi=300, bbox_inches="tight") + plt.close(fig) + # Full-resolution export. These raster figures_source CSVs are gated behind + # --figure-source-tables (off by default), so the ~86-Mpx / ~1 GB dump only + # happens when a user explicitly asks for the raw plotting data — and then it + # must match the analysis exactly, not a thumbnail. The rendered PNG still uses + # the _imshow_thumb / _thumb display-resolution path for speed regardless. + if figures_source_dir is not None: + pd.DataFrame(distance_map).to_csv( + figures_source_dir / "distance_map_edge.csv", + index=False, + header=False, + ) + logging.info( + f"[TIMING] Figure 1 (distance map edge): {time.time() - t_fig:.1f}s" + ) + + def _fig2_distance_holes(): + logging.info("Generating Figure 2: Distance map (holes)...") + t_fig = time.time() + _h, _w = distance_map2.shape + _aspect = (_h / _w) if _w > 0 else 1.0 + # Floor the height at 50% of width so very wide slides (e.g. brain) don't get squashed + fig, ax = plt.subplots(1, 1, figsize=(6, max(3, 6 * _aspect))) + # Phase v5: distance-to-holes readability fix. + # - viridis colormap (same as edge map) for visual consistency. + # - Absolute distance in µm: |signed_maurer_distance| × 8 × 0.2125. + # - Linear vmin=0 / vmax=300 µm — matches the edge map for visual + # consistency across both panels of §2.4 (edge + holes). Tiles beyond 300 µm saturate + # at yellow with an "extend max" arrow on the colorbar. + # - NaN outside tissue mask → white via cmap.set_bad. + # - Thin black tissue outline as boundary marker. + # TODO: pixel-size conversion hardcoded — see task #15. + from copy import copy as _copy_cmap + + _cmap_holes = _copy_cmap(plt.cm.viridis) + _cmap_holes.set_bad(color="white") + _HOLES_VMAX_UM = 300.0 + if _dm_mask.shape == distance_map2.shape: + _dist_um_h = np.abs(distance_map2) * 8 * 0.2125 + _dm_holes = np.where(_dm_mask > 0, _dist_um_h, np.nan) + im = _imshow_thumb( + ax, _dm_holes, cmap=_cmap_holes, vmin=0.0, vmax=_HOLES_VMAX_UM + ) + _holes_cbar_extend = "max" + else: + _dm_holes = distance_map2 # fallback: shapes mismatch + im = _imshow_thumb(ax, _dm_holes, cmap=_cmap_holes) + _holes_cbar_extend = "neither" + if _dm_mask.shape == distance_map2.shape: + _contour_thumb( + ax, + (_dm_mask > 0).astype(np.uint8), + levels=[0.5], + colors="black", + linewidths=0.5, + ) + ax.set_title("Distance to Nearest Hole") + ax.set_aspect("equal") + cbar = fig.colorbar( + im, ax=ax, fraction=0.035, pad=0.04, shrink=0.45, extend=_holes_cbar_extend + ) + cbar.set_label("Distance from nearest hole (µm)") + # Explicit "300+" label at the cap when extend="max" is active. + if _holes_cbar_extend == "max": + cbar.set_ticks([0, 50, 100, 150, 200, 250, _HOLES_VMAX_UM]) + cbar.set_ticklabels( + ["0", "50", "100", "150", "200", "250", f"{int(_HOLES_VMAX_UM)}+"] + ) + plt.tight_layout() + plt.savefig( + figures_dir / "distance_map_holes.pdf", dpi=300, bbox_inches="tight" + ) + plt.savefig( + figures_dir / "distance_map_holes.png", dpi=300, bbox_inches="tight" + ) + plt.close(fig) + # Full-resolution export (see distance_map_edge.csv note): gated behind + # --figure-source-tables (off by default); rendering uses the display-res thumbnail. + if figures_source_dir is not None: + pd.DataFrame(distance_map2).to_csv( + figures_source_dir / "distance_map_holes.csv", + index=False, + header=False, + ) + logging.info( + f"[TIMING] Figure 2 (distance map holes): {time.time() - t_fig:.1f}s" + ) + + def _fig2b_distance_combined(): + """Two-panel combined figure: distance to edge (left) + distance to + holes (right), sharing coordinate system and colorbar conventions. + Standalone _fig1_distance_edge / _fig2_distance_holes continue to + render the individual PNGs as latent artefacts; the combined figure + is what the QMD §2.4 embeds. + """ + logging.info("Generating Figure 2b: Distance combined (edge + holes)...") + t_fig = time.time() + _h, _w = distance_map.shape + _aspect = (_h / _w) if _w > 0 else 1.0 + _panel_width = 6 + _panel_height = max(3, _panel_width * _aspect) + fig, axes = plt.subplots(1, 2, figsize=(2 * _panel_width + 2, _panel_height)) + + from copy import copy as _copy_cmap + + _VMAX_UM = 300.0 + _cmap = _copy_cmap(plt.cm.viridis) + _cmap.set_bad(color="white") + + # ── Left panel: distance to edge ─────────────────────────────── + ax = axes[0] + if _dm_mask.shape == distance_map.shape: + _dist_um = np.abs(distance_map) * 8 * 0.2125 + _dm_edge = np.where(_dm_mask > 0, _dist_um, np.nan) + im_e = _imshow_thumb(ax, _dm_edge, cmap=_cmap, vmin=0.0, vmax=_VMAX_UM) + _contour_thumb( + ax, + (_dm_mask > 0).astype(np.uint8), + levels=[0.5], + colors="black", + linewidths=0.5, + ) + _edge_extend = "max" + else: + im_e = _imshow_thumb(ax, distance_map, cmap=_cmap) + _edge_extend = "neither" + ax.set_title("Distance to Edge", fontsize=14) + ax.set_aspect("equal") + cbar_e = fig.colorbar( + im_e, ax=ax, fraction=0.035, pad=0.04, shrink=0.45, extend=_edge_extend + ) + cbar_e.set_label("Distance from edge (µm)") + if _edge_extend == "max": + cbar_e.set_ticks([0, 50, 100, 150, 200, 250, _VMAX_UM]) + cbar_e.set_ticklabels( + ["0", "50", "100", "150", "200", "250", f"{int(_VMAX_UM)}+"] + ) + + # ── Right panel: distance to holes ───────────────────────────── + ax = axes[1] + if _dm_mask.shape == distance_map2.shape: + _dist_um_h = np.abs(distance_map2) * 8 * 0.2125 + _dm_holes = np.where(_dm_mask > 0, _dist_um_h, np.nan) + im_h = _imshow_thumb(ax, _dm_holes, cmap=_cmap, vmin=0.0, vmax=_VMAX_UM) + _contour_thumb( + ax, + (_dm_mask > 0).astype(np.uint8), + levels=[0.5], + colors="black", + linewidths=0.5, + ) + _holes_extend = "max" + else: + im_h = _imshow_thumb(ax, distance_map2, cmap=_cmap) + _holes_extend = "neither" + ax.set_title("Distance to Nearest Hole", fontsize=14) + ax.set_aspect("equal") + cbar_h = fig.colorbar( + im_h, ax=ax, fraction=0.035, pad=0.04, shrink=0.45, extend=_holes_extend + ) + cbar_h.set_label("Distance from nearest hole (µm)") + if _holes_extend == "max": + cbar_h.set_ticks([0, 50, 100, 150, 200, 250, _VMAX_UM]) + cbar_h.set_ticklabels( + ["0", "50", "100", "150", "200", "250", f"{int(_VMAX_UM)}+"] + ) + + plt.tight_layout() + plt.savefig(figures_dir / "distance_maps.png", dpi=300, bbox_inches="tight") + plt.savefig(figures_dir / "distance_maps.pdf", dpi=300, bbox_inches="tight") + plt.close(fig) + logging.info( + f"[TIMING] Figure 2b (distance combined): {time.time() - t_fig:.1f}s" + ) + + def _fig3_morphology_overview(): + logging.info("Generating Figure 3: Morphology overview...") + t_fig = time.time() + # Phase v5 TODO #12: 1×3 layout (DAPI + Boundary + Interior only). The + # artefact / dense-intensity-regions panel that used to live in this + # figure's bottom-right is also rendered in imageqc_masks.png (§2.3 Masks), + # so showing it here too duplicates the same plot — dropped here. + _img_aspect = small0.shape[0] / small0.shape[1] if small0.shape[1] > 0 else 1.0 + _panel_width = 6 + _panel_height = max(_panel_width * 0.5, _panel_width * _img_aspect) + fig, ax = plt.subplots(1, 3, figsize=(3 * _panel_width, _panel_height)) + _imshow_thumb(ax[0], small0, cmap="Greys_r", vmax=np.percentile(small0, 99)) + ax[0].set_title("DAPI", fontsize=14) + ax[0].set_aspect("equal") + ax[0].axis("off") + _imshow_thumb(ax[1], small1, cmap="Greys_r", vmax=np.percentile(small1, 99)) + ax[1].set_title("Boundary", fontsize=14) + ax[1].set_aspect("equal") + ax[1].axis("off") + _imshow_thumb(ax[2], small2, cmap="Greys_r", vmax=np.percentile(small2, 99)) + ax[2].set_title("Interior", fontsize=14) + ax[2].set_aspect("equal") + ax[2].axis("off") + plt.tight_layout() + plt.savefig( + figures_dir / "morphology_overview.pdf", dpi=300, bbox_inches="tight" + ) + plt.savefig( + figures_dir / "morphology_overview.png", dpi=300, bbox_inches="tight" + ) + plt.close(fig) + # Full-resolution export (see distance_map_edge.csv note): four ~86-Mpx channel + # grids, gated behind --figure-source-tables (off by default); rendering uses + # the display-res thumbnail so the figure wall time is unaffected. + if figures_source_dir is not None: + pd.DataFrame(small0).to_csv( + figures_source_dir / "morphology_overview_DAPI.csv", + index=False, + header=False, + ) + pd.DataFrame(small1).to_csv( + figures_source_dir / "morphology_overview_Boundary.csv", + index=False, + header=False, + ) + pd.DataFrame(small2).to_csv( + figures_source_dir / "morphology_overview_Interior.csv", + index=False, + header=False, + ) + pd.DataFrame(dense_intensity_regions).to_csv( + figures_source_dir / "morphology_overview_DenseIntensityRegions.csv", + index=False, + header=False, + ) + logging.info( + f"[TIMING] Figure 3 (morphology overview): {time.time() - t_fig:.1f}s" + ) + + def _fig4_imageqc_masks(): + logging.info("Generating Figure 4: ImageQC masks...") + t_fig = time.time() + # Phase v5: 1x3 triplet — three mask panels only. DAPI morphology + # image was dropped (it's a staining, already shown under §2.4 + # Stainings via morphology_overview.png). + _img_aspect = small0.shape[0] / small0.shape[1] if small0.shape[1] > 0 else 1.0 + _panel_width = 6 + _panel_height = max(_panel_width * 0.5, _panel_width * _img_aspect) + # §2.3 tissue mask shows the EXTENT mask (all available stains) so it matches the + # reported tissue coverage; falls back to the DAPI mask on single-stain slides. + _extent_ws = ( + multistain_whole_sample + if multistain_whole_sample is not None + else whole_sample + ) + fig, ax = plt.subplots(1, 3, figsize=(3 * _panel_width, _panel_height)) + ax[0].set_title("Tissue mask (all stains)", fontsize=14) + _imshow_thumb(ax[0], _extent_ws, rgb=lambda d: color.label2rgb(d, bg_label=0)) + ax[0].set_aspect("equal") + ax[0].axis("off") + ax[1].set_title("Holes in sample", fontsize=14) + _imshow_thumb(ax[1], holes, rgb=lambda d: color.label2rgb(d, bg_label=0)) + ax[1].set_aspect("equal") + ax[1].axis("off") + ax[2].set_title("Optically dense regions", fontsize=14) + _imshow_thumb( + ax[2], dense_intensity_regions, rgb=lambda d: color.label2rgb(d, bg_label=0) + ) + ax[2].set_aspect("equal") + ax[2].axis("off") + plt.tight_layout() + plt.savefig(figures_dir / "imageqc_masks.pdf", dpi=300, bbox_inches="tight") + plt.savefig(figures_dir / "imageqc_masks.png", dpi=300, bbox_inches="tight") + plt.close(fig) + # DAPI csv export dropped — same data exposed by §2.4 Stainings. + # Full-resolution export (see distance_map_edge.csv note): three ~86-Mpx mask + # grids, gated behind --figure-source-tables (off by default); rendering uses + # the display-res thumbnail so the figure wall time is unaffected. + if figures_source_dir is not None: + pd.DataFrame(_extent_ws).to_csv( + figures_source_dir / "imageqc_masks_WholeSample.csv", + index=False, + header=False, + ) + pd.DataFrame(holes).to_csv( + figures_source_dir / "imageqc_masks_Holes.csv", + index=False, + header=False, + ) + pd.DataFrame(dense_intensity_regions).to_csv( + figures_source_dir / "imageqc_masks_DenseIntensityRegions.csv", + index=False, + header=False, + ) + logging.info(f"[TIMING] Figure 4 (ImageQC masks): {time.time() - t_fig:.1f}s") + + def _fig5_focus_heatmap(): + logging.info("Generating Figure 5: Grid tile focus heatmap...") + t_fig = time.time() + plot_grid_roi_focus_heatmap( + df_grid_roi, + small0, + figures_dir, + figures_source_dir, + threshold=-1.0, + focus_maps=focus_maps, + focus_heatmap=focus_heatmap, + ) + logging.info(f"[TIMING] Figure 5 (focus heatmap): {time.time() - t_fig:.1f}s") + + def _fig5b_focus_vs_intensity(): + logging.info("Generating Figure 5b: Tile focus score vs intensity...") + t_fig = time.time() + plot_roi_focus_vs_intensity( + df_grid_roi, figures_dir, figures_source_dir, threshold=-1.0 + ) + logging.info( + f"[TIMING] Figure 5b (focus vs intensity): {time.time() - t_fig:.1f}s" + ) + + def _fig5c_focus_distribution(): + logging.info("Generating Figure 5c: Tile focus score distribution...") + t_fig = time.time() + plot_roi_focus_distribution( + df_grid_roi, figures_dir, figures_source_dir, threshold=-1.0 + ) + logging.info( + f"[TIMING] Figure 5c (focus distribution): {time.time() - t_fig:.1f}s" + ) + + def _fig5d_focus_vs_laplacian(): + logging.info( + "Generating Figure 5d: Focus score vs Laplacian variance comparison..." + ) + t_fig = time.time() + plot_focus_score_vs_laplacian(df_grid_roi, figures_dir, figures_source_dir) + logging.info( + f"[TIMING] Figure 5d (focus vs Laplacian): {time.time() - t_fig:.1f}s" + ) + + def _fig6_intensity_assessment(): + logging.info("Generating Figure 6: Intensity assessment...") + t_fig = time.time() + plot_intensity_assessment( + df_roi_intensities, intensity_stats, small0, figures_dir, figures_source_dir + ) + logging.info( + f"[TIMING] Figure 6 (intensity assessment): {time.time() - t_fig:.1f}s" + ) + + def _fig7_snr_heatmap(): + if "neg_pct" not in df_grid_roi.columns: + logging.info("Skipping SNR heatmap: no SNR columns in df_grid_roi") + return + logging.info("Generating Figure 7: SNR spatial heatmap...") + t_fig = time.time() + plot_snr_roi_heatmap( + df_grid_roi, + small0, + figures_dir, + figures_source_dir, + snr_thresholds=snr_thresholds, + ) + logging.info(f"[TIMING] Figure 7 (SNR heatmap): {time.time() - t_fig:.1f}s") + + def _fig8_concordance(): + if "roi_tx_snr_ratio" not in df_grid_roi.columns: + logging.info("Skipping concordance plot: no SNR columns in df_grid_roi") + return + logging.info("Generating Figure 8: Cross-section concordance...") + t_fig = time.time() + plot_cross_section_concordance( + df_grid_roi, + figures_dir, + figures_source_dir, + snr_thresholds=snr_thresholds, + ) + logging.info(f"[TIMING] Figure 8 (concordance): {time.time() - t_fig:.1f}s") + + tasks = [ + _fig1_distance_edge, + _fig2_distance_holes, + _fig2b_distance_combined, + _fig3_morphology_overview, + _fig4_imageqc_masks, + _fig5_focus_heatmap, + _fig5b_focus_vs_intensity, + _fig5c_focus_distribution, + # _fig5d_focus_vs_laplacian disabled 2026-05-20 — produces + # focus_score_vs_laplacian.png + focus_score_vs_laplacian_distributions.png + # in methodology_figures_dir; neither is embedded in the report. + # Function definition retained above as latent code. + _fig6_intensity_assessment, + _fig7_snr_heatmap, + _fig8_concordance, + ] + + _run_figure_pool(tasks, phase="tile-based figures") + + logging.info("All tile-based figures generated successfully!") + logging.info(f"Saved figures to {figures_dir}") + if figures_source_dir is not None: + logging.info(f"Saved source data to {figures_source_dir}") + + +def compute_whole_grid_stain_percentiles(df_grid_roi): + """Mask-independent stain percentiles (p95 / p99) per channel over the WHOLE + tile grid (no tissue-mask filter), so they survive a mask-generation failure + (qc_threshold_refinement §5.5). DIAGNOSTIC ONLY — used to triage why a mask + failed (collapsed p99 = dim stain like skin; healthy p99 = a structural + mask-detection failure on bright tissue). NOT a gate: an absolute p99 does + not separate mask-PASS from mask-FAIL (XOA-version confound). + + Returns ``{channel: {"p95": float|None, "p99": float|None}}`` for dapi, + boundary, intrna. None when the channel column is absent or all-NaN. + """ + out = {} + for ch, col in ( + ("dapi", "dapi_intensity"), + ("boundary", "boundary_intensity"), + ("intrna", "intrna_intensity"), + ): + vals = None + if col in df_grid_roi.columns: + vals = pd.to_numeric(df_grid_roi[col], errors="coerce").to_numpy() + vals = vals[np.isfinite(vals)] + if vals is not None and vals.size > 0: + out[ch] = { + "p95": float(np.percentile(vals, 95)), + "p99": float(np.percentile(vals, 99)), + } + else: + out[ch] = {"p95": None, "p99": None} + return out + + +def save_roi_qc_metrics( + df_grid_roi, + intensity_stats, + outdir, + roi_size=None, + snr_summary=None, + distance_map=None, + distance_map2=None, + multistain_whole_sample=None, + multistain_distance_map=None, + multistain_distance_map2=None, + edge_distance_threshold: float = -25.0, + hole_distance_threshold: float = -25.0, + min_tissue_coverage_for_qc: float = ROI_MIN_TISSUE_COVERAGE_FOR_INTENSITY_QC, + qc_thresholds: dict | None = None, + lap_sigma: float | None = None, + segmentation_software: str | None = None, + xoa_version: str | None = None, +): + """ + Save tile-based QC metrics to JSON file. + + Parameters + ---------- + snr_summary : dict or None + If provided, stored under ``snr`` (from :mod:`snr_metrics`). + segmentation_software : str or None + Human-readable label for the segmentation software that produced the + bundle this report describes (e.g. ``"Xenium Onboard Analysis v4.0.1"``). + """ + import json + + # Tissue EXTENT coverage: from the multi-stain mask (all available stains) when provided, + # else the DAPI tissue_coverage column. EXTENT metrics (coverage stats, mask status, + # edge/hole) use this; DAPI-QUALITY metrics (focus, blur, usable, cluster) keep the DAPI + # tissue_coverage column. DAPI-only bundles pass multistain_whole_sample=None, so extent + # == DAPI and the output is identical. See plans/2026-06-26_PLAN_multistain-mask.md. + _has_xy = {"x1", "x2", "y1", "y2"}.issubset(df_grid_roi.columns) + if multistain_whole_sample is not None and _has_xy: + _ext_cov = _per_tile_coverage( + multistain_whole_sample, + df_grid_roi["x1"].to_numpy(np.int64, copy=False), + df_grid_roi["x2"].to_numpy(np.int64, copy=False), + df_grid_roi["y1"].to_numpy(np.int64, copy=False), + df_grid_roi["y2"].to_numpy(np.int64, copy=False), + ) + _ext_dmap = ( + distance_map if multistain_distance_map is None else multistain_distance_map + ) + _ext_dmap2 = ( + distance_map2 + if multistain_distance_map2 is None + else multistain_distance_map2 + ) + else: + _ext_cov = ( + df_grid_roi["tissue_coverage"].to_numpy(np.float64, copy=False) + if "tissue_coverage" in df_grid_roi.columns + else None + ) + _ext_dmap, _ext_dmap2 = distance_map, distance_map2 + _ext_rois_in_tissue = int((_ext_cov > 0).sum()) if _ext_cov is not None else 0 + + # Existing stats + roi_metrics = { + "roi_size_pixels": roi_size, + "xoa_version": xoa_version, + "segmentation_software": segmentation_software, + "total_rois": len(df_grid_roi), + "rois_in_tissue": _ext_rois_in_tissue, + "focus_score": { + "mean": float(df_grid_roi["focus_score"].mean()), + "median": float(df_grid_roi["focus_score"].median()), + "std": float(df_grid_roi["focus_score"].std()), + "min": float(df_grid_roi["focus_score"].min()), + "max": float(df_grid_roi["focus_score"].max()), + "tissue_median": float( + df_grid_roi.loc[ + df_grid_roi["tissue_coverage"] >= min_tissue_coverage_for_qc, + "focus_score", + ].median() + ) + if "tissue_coverage" in df_grid_roi.columns + else float(df_grid_roi["focus_score"].median()), + }, + "focus_score_norm": { + "mean": float(df_grid_roi["focus_score_norm"].mean()), + "median": float(df_grid_roi["focus_score_norm"].median()), + "std": float(df_grid_roi["focus_score_norm"].std()), + "min": float(df_grid_roi["focus_score_norm"].min()), + "max": float(df_grid_roi["focus_score_norm"].max()), + }, + "raw_intensity": { + "mean": float(df_grid_roi["raw_intensity"].mean()), + "median": float(df_grid_roi["raw_intensity"].median()), + "std": float(df_grid_roi["raw_intensity"].std()), + "min": float(df_grid_roi["raw_intensity"].min()), + "max": float(df_grid_roi["raw_intensity"].max()), + }, + # tissue_coverage (extent) reflects all available stains (multi-stain mask). + "tissue_coverage": { + "mean": float(np.mean(_ext_cov)) if _ext_cov is not None else 0.0, + "median": float(np.median(_ext_cov)) if _ext_cov is not None else 0.0, + "min": float(np.min(_ext_cov)) if _ext_cov is not None else 0.0, + "max": float(np.max(_ext_cov)) if _ext_cov is not None else 0.0, + }, + "intensity_quality": intensity_stats, + } + total_rois = int(len(df_grid_roi)) + rois_in_tissue = _ext_rois_in_tissue # extent: any tissue tile (all stains) + roi_metrics["tissue_mask_qc"] = { + "tissue_mask_generated": bool(rois_in_tissue > 0), + "status": "PASS" if rois_in_tissue > 0 else "FAIL", + "rois_in_tissue": rois_in_tissue, + "total_rois": total_rois, + "tissue_roi_fraction": float(rois_in_tissue / total_rois) + if total_rois > 0 + else 0.0, + } + + # Mask-independent stain percentiles (p95 / p99 over the WHOLE tile grid, + # 2026-06-23, qc_threshold_refinement §5.5). See + # compute_whole_grid_stain_percentiles for the diagnostic-only rationale. + roi_metrics["stain_percentiles_whole_grid"] = compute_whole_grid_stain_percentiles( + df_grid_roi + ) + + # Optional: add GMM-based blur summary if columns are present + if ( + "is_blurred_gmm" in df_grid_roi.columns + and "is_low_intensity" in df_grid_roi.columns + ): + total_rois = len(df_grid_roi) + n_blurred = int(df_grid_roi["is_blurred_gmm"].sum()) + n_low_int = int(df_grid_roi["is_low_intensity"].sum()) + if "tissue_coverage" in df_grid_roi.columns: + tissue_mask_qc = ( + df_grid_roi["tissue_coverage"] >= min_tissue_coverage_for_qc + ) + total_rois_tissue = int(tissue_mask_qc.sum()) + n_blurred_tissue = int( + df_grid_roi.loc[tissue_mask_qc, "is_blurred_gmm"].sum() + ) + n_low_int_tissue = int( + df_grid_roi.loc[tissue_mask_qc, "is_low_intensity"].sum() + ) + else: + total_rois_tissue = total_rois + n_blurred_tissue = n_blurred + n_low_int_tissue = n_low_int + + roi_metrics["blur_gmm_1d"] = { + "total_rois": total_rois, + "rois_blurred_gmm": n_blurred, + "rois_low_intensity": n_low_int, + "pct_blurred_gmm": float(n_blurred / total_rois * 100.0) + if total_rois > 0 + else 0.0, + "pct_low_intensity": float(n_low_int / total_rois * 100.0) + if total_rois > 0 + else 0.0, + # Tissue-aware denominator for report-level interpretation + "total_rois_tissue_filtered": total_rois_tissue, + "rois_blurred_gmm_tissue_filtered": n_blurred_tissue, + "rois_low_intensity_tissue_filtered": n_low_int_tissue, + "pct_blurred_gmm_tissue_filtered": float( + n_blurred_tissue / total_rois_tissue * 100.0 + ) + if total_rois_tissue > 0 + else 0.0, + "pct_low_intensity_tissue_filtered": float( + n_low_int_tissue / total_rois_tissue * 100.0 + ) + if total_rois_tissue > 0 + else 0.0, + "tissue_coverage_min_for_qc": float(min_tissue_coverage_for_qc), + } + + if "blur_prob_gmm" in df_grid_roi.columns: + valid_probs = df_grid_roi["blur_prob_gmm"].dropna() + if len(valid_probs) > 0: + roi_metrics["blur_gmm_1d"]["blur_prob_mean"] = float(valid_probs.mean()) + roi_metrics["blur_gmm_1d"]["blur_prob_median"] = float( + valid_probs.median() + ) + + # Optional: add 2D GMM-based blur summary if columns are present + if "is_blurred_gmm_2d" in df_grid_roi.columns: + total_rois = len(df_grid_roi) + n_blurred_2d = int(df_grid_roi["is_blurred_gmm_2d"].sum()) + n_low_int = ( + int(df_grid_roi["is_low_intensity"].sum()) + if "is_low_intensity" in df_grid_roi.columns + else 0 + ) + if "tissue_coverage" in df_grid_roi.columns: + tissue_mask_qc = ( + df_grid_roi["tissue_coverage"] >= min_tissue_coverage_for_qc + ) + total_rois_tissue = int(tissue_mask_qc.sum()) + n_blurred_tissue = int( + df_grid_roi.loc[tissue_mask_qc, "is_blurred_gmm_2d"].sum() + ) + n_low_int_tissue = ( + int(df_grid_roi.loc[tissue_mask_qc, "is_low_intensity"].sum()) + if "is_low_intensity" in df_grid_roi.columns + else 0 + ) + else: + total_rois_tissue = total_rois + n_blurred_tissue = n_blurred_2d + n_low_int_tissue = n_low_int + + roi_metrics["blur_gmm_2d"] = { + "total_rois": total_rois, + "rois_blurred_gmm": n_blurred_2d, + "rois_low_intensity": n_low_int, + "pct_blurred_gmm": float(n_blurred_2d / total_rois * 100.0) + if total_rois > 0 + else 0.0, + "pct_low_intensity": float(n_low_int / total_rois * 100.0) + if total_rois > 0 + else 0.0, + # Tissue-aware denominator for report-level interpretation + "total_rois_tissue_filtered": total_rois_tissue, + "rois_blurred_gmm_tissue_filtered": n_blurred_tissue, + "rois_low_intensity_tissue_filtered": n_low_int_tissue, + "pct_blurred_gmm_tissue_filtered": float( + n_blurred_tissue / total_rois_tissue * 100.0 + ) + if total_rois_tissue > 0 + else 0.0, + "pct_low_intensity_tissue_filtered": float( + n_low_int_tissue / total_rois_tissue * 100.0 + ) + if total_rois_tissue > 0 + else 0.0, + "tissue_coverage_min_for_qc": float(min_tissue_coverage_for_qc), + } + + if "blur_prob_gmm_2d" in df_grid_roi.columns: + valid_probs = df_grid_roi["blur_prob_gmm_2d"].dropna() + if len(valid_probs) > 0: + roi_metrics["blur_gmm_2d"]["blur_prob_mean"] = float(valid_probs.mean()) + roi_metrics["blur_gmm_2d"]["blur_prob_median"] = float( + valid_probs.median() + ) + + # Compare 1D vs 2D if both exist + if "is_blurred_gmm" in df_grid_roi.columns: + both_blurred = ( + df_grid_roi["is_blurred_gmm"] & df_grid_roi["is_blurred_gmm_2d"] + ).sum() + both_in_focus = ( + (~df_grid_roi["is_blurred_gmm"]) & (~df_grid_roi["is_blurred_gmm_2d"]) + ).sum() + agreement = ( + (both_blurred + both_in_focus) / total_rois * 100.0 + if total_rois > 0 + else 0.0 + ) + roi_metrics["gmm_comparison"] = { + "agreement_pct": float(agreement), + "both_blurred": int(both_blurred), + "both_in_focus": int(both_in_focus), + "only_1d_blurred": int( + ( + df_grid_roi["is_blurred_gmm"] + & ~df_grid_roi["is_blurred_gmm_2d"] + ).sum() + ), + "only_2d_blurred": int( + ( + ~df_grid_roi["is_blurred_gmm"] + & df_grid_roi["is_blurred_gmm_2d"] + ).sum() + ), + } + + if snr_summary is not None: + roi_metrics["snr"] = snr_summary + + # Optional: morphology ROI summary metrics (vectorized, memory-friendly). + if ( + distance_map is not None + and distance_map2 is not None + and {"x1", "x2", "y1", "y2", "tissue_coverage"}.issubset(df_grid_roi.columns) + ): + x1 = df_grid_roi["x1"].to_numpy(np.int64, copy=False) + x2 = df_grid_roi["x2"].to_numpy(np.int64, copy=False) + y1 = df_grid_roi["y1"].to_numpy(np.int64, copy=False) + y2 = df_grid_roi["y2"].to_numpy(np.int64, copy=False) + # Two coverage arrays: DAPI for QUALITY (usable, cluster), multi-stain for EXTENT + # (edge/hole, tissue-tile count). ext_cov == dapi_cov on DAPI-only bundles. + dapi_cov = df_grid_roi["tissue_coverage"].to_numpy(np.float64, copy=False) + ext_cov = _ext_cov if _ext_cov is not None else dapi_cov + + # EXTENT distance maps (multi-stain when present) sampled at ROI centroids. + cx_ds = np.clip(((x1 + x2) // 2) // 8, 0, _ext_dmap.shape[1] - 1) + cy_ds = np.clip(((y1 + y2) // 2) // 8, 0, _ext_dmap.shape[0] - 1) + dist_edge = _ext_dmap[cy_ds, cx_ds] + dist_hole = _ext_dmap2[cy_ds, cx_ds] + + # DAPI-quality masks (usable, cluster) and EXTENT masks (edge/hole, counts). + dapi_tissue_mask = dapi_cov > 0.0 + qc_tissue_mask = dapi_cov >= float(min_tissue_coverage_for_qc) + n_qc_tissue = int(qc_tissue_mask.sum()) + ext_tissue_mask = ext_cov > 0.0 + n_ext_tissue = int(ext_tissue_mask.sum()) + + edge_zone_frac = ( + float( + (ext_tissue_mask & (dist_edge > float(edge_distance_threshold))).sum() + / n_ext_tissue + ) + if n_ext_tissue > 0 + else None + ) + hole_area_frac = ( + float( + (ext_tissue_mask & (dist_hole > float(hole_distance_threshold))).sum() + / n_ext_tissue + ) + if n_ext_tissue > 0 + else None + ) + + if "is_blurred_gmm_2d" in df_grid_roi.columns: + blurred_mask = df_grid_roi["is_blurred_gmm_2d"].to_numpy(bool, copy=False) + elif "is_blurred_gmm" in df_grid_roi.columns: + blurred_mask = df_grid_roi["is_blurred_gmm"].to_numpy(bool, copy=False) + else: + blurred_mask = np.zeros(len(df_grid_roi), dtype=bool) + # usable_tissue uses its OWN low-intensity floor (DAPI_LOW_INTENSITY_FLOOR=50), + # independent of the is_low_intensity column (which stays at ROI_INTENSITY_THRESHOLD + # for the 1D-GMM tissue scope). A dim-but-real tissue tile shouldn't count as + # unusable on absolute brightness alone. See 2026-06-26_PLAN_tissue-mask-recalibration. + _intensity_col = ( + "dapi_intensity" + if "dapi_intensity" in df_grid_roi.columns + else ("raw_intensity" if "raw_intensity" in df_grid_roi.columns else None) + ) + if _intensity_col is not None: + low_intensity_mask = ( + df_grid_roi[_intensity_col].to_numpy() < DAPI_LOW_INTENSITY_FLOOR + ) + else: + low_intensity_mask = np.zeros(len(df_grid_roi), dtype=bool) + bad_mask = blurred_mask | low_intensity_mask + + usable_tissue_frac = ( + float((qc_tissue_mask & (~bad_mask)).sum() / n_qc_tissue) + if n_qc_tissue > 0 + else None + ) + + # Largest contiguous bad zone (4-neighbour) over ROI grid. + x_unique = np.unique(x1) + y_unique = np.unique(y1) + x_idx = np.searchsorted(x_unique, x1) + y_idx = np.searchsorted(y_unique, y1) + bad_grid = np.zeros((len(y_unique), len(x_unique)), dtype=bool) + tissue_grid = np.zeros((len(y_unique), len(x_unique)), dtype=bool) + bad_grid[y_idx, x_idx] = dapi_tissue_mask & bad_mask + tissue_grid[y_idx, x_idx] = dapi_tissue_mask + n_tissue_grid = int(tissue_grid.sum()) + if n_tissue_grid > 0 and bad_grid.any(): + labels, n_comp = ndimage.label( + bad_grid, + structure=np.array([[0, 1, 0], [1, 1, 1], [0, 1, 0]], dtype=np.uint8), + ) + component_sizes = np.bincount(labels.ravel()) + largest_bad_component = int(component_sizes[1:].max()) if n_comp > 0 else 0 + cluster_zone_bad_fraction = float(largest_bad_component / n_tissue_grid) + else: + cluster_zone_bad_fraction = 0.0 if n_tissue_grid > 0 else None + + roi_metrics["morphology"] = { + "edge_zone_frac": edge_zone_frac, + "hole_area_frac": hole_area_frac, + "usable_tissue_frac": usable_tissue_frac, + "cluster_zone_bad_fraction": cluster_zone_bad_fraction, + "edge_distance_threshold_px_ds": float(edge_distance_threshold), + "hole_distance_threshold_px_ds": float(hole_distance_threshold), + "min_tissue_coverage_for_qc": float(min_tissue_coverage_for_qc), + "n_tissue_rois": n_ext_tissue, + "n_qc_tissue_rois": n_qc_tissue, + } + + # Optional: Laplacian sharpness summary for section-6 report metrics. + _qc = qc_thresholds or {} + _ch_dapi = (_qc.get("channels") or {}).get("DAPI") or {} + _focus_cut = _qc.get("focus") or {} + # The standalone absolute Laplacian-variance floor verdict was dropped + # (2026-06-22, qc_drift_analysis): lap_var scales with brightness² (rho≈0.91 + # with DAPI) and collapses to near-zero on dim XOA-4.0 images, over-flagging + # good dim tissue. It is redundant with the GMM-2D classifier, which already + # uses lap_var as a feature. lap_var_median_raw is still emitted as an + # informational calibration stat below. + + if {"dapi_lap_var", "tissue_coverage"}.issubset(df_grid_roi.columns): + lap_var = df_grid_roi["dapi_lap_var"].to_numpy(np.float64, copy=False) + tissue_cov = df_grid_roi["tissue_coverage"].to_numpy(np.float64, copy=False) + focus_col = ( + "dapi_focus_score" + if "dapi_focus_score" in df_grid_roi.columns + else ("focus_score" if "focus_score" in df_grid_roi.columns else None) + ) + focus_vals = ( + df_grid_roi[focus_col].to_numpy(np.float64, copy=False) + if focus_col is not None + else np.full_like(lap_var, np.nan, dtype=np.float64) + ) + + mask = ( + (tissue_cov >= float(min_tissue_coverage_for_qc)) + & np.isfinite(lap_var) + & np.isfinite(focus_vals) + & (lap_var >= 0.0) + ) + # P5: Exclude boundary ROIs (centre within window_size//2 of image edge) + if "is_boundary_roi" in df_grid_roi.columns: + boundary_mask = df_grid_roi["is_boundary_roi"].to_numpy(bool, copy=False) + mask = mask & ~boundary_mask + n_boundary_excluded = int(boundary_mask.sum()) + else: + n_boundary_excluded = 0 + + n_used = int(mask.sum()) + if n_used > 1: + lap_used = lap_var[mask] + focus_used = focus_vals[mask] + + med = float(np.nanmedian(lap_used)) + q25, q75 = np.nanpercentile(lap_used, [25, 75]) + iqr = float(q75 - q25) + + iqr_uniform = iqr <= 1e-12 + + # P4: Guard Spearman correlation on uniform slides + cv_focus = float(np.std(focus_used) / (np.mean(focus_used) + 1e-12)) + cv_lap = float(np.std(lap_used) / (np.mean(lap_used) + 1e-12)) + if cv_focus < 0.1 and cv_lap < 0.1: + focus_lap_corr = None + focus_lap_corr_status = "uniform" + else: + focus_lap_corr = float( + pd.Series(focus_used).corr(pd.Series(lap_used), method="spearman") + ) + focus_lap_corr_status = None + + # Separation of 2D-GMM blur components in log1p(lap_var) space (Cohen's d). + component_separation = None + if "is_blurred_gmm_2d" in df_grid_roi.columns: + blur2d = df_grid_roi["is_blurred_gmm_2d"].to_numpy(bool, copy=False)[ + mask + ] + if blur2d.any() and (~blur2d).any(): + lap_log = np.log1p(np.maximum(lap_used, 0.0)) + lap_blur = lap_log[blur2d] + lap_focus = lap_log[~blur2d] + if lap_blur.size > 1 and lap_focus.size > 1: + n_b, n_f = lap_blur.size, lap_focus.size + v_blur = float(np.var(lap_blur, ddof=1)) + v_focus = float(np.var(lap_focus, ddof=1)) + pooled_sd = np.sqrt( + max( + ((n_b - 1) * v_blur + (n_f - 1) * v_focus) + / (n_b + n_f - 2), + 0.0, + ) + + 1e-12 + ) + component_separation = float( + abs(float(np.mean(lap_focus)) - float(np.mean(lap_blur))) + / pooled_sd + ) + + roi_metrics["laplacian_sharpness"] = { + "lap_sigma": float(lap_sigma) if lap_sigma is not None else None, + "n_rois_used": n_used, + "n_boundary_excluded": n_boundary_excluded, + "min_tissue_coverage_for_qc": float(min_tissue_coverage_for_qc), + "iqr_uniform": iqr_uniform, + "lap_var_median_raw": med, + "focus_lap_spearman_corr": focus_lap_corr, + "focus_lap_corr_status": focus_lap_corr_status, + "component_separation": component_separation, + } + + # Save to JSON + metrics_file = Path(outdir) / "roi_qc_metrics.json" + with open(metrics_file, "w") as f: + json.dump(roi_metrics, f, indent=2, default=str) + logging.info(f"[OK] Saved tile QC metrics to {metrics_file}") + + +def save_pixel_focus_maps(focus_maps, outdir, compress=True): + """ + Save per-pixel focus maps as compressed tiled TIFF files. + + Uses lz4 compression (5-10x faster than zlib) and parallel I/O via + ThreadPoolExecutor to minimize wall-clock time on multi-map datasets. + + Args: + focus_maps: dict returned by compute_all_focus_maps() + outdir: Path to output directory + compress: Use lz4 compression (default: True) + """ + from concurrent.futures import ThreadPoolExecutor + + outdir = Path(outdir) + items = [(name, arr) for name, arr in focus_maps.items() if arr is not None] + + if not items: + logging.info(" No focus maps to save.") + return + + def _save_one(name, array, outdir_path): + t_start = time.time() + path = outdir_path / f"{name}.tif" + tile = (256, 256) if array.shape[0] >= 256 and array.shape[1] >= 256 else None + tifffile.imwrite( + str(path), + np.asarray(array, dtype=np.float32), + compression="zstd" if compress else None, + tile=tile, + ) + elapsed = time.time() - t_start + return name, path, array.shape, elapsed + + with ThreadPoolExecutor(max_workers=min(len(items), 4)) as executor: + futures = [executor.submit(_save_one, name, arr, outdir) for name, arr in items] + for future in futures: + name, path, shape, elapsed = future.result() + logging.info(f" Saved {name}: {path} ({shape}, float32, {elapsed:.1f}s)") + + +def save_versions_file(outdir): + """Save package versions to YAML file""" + import matplotlib + + def get_version(package, package_name=None): + """Safely get version of a package""" + if package_name is None: + package_name = ( + package.__name__ if hasattr(package, "__name__") else str(package) + ) + + try: + if hasattr(package, "__version__"): + return package.__version__ + else: + # Try to get version via importlib + import importlib.metadata + + return importlib.metadata.version(package_name) + except Exception: + return "unknown" + + # Get Python version + python_version = ( + f"{sys.version_info.major}.{sys.version_info.minor}.{sys.version_info.micro}" + ) + + # Get package versions safely + packages = { + "python": python_version, + "numpy": get_version(np, "numpy"), + "pandas": get_version(pd, "pandas"), + "matplotlib": get_version(matplotlib, "matplotlib"), + "seaborn": get_version(sns, "seaborn"), + "click": get_version(click, "click"), + "pathlib": "built-in", + } + + # Get versions for other packages + try: + import skimage + + packages["scikit-image"] = get_version(skimage, "scikit-image") + except Exception: + packages["scikit-image"] = "unknown" + + try: + packages["tifffile"] = get_version(tifffile, "tifffile") + except Exception: + packages["tifffile"] = "unknown" + + try: + packages["zarr"] = get_version(zarr, "zarr") + except Exception: + packages["zarr"] = "unknown" + + try: + import napari_skimage_regionprops + + packages["napari-skimage-regionprops"] = get_version( + napari_skimage_regionprops, "napari-skimage-regionprops" + ) + except Exception: + packages["napari-skimage-regionprops"] = "unknown" + + try: + packages["napari-simpleitk-image-processing"] = get_version( + nsitk, "napari-simpleitk-image-processing" + ) + except Exception: + packages["napari-simpleitk-image-processing"] = "unknown" + + # Create YAML content + yaml_content = f"""XENIUM_IMAGE_QC: + python: {packages["python"]} + numpy: {packages["numpy"]} + pandas: {packages["pandas"]} + matplotlib: {packages["matplotlib"]} + seaborn: {packages["seaborn"]} + scikit-image: {packages["scikit-image"]} + tifffile: {packages["tifffile"]} + zarr: {packages["zarr"]} + napari-skimage-regionprops: {packages["napari-skimage-regionprops"]} + napari-simpleitk-image-processing: {packages["napari-simpleitk-image-processing"]} + click: {packages["click"]} + pathlib: {packages["pathlib"]} +""" + + # Save to file + versions_file = outdir / "versions.yml" + with open(versions_file, "w") as f: + f.write(yaml_content) + + logging.info(f"[OK] Saved package versions to {versions_file}") + + +# ===== FUNCTIONS FROM image_qc_mapping_to_cells.py ===== + + +def load_spatial_data(data): + """Load and prepare spatial data. This only works for the original Xenium bundle generated from XOA""" + + # Load cell masks + cell_masks_zarr = open_zarr(data["cell_masks_path"]) + # cell_masks_zarr will only have one channel if its the result from xeniumranger import-segmentation + # Lazily sliced, not materialised: np.array() here cost 22 GB of uint32 on a + # 5.5 gigapixel sample. Its consumers are row-block reductions + # (_labeled_sums_chunked, LabeledSumAccumulator) and scattered point lookups, + # both of which LazyLabelPlane serves straight from zarr. The legacy + # --legacy-focus path calls regionprops_table, which needs a real array, and + # materialises it explicitly there. + cellseg_mask = LazyLabelPlane(cell_masks_zarr.get("masks").get("1")) + + # Load spatial data + df_spatial = pd.read_parquet( + data["cells_parquet_path"], + columns=[ + "x_centroid", + "y_centroid", + "transcript_counts", + "cell_area", + "nucleus_area", + "nucleus_count", + "segmentation_method", + "cell_id", + ], + ) + + # Rename columns for simplicity + df_spatial.rename(columns={"x_centroid": "x", "y_centroid": "y"}, inplace=True) + + # Convert to pixel coordinates + df_spatial["x"] = (df_spatial.x / XENIUM_PIXEL_SIZE_UM).astype(int) + df_spatial["y"] = (df_spatial.y / XENIUM_PIXEL_SIZE_UM).astype(int) + + # Measure CellID - OPTIMIZED: vectorized NumPy indexing (50-100x faster) + x_coords = np.clip(df_spatial["x"].values, 0, cellseg_mask.shape[1] - 1) + y_coords = np.clip(df_spatial["y"].values, 0, cellseg_mask.shape[0] - 1) + df_spatial["CellID"] = cellseg_mask[y_coords, x_coords] + + # Load and merge clustering and UMAP data + df_clusters = pd.read_csv(data["clusters_csv_path"]).rename( + columns={"Barcode": "cell_id", "Cluster": "Cluster_kmeans10"} + ) + df_UMAP = pd.read_csv(data["umap_path"]).rename(columns={"Barcode": "cell_id"}) + df_spatial = df_spatial.merge(df_clusters, on="cell_id").merge( + df_UMAP, on="cell_id" + ) + return df_spatial, cellseg_mask, cell_masks_zarr + + +def map_distances_to_cells( + df_spatial, + distance_map, + distance_map2, + dense_intensity_regions, + downsample_factor=8, +): + """ + Map distance maps and dense intensity region masks to cells based on cell centroid locations. + + Extracted from generate_sample_masks_and_distances() for cell-independent workflow. + + Parameters: + ----------- + df_spatial : pandas DataFrame + Cell DataFrame with 'x' and 'y' centroid columns + distance_map : numpy.ndarray + Distance to edge map (downsampled) + distance_map2 : numpy.ndarray + Distance to nearest hole map (downsampled) + dense_intensity_regions : numpy.ndarray + Dense intensity region mask (downsampled) + downsample_factor : int, optional + Downsampling factor (default: 8) + + Returns: + -------- + pandas DataFrame + df_spatial with added columns: + - Distance-to-edge: Distance to sample edge for each cell + - Distance-to-nearest-hole: Distance to nearest hole for each cell + - In-Area-with-Dense-Intensity-Region: Binary indicator if cell is in dense intensity region area + """ + # Vectorized coordinate mapping -- compute pixel indices once, clipped to array bounds + X = np.clip( + (df_spatial["x"].values / downsample_factor).astype(int), + 0, + distance_map.shape[1] - 1, + ) + Y = np.clip( + (df_spatial["y"].values / downsample_factor).astype(int), + 0, + distance_map.shape[0] - 1, + ) + + # Vectorized lookups -- three lines instead of three loops + df_spatial["Distance-to-edge"] = distance_map[Y, X] + df_spatial["Distance-to-nearest-hole"] = distance_map2[Y, X] + df_spatial["Dense-Intensity-Region-ID"] = dense_intensity_regions[Y, X].astype(int) + + return df_spatial + + +def map_grid_roi_to_cells(df_grid_roi, df_cells, overlapping=False): + """ + Map cell-independent grid tile focus scores to cells based on cell centroid location. + + This function replaces the previous cell-based tile approach with a cell-independent + grid-based tile approach that provides uniform coverage across the tissue. + + For each cell, finds which tile its centroid (x, y) falls into and assigns + that tile's focus scores to the cell. + + Parameters: + ----------- + df_grid_roi : pandas DataFrame + Grid ROI DataFrame with columns: roi_id, x1, x2, y1, y2, focus_score, + focus_score_norm, raw_intensity, tissue_coverage, and channel-specific columns + df_cells : pandas DataFrame + Cell DataFrame with columns: x, y (centroid coordinates) + overlapping : bool, optional + Whether grid is overlapping. If True and cell falls into multiple ROIs, + uses ROI with highest tissue_coverage (default: False) + + Returns: + -------- + pandas DataFrame + Cell DataFrame with added columns: + - roi_id: Integer ID of ROI containing cell centroid (0, 1, 2, ...) or None if cell outside all ROIs + - DAPI_mean_roi: Mean intensity from grid ROI (NaN if cell outside all ROIs) + - DAPI_RFS_roi: Raw focus score from grid ROI (NaN if cell outside all ROIs) + - DAPI_RFSnorm_roi: Normalized focus score from grid ROI (NaN if cell outside all ROIs) + - roi_tissue_coverage: Tissue coverage of assigned ROI (NaN if cell outside all ROIs) + - Additional channel-specific columns if available (boundary_*, intrna_*) + """ + # Create output DataFrame + df_cells_mapped = df_cells.copy() + + # Initialize columns using old naming convention for compatibility + df_cells_mapped["roi_id"] = None + df_cells_mapped["DAPI_mean_roi"] = np.nan + df_cells_mapped["DAPI_RFS_roi"] = np.nan + df_cells_mapped["DAPI_RFSnorm_roi"] = np.nan + df_cells_mapped["roi_tissue_coverage"] = np.nan + + # Check for additional channel columns + has_boundary = "boundary_focus_score" in df_grid_roi.columns + has_intrna = "intrna_focus_score" in df_grid_roi.columns + + if has_boundary: + df_cells_mapped["boundary_focus_score_roi"] = np.nan + df_cells_mapped["boundary_focus_score_norm_roi"] = np.nan + df_cells_mapped["boundary_intensity_roi"] = np.nan + + if has_intrna: + df_cells_mapped["intrna_focus_score_roi"] = np.nan + df_cells_mapped["intrna_focus_score_norm_roi"] = np.nan + df_cells_mapped["intrna_intensity_roi"] = np.nan + + # OPTIMIZED: Fast vectorized lookup using 2D grid + logging.info(f" Mapping {len(df_cells):,} cells to {len(df_grid_roi):,} tiles...") + + # Get cell coordinates as arrays (faster than DataFrame access) + cell_x = df_cells["x"].values + cell_y = df_cells["y"].values + n_cells = len(df_cells) + + # Create arrays to store matches + cell_roi_matches = np.full(n_cells, -1, dtype=np.int32) # -1 means no match + cell_roi_tissue_coverage_arr = np.full(n_cells, np.nan, dtype=np.float64) + + # Always try the fast vectorized approach: build a 2D lookup grid + is_regular_grid = False + if len(df_grid_roi) > 1 and not overlapping: + roi_size = df_grid_roi.iloc[0]["x2"] - df_grid_roi.iloc[0]["x1"] + stride = roi_size # non-overlapping grid + + x_min = df_grid_roi["x1"].min() + y_min = df_grid_roi["y1"].min() + x_max = df_grid_roi["x1"].max() + y_max = df_grid_roi["y1"].max() + + n_cols = int(round((x_max - x_min) / stride)) + 1 + n_rows = int(round((y_max - y_min) / stride)) + 1 + + # Build 2D grid: grid[row, col] -> index in df_grid_roi + grid_lookup = np.full((n_rows, n_cols), -1, dtype=np.int32) + roi_x1_arr = df_grid_roi["x1"].values + roi_y1_arr = df_grid_roi["y1"].values + gx_arr = np.round((roi_x1_arr - x_min) / stride).astype(int) + gy_arr = np.round((roi_y1_arr - y_min) / stride).astype(int) + + # Check if this is a valid grid (no out-of-bounds) + valid = (gx_arr >= 0) & (gx_arr < n_cols) & (gy_arr >= 0) & (gy_arr < n_rows) + if valid.all(): + # Populate grid (vectorized) + grid_lookup[gy_arr, gx_arr] = np.arange(len(df_grid_roi)) + + # Vectorized lookup for all cells -- O(n_cells). + # floor, NOT round: tile x1/y1 above are exact multiples of stride so + # rounding them is a no-op, but cell centroids fall anywhere inside a + # tile. Rounding pushed every cell past the half-stride mark into the + # next tile, misassigning ~74% of cells (measured on 4 calibration + # samples) and manufacturing cell-vs-tile disagreement. The slow + # containment path below is the reference; see + # tests/test_cell_roi_mapping.py. + cell_gx = np.clip( + np.floor((cell_x - x_min) / stride).astype(int), 0, n_cols - 1 + ) + cell_gy = np.clip( + np.floor((cell_y - y_min) / stride).astype(int), 0, n_rows - 1 + ) + cell_roi_idx = grid_lookup[cell_gy, cell_gx] # vectorized! + + matched_mask = cell_roi_idx >= 0 + cell_roi_matches[matched_mask] = df_grid_roi["roi_id"].values[ + cell_roi_idx[matched_mask] + ] + cell_roi_tissue_coverage_arr[matched_mask] = df_grid_roi[ + "tissue_coverage" + ].values[cell_roi_idx[matched_mask]] + + is_regular_grid = True + logging.info( + f" Using fast grid lookup (stride={stride}px, grid={n_cols}x{n_rows})..." + ) + + if not is_regular_grid: + # SLOW PATH: Irregular/filtered grid - fallback for grids that don't validate + if len(df_grid_roi) > 1000: + logging.info( + f" Processing {len(df_cells):,} cells against {len(df_grid_roi):,} tiles (irregular grid)..." + ) + + # Convert ROI boundaries to arrays for fast lookup + roi_x1 = df_grid_roi["x1"].values + roi_x2 = df_grid_roi["x2"].values + roi_y1 = df_grid_roi["y1"].values + roi_y2 = df_grid_roi["y2"].values + roi_ids = df_grid_roi["roi_id"].values + roi_tissue_coverage = df_grid_roi["tissue_coverage"].values + + for cell_idx in range(n_cells): + x_cell = cell_x[cell_idx] + y_cell = cell_y[cell_idx] + + matches = ( + (roi_x1 <= x_cell) + & (x_cell < roi_x2) + & (roi_y1 <= y_cell) + & (y_cell < roi_y2) + ) + + if np.any(matches): + match_idx = np.where(matches)[0][0] + cell_roi_matches[cell_idx] = roi_ids[match_idx] + cell_roi_tissue_coverage_arr[cell_idx] = roi_tissue_coverage[match_idx] + + # Check for GMM columns + has_gmm_1d = "is_blurred_gmm" in df_grid_roi.columns + has_gmm_2d = "is_blurred_gmm_2d" in df_grid_roi.columns + has_blur_prob_1d = "blur_prob_gmm" in df_grid_roi.columns + has_blur_prob_2d = "blur_prob_gmm_2d" in df_grid_roi.columns + + # Initialize GMM columns if available + if has_gmm_1d: + df_cells_mapped["is_blurred_gmm_roi"] = False + if has_blur_prob_1d: + df_cells_mapped["blur_prob_gmm_roi"] = np.nan + if has_gmm_2d: + df_cells_mapped["is_blurred_gmm_2d_roi"] = False + if has_blur_prob_2d: + df_cells_mapped["blur_prob_gmm_2d_roi"] = np.nan + + # Build vectorized column arrays from df_grid_roi for direct indexing + # Map roi_id -> index in df_grid_roi for O(1) lookup + roi_id_to_idx = pd.Series( + range(len(df_grid_roi)), index=df_grid_roi["roi_id"].values + ) + + # Resolve column names (handle legacy naming) + _dapi_intensity_col = ( + "dapi_intensity" if "dapi_intensity" in df_grid_roi.columns else "raw_intensity" + ) + _dapi_focus_col = ( + "dapi_focus_score" + if "dapi_focus_score" in df_grid_roi.columns + else "focus_score" + ) + _dapi_focus_norm_col = ( + "dapi_focus_score_norm" + if "dapi_focus_score_norm" in df_grid_roi.columns + else "focus_score_norm" + ) + + dapi_intensity_arr = df_grid_roi[_dapi_intensity_col].values + dapi_focus_arr = df_grid_roi[_dapi_focus_col].values + dapi_focus_norm_arr = df_grid_roi[_dapi_focus_norm_col].values + + # Assign ROI values to cells using vectorized operations + matched_mask = cell_roi_matches >= 0 + matched_roi_ids = cell_roi_matches[matched_mask] + + # Map matched roi_ids to df_grid_roi indices for vectorized column access + matched_df_indices = roi_id_to_idx[matched_roi_ids].values + + # Assign ROI ID and tissue coverage + df_cells_mapped.loc[matched_mask, "roi_id"] = matched_roi_ids + df_cells_mapped.loc[matched_mask, "roi_tissue_coverage"] = ( + cell_roi_tissue_coverage_arr[matched_mask] + ) + + # Assign DAPI data via direct array indexing (no dict lookups) + df_cells_mapped.loc[matched_mask, "DAPI_mean_roi"] = dapi_intensity_arr[ + matched_df_indices + ] + df_cells_mapped.loc[matched_mask, "DAPI_RFS_roi"] = dapi_focus_arr[ + matched_df_indices + ] + df_cells_mapped.loc[matched_mask, "DAPI_RFSnorm_roi"] = dapi_focus_norm_arr[ + matched_df_indices + ] + + # Assign GMM classifications if available + if has_gmm_1d: + df_cells_mapped.loc[matched_mask, "is_blurred_gmm_roi"] = df_grid_roi[ + "is_blurred_gmm" + ].values[matched_df_indices] + if has_blur_prob_1d: + df_cells_mapped.loc[matched_mask, "blur_prob_gmm_roi"] = df_grid_roi[ + "blur_prob_gmm" + ].values[matched_df_indices] + if has_gmm_2d: + df_cells_mapped.loc[matched_mask, "is_blurred_gmm_2d_roi"] = df_grid_roi[ + "is_blurred_gmm_2d" + ].values[matched_df_indices] + if has_blur_prob_2d: + df_cells_mapped.loc[matched_mask, "blur_prob_gmm_2d_roi"] = df_grid_roi[ + "blur_prob_gmm_2d" + ].values[matched_df_indices] + + # Assign additional channel data if available + if has_boundary: + df_cells_mapped.loc[matched_mask, "boundary_focus_score_roi"] = df_grid_roi[ + "boundary_focus_score" + ].values[matched_df_indices] + df_cells_mapped.loc[matched_mask, "boundary_focus_score_norm_roi"] = ( + df_grid_roi["boundary_focus_score_norm"].values[matched_df_indices] + ) + df_cells_mapped.loc[matched_mask, "boundary_intensity_roi"] = df_grid_roi[ + "boundary_intensity" + ].values[matched_df_indices] + + if has_intrna: + df_cells_mapped.loc[matched_mask, "intrna_focus_score_roi"] = df_grid_roi[ + "intrna_focus_score" + ].values[matched_df_indices] + df_cells_mapped.loc[matched_mask, "intrna_focus_score_norm_roi"] = df_grid_roi[ + "intrna_focus_score_norm" + ].values[matched_df_indices] + df_cells_mapped.loc[matched_mask, "intrna_intensity_roi"] = df_grid_roi[ + "intrna_intensity" + ].values[matched_df_indices] + + # Handle overlapping grids: if cell falls into multiple ROIs, use highest tissue_coverage + if overlapping: + # Find cells with multiple ROI assignments + cell_roi_counts = df_cells_mapped.groupby(df_cells_mapped.index)[ + "roi_id" + ].count() + cells_with_multiple = cell_roi_counts[cell_roi_counts > 1].index + + if len(cells_with_multiple) > 0: + # For each cell with multiple ROIs, find the one with highest tissue_coverage + for cell_idx in cells_with_multiple: + # Find all ROIs this cell falls into + x_cell = df_cells.loc[cell_idx, "x"] + y_cell = df_cells.loc[cell_idx, "y"] + + matching_rois = df_grid_roi[ + (df_grid_roi["x1"] <= x_cell) + & (x_cell < df_grid_roi["x2"]) + & (df_grid_roi["y1"] <= y_cell) + & (y_cell < df_grid_roi["y2"]) + ] + + if len(matching_rois) > 0: + # Select ROI with highest tissue_coverage + best_roi = matching_rois.loc[ + matching_rois["tissue_coverage"].idxmax() + ] + + # Update cell with best ROI + df_cells_mapped.loc[cell_idx, "roi_id"] = best_roi["roi_id"] + df_cells_mapped.loc[cell_idx, "DAPI_mean_roi"] = best_roi.get( + "dapi_intensity", best_roi.get("raw_intensity", np.nan) + ) + df_cells_mapped.loc[cell_idx, "DAPI_RFS_roi"] = best_roi.get( + "dapi_focus_score", best_roi.get("focus_score", np.nan) + ) + df_cells_mapped.loc[cell_idx, "DAPI_RFSnorm_roi"] = best_roi.get( + "dapi_focus_score_norm", + best_roi.get("focus_score_norm", np.nan), + ) + df_cells_mapped.loc[cell_idx, "roi_tissue_coverage"] = best_roi[ + "tissue_coverage" + ] + + # Update GMM classifications if available + if has_gmm_1d: + df_cells_mapped.loc[cell_idx, "is_blurred_gmm_roi"] = ( + best_roi.get("is_blurred_gmm", False) + ) + if has_blur_prob_1d: + df_cells_mapped.loc[cell_idx, "blur_prob_gmm_roi"] = ( + best_roi.get("blur_prob_gmm", np.nan) + ) + if has_gmm_2d: + df_cells_mapped.loc[cell_idx, "is_blurred_gmm_2d_roi"] = ( + best_roi.get("is_blurred_gmm_2d", False) + ) + if has_blur_prob_2d: + df_cells_mapped.loc[cell_idx, "blur_prob_gmm_2d_roi"] = ( + best_roi.get("blur_prob_gmm_2d", np.nan) + ) + + if has_boundary: + df_cells_mapped.loc[cell_idx, "boundary_focus_score_roi"] = ( + best_roi.get("boundary_focus_score", np.nan) + ) + df_cells_mapped.loc[ + cell_idx, "boundary_focus_score_norm_roi" + ] = best_roi.get("boundary_focus_score_norm", np.nan) + df_cells_mapped.loc[cell_idx, "boundary_intensity_roi"] = ( + best_roi.get("boundary_intensity", np.nan) + ) + + if has_intrna: + df_cells_mapped.loc[cell_idx, "intrna_focus_score_roi"] = ( + best_roi.get("intrna_focus_score", np.nan) + ) + df_cells_mapped.loc[cell_idx, "intrna_focus_score_norm_roi"] = ( + best_roi.get("intrna_focus_score_norm", np.nan) + ) + df_cells_mapped.loc[cell_idx, "intrna_intensity_roi"] = ( + best_roi.get("intrna_intensity", np.nan) + ) + + return df_cells_mapped + + +def load_roi_blur_threshold(outdir): + """ + Load ROI blur threshold configuration from JSON file saved by image_qc_roi_processing.py. + + Parameters: + ----------- + outdir : Path + Output directory where roi_blur_threshold.json should be located + + Returns: + -------- + tuple + (roi_focus_score_threshold, roi_intensity_threshold) or (None, None) if not found + """ + threshold_json = Path(outdir) / "roi_blur_threshold.json" + if not threshold_json.exists(): + return None, None + + try: + with open(threshold_json, "r") as f: + threshold_config = json.load(f) + roi_threshold = threshold_config.get("roi_focus_score_threshold") + intensity_threshold = threshold_config.get("roi_intensity_threshold") + return roi_threshold, intensity_threshold + except (json.JSONDecodeError, KeyError) as e: + logging.warning(f" Warning: Could not load threshold configuration: {e}") + return None, None + + +def create_final_merged_data( + df_spatial, + myData, + roi_data=None, + ccfs_threshold=DEFAULT_CCFS_LOW_TEXTURE_THRESHOLD, + edge_distance_threshold=-25, + hole_distance_threshold=-25, + roi_threshold=None, + roi_intensity_threshold=None, +): + """Create final merged dataset with boolean columns for thresholds""" + + # Link the Spatial Cell matrix (df_spatial) with the FocusScore table + new_df = pd.merge(df_spatial, myData, how="inner", on="CellID") + + # Create new segmentation names with colour palette for plotting. + # XR import-segmentation (e.g. from the CROP subworkflow) writes + # `segmentation_method = 'Imported Cell Segmentation'` into cells.parquet, + # which doesn't substring-match any of the three OBA categories + # ("boundary"/"interior"/"nucleus"). Without this 4th row the match-filter + # below drops every cell and downstream figures fail on empty arrays. + seg = pd.DataFrame( + { + "segmentation": [ + "boundary", + "interior", + "nucleus", + "Imported Cell Segmentation", + ], + "segPal": ["#FFABC3", "#A9A800", "#A9CEFF", "#7FBFFF"], + "join": [1, 1, 1, 1], + } + ) + + # Link the Spatial Cell matrix to the new segmentation colour palette + new_df["join"] = 1 + new_df = new_df.merge(seg, on="join").drop("join", axis=1) + seg.drop("join", axis=1, inplace=True) + new_df["match"] = new_df.apply( + lambda x: x.segmentation_method.find(x.segmentation), axis=1 + ).ge(0) + new_df = new_df[new_df["match"]] + + # Add boolean columns for various thresholds (nuclei-based method) + new_df["is_low_nuclear_texture"] = new_df["CCFS_DAPI"] <= ccfs_threshold + new_df["is_high_nuclear_texture"] = new_df["CCFS_DAPI"] > ccfs_threshold + new_df["is_near_edge"] = new_df["Distance-to-edge"] > edge_distance_threshold + new_df["is_far_from_edge"] = new_df["Distance-to-edge"] <= edge_distance_threshold + new_df["is_near_hole"] = ( + new_df["Distance-to-nearest-hole"] > hole_distance_threshold + ) + new_df["is_far_from_hole"] = ( + new_df["Distance-to-nearest-hole"] <= hole_distance_threshold + ) + # Create boolean indicator for cells in any dense intensity region (backward compatibility) + new_df["has_dense_intensity_regions"] = new_df["Dense-Intensity-Region-ID"] > 0 + + # Merge ROI data if provided + calculated_roi_threshold = None + if roi_data is not None: + # Merge ROI data using x and y coordinates + new_df = pd.merge( + new_df, roi_data, on=["x", "y"], how="left", suffixes=("", "_roi_merge") + ) + + # Use new threshold approach: raw score percentile + intensity threshold + # If roi_threshold is provided (e.g., for backward compatibility), use it + # Otherwise, calculate from raw scores using configured parameters + if roi_threshold is None: + # roi_threshold should have been calculated from df_grid_roi and passed here + # If not provided, we can't calculate it here (need df_grid_roi) + # This should not happen in normal flow, but provide fallback + logging.warning( + " Warning: roi_threshold not provided, using default calculation" + ) + roi_threshold = -1.0 # Old default for backward compatibility + + # Use intensity threshold from configuration if not provided + # If None, it means threshold config wasn't loaded (backward compatibility) + # In that case, skip intensity check and only use focus score threshold + if roi_intensity_threshold is None: + roi_intensity_threshold = ( + None # Will skip intensity check in classification + ) + + calculated_roi_threshold = roi_threshold + + # Add boolean columns for tile-based blur detection + # New approach: blurred if (raw_focus_score <= threshold) OR (intensity < intensity_threshold) + # Use raw scores (DAPI_RFS_roi) not normalized (DAPI_RFSnorm_roi) + has_intensity = "DAPI_mean_roi" in new_df.columns + has_raw_focus = "DAPI_RFS_roi" in new_df.columns + + if has_raw_focus: + # Combined threshold: focus score OR intensity (if intensity threshold provided) + if has_intensity and roi_intensity_threshold is not None: + new_df["is_blurred_roi"] = (new_df["DAPI_RFS_roi"] <= roi_threshold) | ( + new_df["DAPI_mean_roi"] < roi_intensity_threshold + ) + else: + # Fallback: just use focus score if intensity not available or threshold not provided + new_df["is_blurred_roi"] = new_df["DAPI_RFS_roi"] <= roi_threshold + else: + # Fallback: use normalized scores if raw scores not available (backward compatibility) + new_df["is_blurred_roi"] = new_df["DAPI_RFSnorm_roi"] <= roi_threshold + + new_df["is_high_focus_roi"] = ~new_df["is_blurred_roi"] + + return new_df, calculated_roi_threshold + + +# ===== CELL-LEVEL PLOTTING FUNCTIONS ===== + + +def plot_nuclear_texture_proportions( + new_df, + figures_dir, + figures_source_dir, + texture_threshold=DEFAULT_CCFS_LOW_TEXTURE_THRESHOLD, + GROUP_BY_COLUMN="Cluster_kmeans10", +): + """ + Create a stacked bar plot showing proportion of cells with high and low blur scores per cluster. + + Parameters: + ----------- + new_df : pandas DataFrame + DataFrame containing CCFS_DAPI and Cluster_kmeans10 columns + texture_threshold : float, optional + Threshold to classify cells as high/low nuclear texture (default: DEFAULT_CCFS_LOW_TEXTURE_THRESHOLD) + """ + # Create nuclear texture categories (create temporary column to avoid modifying original) + df_temp = new_df.copy() + # Low CCFS_DAPI values (<= threshold) = low nuclear texture quality + # High CCFS_DAPI values (> threshold) = high nuclear texture quality + df_temp["texture_category"] = np.where( + df_temp["CCFS_DAPI"] <= texture_threshold, "Low Quality", "High Quality" + ) + + # Calculate proportions + proportions = ( + df_temp.groupby([GROUP_BY_COLUMN, "texture_category"]) + .size() + .unstack(fill_value=0) + ) + proportions = ( + proportions.div(proportions.sum(axis=1), axis=0) * 100 + ) # Convert to percentages + + # Reorder so 'Low Quality' (red) is plotted first → at the bottom of + # each stacked bar. Matches the convention in plot_tile_blur_proportions_roi + # (Blurred at bottom). pandas stacks columns from bottom up in the order + # they appear in the DataFrame. + proportions = proportions[["Low Quality", "High Quality"]] + + # Set up the plot + fig = plt.figure(figsize=(12, 6)) + + # Create stacked bar plot with specified colors + # Column order is Low Quality (red, bottom) then High Quality (blue, top). + proportions.plot( + kind="bar", + stacked=True, + color=[ + "#d62728", + "#1f77b4", + ], # Red for low quality (bottom), blue for high quality (top) + figsize=(12, 6), + ) + + # Customize the plot + plt.title( + f"Proportion of Cells by Nuclear Texture Quality (Threshold: {texture_threshold})", + fontsize=14, + pad=20, + ) + plt.xlabel(GROUP_BY_COLUMN, fontsize=12) + plt.ylabel("Percentage of Cells", fontsize=12) + + # Rotate x-axis labels if needed + plt.xticks(rotation=45, ha="right") + + # Add percentage labels on the bars + for c in plt.gca().containers: + # Add labels + plt.gca().bar_label(c, fmt="%.1f%%", label_type="center") + + # Add a grid for better readability + plt.grid(True, axis="y", linestyle="--", alpha=0.7) + + # Adjust layout to prevent label cutoff + plt.tight_layout() + + # Save figure instead of showing + plt.savefig( + figures_dir / "nuclear_texture_proportions.png", dpi=300, bbox_inches="tight" + ) + plt.close(fig) + + # Save data as CSV + df_texture_proportions = ( + df_temp.groupby([GROUP_BY_COLUMN, "texture_category"]) + .size() + .reset_index(name="count") + ) + if figures_source_dir is not None: + df_texture_proportions.to_csv( + figures_source_dir / "nuclear_texture_proportions.csv", index=False + ) + + # Print summary statistics + logging.info("Summary statistics by cluster:") + summary_stats = ( + df_temp.groupby([GROUP_BY_COLUMN, "texture_category"]) + .size() + .unstack(fill_value=0) + ) + logging.info("Count of cells by texture category:") + logging.info(summary_stats) + logging.info("Percentage of cells by texture category:") + logging.info(summary_stats.div(summary_stats.sum(axis=1), axis=0) * 100) + + +def plot_tile_blur_proportions_roi( + new_df, + figures_dir, + figures_source_dir, + roi_threshold=-1.0, + GROUP_BY_COLUMN="Cluster_kmeans10", +): + """ + Create a stacked bar plot showing proportion of cells with high and low tile-based blur scores per cluster. + Uses GMM 2D classification if available, otherwise falls back to threshold-based method. + + Parameters: + ----------- + new_df : pandas DataFrame + DataFrame containing DAPI_RFSnorm_roi, is_blurred_roi (or is_blurred_gmm_2d_roi), and Cluster_kmeans10 columns + roi_threshold : float, optional + Threshold to classify cells as high/low blur (default: -1.0) - used only if GMM 2D not available + """ + # Check if ROI data is available - prefer GMM 2D, fall back to threshold-based + has_gmm_2d = "is_blurred_gmm_2d_roi" in new_df.columns + has_threshold = "is_blurred_roi" in new_df.columns + + if not has_gmm_2d and not has_threshold: + logging.warning( + "Warning: Tile-based blur scores not available. Skipping tile blur score proportions plot." + ) + return + + # Create blur score categories (create temporary column to avoid modifying original) + df_temp = new_df.copy() + + # Use GMM 2D if available, otherwise use threshold-based + if has_gmm_2d: + df_temp["blur_category"] = np.where( + df_temp["is_blurred_gmm_2d_roi"], "Blurred", "In Focus" + ) + method_name = "2D GMM" + else: + df_temp["blur_category"] = np.where( + df_temp["is_blurred_roi"], "Blurred", "In Focus" + ) + method_name = "Threshold" + + # Calculate proportions + proportions = ( + df_temp.groupby([GROUP_BY_COLUMN, "blur_category"]).size().unstack(fill_value=0) + ) + proportions = ( + proportions.div(proportions.sum(axis=1), axis=0) * 100 + ) # Convert to percentages + + # Set up the plot + fig = plt.figure(figsize=(12, 6)) + + # Create stacked bar plot with specified colors + # Colors are applied in alphabetical order: 'Blurred' (red), 'In Focus' (blue) + proportions.plot( + kind="bar", + stacked=True, + color=["#d62728", "#1f77b4"], # Red for blurred, blue for in focus + figsize=(12, 6), + ) + + # Customize the plot + if has_gmm_2d: + plt.title( + "Proportion of Cells by Tile-based Blur Score (2D GMM Classification)", + fontsize=14, + pad=20, + ) + else: + # Note: roi_threshold is in raw score units, classification uses combined approach + # Try to get intensity threshold from the threshold config if available + intensity_threshold_display = getattr( + plot_tile_blur_proportions_roi, "_intensity_threshold", None + ) + if intensity_threshold_display is None: + intensity_threshold_display = 20.0 # Default + plt.title( + f"Proportion of Cells by Tile-based Blur Score\n(Raw score threshold: {roi_threshold:.2f}, Intensity threshold: {intensity_threshold_display})", + fontsize=14, + pad=20, + ) + plt.xlabel(GROUP_BY_COLUMN, fontsize=12) + plt.ylabel("Percentage of Cells", fontsize=12) + + # Rotate x-axis labels if needed + plt.xticks(rotation=45, ha="right") + + # Add percentage labels on the bars + for c in plt.gca().containers: + # Add labels + plt.gca().bar_label(c, fmt="%.1f%%", label_type="center") + + # Add a grid for better readability + plt.grid(True, axis="y", linestyle="--", alpha=0.7) + + # Adjust layout to prevent label cutoff + plt.tight_layout() + + # Save figure instead of showing + plt.savefig( + figures_dir / "tile_blur_proportions_roi.png", dpi=300, bbox_inches="tight" + ) + plt.close(fig) + + # Save data as CSV + df_blur_proportions = ( + df_temp.groupby([GROUP_BY_COLUMN, "blur_category"]) + .size() + .reset_index(name="count") + ) + if figures_source_dir is not None: + df_blur_proportions.to_csv( + figures_source_dir / "tile_blur_proportions_roi.csv", index=False + ) + + # Print summary statistics + logging.info(f"Summary statistics by cluster (tile-based, {method_name}):") + summary_stats = ( + df_temp.groupby([GROUP_BY_COLUMN, "blur_category"]).size().unstack(fill_value=0) + ) + logging.info("Count of cells by blur category:") + logging.info(summary_stats) + logging.info("Percentage of cells by blur category:") + logging.info(summary_stats.div(summary_stats.sum(axis=1), axis=0) * 100) + + +def plot_cell_focus_distribution( + new_df, + figures_dir, + figures_source_dir, + GROUP_BY_COLUMN="Cluster_kmeans10", +): + """Plot cell-level focus score distributions using tile-propagated metrics. + + Creates a 1×2 panel (v4 r5 trim): + - Left: histogram of DAPI_RFSnorm_roi (tile focus score per cell) + - Right: histogram coloured by GMM blur classification + + The previous bottom row (focus density per cluster, blur prob density per + cluster) was clustering-related and overlapped with figures in the + Per-cluster subsection of 8.B. The blur-prob-density-per-cluster view + moved to its own dedicated function `plot_blur_prob_density_by_cluster`; + the focus-density-per-cluster view was dropped as redundant with the + nuclear texture density per cluster figure. + """ + has_focus = "DAPI_RFSnorm_roi" in new_df.columns + has_gmm_prob = ( + "blur_prob_gmm_2d_roi" in new_df.columns + ) # used for source-CSV column inclusion below + has_gmm_class = "is_blurred_gmm_2d_roi" in new_df.columns + + if not has_focus: + logging.warning( + "No DAPI_RFSnorm_roi column — skipping cell focus distribution plot" + ) + return + + fig, axes = plt.subplots(1, 2, figsize=(16, 6)) + + focus_vals = new_df["DAPI_RFSnorm_roi"].dropna() + + # --- Left: overall focus score histogram --- + ax = axes[0] + ax.hist(focus_vals, bins=60, alpha=0.7, edgecolor="black", color="steelblue") + ax.set_xlabel("Tile focus score (DAPI_RFSnorm_roi)", fontsize=11) + ax.set_ylabel("Number of cells", fontsize=11) + ax.set_title("Overall histogram of tile focus scores across all cells", fontsize=12) + stats_text = f"n={len(focus_vals):,}\nMedian={focus_vals.median():.4f}\nMean={focus_vals.mean():.4f}" + ax.text( + 0.95, + 0.95, + stats_text, + transform=ax.transAxes, + va="top", + ha="right", + bbox=dict(boxstyle="round", facecolor="wheat", alpha=0.7), + fontsize=10, + ) + ax.grid(True, axis="y", linestyle="--", alpha=0.7) + + # --- Right: histogram split by GMM blur class --- + ax = axes[1] + if has_gmm_class: + gmm_col = new_df["is_blurred_gmm_2d_roi"].astype(bool) + valid = new_df["DAPI_RFSnorm_roi"].notna() + blurred = new_df.loc[valid & gmm_col, "DAPI_RFSnorm_roi"] + sharp = new_df.loc[valid & ~gmm_col, "DAPI_RFSnorm_roi"] + ax.hist( + sharp, + bins=60, + alpha=0.6, + label=f"In focus ({len(sharp):,})", + color="#1f77b4", + ) + ax.hist( + blurred, + bins=60, + alpha=0.6, + label=f"Blurred ({len(blurred):,})", + color="#d62728", + ) + ax.legend(fontsize=10) + ax.set_title( + "Histogram split by GMM blur classification (blue = in focus, red = blurred)", + fontsize=12, + ) + else: + ax.hist(focus_vals, bins=60, alpha=0.7, color="steelblue") + ax.set_title( + "Histogram split by GMM blur classification (no GMM data)", fontsize=12 + ) + ax.set_xlabel("Tile focus score (DAPI_RFSnorm_roi)", fontsize=11) + ax.set_ylabel("Number of cells", fontsize=11) + ax.grid(True, axis="y", linestyle="--", alpha=0.7) + + plt.tight_layout() + plt.savefig( + figures_dir / "cell_focus_distribution.png", dpi=300, bbox_inches="tight" + ) + plt.close(fig) + + # Save source data + src_cols = ["DAPI_RFSnorm_roi"] + if has_gmm_class: + src_cols.append("is_blurred_gmm_2d_roi") + if has_gmm_prob: + src_cols.append("blur_prob_gmm_2d_roi") + if GROUP_BY_COLUMN in new_df.columns: + src_cols.append(GROUP_BY_COLUMN) + src = new_df[[c for c in src_cols if c in new_df.columns]].dropna( + subset=["DAPI_RFSnorm_roi"] + ) + if figures_source_dir is not None: + src.to_csv(figures_source_dir / "cell_focus_distribution.csv", index=False) + + +def plot_tile_focus_gmm_spatial(new_df, figures_dir, figures_source_dir): + """Single-panel spatial map of tile-based focus, coloured by GMM 2D blur + classification (red = blurred, blue = in-focus). Falls back to RFSnorm + viridis colormap when GMM 2D classification is unavailable. + + Replaces the right panel of the now-latent `plot_spatial_comparison` 2-panel + figure (v4 r5: spatial concordance dropped per user feedback — known low + concordance not informative; tile-blur spatial overview kept on its own). + """ + if "DAPI_RFSnorm_roi" not in new_df.columns: + logging.warning("No DAPI_RFSnorm_roi — skipping tile-focus GMM spatial plot") + return + has_gmm_2d = "is_blurred_gmm_2d_roi" in new_df.columns + valid_mask = new_df["DAPI_RFSnorm_roi"].notna() + if valid_mask.sum() == 0: + logging.warning( + "No valid DAPI_RFSnorm_roi values — skipping tile-focus GMM spatial plot" + ) + return + + # Phase 11 (v5): aspect-adaptive figure size matching the slide proportions + # rather than a fixed (8, 8) square. Mirrors plot_grid_roi_focus_heatmap and + # plot_snr_roi_heatmap so all whole-sample maps render at consistent width. + _x = new_df.loc[valid_mask, "x"] + _y = new_df.loc[valid_mask, "y"] + _x_range = float(_x.max() - _x.min()) if len(_x) else 1.0 + _y_range = float(_y.max() - _y.min()) if len(_y) else 1.0 + _img_aspect = _y_range / _x_range if _x_range > 0 else 1.0 + _panel_width = 6 + _panel_height = max(_panel_width * 0.5, _panel_width * _img_aspect) + fig, ax = plt.subplots(1, 1, figsize=(_panel_width, _panel_height)) + if has_gmm_2d: + gmm_valid = valid_mask & new_df["is_blurred_gmm_2d_roi"].notna() + if gmm_valid.sum() > 0: + colors = [ + "red" if blurred else "blue" + for blurred in new_df.loc[gmm_valid, "is_blurred_gmm_2d_roi"] + ] + ax.scatter( + new_df.loc[gmm_valid, "x"], + -new_df.loc[gmm_valid, "y"], + s=0.1, + c=colors, + rasterized=True, + ) + ax.set_title( + "Focus score, coloured by GMM 2D blur classification\n" + "(red = blurred, blue = in focus)", + fontsize=12, + ) + else: + ax.text( + 0.5, + 0.5, + "No valid GMM 2D classification data", + transform=ax.transAxes, + ha="center", + va="center", + fontsize=14, + ) + else: + roi_values = new_df.loc[valid_mask, "DAPI_RFSnorm_roi"] + scatter = ax.scatter( + new_df.loc[valid_mask, "x"], + -new_df.loc[valid_mask, "y"], + s=0.1, + c=roi_values, + cmap="viridis", + rasterized=True, + ) + ax.set_title("Focus score (DAPI_RFSnorm_roi)", fontsize=12) + plt.colorbar(scatter, ax=ax, label="DAPI_RFSnorm_roi") + ax.set_facecolor("black") + ax.set_aspect("equal") + plt.tight_layout() + plt.savefig( + figures_dir / "tile_focus_gmm_spatial.png", dpi=300, bbox_inches="tight" + ) + plt.close(fig) + # Save source data + src_cols = ["x", "y", "DAPI_RFSnorm_roi"] + if has_gmm_2d: + src_cols.append("is_blurred_gmm_2d_roi") + src = new_df[[c for c in src_cols if c in new_df.columns]].dropna( + subset=["DAPI_RFSnorm_roi"] + ) + if figures_source_dir is not None: + src.to_csv(figures_source_dir / "tile_focus_gmm_spatial.csv", index=False) + + +def plot_cell_flagged_maps_combined(new_df, figures_dir, figures_source_dir): + """Two-panel spatial map combining the nuclear texture flagged and + blurriness flagged cell views into one figure for direct visual + comparison. Left panel mirrors the standalone ccfs_thresholded.png; + right panel mirrors the standalone tile_focus_gmm_spatial.png. Both + panels share the sample's coordinate system and use the aspect-adaptive + sizing pattern from plot_snr_roi_heatmap so the rendered figure aligns + with the §3.4 SNR heatmap panel layout. + """ + needed = ("is_low_nuclear_texture", "is_blurred_gmm_2d_roi") + if all(c not in new_df.columns for c in needed): + logging.warning( + "No nuclear texture or blurry-(GMM) columns — skipping combined cell-flagged maps" + ) + return + if "x" not in new_df.columns or "y" not in new_df.columns: + logging.warning("No x / y coordinates — skipping combined cell-flagged maps") + return + + _x = new_df["x"].dropna() + _y = new_df["y"].dropna() + if len(_x) == 0 or len(_y) == 0: + logging.warning("Empty x / y — skipping combined cell-flagged maps") + return + _x_range = float(_x.max() - _x.min()) if _x.max() > _x.min() else 1.0 + _y_range = float(_y.max() - _y.min()) if _y.max() > _y.min() else 1.0 + _img_aspect = _y_range / _x_range if _x_range > 0 else 1.0 + panel_width = 6 + panel_height = max(panel_width * 0.5, panel_width * _img_aspect) + fig, axes = plt.subplots(1, 2, figsize=(2 * panel_width + 2, panel_height)) + + # ── Left panel: nuclear texture-flagged ──────────────────────────── + ax = axes[0] + if "is_low_nuclear_texture" in new_df.columns: + _mask_low = new_df["is_low_nuclear_texture"].fillna(False).astype(bool) + _high = new_df[~_mask_low] + _low = new_df[_mask_low] + if len(_high) > 0: + ax.scatter(_high["x"], -_high["y"], s=0.1, color="#64B5F6", rasterized=True) + if len(_low) > 0: + # Slightly larger red points so flagged cells remain visible + # against the dense blue background and on the white figure bg. + ax.scatter(_low["x"], -_low["y"], s=0.5, color="red", rasterized=True) + ax.set_title( + "Spatial distribution of low nuclear texture cells\n(red = low nuclear texture, blue = high)", + fontsize=14, + ) + ax.set_aspect("equal") + ax.set_xticks([]) + ax.set_yticks([]) + for _spine in ax.spines.values(): + _spine.set_edgecolor("black") + _spine.set_linewidth(1.2) + + # ── Right panel: blurriness-flagged (tile-GMM inherited) ─────────── + ax = axes[1] + if "is_blurred_gmm_2d_roi" in new_df.columns: + _valid_mask = new_df["is_blurred_gmm_2d_roi"].notna() + if _valid_mask.any(): + _valid_df = new_df.loc[_valid_mask] + _blur_mask = _valid_df["is_blurred_gmm_2d_roi"].astype(bool) + _in_focus = _valid_df[~_blur_mask] + _blurred = _valid_df[_blur_mask] + if len(_in_focus) > 0: + ax.scatter( + _in_focus["x"], + -_in_focus["y"], + s=0.1, + color="#64B5F6", + rasterized=True, + ) + if len(_blurred) > 0: + # Same size as in-focus: GMM blur flags typically cover contiguous + # regions, so enlargement is not needed for visibility and would + # swamp the panel. + ax.scatter( + _blurred["x"], -_blurred["y"], s=0.1, color="red", rasterized=True + ) + ax.set_title( + "Spatial distribution of blurry cells\n(red = blurred, blue = in focus)", + fontsize=14, + ) + ax.set_aspect("equal") + ax.set_xticks([]) + ax.set_yticks([]) + for _spine in ax.spines.values(): + _spine.set_edgecolor("black") + _spine.set_linewidth(1.2) + + plt.tight_layout() + plt.savefig(figures_dir / "cell_flagged_maps.png", dpi=300, bbox_inches="tight") + plt.close(fig) + + +def plot_blur_prob_density_by_cluster( + new_df, + figures_dir, + figures_source_dir, + GROUP_BY_COLUMN="Cluster_kmeans10", +): + """Single-panel KDE of `blur_prob_gmm_2d_roi` per expression cluster, with + a vertical line at the blur threshold (0.5). Parallels the existing + nuclear texture density by cluster figure (CCFS density per cluster) but + for tile-blur probability — sits in 8.B's Per-cluster subsection. + + Extracted from the bottom-right panel of the v4-r4 `plot_cell_focus_distribution` + (which was a 2×2 grid; trimmed to 1×2 in v4 r5 because the bottom row + was clustering-related and belonged in the Per-cluster subsection). + """ + if "blur_prob_gmm_2d_roi" not in new_df.columns: + logging.warning( + "No blur_prob_gmm_2d_roi column — skipping per-cluster blur probability density" + ) + return + + # Aesthetic harmonization with plot_nuclear_texture_density: same + # `palette="husl"` + `fill=True` + `alpha=0.3` so cluster colours match + # across the two density figures (same Cluster_kmeans10 hue → same colour + # assignment by seaborn) and both have semi-transparent fill. + fig = plt.figure(figsize=(12, 6)) + if GROUP_BY_COLUMN in new_df.columns: + df_kde = new_df[[GROUP_BY_COLUMN, "blur_prob_gmm_2d_roi"]].dropna() + if len(df_kde) > 0: + ax = sns.kdeplot( + data=df_kde, + x="blur_prob_gmm_2d_roi", + hue=GROUP_BY_COLUMN, + palette="husl", + common_norm=False, + fill=True, + alpha=0.3, + ) + plt.axvline( + x=0.5, + color="black", + linestyle="--", + alpha=0.5, + label="Blur threshold (0.5)", + ) + # Reformat legend labels to "Cluster N" prefix and include the + # threshold line — matches plot_nuclear_texture_density legend. + legend = ax.get_legend() + if legend is not None: + handles = legend.legend_handles + labels = [f"Cluster {label.get_text()}" for label in legend.get_texts()] + threshold_line = plt.Line2D( + [0], [0], color="black", linestyle="--", alpha=0.5 + ) + handles = [threshold_line] + handles + labels = ["Blur threshold (0.5)"] + labels + plt.legend( + handles, + labels, + title=GROUP_BY_COLUMN, + bbox_to_anchor=(1.05, 1), + loc="upper left", + ) + plt.title( + f"Distribution of GMM blur probability by {GROUP_BY_COLUMN}", + fontsize=14, + pad=20, + ) + else: + ax = plt.gca() + new_df["blur_prob_gmm_2d_roi"].dropna().plot.kde(ax=ax, color="steelblue") + plt.axvline(x=0.5, color="black", linestyle="--", alpha=0.5) + plt.title( + "GMM blur probability density (threshold at 0.5)", fontsize=14, pad=20 + ) + plt.xlabel("P(blur component)", fontsize=12) + plt.ylabel("Density", fontsize=12) + plt.grid(True, linestyle="--", alpha=0.7) + plt.tight_layout() + plt.savefig( + figures_dir / "blur_prob_density_by_cluster.png", + dpi=300, + bbox_inches="tight", + ) + plt.close(fig) + src_cols = ["blur_prob_gmm_2d_roi"] + if GROUP_BY_COLUMN in new_df.columns: + src_cols.append(GROUP_BY_COLUMN) + src = new_df[[c for c in src_cols if c in new_df.columns]].dropna( + subset=["blur_prob_gmm_2d_roi"] + ) + if figures_source_dir is not None: + src.to_csv(figures_source_dir / "blur_prob_density_by_cluster.csv", index=False) + + +def plot_intensity_transcript_correlation(new_df, figures_dir, figures_source_dir): + """Spearman correlation heatmap between per-cell quality metrics and + transcript counts. + + Includes six per-cell variables: nuclear texture (CCFS_DAPI), focus + score (tile-level, propagated to cell), three channel intensities + (DAPI / Boundary / IntRNA), and transcript counts. + + Diagnostic for "which quality axis is critical for transcript yield + in THIS sample's tissue type?". Per-channel intensity_critical cutoffs + (DAPI=500, Boundary=100, IntRNA=300) are calibrated on lung/liver and + may not be appropriate for tissues like brain (where DAPI is + unreliable due to large neurons with naturally lower nuclear + contrast). The heatmap shows which axis actually correlates with + transcript yield, helping the analyst weight per-channel verdicts in + §4.1 appropriately. + + Saves `figures/intensity_transcript_correlation.png` + source CSV. + Skips silently if fewer than 2 of the input columns are available. + """ + # All columns are read from `new_df`, the post-merge cell-aligned + # frame produced at line ~5502 by `pd.merge(df_spatial, myData, on="CellID")`. + # `mean_intensity` (DAPI) comes in via the merge from `myData`. Reading + # all columns from a single index-aligned DataFrame guarantees that + # row N is the SAME cell across all series — critical for the + # correlation to be scientifically meaningful (reading half from `myData` + # and half from `new_df` would pair values across DIFFERENT cells). + cols = {} + # Image-quality metrics first (nuclear texture + focus) — these are the + # axes a reader is checking "is X the operative quality signal for my + # tissue?". Stain intensities follow; transcript counts last. + if "CCFS_DAPI" in new_df.columns: + cols["Nuclear texture"] = new_df["CCFS_DAPI"] + if "DAPI_RFSnorm_roi" in new_df.columns: + cols["Focus score"] = new_df["DAPI_RFSnorm_roi"] + if "mean_intensity" in new_df.columns: + cols["DAPI"] = new_df["mean_intensity"] + if "mean_intensity_Boundary" in new_df.columns: + cols["Boundary"] = new_df["mean_intensity_Boundary"] + if "mean_intensity_IntRNA" in new_df.columns: + cols["IntRNA"] = new_df["mean_intensity_IntRNA"] + if "transcript_counts" in new_df.columns: + cols["transcripts"] = new_df["transcript_counts"] + + if len(cols) < 2: + logging.warning( + "Fewer than 2 quality / transcript columns available — skipping " + "per-cell correlation heatmap." + ) + return + + df_corr_input = pd.DataFrame(cols).dropna() + if len(df_corr_input) < 30: + logging.warning( + "Fewer than 30 cells with all quality / transcript columns — " + "skipping per-cell correlation heatmap (rho unstable)." + ) + return + + # Spearman (rank-based) — robust to non-normal distributions and outliers + # which intensity / transcript count data often exhibits. + corr_matrix = df_corr_input.corr(method="spearman") + + fig, ax = plt.subplots(1, 1, figsize=(8, 7)) + sns.heatmap( + corr_matrix, + annot=True, + fmt=".2f", + cmap="RdBu_r", + vmin=-1.0, + vmax=1.0, + center=0.0, + square=True, + cbar_kws={"label": "Spearman ρ", "shrink": 0.8}, + ax=ax, + linewidths=0.5, + linecolor="white", + ) + ax.set_title( + f"Per-cell quality metrics vs transcript count correlation (Spearman ρ; n={len(df_corr_input):,} cells)", + fontsize=12, + ) + plt.tight_layout() + plt.savefig( + figures_dir / "intensity_transcript_correlation.png", + dpi=300, + bbox_inches="tight", + ) + plt.close(fig) + # Save the input data (long form: cell + 4 columns) for reproducibility. + if figures_source_dir is not None: + df_corr_input.to_csv( + figures_source_dir / "intensity_transcript_correlation.csv", index=False + ) + + +def plot_nuclear_texture_density( + new_df, + figures_dir, + figures_source_dir, + GROUP_BY_COLUMN="Cluster_kmeans10", + ccfs_low_texture_threshold=DEFAULT_CCFS_LOW_TEXTURE_THRESHOLD, +): + """ + Create a density plot of CCFS_DAPI values grouped by Cluster_kmeans10. + + Parameters: + ----------- + new_df : pandas DataFrame + DataFrame containing CCFS_DAPI and Cluster_kmeans10 columns + """ + # Set up the plot + fig = plt.figure(figsize=(12, 6)) + + # Create the density plot + ax = sns.kdeplot( + data=new_df, + x="CCFS_DAPI", + hue=GROUP_BY_COLUMN, + palette="husl", + common_norm=False, # Normalize each cluster separately + fill=True, + alpha=0.3, + ) # Make the fill semi-transparent + + # Add a vertical line for the blur threshold + plt.axvline( + x=ccfs_low_texture_threshold, + color="black", + linestyle="--", + alpha=0.5, + label=f"Low Texture Threshold ({ccfs_low_texture_threshold})", + ) + + # Customize the plot + plt.title( + f"Distribution of Nuclear Texture Scores (CCFS_DAPI) by {GROUP_BY_COLUMN}", + fontsize=14, + pad=20, + ) + plt.xlabel("CCFS_DAPI (Nuclear Texture Score)", fontsize=12) + plt.ylabel("Density", fontsize=12) + + # Add a grid for better readability + plt.grid(True, linestyle="--", alpha=0.7) + + # Get the current legend + legend = ax.get_legend() + + # Get the handles and labels + handles = legend.legend_handles + labels = [f"Cluster {label.get_text()}" for label in legend.get_texts()] + + # Add the threshold line to the legend + threshold_line = plt.Line2D([0], [0], color="black", linestyle="--", alpha=0.5) + handles = [threshold_line] + handles + labels = ["Default Threshold"] + labels + + # Create new legend + plt.legend( + handles, + labels, + title=GROUP_BY_COLUMN, + bbox_to_anchor=(1.05, 1), + loc="upper left", + ) + + # Adjust layout to prevent label cutoff + plt.tight_layout() + + # Save figure instead of showing + plt.savefig( + figures_dir / "nuclear_texture_density.png", dpi=300, bbox_inches="tight" + ) + plt.close(fig) + + # Save data as CSV + if figures_source_dir is not None: + df_density_data = new_df[["CCFS_DAPI", GROUP_BY_COLUMN]].copy() + df_density_data.to_csv( + figures_source_dir / "nuclear_texture_density.csv", index=False + ) + + +def plot_nuclear_texture_vs_transcripts( + new_df, + figures_dir, + figures_source_dir, + log_scale=False, + ccfs_low_texture_threshold=DEFAULT_CCFS_LOW_TEXTURE_THRESHOLD, +): + """ + Create a scatter plot of nuclear texture scores (CCFS_DAPI) vs transcript counts. + + Parameters: + ----------- + new_df : pandas DataFrame + DataFrame containing CCFS_DAPI and transcript_counts columns + log_scale : bool, optional + Whether to use log scale for x-axis (default: False) + """ + # Set up the plot + fig = plt.figure(figsize=(12, 6)) + ax = plt.gca() + + _valid = new_df[["transcript_counts", "CCFS_DAPI"]].dropna() + x_vals = _valid["transcript_counts"].values.astype(np.float64) + y_vals = _valid["CCFS_DAPI"].values.astype(np.float64) + + # Density-coloured scatter (same idiom as §4.4.b and the CCFS rank + # comparison plot at line ~6692). High-overlap regions paint at the + # high-density end of viridis, revealing trend / correlation that + # pure alpha-blending obscures at large N. + try: + from scipy.stats import gaussian_kde + + # KDE in display coordinates: log space when axes are log-scaled + # so density reflects what the reader sees. + if log_scale: + x_kde = np.log10(np.clip(x_vals, 1e-9, None)) + y_kde = np.log10(np.clip(y_vals, 1e-9, None)) + else: + x_kde, y_kde = x_vals, y_vals + + max_kde_pts = 8000 + n_pts = len(x_kde) + if n_pts > max_kde_pts: + rng = np.random.default_rng(42) + sidx = rng.choice(n_pts, max_kde_pts, replace=False) + kde = gaussian_kde(np.vstack([x_kde[sidx], y_kde[sidx]])) + else: + kde = gaussian_kde(np.vstack([x_kde, y_kde])) + # Subsample the PLOTTED points to the standard cap (see + # plot_per_cell_intensity_vs_transcripts): ~15M cells overplot into a solid + # cloud, so ~10k render identically in ~1s vs ~35s. Density colour is still + # from the full-data KDE fit. + pidx = _subsample_idx(np.ones(n_pts, dtype=bool)) + zc = _kde_density_grid(kde, x_kde[pidx], y_kde[pidx]) + order = zc.argsort() # draw high-density points last (on top) + scatter = ax.scatter( + x_vals[pidx][order], + y_vals[pidx][order], + c=zc[order], + s=8, + cmap="viridis", + alpha=0.6, + rasterized=True, + ) + plt.colorbar(scatter, ax=ax, label="Cell density (relative)") + except Exception as _kde_err: + logging.warning( + f"Density coloring failed for nuclear texture scatter ({_kde_err}); " + "falling back to alpha-blended scatter." + ) + fidx = _subsample_idx(np.ones(len(x_vals), dtype=bool)) + ax.scatter( + x_vals[fidx], + y_vals[fidx], + alpha=0.3, + s=8, + color="C0", + rasterized=True, + ) + + # Add a horizontal line for the blur threshold + plt.axhline( + y=ccfs_low_texture_threshold, + color="red", + linestyle="--", + alpha=0.5, + label=f"Low Texture Threshold ({ccfs_low_texture_threshold})", + ) + + # Customize the plot + title = "Per-cell nuclear texture score vs transcript count" + if log_scale: + title += " (log scale)" + plt.title(title, fontsize=14, pad=20) + plt.ylabel("CCFS_DAPI (Nuclear Texture Score)", fontsize=12) + plt.xlabel("Transcript Counts", fontsize=12) + + # Set log scale for both axes if requested + if log_scale: + plt.xscale("log") + plt.xlabel("Transcript Counts (log scale)", fontsize=12) + plt.yscale("log") + plt.ylabel("CCFS_DAPI (Nuclear Texture Score) (log scale)", fontsize=12) + + # Add a grid for better readability + plt.grid(True, linestyle="--", alpha=0.7) + + # Add legend + plt.legend() + + # Adjust layout to prevent label cutoff + plt.tight_layout() + + # Save figure instead of showing + suffix = "_log" if log_scale else "" + plt.savefig( + figures_dir / f"nuclear_texture_vs_transcripts{suffix}.png", + dpi=300, + bbox_inches="tight", + ) + plt.close(fig) + + # Save data as CSV + if figures_source_dir is not None: + df_scatter_data = new_df[["transcript_counts", "CCFS_DAPI"]].copy() + df_scatter_data.to_csv( + figures_source_dir / f"nuclear_texture_vs_transcripts{suffix}.csv", + index=False, + ) + + # Print summary statistics + logging.info("Summary statistics:") + logging.info("Nuclear Texture Score (CCFS_DAPI):") + logging.info(new_df["CCFS_DAPI"].describe().round(4)) + logging.info("Transcript Counts:") + logging.info(new_df["transcript_counts"].describe()) + + +def plot_per_cell_intensity_vs_transcripts( + new_df, + figures_dir, + figures_source_dir, + log_scale=False, +): + """Per-cell mean intensity (DAPI / Boundary / IntRNA) vs transcript counts. + + Three separate scatter plots, same format as + `plot_nuclear_texture_vs_transcripts` so readers learn the axes once. + Used in §4.4.b "Stain intensity vs transcript count" alongside the + higher-level `intensity_transcript_correlation.png` heatmap. + + Skips silently when a channel intensity column is missing (e.g. DAPI-only + bundles produce no Boundary / IntRNA columns). + + Parameters + ---------- + new_df : pandas.DataFrame + Per-cell DataFrame; expected columns: `transcript_counts`, + `mean_intensity`, `mean_intensity_Boundary`, `mean_intensity_IntRNA`. + log_scale : bool, optional + If True, render both axes on a log scale (default False). + """ + channels = [ + ("DAPI", "mean_intensity"), + ("Boundary", "mean_intensity_Boundary"), + ("IntRNA", "mean_intensity_IntRNA"), + ] + + for ch_label, col in channels: + if col not in new_df.columns: + logging.warning( + f"Column {col!r} not in per-cell DataFrame — " + f"skipping {ch_label} intensity vs transcripts scatter." + ) + continue + + _valid = new_df[[col, "transcript_counts"]].dropna() + if len(_valid) == 0: + logging.warning( + f"No valid (non-NaN) cells for {ch_label} intensity vs transcripts — skipping." + ) + continue + + fig = plt.figure(figsize=(12, 6)) + ax = plt.gca() + x_vals = _valid["transcript_counts"].values.astype(np.float64) + y_vals = _valid[col].values.astype(np.float64) + + # Density-colored scatter — high-overlap regions paint in the + # high-density end of the colormap, making trend / correlation + # readable even at large N where simple alpha blending saturates. + # Pattern matches plot_ccfs_vs_roi_comparison (line ~6692). + try: + from scipy.stats import gaussian_kde + + # Compute KDE in display coordinates: log space when the axes + # are log-scaled, so the density gradient reflects what the + # reader sees rather than the raw coordinate distance. + if log_scale: + x_kde = np.log10(np.clip(x_vals, 1e-9, None)) + y_kde = np.log10(np.clip(y_vals, 1e-9, None)) + else: + x_kde, y_kde = x_vals, y_vals + + max_kde_pts = 8000 + n_pts = len(x_kde) + if n_pts > max_kde_pts: + rng = np.random.default_rng(42) + sidx = rng.choice(n_pts, max_kde_pts, replace=False) + kde = gaussian_kde(np.vstack([x_kde[sidx], y_kde[sidx]])) + else: + kde = gaussian_kde(np.vstack([x_kde, y_kde])) + # Subsample the PLOTTED points to the standard scatter cap. A + # density-colored scatter of ~15M cells overplots into a solid cloud at + # display resolution, so ~10k points render the identical cloud in ~1s + # vs ~32s/channel (measured 96s total). Density colour is still computed + # per plotted point from the full-data KDE fit; matches _subsample_idx as + # used by roi_focus_vs_intensity and the other density scatters. + pidx = _subsample_idx(np.ones(n_pts, dtype=bool)) + zc = _kde_density_grid(kde, x_kde[pidx], y_kde[pidx]) + order = zc.argsort() # draw high-density points last (on top) + scatter = ax.scatter( + x_vals[pidx][order], + y_vals[pidx][order], + c=zc[order], + s=8, + cmap="viridis", + alpha=0.6, + rasterized=True, + ) + plt.colorbar(scatter, ax=ax, label="Cell density (relative)") + except Exception as _kde_err: + logging.warning( + f"Density coloring failed for {ch_label} ({_kde_err}); " + "falling back to alpha-blended scatter." + ) + fidx = _subsample_idx(np.ones(len(x_vals), dtype=bool)) + ax.scatter( + x_vals[fidx], + y_vals[fidx], + alpha=0.3, + s=8, + color="C0", + rasterized=True, + ) + + title = f"Per-cell {ch_label} mean intensity vs transcript count" + if log_scale: + title += " (log scale)" + plt.title(title, fontsize=14, pad=20) + plt.ylabel(f"{ch_label} mean intensity (16-bit counts)", fontsize=12) + plt.xlabel("Transcript counts", fontsize=12) + + if log_scale: + plt.xscale("log") + plt.yscale("log") + plt.xlabel("Transcript counts (log scale)", fontsize=12) + plt.ylabel( + f"{ch_label} mean intensity (16-bit counts, log scale)", fontsize=12 + ) + + plt.grid(True, linestyle="--", alpha=0.7) + plt.tight_layout() + + suffix = "_log" if log_scale else "" + fname = f"{ch_label.lower()}_intensity_vs_transcripts{suffix}" + plt.savefig(figures_dir / f"{fname}.png", dpi=300, bbox_inches="tight") + plt.close(fig) + + if figures_source_dir is not None: + _valid.to_csv(figures_source_dir / f"{fname}.csv", index=False) + + logging.info( + f"Saved {fname}.png (n={len(_valid):,}, " + f"median intensity={_valid[col].median():.0f})" + ) + + +def plot_gmm_focus_vs_transcripts( + new_df, + figures_dir, + figures_source_dir, +): + """Scatter plot of tile-level GMM focus score vs transcript counts, coloured by blur class. + + Mirrors plot_nuclear_texture_vs_transcripts but uses the tile-level Laplacian + GMM focus score (DAPI_RFSnorm_roi) propagated to cells, with points coloured + by GMM blur classification. + """ + focus_col = "DAPI_RFSnorm_roi" + tx_col = "transcript_counts" + gmm_col = "is_blurred_gmm_2d_roi" + + if focus_col not in new_df.columns or tx_col not in new_df.columns: + logging.warning( + "Missing %s or %s — skipping GMM focus vs transcripts plot.", + focus_col, + tx_col, + ) + return + + df = new_df[ + [tx_col, focus_col] + ([gmm_col] if gmm_col in new_df.columns else []) + ].dropna() + if df.empty: + return + + has_gmm = gmm_col in df.columns + fig, ax = plt.subplots(figsize=(12, 6)) + + if has_gmm: + sharp = df[~df[gmm_col].astype(bool)] + blurred = df[df[gmm_col].astype(bool)] + ax.scatter( + sharp[tx_col], + sharp[focus_col], + alpha=0.3, + s=8, + color="#1f77b4", + label=f"In Focus ({len(sharp):,})", + rasterized=True, + ) + ax.scatter( + blurred[tx_col], + blurred[focus_col], + alpha=0.3, + s=8, + color="#d62728", + label=f"Blurred ({len(blurred):,})", + rasterized=True, + ) + ax.legend(fontsize=10) + else: + ax.scatter(df[tx_col], df[focus_col], alpha=0.3, s=8, rasterized=True) + + ax.set_xscale("log") + ax.set_yscale("log") + ax.set_xlabel("Transcript Counts (log scale)", fontsize=12) + ax.set_ylabel("Tile Focus Score (DAPI_RFSnorm_roi, log scale)", fontsize=12) + ax.set_title("GMM Tile Focus Score vs Transcript Counts", fontsize=14, pad=20) + ax.grid(True, linestyle="--", alpha=0.7) + + plt.tight_layout() + plt.savefig( + figures_dir / "gmm_focus_vs_transcripts.png", dpi=300, bbox_inches="tight" + ) + plt.close(fig) + + # Save source data + if figures_source_dir is not None: + src_cols = [tx_col, focus_col] + if has_gmm: + src_cols.append(gmm_col) + df[src_cols].to_csv( + figures_source_dir / "gmm_focus_vs_transcripts.csv", index=False + ) + + logging.info("GMM focus vs transcripts: %d cells plotted.", len(df)) + + +def plot_ccfs_vs_roi_comparison(new_df, figures_dir, figures_source_dir): + """ + Create comparison plots between CCFS (nuclei-based) and cell-independent tile-based focus scores. + + Parameters: + ----------- + new_df : pandas DataFrame + DataFrame containing both CCFS_DAPI and DAPI_RFSnorm_roi columns + (DAPI_RFSnorm_roi contains cell-independent grid ROI focus scores transferred to cells) + figures_dir : Path + Directory to save figures + figures_source_dir : Path + Directory to save source data + """ + # Create figure with subplots + fig, axes = plt.subplots(2, 2, figsize=(15, 15)) + + # Plot 1: Ranked scatter plot with density coloring + ax = axes[0, 0] + # Calculate ranks (higher rank = higher focus score) + ccfs_ranks = new_df["CCFS_DAPI"].rank(method="average") + roi_ranks = new_df["DAPI_RFSnorm_roi"].rank(method="average") + + # Create scatter plot with density coloring (subsample for KDE performance) + try: + from scipy.stats import gaussian_kde + + valid = ccfs_ranks.dropna().index.intersection(roi_ranks.dropna().index) + x_vals, y_vals = ccfs_ranks[valid].values, roi_ranks[valid].values + max_kde_pts = 8000 + if len(x_vals) > max_kde_pts: + rng = np.random.default_rng(42) + sidx = rng.choice(len(x_vals), max_kde_pts, replace=False) + xy_sub = np.vstack([x_vals[sidx], y_vals[sidx]]) + kde = gaussian_kde(xy_sub) + z = _kde_density_grid(kde, x_vals, y_vals) + else: + z = gaussian_kde(np.vstack([x_vals, y_vals]))(np.vstack([x_vals, y_vals])) + idx = z.argsort() + scatter = ax.scatter( + x_vals[idx], + y_vals[idx], + c=z[idx], + s=1, + cmap="viridis", + alpha=0.6, + rasterized=True, + ) + plt.colorbar(scatter, ax=ax, label="Density") + except (ImportError, Exception): + # Fallback if scipy not available or density calculation fails + ax.scatter(ccfs_ranks, roi_ranks, alpha=0.3, s=1, marker="o", rasterized=True) + + ax.set_xlabel("CCFS_DAPI Rank (Nuclei-based)", fontsize=12) + ax.set_ylabel("DAPI_RFSnorm_roi Rank (Cell-Independent Tile-based)", fontsize=12) + ax.set_title("CCFS vs Cell-Independent Tile Focus Score (Ranked)", fontsize=14) + ax.grid(True, linestyle="--", alpha=0.7) + + # Add correlation coefficient (Spearman rank-based correlation) + correlation = new_df["CCFS_DAPI"].corr( + new_df["DAPI_RFSnorm_roi"], method="spearman" + ) + ax.text( + 0.05, + 0.95, + f"Spearman Correlation: {correlation:.4f}", + transform=ax.transAxes, + fontsize=12, + verticalalignment="top", + bbox=dict(boxstyle="round", facecolor="wheat", alpha=0.5), + ) + + # Plot 2: Density plot overlay + ax = axes[0, 1] + ax.scatter( + new_df["CCFS_DAPI"], + new_df["DAPI_RFSnorm_roi"], + alpha=0.1, + s=0.5, + marker="o", + c="blue", + label="Data points", + rasterized=True, + ) + # Add 2D density contour (subsample for KDE performance) + try: + from scipy.stats import gaussian_kde + + valid_mask = new_df["CCFS_DAPI"].notna() & new_df["DAPI_RFSnorm_roi"].notna() + x_vals = new_df.loc[valid_mask, "CCFS_DAPI"].values + y_vals = new_df.loc[valid_mask, "DAPI_RFSnorm_roi"].values + max_kde_pts = 8000 + if len(x_vals) > max_kde_pts: + rng = np.random.default_rng(42) + sidx = rng.choice(len(x_vals), max_kde_pts, replace=False) + kde = gaussian_kde(np.vstack([x_vals[sidx], y_vals[sidx]])) + z = _kde_density_grid(kde, x_vals, y_vals) + else: + z = gaussian_kde(np.vstack([x_vals, y_vals]))(np.vstack([x_vals, y_vals])) + idx = z.argsort() + ax.scatter( + x_vals[idx], + y_vals[idx], + c=z[idx], + s=1, + cmap="viridis", + alpha=0.6, + rasterized=True, + ) + except (ImportError, Exception): + pass # Skip density overlay if unavailable + ax.set_xlabel("CCFS_DAPI (Nuclei-based)", fontsize=12) + ax.set_ylabel("DAPI_RFSnorm_roi (Cell-Independent Tile-based)", fontsize=12) + ax.set_title("CCFS vs Cell-Independent Tile Focus Score (Density)", fontsize=14) + ax.grid(True, linestyle="--", alpha=0.7) + + # Plot 3: Agreement/disagreement classification + ax = axes[1, 0] + # Prefer GMM 2D if available, otherwise use threshold-based + has_gmm_2d = "is_blurred_gmm_2d_roi" in new_df.columns + has_threshold = "is_blurred_roi" in new_df.columns + + if "is_low_nuclear_texture" in new_df.columns and (has_gmm_2d or has_threshold): + # Use GMM 2D if available, otherwise threshold-based + if has_gmm_2d: + roi_blurred_col = "is_blurred_gmm_2d_roi" + roi_high_focus = ~new_df["is_blurred_gmm_2d_roi"] + method_label = "2D GMM" + else: + roi_blurred_col = "is_blurred_roi" + roi_high_focus = new_df["is_high_focus_roi"] + method_label = "Threshold" + + # Create agreement categories + agreement = new_df["is_low_nuclear_texture"] == new_df[roi_blurred_col] + agree_high = new_df["is_high_nuclear_texture"] & roi_high_focus + agree_low = new_df["is_low_nuclear_texture"] & new_df[roi_blurred_col] + disagree = ~agreement + + # Plot + ax.scatter( + new_df.loc[agree_high, "CCFS_DAPI"], + new_df.loc[agree_high, "DAPI_RFSnorm_roi"], + alpha=0.3, + s=1, + c="green", + label="Both high focus", + marker="o", + rasterized=True, + ) + ax.scatter( + new_df.loc[agree_low, "CCFS_DAPI"], + new_df.loc[agree_low, "DAPI_RFSnorm_roi"], + alpha=0.3, + s=1, + c="red", + label="Both blurred", + marker="o", + rasterized=True, + ) + ax.scatter( + new_df.loc[disagree, "CCFS_DAPI"], + new_df.loc[disagree, "DAPI_RFSnorm_roi"], + alpha=0.5, + s=2, + c="orange", + label="Disagree", + marker="x", + rasterized=True, + ) + ax.set_xlabel("CCFS_DAPI (Nuclei-based)", fontsize=12) + ax.set_ylabel("DAPI_RFSnorm_roi (Cell-Independent Tile-based)", fontsize=12) + ax.set_title( + f"Classification Agreement (CCFS vs Tile-based {method_label})", fontsize=14 + ) + ax.legend(fontsize=10) + ax.grid(True, linestyle="--", alpha=0.7) + + # Add agreement percentage + agreement_rate = agreement.sum() / len(new_df) * 100 + ax.text( + 0.05, + 0.95, + f"Agreement: {agreement_rate:.2f}%", + transform=ax.transAxes, + fontsize=12, + verticalalignment="top", + bbox=dict(boxstyle="round", facecolor="wheat", alpha=0.5), + ) + + # Plot 4: Distribution comparison + ax = axes[1, 1] + ax.hist( + new_df["CCFS_DAPI"].dropna(), + bins=50, + alpha=0.5, + label="CCFS_DAPI (Nuclei-based)", + color="blue", + density=True, + ) + ax.hist( + new_df["DAPI_RFSnorm_roi"].dropna(), + bins=50, + alpha=0.5, + label="DAPI_RFSnorm_roi (Cell-Independent Tile)", + color="red", + density=True, + ) + ax.set_xlabel("Focus Score", fontsize=12) + ax.set_ylabel("Density", fontsize=12) + ax.set_title("Distribution Comparison (CCFS vs Cell-Independent Tile)", fontsize=14) + ax.legend(fontsize=10) + ax.grid(True, linestyle="--", alpha=0.7) + + plt.tight_layout() + plt.savefig( + figures_dir / "ccfs_vs_roi_comparison.pdf", dpi=300, bbox_inches="tight" + ) + plt.savefig( + figures_dir / "ccfs_vs_roi_comparison.png", dpi=300, bbox_inches="tight" + ) + plt.close(fig) + + # Save data as CSV (including ranks) + df_comparison = new_df[["CCFS_DAPI", "DAPI_RFS_roi", "DAPI_RFSnorm_roi"]].copy() + # Add ranks for analysis + df_comparison["CCFS_DAPI_rank"] = new_df["CCFS_DAPI"].rank(method="average") + df_comparison["DAPI_RFSnorm_roi_rank"] = new_df["DAPI_RFSnorm_roi"].rank( + method="average" + ) + + # Add classification columns - prefer GMM 2D, include threshold-based if available + if "is_low_nuclear_texture" in new_df.columns: + df_comparison["is_low_nuclear_texture"] = new_df["is_low_nuclear_texture"] + + # GMM 2D classification (preferred) + if "is_blurred_gmm_2d_roi" in new_df.columns: + df_comparison["is_blurred_gmm_2d_roi"] = new_df["is_blurred_gmm_2d_roi"] + df_comparison["classification_agreement_gmm_2d"] = ( + new_df["is_low_nuclear_texture"] == new_df["is_blurred_gmm_2d_roi"] + ) + if "blur_prob_gmm_2d_roi" in new_df.columns: + df_comparison["blur_prob_gmm_2d_roi"] = new_df["blur_prob_gmm_2d_roi"] + + # Threshold-based classification (for comparison) + if "is_blurred_roi" in new_df.columns: + df_comparison["is_blurred_roi"] = new_df["is_blurred_roi"] + df_comparison["classification_agreement"] = ( + new_df["is_low_nuclear_texture"] == new_df["is_blurred_roi"] + ) + + if figures_source_dir is not None: + df_comparison.to_csv( + figures_source_dir / "ccfs_vs_roi_comparison.csv", index=False + ) + + +def plot_spatial_comparison(new_df, myData, figures_dir, figures_source_dir): + """ + Create spatial comparison plots showing both nuclei-based and tile-based focus scores. + + Parameters: + ----------- + new_df : pandas DataFrame + DataFrame containing spatial and focus score data + myData : pandas DataFrame + DataFrame with nuclei-based measurements (for spatial coordinates) + figures_dir : Path + Directory to save figures + figures_source_dir : Path + Directory to save source data + """ + # Compute figure size from data extents to avoid empty white space + _x_range = ( + myData["centroid-1"].max() - myData["centroid-1"].min() + if "centroid-1" in myData.columns + else 1 + ) + _y_range = ( + myData["centroid-0"].max() - myData["centroid-0"].min() + if "centroid-0" in myData.columns + else 1 + ) + _data_aspect = _y_range / _x_range if _x_range > 0 else 1.0 + _pw = 7 + _ph = max(4, _pw * _data_aspect) + fig, axes = plt.subplots(1, 2, figsize=(2 * _pw + 2, _ph)) + + # Plot 1: Nuclei-based CCFS spatial + ax = axes[0] + if "centroid-1" in myData.columns and "centroid-0" in myData.columns: + # Filter out NaN values for plotting + valid_mask = myData["CCFS_DAPI"].notna() + if valid_mask.sum() > 0: + # Use CCFS_DAPI directly (not negated) with auto-scaling + ccfs_values = myData.loc[valid_mask, "CCFS_DAPI"] + scatter = ax.scatter( + myData.loc[valid_mask, "centroid-1"], + -myData.loc[valid_mask, "centroid-0"], + s=0.1, + c=ccfs_values, + cmap="viridis", + rasterized=True, + ) + ax.set_title("Nuclei-based Focus Score (CCFS_DAPI)", fontsize=14) + ax.set_facecolor("black") + ax.set_aspect("equal") + plt.colorbar(scatter, ax=ax, label="CCFS_DAPI") + else: + ax.text( + 0.5, + 0.5, + "No valid CCFS_DAPI data", + transform=ax.transAxes, + ha="center", + va="center", + fontsize=14, + bbox=dict(boxstyle="round", facecolor="lightgray", alpha=0.5), + ) + ax.set_facecolor("black") + ax.set_aspect("equal") + + # Plot 2: Tile-based RFS spatial (colored by GMM 2D classification if available) + ax = axes[1] + if "DAPI_RFSnorm_roi" in new_df.columns: + # Prefer GMM 2D classification for coloring, otherwise use raw focus scores + has_gmm_2d = "is_blurred_gmm_2d_roi" in new_df.columns + valid_mask = new_df["DAPI_RFSnorm_roi"].notna() + + if has_gmm_2d: + # Color by GMM 2D classification (blurred vs in-focus) + gmm_valid_mask = valid_mask & new_df["is_blurred_gmm_2d_roi"].notna() + if gmm_valid_mask.sum() > 0: + # Color: red for blurred, blue for in-focus + colors = [ + "red" if blurred else "blue" + for blurred in new_df.loc[gmm_valid_mask, "is_blurred_gmm_2d_roi"] + ] + ax.scatter( + new_df.loc[gmm_valid_mask, "x"], + -new_df.loc[gmm_valid_mask, "y"], + s=0.1, + c=colors, + alpha=0.5, + rasterized=True, + ) + ax.set_title( + "Tile-based Focus Score (GMM 2D Classification)", fontsize=14 + ) + # Add legend + from matplotlib.patches import Patch + + legend_elements = [ + Patch(facecolor="blue", label="In-Focus (2D GMM)"), + Patch(facecolor="red", label="Blurred (2D GMM)"), + ] + ax.legend(handles=legend_elements, fontsize=10, loc="upper right") + else: + ax.text( + 0.5, + 0.5, + "No valid GMM 2D classification data", + transform=ax.transAxes, + ha="center", + va="center", + fontsize=14, + bbox=dict(boxstyle="round", facecolor="lightgray", alpha=0.5), + ) + else: + # Fallback: use raw focus scores with color scale + if valid_mask.sum() > 0: + roi_values = new_df.loc[valid_mask, "DAPI_RFSnorm_roi"] + scatter = ax.scatter( + new_df.loc[valid_mask, "x"], + -new_df.loc[valid_mask, "y"], + s=0.1, + c=roi_values, + cmap="viridis", + rasterized=True, + ) + ax.set_title("Tile-based Focus Score (DAPI_RFSnorm_roi)", fontsize=14) + plt.colorbar(scatter, ax=ax, label="DAPI_RFSnorm_roi") + else: + ax.text( + 0.5, + 0.5, + "No valid DAPI_RFSnorm_roi data", + transform=ax.transAxes, + ha="center", + va="center", + fontsize=14, + bbox=dict(boxstyle="round", facecolor="lightgray", alpha=0.5), + ) + + ax.set_facecolor("black") + ax.set_aspect("equal") + + plt.tight_layout() + plt.savefig( + figures_dir / "spatial_comparison_nuclei_vs_roi.pdf", + dpi=300, + bbox_inches="tight", + ) + plt.savefig( + figures_dir / "spatial_comparison_nuclei_vs_roi.png", + dpi=300, + bbox_inches="tight", + ) + plt.close(fig) + + # Save data as CSV + if "centroid-1" in myData.columns: + df_spatial_comparison = pd.DataFrame( + { + "x_nuclei": myData["centroid-1"], + "y_nuclei": -myData["centroid-0"], + "CCFS_DAPI": myData["CCFS_DAPI"], + } + ) + if "DAPI_RFSnorm_roi" in new_df.columns: + # Try to merge on coordinates + df_spatial_comparison = df_spatial_comparison.merge( + new_df[["x", "y", "DAPI_RFSnorm_roi"]], + left_on=["x_nuclei", "y_nuclei"], + right_on=["x", "y"], + how="left", + ) + if figures_source_dir is not None: + df_spatial_comparison.to_csv( + figures_source_dir / "spatial_comparison_nuclei_vs_roi.csv", + index=False, + ) + + +def save_cell_qc_metrics(new_df, outdir, roi_size=None): + """Save QC metrics to CSV and JSON""" + + # Move cell_id to first position + cell_id_col = new_df.pop("cell_id") + new_df.insert(0, "cell_id", cell_id_col) + # Save full dataset + new_df.to_csv(outdir / "image_qc_cell_metrics.csv", index=False) + + # Generate QC summary statistics + qc_metrics = { + "total_cells": len(new_df), + "cells_with_transcripts": len(new_df[new_df["transcript_counts"] > 0]), + "mean_transcript_count": float(new_df["transcript_counts"].mean()), + "median_transcript_count": float(new_df["transcript_counts"].median()), + "std_transcript_count": float(new_df["transcript_counts"].std()), + "mean_ccfs_dapi": float(new_df["CCFS_DAPI"].mean()), + "cells_high_nuclear_texture": len(new_df[new_df["is_high_nuclear_texture"]]), + "cells_low_nuclear_texture": len(new_df[new_df["is_low_nuclear_texture"]]), + "cells_near_edge": len(new_df[new_df["is_near_edge"]]), + "cells_near_holes": len(new_df[new_df["is_near_hole"]]), + "cells_with_dense_intensity_regions": len( + new_df[new_df["has_dense_intensity_regions"]] + ), + "unique_dense_intensity_regions": sorted( + new_df["Dense-Intensity-Region-ID"].unique().tolist() + ) + if "Dense-Intensity-Region-ID" in new_df.columns + else [], + "total_dense_intensity_regions": len( + new_df[new_df["Dense-Intensity-Region-ID"] > 0][ + "Dense-Intensity-Region-ID" + ].unique() + ) + if "Dense-Intensity-Region-ID" in new_df.columns + else 0, + "clusters_present": sorted(new_df["Cluster_kmeans10"].unique().tolist()), + "segmentation_methods": sorted(new_df["segmentation_method"].unique().tolist()), + } + + # Add tile-based metrics if available + if "DAPI_RFS_roi" in new_df.columns: + # Save ROI size used for this analysis + if roi_size is not None: + qc_metrics["roi_size_used"] = int(roi_size) + qc_metrics["mean_dapi_rfs_roi"] = float(new_df["DAPI_RFS_roi"].mean()) + qc_metrics["median_dapi_rfs_roi"] = float(new_df["DAPI_RFS_roi"].median()) + qc_metrics["mean_dapi_rfsnorm_roi"] = float(new_df["DAPI_RFSnorm_roi"].mean()) + qc_metrics["median_dapi_rfsnorm_roi"] = float( + new_df["DAPI_RFSnorm_roi"].median() + ) + + # GMM 2D metrics (preferred method) + if "is_blurred_gmm_2d_roi" in new_df.columns: + qc_metrics["cells_high_focus_gmm_2d_roi"] = len( + new_df[~new_df["is_blurred_gmm_2d_roi"]] + ) + qc_metrics["cells_blurred_gmm_2d_roi"] = len( + new_df[new_df["is_blurred_gmm_2d_roi"]] + ) + if "blur_prob_gmm_2d_roi" in new_df.columns: + valid_probs = new_df["blur_prob_gmm_2d_roi"].dropna() + if len(valid_probs) > 0: + qc_metrics["mean_blur_prob_gmm_2d_roi"] = float(valid_probs.mean()) + qc_metrics["median_blur_prob_gmm_2d_roi"] = float( + valid_probs.median() + ) + + # Threshold-based metrics (for comparison/backward compatibility) + if "is_blurred_roi" in new_df.columns: + qc_metrics["cells_high_focus_roi"] = len( + new_df[new_df["is_high_focus_roi"]] + ) + qc_metrics["cells_blurred_roi"] = len(new_df[new_df["is_blurred_roi"]]) + + # Comparison metrics between nuclei-based and tile-based methods + # Calculate Spearman rank-based correlation between CCFS_DAPI and DAPI_RFSnorm_roi + # Spearman is better suited for different scales and non-linear relationships + if "CCFS_DAPI" in new_df.columns and "DAPI_RFSnorm_roi" in new_df.columns: + correlation = new_df["CCFS_DAPI"].corr( + new_df["DAPI_RFSnorm_roi"], method="spearman" + ) + qc_metrics["ccfs_vs_rfs_correlation"] = ( + float(correlation) if not np.isnan(correlation) else None + ) + qc_metrics["ccfs_vs_rfs_correlation_method"] = "spearman" + + # Calculate agreement rate - prefer GMM 2D, include threshold-based for comparison + if "is_low_nuclear_texture" in new_df.columns: + # GMM 2D agreement (preferred) + if "is_blurred_gmm_2d_roi" in new_df.columns: + agreement_gmm_2d = ( + new_df["is_low_nuclear_texture"] + == new_df["is_blurred_gmm_2d_roi"] + ).sum() + agreement_rate_gmm_2d = agreement_gmm_2d / len(new_df) + qc_metrics["classification_agreement_gmm_2d"] = float( + agreement_rate_gmm_2d + ) + qc_metrics["classification_agreement_gmm_2d_count"] = int( + agreement_gmm_2d + ) + qc_metrics["classification_disagreement_gmm_2d_count"] = int( + len(new_df) - agreement_gmm_2d + ) + + # Threshold-based agreement (for comparison) + if "is_blurred_roi" in new_df.columns: + agreement = ( + new_df["is_low_nuclear_texture"] == new_df["is_blurred_roi"] + ).sum() + agreement_rate = agreement / len(new_df) + qc_metrics["classification_agreement"] = float(agreement_rate) + qc_metrics["classification_agreement_count"] = int(agreement) + qc_metrics["classification_disagreement_count"] = int( + len(new_df) - agreement + ) + + # Save metrics + with open(outdir / "image_qc_cell_metrics.json", "w") as f: + json.dump(qc_metrics, f, indent=2) + + # Save ROI size to simple text file for easy retrieval + if roi_size is not None: + with open(outdir / "roi_size.txt", "w") as f: + f.write(f"{roi_size}\n") + + logging.info("\n=== IMAGE QC SUMMARY ===") + logging.info(f"Total cells analyzed: {qc_metrics['total_cells']:,}") + logging.info(f"Cells with transcripts: {qc_metrics['cells_with_transcripts']:,}") + logging.info(f"Mean transcript count: {qc_metrics['mean_transcript_count']:.1f}") + logging.info(f"Mean CCFS DAPI: {qc_metrics['mean_ccfs_dapi']:.6f}") + logging.info( + f"Cells high nuclear texture: {qc_metrics['cells_high_nuclear_texture']:,}" + ) + logging.info( + f"Cells low nuclear texture: {qc_metrics['cells_low_nuclear_texture']:,}" + ) + logging.info(f"Cells near edge: {qc_metrics['cells_near_edge']:,}") + logging.info(f"Cells near holes: {qc_metrics['cells_near_holes']:,}") + logging.info( + f"Cells with dense intensity regions: {qc_metrics['cells_with_dense_intensity_regions']:,}" + ) + if ( + "total_dense_intensity_regions" in qc_metrics + and qc_metrics["total_dense_intensity_regions"] > 0 + ): + logging.info( + f"Total unique dense intensity regions detected: {qc_metrics['total_dense_intensity_regions']}" + ) + logging.info( + f" (Region IDs: {qc_metrics['unique_dense_intensity_regions'][:10]}{'...' if len(qc_metrics['unique_dense_intensity_regions']) > 10 else ''})" + ) + logging.info(f"Clusters present: {qc_metrics['clusters_present']}") + logging.info(f"Segmentation methods: {qc_metrics['segmentation_methods']}") + + # Print tile-based metrics if available + if "mean_dapi_rfs_roi" in qc_metrics: + logging.info("\n=== TILE-BASED FOCUS SCORE SUMMARY ===") + logging.info(f"Mean DAPI RFS (tile): {qc_metrics['mean_dapi_rfs_roi']:.6f}") + logging.info( + f"Mean DAPI RFS normalized (tile): {qc_metrics['mean_dapi_rfsnorm_roi']:.6f}" + ) + + # GMM 2D metrics (preferred) + if "cells_blurred_gmm_2d_roi" in qc_metrics: + logging.info("\n--- 2D GMM Classification (Preferred) ---") + logging.info( + f"Cells with high focus (2D GMM): {qc_metrics['cells_high_focus_gmm_2d_roi']:,}" + ) + logging.info( + f"Cells blurred (2D GMM): {qc_metrics['cells_blurred_gmm_2d_roi']:,}" + ) + if "mean_blur_prob_gmm_2d_roi" in qc_metrics: + logging.info( + f"Mean blur probability (2D GMM): {qc_metrics['mean_blur_prob_gmm_2d_roi']:.4f}" + ) + + # Threshold-based metrics (for comparison) + if "cells_blurred_roi" in qc_metrics: + logging.info("\n--- Threshold-based Classification (Comparison) ---") + logging.info( + f"Cells with high focus (Threshold): {qc_metrics['cells_high_focus_roi']:,}" + ) + logging.info( + f"Cells blurred (Threshold): {qc_metrics['cells_blurred_roi']:,}" + ) + + if ( + "ccfs_vs_rfs_correlation" in qc_metrics + and qc_metrics["ccfs_vs_rfs_correlation"] is not None + ): + logging.info("\n=== METHOD COMPARISON ===") + logging.info( + f"Spearman Correlation (CCFS vs RFS): {qc_metrics['ccfs_vs_rfs_correlation']:.4f}" + ) + + # GMM 2D agreement (preferred) + if "classification_agreement_gmm_2d" in qc_metrics: + logging.info("\n--- CCFS vs 2D GMM Agreement (Preferred) ---") + logging.info( + f"Classification agreement: {qc_metrics['classification_agreement_gmm_2d'] * 100:.2f}%" + ) + logging.info( + f" - Agreeing cells: {qc_metrics['classification_agreement_gmm_2d_count']:,}" + ) + logging.info( + f" - Disagreeing cells: {qc_metrics['classification_disagreement_gmm_2d_count']:,}" + ) + + # Threshold-based agreement (for comparison) + if "classification_agreement" in qc_metrics: + logging.info("\n--- CCFS vs Threshold Agreement (Comparison) ---") + logging.info( + f"Classification agreement: {qc_metrics['classification_agreement'] * 100:.2f}%" + ) + logging.info( + f" - Agreeing cells: {qc_metrics['classification_agreement_count']:,}" + ) + logging.info( + f" - Disagreeing cells: {qc_metrics['classification_disagreement_count']:,}" + ) + + +def generate_cell_figures( + data, + new_df, + myData, + figures_dir, + figures_source_dir, + roi_threshold=None, + roi_intensity_threshold=None, + ccfs_low_texture_threshold=DEFAULT_CCFS_LOW_TEXTURE_THRESHOLD, +): + """ + Generate cell-based figures from merged data using multithreading. + + Parameters: + ----------- + data : dict + Dictionary with paths and directories + new_df : pandas DataFrame + Merged DataFrame with cell data and ROI mappings + myData : pandas DataFrame + Nuclei-based measurements DataFrame + figures_dir : Path + Directory to save figures + figures_source_dir : Path + Directory to save source data + roi_threshold : float, optional + Tile-based blur threshold for display + roi_intensity_threshold : float, optional + Tile intensity threshold for display + """ + if figures_source_dir is not None: + figures_source_dir.mkdir(parents=True, exist_ok=True) + + logging.info("Generating cell-based figures...") + + def _plot1_nuclear_texture_proportions(): + logging.info(" - Nuclear texture proportions (CCFS)...") + plot_nuclear_texture_proportions( + new_df, + figures_dir, + figures_source_dir, + texture_threshold=ccfs_low_texture_threshold, + ) + + def _plot2_blur_proportions_roi(): + if ( + "is_blurred_gmm_2d_roi" in new_df.columns + or "is_blurred_roi" in new_df.columns + ): + logging.info(" - Blur score proportions (tile-based)...") + roi_threshold_display = roi_threshold if roi_threshold is not None else -1.0 + plot_tile_blur_proportions_roi._intensity_threshold = ( + roi_intensity_threshold if roi_intensity_threshold is not None else 20.0 + ) + plot_tile_blur_proportions_roi( + new_df, + figures_dir, + figures_source_dir, + roi_threshold=roi_threshold_display, + ) + + def _plot3_nuclear_texture_density(): + logging.info(" - Nuclear texture density...") + plot_nuclear_texture_density( + new_df, + figures_dir, + figures_source_dir, + ccfs_low_texture_threshold=ccfs_low_texture_threshold, + ) + + def _plot5_ccfs_vs_roi(): + if "DAPI_RFSnorm_roi" in new_df.columns: + logging.info(" - CCFS vs tile comparison...") + plot_ccfs_vs_roi_comparison(new_df, figures_dir, figures_source_dir) + + def _plot6_spatial_comparison(): + logging.info(" - Spatial comparison...") + plot_spatial_comparison(new_df, myData, figures_dir, figures_source_dir) + + def _plot7_cell_focus_distribution(): + logging.info(" - Cell-level focus score distribution...") + plot_cell_focus_distribution(new_df, figures_dir, figures_source_dir) + + def _plot8_gmm_focus_vs_transcripts(): + logging.info(" - GMM focus vs transcripts...") + plot_gmm_focus_vs_transcripts(new_df, figures_dir, figures_source_dir) + + def _plot9_tile_focus_gmm_spatial(): + logging.info(" - Tile-focus GMM spatial...") + plot_tile_focus_gmm_spatial(new_df, figures_dir, figures_source_dir) + + def _plot9b_cell_flagged_maps_combined(): + logging.info(" - Cell-flagged maps (combined)...") + plot_cell_flagged_maps_combined(new_df, figures_dir, figures_source_dir) + + def _plot10_blur_prob_density_by_cluster(): + logging.info(" - Blur probability density by cluster...") + plot_blur_prob_density_by_cluster(new_df, figures_dir, figures_source_dir) + + def _plot11_intensity_transcript_correlation(): + logging.info(" - Intensity-transcript correlation heatmap...") + plot_intensity_transcript_correlation(new_df, figures_dir, figures_source_dir) + + def _plot12_per_cell_intensity_vs_transcripts(): + logging.info( + " - Per-cell intensity vs transcripts (DAPI / Boundary / IntRNA)..." + ) + plot_per_cell_intensity_vs_transcripts( + new_df, figures_dir, figures_source_dir, log_scale=True + ) + + tasks = [ + _plot1_nuclear_texture_proportions, + _plot2_blur_proportions_roi, + # Latent figures disabled 2026-05-20 (user request) — function defs + # retained above as latent code: _plot3_nuclear_texture_density, + # _plot5_ccfs_vs_roi, _plot6_spatial_comparison, + # _plot7_cell_focus_distribution, _plot10_blur_prob_density_by_cluster. + _plot8_gmm_focus_vs_transcripts, + _plot9_tile_focus_gmm_spatial, + _plot9b_cell_flagged_maps_combined, + _plot11_intensity_transcript_correlation, + _plot12_per_cell_intensity_vs_transcripts, + ] + + _run_figure_pool(tasks, phase="cell-based figures") + + logging.info("All cell-based figures generated successfully!") + + +# ===== QUARTO FIGURE GENERATION (from image_qc_processing.py) ===== + + +def generate_all_figures( + data, + df_spatial, + new_df, + myData, + small0, + small1, + small2, + distance_map, + distance_map2, + whole_sample, + holes, + artefacts, + ccfs_low_texture_threshold=DEFAULT_CCFS_LOW_TEXTURE_THRESHOLD, + multistain_whole_sample=None, + multistain_distance_map=None, + multistain_distance_map2=None, + figure_source_tables=False, +): + """Generate ALL 13 Quarto-required figures using multithreading, and save the data used for each plot as CSV.""" + figures_dir = data["figures_dir"] + figures_source_dir = ( + (figures_dir / "figures_source") if figure_source_tables else None + ) + if figures_source_dir is not None: + figures_source_dir.mkdir(parents=True, exist_ok=True) + # if df_spatial['In-Area-with-Artefact'] have a single unique value, use that value, otherwise use 0.5 + if len(df_spatial["In-Area-with-Artefact"].unique()) == 1: + iawa_vmax = df_spatial["In-Area-with-Artefact"].unique()[0] + else: + iawa_vmax = 0.5 + + # §2.4 distance figures use the multi-stain extent mask + its distance maps when present, + # matching the edge/hole burden metrics in save_roi_qc_metrics. DAPI fallback (None on + # DAPI-only bundles) keeps those bundles byte-identical. distance_map/distance_map2 are + # referenced only by the distance closures, so rebinding them here is safe; the mask is + # aliased as _dm_mask because whole_sample is also used by the masks figure below. + if multistain_distance_map is not None: + distance_map = multistain_distance_map + if multistain_distance_map2 is not None: + distance_map2 = multistain_distance_map2 + _dm_mask = ( + whole_sample if multistain_whole_sample is None else multistain_whole_sample + ) + + def _fig1_distance_edge(): + logging.info("Generating Figure 1: Distance map (edge)...") + _h, _w = distance_map.shape + _aspect = (_h / _w) if _w > 0 else 1.0 + # Floor the height at 50% of width so very wide slides (e.g. brain) don't get squashed + fig, ax = plt.subplots(1, 1, figsize=(6, max(3, 6 * _aspect))) + # Phase v5: distance-to-edge readability fix. + # - viridis so near-edge tissue (low |distance|) renders as bright + # yellow against the white background — visible diagnostic region. + # - Absolute distance in µm: signed maurer is negative inside; + # |·| × 8 × 0.2125 → 0-at-boundary → max-deep-inside gradient. + # - NaN outside tissue mask → white via cmap.set_bad. + # - Thin black tissue outline as unambiguous boundary marker. + # TODO: 8 (downsample factor) and 0.2125 (Xenium native µm/px) are + # hardcoded here and in three other sites. See task #15 — future + # plumbing reads pixel_size from the bundle's experiment.xenium. + from copy import copy as _copy_cmap + + _cmap_edge = _copy_cmap(plt.cm.viridis) + _cmap_edge.set_bad(color="white") + # Cap the colorscale at 300 µm so the near-edge band uses most of the + # spectrum; tiles further than 300 µm from the boundary saturate at + # yellow and the colorbar shows an "extend max" arrow. Beyond 300 µm + # the tile is unambiguously deep-tissue and not edge-affected. + _EDGE_VMAX_UM = 300.0 + if _dm_mask.shape == distance_map.shape: + _dist_um = np.abs(distance_map) * 8 * 0.2125 + _dm_edge = np.where(_dm_mask > 0, _dist_um, np.nan) + im = _imshow_thumb( + ax, _dm_edge, cmap=_cmap_edge, vmin=0.0, vmax=_EDGE_VMAX_UM + ) + _edge_cbar_extend = "max" + else: + _dm_edge = ( + distance_map # fallback: shapes mismatch, preserve prior behaviour + ) + im = _imshow_thumb(ax, _dm_edge, cmap=_cmap_edge) + _edge_cbar_extend = "neither" + if _dm_mask.shape == distance_map.shape: + _contour_thumb( + ax, + (_dm_mask > 0).astype(np.uint8), + levels=[0.5], + colors="black", + linewidths=0.5, + ) + ax.set_title("Distance to Edge") + ax.set_aspect("equal") + cbar = fig.colorbar( + im, ax=ax, fraction=0.035, pad=0.04, shrink=0.45, extend=_edge_cbar_extend + ) + cbar.set_label("Distance from edge (µm)") + # Explicit "300+" label at the cap when extend="max" is active + if _edge_cbar_extend == "max": + cbar.set_ticks([0, 50, 100, 150, 200, 250, _EDGE_VMAX_UM]) + cbar.set_ticklabels( + ["0", "50", "100", "150", "200", "250", f"{int(_EDGE_VMAX_UM)}+"] + ) + plt.tight_layout() + plt.savefig(figures_dir / "distance_map_edge.pdf", dpi=300, bbox_inches="tight") + plt.savefig(figures_dir / "distance_map_edge.png", dpi=300, bbox_inches="tight") + plt.close(fig) + # Full-resolution export. These raster figures_source CSVs are gated behind + # --figure-source-tables (off by default), so the ~86-Mpx / ~1 GB dump only + # happens when a user explicitly asks for the raw plotting data — and then it + # must match the analysis exactly, not a thumbnail. The rendered PNG still uses + # the _imshow_thumb / _thumb display-resolution path for speed regardless. + if figures_source_dir is not None: + pd.DataFrame(distance_map).to_csv( + figures_source_dir / "distance_map_edge.csv", + index=False, + header=False, + ) + + def _fig2_distance_holes(): + logging.info("Generating Figure 2: Distance map (holes)...") + _h, _w = distance_map2.shape + _aspect = (_h / _w) if _w > 0 else 1.0 + # Floor the height at 50% of width so very wide slides (e.g. brain) don't get squashed + fig, ax = plt.subplots(1, 1, figsize=(6, max(3, 6 * _aspect))) + # Phase v5: distance-to-holes readability fix. + # - viridis colormap (same as edge map) for visual consistency. + # - Absolute distance in µm: |signed_maurer_distance| × 8 × 0.2125. + # - Linear vmin=0 / vmax=300 µm — matches the edge map for visual + # consistency across both panels of §2.4 (edge + holes). Tiles beyond 300 µm saturate + # at yellow with an "extend max" arrow on the colorbar. + # - NaN outside tissue mask → white via cmap.set_bad. + # - Thin black tissue outline as boundary marker. + # TODO: pixel-size conversion hardcoded — see task #15. + from copy import copy as _copy_cmap + + _cmap_holes = _copy_cmap(plt.cm.viridis) + _cmap_holes.set_bad(color="white") + _HOLES_VMAX_UM = 300.0 + if _dm_mask.shape == distance_map2.shape: + _dist_um_h = np.abs(distance_map2) * 8 * 0.2125 + _dm_holes = np.where(_dm_mask > 0, _dist_um_h, np.nan) + im = _imshow_thumb( + ax, _dm_holes, cmap=_cmap_holes, vmin=0.0, vmax=_HOLES_VMAX_UM + ) + _holes_cbar_extend = "max" + else: + _dm_holes = distance_map2 # fallback: shapes mismatch + im = _imshow_thumb(ax, _dm_holes, cmap=_cmap_holes) + _holes_cbar_extend = "neither" + if _dm_mask.shape == distance_map2.shape: + _contour_thumb( + ax, + (_dm_mask > 0).astype(np.uint8), + levels=[0.5], + colors="black", + linewidths=0.5, + ) + ax.set_title("Distance to Nearest Hole") + ax.set_aspect("equal") + cbar = fig.colorbar( + im, ax=ax, fraction=0.035, pad=0.04, shrink=0.45, extend=_holes_cbar_extend + ) + cbar.set_label("Distance from nearest hole (µm)") + # Explicit "300+" label at the cap when extend="max" is active. + if _holes_cbar_extend == "max": + cbar.set_ticks([0, 50, 100, 150, 200, 250, _HOLES_VMAX_UM]) + cbar.set_ticklabels( + ["0", "50", "100", "150", "200", "250", f"{int(_HOLES_VMAX_UM)}+"] + ) + plt.tight_layout() + plt.savefig( + figures_dir / "distance_map_holes.pdf", dpi=300, bbox_inches="tight" + ) + plt.savefig( + figures_dir / "distance_map_holes.png", dpi=300, bbox_inches="tight" + ) + plt.close(fig) + # Full-resolution export (see distance_map_edge.csv note): gated behind + # --figure-source-tables (off by default); rendering uses the display-res thumbnail. + if figures_source_dir is not None: + pd.DataFrame(distance_map2).to_csv( + figures_source_dir / "distance_map_holes.csv", + index=False, + header=False, + ) + + def _fig2b_distance_combined(): + """Two-panel combined figure: distance to edge (left) + distance to + holes (right), sharing coordinate system and colorbar conventions. + Standalone _fig1_distance_edge / _fig2_distance_holes continue to + render the individual PNGs as latent artefacts; the combined figure + is what the QMD §2.4 embeds. + """ + logging.info("Generating Figure 2b: Distance combined (edge + holes)...") + _h, _w = distance_map.shape + _aspect = (_h / _w) if _w > 0 else 1.0 + _panel_width = 6 + _panel_height = max(3, _panel_width * _aspect) + fig, axes = plt.subplots(1, 2, figsize=(2 * _panel_width + 2, _panel_height)) + + from copy import copy as _copy_cmap + + _VMAX_UM = 300.0 + _cmap = _copy_cmap(plt.cm.viridis) + _cmap.set_bad(color="white") + + # ── Left panel: distance to edge ─────────────────────────────── + ax = axes[0] + if _dm_mask.shape == distance_map.shape: + _dist_um = np.abs(distance_map) * 8 * 0.2125 + _dm_edge = np.where(_dm_mask > 0, _dist_um, np.nan) + im_e = _imshow_thumb(ax, _dm_edge, cmap=_cmap, vmin=0.0, vmax=_VMAX_UM) + _contour_thumb( + ax, + (_dm_mask > 0).astype(np.uint8), + levels=[0.5], + colors="black", + linewidths=0.5, + ) + _edge_extend = "max" + else: + im_e = _imshow_thumb(ax, distance_map, cmap=_cmap) + _edge_extend = "neither" + ax.set_title("Distance to Edge", fontsize=14) + ax.set_aspect("equal") + cbar_e = fig.colorbar( + im_e, ax=ax, fraction=0.035, pad=0.04, shrink=0.45, extend=_edge_extend + ) + cbar_e.set_label("Distance from edge (µm)") + if _edge_extend == "max": + cbar_e.set_ticks([0, 50, 100, 150, 200, 250, _VMAX_UM]) + cbar_e.set_ticklabels( + ["0", "50", "100", "150", "200", "250", f"{int(_VMAX_UM)}+"] + ) + + # ── Right panel: distance to holes ───────────────────────────── + ax = axes[1] + if _dm_mask.shape == distance_map2.shape: + _dist_um_h = np.abs(distance_map2) * 8 * 0.2125 + _dm_holes = np.where(_dm_mask > 0, _dist_um_h, np.nan) + im_h = _imshow_thumb(ax, _dm_holes, cmap=_cmap, vmin=0.0, vmax=_VMAX_UM) + _contour_thumb( + ax, + (_dm_mask > 0).astype(np.uint8), + levels=[0.5], + colors="black", + linewidths=0.5, + ) + _holes_extend = "max" + else: + im_h = _imshow_thumb(ax, distance_map2, cmap=_cmap) + _holes_extend = "neither" + ax.set_title("Distance to Nearest Hole", fontsize=14) + ax.set_aspect("equal") + cbar_h = fig.colorbar( + im_h, ax=ax, fraction=0.035, pad=0.04, shrink=0.45, extend=_holes_extend + ) + cbar_h.set_label("Distance from nearest hole (µm)") + if _holes_extend == "max": + cbar_h.set_ticks([0, 50, 100, 150, 200, 250, _VMAX_UM]) + cbar_h.set_ticklabels( + ["0", "50", "100", "150", "200", "250", f"{int(_VMAX_UM)}+"] + ) + + plt.tight_layout() + plt.savefig(figures_dir / "distance_maps.png", dpi=300, bbox_inches="tight") + plt.savefig(figures_dir / "distance_maps.pdf", dpi=300, bbox_inches="tight") + plt.close(fig) + + def _fig3_morphology_overview(): + logging.info("Generating Figure 3: Morphology overview...") + # Phase v5 TODO #12: 1×3 layout (DAPI + Boundary + Interior only). The + # artefacts panel that used to live in this figure's bottom-right is + # also rendered in imageqc_masks.png (§2.3 Masks), so showing it here + # duplicates the same plot — dropped here. + _img_aspect = small0.shape[0] / small0.shape[1] if small0.shape[1] > 0 else 1.0 + _panel_width = 6 + _panel_height = max(_panel_width * 0.5, _panel_width * _img_aspect) + fig, ax = plt.subplots(1, 3, figsize=(3 * _panel_width, _panel_height)) + _imshow_thumb(ax[0], small0, cmap="Greys_r", vmax=np.percentile(small0, 99)) + ax[0].set_title("DAPI", fontsize=14) + ax[0].set_aspect("equal") + ax[0].axis("off") + _imshow_thumb(ax[1], small1, cmap="Greys_r", vmax=np.percentile(small1, 99)) + ax[1].set_title("Boundary", fontsize=14) + ax[1].set_aspect("equal") + ax[1].axis("off") + _imshow_thumb(ax[2], small2, cmap="Greys_r", vmax=np.percentile(small2, 99)) + ax[2].set_title("Interior", fontsize=14) + ax[2].set_aspect("equal") + ax[2].axis("off") + plt.tight_layout() + plt.savefig( + figures_dir / "morphology_overview.pdf", dpi=300, bbox_inches="tight" + ) + plt.savefig( + figures_dir / "morphology_overview.png", dpi=300, bbox_inches="tight" + ) + plt.close(fig) + # Full-resolution export (see distance_map_edge.csv note): four ~86-Mpx channel + # grids, gated behind --figure-source-tables (off by default); rendering uses + # the display-res thumbnail so the figure wall time is unaffected. + if figures_source_dir is not None: + pd.DataFrame(small0).to_csv( + figures_source_dir / "morphology_overview_DAPI.csv", + index=False, + header=False, + ) + pd.DataFrame(small1).to_csv( + figures_source_dir / "morphology_overview_Boundary.csv", + index=False, + header=False, + ) + pd.DataFrame(small2).to_csv( + figures_source_dir / "morphology_overview_Interior.csv", + index=False, + header=False, + ) + pd.DataFrame(artefacts).to_csv( + figures_source_dir / "morphology_overview_Artefacts.csv", + index=False, + header=False, + ) + + def _fig4_sample_qc_metrics(): + logging.info("Generating Figure 4: Sample QC metrics...") + idx4 = _subsample_idx(np.ones(len(df_spatial), dtype=bool)) + fig, ax = plt.subplots(2, 2, figsize=(12, 10)) + _imshow_thumb(ax[0, 0], small0, cmap="Greys_r", vmax=np.percentile(small0, 99)) + ax[0, 0].set_title("DAPI", fontsize=14) + ax[0, 0].set_aspect("equal") + ax[0, 1].scatter( + df_spatial["x"].iloc[idx4], + -df_spatial["y"].iloc[idx4], + c=df_spatial["Distance-to-edge"].iloc[idx4], + s=0.1, + marker="o", + rasterized=True, + ) + ax[0, 1].set_title("Distance to sample Edge") + ax[0, 1].set_aspect("equal") + ax[0, 1].set_facecolor("black") + ax[1, 0].scatter( + df_spatial["x"].iloc[idx4], + -df_spatial["y"].iloc[idx4], + c=df_spatial["Distance-to-nearest-hole"].iloc[idx4], + s=0.1, + marker="o", + rasterized=True, + ) + ax[1, 0].set_title("Distance to nearest hole") + ax[1, 0].set_aspect("equal") + ax[1, 0].set_facecolor("black") + ax[1, 1].scatter( + df_spatial["x"].iloc[idx4], + -df_spatial["y"].iloc[idx4], + c=df_spatial["In-Area-with-Artefact"].iloc[idx4], + cmap="Blues_r", + s=0.1, + marker="o", + vmax=iawa_vmax, + rasterized=True, + ) + ax[1, 1].set_title("Overlap with sample artefact") + ax[1, 1].set_aspect("equal") + ax[1, 1].set_facecolor("black") + plt.tight_layout() + plt.savefig(figures_dir / "sample_qc_metrics.png", dpi=300, bbox_inches="tight") + plt.close(fig) + df_sample_qc_metrics = pd.DataFrame( + { + "x": df_spatial["x"], + "y": df_spatial["y"], + "Distance-to-edge": df_spatial["Distance-to-edge"], + "Distance-to-nearest-hole": df_spatial["Distance-to-nearest-hole"], + "In-Area-with-Artefact": df_spatial["In-Area-with-Artefact"], + } + ) + if figures_source_dir is not None: + df_sample_qc_metrics.to_csv( + figures_source_dir / "sample_qc_metrics.csv", index=False + ) + + def _fig5_imageqc_masks(): + logging.info("Generating Figure 5: ImageQC masks...") + # Phase v5: 1x3 triplet — three mask panels only. DAPI morphology + # image was dropped (already shown under §2.4 Stainings). + _img_aspect = small0.shape[0] / small0.shape[1] if small0.shape[1] > 0 else 1.0 + _panel_width = 6 + _panel_height = max(_panel_width * 0.5, _panel_width * _img_aspect) + # §2.3 tissue mask shows the EXTENT mask (all available stains) so it matches the + # reported tissue coverage; falls back to the DAPI mask on single-stain slides. + _extent_ws = ( + multistain_whole_sample + if multistain_whole_sample is not None + else whole_sample + ) + fig, ax = plt.subplots(1, 3, figsize=(3 * _panel_width, _panel_height)) + ax[0].set_title("Tissue mask (all stains)", fontsize=14) + _imshow_thumb(ax[0], _extent_ws, rgb=lambda d: color.label2rgb(d, bg_label=0)) + ax[0].set_aspect("equal") + ax[0].axis("off") + ax[1].set_title("Holes in sample", fontsize=14) + _imshow_thumb(ax[1], holes, rgb=lambda d: color.label2rgb(d, bg_label=0)) + ax[1].set_aspect("equal") + ax[1].axis("off") + ax[2].set_title("Optically dense regions", fontsize=14) + _imshow_thumb(ax[2], artefacts, rgb=lambda d: color.label2rgb(d, bg_label=0)) + ax[2].set_aspect("equal") + ax[2].axis("off") + plt.tight_layout() + plt.savefig(figures_dir / "imageqc_masks.pdf", dpi=300, bbox_inches="tight") + plt.savefig(figures_dir / "imageqc_masks.png", dpi=300, bbox_inches="tight") + plt.close(fig) + # DAPI csv export dropped — same data exposed by §2.4 Stainings. + # Full-resolution export (see distance_map_edge.csv note): three ~86-Mpx mask + # grids, gated behind --figure-source-tables (off by default); rendering uses + # the display-res thumbnail so the figure wall time is unaffected. + if figures_source_dir is not None: + pd.DataFrame(_extent_ws).to_csv( + figures_source_dir / "imageqc_masks_WholeSample.csv", + index=False, + header=False, + ) + pd.DataFrame(holes).to_csv( + figures_source_dir / "imageqc_masks_Holes.csv", + index=False, + header=False, + ) + pd.DataFrame(artefacts).to_csv( + figures_source_dir / "imageqc_masks_Artefacts.csv", + index=False, + header=False, + ) + + def _fig6_ccfs_spatial(): + logging.info("Generating Figure 6: CCFS Spatial...") + idx6 = _subsample_idx(np.ones(len(myData), dtype=bool)) + # Phase 11 (v5): aspect-adaptive figure size matching slide proportions + # (was a fixed 8x8 square). Mirrors plot_grid_roi_focus_heatmap. + _x = myData["centroid-1"] + _y = myData["centroid-0"] + _x_range = float(_x.max() - _x.min()) if len(_x) else 1.0 + _y_range = float(_y.max() - _y.min()) if len(_y) else 1.0 + _img_aspect = _y_range / _x_range if _x_range > 0 else 1.0 + _panel_width = 6 + _panel_height = max(_panel_width * 0.5, _panel_width * _img_aspect) + fig, ax = plt.subplots(1, 1, figsize=(_panel_width, _panel_height)) + ax.scatter( + myData["centroid-1"].iloc[idx6], + -myData["centroid-0"].iloc[idx6], + s=0.1, + c=-myData["CCFS_DAPI"].iloc[idx6], + cmap="viridis", + vmin=-0.012, + rasterized=True, + ) + ax.set_title("Calculated Cell Focus Score") + ax.set_facecolor("black") + ax.set_aspect("equal") + plt.tight_layout() + plt.savefig(figures_dir / "ccfs_spatial.png", dpi=300, bbox_inches="tight") + plt.close(fig) + df_ccfs_spatial = pd.DataFrame( + { + "centroid-1": myData["centroid-1"], + "centroid-0": myData["centroid-0"], + "CCFS_DAPI": myData["CCFS_DAPI"], + } + ) + if figures_source_dir is not None: + df_ccfs_spatial.to_csv(figures_source_dir / "ccfs_spatial.csv", index=False) + + def _fig7_ccfs_thresholded(): + logging.info("Generating Figure 7: CCFS Thresholded...") + highC = myData[myData["is_high_nuclear_texture"]] + lowC = myData[myData["is_low_nuclear_texture"]] + idx7h = _subsample_idx(np.ones(len(highC), dtype=bool)) + idx7l = _subsample_idx(np.ones(len(lowC), dtype=bool)) + # Phase 11 (v5): aspect-adaptive figure size matching slide proportions + # (was a fixed 8x8 square). Uses full myData coordinate range so both + # high/low subsets share the same axis layout. + _x = myData["centroid-1"] + _y = myData["centroid-0"] + _x_range = float(_x.max() - _x.min()) if len(_x) else 1.0 + _y_range = float(_y.max() - _y.min()) if len(_y) else 1.0 + _img_aspect = _y_range / _x_range if _x_range > 0 else 1.0 + _panel_width = 6 + _panel_height = max(_panel_width * 0.5, _panel_width * _img_aspect) + fig, ax = plt.subplots(1, 1, figsize=(_panel_width, _panel_height)) + # High-texture cells were `#1B2631` (near-black) on black background + # — invisible. Lightened to `#999999` (mid-grey) for clear contrast + # against the black facecolor while keeping the red low-texture cells + # visually dominant. + ax.scatter( + highC["centroid-1"].iloc[idx7h], + -highC["centroid-0"].iloc[idx7h], + s=0.1, + color="#999999", + rasterized=True, + ) + ax.scatter( + lowC["centroid-1"].iloc[idx7l], + -lowC["centroid-0"].iloc[idx7l], + s=0.1, + color="red", + rasterized=True, + ) + ax.set_title("Thresholded CCFS (red = low nuclear texture cells)") + ax.set_facecolor("black") + ax.set_aspect("equal") + plt.tight_layout() + plt.savefig(figures_dir / "ccfs_thresholded.png", dpi=300, bbox_inches="tight") + plt.close(fig) + df_ccfs_thresholded = pd.DataFrame( + { + "centroid-1": myData["centroid-1"], + "centroid-0": myData["centroid-0"], + "CCFS_DAPI": myData["CCFS_DAPI"], + "is_low_nuclear_texture": myData["is_low_nuclear_texture"], + } + ) + if figures_source_dir is not None: + df_ccfs_thresholded.to_csv( + figures_source_dir / "ccfs_thresholded.csv", index=False + ) + + def _fig8_umap_multiple_metrics(): + logging.info("Generating Figure 8: UMAP by multiple metrics...") + fig, ax = plt.subplots(2, 2, figsize=(15, 15)) + ax[0, 0].scatter( + new_df["UMAP-1"], + new_df["UMAP-2"], + s=0.1, + c=-new_df["CCFS_DAPI"], + marker="o", + cmap="viridis", + vmin=-0.012, + rasterized=True, + ) + ax[0, 0].set_title("UMAP by Nuclear Texture Score") + ax[0, 0].set_facecolor("black") + ax[0, 0].set_aspect("equal") + ax[0, 1].scatter( + new_df["UMAP-1"], + new_df["UMAP-2"], + s=0.1, + c=new_df["Cluster_kmeans10"], + marker="o", + rasterized=True, + ) + ax[0, 1].set_title("UMAP by Cluster Allocation") + ax[0, 1].set_facecolor("black") + ax[0, 1].set_aspect("equal") + ax[1, 0].scatter( + new_df["UMAP-1"], + new_df["UMAP-2"], + s=0.1, + c=new_df["segPal"], + marker="o", + rasterized=True, + ) + ax[1, 0].set_title("UMAP by Segmentation method") + ax[1, 0].set_facecolor("black") + ax[1, 0].set_aspect("equal") + ax[1, 1].scatter( + new_df["UMAP-1"], + new_df["UMAP-2"], + s=0.1, + c=new_df["transcript_counts"], + marker="o", + rasterized=True, + ) + ax[1, 1].set_title("UMAP by Transcript counts") + ax[1, 1].set_facecolor("black") + ax[1, 1].set_aspect("equal") + plt.tight_layout() + plt.savefig( + figures_dir / "umap_multiple_metrics.png", dpi=300, bbox_inches="tight" + ) + plt.close(fig) + df_umap_multiple_metrics = new_df[ + [ + "UMAP-1", + "UMAP-2", + "CCFS_DAPI", + "Cluster_kmeans10", + "segPal", + "transcript_counts", + ] + ] + if figures_source_dir is not None: + df_umap_multiple_metrics.to_csv( + figures_source_dir / "umap_multiple_metrics.csv", index=False + ) + + def _fig9_umap_distance_metrics(): + logging.info("Generating Figure 9: UMAP by distance metrics...") + highC_new = new_df[new_df["is_high_nuclear_texture"]] + lowC_new = new_df[new_df["is_low_nuclear_texture"]] + fig, ax = plt.subplots(2, 2, figsize=(15, 15)) + ax[0, 0].scatter( + new_df["UMAP-1"], + new_df["UMAP-2"], + s=0.1, + c=-new_df["Distance-to-edge"], + marker="o", + rasterized=True, + ) + ax[0, 0].set_title("UMAP by Distance to edge") + ax[0, 0].set_facecolor("black") + ax[0, 0].set_aspect("equal") + ax[0, 1].scatter( + new_df["UMAP-1"], + new_df["UMAP-2"], + s=0.1, + c=-new_df["Distance-to-nearest-hole"], + marker="o", + rasterized=True, + ) + ax[0, 1].set_title("UMAP by Distance to nearest hole") + ax[0, 1].set_facecolor("black") + ax[0, 1].set_aspect("equal") + ax[1, 0].scatter( + new_df["UMAP-1"], + new_df["UMAP-2"], + s=0.1, + c=new_df["In-Area-with-Artefact"], + cmap="Blues_r", + marker="o", + vmax=iawa_vmax, + rasterized=True, + ) + ax[1, 0].set_title("UMAP by Overlap with artefact") + ax[1, 0].set_facecolor("black") + ax[1, 0].set_aspect("equal") + ax[1, 1].scatter( + highC_new["UMAP-1"], + highC_new["UMAP-2"], + s=0.1, + color="#1B2631", + rasterized=True, + ) + ax[1, 1].scatter( + lowC_new["UMAP-1"], lowC_new["UMAP-2"], s=0.1, color="red", rasterized=True + ) + ax[1, 1].set_title("UMAP by Blurred cells (in red)") + ax[1, 1].set_facecolor("black") + ax[1, 1].set_aspect("equal") + plt.tight_layout() + plt.savefig( + figures_dir / "umap_distance_metrics.png", dpi=300, bbox_inches="tight" + ) + plt.close(fig) + df_umap_distance_metrics = new_df[ + [ + "UMAP-1", + "UMAP-2", + "Distance-to-edge", + "Distance-to-nearest-hole", + "In-Area-with-Artefact", + "CCFS_DAPI", + "is_low_nuclear_texture", + ] + ] + if figures_source_dir is not None: + df_umap_distance_metrics.to_csv( + figures_source_dir / "umap_distance_metrics.csv", index=False + ) + + def _fig10_umap_thresholded_metrics(): + logging.info("Generating Figure 10: UMAP thresholded metrics...") + highE = new_df[new_df["is_far_from_edge"]] + lowE = new_df[new_df["is_near_edge"]] + highH = new_df[new_df["is_far_from_hole"]] + lowH = new_df[new_df["is_near_hole"]] + fig, ax = plt.subplots(2, 2, figsize=(15, 15)) + ax[0, 0].scatter( + highE["UMAP-1"], highE["UMAP-2"], s=0.1, color="#1B2631", rasterized=True + ) + ax[0, 0].scatter( + lowE["UMAP-1"], lowE["UMAP-2"], s=0.1, color="red", rasterized=True + ) + ax[0, 0].set_title("UMAP by Distance to edge") + ax[0, 0].set_facecolor("black") + ax[0, 0].set_aspect("equal") + ax[0, 1].scatter( + highH["UMAP-1"], highH["UMAP-2"], s=0.1, color="#1B2631", rasterized=True + ) + ax[0, 1].scatter( + lowH["UMAP-1"], lowH["UMAP-2"], s=0.1, color="red", rasterized=True + ) + ax[0, 1].set_title("UMAP by Distance to nearest hole") + ax[0, 1].set_facecolor("black") + ax[0, 1].set_aspect("equal") + ax[1, 0].scatter( + new_df["UMAP-1"], + new_df["UMAP-2"], + s=0.1, + c=new_df["In-Area-with-Artefact"], + cmap="Blues_r", + marker="o", + vmax=iawa_vmax, + rasterized=True, + ) + ax[1, 0].set_title("UMAP by Overlap with artefact") + ax[1, 0].set_facecolor("black") + ax[1, 0].set_aspect("equal") + ax[1, 1].scatter( + new_df[new_df["is_high_nuclear_texture"]]["UMAP-1"], + new_df[new_df["is_high_nuclear_texture"]]["UMAP-2"], + s=0.1, + color="#1B2631", + rasterized=True, + ) + ax[1, 1].scatter( + new_df[new_df["is_low_nuclear_texture"]]["UMAP-1"], + new_df[new_df["is_low_nuclear_texture"]]["UMAP-2"], + s=0.1, + color="red", + rasterized=True, + ) + ax[1, 1].set_title("UMAP by Blurred cells (in red)") + ax[1, 1].set_facecolor("black") + ax[1, 1].set_aspect("equal") + plt.tight_layout() + plt.savefig( + figures_dir / "umap_thresholded_metrics.png", dpi=300, bbox_inches="tight" + ) + plt.close(fig) + # Save data as CSV + df_umap_thresholded_metrics = new_df[ + [ + "UMAP-1", + "UMAP-2", + "Distance-to-edge", + "Distance-to-nearest-hole", + "In-Area-with-Artefact", + "CCFS_DAPI", + "is_near_edge", + "is_near_hole", + "is_low_nuclear_texture", + ] + ] + if figures_source_dir is not None: + df_umap_thresholded_metrics.to_csv( + figures_source_dir / "umap_thresholded_metrics.csv", index=False + ) + + def _fig11_nuclear_texture_proportions(): + logging.info("Generating Figure 11: Nuclear Texture Proportions by Cluster...") + plot_nuclear_texture_proportions( + new_df, + figures_dir, + figures_source_dir, + texture_threshold=ccfs_low_texture_threshold, + GROUP_BY_COLUMN="Cluster_kmeans10", + ) + + def _fig12_nuclear_texture_density(): + logging.info("Generating Figure 12: Nuclear Texture Density by Cluster...") + plot_nuclear_texture_density( + new_df, + figures_dir, + figures_source_dir, + GROUP_BY_COLUMN="Cluster_kmeans10", + ccfs_low_texture_threshold=ccfs_low_texture_threshold, + ) + + def _fig13_nuclear_texture_vs_transcripts(): + logging.info( + "Generating Figure 13: Nuclear Texture vs Transcripts (Log Scale)..." + ) + plot_nuclear_texture_vs_transcripts( + new_df, + figures_dir, + figures_source_dir, + log_scale=True, + ccfs_low_texture_threshold=ccfs_low_texture_threshold, + ) + + def _fig14_cell_focus_distribution(): + logging.info("Generating Figure 14: Cell-level focus score distribution...") + plot_cell_focus_distribution(new_df, figures_dir, figures_source_dir) + + def _fig15_gmm_blur_proportions_by_cluster(): + if ( + "is_blurred_gmm_2d_roi" in new_df.columns + or "is_blurred_roi" in new_df.columns + ): + logging.info("Generating Figure 15: GMM Blur Proportions by Cluster...") + plot_tile_blur_proportions_roi( + new_df, + figures_dir, + figures_source_dir, + GROUP_BY_COLUMN="Cluster_kmeans10", + ) + + def _fig16_gmm_focus_vs_transcripts(): + logging.info("Generating Figure 16: GMM Focus vs Transcripts...") + plot_gmm_focus_vs_transcripts(new_df, figures_dir, figures_source_dir) + + def _fig17_tile_focus_gmm_spatial(): + logging.info("Generating Figure 17: Tile-focus GMM spatial...") + plot_tile_focus_gmm_spatial(new_df, figures_dir, figures_source_dir) + + def _fig17b_cell_flagged_maps_combined(): + logging.info("Generating Figure 17b: Cell-flagged maps (combined)...") + plot_cell_flagged_maps_combined(new_df, figures_dir, figures_source_dir) + + def _fig18_blur_prob_density_by_cluster(): + logging.info("Generating Figure 18: Blur probability density by cluster...") + plot_blur_prob_density_by_cluster(new_df, figures_dir, figures_source_dir) + + def _fig19_intensity_transcript_correlation(): + logging.info( + "Generating Figure 19: Intensity-transcript correlation heatmap..." + ) + plot_intensity_transcript_correlation(new_df, figures_dir, figures_source_dir) + + def _fig20_per_cell_intensity_vs_transcripts(): + logging.info( + "Generating Figure 20: Per-cell intensity vs transcripts (DAPI / Boundary / IntRNA)..." + ) + plot_per_cell_intensity_vs_transcripts( + new_df, figures_dir, figures_source_dir, log_scale=True + ) + + tasks = [ + # Deduped 2026-08-01: distance_map_edge/holes, distance_maps, + # morphology_overview and imageqc_masks are already rendered by the + # always-run tile path (generate_roi_figures); render once, not twice. + _fig7_ccfs_thresholded, + # Latent figures disabled 2026-05-20 (user request) — function defs + # retained above as latent code: _fig4_sample_qc_metrics, + # _fig6_ccfs_spatial, _fig8/_fig9/_fig10 UMAP, _fig12_nuclear_texture_density, + # _fig14_cell_focus_distribution, _fig18_blur_prob_density_by_cluster. + _fig11_nuclear_texture_proportions, + _fig13_nuclear_texture_vs_transcripts, + _fig15_gmm_blur_proportions_by_cluster, + _fig16_gmm_focus_vs_transcripts, + _fig17_tile_focus_gmm_spatial, + _fig17b_cell_flagged_maps_combined, + _fig19_intensity_transcript_correlation, + _fig20_per_cell_intensity_vs_transcripts, + ] + + _run_figure_pool(tasks, phase="figures") + + logging.info("All figures generated successfully!") + logging.info(f"[OK] Saved figures to {figures_dir}") + if figures_source_dir is not None: + logging.info(f"[OK] Saved source data to {figures_source_dir}") + + +# ===== SIMPLE QC METRICS (from image_qc_processing.py) ===== + + +def save_simple_qc_metrics(new_df, outdir): + """Save QC metrics to CSV and JSON""" + + # Move cell_id to first position + cell_id_col = new_df.pop("cell_id") + new_df.insert(0, "cell_id", cell_id_col) + # Save full dataset + new_df.to_csv(outdir / "image_qc_metrics.csv", index=False) + + # Generate QC summary statistics + qc_metrics = { + "total_cells": len(new_df), + "cells_with_transcripts": len(new_df[new_df["transcript_counts"] > 0]), + "mean_transcript_count": float(new_df["transcript_counts"].mean()), + "median_transcript_count": float(new_df["transcript_counts"].median()), + "std_transcript_count": float(new_df["transcript_counts"].std()), + "mean_ccfs_dapi": float(new_df["CCFS_DAPI"].mean()), + "cells_high_nuclear_texture": len(new_df[new_df["is_high_nuclear_texture"]]), + "cells_low_nuclear_texture": len(new_df[new_df["is_low_nuclear_texture"]]), + "cells_near_edge": len(new_df[new_df["is_near_edge"]]), + "cells_near_holes": len(new_df[new_df["is_near_hole"]]), + "cells_with_artifacts": len(new_df[new_df["has_artifacts"]]), + "clusters_present": sorted(new_df["Cluster_kmeans10"].unique().tolist()), + "segmentation_methods": sorted(new_df["segmentation_method"].unique().tolist()), + } + + # Save metrics + with open(outdir / "image_qc_metrics.json", "w") as f: + json.dump(qc_metrics, f, indent=2) + + logging.info("\n=== IMAGE QC SUMMARY ===") + logging.info(f"Total cells analyzed: {qc_metrics['total_cells']:,}") + logging.info(f"Cells with transcripts: {qc_metrics['cells_with_transcripts']:,}") + logging.info(f"Mean transcript count: {qc_metrics['mean_transcript_count']:.1f}") + logging.info(f"Mean CCFS DAPI: {qc_metrics['mean_ccfs_dapi']:.6f}") + logging.info( + f"Cells high nuclear texture: {qc_metrics['cells_high_nuclear_texture']:,}" + ) + logging.info( + f"Cells low nuclear texture: {qc_metrics['cells_low_nuclear_texture']:,}" + ) + logging.info(f"Cells near edge: {qc_metrics['cells_near_edge']:,}") + logging.info(f"Cells near holes: {qc_metrics['cells_near_holes']:,}") + logging.info(f"Cells with artifacts: {qc_metrics['cells_with_artifacts']:,}") + logging.info(f"Clusters present: {qc_metrics['clusters_present']}") + logging.info(f"Segmentation methods: {qc_metrics['segmentation_methods']}") + + +# ===== NEW: PIXEL-MAP AGGREGATION FOR CCFS ===== + + +# Pixels per row block for labeled reductions. 32 M px keeps each transient +# float64/intp cast at ~256 MB, so a full pass costs ~1.3 GB regardless of how +# large the image is. +_LABEL_CHUNK_PIXELS = 32 * 1024 * 1024 + + +def _grow_to(arr: NDArray[Any], n: int) -> NDArray[Any]: + """Zero-extend *arr* to length *n*; no-op when it is already long enough.""" + if arr.size >= n: + return arr + out = np.zeros(n, dtype=arr.dtype) + out[: arr.size] = arr + return out + + +def _labeled_sums_chunked( + labels_img, + value_planes: dict[str, Any], + include_coords: bool = False, + rows_per_chunk: int | None = None, +) -> tuple[NDArray[np.int64], dict[str, NDArray[np.float64]]]: + """Per-label pixel counts and value sums, accumulated in row blocks. + + Equivalent to ``scipy.ndimage.sum`` over every label, but it never casts a + full-resolution plane. ``scipy.ndimage.mean`` reduces via ``np.bincount``, + which requires ``intp`` labels and ``float64`` weights, so a whole-image + call materialises an int64 copy of the label plane *and* a float64 copy of + the value plane — 88 GB combined on a 5.5 gigapixel sample, invisible in the + source. Blocking by rows bounds both casts to one block. + + Counts and sums are additive, so a label straddling a block boundary + accumulates partial contributions from each block and its final + ``sum / count`` is exact — not an approximation. + + Args: + labels_img: 2-D integer label plane. May be a lazily-sliced handle + (zarr array, memmap); only one row block is materialised at a time. + value_planes: Named 2-D value planes aligned with *labels_img*. May be + empty to collect counts only. + include_coords: Also accumulate coordinate sums, returned under the + ``centroid_y_sum`` / ``centroid_x_sum`` keys. ``sum / count`` of + these is exactly ``skimage.measure.regionprops`` ``centroid``. + rows_per_chunk: Rows per block. Defaults to ~32 M pixels per block. + + Returns: + ``(counts, sums)`` indexed by raw label value, index 0 = background. + ``counts[i]`` is the pixel count for label ``i``; ``sums[name][i]`` the + summed value. Both are sized to the largest label actually seen. + """ + height, width = labels_img.shape + if rows_per_chunk is None: + rows_per_chunk = max(1, _LABEL_CHUNK_PIXELS // max(width, 1)) + + counts = np.zeros(1, dtype=np.int64) + sums: dict[str, NDArray[np.float64]] = { + name: np.zeros(1, dtype=np.float64) for name in value_planes + } + if include_coords: + sums["centroid_y_sum"] = np.zeros(1, dtype=np.float64) + sums["centroid_x_sum"] = np.zeros(1, dtype=np.float64) + + def _accumulate(key: str, block_sums: NDArray[np.float64]) -> None: + sums[key] = _grow_to(sums[key], block_sums.size) + sums[key][: block_sums.size] += block_sums + + for y0 in range(0, height, rows_per_chunk): + y1 = min(y0 + rows_per_chunk, height) + lab = np.asarray(labels_img[y0:y1]).ravel() + + block_counts = np.bincount(lab) + counts = _grow_to(counts, block_counts.size) + counts[: block_counts.size] += block_counts + n = block_counts.size + + for name, plane in value_planes.items(): + vals = np.asarray(plane[y0:y1], dtype=np.float64).ravel() + _accumulate(name, np.bincount(lab, weights=vals, minlength=n)) + del vals + + if include_coords: + n_rows = y1 - y0 + rows = np.repeat(np.arange(y0, y1, dtype=np.float64), width) + _accumulate("centroid_y_sum", np.bincount(lab, weights=rows, minlength=n)) + del rows + cols = np.tile(np.arange(width, dtype=np.float64), n_rows) + _accumulate("centroid_x_sum", np.bincount(lab, weights=cols, minlength=n)) + del cols + + del lab, block_counts + + return counts, sums + + +def calculate_ccfs_from_focus_maps( + focus_maps, cell_masks_zarr, cellseg_mask, xoa_morphology_files, streamed=None +): + """ + Derive per-cell CCFS from pixel focus maps using scipy.ndimage.mean. + + This replaces the old regionprops-based calculate_ccfs_measurements() for + DAPI focus scores, but still uses regionprops for boundary/RNA per-cell + intensities (different channels). + + Args: + focus_maps: dict from compute_all_focus_maps() with at least 'dapi_focus_map' and 'dapi_mean_map' + cell_masks_zarr: zarr group with 'masks/0' (nuclear mask) and 'masks/1' (cell mask) + cellseg_mask: numpy array of cell segmentation mask (from masks/1) + xoa_morphology_files: list of morphology file paths (for boundary/RNA channels) + + Returns: + pandas DataFrame with columns: CellID, centroid-0, centroid-1, + CCFS_DAPI, mean_intensity, area_nucleus, area_cell, + mean_intensity_Boundary, mean_intensity_IntRNA + """ + if streamed is None: + focus_map = focus_maps.get("dapi_focus_map") + mean_map = focus_maps.get("dapi_mean_map") + if focus_map is None or mean_map is None: + raise ValueError( + "focus_maps must contain 'dapi_focus_map' and 'dapi_mean_map'" + ) + elif not streamed.has_per_cell: + raise ValueError( + "streamed results carry no per-cell reduction; " + "cell_masks_path was not passed to calculate_roi_focusscore" + ) + + # The nuclear label plane stays lazy: _labeled_sums_chunked slices row + # blocks straight out of zarr, so the full-res uint32 plane is never + # materialised (22 GB on a 5.5 GP sample). One blocked pass replaces + # np.unique() (a 22 GB flatten copy), two whole-image ndimage.mean() calls + # (88 GB of intp/float64 casts each), and regionprops() (a whole-plane + # pass) — centroids and areas fall out of the same accumulators. + if streamed is not None: + # The tile pass already folded every tile into these, so there is no plane + # left to read. Keys are the accumulator's, mapped onto this function's. + logging.info(" Per-nucleus sums reduced during the tile pass (streamed)") + nuc_counts = streamed.nuclear_counts + nuc_sums = { + "focus": streamed.nuclear_sums["focus_map"], + "intensity": streamed.nuclear_sums["mean_map"], + "centroid_y_sum": streamed.nuclear_sums["centroid_y_sum"], + "centroid_x_sum": streamed.nuclear_sums["centroid_x_sum"], + } + else: + nuclear_mask = cell_masks_zarr.get("masks").get("0") + logging.info( + " Aggregating focus/intensity over nuclear masks (row-blocked)..." + ) + t0 = time.time() + nuc_counts, nuc_sums = _labeled_sums_chunked( + nuclear_mask, + {"focus": focus_map, "intensity": mean_map}, + include_coords=True, + ) + logging.info(f" [TIMING] labeled nuclear aggregation: {time.time() - t0:.1f}s") + + # Labels present, ascending, background excluded — same set and order as + # regionprops(), which yields one region per distinct label value. + labels = np.nonzero(nuc_counts)[0] + labels = labels[labels > 0] + if labels.size == 0: + raise ValueError("nuclear mask contains no labelled pixels") + counts = nuc_counts[labels].astype(np.float64) + + cell_focus = nuc_sums["focus"][labels] / counts + cell_intensity = nuc_sums["intensity"][labels] / counts + # sum(coord)/count is exactly regionprops' centroid definition. + centroid_y = nuc_sums["centroid_y_sum"][labels] / counts + centroid_x = nuc_sums["centroid_x_sum"][labels] / counts + + # Normalize: 99th percentile of mean intensity + dapi_norm = np.percentile(cell_intensity, 99) + if dapi_norm == 0: + dapi_norm = 1.0 + ccfs_dapi = cell_focus / dapi_norm + + # Map each nucleus centroid to a cell ID (point lookups only) + iy = np.minimum(centroid_y.astype(np.int64), cellseg_mask.shape[0] - 1) + ix = np.minimum(centroid_x.astype(np.int64), cellseg_mask.shape[1] - 1) + cell_ids = np.asarray(cellseg_mask[iy, ix]).astype(np.int64) + + nucleus_props = pd.DataFrame( + { + "label": labels, + "centroid-0": centroid_y, + "centroid-1": centroid_x, + "area_nucleus": nuc_counts[labels], + "CCFS_DAPI": ccfs_dapi, + "mean_intensity": cell_intensity, + "CellID": cell_ids, + } + ) + del nuc_counts, nuc_sums + + # One blocked pass over the cell mask yields cell areas *and* the per-cell + # boundary/IntRNA means, replacing np.bincount() on the full-res uint32 + # plane (a 44 GB intp cast) plus one ndimage.mean() per channel. + if streamed is not None: + logging.info(" Per-cell sums reduced during the tile pass (streamed)") + cell_counts = streamed.cell_counts + cell_sums = streamed.cell_sums or {} + else: + cell_value_planes: dict[str, Any] = {} + boundary_mean = focus_maps.get("boundary_mean_map") + intrna_mean = focus_maps.get("intrna_mean_map") + if boundary_mean is not None: + cell_value_planes["boundary"] = boundary_mean + if intrna_mean is not None: + cell_value_planes["intrna"] = intrna_mean + + logging.info( + " Aggregating cell areas/intensities over cell masks (row-blocked)..." + ) + t0 = time.time() + cell_counts, cell_sums = _labeled_sums_chunked(cellseg_mask, cell_value_planes) + logging.info(f" [TIMING] labeled cell aggregation: {time.time() - t0:.1f}s") + + cids = nucleus_props["CellID"].to_numpy() + in_range = cids < cell_counts.size + + # area_cell: index by raw CellID, 0 when out of range. CellID 0 keeps + # resolving to the background pixel count, matching the previous mapping. + area_cell = np.zeros(cids.size, dtype=np.int64) + area_cell[in_range] = cell_counts[cids[in_range]] + nucleus_props["area_cell"] = area_cell + + # Per-cell channel means. CellID 0 stays NaN: the old code built its + # lookup from the >0 labels only, so background mapped to NaN. + labelled = in_range & (cids > 0) + for name, column in ( + ("boundary", "mean_intensity_Boundary"), + ("intrna", "mean_intensity_IntRNA"), + ): + if name not in cell_sums: + nucleus_props[column] = np.nan + continue + with np.errstate(invalid="ignore", divide="ignore"): + per_cell = cell_sums[name] / cell_counts + values = np.full(cids.size, np.nan, dtype=np.float64) + values[labelled] = per_cell[cids[labelled]] + nucleus_props[column] = values + + return nucleus_props + + +# ===== FALLBACK: REGIONPROPS-BASED CCFS (for legacy mode) ===== + + +def calculate_ccfs_measurements(xoa_morphology_files, cellseg_mask, cell_masks_zarr): + """ + Calculate CCFS measurements with improved variable naming and memory management. + Loads data just before use and deletes it immediately after. + + Legacy path (--legacy-focus). regionprops_table needs a materialised label + array, so a LazyLabelPlane is realised here rather than in the caller -- the + streaming path never does this. + """ + if isinstance(cellseg_mask, LazyLabelPlane): + logging.info( + " Materialising the cell mask for regionprops (legacy focus path)..." + ) + # A full-height slice; LazyLabelPlane already returns numpy. Not + # np.asarray(...): this function has its own local `import numpy as np` + # further down, so `np` is a local name here and referencing it before + # that import raises UnboundLocalError. + cellseg_mask = cellseg_mask[0 : cellseg_mask.shape[0]] + import numpy as np + import pandas as pd + + # Load full resolution image channels only when needed + fullres_channels = imread( + xoa_morphology_files[0], is_ome=False, level=0, aszarr=False + ) + + # Check number of channels (could be 2D array for single channel, or 3D array for multi-channel) + if len(fullres_channels.shape) == 2: + # Single channel (2D array) + dapi_image = fullres_channels + boundary_image = None + rna_image = None + else: + # Multi-channel (3D array: channels, height, width) + dapi_image = fullres_channels[0] + boundary_image = fullres_channels[1] if fullres_channels.shape[0] > 1 else None + rna_image = fullres_channels[2] if fullres_channels.shape[0] > 2 else None + del fullres_channels + + # Load nuclear segmentation mask + nuclear_mask = np.array(cell_masks_zarr.get("masks").get("0")) + + # Per-nucleus measurements on DAPI image + nucleus_props = pd.DataFrame( + regionprops_table(dapi_image, nuclear_mask, position=True) + ) + # Calculate CCFS_DAPI + mean_intensity = nucleus_props["mean_intensity"] + std_intensity = nucleus_props["standard_deviation_intensity"] + dapi_norm = np.percentile(mean_intensity, 99) + ccfs_dapi = (std_intensity * std_intensity) / (mean_intensity * dapi_norm) + nucleus_props["CCFS_DAPI"] = ccfs_dapi + del dapi_image + + # Get CellID for each nucleus by measuring mean intensity in cellseg_mask + cellid_props = pd.DataFrame( + regionprops_table(cellseg_mask, nuclear_mask, position=True) + ) + nucleus_props["CellID"] = cellid_props["mean_intensity"].astype(int) + del nuclear_mask, cellid_props + + # Measure boundary (red channel) intensity per cell + if boundary_image is not None: + boundary_props = pd.DataFrame(regionprops_table(boundary_image, cellseg_mask)) + boundary_props["CellID"] = boundary_props["label"].astype(int) + del boundary_image + else: + # Create empty boundary_props if boundary channel not available + boundary_props = pd.DataFrame( + {"CellID": nucleus_props["CellID"], "mean_intensity": np.nan} + ) + + # Measure RNA (interior) intensity per cell + if rna_image is not None: + rna_props = pd.DataFrame(regionprops_table(rna_image, cellseg_mask)) + rna_props["CellID"] = rna_props["label"].astype(int) + rna_props["mean_intensity_IntRNA"] = rna_props["mean_intensity"] + del rna_image + else: + # Create empty rna_props if RNA channel not available + rna_props = pd.DataFrame( + { + "CellID": nucleus_props["CellID"], + "area": np.nan, + "mean_intensity_IntRNA": np.nan, + } + ) + + # Merge all measurements into a single DataFrame + merged_data = pd.merge( + nucleus_props, + rna_props[["CellID", "area"]], + how="inner", + on="CellID", + suffixes=("_nucleus", "_cell"), + ) + merged_data = pd.merge( + merged_data, + boundary_props[["CellID", "mean_intensity"]], + how="inner", + on="CellID", + suffixes=("_DAPI", "_Boundary"), + ) + merged_data = pd.merge( + merged_data, + rna_props[["CellID", "mean_intensity_IntRNA"]], + how="inner", + on="CellID", + ) + + return merged_data + + +# ===== COMBINED MAIN ===== + + +def _check_cell_data_exists(xenium_bundle_dir): + """ + Check if cell data files exist in the Xenium bundle directory. + + Args: + xenium_bundle_dir: Path to Xenium bundle directory + + Returns: + bool: True if required cell data files exist + """ + xenium_bundle_dir = Path(xenium_bundle_dir) + required_files = [ + xenium_bundle_dir / "cells.parquet", + xenium_bundle_dir / "cells.zarr.zip", + ] + # Check clustering path + clusters_path = ( + xenium_bundle_dir + / "analysis" + / "clustering" + / "gene_expression_kmeans_10_clusters" + / "clusters.csv" + ) + + # Also check for analysis.tar.gz (test data) + has_analysis = ( + clusters_path.exists() or (xenium_bundle_dir / "analysis.tar.gz").is_file() + ) + + for f in required_files: + if not f.exists(): + logging.info(f" Cell data check: {f.name} not found") + return False + + if not has_analysis: + logging.info(" Cell data check: clustering/UMAP data not found") + return False + + return True + + +@click.command() +@click.option( + "--xenium-bundle-dir", required=True, help="Path to Xenium bundle directory" +) +@click.option("--outdir", required=True, help="Output directory for results") +@click.option( + "--stain-names", + default=None, + help="Semicolon-separated list of stain names. If not provided, uses defaults.", +) +@click.option( + "--roi-size", default=35, type=int, show_default=True, help="Tile size in pixels" +) +@click.option( + "--max-scatter-points", + default=10000, + type=int, + show_default=True, + help="Maximum number of points to plot in scatter figures. Set to 0 to plot all points.", +) +@click.option( + "--legacy-focus", + is_flag=True, + default=False, + help="Use legacy per-tile loop focus scoring instead of the default convolution-based GPU-accelerated method.", +) +@click.option( + "--sample-id", + default=None, + help="Sample identifier for logging and metrics output.", +) +@click.option( + "--no-snr", + is_flag=True, + default=False, + help="Disable SNR metrics (image Otsu / quartiles, transcripts, slide matrix, neg spatial).", +) +@click.option( + "--snr-no-roi-tx-table", + is_flag=True, + default=False, + help="Do not write SNR_roi_tx.parquet (or .csv.gz) alongside roi_qc_metrics.", +) +@click.option( + "--snr-otsu-max-rois", + type=int, + default=None, + help="Cap tiles for per-tile Otsu image SNR (default: all tiles).", +) +@click.option( + "--snr-with-moran", + is_flag=True, + default=False, + help="SNR only: enable Moran's I for neg spatial (needs PySAL/esda; default off).", +) +@click.option( + "--stream-tiles/--no-stream-tiles", + "stream_tiles", + default=True, + help=( + "Reduce each tile as it is computed instead of assembling full-resolution " + "pixel planes (default: stream). The planes cost ~154 GB of scratch on a " + "5.5 gigapixel sample and mmap over a FUSE/S3 work directory is " + "pathological; no downstream metric needs a whole plane. " + "--no-stream-tiles restores the plane-based path, and " + "--save-dapi-maps-tiff implies it." + ), +) +@click.option( + "--save-dapi-maps-tiff", + "save_dapi_maps_tiff", + is_flag=True, + default=False, + help=( + "Write full-resolution per-pixel maps as tiled float32 TIFF " + "(dapi_focus/mean/lap_var and boundary/intrna focus/mean when present). " + "Default off — QC and SNR use in-memory arrays only; enable for archival " + "or external tools (large files)." + ), +) +@click.option( + "--max-gpus", + "max_gpus", + default=0, + type=int, + help=( + "Cap the number of CUDA devices used (0 = use every device detected). " + "Nextflow's `accelerator` directive only sizes the Batch request; it does " + "not restrict CUDA visibility, so a task that asked for one GPU but landed " + "on a multi-GPU instance would otherwise use all of them." + ), +) +@click.option( + "--roi-thresholds-yaml", + default=None, + type=click.Path(exists=True), + help="Path to tile image QC thresholds YAML. Overrides hardcoded defaults.", +) +@click.option( + "--lap-sigma", + default=1.0, + type=float, + show_default=True, + help="Gaussian sigma for Laplacian of Gaussian (LoG) pre-smoothing.", +) +@click.option( + "--pipeline-segmentation", + default="skip", + show_default=True, + help="Pipeline segmentation method (params.segmentation); 'skip' for none.", +) +@click.option( + "--is-resegmented", + is_flag=True, + default=False, + help="Set when this run analyses a pipeline-resegmented bundle (post-seg).", +) +@click.option( + "--figure-source-tables/--no-figure-source-tables", + "figure_source_tables", + default=False, + help=( + "Write the per-figure figures_source/*.csv source-data exports " + "(unused downstream; default off; ~80s on a 5.5 GP sample)." + ), +) +@click.option( + "--figures/--no-figures", + "figures", + default=True, + help=( + "Generate QC figures (default true; --no-figures skips all figure " + "rendering for a metrics-only fast run). Metric/JSON/parquet outputs " + "are always computed regardless of this flag." + ), +) +def main( + xenium_bundle_dir, + outdir, + stain_names, + roi_size, + max_scatter_points, + legacy_focus, + sample_id, + no_snr, + snr_no_roi_tx_table, + snr_otsu_max_rois, + snr_with_moran, + save_dapi_maps_tiff, + stream_tiles, + max_gpus, + roi_thresholds_yaml, + lap_sigma, + pipeline_segmentation, + is_resegmented, + figure_source_tables, + figures, +): + """ + Combined Xenium Image QC pipeline. + + Performs pixel-level focus analysis (GPU-accelerated), tile-level analysis, + and optionally cell-level analysis when cell data is available. + + Produces: tile figures, cell figures, image_qc_metrics.json, image_qc_metrics.csv, + 13 Quarto-required PNGs, and versions.yml. + """ + logging.basicConfig( + level=logging.INFO, + format="%(asctime)s [%(levelname)s] %(message)s", + datefmt="%Y-%m-%d %H:%M:%S", + ) + + global _MAX_SCATTER_POINTS + if max_scatter_points > 0: + _MAX_SCATTER_POINTS = max_scatter_points + + t_total_start = time.time() + logging.info("=" * 60) + logging.info("Starting Combined Xenium Image QC Pipeline") + if sample_id: + logging.info("Sample ID: %s", sample_id) + logging.info("=" * 60) + logging.info(f"Input directory: {xenium_bundle_dir}") + logging.info(f"Output directory: {outdir}") + + # Load QC thresholds from YAML (or use hardcoded defaults) + qc_thresholds = _load_qc_thresholds(roi_thresholds_yaml) + if roi_thresholds_yaml: + logging.info(f"Loaded QC thresholds from: {roi_thresholds_yaml}") + else: + logging.info("Using built-in default QC thresholds") + # Resolve lap_sigma from YAML (CLI value is the fallback) + _yaml_sigma = qc_thresholds.get("lap_sigma", lap_sigma) + try: + _lap_sigma = float(_yaml_sigma) + except (TypeError, ValueError): + logging.warning( + "Invalid lap_sigma value %r in YAML, falling back to CLI=%s", + _yaml_sigma, + lap_sigma, + ) + _lap_sigma = float(lap_sigma) + logging.info( + f"LoG sigma: {_lap_sigma} (CLI={lap_sigma}, YAML={qc_thresholds.get('lap_sigma', 'not set')})" + ) + + # Resolve top-level operational thresholds from YAML (with hardcoded fallbacks) + _roi_intensity_threshold = float( + qc_thresholds.get("roi_intensity_threshold", ROI_INTENSITY_THRESHOLD) + ) + _min_tissue_cov = float( + qc_thresholds.get( + "min_tissue_coverage_for_intensity_qc", + ROI_MIN_TISSUE_COVERAGE_FOR_INTENSITY_QC, + ) + ) + _roi_focus_pct = float( + qc_thresholds.get("roi_focus_score_percentile", ROI_FOCUS_SCORE_PERCENTILE) + ) + _blur_prob_thresh = float(qc_thresholds.get("blur_prob_threshold", 0.5)) + _focus_cfg = qc_thresholds.get("focus") or {} + _ccfs_low_texture_threshold = float( + _focus_cfg.get("ccfs_low_texture_threshold", DEFAULT_CCFS_LOW_TEXTURE_THRESHOLD) + ) + logging.info( + f" roi_intensity_threshold={_roi_intensity_threshold}, " + f"min_tissue_cov={_min_tissue_cov}, " + f"roi_focus_pct={_roi_focus_pct}, " + f"blur_prob_thresh={_blur_prob_thresh}, " + f"ccfs_low_texture_threshold={_ccfs_low_texture_threshold}" + ) + + # ===== PHASE 1: LOAD ===== + logging.info("\n--- PHASE 1: LOAD ---") + + # Handle stain_names (default if None) + if stain_names is None: + stain_names_list = [ + "DAPI", + "Boundary (ATP1A1/E-Cadherin/CD45)", + "Interior - RNA (18S)", + "Protein (alphaSMA/Vimentin)", + ] + logging.info(f"Using default stain names: {stain_names_list}") + else: + logging.info(f"Stain names: {stain_names}") + if ";" in stain_names: + stain_names_list = stain_names.split(";") + else: + stain_names_list = [stain_names] + + # Load downsampled images for sample QC + morphology_focus_dir = Path(xenium_bundle_dir) / "morphology_focus" + if (morphology_focus_dir / "morphology_focus_0000.ome.tif").exists(): + xoa_morphology_files = [ + morphology_focus_dir / "morphology_focus_0000.ome.tif", + morphology_focus_dir / "morphology_focus_0001.ome.tif", + morphology_focus_dir / "morphology_focus_0002.ome.tif", + morphology_focus_dir / "morphology_focus_0003.ome.tif", + ] + else: + xoa_morphology_files = sorted( + list(morphology_focus_dir.glob("ch000*.ome.tif")), + key=lambda x: x.stem.split("_")[0], + ) + + # Boundary / IntRNA / Protein channels are optional — DAPI-only morphology + # bundles are valid (e.g. samples staining nuclei only). The downstream + # loader `_load_morphology_channels` returns `None` for missing channels; + # `generate_tissue_mask` and per-cell intensity emissions handle None. + if len(xoa_morphology_files) < 1: + raise ValueError("No morphology files found in morphology_focus/ directory") + if not xoa_morphology_files[0].exists(): + raise ValueError( + f"Morphology focus file does not exist: {xoa_morphology_files[0]}" + ) + + # Load and prepare data (creates output directories) + data = load_and_prepare_data(xenium_bundle_dir, outdir) + + # Load morphology images (level 3, downsampled) + logging.info("Loading morphology images...") + t0 = time.time() + # Use the robust loader that already handles 1/2/3-channel inputs + # (returns None for missing Boundary / IntRNA channels). 5 other call + # sites already exercise this loader (lines ~1366/1389/1659/1677/3456), + # so the None-tolerant code path is well-tested for DAPI-only bundles. + small0, small1, small2 = _load_morphology_channels(xoa_morphology_files, level=3) + _n_channels_loaded = sum(x is not None for x in (small0, small1, small2)) + _shape_str = f"DAPI={small0.shape}" + if small1 is not None: + _shape_str += f", Boundary={small1.shape}" + if small2 is not None: + _shape_str += f", IntRNA={small2.shape}" + logging.info( + f"Loaded morphology images ({_n_channels_loaded} channel(s)): {_shape_str}" + ) + logging.info(f"[TIMING] Loading morphology images: {time.time() - t0:.1f}s") + _log_mem("morphology loaded") + + # Check if cell data exists + has_cell_data = _check_cell_data_exists(xenium_bundle_dir) + if has_cell_data: + logging.info("Cell data detected - will perform cell-level analysis") + else: + logging.info("No cell data detected - will skip cell-level analysis") + + # ===== PHASE 2: PIXEL-LEVEL FOCUS MAPS ===== + logging.info("\n--- PHASE 2: PIXEL-LEVEL FOCUS MAPS ---") + + # Generate tissue masks and distance maps (cell-independent) + logging.info("Generating tissue masks and distance maps...") + t0 = time.time() + ( + whole_sample, + holes, + dense_intensity_regions, + distance_map, + distance_map2, + ms_whole_sample, + ms_distance_map, + ms_distance_map2, + ) = generate_tissue_mask(xoa_morphology_files, small0, small1, small2) + logging.info("Generated tissue masks and distance maps") + logging.info(f"[TIMING] Tissue mask generation: {time.time() - t0:.1f}s") + _log_mem("tissue mask") + + # Validate ROI size + if roi_size <= 0: + roi_size = 35 + logging.info(f"Using default tile size: {roi_size}px") + else: + logging.info(f"Using tile size: {roi_size}px") + + # Auto-detect available GPUs + available_gpus = detect_gpu_ids() + if available_gpus and max_gpus and len(available_gpus) > max_gpus: + logging.info( + f"Detected {len(available_gpus)} GPU(s) {available_gpus} but --max-gpus " + f"={max_gpus}; using {available_gpus[:max_gpus]}. The accelerator " + "directive sizes the Batch request only, it does not limit CUDA " + "visibility." + ) + available_gpus = available_gpus[:max_gpus] + if available_gpus: + logging.info(f"Detected {len(available_gpus)} GPU(s): {available_gpus}") + else: + logging.info("No GPUs detected, using CPU backend") + + # Calculate cell-independent grid ROI focus scores + logging.info("Calculating cell-independent grid tile focus scores...") + t0 = time.time() + if legacy_focus: + logging.info(" Using LEGACY per-tile loop method (--legacy-focus)") + df_grid_roi = calculate_roi_focusscore_without_laplace( + xoa_morphology_files, + roi_size=roi_size, + stride=None, + tissue_filter=True, + min_tissue_coverage=0.0, + ) + focus_maps = None + streamed = None + else: + logging.info(" Using convolution-based GPU-accelerated method") + # Streaming folds each tile into small reductions and never assembles a + # pixel plane. --save-dapi-maps-tiff is the one output that genuinely needs + # the planes, so asking for it selects the plane-based path. + _stream = stream_tiles and not save_dapi_maps_tiff + if stream_tiles and save_dapi_maps_tiff: + logging.info( + " --save-dapi-maps-tiff requires full pixel planes; " + "streaming disabled for this run" + ) + logging.info( + f" Tile reduction mode: {'streamed' if _stream else 'pixel planes'}" + ) + df_grid_roi, focus_maps, streamed = calculate_roi_focusscore( + xoa_morphology_files, + roi_size=roi_size, + stride=None, + tissue_filter=True, + min_tissue_coverage=0.0, + gpu_ids=available_gpus if available_gpus else None, + return_pixel_maps=True, + lap_sigma=_lap_sigma, + stream_tiles=_stream, + # Gated on has_cell_data for the same reason the CCFS block below is: + # a bundle without cells.zarr.zip has no masks to reduce over, and + # opening it during the tile pass would fail where the plane path + # simply skipped the whole cell section. + cell_masks_path=( + data["cell_masks_path"] if _stream and has_cell_data else None + ), + # Reuse the already-loaded level-3 DAPI plane and the tissue mask + # generate_tissue_mask just computed from it, instead of re-decoding + # level 3 and recomputing the mask inside calculate_roi_focusscore. + # `whole_sample` == compute_tissue_mask(small0)[0] (same small0, same + # min_size_hole=1500), so the result is bit-identical. + small0_ds=small0, + tissue_mask=whole_sample, + ) + logging.info(f"Calculated grid tile focus scores for {len(df_grid_roi):,} tiles") + logging.info(f"[TIMING] calculate_roi_focusscore(): {time.time() - t0:.1f}s") + _log_mem("focus maps built") + + # ===== PHASE 3: ROI-LEVEL ANALYSIS ===== + logging.info("\n--- PHASE 3: TILE-LEVEL ANALYSIS ---") + + # Calculate ROI blur threshold + logging.info("Calculating tile blur threshold...") + roi_threshold = calculate_roi_blur_threshold( + df_grid_roi, + intensity_threshold=_roi_intensity_threshold, + focus_percentile=_roi_focus_pct, + ) + logging.info(f" Calculated threshold: {roi_threshold:.2f} (raw score units)") + logging.info(f" Intensity threshold: {_roi_intensity_threshold}") + + # Fit 1D GMM model. On a very dim sample the GMM can fail to fit (e.g. no + # tissue tiles clear the intensity gate); fall back to the percentile + # threshold so the run still completes and produces a report, mirroring the + # try/except used by the 2D GMM below. + logging.info("Fitting 1D GMM model for tile focus scores...") + try: + gmm, blur_component_idx = fit_focus_gmm( + df_grid_roi, + intensity_threshold=_roi_intensity_threshold, + focus_col_name="dapi_focus_score", + ) + logging.info("Classifying tiles (1D GMM)...") + df_grid_roi = classify_roi_blur( + df_grid_roi, + gmm=gmm, + blur_component_idx=blur_component_idx, + blur_prob_threshold=_blur_prob_thresh, + intensity_threshold=_roi_intensity_threshold, + focus_col_name="dapi_focus_score", + ) + except Exception as e: + logging.warning(f" Warning: 1D GMM failed: {e}") + logging.info(" Using percentile-threshold fallback for 1D blur classification") + gmm = None + blur_component_idx = None + df_grid_roi = classify_roi_blur_by_threshold( + df_grid_roi, + roi_threshold=roi_threshold, + intensity_threshold=_roi_intensity_threshold, + focus_col_name="dapi_focus_score", + ) + n_blurred = int(df_grid_roi["is_blurred_gmm"].sum()) + n_in_focus = int((~df_grid_roi["is_blurred_gmm"]).sum()) + logging.info(f" Blurred: {n_blurred:,}, In-focus: {n_in_focus:,} (1D GMM)") + + # Fit 2D GMM if Laplacian variance available + gmm_2d = None + blur_component_idx_2d = None + if "dapi_lap_var" in df_grid_roi.columns: + logging.info("\nFitting 2D GMM model (focus_score + Laplacian variance)...") + try: + gmm_2d, blur_component_idx_2d = fit_focus_gmm_2d( + df_grid_roi, + intensity_threshold=_roi_intensity_threshold, + focus_col_name="dapi_focus_score", + ) + logging.info("Classifying tiles (2D GMM)...") + df_grid_roi = classify_roi_blur_2d( + df_grid_roi, + gmm=gmm_2d, + blur_component_idx=blur_component_idx_2d, + blur_prob_threshold=_blur_prob_thresh, + intensity_threshold=_roi_intensity_threshold, + focus_col_name="dapi_focus_score", + ) + n_blurred_2d = int(df_grid_roi["is_blurred_gmm_2d"].sum()) + n_in_focus_2d = int((~df_grid_roi["is_blurred_gmm_2d"]).sum()) + logging.info( + f" Blurred: {n_blurred_2d:,}, In-focus: {n_in_focus_2d:,} (2D GMM)" + ) + except Exception as e: + logging.warning(f" Warning: 2D GMM failed: {e}") + logging.info(" Continuing with 1D GMM results only") + gmm_2d = None + blur_component_idx_2d = None + else: + logging.info("\nSkipping 2D GMM: dapi_lap_var column not found") + + # Save grid ROI data to CSV + grid_roi_csv = data["outdir"] / "grid_roi_focus_scores.csv" + df_grid_roi.to_csv(grid_roi_csv, index=False) + logging.info(f"Saved grid tile data to {grid_roi_csv}") + + # Optional: write full-res per-pixel maps (very large); not needed for QC/SNR in-process + if focus_maps is not None and save_dapi_maps_tiff: + logging.info("Saving pixel-level focus maps as TIFF (--save-dapi-maps-tiff)...") + t0 = time.time() + save_pixel_focus_maps(focus_maps, data["outdir"]) + logging.info(f"[TIMING] save_pixel_focus_maps(): {time.time() - t0:.1f}s") + elif focus_maps is not None: + logging.info( + "Skipping pixel-map TIFF export (default). " + "Pass --save-dapi-maps-tiff to write dapi_*/boundary_*/intrna_* map TIFFs." + ) + elif streamed is not None: + logging.info( + "No pixel-map TIFF export: tiles were streamed, so no full-resolution " + "planes exist. Pass --save-dapi-maps-tiff to build them." + ) + + # Free Laplacian map — no longer needed after TIFF export / GMM fitting. + # Remaining consumers (SNR, figures, CCFS) only use focus/mean maps. + if focus_maps is not None: + for _k in ("dapi_lap_var_map",): + focus_maps.pop(_k, None) + + # Save threshold configuration + total_rois_gmm = int(len(df_grid_roi)) + n_blurred_gmm = int(df_grid_roi["is_blurred_gmm"].sum()) + pct_blurred_gmm = ( + (n_blurred_gmm / total_rois_gmm * 100.0) if total_rois_gmm > 0 else 0.0 + ) + + threshold_config = { + "roi_focus_score_threshold": float(roi_threshold), + "roi_intensity_threshold": float(_roi_intensity_threshold), + "roi_focus_score_percentile": float(_roi_focus_pct), + "threshold_method": "Option B: Percentile from tissue tiles (intensity >= threshold) + fixed intensity threshold", + "units": { + "roi_focus_score_threshold_percentile": "raw score units", + "roi_intensity_threshold": "raw pixel intensity (16-bit, 0-65535)", + "component_means_log1p_focus": "log1p(raw focus score)", + }, + } + + # The 1D GMM may have failed on a dim sample (gmm is None), in which case + # the percentile-threshold fallback was used. Guard the dereference the same + # way the gmm_2d stanza below is guarded, and emit a fallback marker instead. + if gmm is not None: + threshold_config["gmm_1d"] = { + "n_components": int(gmm.n_components), + "blur_component_index": int(blur_component_idx), + "component_means_log1p_focus": [float(m) for m in gmm.means_.flatten()], + "component_weights": [float(w) for w in gmm.weights_.flatten()], + "blur_prob_threshold": float(_blur_prob_thresh), + "fraction_rois_blurred_gmm": float(pct_blurred_gmm), + "features": ["log1p(dapi_focus_score)"], + } + else: + threshold_config["gmm_1d"] = { + "status": "fallback", + "method": "percentile_threshold", + "roi_focus_score_threshold": float(roi_threshold), + "blur_prob_threshold": float(_blur_prob_thresh), + "fraction_rois_blurred_gmm": float(pct_blurred_gmm), + "features": ["dapi_focus_score <= roi_focus_score_threshold"], + } + + if "is_blurred_gmm_2d" in df_grid_roi.columns and gmm_2d is not None: + n_blurred_gmm_2d = int(df_grid_roi["is_blurred_gmm_2d"].sum()) + pct_blurred_gmm_2d = ( + (n_blurred_gmm_2d / total_rois_gmm * 100.0) if total_rois_gmm > 0 else 0.0 + ) + threshold_config["gmm_2d"] = { + "n_components": int(gmm_2d.n_components), + "blur_component_index": int(blur_component_idx_2d), + "component_means": [[float(m[0]), float(m[1])] for m in gmm_2d.means_], + "component_weights": [float(w) for w in gmm_2d.weights_.flatten()], + "blur_prob_threshold": float(_blur_prob_thresh), + "fraction_rois_blurred_gmm": float(pct_blurred_gmm_2d), + "features": ["log1p(dapi_focus_score)", "log1p(dapi_lap_var)"], + } + threshold_config["units"]["component_means_2d"] = ( + "[log1p(focus_score), log1p(lap_var)]" + ) + + threshold_json = data["outdir"] / "roi_blur_threshold.json" + with open(threshold_json, "w") as f: + json.dump(threshold_config, f, indent=2) + logging.info(f"Saved tile blur threshold configuration to {threshold_json}") + + # Save ROI count + roi_count_file = data["outdir"] / "roi_count.txt" + with open(roi_count_file, "w") as f: + f.write(str(len(df_grid_roi))) + + # Calculate ROI intensities + logging.info("Calculating tile intensities...") + t0 = time.time() + df_roi_intensities = calculate_roi_intensities(xoa_morphology_files, df_grid_roi) + logging.info(f"[TIMING] Tile intensity calculation: {time.time() - t0:.1f}s") + + snr_summary = None + if not no_snr: + logging.info("Computing SNR metrics (roi_qc / snr_metrics)...") + t_snr = time.time() + pix_um = snr_metrics.read_xenium_pixel_size_um(Path(xenium_bundle_dir)) + if pix_um is None: + logging.info( + " experiment.xenium missing or invalid pixel_size — transcript SNR may skip" + ) + try: + df_roi_intensities, snr_summary = snr_metrics.compute_snr_summary( + df_roi_intensities, + bundle_dir=Path(xenium_bundle_dir), + outdir=data["outdir"], + focus_maps=focus_maps, + # Positional order, not a join: roi_snr_db is indexed by the + # compute_roi_grid() order that built df_grid_roi, and + # calculate_roi_intensities() returns df_grid_roi.copy() without + # filtering or reordering, so row i is the same ROI in both. If that + # function ever starts filtering, this silently mislabels every tile + # -- the grid/DataFrame half of the invariant is pinned by + # test_agrees_with_the_dataframe_coordinates. + roi_snr_db=streamed.roi_snr_db if streamed else None, + pixel_size_um=pix_um, + otsu_max_rois=snr_otsu_max_rois, + save_roi_tx_table=not snr_no_roi_tx_table, + snr_include_moran=snr_with_moran, + # Same stride as grid in calculate_roi_focusscore (stride=None → roi_size). + roi_grid_stride=(roi_size, roi_size), + snr_thresholds=qc_thresholds.get("snr") or {}, + ) + except Exception as e: + logging.warning("SNR metrics failed (continuing image QC): %s", e) + snr_summary = { + "status": "error", + "error": str(e), + "components": {}, + "verdict": {"overall_snr_verdict": "NOT_COMPUTED"}, + } + logging.info(f"[TIMING] SNR metrics: {time.time() - t_snr:.1f}s") + _log_mem("SNR metrics") + else: + logging.info("SNR metrics disabled (--no-snr)") + df_grid_roi = df_roi_intensities + + # Assess raw intensity quality (thresholds from YAML or defaults) + _ch_cfg = qc_thresholds.get("channels") or {} + + # XOA-version-specific intensity floors (2026-06-22, qc_drift_analysis). + # XOA 4.0 images are ~14x dimmer than 3.x, so a single floor can't serve both. + # Pick `intensity_critical_v{major}` when present, else the legacy + # `intensity_critical`, else the module default. Version unknown → legacy/default. + _xoa_major = read_xenium_major_version(Path(xenium_bundle_dir)) + logging.info(f" XOA major version for intensity floors: {_xoa_major}") + + def _pick_critical(ch_key, default): + # Case-insensitive channel lookup so a casing change in the YAML raises + # instead of silently reverting every floor to the module default. + ch = _yaml_channel_cfg(_ch_cfg, ch_key) + if _xoa_major is not None: + v = ch.get(f"intensity_critical_v{_xoa_major}") + if isinstance(v, (int, float)): + return v + return ch.get("intensity_critical", default) + + _ic_dapi = _pick_critical("DAPI", _INTENSITY_CRITICAL_DEFAULTS["dapi"]) + _ic_boundary = _pick_critical("boundary", _INTENSITY_CRITICAL_DEFAULTS["boundary"]) + _ic_intrna = _pick_critical("intRNA", _INTENSITY_CRITICAL_DEFAULTS["intrna"]) + logging.info("Assessing raw intensity quality...") + intensity_stats = assess_raw_intensity_quality( + df_roi_intensities, + dapi_threshold_critical=_ic_dapi, + boundary_threshold_critical=_ic_boundary, + intrna_threshold_critical=_ic_intrna, + min_tissue_coverage=_min_tissue_cov, + channel_pct_thresholds=_ch_cfg, + ) + + # Save intensity statistics + intensity_json = data["outdir"] / "intensity_assessment.json" + with open(intensity_json, "w") as f: + json.dump(intensity_stats, f, indent=2) + + # Generate ROI figures (cell-independent) + logging.info("Generating tile-level figures...") + t0 = time.time() + if figures: + generate_roi_figures( + data, + small0, + small1, + small2, + distance_map, + distance_map2, + whole_sample, + holes, + dense_intensity_regions, + df_grid_roi, + df_roi_intensities, + intensity_stats, + xoa_morphology_files, + focus_maps=focus_maps, + focus_heatmap=streamed.focus_heatmap if streamed else None, + snr_thresholds=(qc_thresholds.get("snr") or {}).get("roi_tx") or {}, + multistain_whole_sample=ms_whole_sample, + multistain_distance_map=ms_distance_map, + multistain_distance_map2=ms_distance_map2, + figure_source_tables=figure_source_tables, + ) + logging.info(f"[TIMING] generate_roi_figures(): {time.time() - t0:.1f}s") + _log_mem("tile figures") + + # Free focus-score maps no longer needed. CCFS only requires + # dapi_focus_map, dapi_mean_map, boundary_mean_map, intrna_mean_map. + if focus_maps is not None: + for _k in ("boundary_focus_map", "intrna_focus_map"): + focus_maps.pop(_k, None) + + # Save ROI QC metrics + logging.info("Saving tile QC metrics...") + save_roi_qc_metrics( + df_grid_roi, + intensity_stats, + data["outdir"], + roi_size=roi_size, + snr_summary=snr_summary, + distance_map=distance_map, + distance_map2=distance_map2, + multistain_whole_sample=ms_whole_sample, + multistain_distance_map=ms_distance_map, + multistain_distance_map2=ms_distance_map2, + edge_distance_threshold=-25.0, + hole_distance_threshold=-25.0, + min_tissue_coverage_for_qc=_min_tissue_cov, + qc_thresholds=qc_thresholds, + lap_sigma=_lap_sigma, + segmentation_software=resolve_segmentation_software( + xenium_bundle_dir, pipeline_segmentation, is_resegmented + ), + xoa_version=read_xenium_analysis_sw_version(xenium_bundle_dir), + ) + + # ===== PHASE 4: CELL-LEVEL ANALYSIS (conditional on cell data) ===== + if has_cell_data: + logging.info("\n--- PHASE 4: CELL-LEVEL ANALYSIS ---") + + # Prepare cell-centred output directory + figures_dir_cell = data["outdir"] / "figures" + figures_dir_cell.mkdir(parents=True, exist_ok=True) + if figure_source_tables: + figures_source_dir_cell = figures_dir_cell / "figures_source" + figures_source_dir_cell.mkdir(parents=True, exist_ok=True) + + # Prepare data dict for cell-level functions + cell_data = dict(data) + cell_data["figures_dir"] = figures_dir_cell + + # Unpack analysis.tar.gz if needed (test data) + xenium_bundle_dir_path = Path(xenium_bundle_dir) + if (xenium_bundle_dir_path / "analysis.tar.gz").is_file(): + import shutil + + shutil.unpack_archive( + xenium_bundle_dir_path / "analysis.tar.gz", extract_dir="." + ) + cell_data["clusters_csv_path"] = ( + Path("analysis") + / "clustering" + / "gene_expression_kmeans_10_clusters" + / "clusters.csv" + ) + cell_data["umap_path"] = ( + Path("analysis") + / "umap" + / "gene_expression_2_components" + / "projection.csv" + ) + else: + cell_data["clusters_csv_path"] = ( + xenium_bundle_dir_path + / "analysis" + / "clustering" + / "gene_expression_kmeans_10_clusters" + / "clusters.csv" + ) + cell_data["umap_path"] = ( + xenium_bundle_dir_path + / "analysis" + / "umap" + / "gene_expression_2_components" + / "projection.csv" + ) + cell_data["cells_parquet_path"] = xenium_bundle_dir_path / "cells.parquet" + cell_data["cell_masks_path"] = xenium_bundle_dir_path / "cells.zarr.zip" + + # Load spatial data + logging.info("Loading spatial data and masks...") + t0 = time.time() + try: + df_spatial, cellseg_mask, cell_masks_zarr = load_spatial_data(cell_data) + logging.info(f"Loaded {len(df_spatial):,} cells") + logging.info(f"[TIMING] Loading spatial data: {time.time() - t0:.1f}s") + except (FileNotFoundError, KeyError, AttributeError) as e: + logging.warning(f"Warning: Could not load spatial data: {e}") + logging.info("Skipping cell-level analysis") + has_cell_data = False + + if has_cell_data: + # Map distances to cells + logging.info("Mapping distances to cells...") + t0 = time.time() + df_spatial = map_distances_to_cells( + df_spatial, distance_map, distance_map2, dense_intensity_regions + ) + logging.info(f"[TIMING] map_distances_to_cells(): {time.time() - t0:.1f}s") + + # Calculate CCFS measurements + logging.info("Calculating CCFS measurements...") + t0 = time.time() + if streamed is not None and streamed.has_per_cell: + logging.info(" Using per-cell sums reduced during the tile pass") + myData = calculate_ccfs_from_focus_maps( + None, + cell_masks_zarr, + cellseg_mask, + xoa_morphology_files, + streamed=streamed, + ) + elif ( + focus_maps is not None + and "dapi_focus_map" in focus_maps + and "dapi_mean_map" in focus_maps + ): + # NEW: Use pixel-map aggregation + logging.info(" Using pixel-map aggregation (scipy.ndimage.mean)") + myData = calculate_ccfs_from_focus_maps( + focus_maps, cell_masks_zarr, cellseg_mask, xoa_morphology_files + ) + else: + # Fallback: use regionprops-based method + logging.info(" Using regionprops-based method (legacy)") + myData = calculate_ccfs_measurements( + xoa_morphology_files, cellseg_mask, cell_masks_zarr + ) + logging.info(f"Calculated CCFS for {len(myData):,} cells") + logging.info(f"[TIMING] CCFS calculation: {time.time() - t0:.1f}s") + _log_mem("per-cell CCFS") + + # All pixel-level focus maps consumed — free remaining memory. The streamed + # reductions are kept: nothing downstream re-reads them, but they are tens + # of MB, and dropping the name here would break the `streamed` references + # in the figure calls that follow. + del focus_maps + focus_maps = None + + # Add boolean columns to myData for thresholding (needed by generate_all_figures) + ccfs_threshold = _ccfs_low_texture_threshold + myData["is_low_nuclear_texture"] = myData["CCFS_DAPI"] <= ccfs_threshold + myData["is_high_nuclear_texture"] = myData["CCFS_DAPI"] > ccfs_threshold + + # Map ROI focus scores to cells + logging.info("Mapping tile focus scores to cells...") + t0 = time.time() + roi_mapped = map_grid_roi_to_cells(df_grid_roi, df_spatial, overlapping=False) + logging.info( + f"Mapped tile focus scores to {len(roi_mapped[roi_mapped['DAPI_RFSnorm_roi'].notna()]):,} cells" + ) + logging.info(f"[TIMING] map_grid_roi_to_cells(): {time.time() - t0:.1f}s") + + # Load ROI blur threshold + roi_threshold_cell, roi_intensity_threshold = load_roi_blur_threshold( + data["outdir"] + ) + if roi_threshold_cell is None or roi_intensity_threshold is None: + roi_threshold_cell = roi_threshold + roi_intensity_threshold = _roi_intensity_threshold + + # Create merged dataset (superset version with ROI data) + logging.info("Creating final merged dataset...") + t0 = time.time() + new_df, calculated_roi_threshold = create_final_merged_data( + df_spatial, + myData, + roi_data=roi_mapped, + ccfs_threshold=_ccfs_low_texture_threshold, + roi_threshold=roi_threshold_cell, + roi_intensity_threshold=roi_intensity_threshold, + ) + logging.info(f"Created merged dataset with {len(new_df):,} cells") + logging.info(f"[TIMING] create_final_merged_data(): {time.time() - t0:.1f}s") + + # Alias dense_intensity_regions as artefacts for generate_all_figures compatibility + artefacts = dense_intensity_regions + + # Ensure In-Area-with-Artefact column exists (needed by generate_all_figures) + if "In-Area-with-Artefact" not in df_spatial.columns: + # Map dense_intensity_regions to the artefact column name + if "Dense-Intensity-Region-ID" in df_spatial.columns: + df_spatial["In-Area-with-Artefact"] = df_spatial[ + "Dense-Intensity-Region-ID" + ] + else: + df_spatial["In-Area-with-Artefact"] = 0 + + if "In-Area-with-Artefact" not in new_df.columns: + if "Dense-Intensity-Region-ID" in new_df.columns: + new_df["In-Area-with-Artefact"] = new_df["Dense-Intensity-Region-ID"] + else: + new_df["In-Area-with-Artefact"] = 0 + + # Ensure has_artifacts column exists + if "has_artifacts" not in new_df.columns: + new_df["has_artifacts"] = new_df.get( + "has_dense_intensity_regions", + new_df.get("In-Area-with-Artefact", 0) > 0, + ) + + # Generate 13 Quarto-required figures + logging.info("Generating Quarto-required figures (13 PNGs)...") + t0 = time.time() + if figures: + generate_all_figures( + cell_data, + df_spatial, + new_df, + myData, + small0, + small1, + small2, + distance_map, + distance_map2, + whole_sample, + holes, + artefacts, + ccfs_low_texture_threshold=_ccfs_low_texture_threshold, + multistain_whole_sample=ms_whole_sample, + multistain_distance_map=ms_distance_map, + multistain_distance_map2=ms_distance_map2, + figure_source_tables=figure_source_tables, + ) + logging.info(f"[TIMING] generate_all_figures(): {time.time() - t0:.1f}s") + + # Generate cell-centred comparison figures (ROI vs CCFS) + if figures: + figures_cell_centred_dir = data["outdir"] / "figures_cell_centred" + figures_cell_centred_dir.mkdir(parents=True, exist_ok=True) + figures_cell_centred_source = ( + (figures_cell_centred_dir / "figures_source") + if figure_source_tables + else None + ) + if figures_cell_centred_source is not None: + figures_cell_centred_source.mkdir(parents=True, exist_ok=True) + t0 = time.time() + generate_cell_figures( + cell_data, + new_df, + myData, + figures_cell_centred_dir, + figures_cell_centred_source, + roi_threshold=calculated_roi_threshold, + roi_intensity_threshold=roi_intensity_threshold, + ccfs_low_texture_threshold=_ccfs_low_texture_threshold, + ) + logging.info(f"[TIMING] generate_cell_figures(): {time.time() - t0:.1f}s") + _log_mem("cell figures") + + # Save cell QC metrics (superset version with ROI metrics) + logging.info("Saving cell QC metrics...") + t0 = time.time() + save_cell_qc_metrics(new_df, data["outdir"], roi_size=roi_size) + logging.info(f"[TIMING] save_cell_qc_metrics(): {time.time() - t0:.1f}s") + + # Save simple image_qc_metrics.json for Quarto compatibility + logging.info("Saving Quarto-compatible image_qc_metrics.json...") + n_total = len(new_df) + n_ccfs_low_texture = int(new_df["is_low_nuclear_texture"].sum()) + qc_metrics = { + "total_cells": n_total, + "cells_with_transcripts": len(new_df[new_df["transcript_counts"] > 0]), + "mean_transcript_count": float(new_df["transcript_counts"].mean()), + "median_transcript_count": float(new_df["transcript_counts"].median()), + "mean_ccfs_dapi": float(new_df["CCFS_DAPI"].mean()), + "median_ccfs_dapi": float(new_df["CCFS_DAPI"].median()), + "ccfs_low_texture_threshold": float(_ccfs_low_texture_threshold), + "cells_high_nuclear_texture": len( + new_df[new_df["is_high_nuclear_texture"]] + ), + "cells_low_nuclear_texture": n_ccfs_low_texture, + "pct_low_nuclear_texture": round(100.0 * n_ccfs_low_texture / n_total, 4) + if n_total > 0 + else 0.0, + "cells_near_edge": len(new_df[new_df["is_near_edge"]]), + "cells_near_holes": len(new_df[new_df["is_near_hole"]]), + "cells_with_artifacts": int(new_df["has_artifacts"].sum()), + "clusters_present": sorted(new_df["Cluster_kmeans10"].unique().tolist()), + "segmentation_methods": sorted( + new_df["segmentation_method"].unique().tolist() + ), + } + # Phase 2a-revised: per-cell aggregate emissions for Section 9.A's + # sample-level aggregates table + 9.B's roi_tissue_coverage row. + # Naming asymmetry note: per-cell DAPI intensity column is `mean_intensity` + # (no suffix); aggregate JSON key adds `_DAPI` for sibling-channel + # consistency with mean_intensity_Boundary / mean_intensity_IntRNA. See + # v3 plan §9 assumption #9. + for _src_col, _agg_key_base, _decimals in ( + ("mean_intensity", "intensity_DAPI", 4), + ("mean_intensity_Boundary", "intensity_Boundary", 4), + ("mean_intensity_IntRNA", "intensity_IntRNA", 4), + ("area_nucleus", "area_nucleus", 2), + ("area_cell", "area_cell", 2), + ): + if _src_col not in new_df.columns: + continue + _series = new_df[_src_col] + _mean = _series.mean() + _median = _series.median() + if pd.notna(_mean): + qc_metrics[f"mean_{_agg_key_base}"] = round(float(_mean), _decimals) + if pd.notna(_median): + qc_metrics[f"median_{_agg_key_base}"] = round(float(_median), _decimals) + + # Phase 13 (v4): % cells below per-channel intensity floor for the + # 9.A Tier 1 verdict rows. 2026-06-23: use the SAME XOA-version-specific + # floors as the tile-level intensity QC (via _pick_critical), so a dim + # XOA-4.0 sample isn't flagged against the bright-era 500/100/300 floors. + # NaN-intensity cells are excluded from the numerator (NaN < cutoff → + # False) but stay in the n_total denominator — same convention as + # `pct_cells_in_low_coverage_tiles`. + if n_total > 0: + _cell_intensity_emissions = ( + ( + "mean_intensity", + _pick_critical("DAPI", _INTENSITY_CRITICAL_DEFAULTS["dapi"]), + "pct_cells_below_intensity_DAPI", + ), + ( + "mean_intensity_Boundary", + _pick_critical( + "boundary", _INTENSITY_CRITICAL_DEFAULTS["boundary"] + ), + "pct_cells_below_intensity_Boundary", + ), + ( + "mean_intensity_IntRNA", + _pick_critical("intRNA", _INTENSITY_CRITICAL_DEFAULTS["intrna"]), + "pct_cells_below_intensity_IntRNA", + ), + ) + for _src_col, _cutoff, _emit_key in _cell_intensity_emissions: + if _src_col not in new_df.columns: + continue + _n_below = int((new_df[_src_col] < _cutoff).sum()) + qc_metrics[_emit_key] = round(100.0 * _n_below / n_total, 4) + + # 9.B (Phase 2a-revised, folds Phase 11): % cells in low-coverage tiles + # (roi_tissue_coverage < 0.5). Informational; no PASS/WARN/FAIL pill until + # calibration. NaN coverage cells are excluded (NaN < 0.5 → False), so the + # count reflects only cells with assigned ROI tile coverage data. + # NOT disjoint from cells_evaluated_for_blur below, which uses + # ROI_MIN_TISSUE_COVERAGE_FOR_INTENSITY_QC (0.2) as its floor: a cell at + # coverage 0.35 is counted in both, so the two do not partition the cells and + # can sum above n_total. The 0.5 here is deliberately independent of the blur + # denominator -- do not "tidy" it to the constant without recalibrating, since + # it would move a published number. + # 2026-06-26 (multi-stain): roi_tissue_coverage here is DAPI-based (it is also the + # denominator of pct_blurred_gmm_2d_roi below, which MUST stay DAPI to match the + # tile-level DAPI blur figure). A multi-stain cell-coverage view for this + # informational count is a deferred follow-up; tile-level extent already uses all + # stains. Report prose marks this as DAPI-derived. + if "roi_tissue_coverage" in new_df.columns and n_total > 0: + _n_low_cov = int((new_df["roi_tissue_coverage"] < 0.5).sum()) + qc_metrics["cells_in_low_coverage_tiles"] = _n_low_cov + qc_metrics["pct_cells_in_low_coverage_tiles"] = round( + 100.0 * _n_low_cov / n_total, 4 + ) + + # GMM-ROI blur metrics (if cell-to-ROI mapping was performed). + # 2026-06-24: pct_blurred_gmm_2d_roi is reported over SOLID-tissue cells + # (roi_tissue_coverage >= ROI_MIN_TISSUE_COVERAGE_FOR_INTENSITY_QC) so it + # matches the tile-level "tiles in + # focus" metric (also tissue-filtered). Cells in low-coverage / edge tiles + # are force-labelled blurred regardless of optical focus and otherwise + # inflate this far above the tile figure (e.g. 50% cells vs 9% tiles). + # The all-cells value is kept for context / the report's "why it differs" + # note, alongside the existing pct_cells_in_low_coverage_tiles. + if "is_blurred_gmm_2d_roi" in new_df.columns: + _blur = new_df["is_blurred_gmm_2d_roi"] + _n_blur_all = int(_blur.sum()) + qc_metrics["pct_blurred_gmm_2d_roi_all_cells"] = ( + round(100.0 * _n_blur_all / n_total, 4) if n_total > 0 else 0.0 + ) + if "roi_tissue_coverage" in new_df.columns: + _solid = ( + new_df["roi_tissue_coverage"] + >= ROI_MIN_TISSUE_COVERAGE_FOR_INTENSITY_QC + ) + else: + _solid = _blur.notna() + _n_solid = int(_solid.sum()) + n_gmm = int(_blur[_solid].sum()) + qc_metrics["cells_blurred_gmm_2d_roi"] = n_gmm + qc_metrics["cells_evaluated_for_blur"] = _n_solid + qc_metrics["pct_blurred_gmm_2d_roi"] = ( + round(100.0 * n_gmm / _n_solid, 4) if _n_solid > 0 else 0.0 + ) + qc_metrics["pct_blurred_gmm_2d_roi_denominator"] = ( + # Derived from the constant actually applied above. It was + # hardcoded "..._ge_0.5" while the filter used 0.2, so the + # published label named a cutoff the code did not use. + "solid_tissue_cells_coverage_ge_" + f"{ROI_MIN_TISSUE_COVERAGE_FOR_INTENSITY_QC:g}" + if "roi_tissue_coverage" in new_df.columns + else "all_cells" + ) + agreement = int( + ( + new_df["is_low_nuclear_texture"] == new_df["is_blurred_gmm_2d_roi"] + ).sum() + ) + qc_metrics["ccfs_gmm_agreement_pct"] = ( + round(100.0 * agreement / n_total, 4) if n_total > 0 else 0.0 + ) + # Per-cluster blur stats for cluster-outlier detection + if "Cluster_kmeans10" in new_df.columns: + blur_col = ( + "is_blurred_gmm_2d_roi" + if "is_blurred_gmm_2d_roi" in new_df.columns + else "is_low_nuclear_texture" + ) + cluster_blur = {} + for cl, grp in new_df.groupby("Cluster_kmeans10"): + n_cl = len(grp) + n_blur_cl = int(grp[blur_col].sum()) + pct_blur_cl = round(100.0 * n_blur_cl / n_cl, 2) if n_cl > 0 else 0.0 + cluster_blur[int(cl)] = { + "n_cells": n_cl, + "n_blurred": n_blur_cl, + "pct_blurred": pct_blur_cl, + "median_ccfs_dapi": round(float(grp["CCFS_DAPI"].median()), 4) + if not pd.isna(grp["CCFS_DAPI"].median()) + else 0.0, + } + qc_metrics["cluster_blur"] = cluster_blur + qc_metrics["cluster_blur_method"] = blur_col + # Flag clusters that stand apart from the rest (robust MAD z-score + + # absolute floor). See detect_cluster_outliers. + outlier_clusters = detect_cluster_outliers(cluster_blur, "pct_blurred") + if outlier_clusters: + qc_metrics["cluster_blur_outliers"] = outlier_clusters + + # Per-cluster CCFS-low-texture stats for cluster-outlier detection. + # Parallel to cluster_blur above, but always derived from + # is_low_nuclear_texture (independent of whether tile-mapped GMM blur is + # available — distinct from cluster_blur, which falls back to + # is_low_nuclear_texture only when is_blurred_gmm_2d_roi is absent). + # Consumed by the §4.3 Tier-3 row and the §4.6 per-cluster breakdown in + # notebooks/xenium_image_qc_report.qmd. Hidden by the qmd until this + # field is present in the JSON. + if ( + "Cluster_kmeans10" in new_df.columns + and "is_low_nuclear_texture" in new_df.columns + ): + cluster_ccfs = {} + for cl, grp in new_df.groupby("Cluster_kmeans10"): + n_cl = len(grp) + n_low_cl = int(grp["is_low_nuclear_texture"].sum()) + pct_low_cl = round(100.0 * n_low_cl / n_cl, 2) if n_cl > 0 else 0.0 + cluster_ccfs[int(cl)] = { + "n_cells": n_cl, + "n_low_texture": n_low_cl, + "pct_low_texture": pct_low_cl, + "median_ccfs_dapi": round(float(grp["CCFS_DAPI"].median()), 4) + if not pd.isna(grp["CCFS_DAPI"].median()) + else 0.0, + } + qc_metrics["cluster_ccfs"] = cluster_ccfs + # Same robust rule as cluster_blur. The absolute floor backstops the + # MAD=0 case — most clusters sit near 0% low-texture, so a real spike + # is caught by the floor even when the spread is degenerate. + ccfs_outlier_clusters = detect_cluster_outliers( + cluster_ccfs, "pct_low_texture" + ) + if ccfs_outlier_clusters: + qc_metrics["cluster_ccfs_outliers"] = ccfs_outlier_clusters + + with open(data["outdir"] / "image_qc_metrics.json", "w") as f: + json.dump(qc_metrics, f, indent=2) + + # Save image_qc_metrics.csv + cell_id_col = new_df.pop("cell_id") + new_df.insert(0, "cell_id", cell_id_col) + new_df.to_csv(data["outdir"] / "image_qc_metrics.csv", index=False) + + # Save dense intensity region summary + if "Dense-Intensity-Region-ID" in new_df.columns: + region_summary = ( + new_df[new_df["Dense-Intensity-Region-ID"] > 0] + .groupby("Dense-Intensity-Region-ID") + .agg({"cell_id": "count", "x": "mean", "y": "mean"}) + .reset_index() + ) + region_summary.columns = [ + "Dense-Intensity-Region-ID", + "cell_count", + "mean_x", + "mean_y", + ] + region_summary = region_summary.sort_values("Dense-Intensity-Region-ID") + region_summary["annotation"] = "" + region_summary.to_csv( + data["outdir"] / "dense_intensity_regions_summary.csv", index=False + ) + else: + logging.info("\n--- PHASE 4: SKIPPED (no cell data) ---") + + # ===== PHASE 5: FINALIZE ===== + logging.info("\n--- PHASE 5: FINALIZE ---") + + # Save versions file + logging.info("Saving versions file...") + save_versions_file(data["outdir"]) + + logging.info(f"\n[TIMING] Total pipeline time: {time.time() - t_total_start:.1f}s") + _log_mem_summary() + logging.info("=" * 60) + logging.info("Combined Image QC Pipeline completed successfully!") + logging.info(f" Tile figures: {data['figures_dir']}") + if has_cell_data: + logging.info(f" Cell figures: {data['outdir'] / 'figures'}") + logging.info(f" Cell metrics: {data['outdir'] / 'image_qc_metrics.json'}") + logging.info(f" Cell CSV: {data['outdir'] / 'image_qc_metrics.csv'}") + logging.info(f" Tile metrics: {data['outdir'] / 'roi_qc_metrics.json'}") + logging.info(f" Versions: {data['outdir'] / 'versions.yml'}") + logging.info("=" * 60) + + +if __name__ == "__main__": + main() diff --git a/bin/roi_image_qc_thresholds.yaml b/bin/roi_image_qc_thresholds.yaml new file mode 100644 index 00000000..f45caec8 --- /dev/null +++ b/bin/roi_image_qc_thresholds.yaml @@ -0,0 +1,344 @@ +image_qc: + # Gaussian sigma for Laplacian of Gaussian (LoG) pre-smoothing. + # Controls the spatial scale of edges detected; suppresses noise below that scale. + # Higher values detect coarser edges and are more tolerant of noise; lower values + # are more sensitive to fine detail but noisier. Stored in roi_qc_metrics.json + # for calibration tracking. + lap_sigma: 1.0 + # --- Tile pre-filtering thresholds (these three are NOT redundant) --- + # 1) TISSUE GATE: Minimum mean DAPI intensity (16-bit, 0-65535) for a tile to + # be treated as tissue in blur/focus classification. Tiles below this are + # auto-marked as background and excluded from GMM fitting, intensity QC, and + # SNR calculations. Prevents empty-slide / coverslip tiles from contaminating + # focus statistics. + roi_intensity_threshold: 100.0 + # 2) COVERAGE GATE: Minimum fraction of a tile that must overlap the binary + # tissue mask to be included in per-channel intensity assessments (DAPI, + # boundary, intRNA warn/fail percentages). Partial-tissue tiles (e.g. at + # tissue edges) are excluded so they do not artificially inflate the "dim + # tile" fraction. Does NOT affect blur/focus classification (which uses the + # intensity gate above). + # 2026-06-26: lowered 0.5 -> 0.2. The background-from-nonzero tissue mask is tighter + # (un-flooded), so real tissue tiles are only thinly covered; 0.5 discarded them and + # left dim/sparse samples with too few tissue tiles for the GMM / intensity QC. + min_tissue_coverage_for_intensity_qc: 0.2 + # 3) GMM FALLBACK: Percentile of raw focus scores (after excluding + # low-intensity tiles) used as the blur/focus separation threshold ONLY when + # GMM fitting fails (too few tiles, singular covariance, convergence failure). + # In normal operation the GMM posterior probability (blur_prob_threshold + # below) is used instead. + roi_focus_score_percentile: 5.0 + # GMM posterior probability threshold: tile classified as blurred when + # P(blur_component) > this value. This is the primary blur classification + # gate used during normal GMM operation. + blur_prob_threshold: 0.5 + channels: + DAPI: + # Absolute minimum tile intensity (16-bit) below which a tile is "too dim". + # 2026-06-22 (qc_drift_analysis): floors are now XOA-version-specific — + # XOA 4.0 images are ~14x dimmer than 3.x, so one floor can't serve both. + # image_qc.py picks intensity_critical_v{major} by the bundle's + # analysis_sw_version; intensity_critical below is the fallback when the + # version is unknown. CAVEAT: the report's v4 cohort is small (~29 samples, + # mostly one site) — treat v4 floors as provisional. + intensity_critical_v3: 50 + intensity_critical_v4: 50 + intensity_critical: 500 # fallback when XOA version is unknown + # Fraction of tissue tiles below intensity_critical that triggers WARN. + # "Tissue tile" here means coverage >= min_tissue_coverage_for_intensity_qc + # (0.2 since 2026-06-26); this comment said 0.5 until 2026-08-24. + # + # 2026-06-22 (qc_drift_analysis): WARN-only — the FAIL tier (intensity_fail) + # is removed. Intensity does not track quality post-XOA-4.0, so a dim sample + # is flagged for review, never hard-failed on intensity alone. intensity_warn + # stays 0.40 (WARN when >= 40% of tissue tiles fall below the floor). + # + # Brain/neural tissue caveat: large diffuse nuclei produce inherently dim + # DAPI tiles. DAPI intensity WARN should be interpreted with caution for + # neural tissue — cross-check with boundary/intRNA and the blurry-tile %. + intensity_warn: 0.40 + # --- PLANNED: Spatial CV per channel (not yet implemented) --- + # Coefficient of variation (std/mean) of per-tile DAPI intensity across + # tissue tiles. Captures spatially heterogeneous staining: current + # intensity stats (p10-p90) show the range but not whether dim tiles are + # scattered or concentrated in one region. High spatial CV flags slides + # with localized staining failures that aggregate statistics average out. + # spatial_cv_warn: 0.40 + # spatial_cv_fail: 0.65 + # Tile blur fraction (GMM-2D, tissue-filtered). Cutoffs empirically + # calibrated 2026-05-14 against 9 calibration samples + # (lung/liver/pancreas/brain). GMM is trained on tissue-only tiles + # (intensity >= ROI_INTENSITY_THRESHOLD) — see + # bin/image_qc.py:fit_focus_gmm_2d. + # + # Observed tissue-filtered % blurry floor: ~27% even on best samples, + # because the GMM-2D classifier (a) always fits two components → some + # tiles always tagged blurry, and (b) force-classifies low-intensity / + # missing-lap_var tiles as blurry (bin/image_qc.py classify_focus_gmm_2d). + # Previous 0.15/0.30 cutoffs were unreachable: 0/9 calibration samples + # ever passed. + # + # An "all-tiles training" alternative (Scope B, briefly implemented in + # commit e8e6731 on 2026-05-14 then reverted) was tested and rejected: + # it broke the within-tissue focus interpretation (GMM bimodality became + # background-vs-tissue rather than blur-vs-focus, and the + # component_separation diagnostic no longer computed within tissue). + # See plans/2026-05-14_SPIKE_gmm_scope.md for the full analysis. + # + # CAVEAT: thresholds tuned on a small calibration set — recalibrate when + # panel widens. Brain samples tend to flag elevated values due to lower + # DAPI contrast in neural nuclei (biology-confounded, not an imaging + # defect); this is disclosed in the §3.2 metric explanation, not at the + # cutoff level. + # 2026-06-25 (hysteresis mask change): UNCHANGED. The 14-tissue panel showed + # tissue-filtered blur % is tissue-confounded (normal brain/pancreas blur as much + # as named-bad samples), so it stays advisory; no sample reaches focus_fail under + # the hysteresis mask (max ~39%). See plans/2026-06-25_PLAN_tissue-mask-hysteresis.md. + focus_warn: 0.40 + focus_fail: 0.60 + # --- Report-only thresholds (read by Quarto report, not by image_qc.py) --- + # Agreement between focus_score and lap_var across tissue tiles. + # Low correlation can indicate one metric is unstable for that sample. + lap_focus_corr_warn: 0.50 + lap_focus_corr_fail: 0.25 + boundary: + # Per-XOA-version floors (see DAPI block). Report v3=100, v4=20 (boundary + # signal collapses on dim v4). intensity_critical is the unknown-version + # fallback. + intensity_critical_v3: 100 + intensity_critical_v4: 20 + intensity_critical: 100 # fallback when XOA version is unknown + # Boundary channel is biologically patchy across tissues (neural/stromal/immune). + # WARN-only (no FAIL tier) since 2026-06-22 — see DAPI block. + intensity_warn: 0.40 + # Tissue-type note (not parameterized yet): + # - neural/stromal tissues may need further relaxation to 0.25 / 0.50 + # - epithelial tissues often tolerate stricter defaults + # --- PLANNED: Spatial CV (not yet implemented) --- + # See DAPI section for rationale. Boundary channel is biologically + # patchy (neural/stromal/immune), so thresholds are more permissive. + # spatial_cv_warn: 0.45 + # spatial_cv_fail: 0.65 + # --- PLANNED: DAPI correlation (not yet implemented) --- + # Correlation between boundary and DAPI per-tile intensity. Catches + # channel-specific staining failures independent of DAPI quality: + # if boundary antibody fails but DAPI is fine, per-channel thresholds + # may still pass, but near-zero correlation flags it immediately. + # Tissue-dependent; permissive thresholds to avoid false positives. + # dapi_corr_warn: 0.20 + # dapi_corr_fail: 0.05 + intRNA: + # Per-XOA-version floors (see DAPI block). Report v3=20, v4=50. + # intensity_critical is the unknown-version fallback. + intensity_critical_v3: 20 + intensity_critical_v4: 50 + intensity_critical: 300 # fallback when XOA version is unknown + # Most biology-dependent channel (necrosis/adipose/white matter strongly + # affect signal). WARN-only (no FAIL tier) since 2026-06-22 — see DAPI block. + intensity_warn: 0.40 + # --- PLANNED: Spatial CV (not yet implemented) --- + # See DAPI section for rationale. intRNA has highest biological variance + # across tissue types; permissive thresholds reflect this. + # spatial_cv_warn: 0.55 + # spatial_cv_fail: 0.75 + # --- PLANNED: DAPI correlation (not yet implemented) --- + # See boundary section for rationale. intRNA-DAPI coupling is weak in + # many valid mixed tissues, so very permissive thresholds. + # dapi_corr_warn: 0.10 + # dapi_corr_fail: 0.02 + slide_level: + # --- Report-only thresholds (read by Quarto report, not by image_qc.py) --- + # NOTE: semantics matter. + # Recommended definition: + # usable = tile is not blurred AND not below intensity threshold + # usable_tissue_* should be computed as "% of tissue-overlapping tiles that are usable" + # (not "% of all slide tiles"), so small inputs (TMA/biopsy) are not unfairly penalized. + # 2026-06-26: usable_tissue is now WARN-only (the *_fail key is removed). Under the + # background-from-nonzero mask + gate 0.2 + is_low=50, usable is partly intensity- + # confounded for dim tissue (e.g. skin), and the FAIL cutoff was overfit to ~3 named-bad + # samples. The report uses _advisory_pass_warn for this row; reintroduce a FAIL tier only + # with a larger calibration panel. warn lowered 0.70 -> 0.60 (normals span 50-79%; 0.60 + # flags the low end without over-flagging). See 2026-06-26_PLAN_tissue-mask-recalibration.md. + usable_tissue_warn: 0.60 + # Contiguous bad-tile area >15% usually indicates a localized physical artefact (fold/chip/shadow). + cluster_zone_fail: 0.15 + morphology: + # 2026-06-24: Mean tissue coverage is WARN-only (no FAIL). Low coverage is + # tissue geometry / biology (small or sparse sections), not a data-quality + # failure — the data on the tissue present is judged by the focus/intensity/ + # SNR metrics, and a genuinely empty slide is caught by the tissue-mask gate. + # The *_fail key is removed; the report uses _advisory_pass_warn for this row, + # and the §1 Morphology aggregation reads None for fail (WARN-only). + # 2026-06-25 (hysteresis mask change): mean coverage is tissue-confounded (a normal + # liver has the same low coverage as bad skin), so it stays advisory/WARN-only. + # 2026-06-26: lowered 0.30 -> 0.10. The background-from-nonzero mask is tighter, so mean + # coverage drops (normals now span ~0.12-0.44); 0.30 would WARN almost every normal + # sample. 0.10 flags only genuinely sparse fields (e.g. skin_bad at 0.08). + # 2026-06-26 (multi-stain): mean tissue coverage is now computed from the MULTI-STAIN + # extent mask (all available stains), so coverage rises further (e.g. muscle 0.16 -> 0.33). + # 0.10 is kept conservatively for v1 (higher coverage => fewer false WARNs); recalibrate + # on the multi-stain Seqera run across the panel, as the DAPI value was. WARN-only/advisory. + tissue_coverage_warn: 0.10 + # 2026-06-23: Edge-zone and Hole-area are WARN-only (no FAIL). Both are + # biology-driven — sparse/branching tissues (lung, intestine, lymph node) + # legitimately have high edge fractions, and the hole mask does not yet + # separate anatomical lumens from artefactual tears — so they flag for + # review rather than failing the sample. The *_fail keys are removed; the + # report uses _advisory_pass_warn for these two rows. + # 2026-06-23 cohort calibration (160+ sample cohort, n=79 carry these metrics): WARN set just + # above the per-tissue p90 so it flags the unusual high-burden tail, not normal tissue. + edge_zone_warn: 0.40 # was 0.30 (flagged 16% of cohort, ~p87); 0.40 ~ p93, flags ~9%, above per-tissue p90 (0.33-0.38) + # hole_area is strongly tissue-driven (lung median 0.32, p90 0.46; liver/muscle ~0.18). The old + # 0.25 sat at the cohort median (0.23) and flagged 44%, i.e. it fired on normal tissue, esp. lung. + hole_area_warn: 0.45 # was 0.25; 0.45 ~ p93, flags ~9% and clears the lung biology tail. Consider relaxing for small inputs (TMA/biopsy/needle cores). + spatial_context: + # --- Report-only thresholds (read by Quarto report, not by image_qc.py) --- + # PLACEHOLDER DEFAULTS pending calibration on >50 samples (Phase 9 of cell-centered + # restructure plan v3, 2026-05-01). Drive the % + PASS/WARN/FAIL Status column in + # Section 9.C of the image QC report. The rendered Status pills carry an + # "(uncalibrated)" badge until the calibration sweep produces real thresholds. + # Triggers: fraction of cells flagged by each per-cell spatial filter. + near_edge_warn: 0.20 + near_edge_fail: 0.40 + # near_holes relaxed from 0.10/0.25 → 0.30/0.50 (Phase 9 follow-up, 2026-05-01) + # after first render exposed eager triggering on lumen-rich tissues (lung, + # intestine, glandular epithelia) where 15-25% near-holes cells is biology, + # not artefact. Other two axes still use original placeholders. + near_holes_warn: 0.30 + near_holes_fail: 0.50 + in_artifact_warn: 0.05 + in_artifact_fail: 0.15 + focus: + # Cell-level CCFS nuclear texture threshold: cells with CCFS_DAPI below + # this are classified as low nuclear texture quality. CCFS measures per-cell + # nuclear contrast (local_var/local_mean), NOT optical blur — it is distinct + # from the tile-level GMM blur metric. + # Calibrated on 5 samples (2026-04-02): cells below 0.02 lose >50% + # transcripts vs top-50% CCFS cells in lung/liver. + # Brain/neural tissues: CCFS is confounded by nucleus size (large neurons + # score lower); interpret with caution. + ccfs_low_texture_threshold: 0.02 + # --- Report-only thresholds (read by Quarto report, not by image_qc.py) --- + # Cell-level only: applies to per-cell CCFS nuclear texture calls + # (not tile-level GMM blur fractions). + # 10%/25% is a practical default for many tissues; review for very + # small-nucleus contexts (lymphoid / some brain regions) where cell-level + # CCFS can overcall low texture. + low_texture_cell_warn: 0.10 + low_texture_cell_fail: 0.25 + # Median raw DAPI focus score (std²/mean) across tissue tiles. + # Highly tissue-dependent: brain/small-nucleus tissues score lower than + # lung/liver. Calibrated on 8 samples (2026-04-01): + # bad (panc/brain): 0.95 - 126, moderate: 137 - 797, good (lung): 1079. + # WARN-only; no FAIL — insufficient cross-tissue data for a hard cutoff. + focus_median_warn: 100 + # --- PLANNED: Focus Moran's I (not yet implemented) --- + # Spatial autocorrelation of tile-level focus scores (not negative probes). + # Complements cluster_zone_bad_fraction by detecting distributed spatial + # blur patterns (many small patches) vs one large contiguous patch. + # cluster_zone catches "one bad region"; Moran's I catches "many scattered + # bad spots" that collectively degrade quality. Can reuse the Moran's I + # implementation from snr_metrics.py applied to focus_score or blur flags. + # NOTE: Do NOT confuse with snr.neg_spatial.moran_warn/fail (0.5/1.0) + # which is for negative-probe spatial clustering — a different metric. + # morans_i_warn: 0.30 + # morans_i_fail: 0.60 + # Absolute Laplacian variance floor — REMOVED 2026-06-22 (qc_drift_analysis). + # lap_var scales with brightness² (rho ≈ 0.91 with DAPI) and collapses to + # near-zero on dim XOA-4.0 images, over-flagging good dim tissue. It was also + # redundant with the GMM-2D classifier, which already uses lap_var as a + # feature. lap_var_median_raw is still emitted as an informational stat in + # roi_qc_metrics.json. The GMM-unreliable focus fallback no longer depends on + # this floor (see notebooks/xenium_image_qc_report.qmd §3.2 "honest fallback"). + # --- Report-only threshold (read by Quarto report, not by image_qc.py) --- + # 2D GMM component separation (Cohen's d between blur and focus populations + # in log1p(lap_var) space). Low values indicate the GMM cannot reliably + # distinguish blurred from focused tiles. + # Calibrated range across 6 samples: 2.98-4.64 (all well-separated). + gmm_2d_laplacian_component_separation_warn: 1.0 + gmm_2d_laplacian_component_separation_fail: 0.5 + snr: + # Per-channel image SNR dB thresholds. + # RESERVED: not yet consumed by code (quartile/Otsu currently use + # image_snr_db fallback below). Will be wired in when per-channel + # SNR assessment is implemented. + dapi: + warn_db: 15 + fail_db: 8 + boundary: + warn_db: 10 + fail_db: 5 + intrna: + warn_db: 8 + fail_db: 3 + # Fallback image SNR dB thresholds when channel is not specified. + # 2026-05-15: three-tier with headroom for future "extremely bad" samples. + # Calibration set (n=11) all PASS at warn=15 (observed quartile range + # 33-71 dB, Otsu range 18-27 dB); WARN and FAIL tiers are reserved for + # known-failure samples (mounting drift, washing artefacts) once captured. + image_snr_db: + warn: 15.0 + fail: 10.0 + # Per-tile transcript SNR (target-to-negative ratio and negative fraction). + # 2026-06-23: ratio cutoffs set to 60 / 30 against the qc_threshold_refinement + # report. The deduped p05 / p01 are 77 / 32; 60 / 30 is a deliberately + # less-aggressive operating point chosen pending more data. (An earlier + # 50 / 25 came from a flawed dedup key, since corrected in the report; the + # original 30 / 10 never fired.) WARN below 60, FAIL below 30. + # 2026-06-25 (hysteresis mask change): UNCHANGED. The roi_tx SNR verdict is computed + # over all transcript-bearing tiles (snr_metrics.py compute_roi_snr), NOT the DAPI + # tissue mask, so it does not drift when the mask changes. Validated on the 14-tissue + # panel: 60/30 still separates the SNR-bad samples from all normal tissues. + roi_tx: + ratio_warn: 60.0 + ratio_fail: 30.0 + neg_pct_warn: 0.15 + neg_pct_fail: 0.30 + # Negative-probe spatial autocorrelation (quadrant spread + Moran's I). + # Loosened from 0.3/0.6 — heterogeneous tissues (pancreas, liver) can + # produce biologically-driven spatial variation in negative probe rates. + neg_spatial: + quadrant_spread_warn: 0.5 + quadrant_spread_fail: 1.0 + moran_warn: 0.5 + moran_fail: 1.0 + # Slide-level Plummer SNR (log-ratio of real vs negative gene means). + slide_plummer: + pass_min: 0.12 + fail_below: 0.10 + # Slide-level SpatialQM — verdict based on pct_genes_above_neg (% of genes + # whose mean expression exceeds the mean negative-control level). + slide_spatialqm: + metric: pct_genes_above_neg + warn: 45 + fail: 35 + # Cell-level advisory thresholds (§4.1 cell quality scores summary). + # PASS / WARN only — these are advisory flags-to-investigate, not strict + # gates. WARN triggers when the % of flagged cells reaches the cutoff. + # Calibrated 2026-05-15 against n=11 calibration_data samples: + # - 3 clean PASS (R4_bleo_d14_lung, R1_303Liv, 0060254_pancreas) + # - 5 samples 1-flag WARN (OSMK named-bad + 4 tissue-biology cases: + # TSS_41190_brain, TSS_51107_brain, 0060257_pancreas, r32_37Liv) + # - 1 sample 2-flag WARN (XEN-85A — named-bad) + # - 2 samples 4-flag WARN (R1_skin_good + R1_skin_bad) + # Tissue caveat: high % low IntRNA in skin samples and high % artefacts + # in dense tissue (pancreas / brain) reflect tissue biology — investigate + # before excluding cells. See §4.1 metric-explanation callout in QMD. + cell_level: + pct_low_nuclear_texture_warn: 5.0 + # Single uniform cutoff applies to DAPI / Boundary / IntRNA. Channel- + # specific differences are already encoded in the per-channel + # intensity_critical floors (DAPI=500, Boundary=100, IntRNA=300) above. + pct_cells_below_intensity_warn: 10.0 + pct_cells_in_optically_dense_regions_warn: 30.0 + # §4.2 row: % cells inheriting the tile-level blurry (GMM) label. + # Calibration (n=11) cleanly bimodal: healthy 11.0–19.2%, named-bad + # 22.6–33.1%. WARN ≥ 20% catches all 3 named-bad + skin_good (tissue + # biology — Xenium IntRNA channel struggles on skin regardless of run). + pct_blurred_gmm_2d_roi_warn: 20.0 + # Note: % cells in low-coverage tiles (§4.2 row 2) is intentionally + # informational — no Advisory threshold. Biology-driven: skin and + # thin epithelial sections are at 100% regardless of run quality, and + # named-bad correlation is weak (XEN-85A 0.3%, OSMK 0.8%). Surfaces as + # "—" pill so readers treat the % as context (patchy tissue mount) + # rather than a verdict. diff --git a/bin/snr_metrics.py b/bin/snr_metrics.py new file mode 100755 index 00000000..a33fdf79 --- /dev/null +++ b/bin/snr_metrics.py @@ -0,0 +1,1892 @@ +#!/usr/bin/env python3 +""" +SNR metrics for Xenium image QC (ROI image, transcripts, slide matrix, neg spatial). + +Used by ``bin/image_qc.py``. See ``plans/image_qc_report/SNR_plan.md``. + +Dependencies: numpy, pandas; optional h5py, pyarrow, scipy, libpysal/esda, scikit-image. +""" + +from __future__ import annotations + +import json +import logging +import math +import os +import re +import threading +from collections import deque +from concurrent.futures import ThreadPoolExecutor +from pathlib import Path +from typing import Any, Dict, List, Optional, Tuple + +import numpy as np +import pandas as pd +import pyarrow.parquet as pq + +logger = logging.getLogger(__name__) + +# JSON ``summary["components"]`` keys — ``SNR_`` prefix keeps this module separate from +# generic ``roi_qc_metrics`` / image_qc fields when integrated into ``bin/image_qc.py``. +SNR_CKEY_IMAGE_ROI_QUARTILE_DB = "SNR_image_roi_quartile_db" +SNR_CKEY_IMAGE_OTSU = "SNR_image_otsu" +SNR_CKEY_ROI_TX = "SNR_roi_tx" +SNR_CKEY_SLIDE_PLUMMER = "SNR_slide_plummer" +SNR_CKEY_SLIDE_SPATIALQM = "SNR_slide_spatialqm" +SNR_CKEY_ROI_NEG_SPATIAL = "SNR_roi_neg_spatial" + +# Per-ROI transcript SNR table (same rows as input grid + count/ratio columns) for histograms / maps. +SNR_ROI_TX_TABLE_BASENAME = "SNR_roi_tx" + +# --------------------------------------------------------------------------- +# SNR_plan.md: snr_verdict_to_quality_status +# --------------------------------------------------------------------------- + + +def snr_verdict_to_quality_status(snr_summary: dict) -> str: + """ + Convert SNR module verdict to the 'pass'/'warn'/'fail' scale used by + assess_raw_intensity_quality() so the HTML report scorecard can treat + SNR like any other channel quality metric. + + PASS → 'pass' + WARN → 'warn' + FAIL → 'fail' + NOT_COMPUTED / ERROR → 'not_available' + """ + mapping = { + "PASS": "pass", + "WARN": "warn", + "FAIL": "fail", + } + overall = snr_summary.get("verdict", {}).get("overall_snr_verdict", "NOT_COMPUTED") + return mapping.get(overall, "not_available") + + +# --------------------------------------------------------------------------- +# Neg-control classification (Xenium-style) +# --------------------------------------------------------------------------- + +_NEG_PATTERNS = ( + re.compile(r"^NegControlProbe_", re.I), + re.compile(r"^NegControlCodeword_", re.I), + re.compile(r"^Blank[-_]", re.I), + re.compile(r"^BLANK[-_]", re.I), + re.compile(r"^antisense_", re.I), +) +_NEG_REGEX = r"^(?:NegControlProbe_|NegControlCodeword_|Blank[-_]|antisense_)" + +# Features that are neither real signal NOR a background estimate, and so are +# dropped from both sides of the SNR ratio rather than being reclassified. +# +# `UnassignedCodeword_*` is a decoding failure: the readout did not match any valid +# barcode. It was previously counted as real gene signal (`is_real = ~is_neg` had no +# third category), which inflated the numerator of the transcript SNR ratio and +# biased the verdict toward PASS. Counting it as a negative control would be equally +# wrong -- a negative-control probe measures background because it is designed not to +# bind, whereas a failed read measures nothing at all. So it is excluded from real, +# from neg, and from the total. +# +# NOT included, deliberately: `DeprecatedCodeword_*`. Those are retired-but-valid +# barcodes, so they may be genuine detections of a gene no longer in the panel +# definition. That is a different question and needs its own decision; adding the +# prefix here is the whole change if that decision comes out the same way. +_EXCLUDED_PATTERNS = (re.compile(r"^UnassignedCodeword_", re.I),) +_EXCLUDED_REGEX = r"^(?:UnassignedCodeword_)" + + +def is_neg_probe_feature(name: str) -> bool: + if not isinstance(name, str) or not name: + return False + return any(p.search(name) for p in _NEG_PATTERNS) + + +def is_excluded_feature(name: str) -> bool: + """True for features that count as neither signal nor background.""" + if not isinstance(name, str) or not name: + return False + return any(p.search(name) for p in _EXCLUDED_PATTERNS) + + +# --------------------------------------------------------------------------- +# Image SNR — lightweight path (ROI DataFrame columns) +# SNR_plan: top ~75th pct tissue ROIs = signal; bottom quartile = background; +# SNR_dB = 20 * log10(mean_fg / std_bg) +# --------------------------------------------------------------------------- + + +def compute_image_snr_from_roi_df( + df_grid_roi: pd.DataFrame, + intensity_threshold: float = 0.0, + intensity_col: Optional[str] = None, + snr_thresholds: Optional[Dict[str, Any]] = None, +) -> Dict[str, Any]: + """ + Slide-level image SNR (dB) from pre-aggregated ROI intensities; same value + conceptually applies to all tissue ROIs (SNR_plan lightweight path). + """ + if intensity_col is None: + intensity_col = ( + "dapi_intensity" + if "dapi_intensity" in df_grid_roi.columns + else "raw_intensity" + ) + if intensity_col not in df_grid_roi.columns: + return {"status": "skipped", "reason": f"missing column {intensity_col!r}"} + + df = df_grid_roi.copy() + if "overlaps_tissue" in df.columns: + tissue = df[df["overlaps_tissue"]].copy() + elif "tissue_coverage" in df.columns: + tissue = df[df["tissue_coverage"] > 0].copy() + else: + tissue = df + + tissue = tissue[tissue[intensity_col] >= intensity_threshold] + if len(tissue) < 8: + return { + "status": "skipped", + "reason": "too_few_tissue_rois", + "n": int(len(tissue)), + } + + vals = tissue[intensity_col].astype(np.float64).values + q25, q75 = np.percentile(vals, [25, 75]) + bg = vals[vals <= q25] + fg = vals[vals >= q75] + if len(bg) < 2 or len(fg) < 2: + return {"status": "skipped", "reason": "empty_quartile_split"} + + mean_fg = float(np.mean(fg)) + std_bg = float(np.std(bg, ddof=1)) if len(bg) > 1 else float(np.std(bg)) + eps = 1e-12 + if std_bg < eps: + return {"status": "skipped", "reason": "zero_background_std"} + + ratio = mean_fg / std_bg + snr_db = 20.0 * math.log10(max(ratio, eps)) + _t = snr_thresholds or {} + _img = _t.get("image_snr_db") or {} + warn_db = float(_img.get("warn", 15.0)) + fail_db = float(_img.get("fail", 10.0)) + verdict = "PASS" if snr_db >= warn_db else ("WARN" if snr_db >= fail_db else "FAIL") + + return { + "status": "ok", + "method": "roi_df_quartiles", + "intensity_col": intensity_col, + "snr_db": snr_db, + "mean_foreground": mean_fg, + "std_background": std_bg, + "n_tissue_rois": int(len(tissue)), + "verdict": verdict, + } + + +# --------------------------------------------------------------------------- +# Otsu (numpy-only fallback; optional skimage) +# --------------------------------------------------------------------------- + + +def _otsu_threshold_uint16(arr: np.ndarray) -> float: + """Histogram Otsu for non-negative array (2D).""" + a = np.asarray(arr, dtype=np.float64).ravel() + a = a[np.isfinite(a)] + if a.size < 4: + return float(np.median(a)) if a.size else 0.0 + # uint16-style bins for Xenium-ish data; cap bins for speed + a_min, a_max = float(np.min(a)), float(np.max(a)) + if a_max <= a_min: + return a_min + nb = min(256, int(a_max - a_min) + 1) + hist, bin_edges = np.histogram(a, bins=nb, range=(a_min, a_max)) + bin_centers = (bin_edges[:-1] + bin_edges[1:]) / 2.0 + w = hist.astype(np.float64) + total = w.sum() + if total <= 0: + return float(np.median(a)) + w = w / total + mu = (w * bin_centers).sum() + omega = np.cumsum(w) + mu_t = np.cumsum(w * bin_centers) + sigma_b2 = (mu_t - mu * omega) ** 2 / (omega * (1 - omega) + 1e-12) + idx = int(np.nanargmax(sigma_b2)) + return float(bin_centers[idx]) + + +def roi_snr_db(tile: np.ndarray) -> float: + """SNR in dB for one ROI window: Otsu foreground mean over background stdev. + + Returns NaN when the window cannot yield a value — too few pixels, an empty + foreground or background after thresholding, or a degenerate background with + zero spread. NaN rather than an exception because a slide legitimately + contains such windows (empty edge tissue) and they are reported as N/A. + + Factored out so the whole-map path and the streaming tile consumer + (``image_qc.RoiOtsuSnrAccumulator``) share one implementation: equivalence is + then structural rather than something a test has to keep rediscovering. + + The window is reduced in **float64**. Production maps are float32, but the + batched (``roi_snr_db_batch``) and numba/GPU paths promote to float64, and a + float32 masked reduction is order-dependent so it cannot bit-match a batched + one. Computing every path in float64 makes CPU and GPU results identical to + float64 rounding instead of differing by ~2e-6 dB with the instance type. + """ + if tile.size < 4: + return float("nan") + tile = np.asarray(tile, dtype=np.float64) + + try: + from skimage.filters import threshold_otsu as _skimage_threshold_otsu # type: ignore + except Exception: + _skimage_threshold_otsu = None + + if _skimage_threshold_otsu is not None: + try: + threshold = float(_skimage_threshold_otsu(tile)) + except Exception: + threshold = _otsu_threshold_uint16(tile) + else: + threshold = _otsu_threshold_uint16(tile) + + foreground = tile[tile >= threshold] + background = tile[tile < threshold] + if foreground.size < 2 or background.size < 2: + return float("nan") + + mean_fg = float(np.mean(foreground)) + std_bg = float(np.std(background, ddof=1)) + eps = 1e-12 + if std_bg < eps: + return float("nan") + return 20.0 * math.log10(max(mean_fg / std_bg, eps)) + + +def roi_snr_db_batch(tiles: Any, xp: Any = np) -> Any: + """Vectorized ``roi_snr_db`` over a stack of equal-size ROI windows. + + ``tiles`` is ``(K, h, w)`` or ``(K, N)``; returns a ``(K,)`` float64 array of + dB values, one per window, computed as a single batched reduction with **no + Python per-ROI loop**. Pass ``xp=cupy`` to run entirely on the GPU — every op + below (min/max, bincount, cumsum, argmax, masked reductions) exists in both + numpy and cupy, so the same code runs on either device. This is the form the + streaming tile pass uses when the tile's ``mean_map`` is already resident in + VRAM (``image_qc._compute_channel_maps_on_gpu``): the Otsu SNR is folded on + device and only the small ``(K,)`` dB vector returns to host. + + Reproduces ``skimage.filters.threshold_otsu`` (256 bins over each window's own + min..max, the float-corrected uniform-bin assignment numpy uses) and the + ``foreground-mean / background-std(ddof=1)`` dB. Equivalence vs the per-ROI + ``roi_snr_db`` is exact to float64 rounding (max ~1e-14 dB, verified in + ``tests/test_roi_snr_db_batch.py``). + + Assumes finite, full-size windows. Constant windows (zero span) return NaN, + matching the scalar path (an all-foreground split leaves the background empty). + Callers route edge-clipped or non-finite windows to the scalar ``roi_snr_db``. + """ + nbins = 256 + eps = 1e-12 + flat = xp.asarray(tiles).reshape(xp.asarray(tiles).shape[0], -1).astype(xp.float64) + K, N = flat.shape[0], flat.shape[1] + out = xp.full(K, xp.nan, dtype=xp.float64) + if N < 4: + return out + + mn = flat.min(axis=1) + mx = flat.max(axis=1) + ok = mx > mn # constant windows -> NaN, as in roi_snr_db + if not bool(ok.any()): + return out + sub = flat[ok] + mn_s = mn[ok] + step = (mx[ok] - mn_s) / nbins + + # numpy's uniform-bin assignment for np.histogram(row, nbins, (min, max)), + # including its float corrections against the arithmetic edge mn + idx*step. + idx = ((sub - mn_s[:, None]) / step[:, None]).astype(xp.intp) + idx = xp.where(idx == nbins, nbins - 1, idx) + idx = xp.clip(idx, 0, nbins - 1) + left = mn_s[:, None] + idx * step[:, None] + idx = xp.where(sub < left, idx - 1, idx) + idx = xp.clip(idx, 0, nbins - 1) + right = mn_s[:, None] + (idx + 1) * step[:, None] + idx = xp.where((sub >= right) & (idx != nbins - 1), idx + 1, idx) + idx = xp.clip(idx, 0, nbins - 1) + + Ks = sub.shape[0] + flat_idx = (idx + xp.arange(Ks)[:, None] * nbins).reshape(-1) + counts = ( + xp.bincount(flat_idx, minlength=Ks * nbins) + .reshape(Ks, nbins) + .astype(xp.float64) + ) + centers = mn_s[:, None] + (xp.arange(nbins) + 0.5) * step[:, None] + + # skimage.filters.threshold_otsu's histogram math, per row. + with np.errstate(invalid="ignore", divide="ignore"): + w1 = xp.cumsum(counts, axis=1) + w2 = xp.cumsum(counts[:, ::-1], axis=1)[:, ::-1] + cb = counts * centers + mean1 = xp.cumsum(cb, axis=1) / w1 + mean2 = (xp.cumsum(cb[:, ::-1], axis=1) / w2[:, ::-1])[:, ::-1] + var12 = w1[:, :-1] * w2[:, 1:] * (mean1[:, :-1] - mean2[:, 1:]) ** 2 + thr = centers[xp.arange(Ks), xp.argmax(var12, axis=1)] + + fg = sub >= thr[:, None] + nfg = fg.sum(axis=1) + nbg = N - nfg + with np.errstate(invalid="ignore", divide="ignore"): + mean_fg = xp.where(fg, sub, 0.0).sum(axis=1) / nfg + mean_bg = xp.where(~fg, sub, 0.0).sum(axis=1) / nbg + ss_bg = xp.where(~fg, (sub - mean_bg[:, None]) ** 2, 0.0).sum(axis=1) + std_bg = xp.sqrt(ss_bg / (nbg - 1)) + good = (nfg >= 2) & (nbg >= 2) & (std_bg >= eps) + db = xp.full(Ks, xp.nan, dtype=xp.float64) + ratio = xp.maximum(mean_fg / std_bg, eps) + db = xp.where(good, 20.0 * xp.log10(ratio), db) + out[ok] = db + return out + + +try: + import numba as _numba + + _HAS_NUMBA = True +except Exception: # numba is an optional accelerator; callers fall back otherwise + _HAS_NUMBA = False + +if _HAS_NUMBA: + # cache=False deliberately: each Nextflow task is a fresh process with an + # ephemeral work dir, so an on-disk cache never survives to be reused, and + # numba's cache key for a path-loaded module is "", which poisons a + # recompile with `ModuleNotFoundError: No module named ''`. Compiling + # once per task (~seconds) is negligible against the fold. + @_numba.njit(cache=False, fastmath=False) + def _roi_snr_db_one_numba(tile_flat, nbins): # noqa: D401 + """One ROI: skimage-Otsu threshold then fg-mean/bg-std dB, all float64. + + Reproduces ``roi_snr_db`` for a finite, non-constant window; returns NaN + for the same degenerate cases (too small, constant, empty fg/bg, zero bg + spread). Compiled + released-GIL so ``prange`` scales across cores. + """ + n = tile_flat.size + if n < 4: + return np.nan + mn = tile_flat[0] + mx = tile_flat[0] + for v in tile_flat: + if v < mn: + mn = v + if v > mx: + mx = v + if mx <= mn: + return np.nan + step = (mx - mn) / nbins + counts = np.zeros(nbins, dtype=np.float64) + for v in tile_flat: + b = int((v - mn) / step) + if b == nbins: + b = nbins - 1 + if v < mn + b * step: + b -= 1 + elif b != nbins - 1 and v >= mn + (b + 1) * step: + b += 1 + if b < 0: + b = 0 + elif b >= nbins: + b = nbins - 1 + counts[b] += 1.0 + w1 = np.cumsum(counts) + centers = np.empty(nbins, dtype=np.float64) + for j in range(nbins): + centers[j] = mn + (j + 0.5) * step + m1cum = np.cumsum(counts * centers) + total = w1[nbins - 1] + cbtot = m1cum[nbins - 1] + best_var = -1.0 + best_idx = 0 + for t in range(nbins - 1): + wa = w1[t] + wb = total - wa + if wa == 0.0 or wb == 0.0: + continue + ma = m1cum[t] / wa + mb = (cbtot - m1cum[t]) / wb + var = wa * wb * (ma - mb) ** 2 + if var > best_var: + best_var = var + best_idx = t + thr = centers[best_idx] + sfg = 0.0 + nfg = 0 + sbg = 0.0 + nbg = 0 + for v in tile_flat: + if v >= thr: + sfg += v + nfg += 1 + else: + sbg += v + nbg += 1 + if nfg < 2 or nbg < 2: + return np.nan + mean_fg = sfg / nfg + mean_bg = sbg / nbg + ss = 0.0 + for v in tile_flat: + if v < thr: + d = v - mean_bg + ss += d * d + std_bg = np.sqrt(ss / (nbg - 1)) + if std_bg < 1e-12: + return np.nan + ratio = mean_fg / std_bg + if ratio < 1e-12: + ratio = 1e-12 + return 20.0 * np.log10(ratio) + + @_numba.njit(cache=False, parallel=True, fastmath=False) # see cache note above + def _roi_snr_db_numba_kernel(stack2d, nbins): + K = stack2d.shape[0] + out = np.empty(K, dtype=np.float64) + for k in _numba.prange(K): + out[k] = _roi_snr_db_one_numba(stack2d[k], nbins) + return out + + +def roi_snr_db_numba(tiles: Any) -> np.ndarray: + """CPU-parallel batched ``roi_snr_db`` (numba ``prange``). + + The efficient path for **CPU** instances: ~50x the Python per-ROI loop on 8 + cores, versus the pure-numpy ``roi_snr_db_batch(xp=np)`` which is memory-bound + and no faster than the loop. On **GPU** instances use ``roi_snr_db_batch(xp=cp)`` + instead. Equivalent to ``roi_snr_db`` to float64 rounding. + + Thread count follows ``numba.set_num_threads`` / ``NUMBA_NUM_THREADS``; the + caller must cap it so it does not oversubscribe against the tile worker pool. + """ + if not _HAS_NUMBA: + raise RuntimeError("roi_snr_db_numba requires numba, which is not installed") + arr = np.asarray(tiles) + flat = np.ascontiguousarray(arr.reshape(arr.shape[0], -1), dtype=np.float64) + if flat.shape[1] < 4: + return np.full(flat.shape[0], np.nan, dtype=np.float64) + return _roi_snr_db_numba_kernel(flat, 256) + + +def compute_image_snr_from_pixel_maps( + focus_maps: Optional[Dict[str, Any]], + df_grid_roi: pd.DataFrame, + map_key: str = "dapi_mean_map", + max_rois: Optional[int] = None, + snr_thresholds: Optional[Dict[str, Any]] = None, + precomputed_db: Optional[Any] = None, +) -> Dict[str, Any]: + """ + Per-ROI SNR_dB from pixel tiles: Otsu foreground vs background std (SNR_plan accurate path). + Uses ``dapi_mean_map`` (or ``map_key``) slices [y1:y2, x1:x2] per ROI row. + + ``precomputed_db`` supplies one dB value per ROI when the caller already + reduced them during a tiled pass, in which case no pixel map is needed. Both + paths get their dB from :func:`roi_snr_db`, so the aggregate statistics below + are computed identically either way. + """ + if precomputed_db is None: + arr = focus_maps.get(map_key) if focus_maps else None + if arr is None: + return {"status": "skipped", "reason": f"missing focus_maps[{map_key!r}]"} + + # Keep the map lazy (do NOT force a whole-array float64 copy — for a + # full-res / memmap-backed map that materialises tens of GB). Each tile is + # sliced then converted per-tile below. + img = np.asarray(arr) + h, w = img.shape[:2] + + required = {"x1", "x2", "y1", "y2"} + if not required.issubset(df_grid_roi.columns): + return {"status": "skipped", "reason": f"df_grid_roi needs columns {required}"} + + df = df_grid_roi.copy() + if max_rois is not None: + df = df.iloc[: int(max_rois)] + + x1a = df["x1"].to_numpy(dtype=np.int64) + x2a = df["x2"].to_numpy(dtype=np.int64) + y1a = df["y1"].to_numpy(dtype=np.int64) + y2a = df["y2"].to_numpy(dtype=np.int64) + + # Per-tile dB array aligned with df rows (NaN for skipped tiles). + # Used both for aggregate stats and to expose per-tile values to df_grid_roi + # downstream — the §3.4 cross-section-concordance scatter consumes this. + per_tile_db = np.full(len(df), np.nan, dtype=np.float64) + if precomputed_db is not None: + supplied = np.asarray(precomputed_db, dtype=np.float64) + if supplied.size < len(df): + return { + "status": "skipped", + "reason": f"precomputed_db has {supplied.size} values for {len(df)} tiles", + } + per_tile_db = supplied[: len(df)].copy() + dbs = [float(v) for v in per_tile_db[~np.isnan(per_tile_db)]] + else: + dbs = [] + for k in range(len(df)): + x1, x2, y1, y2 = int(x1a[k]), int(x2a[k]), int(y1a[k]), int(y2a[k]) + x1, y1 = max(0, x1), max(0, y1) + x2, y2 = min(w, x2), min(h, y2) + if x2 <= x1 or y2 <= y1: + continue + db_val = roi_snr_db(img[y1:y2, x1:x2]) + if math.isnan(db_val): + continue + per_tile_db[k] = db_val + dbs.append(db_val) + + # Write per-tile values back to the caller's df_grid_roi (in-place, + # parallel to how compute_roi_transcript_snr writes roi_tx_snr_ratio). + # df.index is a subset of df_grid_roi.index when max_rois is set; rows + # outside that subset retain NaN. + df_grid_roi.loc[df.index, "snr_image_otsu_db"] = per_tile_db + + if not dbs: + return {"status": "skipped", "reason": "no_valid_roi_tiles"} + + # Aggregate over TISSUE tiles only, matching the tissue filter in the sibling + # compute_image_snr_from_roi_df -- both feed the same image_snr_db warn/fail + # pair, so grading one on tissue and the other on the whole slide made the two + # incomparable. An empty tile splits its own noise and can score HIGHER than + # real tissue, so on a slide with a small tissue footprint (a TMA core, a + # biopsy, a diagonal section) the median reported the coverslip and passed. + # + # per_tile_db above stays unfiltered on purpose: it is written back to + # df_grid_roi["snr_image_otsu_db"] and consumed by the §3.4 concordance + # scatter, so scoping it here would silently change that figure. Only the + # median and the verdict are tissue-scoped. + if "overlaps_tissue" in df.columns: + _is_tissue = df["overlaps_tissue"].to_numpy(dtype=bool) + elif "tissue_coverage" in df.columns: + _is_tissue = (df["tissue_coverage"] > 0).to_numpy(dtype=bool) + else: + _is_tissue = np.ones(len(df), dtype=bool) + _agg = per_tile_db[_is_tissue & ~np.isnan(per_tile_db)] + if _agg.size == 0: + # Tiles produced values but none of them are tissue. Reporting the + # background median here is what this fix exists to stop, so decline + # instead. aggregate_snr_verdict maps a verdict-less part to NOT_COMPUTED. + return { + "status": "skipped", + "reason": "no_valid_tissue_roi_tiles", + "n_rois_all_tiles": len(dbs), + } + + _t = snr_thresholds or {} + _img = _t.get("image_snr_db") or {} + warn_db = float(_img.get("warn", 15.0)) + fail_db = float(_img.get("fail", 10.0)) + med_db = float(np.median(_agg)) + return { + "status": "ok", + "method": "per_roi_otsu", + "map_key": map_key, + "scope": "tissue_tiles", + "n_rois_computed": int(_agg.size), + "n_rois_all_tiles": len(dbs), + "snr_db_median": med_db, + "snr_db_mean": float(np.mean(_agg)), + "snr_db_p25": float(np.percentile(_agg, 25)), + "snr_db_p75": float(np.percentile(_agg, 75)), + "verdict": "PASS" + if med_db >= warn_db + else ("WARN" if med_db >= fail_db else "FAIL"), + } + + +# --------------------------------------------------------------------------- +# Transcripts +# --------------------------------------------------------------------------- + + +#: Transcript rows per batch for the per-ROI SNR reduction. Bounded so the +#: per-batch temporaries never approach the frame this replaces. +ROI_TX_BATCH_ROWS = 2_000_000 + + +def load_transcripts(path: Path) -> pd.DataFrame: + """Load minimal columns: feature_name, x_location, y_location, cell_id (optional). + + Projects the columns at read time rather than after. A Xenium transcript + table has ~20 columns and can run to several hundred million rows, and this + load happens while the full-resolution focus planes are still live — reading + everything and then slicing cost three overlapping copies of the full frame. + """ + path = Path(path) + cols_want = ["feature_name", "x_location", "y_location"] + + if path.suffix.lower() in (".parquet", ".pq"): + available = set(pq.ParquetFile(path).schema_arrow.names) + missing = [c for c in cols_want if c not in available] + if missing: + raise ValueError(f"transcripts missing column(s) {missing!r}") + keep = cols_want + (["cell_id"] if "cell_id" in available else []) + return pd.read_parquet(path, columns=keep) + + if path.name.endswith(".csv.gz") or path.suffix.lower() == ".csv": + available = set(pd.read_csv(path, nrows=0).columns) + missing = [c for c in cols_want if c not in available] + if missing: + raise ValueError(f"transcripts missing column(s) {missing!r}") + keep = cols_want + (["cell_id"] if "cell_id" in available else []) + return pd.read_csv(path, usecols=keep) + + raise ValueError(f"Unsupported transcript format: {path}") + + +def transcripts_um_to_px(df_tx: pd.DataFrame, pixel_size_um: float) -> pd.DataFrame: + """Convert x_location / y_location from microns to pixels (divide by pixel size).""" + out = df_tx.copy() + ps = float(pixel_size_um) + if ps <= 0: + raise ValueError("pixel_size_um must be positive") + out["x_px"] = out["x_location"].astype(np.float64) / ps + out["y_px"] = out["y_location"].astype(np.float64) / ps + return out + + +# --------------------------------------------------------------------------- +# ROI transcript SNR + neg_pct +# --------------------------------------------------------------------------- + + +def _infer_uniform_grid_strides( + df_grid_roi: pd.DataFrame, +) -> Optional[Tuple[int, int]]: + """ + Infer (stride_x, stride_y) for a regular grid like ``image_qc`` (arange(0, W, stride)). + + Returns None if x1/y1 spacing is irregular (fall back to slow geometric assign). + """ + x1 = df_grid_roi["x1"].to_numpy(dtype=np.int64) + y1 = df_grid_roi["y1"].to_numpy(dtype=np.int64) + ux = np.sort(np.unique(x1)) + uy = np.sort(np.unique(y1)) + if len(ux) >= 2: + dx = np.diff(ux) + if dx.size == 0 or int(dx[0]) <= 0 or not np.all(dx == dx[0]): + return None + sx = int(dx[0]) + else: + sx = int(df_grid_roi["x2"].iloc[0] - df_grid_roi["x1"].iloc[0]) + if sx <= 0: + return None + if len(uy) >= 2: + dy = np.diff(uy) + if dy.size == 0 or int(dy[0]) <= 0 or not np.all(dy == dy[0]): + return None + sy = int(dy[0]) + else: + sy = int(df_grid_roi["y2"].iloc[0] - df_grid_roi["y1"].iloc[0]) + if sy <= 0: + return None + if not bool(np.all((x1 % sx) == 0)) or not bool(np.all((y1 % sy) == 0)): + return None + return sx, sy + + +class _UniformRoiLookup: + """Prebuilt (iy, ix) -> roi_id lattice, so the streaming path builds it once. + + `_roi_grid_assign_fast_uniform` derives the lattice from the ROI table on every + call, and its collision check is an `np.unique` over one entry per ROI -- 424 ms + for 4.49 M ROIs. Calling it per transcript batch would spend that repeatedly on a + lookup that cannot change: at 1,325,798,498 transcripts the reference sample is + 663 batches, so ~291 s of pure rebuild. + + Returns None from :meth:`build` when the ROI layout is not a simple lattice, so + the caller can fall back to the general path. + """ + + __slots__ = ("_grid", "_stride_x", "_stride_y", "_max_ix", "_max_iy") + + def __init__(self, grid, stride_x, stride_y, max_ix, max_iy): + self._grid = grid + self._stride_x = stride_x + self._stride_y = stride_y + self._max_ix = max_ix + self._max_iy = max_iy + + @classmethod + def build( + cls, df_grid_roi: pd.DataFrame, stride_x: int, stride_y: int + ) -> Optional["_UniformRoiLookup"]: + x1 = df_grid_roi["x1"].to_numpy(dtype=np.int64) + y1 = df_grid_roi["y1"].to_numpy(dtype=np.int64) + rids = df_grid_roi["roi_id"].to_numpy(dtype=np.int32) + ix = x1 // stride_x + iy = y1 // stride_y + max_ix, max_iy = int(ix.max()) + 1, int(iy.max()) + 1 + flat_idx = iy.astype(np.int64) * max_ix + ix.astype(np.int64) + if len(np.unique(flat_idx)) != len(flat_idx): + return None + grid = np.full(max_iy * max_ix, -1, dtype=np.int32) + grid[flat_idx] = rids + return cls(grid.reshape(max_iy, max_ix), stride_x, stride_y, max_ix, max_iy) + + def assign(self, x_px: np.ndarray, y_px: np.ndarray) -> np.ndarray: + """roi_id per point. Out-of-lattice coordinates clip to the edge ROI, which + is what `_roi_grid_assign_fast_uniform` has always done.""" + xi = (x_px.astype(np.int64, copy=False) // self._stride_x).clip( + 0, self._max_ix - 1 + ) + yi = (y_px.astype(np.int64, copy=False) // self._stride_y).clip( + 0, self._max_iy - 1 + ) + return self._grid[yi, xi] + + +def _roi_grid_assign_fast_uniform( + df_grid_roi: pd.DataFrame, + x_px: np.ndarray, + y_px: np.ndarray, + stride_x: int, + stride_y: int, +) -> Optional[np.ndarray]: + """ + O(n_tx) assignment: tile index from pixel coords, lookup pre-filled roi_id grid. + + Returns None if ROI layout does not match a simple (iy, ix) lattice (collisions). + """ + lookup = _UniformRoiLookup.build(df_grid_roi, stride_x, stride_y) + if lookup is None: + return None + return lookup.assign(x_px, y_px) + + +def _roi_grid_assign( + df_grid_roi: pd.DataFrame, + x_px: np.ndarray, + y_px: np.ndarray, + stride_xy: Optional[Tuple[int, int]] = None, +) -> np.ndarray: + """ + Assign each transcript pixel to ``roi_id``, or -1. + + **Fast path (typical):** uniform stride grid from ``image_qc`` → O(n_tx) index + lookup. + Pass ``stride_xy=(stride_x, stride_y)`` from the same ``roi_size`` / stride used to build + the grid (see ``bin/image_qc.py``) to skip inference on large ROI tables. + + **Slow path:** O(n_roi × n_tx) rectangle tests (irregular ROIs / spikes). + """ + if stride_xy is not None: + sx, sy = int(stride_xy[0]), int(stride_xy[1]) + if sx > 0 and sy > 0: + fast = _roi_grid_assign_fast_uniform(df_grid_roi, x_px, y_px, sx, sy) + if fast is not None: + logger.debug( + "SNR ROI assignment: fast path (caller stride %d×%d), %d points", + sx, + sy, + int(x_px.shape[0]), + ) + return fast + logger.info( + "SNR ROI assignment: caller stride (%d, %d) incompatible with ROI layout; " + "inferring or slow path", + sx, + sy, + ) + + inferred = _infer_uniform_grid_strides(df_grid_roi) + if inferred is not None: + sx, sy = inferred + fast = _roi_grid_assign_fast_uniform(df_grid_roi, x_px, y_px, sx, sy) + if fast is not None: + logger.info( + "SNR ROI assignment: fast uniform grid (inferred stride %d×%d), %d transcripts", + sx, + sy, + int(x_px.shape[0]), + ) + return fast + logger.info( + "SNR ROI assignment: inferred stride rejected (collision); using slow path" + ) + + rid_out = np.full(x_px.shape[0], -1, dtype=np.int32) + rids = df_grid_roi["roi_id"].to_numpy(dtype=np.int32) + x1a = df_grid_roi["x1"].to_numpy(dtype=np.float64) + x2a = df_grid_roi["x2"].to_numpy(dtype=np.float64) + y1a = df_grid_roi["y1"].to_numpy(dtype=np.float64) + y2a = df_grid_roi["y2"].to_numpy(dtype=np.float64) + n_tx = int(x_px.shape[0]) + n_roi = len(df_grid_roi) + if n_tx > 2_000_000 and n_roi * n_tx > 5e9: + logger.warning( + "Large transcript count (%d) × ROIs (%d): slow ROI assignment O(n_roi×n_tx).", + n_tx, + n_roi, + ) + for i in range(n_roi): + m = (x_px >= x1a[i]) & (x_px < x2a[i]) & (y_px >= y1a[i]) & (y_px < y2a[i]) + rid_out[m] = int(rids[i]) + return rid_out + + +def _prefix_mask_vectorized(feats: np.ndarray, regex: str) -> np.ndarray: + """Vectorised prefix match, for large transcript tables.""" + s = pd.Series(feats, dtype="string") + return s.str.match(regex, case=False).fillna(False).to_numpy(dtype=bool) + + +def _neg_mask_vectorized(feats: np.ndarray) -> np.ndarray: + """Same logic as ``is_neg_probe_feature``, vectorised.""" + return _prefix_mask_vectorized(feats, _NEG_REGEX) + + +def _excluded_mask_vectorized(feats: np.ndarray) -> np.ndarray: + """Same logic as ``is_excluded_feature``, vectorised.""" + return _prefix_mask_vectorized(feats, _EXCLUDED_REGEX) + + +def _mask_for_column(values: pd.Series, regex: str) -> np.ndarray: + """Prefix mask for a transcript ``feature_name`` column. + + When the column is dictionary-encoded -- which is how Xenium writes it, and what + ``ParquetFile.iter_batches`` hands back -- the regex runs over the ~13 k distinct + categories and the result is indexed by the codes, instead of materialising one + Python string per transcript. The mask is identical either way. + """ + if isinstance(values.dtype, pd.CategoricalDtype): + cats = _prefix_mask_vectorized(values.cat.categories.to_numpy(), regex) + codes = values.cat.codes.to_numpy() + out = np.zeros(codes.shape[0], dtype=bool) + known = codes >= 0 + out[known] = cats[codes[known]] + return out + return _prefix_mask_vectorized(values.astype(str).to_numpy(), regex) + + +def _neg_mask_for_column(values: pd.Series) -> np.ndarray: + """Negative-control mask for a transcript ``feature_name`` column.""" + return _mask_for_column(values, _NEG_REGEX) + + +def _excluded_mask_for_column(values: pd.Series) -> np.ndarray: + """Mask of features that count as neither signal nor background.""" + return _mask_for_column(values, _EXCLUDED_REGEX) + + +def _accumulate_roi_tx_counts( + df_grid_roi: pd.DataFrame, + row_ix: pd.Series, + x_px: np.ndarray, + y_px: np.ndarray, + is_neg: np.ndarray, + counters: Tuple[np.ndarray, np.ndarray, np.ndarray], + stride_xy: Optional[Tuple[int, int]], + lookup: Optional["_UniformRoiLookup"] = None, + roi_id_is_arange: bool = False, + is_excluded: Optional[np.ndarray] = None, +) -> None: + """Fold one batch of transcripts into the per-ROI counters. + + ``np.bincount`` rather than ``np.add.at``: both count occurrences exactly, but + ``add.at`` is an order of magnitude slower, and this runs over every transcript. + + Two count-preserving micro-opts: + + * When ``roi_id_is_arange`` (``roi_id == arange(n_rois)``, how ``image_qc`` builds + the grid), the ``roi_id -> row index`` map is the identity: a valid id maps to + itself and an out-of-grid ``-1`` stays ``-1``. The pandas ``reindex`` is then a + no-op, so ``j`` is ``assign`` directly. + * ``real = total - neg`` instead of a third ``bincount``. Every *counted* + transcript is exactly one of neg / real, so this is the same integer per ROI -- + two ``bincount`` passes rather than three. + + ``is_excluded`` marks features that are neither (decoding failures, see + ``_EXCLUDED_PATTERNS``). They are removed from ``ok`` before any counting, so they + leave ``total``, ``neg`` and ``real`` alike, which is what keeps the + ``real = total - neg`` identity above true. Excluding them from ``total`` also + keeps ``neg_pct = neg / total`` a fraction of *decoded* transcripts rather than of + everything the instrument emitted. + """ + if lookup is not None: + assign = lookup.assign(x_px, y_px) + else: + assign = _roi_grid_assign(df_grid_roi, x_px, y_px, stride_xy=stride_xy) + assign = np.asarray(assign, dtype=np.int64) + if roi_id_is_arange: + j = assign + else: + j = row_ix.reindex(assign, fill_value=-1).to_numpy() + ok = (j >= 0) & (assign >= 0) + if is_excluded is not None: + ok &= ~is_excluded + real_c, neg_c, total_c = counters + n = real_c.shape[0] + total_b = np.bincount(j[ok], minlength=n) + neg_b = np.bincount(j[ok & is_neg], minlength=n) + total_c += total_b + neg_c += neg_b + real_c += total_b - neg_b + + +def _stream_roi_tx_counts( + df_grid_roi: pd.DataFrame, + row_ix: pd.Series, + path: Path, + pixel_size_um: float, + stride_xy: Optional[Tuple[int, int]], + batch_rows: int = ROI_TX_BATCH_ROWS, + roi_id_is_arange: bool = False, +) -> Tuple[Tuple[np.ndarray, np.ndarray, np.ndarray], int]: + """Per-ROI real/neg/total transcript counts, read in batches, folded in parallel. + + The whole-frame version materialised ``feature_name``, ``x_location`` and + ``y_location`` for every transcript, then ``transcripts_um_to_px`` copied the + frame and added two more float64 columns. On run 1ZyVIlaKBYxJrQ that OOM-killed + IMAGE_QC (exit 137) at the 180 GB tier *after* the tile pass had completed in + 19.4 minutes. The outputs are three per-ROI int64 arrays, so nothing about this + needs the transcripts resident. + + Batches are decoded (``batch.to_pandas``, which releases the GIL) and folded + (``np.bincount``, also GIL-releasing) on a thread pool sized to ``os.cpu_count()``. + Each worker thread owns a private ``(real, neg, total)`` counter triple and folds + every batch it draws into it via ``_accumulate_roi_tx_counts``; the per-thread + partials are summed on the main thread once the pool drains. Counts are additive, + so the sum is bit-identical to the serial fold no matter how the batches interleave + across threads -- there is no order dependence in integer addition. In-flight + batches are capped at the worker count, so peak memory is bounded exactly as the + serial loop's was (a handful of ``batch_rows`` batches, not the whole table). + """ + n_rois = len(df_grid_roi) + ps = float(pixel_size_um) + if ps <= 0: + raise ValueError("pixel_size_um must be positive") + + # Built once, not per batch: see _UniformRoiLookup. None means the ROI layout is + # not a lattice, and each batch falls back to the general assignment. + lookup = None + if stride_xy is not None and stride_xy[0] > 0 and stride_xy[1] > 0: + lookup = _UniformRoiLookup.build( + df_grid_roi, int(stride_xy[0]), int(stride_xy[1]) + ) + if lookup is None: + logger.info( + "SNR ROI assignment: no uniform lattice; assigning per batch (slower)" + ) + + handle = pq.ParquetFile(str(path)) + columns = ["feature_name", "x_location", "y_location"] + max_workers = max(1, os.cpu_count() or 1) + logger.info( + "SNR ROI transcript counts: streaming %s rows in %s-row batches across %d threads", + f"{handle.metadata.num_rows:,}", + f"{batch_rows:,}", + max_workers, + ) + + # One counter triple per worker thread. Registered under a lock the first time a + # thread runs, so the main thread can sum them after the pool drains. + tls = threading.local() + partials: List[Tuple[np.ndarray, np.ndarray, np.ndarray]] = [] + partials_lock = threading.Lock() + + def _fold_one(batch) -> int: + thread_counters = getattr(tls, "counters", None) + if thread_counters is None: + thread_counters = ( + np.zeros(n_rois, dtype=np.int64), + np.zeros(n_rois, dtype=np.int64), + np.zeros(n_rois, dtype=np.int64), + ) + tls.counters = thread_counters + with partials_lock: + partials.append(thread_counters) + frame = batch.to_pandas() + _accumulate_roi_tx_counts( + df_grid_roi, + row_ix, + frame["x_location"].to_numpy(dtype=np.float64) / ps, + frame["y_location"].to_numpy(dtype=np.float64) / ps, + _neg_mask_for_column(frame["feature_name"]), + thread_counters, + stride_xy, + lookup=lookup, + roi_id_is_arange=roi_id_is_arange, + is_excluded=_excluded_mask_for_column(frame["feature_name"]), + ) + return len(frame) + + n_used = 0 + n_batches = 0 + batch_iter = handle.iter_batches(batch_size=batch_rows, columns=columns) + with ThreadPoolExecutor(max_workers=max_workers) as pool: + # Cap outstanding futures at the worker count so at most ~max_workers batches + # are resident at once (the same memory envelope as the serial loop). + inflight: deque = deque() + for batch in batch_iter: + inflight.append(pool.submit(_fold_one, batch)) + del batch + if len(inflight) >= max_workers: + n_used += inflight.popleft().result() + n_batches += 1 + while inflight: + n_used += inflight.popleft().result() + n_batches += 1 + + real_c = np.zeros(n_rois, dtype=np.int64) + neg_c = np.zeros(n_rois, dtype=np.int64) + total_c = np.zeros(n_rois, dtype=np.int64) + for p_real, p_neg, p_total in partials: + real_c += p_real + neg_c += p_neg + total_c += p_total + logger.info( + "SNR ROI transcript counts: %s rows in %d batches across %d threads", + f"{n_used:,}", + n_batches, + len(partials), + ) + return (real_c, neg_c, total_c), n_used + + +def compute_roi_snr( + df_grid_roi: pd.DataFrame, + df_tx: Optional[pd.DataFrame] = None, + x_col: str = "x_px", + y_col: str = "y_px", + roi_grid_stride: Optional[Tuple[int, int]] = None, + snr_thresholds: Optional[Dict[str, Any]] = None, + *, + transcripts_path: Optional[Path] = None, + pixel_size_um: Optional[float] = None, + batch_rows: int = ROI_TX_BATCH_ROWS, +) -> Tuple[pd.DataFrame, Dict[str, Any]]: + """ + Per-ROI real vs neg transcript counts, ratio, neg_pct; **mutates** *df_grid_roi* in place + (caller should pass a copy if the original must stay unchanged — ``run_snr_module`` does). + + Give either *df_tx* -- a frame with feature_name and pixel columns x_col, y_col, + see ``transcripts_um_to_px`` -- or *transcripts_path* plus *pixel_size_um*, in + which case the table is read in batches and never held whole. The counters are + additive, so both give identical results; the streaming form exists because the + whole-frame one OOM-killed IMAGE_QC on a production sample. + + ``roi_grid_stride`` should match the grid used in ``image_qc.py`` (typically + ``(roi_size, roi_size)`` when stride defaults to roi_size). + """ + if (df_tx is None) == (transcripts_path is None): + raise ValueError("pass exactly one of df_tx or transcripts_path") + + df = df_grid_roi + if "roi_id" not in df.columns: + df = df.copy() + df["roi_id"] = np.arange(len(df), dtype=np.int32) + + n_rois = len(df) + # Map roi_id → row index (avoids O(max(roi_id)) array if ids are sparse) + row_ix = pd.Series(np.arange(n_rois, dtype=np.int32), index=df["roi_id"].values) + + # image_qc builds the grid with roi_id == arange(n_rois); then row_ix is the + # identity and the per-batch reindex can be skipped (see _accumulate_roi_tx_counts). + roi_ids = df["roi_id"].to_numpy() + roi_id_is_arange = bool( + roi_ids.dtype.kind in ("i", "u") + and np.array_equal(roi_ids, np.arange(n_rois, dtype=roi_ids.dtype)) + ) + + if transcripts_path is not None: + if pixel_size_um is None: + raise ValueError("pixel_size_um is required with transcripts_path") + counters, n_used = _stream_roi_tx_counts( + df, + row_ix, + Path(transcripts_path), + pixel_size_um, + roi_grid_stride, + batch_rows=batch_rows, + roi_id_is_arange=roi_id_is_arange, + ) + else: + counters = ( + np.zeros(n_rois, dtype=np.int64), + np.zeros(n_rois, dtype=np.int64), + np.zeros(n_rois, dtype=np.int64), + ) + _accumulate_roi_tx_counts( + df, + row_ix, + df_tx[x_col].to_numpy(dtype=np.float64), + df_tx[y_col].to_numpy(dtype=np.float64), + _neg_mask_for_column(df_tx["feature_name"]), + counters, + roi_grid_stride, + roi_id_is_arange=roi_id_is_arange, + is_excluded=_excluded_mask_for_column(df_tx["feature_name"]), + ) + n_used = int(len(df_tx)) + real_c, neg_c, total_c = counters + + df["snr_real_tx"] = real_c + df["snr_neg_tx"] = neg_c + df["snr_total_tx"] = total_c + df["neg_pct"] = np.where( + total_c > 0, neg_c.astype(np.float64) / total_c.astype(np.float64), np.nan + ) + + ratio = np.divide( + real_c.astype(np.float64), + neg_c.astype(np.float64), + out=np.full(n_rois, np.nan, dtype=np.float64), + where=neg_c > 0, + ) + df["roi_tx_snr_ratio"] = ratio + df["roi_tx_snr_log"] = np.log10(np.maximum(ratio, 1e-12)) + + med_ratio = float(np.nanmedian(ratio[total_c > 0])) + med_neg_pct = float(np.nanmedian(df.loc[total_c > 0, "neg_pct"])) + _t = snr_thresholds or {} + _rt = _t.get("roi_tx") or {} + ratio_warn = float(_rt.get("ratio_warn", 3.0)) + ratio_fail = float(_rt.get("ratio_fail", 1.5)) + neg_pct_warn = float(_rt.get("neg_pct_warn", 0.15)) + neg_pct_fail = float(_rt.get("neg_pct_fail", 0.30)) + # A bundle with no negative-control probes leaves every ratio NaN (the divide + # below is masked to neg_c > 0), so med_ratio is NaN. Both comparisons then + # evaluate False -- NaN < warn is False, NaN > warn is False -- and the verdict + # would stay at its initial "PASS" while a bare NaN went into the JSON. Decline + # explicitly instead: aggregate_snr_verdict maps a verdict-less part to + # NOT_COMPUTED. Note this is the all-NaN case only; the separate question of + # whether zero-neg tiles should be admitted via a pseudocount is a live + # calibration decision and is NOT changed here (it would move + # median_roi_tx_snr_ratio and re-open the ratio_warn/ratio_fail pair). + _ratio_undefined = not math.isfinite(med_ratio) + if _ratio_undefined: + summary = { + "status": "skipped", + "method": "roi_tx_target_vs_neg", + "reason": "no_negative_control_transcripts", + "n_transcripts_used": int(n_used), + "n_rois_with_tx": int(np.sum(total_c > 0)), + "median_neg_pct": med_neg_pct if math.isfinite(med_neg_pct) else None, + } + # run_snr_module owns serialisation (it files this into parts[SNR_CKEY_ROI_TX] + # and writes snr_metrics.json), so just hand the skipped summary back. + return df, summary + + verdict = "PASS" + if med_ratio < ratio_warn or med_neg_pct > neg_pct_warn: + verdict = "WARN" + if med_ratio < ratio_fail or med_neg_pct > neg_pct_fail: + verdict = "FAIL" + + summary = { + "status": "ok", + "method": "roi_tx_target_vs_neg", + "n_transcripts_used": int(n_used), + "n_rois_with_tx": int(np.sum(total_c > 0)), + "median_roi_tx_snr_ratio": med_ratio, + "median_neg_pct": med_neg_pct, + "verdict": verdict, + } + return df, summary + + +# --------------------------------------------------------------------------- +# Spatial clustering of neg_pct — Moran (optional) + quadrant fallback +# --------------------------------------------------------------------------- + + +def compute_neg_spatial_autocorrelation( + df_grid_roi: pd.DataFrame, + moran_max_rois: int = 25_000, + moran_permutations: int = 99, + moran_subsample_seed: int = 42, + *, + include_moran: bool = False, + snr_thresholds: Optional[Dict[str, Any]] = None, +) -> Dict[str, Any]: + """ + SNR_plan: Moran's I with 4-neighbour weights if libpysal/esda available; + else quadrant variance proxy on neg_pct. + + Moran + permutations on hundreds of thousands of ROIs is impractical; when + ``len(df) > moran_max_rois``, a fixed-seed random subsample is used **only** + for the Moran branch. Quadrant summary still uses all finite-``neg_pct`` ROIs. + + If ``include_moran`` is False, the Moran branch is not run (quadrant summary only; + avoids PySAL/esda and saves time). If True but packages are missing, Moran is + skipped inside a try/except and ``moran_note`` is set — the run still succeeds. + """ + if "neg_pct" not in df_grid_roi.columns: + return {"status": "skipped", "reason": "neg_pct not computed"} + + df = df_grid_roi[np.isfinite(df_grid_roi["neg_pct"])].copy() + if len(df) < 8: + return {"status": "skipped", "reason": "too_few_rois"} + + cx = (df["x1"].astype(np.float64) + df["x2"].astype(np.float64)) / 2.0 + cy = (df["y1"].astype(np.float64) + df["y2"].astype(np.float64)) / 2.0 + z = df["neg_pct"].astype(np.float64).values + + # Quadrant proxy + mx, my = float(np.median(cx)), float(np.median(cy)) + quad = np.zeros(4, dtype=np.float64) + nq = np.zeros(4, dtype=np.int64) + for k, mask in enumerate( + [ + (cx < mx) & (cy < my), + (cx >= mx) & (cy < my), + (cx < mx) & (cy >= my), + (cx >= mx) & (cy >= my), + ] + ): + quad[k] = float(np.mean(z[mask])) if np.any(mask) else 0.0 + nq[k] = int(np.sum(mask)) + spread = float((np.max(quad) - np.min(quad)) / (np.mean(z) + 1e-9)) + _t = snr_thresholds or {} + _ns = _t.get("neg_spatial") or {} + qs_warn = float(_ns.get("quadrant_spread_warn", 0.5)) + qs_fail = float(_ns.get("quadrant_spread_fail", 1.0)) + q_verdict = "PASS" + if spread > qs_warn: + q_verdict = "WARN" + if spread > qs_fail: + q_verdict = "FAIL" + + out: Dict[str, Any] = { + "status": "ok", + "method_primary": "quadrant_spread", + "quadrant_neg_pct_means": quad.tolist(), + "quadrant_counts": nq.tolist(), + "quadrant_spread_index": spread, + "quadrant_verdict": q_verdict, + } + + # Optional Moran (subsampled when n is large — see docstring) + if include_moran: + try: + from esda.moran import Moran # type: ignore + from libpysal.weights import W # type: ignore + from scipy.spatial import cKDTree # type: ignore + + n_all = len(df) + if n_all > moran_max_rois: + rng = np.random.default_rng(moran_subsample_seed) + pick = rng.choice(n_all, size=moran_max_rois, replace=False) + cx_m = cx.iloc[pick].to_numpy(dtype=np.float64) + cy_m = cy.iloc[pick].to_numpy(dtype=np.float64) + z_m = z[pick] + out["moran_subsample"] = { + "n_used": int(moran_max_rois), + "n_available": int(n_all), + "seed": int(moran_subsample_seed), + } + else: + cx_m = cx.to_numpy(dtype=np.float64) + cy_m = cy.to_numpy(dtype=np.float64) + z_m = z + + xy = np.column_stack([cx_m, cy_m]) + # kNN weights k=4 as SNR_plan neighbour count + tree = cKDTree(xy) + _d, idx = tree.query(xy, k=min(5, len(xy))) + neighbors = {i: [int(j) for j in idx[i][1:]] for i in range(len(xy))} + + w = W(neighbors, silence_warnings=True) + mi = Moran(z_m, w, permutations=int(moran_permutations)) + moran_i = float(mi.I) + p_sim = float(mi.p_sim) if mi.p_sim is not None else float("nan") + m_warn = float(_ns.get("moran_warn", 0.5)) + m_fail = float(_ns.get("moran_fail", 1.0)) + m_verdict = "PASS" + if moran_i > m_warn: + m_verdict = "WARN" + if moran_i > m_fail: + m_verdict = "FAIL" + if not math.isnan(p_sim) and p_sim >= 0.05: + m_verdict = "PASS" + out["method_secondary"] = "moran_knn4" + out["moran_i"] = moran_i + out["moran_p_sim"] = p_sim + out["moran_verdict"] = m_verdict + except Exception as e: + out["moran_note"] = f"skipped ({type(e).__name__}: {e})" + else: + out["moran_note"] = "skipped (Moran disabled; quadrant summary only)" + + # Combined: escalate per SNR_plan (clustered noise) + prim = out.get("moran_verdict", q_verdict) + if prim == "FAIL" or q_verdict == "FAIL": + out["verdict"] = "FAIL" + elif prim == "WARN" or q_verdict == "WARN": + out["verdict"] = "WARN" + else: + out["verdict"] = "PASS" + + if out["verdict"] in ("WARN", "FAIL"): + out["report_note"] = ( + "Elevated spatial clustering of negative-control probes can reflect " + "genuine biological heterogeneity (e.g. necrosis, adipose tissue, or " + "varying cell density) rather than a technical artefact. Consider " + "reviewing the spatial distribution map before concluding a quality issue." + ) + return out + + +# --------------------------------------------------------------------------- +# Slide-level matrix SNR (h5) — Plummer-style vs SpatialQM-style +# --------------------------------------------------------------------------- + + +def load_expression_matrix_h5( + path: Path, +) -> Tuple[Any, List[str], List[str], Dict[str, Any]]: + """ + Load 10x-style Xenium h5 sparse matrix [features x cells]. + + 10x / Xenium may store **CSR** (``len(indptr) == n_features + 1``) or **CSC** + (``len(indptr) == n_cells + 1``). Returns a **scipy.sparse.csr_matrix** without + densifying (full matrix can be tens of GB). + """ + import h5py + + path = Path(path) + meta: Dict[str, Any] = {"path": str(path)} + with h5py.File(path, "r") as f: + if "matrix" in f: + g = f["matrix"] + shape_t = tuple(g["shape"][:]) if "shape" in g else None + if shape_t is None or len(shape_t) != 2: + raise ValueError("h5 matrix missing valid shape") + n_feat, n_cell = int(shape_t[0]), int(shape_t[1]) + data = g["data"][:] + indices = g["indices"][:] + indptr = g["indptr"][:] + from scipy.sparse import csc_matrix, csr_matrix # type: ignore + + if len(indptr) == n_feat + 1: + mat = csr_matrix( + (data, indices, indptr), shape=shape_t, dtype=np.float64 + ) + meta["sparse_layout"] = "csr" + elif len(indptr) == n_cell + 1: + mat = csc_matrix( + (data, indices, indptr), shape=shape_t, dtype=np.float64 + ).tocsr() + meta["sparse_layout"] = "csc_assembled_csr" + else: + raise ValueError( + f"Unrecognized sparse indptr length {len(indptr)} for shape {shape_t}" + ) + fg = g["features"] + if "name" in fg: + raw = fg["name"][:] + elif "id" in fg: + raw = fg["id"][:] + else: + raise ValueError("h5 matrix/features has neither 'name' nor 'id'") + names = [x.decode() if isinstance(x, bytes) else str(x) for x in raw] + if "feature_type" in fg: + ftype = [ + x.decode() if isinstance(x, bytes) else str(x) + for x in fg["feature_type"][:] + ] + else: + ftype = ["Gene Expression"] * len(names) + else: + raise ValueError("Unrecognized h5 layout (expected 'matrix' group)") + + meta.update({"n_features": mat.shape[0], "n_cells": mat.shape[1]}) + return mat, names, ftype, meta + + +PSEUDOCOUNT = 0.1 +EPS = 1e-8 + +# Fallback defaults for slide-level SNR (used when YAML thresholds not provided). +SLIDE_SNR_THRESHOLDS = { + "plummer_pass_min": 0.12, + "plummer_fail_below": 0.10, +} + + +def _mean_all(x: Any) -> float: + """Mean across all elements; works for scipy sparse and dense arrays.""" + m = x.mean() + if hasattr(m, "A1"): + return float(m.A1[0]) + return float(np.asarray(m).reshape(-1)[0]) + + +def _mean_per_feature(feature_by_cell: Any) -> np.ndarray: + """Row means (features across cells); works for sparse and dense.""" + out = feature_by_cell.mean(axis=1) + return np.asarray(out, dtype=np.float64).ravel() + + +def _build_feature_masks(feature_names: List[str]) -> tuple[np.ndarray, np.ndarray]: + """Signal / background masks for the slide expression matrix. + + ``is_real`` is NOT simply ``~is_neg``: features matching + ``_EXCLUDED_PATTERNS`` (decoding failures) are neither, and were previously + swept into ``is_real``, inflating the signal side of the ratio. They are now + absent from both masks, so a feature is real, negative, or neither. + """ + is_neg = np.array([is_neg_probe_feature(n) for n in feature_names], dtype=bool) + is_excluded = np.array([is_excluded_feature(n) for n in feature_names], dtype=bool) + is_real = ~is_neg & ~is_excluded + return is_real, is_neg + + +def compute_slide_snr_plummer_corrected( + mat: Any, + feature_names: List[str], + snr_thresholds: Optional[Dict[str, Any]] = None, +) -> Dict[str, Any]: + """ + Slide SNR: log10(mean_real + 0.1) - log10(mean_neg + 0.1) over matrix elements + (features × cells), real vs neg probe subsets. + """ + is_real, is_neg = _build_feature_masks(feature_names) + if not np.any(is_real) or not np.any(is_neg): + return {"status": "skipped", "reason": "missing real or neg features"} + + real_mat = mat[is_real] + neg_mat = mat[is_neg] + + mean_real = _mean_all(real_mat) + mean_neg = _mean_all(neg_mat) + log_real = math.log10(mean_real + PSEUDOCOUNT) + log_neg = math.log10(mean_neg + PSEUDOCOUNT) + snr = float(log_real - log_neg) + + real_feature_means = _mean_per_feature(real_mat) + dynamic_range = float( + math.log10(float(real_feature_means.max()) + PSEUDOCOUNT) - log_neg + ) + noise_floor_pct = float(mean_neg / (mean_real + EPS) * 100.0) + + _t = snr_thresholds or {} + _pl = _t.get("slide_plummer") or {} + pmin = float(_pl.get("pass_min", SLIDE_SNR_THRESHOLDS["plummer_pass_min"])) + fbelow = float(_pl.get("fail_below", SLIDE_SNR_THRESHOLDS["plummer_fail_below"])) + if snr < fbelow: + verdict = "FAIL" + elif snr < pmin: + verdict = "WARN" + else: + verdict = "PASS" + + return { + "status": "ok", + "method": "plummer_corrected", + "formula": "log10(mean_real+0.1)-log10(mean_neg+0.1)", + "snr": snr, + "dynamic_range": dynamic_range, + "mean_real": mean_real, + "mean_neg": mean_neg, + "n_real_genes": int(np.sum(is_real)), + "n_neg_probes": int(np.sum(is_neg)), + "noise_floor_pct": noise_floor_pct, + "verdict": verdict, + "thresholds_used": { + "pass_min": pmin, + "fail_below": fbelow, + "note": "PASS if snr >= pass_min; WARN if fail_below <= snr < pass_min; FAIL if snr < fail_below", + }, + } + + +def compute_slide_snr_spatialqm_corrected( + mat: Any, + feature_names: List[str], + snr_thresholds: Optional[Dict[str, Any]] = None, +) -> Dict[str, Any]: + """ + Per-gene variant: snr_g = log10(mean_gene_g + 0.1) - log10(mean_neg + 0.1); slide_snr = mean(snr_g). + """ + is_real, is_neg = _build_feature_masks(feature_names) + if not np.any(is_real) or not np.any(is_neg): + return {"status": "skipped", "reason": "missing real or neg features"} + + real_mat = mat[is_real] + neg_mat = mat[is_neg] + mean_neg = _mean_all(neg_mat) + log_neg = math.log10(mean_neg + PSEUDOCOUNT) + + real_feature_means = _mean_per_feature(real_mat) + per_gene_snr = np.log10(real_feature_means + PSEUDOCOUNT) - log_neg + snr_mean = float(np.mean(per_gene_snr)) + snr_median = float(np.median(per_gene_snr)) + snr_p10 = float(np.percentile(per_gene_snr, 10)) + pct_genes_above_neg = float(np.mean(per_gene_snr > 0.0) * 100.0) + dynamic_range = float( + math.log10(float(real_feature_means.max()) + PSEUDOCOUNT) - log_neg + ) + + _t = snr_thresholds or {} + _sq = _t.get("slide_spatialqm") or {} + verdict_metric = _sq.get("metric", "pct_genes_above_neg") + sq_warn = float(_sq.get("warn", 45)) + sq_fail = float(_sq.get("fail", 35)) + + if verdict_metric == "pct_genes_above_neg": + verdict_value = pct_genes_above_neg + else: + # Legacy: snr_mean + verdict_value = snr_mean + verdict = ( + "PASS" + if verdict_value >= sq_warn + else ("WARN" if verdict_value >= sq_fail else "FAIL") + ) + + return { + "status": "ok", + "method": "spatialqm_per_gene_corrected", + "snr_mean": snr_mean, + "snr_median": snr_median, + "snr_p10": snr_p10, + "pct_genes_above_neg": pct_genes_above_neg, + "dynamic_range": dynamic_range, + "mean_neg": mean_neg, + "n_real_genes": int(np.sum(is_real)), + "n_neg_probes": int(np.sum(is_neg)), + "verdict": verdict, + "thresholds_used": { + "metric": verdict_metric, + "warn": sq_warn, + "fail": sq_fail, + }, + } + + +# --------------------------------------------------------------------------- +# aggregate_snr_verdict + run_snr_module +# --------------------------------------------------------------------------- + + +def save_snr_roi_tx_table( + df: pd.DataFrame, + outdir: Path, + *, + basename: str = SNR_ROI_TX_TABLE_BASENAME, +) -> Optional[Path]: + """ + Persist the grid with ``compute_roi_snr`` columns (``snr_real_tx``, ``snr_neg_tx``, + ``snr_total_tx``, ``neg_pct``, ``roi_tx_snr_ratio``, …) for downstream plots. + + Writes ``{basename}.parquet`` when Parquet is available; otherwise ``{basename}.csv.gz``. + """ + need = {"snr_real_tx", "snr_neg_tx", "snr_total_tx"} + if not need.issubset(df.columns): + return None + outdir = Path(outdir) + outdir.mkdir(parents=True, exist_ok=True) + path_pq = outdir / f"{basename}.parquet" + try: + df.to_parquet(path_pq, index=False) + logger.info("Wrote %s", path_pq) + return path_pq + except Exception as e: + logger.warning("SNR_roi_tx Parquet write failed (%s), trying CSV.gz", e) + path_gz = outdir / f"{basename}.csv.gz" + try: + df.to_csv(path_gz, index=False, compression="gzip") + logger.info("Wrote %s", path_gz) + return path_gz + except Exception as e2: + logger.warning("Could not write SNR_roi_tx table: %s", e2) + return None + + +def _write_snr_json(summary: Dict[str, Any], outdir: Path) -> None: + """Write snr_metrics.json, refusing to emit a bare NaN token. + + ``json.dump(..., default=str)`` does NOT catch float NaN: ``default`` only fires + for objects json cannot serialise, and Python happily writes NaN as the bare + non-standard ``NaN`` literal, which strict JSON parsers reject. ``allow_nan=False`` + turns that into a loud ValueError, and any non-finite value is replaced with null + first so a real metric never silently becomes the string "nan". + """ + + def _clean(v: Any) -> Any: + if isinstance(v, dict): + return {k: _clean(x) for k, x in v.items()} + if isinstance(v, (list, tuple)): + return [_clean(x) for x in v] + if isinstance(v, float) and not math.isfinite(v): + return None + return v + + try: + out_json = outdir / "snr_metrics.json" + with open(out_json, "w") as f: + json.dump(_clean(summary), f, indent=2, default=str, allow_nan=False) + logger.info("Wrote %s", out_json) + except Exception as e: + logger.warning("Could not write snr_metrics.json: %s", e) + + +def aggregate_snr_verdict(parts: Dict[str, Dict[str, Any]]) -> Dict[str, Any]: + """Any FAIL → overall FAIL; clustered neg (FAIL) escalates WARN → FAIL. + + Components that could not run return ``{"status": "skipped"/"error"}`` with no + ``verdict`` key at all. Those match neither branch below, so an aggregate that + starts at "PASS" and only ever moves on FAIL or WARN reported PASS for a module + where nothing was measured -- a green report on an unreadable input, which is + worse than a crash because nothing about the run looks wrong. If NO component + produced a verdict, say NOT_COMPUTED, matching what the whole-module failure + path in image_qc.py already emits. Components that did not report are always + listed, so a partial PASS cannot be mistaken for a complete one. + """ + overall = "PASS" + n_verdicts = 0 + not_computed: list[str] = [] + for key, sub in parts.items(): + if not isinstance(sub, dict): + not_computed.append(str(key)) + continue + v = sub.get("verdict") + if v is None: + not_computed.append(str(key)) + continue + n_verdicts += 1 + if v == "FAIL": + overall = "FAIL" + elif v == "WARN" and overall != "FAIL": + overall = "WARN" + ns = parts.get(SNR_CKEY_ROI_NEG_SPATIAL) or {} + if ns.get("verdict") == "FAIL" and overall == "WARN": + overall = "FAIL" + if n_verdicts == 0: + overall = "NOT_COMPUTED" + out: Dict[str, Any] = {"overall_snr_verdict": overall} + if not_computed: + out["components_not_computed"] = sorted(not_computed) + return out + + +def run_snr_module( + xenium_bundle_dir: Optional[Path], + df_grid_roi: pd.DataFrame, + outdir: Path, + focus_maps: Optional[Dict[str, Any]] = None, + roi_snr_db: Optional[Any] = None, + intensity_threshold: float = 0.0, + transcripts_path: Optional[Path] = None, + cell_matrix_h5: Optional[Path] = None, + pixel_size_um: Optional[float] = None, + otsu_max_rois: Optional[int] = None, + save_roi_tx_table: bool = True, + write_snr_json: bool = True, + snr_include_moran: bool = False, + roi_grid_stride: Optional[Tuple[int, int]] = None, + snr_thresholds: Optional[Dict[str, Any]] = None, +) -> Tuple[pd.DataFrame, Dict[str, Any]]: + """ + Run all SNR sub-components; returns grid (with transcript columns when computed) and summary dict. + + Writes ``snr_metrics.json`` under outdir when ``write_snr_json`` is True. When ROI transcript + SNR succeeds and ``save_roi_tx_table`` is True, also writes ``SNR_roi_tx.parquet`` (or + ``.csv.gz`` fallback) and sets ``components[SNR_roi_tx]["per_roi_table_file"]`` to the basename. + + If ``snr_include_moran`` is False (default), neg-control spatial uses quadrant spread only (no PySAL Moran). + + ``roi_grid_stride``: optional ``(stride_x, stride_y)`` matching ``image_qc`` grid construction + (same as ``roi_size`` when stride is unset). Enables O(n_tx) transcript-to-ROI assignment + without scanning unique x1/y1 on huge grids. + """ + outdir = Path(outdir) + outdir.mkdir(parents=True, exist_ok=True) + bundle = Path(xenium_bundle_dir) if xenium_bundle_dir else None + + df = df_grid_roi.copy() + parts: Dict[str, Dict[str, Any]] = {} + + thresholds = snr_thresholds or {} + + # 1) Image SNR (ROI df quartiles → dB) + parts[SNR_CKEY_IMAGE_ROI_QUARTILE_DB] = compute_image_snr_from_roi_df( + df, intensity_threshold=intensity_threshold, snr_thresholds=thresholds + ) + + # 2) Image SNR (Otsu) + if focus_maps or roi_snr_db is not None: + parts[SNR_CKEY_IMAGE_OTSU] = compute_image_snr_from_pixel_maps( + focus_maps, + df, + max_rois=otsu_max_rois, + snr_thresholds=thresholds, + precomputed_db=roi_snr_db, + ) + else: + parts[SNR_CKEY_IMAGE_OTSU] = {"status": "skipped", "reason": "no focus_maps"} + + # 3) ROI transcript SNR + tx_path = transcripts_path + if tx_path is None and bundle: + cand = bundle / "transcripts.parquet" + if cand.exists(): + tx_path = cand + else: + cg = bundle / "transcripts.csv.gz" + if cg.exists(): + tx_path = cg + if tx_path and tx_path.exists(): + try: + if pixel_size_um is None: + parts[SNR_CKEY_ROI_TX] = { + "status": "skipped", + "reason": "pixel_size_um required to convert transcript coordinates to pixels", + } + else: + # Streamed: see _stream_roi_tx_counts. Reading the table whole and + # then copying it in transcripts_um_to_px OOM-killed this step. + df, rtx = compute_roi_snr( + df, + transcripts_path=Path(tx_path), + pixel_size_um=pixel_size_um, + roi_grid_stride=roi_grid_stride, + snr_thresholds=thresholds, + ) + if save_roi_tx_table and rtx.get("status") == "ok": + pth = save_snr_roi_tx_table(df, outdir) + if pth is not None: + rtx["per_roi_table_file"] = pth.name + parts[SNR_CKEY_ROI_TX] = rtx + except Exception as e: + logger.exception("ROI transcript SNR failed") + parts[SNR_CKEY_ROI_TX] = {"status": "error", "error": str(e)} + else: + parts[SNR_CKEY_ROI_TX] = {"status": "skipped", "reason": "no transcripts file"} + + # 4) Slide SNR + h5_path = cell_matrix_h5 + if h5_path is None and bundle: + hp = bundle / "cell_feature_matrix.h5" + if hp.exists(): + h5_path = hp + if h5_path and Path(h5_path).exists(): + try: + mat, names, _ftype, _meta = load_expression_matrix_h5(Path(h5_path)) + parts[SNR_CKEY_SLIDE_PLUMMER] = compute_slide_snr_plummer_corrected( + mat, names, snr_thresholds=thresholds + ) + parts[SNR_CKEY_SLIDE_SPATIALQM] = compute_slide_snr_spatialqm_corrected( + mat, names, snr_thresholds=thresholds + ) + except Exception as e: + logger.exception("Slide SNR failed") + parts[SNR_CKEY_SLIDE_PLUMMER] = {"status": "error", "error": str(e)} + parts[SNR_CKEY_SLIDE_SPATIALQM] = {"status": "error", "error": str(e)} + else: + parts[SNR_CKEY_SLIDE_PLUMMER] = { + "status": "skipped", + "reason": "no cell_feature_matrix.h5", + } + parts[SNR_CKEY_SLIDE_SPATIALQM] = { + "status": "skipped", + "reason": "no cell_feature_matrix.h5", + } + + # 5) Neg spatial autocorrelation (needs neg_pct) + if "neg_pct" in df.columns: + parts[SNR_CKEY_ROI_NEG_SPATIAL] = compute_neg_spatial_autocorrelation( + df, include_moran=snr_include_moran, snr_thresholds=thresholds + ) + else: + parts[SNR_CKEY_ROI_NEG_SPATIAL] = { + "status": "skipped", + "reason": "neg_pct not available", + } + + verdict = aggregate_snr_verdict(parts) + summary = { + "components": parts, + "verdict": verdict, + } + + if write_snr_json: + _write_snr_json(summary, outdir) + + return df, summary + + +def read_xenium_pixel_size_um(bundle_dir: Path) -> Optional[float]: + """Read ``pixel_size`` (or ``pixel_size_um``) from ``experiment.xenium``.""" + exp = Path(bundle_dir) / "experiment.xenium" + if not exp.is_file(): + return None + try: + with open(exp, encoding="utf-8") as f: + meta = json.load(f) + ps = float(meta.get("pixel_size", meta.get("pixel_size_um", 0.0))) + except (OSError, ValueError, TypeError, json.JSONDecodeError): + return None + if ps <= 0 or not math.isfinite(ps): + return None + return ps + + +def compute_snr_summary( + df_grid_roi: pd.DataFrame, + *, + bundle_dir: Path, + outdir: Path, + focus_maps: Optional[Dict[str, Any]] = None, + roi_snr_db: Optional[Any] = None, + pixel_size_um: Optional[float] = None, + intensity_threshold: float = 0.0, + otsu_max_rois: Optional[int] = None, + save_roi_tx_table: bool = True, + write_snr_json: bool = True, + snr_include_moran: bool = False, + roi_grid_stride: Optional[Tuple[int, int]] = None, + snr_thresholds: Optional[Dict[str, Any]] = None, +) -> Tuple[pd.DataFrame, Dict[str, Any]]: + """High-level API for :mod:`image_qc` — same as :func:`run_snr_module` with keyword-only opts.""" + return run_snr_module( + bundle_dir, + df_grid_roi, + outdir, + focus_maps=focus_maps, + roi_snr_db=roi_snr_db, + intensity_threshold=intensity_threshold, + pixel_size_um=pixel_size_um, + otsu_max_rois=otsu_max_rois, + save_roi_tx_table=save_roi_tx_table, + write_snr_json=write_snr_json, + snr_include_moran=snr_include_moran, + roi_grid_stride=roi_grid_stride, + snr_thresholds=snr_thresholds, + ) + + +__all__ = [ + "SNR_CKEY_IMAGE_OTSU", + "SNR_CKEY_IMAGE_ROI_QUARTILE_DB", + "SNR_CKEY_ROI_NEG_SPATIAL", + "SNR_CKEY_ROI_TX", + "SNR_CKEY_SLIDE_PLUMMER", + "SNR_CKEY_SLIDE_SPATIALQM", + "SNR_ROI_TX_TABLE_BASENAME", + "aggregate_snr_verdict", + "compute_image_snr_from_pixel_maps", + "compute_image_snr_from_roi_df", + "compute_neg_spatial_autocorrelation", + "compute_roi_snr", + "compute_snr_summary", + "compute_slide_snr_plummer_corrected", + "compute_slide_snr_spatialqm_corrected", + "is_neg_probe_feature", + "is_excluded_feature", + "load_expression_matrix_h5", + "load_transcripts", + "read_xenium_pixel_size_um", + "run_snr_module", + "save_snr_roi_tx_table", + "snr_verdict_to_quality_status", + "transcripts_um_to_px", +] diff --git a/bin/tests/conftest.py b/bin/tests/conftest.py new file mode 100644 index 00000000..34986089 --- /dev/null +++ b/bin/tests/conftest.py @@ -0,0 +1,39 @@ +"""Import paths for the pipeline scripts under test. + +Ported from nf-xenium-processing `tests/conftest.py` (dev HEAD 5e35cae). + +The pipeline's Python lives in the pipeline-level `bin/` directory (staged into +the container by Nextflow), not in an installed package, so tests import it by +path. Each test module used to insert those paths itself, which made collection +order load-bearing: a module that inserted only a partial path list imported +`image_qc` successfully *only* when some other module had already inserted the +real directory. Run alone -- or given its own pytest-xdist worker -- it failed. +conftest.py is imported before any test module, so putting the paths here makes +every module importable in isolation. + +ADAPTED FROM UPSTREAM: upstream keeps each script in +`modules/local//resources/usr/bin/` plus a `bin/xenium_helpers/src` +package, so `_MODULE_BIN_DIRS` listed four directories. This pipeline ships all +QC scripts in the single pipeline-level `bin/` and does not ship +`xenium_helpers` at all (its helpers are inlined into the scripts), so the list +collapses to one entry. + +Stubs for heavy optional imports stay in the individual test modules, since they +differ between modules. +""" + +from __future__ import annotations + +import sys +from pathlib import Path + +# This file lives in `bin/tests/`, so parent.parent is the pipeline `bin/`. +_bin = Path(__file__).resolve().parent.parent + +_MODULE_BIN_DIRS = [ + _bin, +] + +for _path in _MODULE_BIN_DIRS: + if str(_path) not in sys.path: + sys.path.insert(0, str(_path)) diff --git a/bin/tests/environment.yml b/bin/tests/environment.yml new file mode 100644 index 00000000..a0cffbb5 --- /dev/null +++ b/bin/tests/environment.yml @@ -0,0 +1,45 @@ +--- +# yaml-language-server: $schema=https://raw.githubusercontent.com/nf-core/modules/master/modules/environment-schema.json +# Environment for running the QC Python test suite (bin/tests/). +# +# This is a self-contained env rather than a merge of the two QC module +# environments: those two cannot be co-installed (image QC pins zarr 2.x while +# scanpy pulls zarr 3.x, and image QC also carries a pip section), so CI needs +# one coherent set. +# +# It must cover everything bin/image_qc.py and bin/transcript_qc_processing.py +# import at module scope, because both are imported during collection. The +# napari plugins and snr_metrics are stubbed by the test modules themselves and +# are deliberately absent here. +# +# Only python and numpy are pinned to the exact module-environment versions; +# the rest float so the solver can find a coherent set. The tests assert pure +# Python logic (chunked accumulation, threshold scaling, metric parsing), so +# they are not sensitive to patch versions of the scientific stack. +name: qc-tests +channels: + - conda-forge + - bioconda +dependencies: + - conda-forge::python=3.11.0 + - conda-forge::numpy=2.3.3 + # image_qc.py module-scope imports + - conda-forge::pandas + - conda-forge::scipy + - conda-forge::matplotlib + - conda-forge::seaborn + - conda-forge::scikit-image + - conda-forge::scikit-learn + - conda-forge::tifffile + - conda-forge::zarr + - conda-forge::numba + - conda-forge::click + - conda-forge::pyyaml + # transcript_qc_processing.py / transcript_stream.py module-scope imports + - conda-forge::pyarrow + - conda-forge::scanpy + - conda-forge::anndata + - conda-forge::h5py + # test runner + - conda-forge::pytest + - conda-forge::pytest-xdist diff --git a/bin/tests/helpers_memmap_worker.py b/bin/tests/helpers_memmap_worker.py new file mode 100644 index 00000000..6f464d57 --- /dev/null +++ b/bin/tests/helpers_memmap_worker.py @@ -0,0 +1,34 @@ +"""Worker helpers for the cross-process plane-sharing test. + +Lives in its own module because ``multiprocessing`` with the ``spawn`` start +method re-imports the function's module in the child, and a pytest test module +is a fragile thing to re-import. This mirrors the production arrangement in +``image_qc._tile_worker_init`` / ``_tile_worker_run``: the child receives a +*path*, not an array, and maps it itself. +""" + +from __future__ import annotations + +from typing import Any + +import numpy as np + +_STATE: dict[str, Any] = {} + + +def init(plane_path: str, shape: tuple[int, int], dtype_str: str) -> None: + """Map the shared plane file into this worker process.""" + _STATE["plane"] = np.memmap( + plane_path, dtype=np.dtype(dtype_str), mode="r+", shape=tuple(shape) + ) + + +def write_tile(task: tuple[dict[str, int], float]) -> float: + """Write *value* into this tile's disjoint write region and flush.""" + spec, value = task + plane = _STATE["plane"] + plane[spec["write_y0"] : spec["write_y1"], spec["write_x0"] : spec["write_x1"]] = ( + value + ) + plane.flush() + return value diff --git a/bin/tests/test_cell_roi_mapping.py b/bin/tests/test_cell_roi_mapping.py new file mode 100644 index 00000000..247ca0cb --- /dev/null +++ b/bin/tests/test_cell_roi_mapping.py @@ -0,0 +1,189 @@ +"""Unit tests for the cell -> grid-tile mapping (map_grid_roi_to_cells). + +`map_grid_roi_to_cells` has two implementations of the same operation: + +* a fast vectorised path, taken when the grid is non-overlapping and complete, + which converts a centroid to a grid index arithmetically; and +* a slow containment path (`roi_x1 <= x < roi_x2`), taken otherwise, which is the + reference definition of "which tile is this cell in". + +The fast path used `np.round` on centroids where it must use `np.floor`. Rounding +is correct for the tile *edges* used to populate the lookup grid (they are exact +multiples of the stride) but wrong for centroids, which fall anywhere inside a +tile: every cell past the half-stride mark was pushed into the next tile. +Measured on four calibration samples, that misassigned 73.5-73.9% of cells and +inflated `pct_blurred_gmm_2d_roi` by 2x to 14x while depressing +`ccfs_gmm_agreement_pct` by 10-19 points. See +`docs/plans/2026-08-24_SPIKE_tile-mapping-impact.md`. + +These tests assert the two paths agree rather than pinning tile IDs the tests +themselves compute, so they cannot enshrine the same off-by-half error, and they +fail loudly if `round` is ever reintroduced. + +`overlapping=True` is what forces the slow path; the two calls are otherwise +identical. +""" + +from __future__ import annotations + +import importlib +import sys +import types + +import numpy as np +import pandas as pd +import pytest + +# image_qc.py has heavy top-level imports (napari, snr_metrics, etc.) not needed +# for the pure function under test. Stub them before importing, matching +# test_cluster_outliers.py. Import paths come from tests/conftest.py. +_stubs = [ + "napari_skimage_regionprops", + "napari_simpleitk_image_processing", + "snr_metrics", + "scanpy", +] +for _mod in _stubs: + if _mod not in sys.modules: + sys.modules[_mod] = types.ModuleType(_mod) +sys.modules["napari_skimage_regionprops"].regionprops_table = lambda *a, **kw: None # type: ignore[attr-defined] +sys.modules["scanpy"].AnnData = object # type: ignore[attr-defined] + +image_qc = importlib.import_module("image_qc") +map_grid_roi_to_cells = image_qc.map_grid_roi_to_cells + +STRIDE = 35 + + +def _grid(n_cols: int, n_rows: int, stride: int = STRIDE) -> pd.DataFrame: + """A complete non-overlapping tile grid, one row per tile.""" + rows = [] + for r in range(n_rows): + for c in range(n_cols): + rows.append( + { + "roi_id": r * n_cols + c, + "x1": c * stride, + "x2": (c + 1) * stride, + "y1": r * stride, + "y2": (r + 1) * stride, + # columns the function joins onto cells; distinct per tile so a + # misassignment is visible, not masked by equal values + "focus_score": float(r * n_cols + c), + "focus_score_norm": float(r * n_cols + c) / 100.0, + "raw_intensity": 1000.0 + (r * n_cols + c), + "tissue_coverage": ((r * n_cols + c) % 11) / 10.0, + } + ) + return pd.DataFrame(rows) + + +def _cells(n_cols: int, n_rows: int, fracs, stride: int = STRIDE) -> pd.DataFrame: + """One cell per (tile, fraction) pair, placed at `frac` of the way into the tile.""" + rows = [] + cid = 0 + for r in range(n_rows): + for c in range(n_cols): + for fx in fracs: + for fy in fracs: + rows.append( + { + "cell_id": f"c{cid}", + "x": c * stride + fx * stride, + "y": r * stride + fy * stride, + } + ) + cid += 1 + return pd.DataFrame(rows) + + +def _map(grid, cells, *, slow: bool): + return map_grid_roi_to_cells(grid, cells, overlapping=slow) + + +# The three fractions that matter: below the halfway point (round and floor agree), +# exactly at it, and above it (round and floor disagree). +_FRACS = (0.25, 0.5, 0.75) + + +def test_fast_path_matches_containment_path(): + """The vectorised lookup must agree with `roi_x1 <= x < roi_x2` exactly.""" + grid = _grid(6, 5) + cells = _cells(6, 5, _FRACS) + + fast = _map(grid, cells, slow=False) + slow = _map(grid, cells, slow=True) + + assert len(fast) == len(cells) + pd.testing.assert_series_equal( + fast["roi_id"].reset_index(drop=True), + slow["roi_id"].reset_index(drop=True), + check_names=False, + ) + + +@pytest.mark.parametrize("frac", _FRACS) +def test_each_offset_within_a_tile_maps_to_that_tile(frac): + """A cell anywhere inside a tile belongs to that tile, not its neighbour. + + Parametrised so a failure names the offending offset. `round` fails at 0.5 and + 0.75; `floor` passes all three. + """ + grid = _grid(4, 4) + cells = _cells(4, 4, (frac,)) + + fast = _map(grid, cells, slow=False) + + # the tile each cell sits in, computed from the tile bounds rather than from + # the same arithmetic the function under test uses + expected = [] + for _, cell in cells.iterrows(): + hit = grid[ + (grid["x1"] <= cell["x"]) + & (cell["x"] < grid["x2"]) + & (grid["y1"] <= cell["y"]) + & (cell["y"] < grid["y2"]) + ] + expected.append(hit["roi_id"].iloc[0]) + + assert list(fast["roi_id"]) == expected + + +def test_joined_tile_attributes_follow_the_correct_tile(): + """The joined columns, not just roi_id, must come from the containing tile.""" + grid = _grid(5, 5) + cells = _cells(5, 5, _FRACS) + + fast = _map(grid, cells, slow=False) + slow = _map(grid, cells, slow=True) + + for col in ("roi_tissue_coverage", "DAPI_RFS_roi"): + if col in fast.columns and col in slow.columns: + np.testing.assert_allclose( + fast[col].to_numpy(dtype=float), + slow[col].to_numpy(dtype=float), + equal_nan=True, + ) + + +def test_cells_on_the_far_edge_are_not_pushed_off_the_grid(): + """A cell just inside the last tile stays in it rather than being clipped.""" + n = 4 + grid = _grid(n, n) + last = grid["roi_id"].max() + eps = 0.01 + cells = pd.DataFrame( + [ + { + "cell_id": "edge", + "x": n * STRIDE - eps, + "y": n * STRIDE - eps, + } + ] + ) + + fast = _map(grid, cells, slow=False) + slow = _map(grid, cells, slow=True) + + assert fast["roi_id"].iloc[0] == last + assert slow["roi_id"].iloc[0] == last diff --git a/bin/tests/test_dense_intensity_mask.py b/bin/tests/test_dense_intensity_mask.py new file mode 100644 index 00000000..0fa2bcb2 --- /dev/null +++ b/bin/tests/test_dense_intensity_mask.py @@ -0,0 +1,182 @@ +"""Regression test for the artefact ("optically dense regions") mask reduction. + +`generate_tissue_mask` finds bright artefacts by thresholding each morphology +channel at the 97th percentile and combining the results. A pixel bright in two or +three channels is the strongest evidence of a real artefact, because genuine signal +is stain-specific while debris, folds and coverslip contamination are not. + +From 2025-06-06 to 2026-05-06 the combination added three *bool* arrays. NumPy `+` +on bool dtype is logical OR and returns bool, so `binary_fill_holes` received a +genuine any-channel union. `119cff9` (2026-05-06, "support DAPI-only morphology +bundles") added `.astype(np.int_)` to each term while making the block None-tolerant, +turning the union into a 0..3 count. `SimpleITK.BinaryFillhole` treats only value 1 +as foreground, so the 2s and 3s -- the strongest evidence -- stopped being +foreground. + +No literal `> 0` ever existed: the reduction rode on bool dtype alone, which is why +the change looked like tidying and went unnoticed for three months. These tests +assert the *contract* at the boundary (what reaches the fill filter) rather than the +filter's output, because `napari_simpleitk_image_processing` is not installed in the +test environment and the existing `test_generate_tissue_mask.py` stubs the filter as +identity -- which is precisely why its tests could never observe this. +""" + +from __future__ import annotations + +import importlib +import sys +import types +from pathlib import Path + +import numpy as np +import pytest + +_bin_dir = Path(__file__).resolve().parent.parent +if str(_bin_dir) not in sys.path: + sys.path.insert(0, str(_bin_dir)) + +# Stub the heavy optional imports image_qc pulls in at module level. scanpy is +# stubbed for the same reason test_cell_roi_mapping.py does it: importing it here +# trips numba's cache locator and aborts collection. Stubbing in this module rather +# than relying on another test file having done it keeps the file runnable alone. +for _mod in ( + "napari_skimage_regionprops", + "napari_simpleitk_image_processing", + "snr_metrics", + "scanpy", +): + if _mod not in sys.modules: + sys.modules[_mod] = types.ModuleType(_mod) +sys.modules["napari_skimage_regionprops"].regionprops_table = lambda *a, **kw: None # type: ignore[attr-defined] +sys.modules["scanpy"].AnnData = object # type: ignore[attr-defined] +_nsitk = sys.modules["napari_simpleitk_image_processing"] +_nsitk.signed_maurer_distance_map = lambda x: np.zeros_like( # type: ignore[attr-defined] + np.asarray(x), dtype=np.float64 +) +# Define a default so monkeypatch.setattr has an attribute to replace. Each test +# swaps in the spy; this identity stand-in matches test_generate_tissue_mask.py. +_nsitk.binary_fill_holes = lambda x: np.asarray(x) # type: ignore[attr-defined] + +image_qc = importlib.import_module("image_qc") + +KW = dict(min_size_edge=10, min_size_hole=5, min_size_dense_intensity_region=5) +SHAPE = (64, 64) + + +class _FillSpy: + """Capture what the artefact mask hands to binary_fill_holes. + + generate_tissue_mask calls the filter more than once; the artefact call is the + one whose argument is not the tissue mask. Recording every call and inspecting + the last is enough, since the artefact block is the final caller. + """ + + def __init__(self): + self.calls: list[np.ndarray] = [] + + def __call__(self, x): + arr = np.asarray(x) + self.calls.append(arr.copy()) + return arr + + +@pytest.fixture +def fill_spy(monkeypatch): + spy = _FillSpy() + monkeypatch.setattr(_nsitk, "binary_fill_holes", spy) + return spy + + +RNG = np.random.default_rng(0) + + +def _channels(n_bright_channels: int): + """Three noisy channels; a 10x10 patch is bright in the first N of them. + + The background must be textured, NOT flat. A flat field's 97th percentile + equals its own value, so `channel >= p97` is True everywhere and every + channel contributes to the count regardless of the patch -- which silently + turns "bright in 1 of 3" into "bright in 3 of 3" and makes the parametrisation + below meaningless. With noise, a channel that lacks the patch has an ordinary + value there and contributes 0, so the counts really are 1, 2 and 3. + """ + chans = [] + for i in range(3): + f = RNG.normal(100.0, 30.0, size=SHAPE).clip(0) + if i < n_bright_channels: + f[20:30, 20:30] = 50_000.0 + chans.append(f) + return chans + + +def _artefact_arg(spy: _FillSpy) -> np.ndarray: + assert spy.calls, "binary_fill_holes was never called" + return spy.calls[-1] + + +@pytest.mark.parametrize("n_bright", [1, 2, 3]) +def test_multi_channel_bright_pixels_reach_the_fill_filter(fill_spy, n_bright): + """The patch must be foreground whether it is bright in 1, 2 or 3 channels. + + Under the count-into-a-binary-filter bug, only n_bright == 1 survived. + """ + c0, c1, c2 = _channels(n_bright) + image_qc.generate_tissue_mask(None, c0, c1, c2, **KW) + + arg = _artefact_arg(fill_spy) + assert arg.dtype == bool, ( + f"the fill filter received {arg.dtype}, not bool. A binary filter treats " + "only one value as foreground, so an integer channel count silently drops " + "the multi-channel-bright pixels." + ) + assert arg[20:30, 20:30].all(), ( + f"a patch bright in {n_bright} of 3 channels is not foreground in the " + "artefact mask" + ) + + +def test_the_reduction_is_any_channel_not_a_count(fill_spy): + """Two disjoint patches, bright in different numbers of channels, both count.""" + c0, c1, c2 = (RNG.normal(100.0, 30.0, size=SHAPE).clip(0) for _ in range(3)) + # patch A: bright in all three + c0[10:16, 10:16] = c1[10:16, 10:16] = c2[10:16, 10:16] = 50_000.0 + # patch B: bright in DAPI only + c0[40:46, 40:46] = 50_000.0 + + image_qc.generate_tissue_mask(None, c0, c1, c2, **KW) + arg = _artefact_arg(fill_spy) + + assert arg[10:16, 10:16].all(), "3-channel-bright patch dropped" + assert arg[40:46, 40:46].all(), "1-channel-bright patch dropped" + assert set(np.unique(arg)) <= {False, True} + + +def test_dapi_only_bundle_still_works(fill_spy): + """small1/small2 are None for DAPI-only bundles; the rule must still apply. + + A `>= 2` rule would make the artefact metric structurally impossible here, + which is one reason the any-channel rule is the correct restoration. + """ + c0 = RNG.normal(100.0, 30.0, size=SHAPE).clip(0) + c0[20:30, 20:30] = 50_000.0 + + image_qc.generate_tissue_mask(None, c0, None, None, **KW) + arg = _artefact_arg(fill_spy) + + assert arg.dtype == bool + assert arg[20:30, 20:30].all() + + +def test_uniform_field_yields_no_artefact_foreground(fill_spy): + """A flat field has no 97th-percentile outlier region worth flagging. + + Guards the other direction: `> 0` on a count must not mark everything. A flat + field's 97th percentile equals its value, so `>=` marks all of it -- this test + pins that this is a property of the percentile threshold, not of the reduction, + by asserting the mask is uniform rather than patchy. + """ + c0 = np.full(SHAPE, 100.0) + image_qc.generate_tissue_mask(None, c0, None, None, **KW) + arg = _artefact_arg(fill_spy) + assert arg.dtype == bool + assert arg.all() or not arg.any(), "flat field produced a patchy artefact mask" diff --git a/bin/tests/test_excluded_features.py b/bin/tests/test_excluded_features.py new file mode 100644 index 00000000..3506f3af --- /dev/null +++ b/bin/tests/test_excluded_features.py @@ -0,0 +1,199 @@ +"""Tests for dropping decoding failures from the transcript SNR ratio. + +Xenium assigns each detected spot to a gene by reading its barcode. Some reads +fail: `UnassignedCodeword_*` means the readout matched no valid barcode. That is a +failure to measure, not a detected transcript. + +The negative-control classifier matched on name prefix only and had no entry for it, +and `is_real` was defined as `~is_neg`, so every unassigned codeword was counted as +real gene signal. That inflated the numerator of the real/negative ratio and biased +the verdict toward PASS. + +Counting them as negative controls would be equally wrong: a negative-control probe +estimates background because it is designed not to bind, while a failed read +estimates nothing. So they are dropped from both sides, and from the total, which +keeps `neg_pct` a fraction of decoded transcripts. + +`DeprecatedCodeword_*` is deliberately NOT excluded: retired-but-valid barcodes may +be genuine detections of a gene no longer in the panel. Its behaviour is pinned below +so a later decision to exclude it is a visible change rather than a silent one. +""" + +from __future__ import annotations + +import importlib +import importlib.util +from pathlib import Path + +import numpy as np +import pandas as pd + +# snr_metrics is stubbed by other test modules (they share sys.modules), so a plain +# import can hand back an empty ModuleType depending on collection order. Load the +# real module by path, the same way test_tile_consumers.py does, so this file works +# whatever ran first. +_snr_path = ( + Path(__file__).resolve().parent.parent / "snr_metrics.py" +) +_spec = importlib.util.spec_from_file_location("_real_snr_metrics", _snr_path) +assert _spec is not None and _spec.loader is not None # narrow for mypy; path is checked above +snr_metrics = importlib.util.module_from_spec(_spec) +_spec.loader.exec_module(snr_metrics) + + +# --------------------------------------------------------------------------- +# classification +# --------------------------------------------------------------------------- + + +def test_unassigned_codeword_is_excluded(): + assert snr_metrics.is_excluded_feature("UnassignedCodeword_0123") + assert snr_metrics.is_excluded_feature("unassignedcodeword_0123"), ( + "case-insensitive" + ) + + +def test_unassigned_codeword_is_not_a_negative_control(): + """It is not background, so it must not join the neg-control pool.""" + assert not snr_metrics.is_neg_probe_feature("UnassignedCodeword_0123") + + +def test_real_genes_and_neg_controls_are_not_excluded(): + for name in ("EPCAM", "NegControlProbe_00042", "BLANK_0097", "antisense_PROX1"): + assert not snr_metrics.is_excluded_feature(name), name + + +def test_deprecated_codeword_is_left_alone_for_now(): + """Pins the open question: still counted as real, deliberately.""" + assert not snr_metrics.is_excluded_feature("DeprecatedCodeword_0001") + assert not snr_metrics.is_neg_probe_feature("DeprecatedCodeword_0001") + + +def test_empty_and_non_string_inputs(): + for bad in ("", None, 3.5, float("nan")): + assert not snr_metrics.is_excluded_feature(bad) + + +# --------------------------------------------------------------------------- +# vectorised masks must agree with the scalar predicate +# --------------------------------------------------------------------------- + +_NAMES = [ + "EPCAM", + "UnassignedCodeword_0001", + "NegControlProbe_00042", + "DeprecatedCodeword_0002", + "BLANK_0097", + "unassignedcodeword_0003", + "antisense_PROX1", +] + + +def test_vectorised_excluded_mask_matches_the_scalar_predicate(): + got = snr_metrics._excluded_mask_vectorized(np.array(_NAMES, dtype=object)) + want = np.array([snr_metrics.is_excluded_feature(n) for n in _NAMES]) + np.testing.assert_array_equal(got, want) + + +def test_dictionary_encoded_column_gives_the_same_mask(): + """Xenium writes feature_name dictionary-encoded; the fast path must agree.""" + plain = pd.Series(_NAMES * 3, dtype="object") + encoded = plain.astype("category") + + np.testing.assert_array_equal( + snr_metrics._excluded_mask_for_column(plain), + snr_metrics._excluded_mask_for_column(encoded), + ) + np.testing.assert_array_equal( + snr_metrics._neg_mask_for_column(plain), + snr_metrics._neg_mask_for_column(encoded), + ) + + +def test_the_three_categories_are_disjoint_and_cover_everything(): + neg = snr_metrics._neg_mask_for_column(pd.Series(_NAMES)) + exc = snr_metrics._excluded_mask_for_column(pd.Series(_NAMES)) + assert not (neg & exc).any(), "a feature cannot be both background and excluded" + + +# --------------------------------------------------------------------------- +# the counts themselves +# --------------------------------------------------------------------------- + + +def _one_tile_frame(): + """A single tile covering everything, so every transcript lands in it.""" + return pd.DataFrame( + {"roi_id": [0], "x1": [0.0], "x2": [100.0], "y1": [0.0], "y2": [100.0]} + ) + + +def _count(names: list[str]): + df = _one_tile_frame() + row_ix = pd.Series(np.arange(1, dtype=np.int32), index=df["roi_id"].values) + counters = ( + np.zeros(1, dtype=np.int64), + np.zeros(1, dtype=np.int64), + np.zeros(1, dtype=np.int64), + ) + col = pd.Series(names) + snr_metrics._accumulate_roi_tx_counts( + df, + row_ix, + np.full(len(names), 50.0), + np.full(len(names), 50.0), + snr_metrics._neg_mask_for_column(col), + counters, + None, + roi_id_is_arange=True, + is_excluded=snr_metrics._excluded_mask_for_column(col), + ) + real, neg, total = (int(c[0]) for c in counters) + return real, neg, total + + +def test_unassigned_codewords_leave_all_three_counts(): + """5 genes, 2 neg controls, 3 unassigned: the 3 vanish from every count.""" + names = ( + ["EPCAM"] * 5 + ["NegControlProbe_00042"] * 2 + ["UnassignedCodeword_0001"] * 3 + ) + real, neg, total = _count(names) + + assert real == 5, "unassigned codewords are still being counted as real signal" + assert neg == 2, "unassigned codewords must not join the negative controls" + assert total == 7, "excluded features must leave the total too" + assert real + neg == total, "real = total - neg identity broken" + + +def test_ratio_is_not_inflated_by_decoding_failures(): + """The point of the fix: adding unassigned codewords must not change the ratio.""" + clean = _count(["EPCAM"] * 20 + ["NegControlProbe_00042"] * 4) + dirty = _count( + ["EPCAM"] * 20 + + ["NegControlProbe_00042"] * 4 + + ["UnassignedCodeword_0001"] * 50 + ) + assert clean == dirty, ( + "50 failed barcode reads changed the counts, so they are still reaching " + f"the ratio: {clean} vs {dirty}" + ) + + +def test_deprecated_codewords_still_count_as_real(): + """Pins current behaviour so changing it later is deliberate.""" + real, neg, total = _count(["EPCAM"] * 3 + ["DeprecatedCodeword_0002"] * 2) + assert (real, neg, total) == (5, 0, 5) + + +def test_all_transcripts_excluded_leaves_empty_counts(): + real, neg, total = _count(["UnassignedCodeword_0001"] * 10) + assert (real, neg, total) == (0, 0, 0) + + +def test_slide_masks_exclude_decoding_failures(): + is_real, is_neg = snr_metrics._build_feature_masks(_NAMES) + for i, name in enumerate(_NAMES): + if snr_metrics.is_excluded_feature(name): + assert not is_real[i], f"{name} counted as real signal" + assert not is_neg[i], f"{name} counted as background" + assert not (is_real & is_neg).any() diff --git a/bin/tests/test_image_qc_memory_bounded.py b/bin/tests/test_image_qc_memory_bounded.py new file mode 100644 index 00000000..2b46a074 --- /dev/null +++ b/bin/tests/test_image_qc_memory_bounded.py @@ -0,0 +1,787 @@ +"""Tests for the bounded-host-memory IMAGE_QC paths. + +The full-resolution focus/mean/Laplacian planes are ~22 GB each on a 5.5 +gigapixel sample. Holding them in host RAM, and reducing over them with +whole-image ``scipy.ndimage`` calls, is what forced the 180 -> 720 GB retry +ladder. These tests pin the replacements: + +1. ``_labeled_sums_chunked`` reduces in row blocks and must agree exactly with + ``scipy.ndimage.mean`` / ``skimage.measure.regionprops``, including when a + label straddles a block boundary. +2. ``calculate_ccfs_from_focus_maps`` must reproduce the previous + ndimage.mean + regionprops + bincount algorithm value-for-value. +3. ``_PlaneStore`` must spill large planes to disk, expose them as ordinary + arrays, and refuse to start when the scratch filesystem is too small. +4. ``_figure_worker_limit`` must cap the forked figure fan-out. +""" + +from __future__ import annotations + +import importlib +import multiprocessing +import os +import sys +import time +import types +from pathlib import Path + +import numpy as np +import pytest +from scipy.ndimage import mean as ndimage_mean +from skimage.measure import regionprops + +# Same stub-then-import dance as upstream: image_qc.py has heavy top-level +# imports that the functions under test do not need. +# ADAPTED FROM UPSTREAM: upstream also inserted +# `modules/local/image_qc/resources/usr/bin`, which does not exist in this +# pipeline -- `image_qc.py` lives in the pipeline-level `bin/`. This file lives +# in `bin/tests/`, so parent.parent is that `bin/`. +_bin_dir = Path(__file__).resolve().parent.parent +sys.path.insert(0, str(_bin_dir)) +# spawn passes sys.path to children, so this makes the worker helper importable there +sys.path.insert(0, str(Path(__file__).resolve().parent)) + +for _mod in ( + "napari_skimage_regionprops", + "napari_simpleitk_image_processing", + "snr_metrics", + "scanpy", +): + if _mod not in sys.modules: + sys.modules[_mod] = types.ModuleType(_mod) +sys.modules["napari_skimage_regionprops"].regionprops_table = lambda *a, **kw: None # type: ignore[attr-defined] +sys.modules["scanpy"].AnnData = object # type: ignore[attr-defined] + +image_qc = importlib.import_module("image_qc") +import helpers_memmap_worker # noqa: E402 (needs the sys.path insert above) + +# /proc is Linux-only. Tests that read it directly -- via _process_rss, +# _tree_rss, or os.listdir("/proc") -- get None or FileNotFoundError off-Linux, +# so a Mac reviewer would otherwise see spurious red. Skip them there; the +# cgroup tests below use monkeypatched sources and run everywhere. +requires_proc = pytest.mark.skipif( + not sys.platform.startswith("linux"), + reason="reads /proc, which exists only on Linux", +) + + +# --------------------------------------------------------------------------- +# Helpers +# --------------------------------------------------------------------------- + + +class _RowSliceOnly: + """Array handle that only supports ``.shape`` and ``[y0:y1]`` row slicing. + + Stands in for a zarr array or memmap: if ``_labeled_sums_chunked`` ever + materialised the whole plane (``np.asarray(handle)``) this would fail, so the + test proves the reduction really is block-wise. + """ + + def __init__(self, array: np.ndarray) -> None: + self._array = array + self.shape = array.shape + + def __getitem__(self, key): + if not isinstance(key, slice): + raise TypeError(f"expected a row slice, got {key!r}") + return self._array[key] + + +class _MasksGroup: + """Minimal stand-in for the zarr group passed as ``cell_masks_zarr``.""" + + def __init__(self, nuclear: np.ndarray, cell: np.ndarray) -> None: + self._masks = {"0": nuclear, "1": cell} + + def get(self, name): + if name == "masks": + return self + return self._masks[name] + + +@pytest.fixture +def labelled_scene(): + """Deterministic label planes plus aligned value planes. + + Labels are non-contiguous and deliberately span many rows so that any + row-block size splits several of them. + """ + rng = np.random.RandomState(1234) + height, width = 64, 48 + nuclear = np.zeros((height, width), dtype=np.int32) + # Tall blobs that straddle row blocks; label values are non-consecutive. + for idx, label in enumerate((3, 7, 8, 15, 42)): + y0 = idx * 12 + nuclear[y0 : y0 + 14, idx * 9 : idx * 9 + 8] = label + + cell = np.zeros((height, width), dtype=np.int32) + for idx, label in enumerate((5, 9, 11, 20, 41)): + y0 = idx * 12 + cell[y0 : y0 + 16, idx * 9 : idx * 9 + 10] = label + + focus = rng.rand(height, width).astype(np.float32) * 100.0 + mean = rng.rand(height, width).astype(np.float32) * 5000.0 + boundary = rng.rand(height, width).astype(np.float32) * 900.0 + intrna = rng.rand(height, width).astype(np.float32) * 700.0 + return nuclear, cell, focus, mean, boundary, intrna + + +def _reference_ccfs(focus_maps, nuclear_mask, cellseg_mask): + """The pre-change algorithm: np.unique + ndimage.mean + regionprops + bincount.""" + import pandas as pd + + labels = np.unique(nuclear_mask) + labels = labels[labels > 0] + focus_map = focus_maps["dapi_focus_map"] + mean_map = focus_maps["dapi_mean_map"] + + cell_focus = ndimage_mean(focus_map, nuclear_mask, labels) + cell_intensity = ndimage_mean(mean_map, nuclear_mask, labels) + + dapi_norm = np.percentile(cell_intensity, 99) + if dapi_norm == 0: + dapi_norm = 1.0 + ccfs_dapi = np.asarray(cell_focus) / dapi_norm + + props = regionprops(nuclear_mask) + label_to_ccfs = dict(zip(labels, ccfs_dapi)) + label_to_intensity = dict(zip(labels, cell_intensity)) + + rows = [] + for p in props: + cy, cx = p.centroid + iy = min(int(cy), cellseg_mask.shape[0] - 1) + ix = min(int(cx), cellseg_mask.shape[1] - 1) + rows.append( + { + "label": p.label, + "centroid-0": cy, + "centroid-1": cx, + "area_nucleus": p.area, + "CCFS_DAPI": label_to_ccfs.get(p.label, np.nan), + "mean_intensity": label_to_intensity.get(p.label, np.nan), + "CellID": int(cellseg_mask[iy, ix]), + } + ) + out = pd.DataFrame(rows) + + cell_areas = np.bincount(cellseg_mask.ravel()) + out["area_cell"] = out["CellID"].map( + lambda cid: int(cell_areas[cid]) if cid < len(cell_areas) else 0 + ) + + for key, column in ( + ("boundary_mean_map", "mean_intensity_Boundary"), + ("intrna_mean_map", "mean_intensity_IntRNA"), + ): + plane = focus_maps.get(key) + if plane is None: + out[column] = np.nan + continue + uniq = out["CellID"].unique() + uniq = uniq[uniq > 0] + per_cell = ndimage_mean(plane, cellseg_mask, uniq) + out[column] = out["CellID"].map(dict(zip(uniq, per_cell))) + + return out + + +# --------------------------------------------------------------------------- +# _labeled_sums_chunked +# --------------------------------------------------------------------------- + + +class TestLabeledSumsChunked: + """Row-blocked per-label reduction must equal the whole-image reduction.""" + + def test_mean_matches_scipy_ndimage(self, labelled_scene): + nuclear, _, focus, mean, _, _ = labelled_scene + counts, sums = image_qc._labeled_sums_chunked( + nuclear, {"focus": focus, "intensity": mean} + ) + + labels = np.nonzero(counts)[0] + labels = labels[labels > 0] + got_focus = sums["focus"][labels] / counts[labels] + got_mean = sums["intensity"][labels] / counts[labels] + + expected_focus = ndimage_mean(focus, nuclear, labels) + expected_mean = ndimage_mean(mean, nuclear, labels) + + np.testing.assert_allclose(got_focus, expected_focus, rtol=1e-12, atol=0) + np.testing.assert_allclose(got_mean, expected_mean, rtol=1e-12, atol=0) + + def test_label_set_matches_np_unique(self, labelled_scene): + nuclear = labelled_scene[0] + counts, _ = image_qc._labeled_sums_chunked(nuclear, {}) + labels = np.nonzero(counts)[0] + labels = labels[labels > 0] + expected = np.unique(nuclear) + np.testing.assert_array_equal(labels, expected[expected > 0]) + + def test_counts_match_regionprops_area(self, labelled_scene): + nuclear = labelled_scene[0] + counts, _ = image_qc._labeled_sums_chunked(nuclear, {}) + for p in regionprops(nuclear): + assert counts[p.label] == p.area, f"area mismatch for label {p.label}" + + def test_centroids_match_regionprops(self, labelled_scene): + nuclear = labelled_scene[0] + counts, sums = image_qc._labeled_sums_chunked(nuclear, {}, include_coords=True) + for p in regionprops(nuclear): + cy = sums["centroid_y_sum"][p.label] / counts[p.label] + cx = sums["centroid_x_sum"][p.label] / counts[p.label] + np.testing.assert_allclose((cy, cx), p.centroid, rtol=1e-12, atol=0) + + @pytest.mark.parametrize("rows_per_chunk", [1, 2, 3, 7, 13, 64, 4096]) + def test_block_size_does_not_change_result(self, labelled_scene, rows_per_chunk): + """Every label straddles a boundary at some block size; sums stay additive.""" + nuclear, _, focus, _, _, _ = labelled_scene + single = image_qc._labeled_sums_chunked( + nuclear, {"focus": focus}, include_coords=True, rows_per_chunk=10_000 + ) + blocked = image_qc._labeled_sums_chunked( + nuclear, + {"focus": focus}, + include_coords=True, + rows_per_chunk=rows_per_chunk, + ) + np.testing.assert_array_equal(single[0], blocked[0]) + for key in single[1]: + np.testing.assert_allclose( + single[1][key], blocked[1][key], rtol=1e-12, atol=0 + ) + + def test_accepts_row_sliceable_handle(self, labelled_scene): + """Must never materialise the whole plane — proves zarr/memmap support.""" + nuclear, _, focus, _, _, _ = labelled_scene + handle = _RowSliceOnly(nuclear) + counts, sums = image_qc._labeled_sums_chunked( + handle, {"focus": _RowSliceOnly(focus)}, rows_per_chunk=5 + ) + expected_counts, expected_sums = image_qc._labeled_sums_chunked( + nuclear, {"focus": focus} + ) + np.testing.assert_array_equal(counts, expected_counts) + np.testing.assert_allclose( + sums["focus"], expected_sums["focus"], rtol=1e-12, atol=0 + ) + + def test_empty_value_planes_gives_counts_only(self, labelled_scene): + nuclear = labelled_scene[0] + counts, sums = image_qc._labeled_sums_chunked(nuclear, {}) + assert sums == {} + assert counts.sum() == nuclear.size + + +# --------------------------------------------------------------------------- +# calculate_ccfs_from_focus_maps +# --------------------------------------------------------------------------- + + +class TestCalculateCcfsFromFocusMaps: + """The blocked rewrite must reproduce the previous algorithm exactly.""" + + def _run(self, labelled_scene): + nuclear, cell, focus, mean, boundary, intrna = labelled_scene + focus_maps = { + "dapi_focus_map": focus, + "dapi_mean_map": mean, + "boundary_mean_map": boundary, + "intrna_mean_map": intrna, + } + got = image_qc.calculate_ccfs_from_focus_maps( + focus_maps, _MasksGroup(nuclear, cell), cell, [] + ) + expected = _reference_ccfs(focus_maps, nuclear, cell) + return got, expected + + def test_same_rows_and_labels(self, labelled_scene): + got, expected = self._run(labelled_scene) + assert len(got) == len(expected) + np.testing.assert_array_equal( + got["label"].to_numpy(), expected["label"].to_numpy() + ) + + @pytest.mark.parametrize( + "column", + [ + "centroid-0", + "centroid-1", + "CCFS_DAPI", + "mean_intensity", + "mean_intensity_Boundary", + "mean_intensity_IntRNA", + ], + ) + def test_float_columns_match_reference(self, labelled_scene, column): + got, expected = self._run(labelled_scene) + np.testing.assert_allclose( + got[column].to_numpy(dtype=np.float64), + expected[column].to_numpy(dtype=np.float64), + rtol=1e-12, + atol=0, + ) + + @pytest.mark.parametrize("column", ["area_nucleus", "area_cell", "CellID"]) + def test_integer_columns_match_reference(self, labelled_scene, column): + got, expected = self._run(labelled_scene) + np.testing.assert_array_equal( + got[column].to_numpy(dtype=np.int64), + expected[column].to_numpy(dtype=np.int64), + ) + + def test_missing_optional_channels_give_nan(self, labelled_scene): + nuclear, cell, focus, mean, _, _ = labelled_scene + got = image_qc.calculate_ccfs_from_focus_maps( + {"dapi_focus_map": focus, "dapi_mean_map": mean}, + _MasksGroup(nuclear, cell), + cell, + [], + ) + assert got["mean_intensity_Boundary"].isna().all() + assert got["mean_intensity_IntRNA"].isna().all() + + def test_requires_dapi_maps(self, labelled_scene): + nuclear, cell, _, mean, _, _ = labelled_scene + with pytest.raises(ValueError, match="dapi_focus_map"): + image_qc.calculate_ccfs_from_focus_maps( + {"dapi_mean_map": mean}, _MasksGroup(nuclear, cell), cell, [] + ) + + def test_reads_nuclear_mask_lazily(self, labelled_scene): + """The nuclear plane must be row-sliced, never materialised whole.""" + nuclear, cell, focus, mean, _, _ = labelled_scene + + class _LazyMasks: + def get(self, name): + if name == "masks": + return self + return _RowSliceOnly(nuclear) + + got = image_qc.calculate_ccfs_from_focus_maps( + {"dapi_focus_map": focus, "dapi_mean_map": mean}, + _LazyMasks(), + cell, + [], + ) + expected = _reference_ccfs( + {"dapi_focus_map": focus, "dapi_mean_map": mean}, nuclear, cell + ) + np.testing.assert_allclose( + got["CCFS_DAPI"].to_numpy(dtype=np.float64), + expected["CCFS_DAPI"].to_numpy(dtype=np.float64), + rtol=1e-12, + atol=0, + ) + + +# --------------------------------------------------------------------------- +# _PlaneStore +# --------------------------------------------------------------------------- + + +class TestPlaneStore: + """Large planes must land on disk; small ones may stay in RAM.""" + + def test_small_plane_stays_in_ram(self, tmp_path): + """Below the spill threshold a file would be pure overhead.""" + store = image_qc._PlaneStore((64, 64), ["focus_map"], plane_dir=tmp_path) + assert store.on_disk is False + assert store.descriptors() is None + assert not isinstance(store.arrays()["focus_map"], np.memmap) + + def test_defaults_to_cwd(self, tmp_path, monkeypatch): + """No plane_dir means the task working directory, per Nextflow convention.""" + monkeypatch.setattr(image_qc, "_PLANE_SPILL_BYTES", 1024) + monkeypatch.chdir(tmp_path) + store = image_qc._PlaneStore((128, 128), ["focus_map"], prefix="dapi") + assert store.on_disk is True + path = Path(store.descriptors()["focus_map"]) + assert path.parent == tmp_path.resolve() + + def test_large_plane_spills_to_disk(self, tmp_path, monkeypatch): + # Shrink the spill threshold rather than allocating a real 2 GB plane. + monkeypatch.setattr(image_qc, "_PLANE_SPILL_BYTES", 1024) + store = image_qc._PlaneStore( + (128, 128), ["focus_map", "mean_map"], plane_dir=tmp_path, prefix="dapi" + ) + assert store.on_disk is True + arrays = store.arrays() + assert isinstance(arrays["focus_map"], np.memmap) + descriptors = store.descriptors() + assert descriptors is not None + assert set(descriptors) == {"focus_map", "mean_map"} + for path in descriptors.values(): + assert Path(path).exists() + + def test_disk_plane_round_trips_through_the_file(self, tmp_path, monkeypatch): + """A second mapping of the same file sees the first mapping's writes. + + This is the property the process-per-GPU pool relies on: workers share + planes by filename, so no pixel data is ever pickled. + """ + monkeypatch.setattr(image_qc, "_PLANE_SPILL_BYTES", 1024) + store = image_qc._PlaneStore( + (32, 48), ["focus_map"], plane_dir=tmp_path, prefix="dapi" + ) + plane = store.arrays()["focus_map"] + plane[4:9, 6:11] = 3.5 + store.flush() + + path = store.descriptors()["focus_map"] + reopened = np.memmap(path, dtype=np.float32, mode="r", shape=(32, 48)) + assert reopened[4:9, 6:11].min() == pytest.approx(3.5) + assert reopened[0, 0] == pytest.approx(0.0) + + def test_insufficient_scratch_space_raises(self, tmp_path, monkeypatch): + monkeypatch.setattr(image_qc, "_PLANE_SPILL_BYTES", 1024) + + class _TinyUsage: + free = 16 + + monkeypatch.setattr(image_qc.shutil, "disk_usage", lambda _p: _TinyUsage()) + with pytest.raises(RuntimeError, match="free"): + image_qc._PlaneStore( + (128, 128), ["focus_map"], plane_dir=tmp_path, prefix="dapi" + ) + + def test_release_deletes_backing_files(self, tmp_path, monkeypatch): + monkeypatch.setattr(image_qc, "_PLANE_SPILL_BYTES", 1024) + store = image_qc._PlaneStore( + (64, 64), ["focus_map"], plane_dir=tmp_path, prefix="dapi" + ) + paths = [Path(p) for p in store.descriptors().values()] + assert all(p.exists() for p in paths) + store.release() + assert not any(p.exists() for p in paths) + + +# --------------------------------------------------------------------------- +# _figure_worker_limit +# --------------------------------------------------------------------------- + + +class TestCrossProcessPlaneSharing: + """Separate processes must share a plane by filename, with no data pickled. + + This is the property the process-per-GPU dispatch rests on, and the reason + disk-backing and multi-GPU are the same change: each worker maps the plane + file itself and writes its own disjoint tile region. Exercised here with a + trivial write instead of a CuPy kernel, so it runs without a GPU. + """ + + def test_spawned_workers_writes_are_visible_to_the_parent( + self, tmp_path, monkeypatch + ): + monkeypatch.setattr(image_qc, "_PLANE_SPILL_BYTES", 1024) + height, width = 200, 160 + + store = image_qc._PlaneStore( + (height, width), ["focus_map"], plane_dir=tmp_path, prefix="dapi" + ) + plane = store.arrays()["focus_map"] + plane[:] = 0.0 + store.flush() + + tiles = image_qc._compute_tile_grid(height, width, tile_size=64, overlap=8) + assert len(tiles) > 1, "need several tiles for this to mean anything" + tasks = [(spec, float(i + 1)) for i, spec in enumerate(tiles)] + + ctx = multiprocessing.get_context("spawn") + with ctx.Pool( + processes=2, + initializer=helpers_memmap_worker.init, + initargs=( + store.descriptors()["focus_map"], + (height, width), + np.dtype(np.float32).str, + ), + ) as pool: + written = list(pool.imap_unordered(helpers_memmap_worker.write_tile, tasks)) + + assert sorted(written) == sorted(v for _, v in tasks) + + expected = np.zeros((height, width), dtype=np.float32) + for spec, value in tasks: + expected[ + spec["write_y0"] : spec["write_y1"], + spec["write_x0"] : spec["write_x1"], + ] = value + + # Every pixel covered: the write regions tile the image exactly. + assert (expected != 0).all(), "tile write regions do not cover the plane" + # MAP_SHARED: the parent's pre-existing mapping sees the children's writes. + np.testing.assert_array_equal(np.asarray(plane), expected) + + def test_worker_writes_land_in_the_backing_file(self, tmp_path, monkeypatch): + """Not just the parent's mapping — the bytes are really on disk.""" + monkeypatch.setattr(image_qc, "_PLANE_SPILL_BYTES", 1024) + height, width = 96, 96 + store = image_qc._PlaneStore( + (height, width), ["mean_map"], plane_dir=tmp_path, prefix="boundary" + ) + store.arrays()["mean_map"][:] = 0.0 + store.flush() + path = store.descriptors()["mean_map"] + + spec = {"write_y0": 10, "write_y1": 20, "write_x0": 30, "write_x1": 40} + ctx = multiprocessing.get_context("spawn") + with ctx.Pool( + processes=1, + initializer=helpers_memmap_worker.init, + initargs=(path, (height, width), np.dtype(np.float32).str), + ) as pool: + pool.apply(helpers_memmap_worker.write_tile, ((spec, 7.5),)) + + fresh = np.memmap(path, dtype=np.float32, mode="r", shape=(height, width)) + assert fresh[10:20, 30:40].min() == pytest.approx(7.5) + assert fresh[0, 0] == pytest.approx(0.0) + + +class TestMemoryInstrumentation: + """working_set must exclude reclaimable page cache. + + This matters specifically because disk-backed planes deliberately generate + page cache: reading the raw usage counter would show a large number and + wrongly suggest the memory fix did nothing. + """ + + def _write_cgroup(self, root, usage, inactive, active, v2=True): + if v2: + (root / "memory.current").write_text(f"{usage}\n") + (root / "memory.stat").write_text( + f"anon 1234\ninactive_file {inactive}\nactive_file {active}\n" + ) + return ( + str(root / "memory.current"), + str(root / "memory.stat"), + "inactive_file", + "active_file", + ) + (root / "memory.usage_in_bytes").write_text(f"{usage}\n") + (root / "memory.stat").write_text( + f"total_inactive_file {inactive}\ntotal_active_file {active}\n" + ) + return ( + str(root / "memory.usage_in_bytes"), + str(root / "memory.stat"), + "total_inactive_file", + "total_active_file", + ) + + def test_subtracts_inactive_file_v2(self, tmp_path, monkeypatch): + source = self._write_cgroup(tmp_path, usage=200, inactive=150, active=10) + monkeypatch.setattr(image_qc, "_CGROUP_SOURCES", (source,)) + working_set, page_cache = image_qc._cgroup_memory() + assert working_set == 50 # 200 usage - 150 reclaimable + assert page_cache == 160 + + def test_supports_cgroup_v1(self, tmp_path, monkeypatch): + source = self._write_cgroup( + tmp_path, usage=500, inactive=400, active=25, v2=False + ) + monkeypatch.setattr(image_qc, "_CGROUP_SOURCES", (source,)) + working_set, page_cache = image_qc._cgroup_memory() + assert working_set == 100 + assert page_cache == 425 + + def test_falls_through_to_second_source(self, tmp_path, monkeypatch): + missing = ("/nonexistent/current", "/nonexistent/stat", "inactive_file", "a") + source = self._write_cgroup(tmp_path, usage=90, inactive=40, active=0) + monkeypatch.setattr(image_qc, "_CGROUP_SOURCES", (missing, source)) + assert image_qc._cgroup_memory()[0] == 50 + + def test_returns_none_when_no_cgroup(self, monkeypatch): + monkeypatch.setattr( + image_qc, + "_CGROUP_SOURCES", + (("/nonexistent/a", "/nonexistent/b", "inactive_file", "active_file"),), + ) + assert image_qc._cgroup_memory() is None + + def test_never_reports_negative_working_set(self, tmp_path, monkeypatch): + source = self._write_cgroup(tmp_path, usage=10, inactive=999, active=0) + monkeypatch.setattr(image_qc, "_CGROUP_SOURCES", (source,)) + assert image_qc._cgroup_memory()[0] == 0.0 + + @requires_proc + def test_process_rss_is_positive(self): + assert image_qc._process_rss() > 0 + + def test_log_mem_tracks_high_water_mark(self, tmp_path, monkeypatch): + monkeypatch.setitem(image_qc._MEM_PEAK, "working_set", 0.0) + (tmp_path / "lo").mkdir(parents=True, exist_ok=True) + low = self._write_cgroup(tmp_path / "lo", usage=0, inactive=0, active=0) + (tmp_path / "hi").mkdir(parents=True, exist_ok=True) + high = self._write_cgroup(tmp_path / "hi", usage=800, inactive=100, active=0) + + monkeypatch.setattr(image_qc, "_CGROUP_SOURCES", (high,)) + image_qc._log_mem("peak stage") + monkeypatch.setattr(image_qc, "_CGROUP_SOURCES", (low,)) + image_qc._log_mem("later quiet stage") + + assert image_qc._MEM_PEAK["working_set"] == 700 + + def test_log_mem_survives_missing_cgroup(self, monkeypatch): + monkeypatch.setattr(image_qc, "_CGROUP_SOURCES", ()) + image_qc._log_mem("no cgroup") # must not raise + image_qc._log_mem_summary() + + +class TestFigureWorkerLimit: + """The forked figure fan-out must be capped, not os.cpu_count().""" + + def test_env_override_wins(self, monkeypatch): + # Override is clamped to n_tasks; with plenty of tasks it wins outright. + monkeypatch.setenv("IMAGE_QC_FIGURE_WORKERS", "2") + assert image_qc._figure_worker_limit(11) == 2 + + def test_capped_at_max(self, monkeypatch): + monkeypatch.delenv("IMAGE_QC_FIGURE_WORKERS", raising=False) + monkeypatch.setattr(image_qc.os, "cpu_count", lambda: 96) + assert image_qc._figure_worker_limit(50) <= image_qc._FIGURE_WORKERS_MAX + + def test_always_at_least_one(self, monkeypatch): + monkeypatch.delenv("IMAGE_QC_FIGURE_WORKERS", raising=False) + monkeypatch.setattr(image_qc.os, "cpu_count", lambda: None) + assert image_qc._figure_worker_limit(5) >= 1 + + +def _hold_200mb(_): + """Worker for the tree-RSS test. Module level because Pool pickles the callable.""" + import numpy as np + + buf = np.ones(25_000_000, dtype=np.float64) # ~200 MB resident + time.sleep(3.0) + return float(buf[0]) + + +@requires_proc +class TestTreeRss: + """`_tree_rss` must count forked children, and only our own. + + The memory summary previously reported `VmHWM` from /proc/self/status, which + excludes children — and the figure phase forks up to `_FIGURE_WORKERS_MAX` of them. + On run 3V53J4ewZt1vsU that module reported 33.4 GB while Tower measured 100.6 GB for + the same task, and the summary line told the reader to size the memory request from + the smaller figure. Following it would under-provision by 3x. + + It sums descendants rather than all of /proc, because these modules also run + directly on shared servers where the latter would add other users' processes. + """ + + def test_counts_forked_children(self): + import multiprocessing as mp + + alone = image_qc._tree_rss() + assert alone is not None and alone > 0 + + ctx = mp.get_context("fork") + with ctx.Pool(3) as pool: + pending = pool.map_async(_hold_200mb, range(3)) + time.sleep(1.5) + with_children = image_qc._tree_rss() + pending.get(timeout=60) + + assert with_children > alone + 300 * 1024**2, ( + f"{(with_children - alone) / 1024**3:.2f} GB delta for three ~200 MB " + "children — forked memory is not being counted" + ) + + def test_excludes_processes_outside_our_tree(self): + """Our tree must be a strict subset of everything readable in /proc. + + The earlier version of this test asserted only that the result was under 64 GB, + which its name did not justify. This compares against the sum over every + readable process: if `_tree_rss` summed all of /proc — correct in a container, + wrong on a shared server — the two would be equal. + """ + everything = 0.0 + others = 0 + me = os.getpid() + for entry in os.listdir("/proc"): + if not entry.isdigit(): + continue + try: + with open(f"/proc/{entry}/status") as fh: + for line in fh: + if line.startswith("VmRSS:"): + everything += float(line.split()[1]) * 1024.0 + if int(entry) != me: + others += 1 + break + except (OSError, ValueError, IndexError): + continue + if others == 0: + pytest.skip("no other readable processes — nothing to exclude") + ours = image_qc._tree_rss() + assert ours is not None + assert ours <= everything, "our tree cannot exceed all of /proc" + assert ours < everything, ( + f"tree RSS {ours / 1024**3:.2f} GB equals the total over " + f"{others} other processes — unrelated processes are being counted" + ) + + def test_returns_none_when_proc_is_unreadable(self, monkeypatch): + monkeypatch.setattr( + image_qc.os, "listdir", lambda *_: (_ for _ in ()).throw(OSError()) + ) + assert image_qc._tree_rss() is None + + +class TestCgroupPeak: + """`_cgroup_peak` reads the kernel's continuous all-process high-water mark. + + The other three figures in the summary each understate the total in a different way, + which is why this one exists and why the summary no longer points at a single number: + + tree_rss all processes, but sampled at stage boundaries -- on run + 1AjX0aBBfdbAFQ it reported 16.2 GB against a main_process_rss of + 19.6 GB, i.e. below a figure it is supposed to contain, because + the main process peaked between samples + working_set OOM-relevant, also sampled + main_process_rss continuous but excludes the forked figure workers + """ + + def test_reads_v2_then_v1(self, tmp_path, monkeypatch): + v2 = tmp_path / "memory.peak" + v1 = tmp_path / "max_usage_in_bytes" + v1.write_text("111\n") + monkeypatch.setattr(image_qc, "_CGROUP_PEAK_PATHS", (str(v2), str(v1))) + assert image_qc._cgroup_peak() == 111.0 # falls through to v1 + v2.write_text("222\n") + assert image_qc._cgroup_peak() == 222.0 # prefers v2 + + def test_none_when_absent(self, tmp_path, monkeypatch): + monkeypatch.setattr( + image_qc, + "_CGROUP_PEAK_PATHS", + (str(tmp_path / "nope"), str(tmp_path / "also-nope")), + ) + assert image_qc._cgroup_peak() is None + + def test_none_on_unparseable(self, tmp_path, monkeypatch): + bad = tmp_path / "memory.peak" + bad.write_text("max\n") + monkeypatch.setattr(image_qc, "_CGROUP_PEAK_PATHS", (str(bad),)) + assert image_qc._cgroup_peak() is None + + def test_summary_reports_every_figure(self, caplog, monkeypatch): + """All four must appear, so no reader takes one for the whole story.""" + monkeypatch.setattr(image_qc, "_cgroup_peak", lambda: 123 * 1024**3) + monkeypatch.setitem(image_qc._MEM_PEAK, "tree_rss", 40 * 1024**3) + monkeypatch.setitem(image_qc._MEM_PEAK, "working_set", 30 * 1024**3) + monkeypatch.setitem(image_qc._MEM_PEAK, "rss", 50 * 1024**3) + with caplog.at_level("INFO"): + image_qc._log_mem_summary() + text = caplog.text + for token in ( + "cgroup_peak=123.0GB", + "tree_rss=40.0GB", + "working_set=30.0GB", + "main_process_rss=50.0GB", + ): + assert token in text, token + assert "peakRss" in text, "must point at the authoritative source for sizing" diff --git a/bin/tests/test_lazy_tiff_channel.py b/bin/tests/test_lazy_tiff_channel.py new file mode 100644 index 00000000..a3863783 --- /dev/null +++ b/bin/tests/test_lazy_tiff_channel.py @@ -0,0 +1,118 @@ +"""Regression test for `_LazyTiffChannel` instances built via `__new__`. + +`_open_morphology_lazy` has a branch for TIFFs that store every channel in a single +3-D page. It builds the channel objects with `_LazyTiffChannel.__new__`, which +creates the instance while skipping `__init__`, then sets five attributes by hand: +`_page`, `shape`, `_lock`, `_source`, `_cached_data`. + +`_is_tiled` was not among them, and it was only ever assigned inside `__init__` +while `__getitem__` reads it unguarded on every region request. So the first tile +read raised `AttributeError: '_LazyTiffChannel' object has no attribute +'_is_tiled'`, inside a thread pool, killing the step. The pre-cached data does not +save it: the cache is consumed in `_read_region_fallback`, which is only reached +after the `_is_tiled` check. + +The fix is a class-level default rather than three more hand-assignments, so a +future `__new__` site cannot reintroduce the bug. These tests assert the object is +usable, not merely that the attribute exists, and cover the profiling counters the +same branch also omits (those are getattr-guarded, so they were never fatal). +""" + +from __future__ import annotations + +import importlib +import sys +import types + +import numpy as np + +for _mod in ( + "napari_skimage_regionprops", + "napari_simpleitk_image_processing", + "snr_metrics", + "scanpy", +): + if _mod not in sys.modules: + sys.modules[_mod] = types.ModuleType(_mod) +sys.modules["napari_skimage_regionprops"].regionprops_table = lambda *a, **kw: None # type: ignore[attr-defined] +sys.modules["scanpy"].AnnData = object # type: ignore[attr-defined] + +image_qc = importlib.import_module("image_qc") +_LazyTiffChannel = image_qc._LazyTiffChannel + + +class _FakePage: + """Minimal stand-in for a tifffile TiffPage/TiffFrame.""" + + def __init__(self, shape, dtype=np.uint16): + self.shape = shape + self.dtype = np.dtype(dtype) + + +def _channel_via_new(data: np.ndarray) -> object: + """Reproduce exactly what `_open_morphology_lazy`'s 3-D-page branch builds.""" + import threading + + ch = _LazyTiffChannel.__new__(_LazyTiffChannel) + ch._page = _FakePage(data.shape) + ch.shape = (data.shape[0], data.shape[1]) + ch._lock = threading.Lock() + ch._source = None + ch._cached_data = data + return ch + + +def test_is_tiled_is_readable_without_init(): + """The attribute must exist on an instance that never ran __init__.""" + ch = _LazyTiffChannel.__new__(_LazyTiffChannel) + assert ch._is_tiled is False + + +def test_is_tiled_is_a_class_level_default(): + """A class attribute, so any future __new__ site inherits it.""" + assert _LazyTiffChannel._is_tiled is False + + +def test_slicing_a_new_built_channel_returns_the_cached_region(): + """The real failure: the first region read raised AttributeError.""" + data = np.arange(64 * 48, dtype=np.uint16).reshape(64, 48) + ch = _channel_via_new(data) + + got = ch[10:20, 5:15] + + assert got.shape == (10, 10) + np.testing.assert_array_equal(got, data[10:20, 5:15]) + + +def test_full_slice_of_a_new_built_channel(): + data = np.arange(16 * 16, dtype=np.uint16).reshape(16, 16) + ch = _channel_via_new(data) + np.testing.assert_array_equal(ch[:, :], data) + + +def test_empty_slice_is_empty_not_an_error(): + """Degenerate request: zero-area slices short-circuit before the tiled check.""" + data = np.zeros((16, 16), dtype=np.uint16) + ch = _channel_via_new(data) + assert ch[5:5, 0:10].shape == (0, 0) + assert ch[0:10, 8:4].shape == (0, 0) + + +def test_init_still_takes_is_tiled_from_the_page(): + """The class default must not shadow a real tiled page.""" + + class _TiledPage(_FakePage): + is_tiled = True + + ch = _LazyTiffChannel(_TiledPage((32, 32))) + assert ch._is_tiled is True + + ch_plain = _LazyTiffChannel(_FakePage((32, 32))) + assert ch_plain._is_tiled is False, "a page without is_tiled means non-tiled" + + +def test_profiling_counters_are_safe_on_a_new_built_channel(): + """The same branch omits these too; they are getattr-guarded, so non-fatal.""" + ch = _channel_via_new(np.zeros((8, 8), dtype=np.uint16)) + for attr in ("_read_seconds", "_decode_seconds", "_read_bytes"): + assert getattr(ch, attr, 0.0) is not None diff --git a/bin/tests/test_qc_threshold_refinements.py b/bin/tests/test_qc_threshold_refinements.py new file mode 100644 index 00000000..35ba922b --- /dev/null +++ b/bin/tests/test_qc_threshold_refinements.py @@ -0,0 +1,248 @@ +"""Unit tests for the 2026-06-22 QC threshold/metric refinements. + +Covers the new code paths: +1. ``read_xenium_major_version`` — XOA version detection for per-version + intensity floors (inlined helper block in ``image_qc``). +2. ``read_bundle_metrics`` / ``_csv_float`` / ``_csv_str`` — the three + 10x-defined gates read from metrics_summary.csv, with the critical + absent-vs-zero distinction (``transcript_qc_processing``). +3. ``assess_raw_intensity_quality`` — WARN-only intensity (no FAIL tier). + +ADAPTED FROM UPSTREAM (nf-xenium-processing dev HEAD 5e35cae, +``tests/test_qc_threshold_refinements.py``): +- ``xenium_helpers`` is not shipped by this pipeline; its helpers are inlined + into the scripts, so ``read_xenium_major_version`` is an attribute of the + ``image_qc`` module (``transcript_qc_processing`` deliberately omits it). + The ``utils = importlib.import_module("xenium_helpers.utils")`` handle and its + ``sys.path`` entry are therefore gone, and ``utils.read_xenium_major_version`` + is now ``image_qc.read_xenium_major_version``. +- ``molecule_qc_processing`` is named ``transcript_qc_processing`` here; the + ``mqc`` alias is kept and repointed. +- Scripts live in the pipeline-level ``bin/`` rather than + ``modules/local/*/resources/usr/bin/``. +No assertion was changed. +""" + +from __future__ import annotations + +import importlib +import sys +import types +from pathlib import Path + +import pandas as pd +import pytest + +# bin modules have heavy optional top-level imports (napari, snr_metrics) that +# are not needed here. Stub them before importing, matching test_focus_score_compute. +_bin_dir = Path(__file__).resolve().parent.parent +sys.path.insert(0, str(_bin_dir)) + +for _mod in ( + "napari_skimage_regionprops", + "napari_simpleitk_image_processing", + "snr_metrics", +): + if _mod not in sys.modules: + sys.modules[_mod] = types.ModuleType(_mod) +sys.modules["napari_skimage_regionprops"].regionprops_table = lambda *a, **kw: None # type: ignore[attr-defined] + +mqc = importlib.import_module("transcript_qc_processing") +image_qc = importlib.import_module("image_qc") + + +# --------------------------------------------------------------------------- +# read_xenium_major_version +# --------------------------------------------------------------------------- +def _write_experiment(tmp_path: Path, version) -> Path: + import json + + d = tmp_path + payload = {} if version is None else {"analysis_sw_version": version} + (d / "experiment.xenium").write_text(json.dumps(payload)) + return d + + +@pytest.mark.parametrize( + "version, expected", + [ + ("xenium-3.2.0.7", 3), + ("xenium-4.0.1.0", 4), + ("xenium-2.0.0", 2), + (None, None), # key absent + ("garbage", None), # unparseable + ], +) +def test_read_xenium_major_version(tmp_path, version, expected): + d = _write_experiment(tmp_path, version) + assert image_qc.read_xenium_major_version(d) == expected + + +def test_read_xenium_major_version_missing_file(tmp_path): + # No experiment.xenium present + assert image_qc.read_xenium_major_version(tmp_path) is None + + +# --------------------------------------------------------------------------- +# _csv_float / _csv_str — absent (None) vs zero (0.0) buckets +# --------------------------------------------------------------------------- +def test_csv_float_distinguishes_absent_from_zero(): + assert mqc._csv_float(0.0) == 0.0 # real zero is kept + assert mqc._csv_float("0") == 0.0 + assert mqc._csv_float("") is None # blank -> absent + assert mqc._csv_float(None) is None + assert mqc._csv_float(float("nan")) is None # pandas blank -> NaN -> absent + assert mqc._csv_float("nan") is None + assert mqc._csv_float("abc") is None + + +def test_csv_str_blank_is_none(): + assert ( + mqc._csv_str("xenium_cell_segmentation_stains_v1") + == "xenium_cell_segmentation_stains_v1" + ) + assert mqc._csv_str("") is None + assert mqc._csv_str(None) is None + assert mqc._csv_str(float("nan")) is None # blank stain_definition + assert mqc._csv_str("nan") is None + + +# --------------------------------------------------------------------------- +# read_bundle_metrics +# --------------------------------------------------------------------------- +def _write_metrics_csv(tmp_path: Path, **cols) -> Path: + pd.DataFrame([cols]).to_csv(tmp_path / "metrics_summary.csv", index=False) + return tmp_path + + +def test_read_bundle_metrics_stain_segmentation_run(tmp_path): + d = _write_metrics_csv( + tmp_path, + nuclear_transcripts_per_100um2=180.5, + fraction_empty_cells=0.00163, + segmented_cell_stain_frac=0.987, + stain_definition="xenium_cell_segmentation_stains_v1", + ) + out = mqc.read_bundle_metrics(d, is_resegmented=False) + assert out["nuclear_transcripts_per_100um2"] == pytest.approx(180.5) + assert out["fraction_empty_cells"] == pytest.approx(0.00163) + assert out["segmented_cell_stain_frac"] == pytest.approx(0.987) + assert out["stain_definition"] == "xenium_cell_segmentation_stains_v1" + assert out["source"] == "metrics_summary.csv" + + +def test_read_bundle_metrics_nuc_expansion_stain_def_blank(tmp_path): + # Nucleus-expansion run: stain_frac is a real 0.0, stain_definition blank. + # The 0.0 must be kept (real state), stain_definition must be None (absent). + d = _write_metrics_csv( + tmp_path, + nuclear_transcripts_per_100um2=35.2, + fraction_empty_cells=0.0056, + segmented_cell_stain_frac=0.0, + stain_definition="", + ) + out = mqc.read_bundle_metrics(d, is_resegmented=False) + assert out["segmented_cell_stain_frac"] == 0.0 # real zero, not None + assert out["stain_definition"] is None # absent, so gate stays informational + + +def test_read_bundle_metrics_resegmented_is_na(tmp_path): + _write_metrics_csv( + tmp_path, + nuclear_transcripts_per_100um2=180.5, + fraction_empty_cells=0.0016, + segmented_cell_stain_frac=0.99, + stain_definition="stains_v1", + ) + out = mqc.read_bundle_metrics(tmp_path, is_resegmented=True) + assert out["nuclear_transcripts_per_100um2"] is None + assert out["fraction_empty_cells"] is None + assert out["segmented_cell_stain_frac"] is None + assert "resegmented" in out["source"] + + +def test_read_bundle_metrics_missing_csv(tmp_path): + out = mqc.read_bundle_metrics(tmp_path, is_resegmented=False) + assert out["nuclear_transcripts_per_100um2"] is None + assert out["source"] is None + + +def test_read_bundle_metrics_missing_column(tmp_path): + d = _write_metrics_csv(tmp_path, some_other_col=1.0) + out = mqc.read_bundle_metrics(d, is_resegmented=False) + assert out["nuclear_transcripts_per_100um2"] is None + assert out["source"] == "metrics_summary.csv" # CSV present, column absent + + +# --------------------------------------------------------------------------- +# assess_raw_intensity_quality — WARN-only (no FAIL tier) +# --------------------------------------------------------------------------- +def test_intensity_quality_never_fails(): + # All tiles far below the critical floor — under the old logic this FAILed. + df = pd.DataFrame( + { + "dapi_intensity": [10.0] * 100, + "boundary_intensity": [5.0] * 100, + "intrna_intensity": [5.0] * 100, + "tissue_coverage": [1.0] * 100, + } + ) + stats = image_qc.assess_raw_intensity_quality( + df, + dapi_threshold_critical=500, + boundary_threshold_critical=100, + intrna_threshold_critical=300, + ) + for ch in ("dapi", "boundary", "intrna"): + assert stats[ch]["quality_status"] in ("warn", "pass", "not_available") + assert stats[ch]["quality_status"] != "fail" + assert stats[ch]["pct_fail_threshold"] is None + assert stats["overall_quality"] in ("warn", "pass", "not_available") + assert stats["overall_quality"] != "fail" + + +# --------------------------------------------------------------------------- +# compute_whole_grid_stain_percentiles — mask-independent diagnostic (§5.5) +# --------------------------------------------------------------------------- +def test_whole_grid_percentiles_all_channels(): + # 100 tiles 1..100; p95/p99 are computed over the WHOLE grid (no mask). + df = pd.DataFrame( + { + "dapi_intensity": list(range(1, 101)), + "boundary_intensity": list(range(1, 101)), + "intrna_intensity": list(range(1, 101)), + "tissue_coverage": [0.0] * 100, # no tissue — must be ignored + } + ) + out = image_qc.compute_whole_grid_stain_percentiles(df) + # Mask-independent: zero tissue coverage must NOT zero the percentiles. + assert out["dapi"]["p99"] == pytest.approx(99.01, abs=0.5) + assert out["dapi"]["p95"] == pytest.approx(95.05, abs=0.5) + for ch in ("dapi", "boundary", "intrna"): + assert out[ch]["p99"] is not None + + +def test_whole_grid_percentiles_missing_channel(): + df = pd.DataFrame({"dapi_intensity": [100.0, 200.0, 300.0]}) + out = image_qc.compute_whole_grid_stain_percentiles(df) + assert out["dapi"]["p99"] is not None + assert out["boundary"] == {"p95": None, "p99": None} + assert out["intrna"] == {"p95": None, "p99": None} + + +def test_whole_grid_percentiles_all_nan_channel(): + df = pd.DataFrame( + { + "dapi_intensity": [float("nan")] * 5, + "boundary_intensity": [10.0] * 5, + } + ) + out = image_qc.compute_whole_grid_stain_percentiles(df) + assert out["dapi"] == {"p95": None, "p99": None} # all-NaN -> None + assert out["boundary"]["p99"] == pytest.approx(10.0) + + +def test_whole_grid_percentiles_empty_df(): + out = image_qc.compute_whole_grid_stain_percentiles(pd.DataFrame()) + for ch in ("dapi", "boundary", "intrna"): + assert out[ch] == {"p95": None, "p99": None} diff --git a/bin/tests/test_snr_verdict_scoping.py b/bin/tests/test_snr_verdict_scoping.py new file mode 100644 index 00000000..08342d46 --- /dev/null +++ b/bin/tests/test_snr_verdict_scoping.py @@ -0,0 +1,259 @@ +"""Tests for SNR verdicts that must not read PASS when nothing was measured. + +Three defects, all of which produced a green report rather than an error, which is +worse than a crash because nothing about the run looks wrong: + +* ``aggregate_snr_verdict`` started at ``"PASS"`` and only moved on an explicit + FAIL or WARN. Components that could not run return ``{"status": "skipped"}`` with + no ``verdict`` key, so they matched neither branch and the overall stayed PASS. +* ``compute_roi_snr`` compared an all-NaN median against its thresholds. Both + ``NaN < warn`` and ``NaN > warn`` are False, so a bundle with no negative-control + probes reported PASS and wrote a bare ``NaN`` token into the JSON. +* ``compute_image_snr_from_pixel_maps`` took its median over every tile on the + slide while its sibling ``compute_image_snr_from_roi_df`` filtered to tissue + first, though both grade against the same ``image_snr_db`` warn/fail pair. An + empty tile splits its own noise and can score higher than real tissue, so a + slide with a small tissue footprint was graded on the coverslip. + +The image-SNR tests assert the two sibling functions agree on which population +they measure, rather than pinning dB values, so they cannot enshrine a scope error. +""" + +from __future__ import annotations + +import importlib +import importlib.util +from pathlib import Path +import json + +import numpy as np +import pandas as pd + +# snr_metrics is stubbed by other test modules (they share sys.modules), so a plain +# import can hand back an empty ModuleType depending on collection order. Load the +# real module by path, the same way test_tile_consumers.py does, so this file works +# whatever ran first. +_snr_path = ( + Path(__file__).resolve().parent.parent / "snr_metrics.py" +) +_spec = importlib.util.spec_from_file_location("_real_snr_metrics", _snr_path) +assert _spec is not None and _spec.loader is not None # narrow for mypy; path is checked above +snr_metrics = importlib.util.module_from_spec(_spec) +_spec.loader.exec_module(snr_metrics) + + +# --------------------------------------------------------------------------- +# aggregate_snr_verdict +# --------------------------------------------------------------------------- + + +def test_no_component_reported_a_verdict_is_not_pass(): + """Every component skipped: the module measured nothing, so PASS is a lie.""" + parts = { + "SNR_image_roi_quartile_db": {"status": "skipped", "reason": "missing column"}, + "SNR_image_otsu": {"status": "error", "error": "unreadable input"}, + "SNR_roi_tx": {"status": "skipped", "reason": "no transcripts file"}, + } + out = snr_metrics.aggregate_snr_verdict(parts) + assert out["overall_snr_verdict"] == "NOT_COMPUTED" + assert out["components_not_computed"] == sorted(parts) + + +def test_partial_skip_keeps_the_real_verdict_but_names_the_gaps(): + """One component reported, others did not. The verdict stands and says so.""" + parts = { + "SNR_image_otsu": {"status": "ok", "verdict": "WARN"}, + "SNR_roi_tx": {"status": "skipped", "reason": "no transcripts file"}, + } + out = snr_metrics.aggregate_snr_verdict(parts) + assert out["overall_snr_verdict"] == "WARN" + assert out["components_not_computed"] == ["SNR_roi_tx"] + + +def test_all_components_pass_reports_pass_with_no_gap_list(): + parts = { + "SNR_image_otsu": {"status": "ok", "verdict": "PASS"}, + "SNR_roi_tx": {"status": "ok", "verdict": "PASS"}, + } + out = snr_metrics.aggregate_snr_verdict(parts) + assert out["overall_snr_verdict"] == "PASS" + assert "components_not_computed" not in out + + +def test_fail_still_wins_over_warn(): + """Pre-existing precedence must survive the NOT_COMPUTED change.""" + parts = { + "a": {"verdict": "WARN"}, + "b": {"verdict": "FAIL"}, + "c": {"status": "skipped"}, + } + assert snr_metrics.aggregate_snr_verdict(parts)["overall_snr_verdict"] == "FAIL" + + +def test_non_dict_part_counts_as_not_computed(): + parts = {"a": None, "b": "unexpected"} + out = snr_metrics.aggregate_snr_verdict(parts) + assert out["overall_snr_verdict"] == "NOT_COMPUTED" + assert out["components_not_computed"] == ["a", "b"] + + +# --------------------------------------------------------------------------- +# _write_snr_json +# --------------------------------------------------------------------------- + + +def test_json_writer_never_emits_a_bare_nan_token(tmp_path): + """`json.dump(default=str)` does not catch float NaN; it writes `NaN`.""" + summary = { + "median_roi_tx_snr_ratio": float("nan"), + "nested": {"also_bad": float("inf"), "fine": 1.5}, + "listed": [float("nan"), 2.0], + } + snr_metrics._write_snr_json(summary, tmp_path) + + raw = (tmp_path / "snr_metrics.json").read_text() + assert "NaN" not in raw + assert "Infinity" not in raw + # strict parsers must accept it + parsed = json.loads(raw, parse_constant=_reject_constant) + assert parsed["median_roi_tx_snr_ratio"] is None + assert parsed["nested"]["also_bad"] is None + assert parsed["nested"]["fine"] == 1.5 + assert parsed["listed"] == [None, 2.0] + + +def _reject_constant(name): # pragma: no cover - only runs if the file is bad + raise AssertionError(f"non-standard JSON constant written: {name}") + + +# --------------------------------------------------------------------------- +# compute_image_snr_from_pixel_maps: tissue scoping +# --------------------------------------------------------------------------- + +_THRESH = {"image_snr_db": {"warn": 15.0, "fail": 10.0}} + + +def _tiled_frame(n: int, tissue_flags: list[bool], stride: int = 8) -> pd.DataFrame: + return pd.DataFrame( + { + "x1": [i * stride for i in range(n)], + "x2": [(i + 1) * stride for i in range(n)], + "y1": [0] * n, + "y2": [stride] * n, + "overlaps_tissue": tissue_flags, + "tissue_coverage": [1.0 if t else 0.0 for t in tissue_flags], + } + ) + + +def test_median_ignores_non_tissue_tiles(): + """Background tiles must not enter the median, however many there are. + + Supplying per-tile dB directly isolates the scoping decision from the Otsu + measurement: tissue tiles are poor (5 dB) and the many background tiles are + excellent (40 dB). Scoped to tissue this must FAIL. Unscoped, the background + median would carry it to PASS. + """ + tissue = [True, True, True] + background = [False] * 12 + df = _tiled_frame(15, tissue + background) + per_tile = [5.0, 5.0, 5.0] + [40.0] * 12 + + out = snr_metrics.compute_image_snr_from_pixel_maps( + None, # focus_maps unused: precomputed_db supplies the per-tile dB + df, + map_key="dapi", + snr_thresholds=_THRESH, + precomputed_db=per_tile, + ) + + assert out["status"] == "ok" + assert out["scope"] == "tissue_tiles" + assert out["snr_db_median"] == 5.0 + assert out["verdict"] == "FAIL" + assert out["n_rois_computed"] == 3 + assert out["n_rois_all_tiles"] == 15 + + +def test_per_tile_column_stays_unfiltered_for_the_concordance_figure(): + """Scoping the verdict must not silently empty the exposed per-tile column.""" + df = _tiled_frame(4, [True, False, True, False]) + per_tile = [20.0, 30.0, 22.0, 31.0] + + snr_metrics.compute_image_snr_from_pixel_maps( + None, # focus_maps unused: precomputed_db supplies the per-tile dB + df, + map_key="dapi", + snr_thresholds=_THRESH, + precomputed_db=per_tile, + ) + + written = df["snr_image_otsu_db"].to_numpy() + assert not np.isnan(written).any(), "background tiles lost their per-tile dB" + np.testing.assert_allclose(written, per_tile) + + +def test_declines_rather_than_grading_a_slide_with_no_tissue(): + df = _tiled_frame(4, [False] * 4) + out = snr_metrics.compute_image_snr_from_pixel_maps( + None, # focus_maps unused: precomputed_db supplies the per-tile dB + df, + map_key="dapi", + snr_thresholds=_THRESH, + precomputed_db=[40.0] * 4, + ) + assert out["status"] == "skipped" + assert out["reason"] == "no_valid_tissue_roi_tiles" + assert "verdict" not in out, "a verdict here would be graded on background" + # and the aggregate must therefore not call it PASS + assert ( + snr_metrics.aggregate_snr_verdict({"SNR_image_otsu": out})[ + "overall_snr_verdict" + ] + == "NOT_COMPUTED" + ) + + +def test_the_two_siblings_agree_on_which_population_they_measure(): + """The reference: both functions grade the same tissue tiles. + + ``compute_image_snr_from_roi_df`` has always filtered to tissue. Giving both + the same frame, where the tissue tiles carry one dB value and the background + another, the pixel-maps median must match the tissue value rather than land + between the two. + """ + tissue = [True] * 10 + background = [False] * 10 + df = _tiled_frame(20, tissue + background) + per_tile = [18.0] * 10 + [45.0] * 10 + + out = snr_metrics.compute_image_snr_from_pixel_maps( + None, # focus_maps unused: precomputed_db supplies the per-tile dB + df.copy(), + map_key="dapi", + snr_thresholds=_THRESH, + precomputed_db=per_tile, + ) + assert out["snr_db_median"] == 18.0 + assert out["n_rois_computed"] == 10 + + +def test_frame_without_tissue_columns_still_grades_every_tile(): + """No tissue information available: fall back to all tiles, do not decline.""" + df = pd.DataFrame( + { + "x1": [0, 8], + "x2": [8, 16], + "y1": [0, 0], + "y2": [8, 8], + } + ) + out = snr_metrics.compute_image_snr_from_pixel_maps( + None, # focus_maps unused: precomputed_db supplies the per-tile dB + df, + map_key="dapi", + snr_thresholds=_THRESH, + precomputed_db=[20.0, 22.0], + ) + assert out["status"] == "ok" + assert out["n_rois_computed"] == 2 diff --git a/bin/transcript_qc.qmd b/bin/transcript_qc.qmd new file mode 100644 index 00000000..92c784b4 --- /dev/null +++ b/bin/transcript_qc.qmd @@ -0,0 +1,1877 @@ +--- +title: "Xenium Transcript QC Report" +format: + html: + embed-resources: true + standalone: true + toc: true + toc-depth: 2 + fontsize: 0.9rem + title-block-style: none +jupyter: python3 +--- + +```{python parameters} +#| tags: [parameters] +#| echo: false + +# Parameters — overridden by -P command line arguments +INDIR = "transcript_qc" +SAMPLE_NAME = "" +XENIUM_BUNDLE = "" +SAMPLE_PUBLISHED_OUTDIR = "" +# Pipeline-staged YAML path (mirrors image QC's ROI_THRESHOLDS_YAML +# convention — the Quarto module passes the YAML's filename here when +# ext.pass_roi_thresholds_yaml is true; empty default means "fall back +# to repo-relative search or the embedded defaults in load_qc_cutoffs"). +ROI_THRESHOLDS_YAML = "" +``` + +```{python setup-environment} +#| echo: false + +from __future__ import annotations + +import base64 +import json +from pathlib import Path + +import numpy as np +import pandas as pd +from IPython.display import HTML, Markdown, display + +base = Path(INDIR) + +# Validate required files +metrics_file = base / "transcript_qc_metrics.json" +if not metrics_file.exists(): + raise FileNotFoundError( + f"CRITICAL ERROR: Required QC metrics file missing: {metrics_file}" + ) +versions_path = base / "versions.yml" +if not versions_path.exists(): + raise FileNotFoundError( + f"CRITICAL ERROR: Required versions file missing: {versions_path}" + ) + +with open(metrics_file, "r", encoding="utf-8") as f: + qc_metrics: dict = json.load(f) + +versions_text: str | None = None +if versions_path.exists(): + versions_text = versions_path.read_text(encoding="utf-8") + + +# ============================================================================= +# === SHARED REPORT HELPERS — copied from xenium_image_qc_report.qmd === +# === (v5, 2026-05-22, image QC HEAD = 6cf4642). === +# === Rule-of-three not yet triggered (2 reports) — extraction to === +# === xenium_helpers.report_helpers deferred. Re-diff against image QC === +# === HEAD at PR-open time per plans/2026-05-22_PLAN_..._v5_redesign.md === +# === Validation row 16. === +# ============================================================================= + + +_EMBEDDED_THRESHOLDS_FALLBACK = { + "transcript_qc": { + "codeword_categories": { + "decoded_gene_combined": { + "display_label": "Decoded gene (custom + predesigned)", + "contributors": ["custom_gene", "predesigned_gene"], + "pass_min": 95.0, + "warn_min": 90.0, + }, + "unassigned_codeword": { + "display_label": "Unassigned codeword", + "pass_max": 5.0, + "warn_max": 10.0, + }, + "negative_control_probe": { + "display_label": "Negative control probe", + "pass_max": 2.0, + "warn_max": 5.0, + }, + }, + "codeword_informational": [ + "custom_gene", + "predesigned_gene", + "genomic_control_probe", + "negative_control_codeword", + "deprecated_codeword", + ], + "category_display_labels": { + "custom_gene": "Custom gene", + "predesigned_gene": "Predesigned gene", + "genomic_control_probe": "Genomic control probe", + "negative_control_probe": "Negative control probe", + "negative_control_codeword": "Negative control codeword", + "unassigned_codeword": "Unassigned codeword", + "deprecated_codeword": "Deprecated codeword", + }, + "yield": { + "total_transcripts_pass_min": 1_000_000, + "total_transcripts_warn_min": 100_000, + }, + "noise_filter": { + "pct_filtered_pass_max": 50.0, + "pct_filtered_warn_max": 80.0, + }, + "qv": { + "pct_above_qv20_pass_min": 70.0, + "pct_above_qv20_warn_min": 60.0, + }, + # SegTraQ Phase A — ADVISORY WARN-only (2026-06-29): pass_max 30 (just + # above pancreas median 29); deduped 164-sample cohort p95 39 / p99 47. + # warn_max null -> WARN above 30, never FAIL. Strongly tissue-driven. + "transcript_assignment": { + "pct_unassigned_pass_max": 30.0, + "pct_unassigned_warn_max": None, + }, + "fov_consistency": { + "n_gate": 100, + "z_warn": 2.0, + "z_fail": 3.0, + "sample_pct_fail_warn": 2.0, + "sample_pct_fail_fail": 5.0, + "sample_pct_warn_warn": 5.0, + }, + # SegTraQ Phase A — advisory, PROVISIONAL (calibrate in Phase B). + "fov_density": { + "z_warn": 2.0, + "z_fail": 3.0, + "sample_pct_dropout_pass_max": 10.0, + "sample_pct_dropout_warn_max": None, + }, + "segmentation": { + "nucleus_to_cell_median_pass_min": 0.10, + "nucleus_to_cell_median_pass_max": 0.60, + "nucleus_to_cell_pct_above_80_pass_max": 10.0, + "nucleus_to_cell_median_warn_min": 0.05, + "nucleus_to_cell_median_warn_max": 0.70, + "nucleus_to_cell_pct_above_80_warn_max": 20.0, + }, + # §5 section-summary additions — lenient PASS/WARN advisory bands. + "cell_size": { + "median_pass_min": 30.0, + "median_pass_max": 500.0, + }, + "nucleus_transcript_fraction": { + # PROVISIONAL — corrected metric (overlaps_nucleus per cell); lenient + # advisory band, not cohort-calibrated. Mirror conf/ YAML. + "median_pass_min": 0.05, + "median_pass_max": 0.70, + "pct_above_60_pass_max": 50.0, + }, + # SegTraQ Phase A. warn 10 / fail 15 (2026-06-23): matches the report's + # deduped p95/p99 = 10/15 (onboard-calibrated, one method-agnostic cut). + # Adds a FAIL tier — the one cell-level row that is not advisory-only. + "nucleus_coverage": { + "pct_cells_no_nucleus_pass_max": 10.0, + "pct_cells_no_nucleus_warn_max": 15.0, + }, + "cells_transcripts": { + "median_pass_x_min": 2.0, + }, + "cells_genes": { + "median_pass_x_min": 2.0, + }, + # 10x-defined bundle metrics (metrics_summary.csv), added 2026-06-22. + "nuclear_density": { + "per_100um2_pass_min": 10.0, + "per_100um2_warn_min": 1.0, + }, + "empty_cells": { + "pct_pass_max": 5.0, + "pct_warn_max": 10.0, + }, + "segmentation_stain": { + "stain_frac_pass_min": 70.0, + "stain_frac_warn_min": None, + }, + }, +} + + +def load_qc_cutoffs() -> dict: + """Load report thresholds. Order of preference: + + 1. YAML at the path the pipeline stages via the ROI_THRESHOLDS_YAML + Quarto parameter (the Quarto module stages the YAML into the + Nextflow work dir and passes its filename here — same mechanism + as image QC). + 2. YAML at repo-relative locations (for local renders from the repo + root). + 3. Embedded fallback dict — keeps the report functional even if the + pipeline-staged YAML never arrives. Values mirror + conf/transcript_qc_thresholds.yaml; the two are versioned together, + so updating one should be paired with updating the other. + """ + try: + import yaml # type: ignore + except Exception: + return _EMBEDDED_THRESHOLDS_FALLBACK + + candidates: list[Path] = [] + # 1. Pipeline-staged YAML (preferred). Use globals().get so the QMD + # still works if Quarto's parameter injection didn't define the var + # (e.g. local rendering, broken parameter wiring). + _staged = str(globals().get("ROI_THRESHOLDS_YAML", "") or "").strip() + if _staged: + candidates.append(Path(_staged)) + # 2. Repo-relative paths for local renders. + candidates.extend( + [ + Path("conf/transcript_qc_thresholds.yaml"), + Path("../conf/transcript_qc_thresholds.yaml"), + Path("../../conf/transcript_qc_thresholds.yaml"), + ] + ) + for p in candidates: + try: + if p.exists(): + with open(p, encoding="utf-8") as f: + d = yaml.safe_load(f) or {} + if isinstance(d, dict) and d: + return d + except Exception: + continue + # 3. Embedded fallback — ensures the report still has thresholds. + return _EMBEDDED_THRESHOLDS_FALLBACK + + +qc_cutoffs = load_qc_cutoffs() + + +def fig_html(rel: str, *, max_width: str = "88%") -> HTML | str: + p = base / rel + if not p.exists(): + return HTML( + f"

Figure not found (optional): {rel}

" + ) + b64 = base64.b64encode(p.read_bytes()).decode("ascii") + return HTML( + f'
' + f'{rel}' + f"
" + ) + + +def display_table(df: pd.DataFrame) -> None: + """Render table without row index for consistent report UX.""" + display(HTML(df.to_html(index=False, border=0))) + + +def display_kv_table(d: dict) -> None: + """Render dict as key/value table without row index.""" + kv = pd.DataFrame( + [{"Metric": str(k), "Value": v} for k, v in d.items()], + columns=["Metric", "Value"], + ) + display_table(kv) + + +def _status_theme(status: str) -> tuple[str, str, str]: + s = str(status).strip().upper() + if s in {"FAIL", "CRITICAL"}: + return ( + "background:#FEF2F2;border:1px solid #FECACA;color:#7F1D1D;", + "background:#DC2626;color:#FFFFFF;", + s, + ) + if s in {"WARN", "WARNING"}: + return ( + "background:#FFF7ED;border:1px solid #FED7AA;color:#7C2D12;", + "background:#EA580C;color:#FFFFFF;", + "WARN", + ) + if s in {"PASS", "GOOD"}: + return ( + "background:#F0FDF4;border:1px solid #BBF7D0;color:#14532D;", + "background:#16A34A;color:#FFFFFF;", + "PASS", + ) + return ( + "background:#F8FAFC;border:1px solid #E2E8F0;color:#334155;", + "background:#64748B;color:#FFFFFF;", + s if s else "N/A", + ) + + +def _status_cell_css(status: str) -> str: + s = str(status).strip().upper() + if s in {"FAIL", "CRITICAL"}: + return "background-color:#FEE2E2;color:#991B1B;font-weight:700;" + if s in {"WARN", "WARNING"}: + return "background-color:#FFEDD5;color:#9A3412;font-weight:700;" + if s in {"PASS", "GOOD"}: + return "background-color:#DCFCE7;color:#166534;font-weight:700;" + return "" + + +def metric_cell_with_caveat(name: str, status: str, body_md: str) -> str: + """Wrap a metric label in an inline
when status is WARN/FAIL.""" + import re as _re + + s = str(status).strip().upper() + if s not in ("WARN", "WARNING", "FAIL", "CRITICAL"): + return name + is_fail = s in ("FAIL", "CRITICAL") + icon = "✕" if is_fail else "⚠" + fg = "#7F1D1D" if is_fail else "#9A3412" + bg = "#FEE2E2" if is_fail else "#FFF7ED" + border = "#DC2626" if is_fail else "#F97316" + chip_bg = "#FCA5A5" if is_fail else "#FED7AA" + body_html = _re.sub(r"\*\*(.+?)\*\*", r"\1", body_md) + chip = ( + f"" + f"{icon}▾" + f"" + ) + return ( + f"
" + f"" + f"{name}{chip}" + f"" + f"
" + f"{body_html}" + f"
" + ) + + +_SUMMARY_ROW_CSS = ( + "border-top:2px solid #94A3B8; background:#F1F5F9; font-weight:600;" +) + + +def _summary_row_styles(row) -> list[str]: + """Apply 'summary row' visual treatment when first cell contains 'Overall'.""" + label = str(row.iloc[0]) if len(row) else "" + if "Overall" in label: + return [_SUMMARY_ROW_CSS] * len(row) + return [""] * len(row) + + +def display_status_table( + df: pd.DataFrame, + status_cols: list[str], + *, + narrow_cols: dict | None = None, + scroll_max_height: str | None = None, +) -> None: + """Render DataFrame without index, with status-coloured cells. + + `scroll_max_height` (e.g. "600px"): wrap the rendered table in a + fixed-height scrollable container so long tables (e.g. §3.1 per-FoV) + cap their vertical footprint. Wrapping the table HTML inline (rather + than emitting separate `
` / `
` HTML displays) is required + because Quarto's render closes orphaned tags before the table arrives. + """ + sty = ( + df.style.set_table_styles( + [ + { + "selector": "th", + "props": "text-align:left; vertical-align:top; background-color:#F8FAFC; white-space:normal;", + }, + { + "selector": "td", + "props": "text-align:left; vertical-align:top; white-space:normal; line-height:1.5; overflow-wrap:anywhere; word-break:break-word;", + }, + { + "selector": "td details[open]", + "props": "white-space:normal;", + }, + { + "selector": "td details[open] > div", + "props": "white-space:normal;", + }, + ] + ).hide(axis="index") + ) + for col in status_cols: + if col in df.columns: + styles = [_status_cell_css(v) for v in df[col]] + sty = sty.apply(lambda _col, s=styles: s, subset=[col]) + sty = sty.apply(_summary_row_styles, axis=1) + if narrow_cols: + for _col, _width in narrow_cols.items(): + if _col in df.columns: + sty = sty.set_properties( + subset=[_col], + **{ + "width": _width, + "max-width": _width, + "white-space": "normal", + "overflow-wrap": "anywhere", + "word-break": "break-word", + }, + ) + _table_html = sty.to_html() + if scroll_max_height: + _table_html = ( + f'
{_table_html}
' + ) + display(HTML(_table_html)) + + +def render_status_note(title: str, status: str, subtitle: str = "") -> None: + """Render a compact status note with themed background + badge.""" + box_css, badge_css, label = _status_theme(status) + sub_html = ( + f"
{subtitle}
" + if subtitle + else "" + ) + display( + HTML( + f"
" + f"
" + f"{title}" + f"{label}" + f"
" + f"{sub_html}" + f"
" + ) + ) + + +def _worst_status(labels: list[str]) -> str: + """Worst-of aggregation: FAIL > WARN > PASS; N/A skipped.""" + rank = {"PASS": 0, "WARN": 1, "FAIL": 2} + vals = [str(x).upper() for x in labels if str(x).upper() in rank] + if not vals: + return "N/A" + return max(vals, key=lambda u: rank[u]) + + +# ----------------------------------------------------------------------------- +# Section-summary builders. Each returns a list of dicts: +# {Metric, Status, Value, Cutoffs, Detail} +# Both the §X.0 summary tables and the §1 at-a-glance scorecard read +# from these — single source of truth for verdicts. +# ----------------------------------------------------------------------------- + + +def _status_higher_better(value, pass_min, warn_min): + """3-tier verdict for higher-is-better metrics (yield, decoded gene, qv). + + `warn_min` may be None for advisory rows (PASS/WARN only, no FAIL tier); + in that case values below `pass_min` get WARN, never FAIL. + """ + if value is None or pass_min is None: + return "N/A" + if value >= pass_min: + return "PASS" + if warn_min is None: + return "WARN" + if value >= warn_min: + return "WARN" + return "FAIL" + + +def _status_lower_better(value, pass_max, warn_max): + """3-tier verdict for lower-is-better metrics (unassigned, neg ctrl).""" + if value is None or pass_max is None: + return "N/A" + if value < pass_max: + return "PASS" + if warn_max is None: + return "WARN" + if value < warn_max: + return "WARN" + return "FAIL" + + +def _link(anchor: str, label: str) -> str: + return f'{label}' + + +def build_section_2_summary(qc_metrics, qc_cutoffs): + """§2 Sample-level — status-bearing rows (yield, quality, codeword + categories, transcripts-not-assigned-to-cell, noise filter).""" + mol_cfg = qc_cutoffs.get("transcript_qc", {}) or {} + rows = [] + + # Row 1: Transcript yield + y_cfg = mol_cfg.get("yield", {}) or {} + tm = qc_metrics.get("total_transcripts") + y_status = _status_higher_better( + tm, y_cfg.get("total_transcripts_pass_min"), y_cfg.get("total_transcripts_warn_min") + ) + rows.append({ + "Metric": "Transcript yield", + "Status": y_status, + "Value": f"{tm:,}" if tm is not None else "—", + "Cutoffs": ( + f"PASS ≥ {int(y_cfg.get('total_transcripts_pass_min', 0)):,}; " + f"WARN ≥ {int(y_cfg.get('total_transcripts_warn_min', 0)):,}" + ) if y_cfg else "—", + "Detail": _link("sec-2-1", "Section 2.1"), + }) + + # Row 2: Transcript quality (pct_above_qv20) — 3-tier + q_cfg = mol_cfg.get("qv", {}) or {} + pq = qc_metrics.get("pct_transcripts_above_qv20") + q_status = _status_higher_better( + pq, q_cfg.get("pct_above_qv20_pass_min"), q_cfg.get("pct_above_qv20_warn_min") + ) + rows.append({ + "Metric": "Transcript quality", + "Status": q_status, + "Value": f"{pq:.1f}% above QV 20" if pq is not None else "—", + "Cutoffs": ( + f"PASS ≥ {q_cfg.get('pct_above_qv20_pass_min'):.0f}%; " + f"WARN ≥ {q_cfg.get('pct_above_qv20_warn_min'):.0f}%" + ) if q_cfg.get('pct_above_qv20_pass_min') is not None else "—", + "Detail": _link("sec-2-3", "Section 2.3"), + }) + + # Rows 3-5: codeword categories + cw_thresh = mol_cfg.get("codeword_categories", {}) or {} + counts = qc_metrics.get("codeword_category_counts", {}) or {} + total = sum(counts.values()) if counts else 0 + + combined_cfg = cw_thresh.get("decoded_gene_combined") + if combined_cfg and total > 0: + contribs = combined_cfg.get("contributors", []) + cc = sum(int(counts.get(c, 0)) for c in contribs) + if cc > 0: + cp = 100.0 * cc / total + cs = _status_higher_better(cp, combined_cfg.get("pass_min"), combined_cfg.get("warn_min")) + cv = f"{cp:.2f}%" + else: + cs = "N/A" + cv = "—" + rows.append({ + "Metric": combined_cfg.get("display_label", "Decoded gene (combined)"), + "Status": cs, + "Value": cv, + "Cutoffs": f"PASS ≥ {combined_cfg['pass_min']:.0f}%; WARN ≥ {combined_cfg['warn_min']:.0f}%", + "Detail": _link("sec-2-2", "Section 2.2"), + }) + + for ck in ("unassigned_codeword", "negative_control_probe"): + cfg = cw_thresh.get(ck) + if not cfg: + continue + if ck in counts and total > 0: + p = 100.0 * int(counts[ck]) / total + s = _status_lower_better(p, cfg.get("pass_max"), cfg.get("warn_max")) + v = f"{p:.3f}%" + else: + s, v = "N/A", "—" + rows.append({ + "Metric": cfg.get("display_label", ck), + "Status": s, + "Value": v, + "Cutoffs": f"PASS < {cfg['pass_max']:.0f}%; WARN < {cfg['warn_max']:.0f}%", + "Detail": _link("sec-2-2", "Section 2.2"), + }) + + # Transcripts not assigned to a cell (SegTraQ Phase A). ADVISORY WARN-only + # (2026-06-29): re-gated at pass_max=30 after deduping the calibration cohort + # to 164 physical samples dropped pooled p95 from ~59% (replicate-inflated, + # non-deduped) to 39% / p99 47%. 30 sits just above the pancreas median (29), + # so expect occasional pancreas WARNs by design; strongly tissue-driven, so + # no FAIL tier (warn_max null -> WARN never FAIL). Falls back to informational + # if pass_max is unset. Distinct from the "Unassigned codeword" row above: + # that is a decoding metric (~0.001%), this is the segmentation cell-assignment + # rate (tens of %). cell_id == "UNASSIGNED". Per-gene detail is in §4.1. + ta = qc_metrics.get("transcript_assignment") or {} + pct_unassigned = ta.get("pct_transcripts_unassigned_to_cell") + ta_cfg = mol_cfg.get("transcript_assignment", {}) or {} + ta_pass = ta_cfg.get("pct_unassigned_pass_max") + ta_warn = ta_cfg.get("pct_unassigned_warn_max") + ta_status = _status_lower_better(pct_unassigned, ta_pass, ta_warn) + rows.append({ + "Metric": "Transcripts not assigned to a cell", + "Status": ta_status, + "Value": f"{pct_unassigned:.1f}%" if pct_unassigned is not None else "—", + "Cutoffs": ( + f"PASS < {ta_pass:.0f}%; WARN ≥ {ta_pass:.0f}% (advisory, no FAIL)" + if ta_pass is not None + else "informational — strongly tissue-driven" + ), + "Detail": _link("sec-4-1", "Section 4.1"), + }) + + # % panel genes filtered out by the negative-control noise bound + # (drives §2.4 verdict). Computed from total_genes_count and + # filtered_genes_count (both added in the Phase-0 schema extension). + nf_cfg = mol_cfg.get("noise_filter", {}) or {} + total_g = qc_metrics.get("total_genes_count") + filt_g = qc_metrics.get("filtered_genes_count") + if total_g is not None and total_g > 0 and filt_g is not None: + pct_filt = 100.0 * filt_g / total_g + nf_status = _status_lower_better( + pct_filt, + nf_cfg.get("pct_filtered_pass_max"), + nf_cfg.get("pct_filtered_warn_max"), + ) + nf_value = f"{pct_filt:.1f}% ({filt_g:,} of {total_g:,} genes)" + else: + nf_status, nf_value = "N/A", "—" + nf_cut = ( + f"PASS < {nf_cfg.get('pct_filtered_pass_max', 50):.0f}%; " + f"WARN < {nf_cfg.get('pct_filtered_warn_max', 80):.0f}%" + if nf_cfg + else "—" + ) + rows.append({ + "Metric": "% genes filtered out (noise)", + "Status": nf_status, + "Value": nf_value, + "Cutoffs": nf_cut, + "Detail": _link("sec-2-4", "Section 2.4"), + }) + + # 10x-defined bundle metrics (metrics_summary.csv). N/A when absent / + # resegmented. Catastrophic-floor gates only. + bm = qc_metrics.get("bundle_metrics") or {} + + # Nuclear transcripts per 100 um² — capture density (higher is better). + nd_cfg = mol_cfg.get("nuclear_density", {}) or {} + nd_val = bm.get("nuclear_transcripts_per_100um2") + nd_pass = nd_cfg.get("per_100um2_pass_min") + nd_warn = nd_cfg.get("per_100um2_warn_min") + nd_status = _status_higher_better(nd_val, nd_pass, nd_warn) + rows.append({ + "Metric": "Nuclear transcripts per 100 µm²", + "Status": nd_status, + "Value": f"{nd_val:.1f}" if isinstance(nd_val, (int, float)) else "—", + "Cutoffs": ( + f"PASS ≥ {nd_pass:.0f}; WARN ≥ {nd_warn:.0f}" + if nd_pass is not None and nd_warn is not None + else "—" + ), + "Detail": _link("sec-2", "Section 2"), + }) + + # Cells with zero transcripts — derived in-house from cells.parquet + # (transcript_counts == 0), not the metrics_summary.csv. warn 5 / fail 10. + ec_cfg = mol_cfg.get("empty_cells", {}) or {} + ec_pct = qc_metrics.get("pct_cells_zero_transcripts") + ec_pass = ec_cfg.get("pct_pass_max") + ec_warn = ec_cfg.get("pct_warn_max") + ec_status = _status_lower_better(ec_pct, ec_pass, ec_warn) + rows.append({ + "Metric": "Cells with zero transcripts", + "Status": ec_status, + "Value": f"{ec_pct:.2f}%" if isinstance(ec_pct, (int, float)) else "—", + "Cutoffs": ( + f"PASS < {ec_pass:.0f}%; WARN {ec_pass:.0f}-{ec_warn:.0f}%; FAIL ≥ {ec_warn:.0f}%" + if ec_pass is not None and ec_warn is not None + else "—" + ), + "Detail": _link("sec-2", "Section 2"), + }) + + return rows + + +def build_section_3_summary(qc_metrics, qc_cutoffs): + """§3 FoV-level — QV-consistency row + transcript-density dropout row.""" + mol_cfg = qc_cutoffs.get("transcript_qc", {}) or {} + fov_cfg = mol_cfg.get("fov_consistency", {}) or {} + elig = qc_metrics.get("fov_eligible_count") + pf = qc_metrics.get("fov_pct_fail") + pw = qc_metrics.get("fov_pct_warn") + out_count = qc_metrics.get("fov_outlier_count") + warn_count = qc_metrics.get("fov_warn_count") + + if elig is None or elig < 2: + status = "N/A" + value = "Single FoV — consistency not computable" if elig == 1 else "—" + else: + f_warn = fov_cfg.get("sample_pct_fail_warn", 5.0) + f_fail = fov_cfg.get("sample_pct_fail_fail", 10.0) + w_warn = fov_cfg.get("sample_pct_warn_warn", 15.0) + if pf is not None and pf > f_fail: + status = "FAIL" + elif (pf is not None and pf > f_warn) or (pw is not None and pw > w_warn): + status = "WARN" + else: + status = "PASS" + value = ( + f"{out_count or 0} FAIL + {warn_count or 0} WARN of {elig} eligible " + f"({pf:.1f}% / {pw:.1f}%)" + ) + + rows = [{ + "Metric": "FoV consistency", + "Status": status, + "Value": value, + "Cutoffs": "PASS ≤ 2 % FAIL; WARN > 2 %; FAIL > 5 %", + "Detail": _link("sec-3-1", "Section 3.1"), + }] + + # Transcript-density dropouts (SegTraQ Phase A, advisory). FoVs carrying far + # fewer transcripts than their neighbours. N/A under subsampling (per-FoV + # counts unreliable) or with too few eligible FoVs. Provisional threshold. + fd_cfg = mol_cfg.get("fov_density", {}) or {} + pct_dropout = qc_metrics.get("fov_density_pct_dropout") + dropout_count = qc_metrics.get("fov_density_dropout_count") + fd_pass = fd_cfg.get("sample_pct_dropout_pass_max") + if pct_dropout is None or elig is None or elig < 2: + d_status, d_value = "N/A", "—" + else: + d_status = _status_lower_better( + pct_dropout, fd_pass, fd_cfg.get("sample_pct_dropout_warn_max") + ) + d_value = f"{dropout_count or 0} of {elig} eligible FoVs ({pct_dropout:.1f}%)" + rows.append({ + "Metric": "FoV transcript-density dropouts", + "Status": d_status, + "Value": d_value, + "Cutoffs": ( + f"PASS < {fd_pass:.0f}% (provisional)" if fd_pass is not None else "—" + ), + "Detail": _link("sec-3-1", "Section 3.1"), + }) + return rows + + +def build_section_4_summary(qc_metrics, qc_cutoffs): + """§4 Feature-level — gene vs control transcript-count shift (advisory).""" + log2_ratio = qc_metrics.get("log2_gene_vs_control_median_ratio") + median_gene = qc_metrics.get("median_transcripts_per_gene_feature") + median_ctrl = qc_metrics.get("median_transcripts_per_control_feature") + + if log2_ratio is None: + status, value = "N/A", "—" + elif log2_ratio > 2.0: + status, value = "PASS", f"log2 ratio = {log2_ratio:.2f} ({2 ** log2_ratio:.1f}× separation)" + else: + status, value = "WARN", f"log2 ratio = {log2_ratio:.2f} ({2 ** log2_ratio:.1f}× separation — gene/ctrl too close)" + + rows = [{ + "Metric": "Gene vs control transcript-count shift", + "Status": status, + "Value": value, + "Cutoffs": "PASS > 2.0 (genes ≥ 4× median count of controls)", + "Detail": _link("sec-4-1", "Section 4.1"), + }] + # Informational reference rows + if median_gene is not None: + rows.append({ + "Metric": "Median transcripts per gene feature", + "Status": "—", + "Value": f"{median_gene:,.1f}", + "Cutoffs": "metric is informational only", + "Detail": _link("sec-4-1", "Section 4.1"), + }) + if median_ctrl is not None: + rows.append({ + "Metric": "Median transcripts per control feature", + "Status": "—", + "Value": f"{median_ctrl:,.1f}", + "Cutoffs": "metric is informational only", + "Detail": _link("sec-4-1", "Section 4.1"), + }) + # Per-gene spread of transcripts-not-assigned-to-cell (SegTraQ Phase A, + # informational). A high CV means a few genes drive the unassigned rate + # rather than it being even across the panel. Absent under subsampling. + cv = (qc_metrics.get("transcript_assignment") or {}).get("unassigned_per_gene_cv") + if cv is not None: + rows.append({ + "Metric": "Per-gene unassigned-to-cell spread (CV)", + "Status": "—", + "Value": f"{cv:.2f}", + "Cutoffs": "metric is informational only", + "Detail": _link("sec-4-1", "Section 4.1"), + }) + return rows + + +def build_section_5_summary(qc_metrics, qc_cutoffs): + """§5 Cell-level — 6 advisory threshold rows + 4 informational rows.""" + mol_cfg = qc_cutoffs.get("transcript_qc", {}) or {} + rows = [] + + # Cell yield (advisory PASS/WARN — flags implausibly low cell counts). + # A standard Xenium section yields ~100k–700k cells; a small biopsy/TMA + # core can legitimately fall below the floor, so this is advisory only. + tc = qc_metrics.get("total_cells") + _cy_warn = int((mol_cfg.get("cell_yield", {}) or {}).get("warn_below", 50000)) + if tc is None or tc == 0: + _cy_status = "N/A" + elif tc < _cy_warn: + _cy_status = "WARN" + else: + _cy_status = "PASS" + rows.append({ + "Metric": "Cell yield", + "Status": _cy_status, + "Value": f"{tc:,} cells" if tc else "—", + "Cutoffs": f"PASS if ≥ {_cy_warn:,}; WARN below (tissue-area dependent)", + "Detail": _link("sec-5-1", "Section 5.1"), + }) + + # §5.3 nucleus-to-cell ratio (calibrated thresholds — advisory) + seg_cfg = mol_cfg.get("segmentation", {}) or {} + ratio = qc_metrics.get("nucleus_to_cell_area_ratio") or {} + median_r = ratio.get("median") + pct80 = ratio.get("pct_cells_above_80pct") + if median_r is not None: + pmin = seg_cfg.get("nucleus_to_cell_median_pass_min", 0.10) + pmax = seg_cfg.get("nucleus_to_cell_median_pass_max", 0.60) + rs = "PASS" if pmin <= median_r <= pmax else "WARN" + rows.append({ + "Metric": "Median nucleus/cell area ratio", + "Status": rs, + "Value": f"{median_r:.3f}", + "Cutoffs": f"PASS [{pmin:.2f}, {pmax:.2f}]", + "Detail": _link("sec-5-3", "Section 5.3"), + }) + if pct80 is not None: + p80_pass = seg_cfg.get("nucleus_to_cell_pct_above_80_pass_max", 10.0) + p80s = "PASS" if pct80 < p80_pass else "WARN" + rows.append({ + "Metric": "% cells with nucleus/cell ratio > 0.8", + "Status": p80s, + "Value": f"{pct80:.2f}%", + "Cutoffs": f"PASS < {p80_pass:.0f}%", + "Detail": _link("sec-5-3", "Section 5.3"), + }) + + # §5.2 nucleus RNA fraction (new — advisory) + nmf = qc_metrics.get("nucleus_transcript_fraction_summary") or {} + nmf_median = nmf.get("median") + nmf_pct60 = nmf.get("pct_cells_above_60pct") + nmf_cfg = mol_cfg.get("nucleus_transcript_fraction", {}) or {} + if nmf_median is not None: + npmin = nmf_cfg.get("median_pass_min", 0.05) + npmax = nmf_cfg.get("median_pass_max", 0.70) # PROVISIONAL (see YAML) + ns = "PASS" if npmin <= nmf_median <= npmax else "WARN" + rows.append({ + "Metric": "Median nucleus RNA fraction", + "Status": ns, + "Value": f"{nmf_median:.3f}", + "Cutoffs": f"PASS [{npmin:.2f}, {npmax:.2f}]", + "Detail": _link("sec-5-2", "Section 5.2"), + }) + if nmf_pct60 is not None: + np60_pass = nmf_cfg.get("pct_above_60_pass_max", 50.0) # PROVISIONAL (see YAML) + n60s = "PASS" if nmf_pct60 < np60_pass else "WARN" + rows.append({ + "Metric": "% cells with nucleus RNA fraction > 0.6", + "Status": n60s, + "Value": f"{nmf_pct60:.2f}%", + "Cutoffs": f"PASS < {np60_pass:.0f}%", + "Detail": _link("sec-5-2", "Section 5.2"), + }) + + # §5.1 cell size (new — advisory) + cs_cfg = mol_cfg.get("cell_size", {}) or {} + mcs = qc_metrics.get("median_cell_size") + if mcs is not None: + cs_pmin = cs_cfg.get("median_pass_min", 30.0) + cs_pmax = cs_cfg.get("median_pass_max", 500.0) + cs_s = "PASS" if cs_pmin <= mcs <= cs_pmax else "WARN" + rows.append({ + "Metric": "Median cell size (px²)", + "Status": cs_s, + "Value": f"{mcs:.0f}", + "Cutoffs": f"PASS [{cs_pmin:.0f}, {cs_pmax:.0f}] px²", + "Detail": _link("sec-5-1", "Section 5.1"), + }) + + # Cells with no in-plane nucleus (SegTraQ Phase A). nucleus_count == 0 over + # cells with total_counts > 0. Expected to be nonzero in thin 2D sections — + # the nucleus often sits in an adjacent z-plane. Carries a FAIL tier + # (calibrated to the qc_threshold_refinement report) — the one cell-level + # row that is not advisory-only. + nc_cfg = mol_cfg.get("nucleus_coverage", {}) or {} + pct_no_nuc = (qc_metrics.get("cells_without_nucleus") or {}).get( + "pct_cells_no_nucleus" + ) + if pct_no_nuc is not None: + nc_pass = nc_cfg.get("pct_cells_no_nucleus_pass_max") + nc_warn = nc_cfg.get("pct_cells_no_nucleus_warn_max") + nc_status = _status_lower_better(pct_no_nuc, nc_pass, nc_warn) + if nc_pass is not None and nc_warn is not None: + nc_cutoffs = ( + f"PASS < {nc_pass:.0f}%, WARN {nc_pass:.0f}-{nc_warn:.0f}%, " + f"FAIL ≥ {nc_warn:.0f}%" + ) + elif nc_pass is not None: + nc_cutoffs = f"PASS < {nc_pass:.0f}%" + else: + nc_cutoffs = "—" + rows.append({ + "Metric": "Cells with no in-plane nucleus", + "Status": nc_status, + "Value": f"{pct_no_nuc:.1f}%", + "Cutoffs": nc_cutoffs, + "Detail": _link("sec-5-2", "Section 5.2"), + }) + + # Informational rows (no status pill) + mm = qc_metrics.get("median_transcripts_per_cell") + mn = qc_metrics.get("min_transcripts_per_cell") + if mm is not None: + rows.append({ + "Metric": "Median transcripts per cell", + "Status": "—", + "Value": f"{mm:,}" + (f" (vs noise floor min: {mn:,})" if mn is not None else ""), + "Cutoffs": "metric is informational only", + "Detail": _link("sec-5-4", "Section 5.4"), + }) + mg = qc_metrics.get("median_genes_per_cell") + gn = qc_metrics.get("min_genes_per_cell") + if mg is not None: + rows.append({ + "Metric": "Median genes per cell", + "Status": "—", + "Value": f"{mg:,}" + (f" (vs noise floor min: {gn:,})" if gn is not None else ""), + "Cutoffs": "metric is informational only", + "Detail": _link("sec-5-5", "Section 5.5"), + }) + + # Cells segmented by stain (10x metrics_summary.csv). Advisory (WARN only), + # and only when stain segmentation was actually used (stain_definition + # present). A nucleus-expansion run reports stain_frac = 0 by design — shown + # as informational, never WARN. This is a segmentation-config signal, not + # XOA-version drift. + bm5 = qc_metrics.get("bundle_metrics") or {} + sf_frac = bm5.get("segmented_cell_stain_frac") + sf_def = bm5.get("stain_definition") + if isinstance(sf_frac, (int, float)): + sf_pct = sf_frac * 100.0 + if sf_def: + sg_cfg = mol_cfg.get("segmentation_stain", {}) or {} + sf_status = _status_higher_better( + sf_pct, + sg_cfg.get("stain_frac_pass_min"), + sg_cfg.get("stain_frac_warn_min"), + ) + sf_cut = ( + f"PASS ≥ {sg_cfg.get('stain_frac_pass_min'):.0f}% (advisory; no FAIL)" + if sg_cfg.get("stain_frac_pass_min") is not None + else "—" + ) + else: + sf_status = "—" + sf_cut = "informational (nucleus-expansion segmentation; no stain used)" + rows.append({ + "Metric": "Cells segmented by stain", + "Status": sf_status, + "Value": f"{sf_pct:.1f}%", + "Cutoffs": sf_cut, + "Detail": _link("sec-5", "Section 5"), + }) + + return rows + + +# ============================================================================= +# === END HELPERS === +# ============================================================================= +``` + +```{python sample-id-banner} +#| echo: false + +from datetime import datetime, timezone + +# Title + Date + Sample banner — title-block-style is suppressed so +# Quarto does not render its built-in "Published ..." block; we render +# the equivalent here in the image-QC format ("Date: YYYY-MM-DD"). +# Must remain the immediate predecessor of the §1 heading (no Python +# cells between this and §1) so Quarto's declaration-order execution +# places the banner above the scorecard. +# Bioinformatics Innovation Hub logo — inlined as base64 (same asset as the +# image QC report) so it travels inside the standalone HTML. The Quarto process +# only stages the qmd + INDIR, not notebooks/assets/, so a relative +# would 404 in the render work dir. Keep this in sync with +# notebooks/xenium_image_qc_report.qmd. +_LOGO_B64 = "iVBORw0KGgoAAAANSUhEUgAAAZAAAAHDCAYAAAAKpUyVAAEAAElEQVR42uz9d3xd53UmjD7rfXc5/aA3AixgFYskSpRkiVazJVe5xB7bsf3ZmcQp43wzzr3fzZ07+c1kJpPMxPPNJJl8KU6bxLHjeBLJtlwlW7JVLMmiWCT2ApIg0TtOb7u86/6x9z44AAEQbBJFncUffgCB0/be717P+6zyLGJmvGWNXYAkrPw5nj75Fyhl+tB9xx/ATGwgsAJIoG51q1vd6ra4aW918ChOvsSTx/4HXLsMJgHlluqrom51q1vd6gCyPHgURp/iySP/FSzDIKFDhlpgxjcQAICovjrqVre61W0Ze+vFaALmMfFTnj78n0EyBJImXCePaMudEFoYzC6AOoDUrW51q1sdQKrgobycR/YUTx/6bUCYIJKAciCEgXjHAx75qINH3epWt7rVAaQGPQAAysnz1MHfBisLJHQADOWWYSY3I5TcTADXk+d1q1vd6lYHkFr88IBh5uSfwsqdgdCiXjgLAqwsxNrvBUiAWdVXRd3qVre61QEkAA+vJLc4/QpnBx+HNBrAyvH/5kJoUURadgGoh6/qVre61a0OIHPoAYDAysL0yT8DCVnzN4JSNvRQO/RIt199VQ9f1a1udatbHUAALyRFhMzwD7icOQ4hox4jAbxSXXYhzaZqPqRudatb3epWBxAfIwTYrWB24DEIGQbjwhxHNZxVt7rVrW51qwOIxz68fo7c1M+4kjsLkqE59uE9ACQN2MUhuFaGAfKfU7e61a1udXtLA0iQEM+M/miJ5DiDSINrZTB98s8BVl5fCPzQVz2kVbe61a1uS9oNLGXile3a5Wkuzh6C0MLz2UeVhCgILYrc2DNwrBQ3rvsUIs23EgXJ9Gq+pJ5cr1vd6la3twSAsM8mirOvwbFmYehJgO0lHq0g9BhKqaMopv8TzOQWjrXvRrxtN/RQC81/TUJd5qRudatb3d4CYor52ddWiDgKpEUAkijnzqCQOYXpge8h1nonN3Tcj0jD5iorYVZeSKwuuFi3utWtDiA3nnm5DEY5exokjEWrrxYDEUCApAlN6lBuBemRZ5Ae/xlCyU3c0L4bidbboOnxNw0r4Tpfqlvd6lYHkEt3m04lzVZpHGLZHg8CSAJc42aZwXAB0iF1EwoCxcwZ5NNnoA8+iUTLTm5ouwvRxNp5rAQgH0yuIyCtr/G61a1udQC5FPxggAhWeQKunYMm9UUT6N5jXSi7CIgIoEUWJMu52ogoNBMCOhy7gOmR5zE9/goiifXc0HY7Gpq3Q9dj1x0rsRUwWVa8KiLqOFK3utWtDiAr4x8MAmCXxqHcCkgai+/NWUHocURaHkQldw7l/HmwMEBaDETSf5VaVqJApEHTQ3AhkM/2I5M5h7GhZ5Fo2spNLbcgkVwzx0rAXjHY68xKgrDVa7MuH5i28fY2jXc0aVQPZ9WtbnWrA8gKzSpPAlC+ZMkC+CAB1y4gtuq9aNn2/yXlllGc3svZsWdRSJ+AY2dBMgrIsNfNvhgrkSYE6XCcEqbG9mFq8jDCsVXc1LwNTc2bETIbqu0nwez51wNMyGcfQ0WFmE44knKwISER1mjJnAgzw3XrTZR1q9s1uy/JC3ELceO0BNzQAOJa6Yvu1UmYHhMRBmLt91Gs/T5U8oOcnXgB2cl9qJQmwCRBWtQDElqalSgIFArjyOXHMTK6F4nkWm5uvgmNydUka1gQM3upl2vABwKAyNjMZRfQBFC0GEMFlzcll0YQIoKmafW7vG51u8amlAIzQwhx3eVM6wBSCyBO8WJbAiin4OU92PVl3wlmbDW1xj6N5rUfRX7mEKcn96CQ7oNt5wARgtAi8MJU5DOTGlYidEihQ7GLmdnTmEqdgxlq4qbGXrQ2rkc81ka1i4aZ/Wrgq7uQMjbgMGCS98ojeRebktoFb+O9P2F6Zob/nz/+Y1iWNY8xeeFAmkdvgv8TUfX/3s/+XxY+ZgFQLQZelwyUzCv+f7Dzq/2M1f8v8vO8z0RzR7+SY7ks0Ge+5N8t9jP7m5ol/77U90Wet+Lvfpj2wu/BBuvir7308XA1clD7OvM3TDwvurBMqcyi123hdV5sjRARSBAEiSqDWPglpax+aZoGTdOg6zoMw4BpmgiFQ4jH4ti2bRt6enqqH8B1XUgp6wByPZonkkjLBnvYLdV4RlHDLBhChpBou4sSbXfBKk1yduYQMjNHUMqPwLHzABlzzKRaxeWDCSQ0zYQiDZaVx8j4IYxMnUI03MrNDavR0rAasXBDFUyudr4kbQU3KSAFYabMcBUgF7Bn13WhaRqefOIJ/Jf/8l/q28O61e0aWlNTE+677z7+7Gc/iw9+8IMkpYRSat4mpg4g14stKz/iSZ0op1DdgdQyE28Pwt4OHQQj3EYt3Q+jpfthlIvjnEv3ITN7AvncEBy7AJJRkNDmARb7QEQkoUsdLiTypVmki7M4P3kKiUgLtzT0oDXZhbARmZcvudLFlLEZwt+RCQIKDiNnK24wxaLJ9MnJSWiahkQiAcepqxPXrW5XdTPrs6disYhvf/vb+Pa3v4077riD/92/+3f4yEc+Qm9WNnJDA4iQJpYTRCSScCszYHarIooLGcqcI/fBhARCkQ4KRTrQ2nUfSsVxnp06hJnpoyiVs4A0ITWtJrxVG+ISkEKDIAkFQrowjZn8LM6On0ZjvI07G7vREm8hKeTcoqNLy5QQvNBVzmEI8iuaATiKkbMZDcEpoQuZiOM41a+61a1uV9+klIjH4wCA/fv346Mf/Sg+9rGP8Z/8yZ+go6OD3mwgckMDiNRiy+4ISBiwi6Owi8NsRHouMpFwMTAhhCMdtGpNB9pX3YeZqSM8MXkQxdIMGBqEFl4Q3vJCVUESXRO6ByasMJEZx3h2GrFQgjsautDd2EEh3Qy4kic9fxFWEuBC1mYuOh7zUD6AKAaKDi+7sKkuzVK3ul1zJhJUO0ajUQgh8Nhjj2H//v149NFHedeuXW8qELmhAUQzGi4SwpJQThHZgW+iZev/VZNIv1iZ3YVgomlhtHfeSW0dtyOd7ufJqWNIZ4dh20UIGYIQGmgeK6kBEwC6poNJQ9Eq4dTEWZyfHeeuhnb0NLYjboYJ1VyJ/9GX8fXTFQWHGfqC31suzwOaeWxNiGplyKWUGXqaYJd6FwVH/8aEEa7X13ujbCUbh+t9c3El1+JynrvU+Vg0QV/zs1IKSql5PyeTSQwMDOChhx7CE088wffcc8+bBkRuaADRQ61eaGqpBcIupB5DZvgHkOEublz381T1cMwrlHAPwCRgJRKNjRupsXEjiqUZnpw+ialUP0qVAlh4iXkiusB5BvkSQRKG1GArB2enRzCYnkFzrJG7Ek1oicbIlHOVVLwgdBXYWElVczjzwlS89ILftWsXNE1DJpN58znHpUqiFwv/LfbYFf7uUirILqfa7Eqd9FLXbLkKtUuq1lqkeup1RLkLruWyVXELrt/FKugu9dzXnouFPwcAsdQ1jsVi1VJeALBtG7FYDMViER/+8Ifx0ksv8caNG0kpdd33jNCNsouaf3U9FlHOneWzP/sVSCEgmCGhIMCo4jppUKSBScJ1yoi03o3GdZ9AuHFHDZDgklV3F5bmOm4FM6nzPD57Fun8NFwGhAxDSB2K/U8lNDBpACQUeV9EGhQkbAYUNJh6CC2RODrjcbRGImQuskPJO4p/MGJ5x6b8xawYRUvhzjYdt7TopNgLby2048eP89TUFIQQ1QXONaWYvERZ5kod2QU36UInvVjJ7CK/u6zvwXvVlmtewfeV/m25n68EbJY715cKChf7CtZC7ffg65JBZQEQXHCdaY7VEhYvr15pOfbCx6302i91/pc6Z4oZrBRc161+2bYN27ZRLpdRLBaRy+UwPj6OH/zgB9i7dy8ikUj1Pqvu5jUN2WwWd955J55//nnSdf267xW5MQHED9K4do5Pv/hZsJWGFAKSlwIQASYDyi2DhYlQ4y1oXP0hxFpupyogAZcxVGou8R5YtjDJYzPnMJUdQ8WugMmAEAYgdSjIeQACSA9UhAbFEi4EHAgokgjrIbSEI+iIRtAcMhDVJVVcxs+mKjxWBnSpwXXnA8juDh1bm5YGkLrVrW7X1lzXxZ/8yZ/wb/3Wb3lREl2fByK6riOTyeC3f/u38bu/+7vXfSjrBgWQORDp3/PrXE4dhqaFINldGkCggYXusRHXAkMi2noXWtZ+FOHEeqplN1zdoazcCy8szbXsMiYzIzyeGka6mIUDgpQhCKFDkQcWAYAwSSj/ZxIeuLgsYbMAkwZDGojoOsqKUHAFNJ/N1DKQkq3wji4DvUmNeIkcSu2O8i1p10gd4E1597yF8jvX+nzV/i0oVvnRj37EH/vYx+A4TjX/GHzewFe8+uqr2LhxIwW5yTqAvK43gFeaO3rsDzk18CgMPQnB9vIAQr7DFjoYEq5TAckIYm1vQ2Pn/QgneskrDa59n0tT3l2sYTBdmOWx9BimcjMoWpYXvtJMCNLAASMJ2IjPToJwFwsNzBocEJg0CNI89lIDIKwYlqPwvtUmOiKyLqpYt7q9gcBs2zYMw8B3vvMd/shHPoJIJLJoKOuXf/mX8Td/8zfXNQu54QEkPfIkjxz6nUsDEJLeo4QBxeQxEmlCD3cgFFuDaMMWxBo2IxRpp4W7jEvZ8SxkJbZrYzo3y2PZKUwX86g4LkgYkFKH8vMhtQDCAahAA0jAJVk9jloAUS6DmPFz60KI6lQHkLrV7Q0227ah6zq+8IUv8J/+6Z/Oa+AlIriui3A4jCNHj2JVV9d1m1C/cUNYfiK9Uhjgcy/9UhU4hJ9pWBJAhOE5YPi7fghA6FAQUIrher3dID2OcGwNki23oKF5GwyzccE8kJVf7MVYScmuYDKX4pFsGjOlElwISGmAhAYX4gIAYfJyIwsBhJWC5TAadOBDa0P1Vo+61e06sIBxpFIp3rFjB2ZmZqDrenUjGrCQL33pS/j85z9PjuNcl2Kn4oa9Qr4DNyI9ZERXQylrBftuhnKKUE4Bjp0Ds1Mj5c4goUPTItB0r0Exn+nH4NnHcfzQn6O/7zHOpM8yMJc09zSxLg7QtdUmQTlvWDexpqmD7lm7he5ZsxHdiQYwAMtvQroUIHCZ0RwSnqo912/eutXtDXe8fgVWc3Mz/R+f+T9gWda8MFUAJE899VT18dflcdzIF8kLYwlEm3aC3cpFvS6zQtumX8aqW34bjaveDU1PwLXzUL6qb5DgYr8qS2gmdD0GVjZmpg7h1Il/xNEjX+aJiVfZcUpzir0+KKwI96o6XHOLqDkSo12r1tD9q9dhXbIBggiWoy7pXHRERP2urVvdrqc9ru9PPvqRj0LTtHl5EKUUNE3DkSNHUCwWuTbRfl0dw41cdROEkgrTe3lo329A1yIQ7C7aB8JMgBbGmt1/B6nHPXEzO8/5mdeQntiDQuY0HNcBaREIGYZLAuxXQXkhIwNMArZyoZhghprR0roD7W07YBrxebmSS5VvZ7+iLHhGwba5P5PHmUwBChqIlg5huUqBFOPDa8x6/qNudbuu/JOXAy0UCrx9+3YMDw/DNM3q75VS0HUdhw4dwtq1a6/LPMgN3YkehIXCjTtIj3Qzl8YBqS0Sx2GAdCg7j0q2D+GmWwFWkHqMkh33ItlxL0r5QU5PvIz09CFY5bQ3+lZG/BAXgVmBQRDSgCQdtl3E0MjLGJs6juamjdzechPi0dpZIOyPbl+BjARqZEwARHWddrQ0Im4YvHcyA10sHijzRBSB7rCog0fd6nadMpBoNEqdnZ18/vx5hEKhOabxJtjb3+Aj6AjMLoQMI9ZyJzLnH4UmEwDcxa4mmF3kJ19GpPl2fx56jWhibDWFY6vRtvoRTs8cxOzEfuRzw1CuBdIiIKEjSDIEUwp1acJxbYxOHMb49GnEY13c0tSL5kQ3Qmasih0rnQUS/FX5fRymlMuvMT/n0ZuQcwtykbdwXXdRCfmLsdPlPu/VfK1r8bwr3Tm+GV7zWr5u3ZZej5fTPW6a5gXrmcHVYVV1AHnDIMS7IPHOdyI7+K2lcxE+0OSnXkHz+s+wNJJUmxAPhkxJPUrNHbvR3LEb+Ww/z0y+hkyqD5VK3mMlWqj6HMU+kGhhMATS+XHM5iag6VHEo23ckuxGS7ILoQWzQC7W0ObhFOPoTBpimce5CkgahJ6opOB5i9mbeSLaG8ls61a3Je89110RkAQbt6UqrC5V3LQOIFf9bvfGKoUbbyYzvpGd/BlAmouWI5HQ4ZSnkB1/Do2rP+TnUGTVa88bMkUCsUQvxRK9sO08Z2ZPYWb2JHK5Udh2ESADwgcTVSPfzkJCsYvZ3DhmchM4O3EKjbF27mjsRnO8tToLpHZx1VogQ3JwapZnyxYMzfAOZcE6FQTYCtjRIKohLlpk4QLAt771LX7++ecxOTkJ27YhhKiO5NQ0rQowgRS167o1ekguXNf72XVdzM7OVhd9bWUZEUFKWW2kchwHyWSyqidERAiFQtXpbMHvA+nrgCXVajEBi3fPB5pddFlRAK5St6WmxC28oYPH8YLnrGRsbvBz8JpL/W6xxyz8qn1M9bFCzNOCEvO+z1UAzmlHAYLEij5T9TjEXOFH7bmp/X31PZc5jwtf90qZ18VEJIPrtVDvbanHK6XgKgXlunCVC+UquMoF2GMQLS0tuOmmm7B79250dHRc0kEsBiCLrac6gLwBxj4TiHc9jJkTRwEZXjyMxQokQ0iP/AjJVe+BkMYicZ+aBe6zEl2PUUv77Whpvx3lcoozmXOYSZ1FLj8B2y6CpAkpNDAFi1JAkzpAEo5yMJ4ewVhmAtFwktuTnehIdiAeitKizouAM6k0n0lnYUodaolQl8tATAc2J+QFKftaxdBPfvKT/Nhjj121c/3Zz34W73//+xGPx6HrnqC8N6TKBZHn8HO5HJ566il85Stfmffc9evXI5VKVROIwc0agFMAFq5SYKVQt7pdyzAUCBBCQgpRvW8c14VSLlgtvTVpaWnBr/7ar/Lv/KffoQAYLub8l2Ig1c9SZyBv5ILwLkC8693I9H8VUBUAcpG9J0NIE5X8EDJjz3Bj93tpTqpk8VgSLciVhEKNFAo1or39NhRLMzw9ewZTqXMolDJgUj4rISh/JgZBQJcaWHizQM5M9OPc9AgaY83c1dCOlmiSTM1zxLZy0Z9K87GZFIwlwCMIVVkKuLlRwpR0AQQqpSClxBe/+EV+7LHHqjv/pVjPSnZPxWKx2vS0kud86lOfwrZt2/jf/tt/i3A4DNu20d/fD9M0IISc9zlqKf7FFFPrIbK6XQ5bWe53tSoTwaZoKVbAzCgUCvj9//r7mJyY5L/5m7+hgFUv995LhZGv91np2ltjiRDAClqojaJt93Fh+NuQesOi4Q1mhtBCmBn4NhId97LUorRk9nkxVlIDJpFwM61e1Yzuzl1IZYd5fOYMZnPjsJwKhCRIIUEgTw6avVkgUtPggjCRn8VEPgNDD3PEiICEhoKtUHBcSKkvGZohP3TVbACbExfOPw9KAUdHR/mLX/wiNE2DbduXnWw1DAPFYhH/5t/8G3z+85+nIATGDGiaB9yDE3lmMFa3xchjP97x/uZv/iZ95zvf4ZdeeqlavmhZNkxzftx3qbBC3er2egL6YmGxpTZUuq7jueeeq845v9jGbFkAEXUAuW4sseajKI4+iaVBgSGEAbs0gZnzj6Ntw2ewLAtZAZgIIdHcsIaaG9agWM7w+Ox5TKaHUah4fRxSmtUhU0ESXZeeWKLjukiVCnAhQUKHLrTFgm8LjgC4vUlC0oU5gKBB6Y//+I+Ry+VgGMYlzUCvjYUTEcrlMj74wQ/ij//4j8lxHEhN88BDEmZzFf7aT87i5FAalu1i18Zm/uX3bSHvhvIYz1e/+lXceeedmJmZgaZpPohYME2zGp9+Kzinul0/7ONyHrPw8Uop6IaOlQohLgsgqAPIdXCnCoAVzORWCrfcyaWpn4H0hiUWgILUYpgd+iES7fdyKL6WVjbqdmkwCYAhEkpSb9ctWNOxDbO5CR5Pj2I6OwXLsSG1hVIGfoUGCQgIqEUmGdaaAFB0GVsbJDpCF/Z9MDOklJienua//du/rcopBI77YjdK7VChILn9i7/4i/jLv/xLL88iRDU5OzRV4L984hQmZkuIGBLCAH56eBx3bGrl2ze3EsPLc/T29tITTzzBn/jEJzAwMFDNf1QqlSqIXI8OxxtPzJf9GrWFAtfC6V3N5y3xYq9bm8KlroErmRK53P+X+rm2qKP277ZlLzud8GIAwssUctQB5I25/T0WsvZTKE29vPzjiMDKxsTpf8Ca2/4DrrQJb2EDoRQaWpOrqDW5CoVKns+M92EiNwNBctFPzRd9faCigLaQxK2N+qJNg67rQtM0/K//9b8wOzsLwzBgWdYlh6za2tpwxx134Fd+5Vfw3ve+d97b5EsOv9Y/i+/uGUSxbCMe1uE4LgQBuiT0DWdw++ZWgL2bxnEc3HnnnfS9732Pd+/ejVKpVE2aF4vFJUNZAQO6LDfGl+BkaOn93/IjVS+cvLfQISxWzXWxn6/W31by/4s64xXcEITlHWAVRGnpa3UlI3mX+/mikxj9SYMXA14p5bwGwOC7X7G4ItchpFiagdQB5HphId589FDLHRRq3sXl2YMgPbm4C2IFqUWQTx1FavQ5bux6kC5VZXdpVjKfZUTNGN2y5ja82PciFy1rrnR4xeAE2AyEJGF3q7lo6CpgH8Vikb/0pS9V2Uc4HMbnPvc5PPjgg2hubq4mCYOxnAHT0HUd0WgUTU1N6OjoQDQa9eRelIIgwnTO4u/vG8G5iTxSuQokAYYm4boMRzEIDCEIs9lycBqq8WKlFLZu3UqHDx/mSqVyQanwwhv7auyol7oxa+PNizm/eWWqgRPlC1/vYqW9gVepfc4Fr32R11rJ35b67JcCIFcycnclj1sJ850XhmVVXeCLleauBCgWG89b+xWsQcdx5o2otW0blUoFpVIJ+XweuVwOjz32GA4dOgTTNKuMI5BkD+6fizIQ8ebsxXrL5UDgy7Enen8BpZnXlt0aMCtIGcbkuW8h3nIba0ZiBQn1SwgnEEC+nuVoaoTLdtkbCHVJcAQoX1/r7e0RxHSxLPv4+te/jqGhIYTDYZRKJfzGb/wG/uAP/uCSDyi4MQImULIcDE4XMJOrQPigRgBcxWiIGbBtF8WSjXTBYzxiEce5Zs2aekKgbm86Gx4e5v379yMcDl8AICsNYS3s+bmSEF4dQK4pC/FyIeGWOynStpsLU68ARnLJGBEJHXZlBlODP0Dnhk9dYkJ9IXTxBbvVfCnD56fOYjQ9DhLGJcm0B/0eDODtHTG0hrRFwSMYiWnbNv7oj/5oXnJ6y5Yt1R1WwD5WsnNfGLNd3RKl//jxHTg2mOb9p2dwdCCFUtlB2VL4F/euwZq2KH7nKweQylZQrDgcMbVFK8RWcg4JVye5/lZLYL9VjvdqrI2VjKgNZnQUi8UlN1krZSBL9Xpc79fsLchA5iy54ZdQmNmPRVu559wapBZBZuIVtKx+H+tGwyWxkLlkuPCqKQheJ3p2nMdmBzCdm4atGFILgy9hsQgAts9idncm0BkxlhRLDNjHY489xidOnICu6/M6uYMywyuRNAnmrG9f00Db1zRgKlPmb754Hi8enUS+ZKO3M0F3bmnjFw+NYmgij009DV7YRdBFb6K61e16BUpN05ZkDUEobCVA9WZd+2/NO7ZakXUTxbreDWVnANIW3+OT8FnILKzixAp2ON68kKBM1duxe6c5V5zmc6Ov8v4TT/Dh/p9iMj0MANCljksR3RBEsJSCIQXu72pCZ8Sk5fKQQbL6i1/84gWL/WrpYAUvq/w+j9ZkiH7lvZtpdVsUh/pnAQAfuHs1hCD8aN/QgjxQ3ep249lKAGQlm6d6COv63KYAYDRt/GUUpvfBtfIgI+lhqj8fhJUF5VagGIi33I5wbPV8gcUqYABBye380l2FfHGKU5lBzGZGkCulvYSyNKDJEEDSC0Et5/0XhKwAQtlx0RSO4s7ONsR0PxS0xPMDmv29732PDx06BF3Xq4zkagJILbiBANdlSEnYva0N//uZfpwdzfL6rgQ9dHs3/+Bn57G2PcYffPs68m40L8Fet7rdSCyoNgdysUbCNysDeQuHsLzudGm2UPst/4nHD/4unMosWOhQkCAZhRlbi1ByM6JNNyPechuRL8xY7TaHJ0xXOyDKtvOcz48hnRlEJjeKQiXrVSoJEyRD0HUNimU1tHUpjtkGw1Uu1jc04ea2VpJEFw2mBQvzS1/60oqEAa8qPgNojJlwFeOpAyP4fFcCH3twPU4NpvAPT53C0GSOf/6hTWhOhKgOInW70axWc+5KGMj1zNLf0jmQakK98WbquecvOT/xIlw7B2m2IJTYCDPeS/Pk3NmtltgGzth1KygVxjiXHUQ2N4RCYQqWU4ZiAskQhDQhNBOAhAtcMnB43QSEsusiZOjY0daJ7niCgItnYgLZknPnzvELL7xQ/d3rufOpWC7CpoYj52YxNJnnnrYY/fIjW/m/f/0Ann11BCfOpfAv37+Fd21pJxXEg+vd2XW7ARjIpYSwLqaVVQeQ6xxENLOFGlZ/eOHVAysHJDS/MUwCYJQKo5zPnEU+cw6FwhgqVt7TdxIGIE1omgmGBgUB5YeycBn9I0TkjcglYE2yBVvbOimk6XM9BBd5fgAgzz33HCqVyjzZktcreTedq0AKQtlysefEJHraYtjYnaTf/Pmd/OffOoKxqQL+5/9+DZ94aBN/8N5eL6TFQbVaXeqjbm9eU1eJgdQB5M0AIvAYRjAnJJgwSKRBuWUUs/2cSx1HPn0a5eIUHGWBSQfJMIQwIaQOJgEXwp8voIDLcH5B85piBdd10BhtwKbWbrTFAtbBl6yNs3fv3td94QaOf2iyAEEAScKJgZTf4QtsXt1Iv/0Ld/BffecoDvZN4cvfP4YT52f4E+/cjLVdCarP3q3bDRDDuipJ9DqAXPcXOqiY0qrbetfOcTF9AvmZQ8inT6FSnvUk2GUIQoah6TEwJBQkOKi8Al12j2Ew8MdVLmzHRSycxLrmHvQ0tFIg2UG4NGG1YFH29fUtSYevxTTCoKR3NlfhoakCdE3AcRnpvIVC2eFYWCdXMVobw/Tvf+EO/HDPeX70x314+cgYDp2axi0bmnnX1nbcuqkNzQ1hcv38SB1T6vbmwA2uhrDqOZAbGjTYkzcJRtDaeS6mDiE3tReF9EnY5RQYAtDCEFoIAjoUguEyAWBc2c7Ba+oDHNeBgkIs3ICeph50N3ZUpxNeDuuoXZRTk1NLLsRrsfMJbqCD/bPIlWxETc0bxOOX+AJensOLVDHee/daamsM83//h/3QJGH/8XG8dHAEuhT4D79yN+/c0ka1wFS3ur1ZgGSlAFLPgbw5Limqela+42dlo5Q6xPnxF1CcfQ1WeQZMEpARCD0KQEJdYVjqgiCVzzYUK9iuBRIGmmKtWNW8Bu3JdgpGilb1jS4DPILn2raNfCF/UZC5mrkGIoKrGC+fmISuCTAYSgHRkI5ISJs3n10pAASUKg4UA/mSjbAu0dvdgFs3taGnI47XTkzwTb3NFDK1OojU7U1jdQZyI4FG0BToV1HZubNcmHgOhYkXUS4MeKkPLQahxwCSUCxqWMaVeazqe4Og2IXjOoBQCJsN6EquQntjD5KRRlro/AOnXrsQF5MRWc4cx4Ft25e1cC/vpvHCTQdOz/DAZAERQ4IVw3Jc9HYloElvRrzw5VSkIPxozwB/5YnjcF2Fd97Rg3e/bS3WdSVJSoLjKPzbfzzAqzsT/G8/9zaETY3qIFK3N4tdSQ7kSiT/6wByxfQxcLhzoOGUxrk49RJK48+gkj4K5ZTBWhRCiwDQoUBzoHFl4u3zWI6rbCjlAMKEaTagLd6FloYeNMbbSQoNSwFHAB5CLD6h72qwhquZA2H/M1VsF0/sG4ahCT9MBWiCcP8tndUHKngg8pUnT/D3XzwHVgq/+Mg2fMCvxPLAT0HTBD7wwAb83bcO46vfPoJ/9fM7/eOvI0jdrl+7VOdfT6JfVyEqWe0Wd+0cF6ZfQW78GVRmDoAr09CEBpJhSKPBr5pSYLhePuQqAIZSNly2wSQhtBhi0TYkEz1oSPQgEW0lKfX5YECLS4cH4DE4OMhPP/00stksbr/9dtx3331UCzgXW5hLzRq42gtXKY9RPLF/hMdSJUQNCWYgnbNw/44ObOhKeL0e5OVA/uFHp/gHL52HIMJ7dq/DB+7tJVd5ysKCCFIKMID33tdL+46O87N7B3HH9k6+fXtHvfGwbm+CTSxflT6QOgN5XYHDA4H87CFOjz2NwuTLcEsjEAB0aUIaDSAwwMor26XLYBtVfSvPwSlle18kIfQIwuEWRGOrEE+sRjzWhXCogRYuimDRLLVwAvD4xje+wZ/73OeQzWarf/voRz/KX/3qVxEKhehiTERKCV3TrzmAuD54HBvM8DOHJhA1NRAY+ZKDdR0xfOKBXmKGP0yL8PKxcf7+z84jbOpoShj41Lu2UBDaCg6HyAOlkKHh/jt6cPr8LJ5+6Rx2bmuv94fU7U0BILVSJnUAuS6Bg6thKtvK8OzYs5gd+RFKmZMgVYaumdD1pNfSx64HGpfLMOAnhF0LLlfA0CD0OMKRNkRiPYgm1iIWW4VQuJkWDp5aCWjUggcRYXh4mD/72c+iVCpB1/Uq4/jmN7+JTZs24fd///erWlfLAchif7+ajYQBeIylSvy1585BkwQpgHzJQVsyhP/zg1sRNjUoP+dh2S4ef/4swqZEueTg4+/cBtOQUIovLGrzT9X6ngbEogYGRjMYnypwV1uMVsLA6la3NxJAVur8lwQQ1JPo1+jiuD7jIJSLYzw2+D3MjP4YdnEUutChaSakZoJ80GBcioTIXMKdGWBlwXXLYOiQRgLhaBfC8XWIJnsRjvXADLXQwgVQO9McuLSxlMGc8n/+53+ugkeQCBdCQEqJr3/96/id3/kdGIaxaCirdkpdPB5fmoFIcQXXwCvDlYIwni7xXz91BmXLhaFJFEoWWhImvvChrWhOmBR0l4OAY+dTPDZbBAHYvKYRd23rIG9mCS12JQAAsYiBkCFRKtmYmimgqy1Wr8iq25uGgVzMltzIMeP1mz7/FgCQoAyXSKJcmuKhc9/A5MiP4FZmYWgh6EbSCyyx65XdXiJoeNImrqfCCwGhxWFGuxFObkKkYQsi8XUwwq20OBNamAi/Mu92+PDhecOfahfl2NgYxsbGeM2aNUvuxINZH729vXj11VerY2xrF+2llggHoEH+aF4C4ehQhh99aRCFkg1Tl6hYDuIRHf/mg1vRmpwTSnT94zh0ZhrMHnN58PZu77MyIJf5KMI/p0oxsgWr7p3qdl1bbQXlFTGQYONVB5CrxzqUsnH+3Ld46Nw3YJenYeoR6EYDiJ1LD1EFTEO5UHYZTBIy1IpwYgsizbci2rANRrR7fkiKlV9xJGpCXHN5mKu1+PL5/AX0Nfi/ZVnIZDIX3QEBwMc+9jE8+uijIKJ54SzHcdDY2Lhi0AjyEwHojMyW+KcnpvDq2RQIDEMTUIpRcRR+8eH1aGsIVZkFA1VNrINnphEyJIolG8mY4ZX0glGbQF9otquqTYiZXAX1KSJ1u9FDWAsrua7HkO2bBECCHb5EKn2Sjx/9c2TTx2BoERhGsia3cYlsAwA7Ba8vIdSBeNOtiLbejXDTLdDMJlrIfHyo8Hs6gHJxgnOZfuRzAyiVUnCUQjSxGqtXvxO6Hrni+emlUmkeECxcWJa1/E5c0zQopfCRj3yEPvWpT/HXv/71eX9/97vfje3bt1PAVC4EDXg6VjWgMZ2z+OxkAceGs+ifyKNsOTB1AVYKrqvADGiScGo4g5a4yY0xg0xDAgzM5iv8v398BqlcBSFdQNME/veP+gAG37yhhWRNCCsIebmKoWuEmXQR5YoDKQgTMwUQUAeRul33djUYyEpfow4gi0NHtcT17Llv88lTfwehbBhGQzW/QSt1JQFouGWwW4YwGhBuuRPRjnci0nInZC1o+LHHuZkfwSwQQmb6IE8MP4tScRK2awNCAygECB358UmUSils3fpJEGmXFaMPFlOgnHupi26xx/zjP/4jfehDH+If/ehHsCwLu3fvxi/+4i9SkENZHDQAy2WMpMp8dqqIc1NFTKTLKJRsEBiaIEQMDY6jAg1KMAO6FHjm0DhePDqBqKmzqRPAjJlMCbmChZChwXVd6JrA6EwB/+0f9mNNW4zv3NqO27a0Y3VHnDQpvHJfQXBdhadeOg8QYOoaTpyZRqlsIxzS68Oo6nZDA8hiG8g6gKyYAnr5DqUcvHb0T3lg8AmE9Tg0TbsM4GAoJw+wDSO6GtHOhxHpejeMWC9dGJryQWMee/DAwypP88DJv/cqhrQIdD0KJukNiSIJ00ggmx3E6Og+7u6+h6rSKddw8a10YX784x+nj3/844s9CMFUwwA0BmbL3DdZxPnpElIFG7bt5ZQMIkRNDa5y4bqMeTM8hE+6mBANaXBsF5mCBcfxJrNJYoQNDY7jVsHG0AV0SRiYyOHUYArf+MkZdLVEuLsthoaYCWbG+ZE0BkdzCBsSkgipTBlf+dYR/tWf30lCUB1E6vaG2tUowa2X8V4j8LDsPL/86u9jamofwkZDtX9jRe6iFjgARBt2oKHng4h2PAihRakKDMG8Dj80tfyFlpDSBATA8zrX/fdiF7oWxtj4AbS2bmfTTFxxKOtqmeu68xajkNLLa/j/H8/ZfHyiiNNTJczmLTjKAwwCICVBFxLliuvpD/P8cFex4sB1FZTrATGBoZHHRgR5Wlhck8eofa5iRkiXMDUBx2GMTOYxNJb1Z4IwTE0gbBq+FDwjEtax/8g4CoVX+Oc/sBWrOhIrbqysW91eLwC5FPZQB5BrAB7lSop/uvd3kMqcQtRsBCvnUi4rlFOAABBvvRtNaz6KWOs9NWzDxVzllVzR6zEzdLOR4k3beGZiH6SeWJQDEQlYTgkjY/vQu/adl1xuGiyYpfo7gr9fqgxJ8Hh/hHv1M/WnKvzaaBFDqTIqtscUTF0gxIRC2UVrTMf9mxrRHjdw4Hwazx2fhi49SAxyHu++tRPxsIZUzsJkuoSpTBljMwU4rvKICc9JnQhBYPayGEFz4VxfCkHTJYQpIfziBPL/TlVm5oHI6fOz+B9/9TIefnsvv+eB9SSlqLORul1nvuzqJdHrAHIJ4FGqpPjZV/4jstl+hI0klLKx8kAQQXEF8YYd6Nz4S4i17JoPHCsGjYWv6llbz7uQnj6y5KRBBkOTJqamT6Cr43YOhRoui4UYhrH8xdMu/fJVPwUBQ1mb944UcWamDIM8vaqILuC4CmBG0VJY3WTi47e1U0j3jvNt6xt5z+lZ2I6CEECl7OKjd3Xj7k3NtPB9hibz/Fc/OIVUrlItbS5ZDioVG4IATQhI6QGKVWEUSg5YAVFTQApRBRaxyA2mFCMU0kCK8b2n+3Di9BR/8sPbsaojUZc5qdt1Y/Uk+uuM1kQCFSvLP9n7O8hkzyOix6HYwcrdPYHZga4nseFtf+4NY2LXC6uQvAK9K3i5AlYIRTqoofU2nh5/BcJILvFQAdutYHTiEHrX3H9J4ZVgx5FIJC7oXA92JaFQCA0NDZd4fj3WUXIYL48U+fhUGSXLxdu6o7i5PYxvHJlBruxAEKHiKDRGNHzMBw9XcfX3Qb7EdhitCRO71jeSYo9NEOaYzeq2GK3tiPNEqoSQIZAve7Imuza1Ym1HHA0xA7omvc9UcTAwlsX+4xM4dGoCmYKFkCYRMsWS5VZKMSSAeMzAwHAa/8/f7MG/eGQr37mzmwLBxXpEq25vVgZSD2Fd0s7Y6wFw3DKe2f/7mM2crYLHpW0mvXJf185j6Oj/zW1rP4ZQfD3NAw52AYjLHjkLAG3dD2B26qCfP5GLMikpDUzN9qG76w42LqOs96abbvJ24DVdqsHPa9asQWdnJy38+5I7IfZy3BNFl398vojZknde37M+jh3tYSraii3XAwkoL9707puaENYFHBXIsACzBRtlWyGkEYquwppWT6I9kGavZSCuYkxlypDSky/5+AO9eM8dPbQUO1jdHse9t67CxEyBD5yYwJ7DY+gfTiGse7kagRqAqrmnXNdjI6wY//iNQ5iYzPMH3r2lnhep2xtmwZq7GgByPTOQ60RD2AuEMzOee/WPeGzmCEJGAuqydKvmGMDM0Hdx9uVfw8DeL/DM2a9yOXPc94RyzgtdUv9IwEIYoUgnNbTsgOuWlqyyEiRhWQVMTp+6pMUkhOeQP/nJTyIUCsGyLEgpIaWsKnx+4QtfgJRyRWqfAXgM5x3+wbki8raCJoB3ro1hR3uYGMBwxkLRVpAEOIrRGNGwviVMgBfa0nynv+dMqip2rxSjLWHO5Sb8JLervGs5Ml3k0ZkiXJfxyNtW4313rSYiD1iUf71rcyDK/317c5Te9/Ze+p3P30Nb1jYhX7RRrjgoVmxULAeOo/zzRNWmQ6UYUgrEYyZ+8vxZ/P3XX+Vyxbmgk79udXszMJBa+ZI6A7mog1MQJPHikb/ic2MvIRZqgKvsi4atyK+aIq8eyvNENSde0xMAOyjMHEBx6mXoMoRwYiPH2u9HpP1+6LG1c8ykNqm+EsADobXz7ZidPgIscYEZDCE0TM72YVXHLSsu5w0kR3p7e+nxxx/nf/2v/zXOnj3rHZOm4Qtf+AJ+/dd/nQLNrJWBh8tPDc51cO9eFcGWZrM6KvbVsSKkz5GkAHJlF986NMU9DSY0QShZLk6N53FusgBTF3CVgq4J9I3lccuaJDfFzLl7wP/+8slJFMsO2htCeNft3eS6DBKe0OKi3I7mQm2O671+YyKEZNzExp5Gr8u94qBQtJHJVpAv29AFIWTo0CShXHbAroJpaDh8dByFgsW/8KnbEIsa9SFUdXtjfNvl5kB4LlytuJ4DWcbBuRAkcfDs43z03PcRMxuglLMsNSJfr8pxCn5ynaHIKzMVwqj+PZA1EVoUElEQO6hkTsBOHUTm7N8j1Hw7RzrfjUjb3RBarFrWOxf2oKXfH4xoYi1FE+s4nx2E0KIX8hhmCKmjUJxBNj/GyfiqFSvIBiDynve8hw4fPsyvvPIKstkstmzZgs2bN180dMVVFgQM5l1+dsSqki5dELrjGnKW4qmCg9fGihjN2tAlQblcBYHj4wUcHclDuQxXKQgoT67Ebxw0NIHzUwX88Q9OozGqcXsyhLWtUXQ2hjE0lcfLxycRCWmwXQVXMYcM78CVqlUmXpTkQdMEShUH5YqD//GbDyIeMaqPrFgOJqeLfGZgFidOT6F/II1svoJ/+bFbkMtXcPzkJGZmijjVN4V/+Pqr+Nwv3AFdF2CuBam5UmSv7aeOLnV740NeizEYVnUGssTJ8ZhH//grvOfEPyBkJvzxs8udaAnHyUMTEs0ttyORWA9JElZpHKXsaVQKA3DcCjQtDBIaANdvEPTVcbUwJCJgdlGc/BnyU3ugRVYj0nYvx7veCTOxiebil6raib7YhSUiNLfehlzm3NJg47/O1Gw/kvFVlxZf9EEkEonQgw8+OG9Xsxh4cM17BmmCo2mXD0zZ1XilC8Blxrf6cgAzChUXzAqGRlDO/IUa0gUgGcplKEVwFeYAxmcKukZwbMbobAnnJ/J4+fg0BDFsx4Xhh75yRRt/9u1jeP/benjjqiSZupz3Gioo0fUbGV2XoWkCR05P8eh0AfGIMS8pbhoaeroS1NOVwIN3r8XUTIF//EI/0pky3vOOjfTQfesxOVXgF18+j5++eA4/fOoUf/CRreS6QXUWLwperDyGVLe6vd4MZLkQWL0Ka4kTQySQLU7wc4f/ApoMYum8LHhYVgYtjTdh69ZfQ0PjVpp/sWwU0ic4NfYsMhM/hVMagS4NkAyh2jDod5sD8Oef63CsaaTPP4bMyJMwG7ZzvPOdiLXeBakvzUoCkEk2b4U++DRc5QBkLHqcQmhIZUfgKge1I2xXCiK1i8jroxDVXg7UOF6qCVuNlJiPpV2MFV1o/h/cGoBx2XOYpkYeOLh8AQtAVQjUb5jkxa6jx3IMTUAXBKUBrnKhS4LruFDK6zY/PZrFHz12BG3JEG/qTmD7uiZs7G5AY9wkucCTaxrBcRUef+YMJmcKmEoVuaVhrgiB/TcOwlKtzVH65Id34OjJSX7tyBjv3NFJ7W0x+uiHtmNoKM0vvXwe69c387ab2oNRVZgYz/GpExPQNIHungZ09zSSkHXwqNv1x0BUPQeyeJCFGXj2yF+jZOUQ0aNgdpYFj4qVxtruh3DLzf9vksLwBQ7nvKgQOuJNN1O86WY4G3+BU6NPIT38A9jZPggCpAz74S01ByhwQaRDGlEoEIqpIyikjkALr0Ks7W5OtN+LcKJ3bt5HjRIvM0PXYxRPruOZ6eMQurkECEiUK1nkClPcEO9cURiLMV/Guco4/OcR5pMeSzHSlstjJYWREmPW8hLupgBcdSEsCwBM8DrEF1mflsOwHQVWHgORxBAL4F0QLSqYH7CK2v+bugRLwnSmhOGJHH6yfwiJiIGe1hhv6E5iXVcCrckwhCCMT+fx5EvnMDiehXIVBseyaG2MwHYYUs6N/63NmSilsH1LGx0+Ps4HDo3w7bd44cKbd3RicCiN733/OIjBXV0JHHxtBK+8PIBy0YIkwNAk2trjfOuubtxye089mlW36yyEVWcgF4SuiASODDzN5yYOIG4mwewsE7oiWHYWm9Z/DDtu+lWqfY0LQckLVWlGA7Wu/ThaVv8ccpMvcmboO6jMHIDrFiG0KCD0moQ5++NtJYQWAUiDY6UxM/g9zI48g3DDFk603Y14y63QjYbqpWa/HDiWWIeZqaPLfHqCYgfp7Aga4p1YSTkvBbRiEXOUQsF2OGs5SFku0pZCxmYUXILDEkJoMIQHDrXEguDLVZHHRmzX6/JeaC4DqxtNNEc0mJLguIyTY3nM5CuQ/qdnAGXLhe4J7cKyFWzHhXIBU1s85MfM0DUBLaJDKYWS5eBI/wwOnJoAMUOTAsSAZTnQBCEW1lEuKzzx07PYsbEVxrzQ1xwDIX9+ulKMm7d20N/94wFmBd61cxXdfedq6u+f4b5T03j0sUOIhDWUijY0IdDYFIGhCyiHMT2ZxxPfPoKxoTS/+4PbicSbo4ekdtJl3a7f67MCBFkSVOohrIU7ayIUyil+ue9RmHp02bwHkYDj5LFl/cexY8vnfHFCWqKiqbbpzp/PIXQkOh6kRMeDKKePcW74eyhPPAe3PAWSEUCP+5MHaxmGCyINmh6BYkIhdQK51Alo51sRbdzKydbbEWvYTFIL++zAuOgxC5LI5Ceqx3Qxy5TznK+UUXEc2MxwFMNWQMlRyNgOSi7BZoIiDQI6SEhoUkIHQfnBHtSo6jIAhz1mwa7HTFpjOnIVF9myXxZLQNllvHNjA3auis67JNs6o/w3LwzO0Q0GPrSrExs7YnAVI1eykcpb6B/PY3/f9AU3QjVhXSPErkkBGSKEDQHXVZ6qrwKiYd3L17gM09TQd34W//WvXuZ7bu3C+p5GrGqPUzg0X+lYqbmk+H13r8OX/nYPWpoivHZNI33207fTT1/o5/37h1Aq2dCkwK5dPdj99nWQmiBWzNlMGU9+5ygOHxjG5q0d3Lu59bruaA8AOWCm9X6XNzkDWeY61wFkQXyDSOCVM99CvjyLmBFfJnRFcF0b4VAztm/+lzQ3LEqs6JIEw50C0Ak1bKNQwza4Gz7HhfGfoDD2Y5SzZ7wyOS0OiLDPSmiOlUBCaGEIknCdElITryA19Sr0cDtHExugmQ1IzZ6EkCaWE70SQqJYzsCySzD08LIs5PDwSR7KTIEhoSDBpFW/k9AAoYGEDoMEWAiACUxeGMtlhiKvJVwpP0SlAJ2AppBAa0iiKyLRGpGIG4IOjZf4x/05hKT3/OaoVgUPxcEhMWIhCUMKOI5C2Xaxe2MT7qmRL+loCAEA7tzYjHLF4VdOTSHsd7CXLQchXcBxPNFF23aqEiWan9SuFVlUyvv8wtcfC5kaBkYzODvoNRU2NYS5pyOBjWsasbm3CWtWJavNiUoxVncnqakxzP/46EH8+q+8jRsbwvTQOzfSzltX8fe+dxRnTk9j0+Y2xOJmEE6kSNQAwCyEQCF//U88DBQKRkdHub29nYIeoTqIvDkZSL0T/RJCV7P5UT429DxCehSK3WX6PRhCSFh2DmMTr3BXxz3V8NVKd/LzHuc/T4ZaKbH255FY+wmUZl/j/NgzKEwfgF2ZAZMO0qIgoYPJl9HwFXeJdGi6CZAG28pgZmIfmCRIC4MotPTsYmaQELCcMgrlNBt6eNG+hGD+iDedz0XYCMOF8OTifQABBFxffp0xJ1LoMqPBlEjoOkxNQBMCAgydBGIakDQE4vqFq7Rgzw3KclzGqqTHpoL+kUAAMVV0ULYVdOFpWN26JolAviSYKOiNpaV5+Q+lGJ9+x3rsWNcEy3YxPlvEwHgOYzMFjM8UMDpVQKFkQfMv0Zw6MFd3ZgwgHNK9PhXFyOYqODg7gYPHxhA2JFa1J3jn9nbcccsqNDWEyTC8FbXrtm6MjWdh6JKjUYOamyPU2Zngk8cnMTiQwrrepurn/OlPTvP0ZAHRmIHutY1LRRWuG+ZhWRY+//nP87e//W1s3LiR/+mf/gnr1q2jpSr06vbmBZCAgVyPQPKG5ED29X8fFaeEqBED+GIKu1590b6D/x29q9/HG9Z9GOFwG80LU1VZw8Veai7nEUiQhJtuo3DTbWiy0lyY3o/81B6UMqdhWRlAhDwwIb/qyX8/QPkhLs1nB+Kig+8JniZXoZhCY7xzyccAwPauTWQp5rFsCoYfJmP/32KHyWDc3ZbAqqhx0SrU2jVIBEwUHH8WufeHjri+KKidnynBUR5raIzqaEuaJGguT0MgCAIsR2FoughTEyhbLtZ3xvHALZ3VT9XZHMHOjS3V159MFfkvHz+CY2dnYOoC+aINjQDTBwHHZSjXRdHvDTKkgGlIRCO6BzqKMTqew8BwCs//7Dxu3drBmiRMThWQTISwdUs7/c3f7eVNG1p4ZrqAEyfGEU+Y2L9vEL3rm7mlNUqv7h3k/XsGQAA2b+9AY3OErtfdfDA98r/9t//Gf//3f49kMol9+/bh3/3Wb+HRf/7n6zrcUbfLv+ZveQYSlO1mSlN8amwPTD1y0Z6PwIURSRCAM/3fwOjoM+juvJd7ut+FRHIj1YapVs5KaE6/KmAlRgMluh5CoushOOVpzk3vR2biZyhl+6G4AtJjNUAyByYMtaKtagAwhXL6oo+VQmLX6q10amKQz8xMAIJAQi4uHe+Dgik9B+7yInpRVAPFNR+14jJSJRdSeKxBk4S2mD4vJks+2zk9UYAmCZatsKojDF2KeRE7xZ6O1sBkgaezFYQ0gULZxuq2mL+LAoSoATD/87U1RigZNVkxo2Q5uOfmTrx/dy9iER1SeBpauYKF6VQJY5N5jIxnMTFdQDZnQbmu34muIRw2YdsKL+0dgCRCOKRj36tDGJ/I8vmBFM71z0IKr8s+mQjDdRT+4e/3oiER4ny2AkMTiCZDuPv+9XMJpOtwNyulRC6X4y9/+csIhUJgZkQiETz33HOYmpri1tZWqoey3ny2nJx7HUC8KDcIEkeHfoqilUPciK+Afcw9GwAMIwnXKeP8uccxOvQkmhq3cWfXA2hpexsMs5HmbjT3MlmJn9wNtVBj93vQ2P0eFFLHODX6LLIzR+A4eQgtBhK+ECNf6iIRKFdyy1LWWmDY0r6amqJJPjw+jLzteJMYFzkzgoAXJ3K4szXO3VF9jk/UFHLVirwo9rJI6bLLgf6VrRhRQ6IhrFXrhAOAmM7bPJ6xYEgBy3KxtjU8j51U34CAvtEsHJdBvvx7Z1PYj9fPb95T7CWoB8ZzfODUJKQgrG5P4P/z6dvpYucmX7B4ZCKH88Np9A+kMDyaRTpbQkiXiEW8EBwxMDaew/BQBtGIAUlecPDhhzbippvaoRTj+WfO4OihEcRjIRRzFdz/8EZEosYlJ89fL4cdsI/XXnsNw8PDiEajcF0XUkrMzsygr68Pra2tWGzGfd1eX+d/qSGs5Z5fBxB4woKOsnFibA90afqAsryzJT8jQKT8xjFP9kQ3kiB2MDvzKtJTexEJt6K59XZu63oIyebbaliJC1qx6m5txzlXGVO0cRtFG7ehnB/imdHnkJk+BNvOg4UJCvpKSMxVcS2DgYIEKnbR1/4SK1o8bbEk3bs2ilfHhnisUIChmYucW4LLwAsTRWxKhnhHo0HGAgdY+7+gX+70bAUOM0zhdZknIhKhIBnBXl5FgjA4W0LZUQhrBFMXWNMcmRdyq72Jzo7noUlvzKyhCXS3RC/8ADWA881nz8CyXTiuws89sAFEXhNhoOw7r2HSf59Y1KDNvc3Y3NtcBZQTZ6bx9PNnkJotQtcEBAlIKWBoEkQEy3KwvrcZd9yxuvpJdt/byyeOjsG2HCQbw9h6cxfBz+lcb+BR64xOnz5dVSNQyisQcV0Xw8PD122svA4gl/8ab3kACRzm0MxJns4NI6KHwcso7RIJ2E4BjnIgwdCEgCG9yiMvZOP1jGhaFBKAbecwPvQEpod/iERyE7d1vxdNnQ9CM5JUG6bCimeT15QD+88NxXpo1abPoG3NBzgzfRCp6cMoFsZg2wWAJEhGLzJrxAMkx7XguBU/t7F8P0hAYU1Nw9t61tG+0WEezuWhS1ntKmd4eQIhAUMCJzM2RkrgjQkdq8KEkCRiZtgKbCmG4zIqLmMkZ+PoVAWmFGDlTQ0s2gq5istxUxIRoPnn4MRYAZrwHHtjREdTzBMnDFiK8qXcx1MlHpgsIKRLVCwHbQ1hdDb5BQM1x+kq7/EnB1J84NQUpCCsWdWAO7a2kzfhUMwHvoXFBjzXaEkExKIG3XFLF86dn+WXJnIwdYli0UY04pX6lss2ykUb5bLtqQSPZHhyIo/TpyYhBMG2FTo6EzBN7ZLnthARsjMF1k0N4Zj5uiDJuXPnFnU+4+PjdW9+g1mdgdRY38Q+qEBfahnnbTtFdDXfjLbGLWC3gnxhGLlcPyqlSSgwTM30ejfg+npaGjQ9CQkXhUwfzqeOYuLs19Dc+SA39bwPodi6CycSrjTGXa3g8pLYutlALaseQMuqB1AuTnAuew6uU8bU5CGUrRxA5rLH5ioHjmOxoXky6hefwU5VmLm9cxXlrQHO2A4kSdjM0AWhKawj5zBy/jyPvMM4MKtwmBgamF1XeZ3lroLjKjiOC8dlGDQX2tIEIVt28LUDU4ibgqO6QGtMR6Hs4PxMGYZGsGxf/6pW8oO8yitXMb79yjAcV3klu65Cc9xEre5V4Pyl8IDx60/3QQhCpaLwzjt6IIXwwOhi4T2aa7RUypsjc34ozXsPjiAc1lGuOHjH/b248/ZuSCJMTRdw/nwKzX447ekf9eHs6WnEohpMQ4NyXLiuqn6+leBHoJt17ug4v/p0H971C7su6flXstMNAGTh7nZycrLucW9AFvOWBxBBAq6yMTB9Aro0PEayFPOwc9i58ePYufnT886oZed4ZvYoxkafx/TUXlhWGro0oUnDK3z1E9pCC0NDBI6VxkT/PyA19DiSrXdxY/cHEG25k+Ynz3nlYEJUbYILwluhSDuFIu0AAD3UxGf7vgUhQ0umRogIrnLhuva8MM4KgmteApUIt7S34/mhEf+phLs7GtAaNqnsKowUHR4ouJi1gAoDZdd3dAwI9kNXAhCSoBPm6V+xHwqrOIxc2YLjKhwd9UKHuvBCUpogzBYsfPfVcd61rgGxkITlKAzPlPDC8Un0j+UQ0iUcV8HUJc5N5PD84XG+aXUDklGdTN2bPpjKVfh//6QPZ0YyCOsSYA27bmoH+WDk6f8ESf+lO8IDBlAoWvzVbxwCM1AqOfjI+2/C/ffMbRqamiLYvKm1+ryW1ihGhtMwDB1EDMOUGBpIYXQozV09DcT+my9ZWumDx8jZGX7usUPY/cFtiDdd+8qtoDx3YGBg0TknExMTdU98A9pbGkCCMtvJ7DDPFsZhSh2AWiJsVURbwybs3Pxp4qBE17t1YOhx6my/G53td6NQGOax0WcxNfZTlHLnIODCkGGQkACUnyvRII0GgB1kx55BfvxZRBObObHq3Yh2PAgt1F7DStTc1vbinGBBt7t3Ezc1b6PRyEtcLKcgZGiZQBZ7wouXsUNhAM3hMHXGojycLyFq6GgKeTLnISmwPm7Q+jhQcJinKozJEmOqzEiXGWXFHmCwL1/Cc7ImNA/sAUMj6EJASc9ZOg5XQUaTAq+cSWF//6yXVLdd5Es2iBmmIeE6c3LwtqPwtWf6EdYFoiHJUVOCGBifLSCVLSNialWp+O+92I+H7ljNLckwmYbEQl1DtUDSOigeIAL+/tFDmJotghRj952rcf8963zlXe+DBNfIq2LymiGDIValsg1T1+DaLv75q/vx3g9t4y3bO6tzUgLl3oXgkUsV+dlHD2LLHT3YdHs3XWsl3ypYFgo8OjrqFVTUHFctA6n3gdQZyI0DIPCkuodTp2G7FYQ0c85hXxDesbC++4Eq8Ih5OYU5Zx2NdtOGjZ9B7/qfx+z0qzw18mOkp/fCrsxAEzqkDPk7NM9RSz0BgkIlewrT6cPInv0yQk23c6TjnQi37ILQk5dXweWDCfvz1mOxVSgUp4DlVF2vZECMz1i643EM58uwFaPouBzXtXmSVlGNKKoR1kYBQCJnM0+XFCaKDiYKLlIlBwVHwXX8c+p3I9KCt6r9XmthQ8JxXVRsj6FETQ2uOxcGmgMjQtTUYLsuZrIVTLou2PXyLRFTg+N6Mvu6JvC9F/rx9CsDaIia3NMWw6Y1jdjQ3Yju9hgaE6Elx+D+8/eO8bG+KWiSsHVzKz72gW3VKqrgngykVIgIxaLN58/PQtclbNvBPff2YstN7ahUHBx5dRjff+wwTh2Z4Nvetho965qq8FrVnBIEq+Lgx/90EMnWKO5675bXpWw2eI/x8XFMT09D1/UL/jY9PV0HkBsQVN7iISzvJIykz1YHMS1+g7gwtCi6mrf7zxJL7vy9m1lBCB0tbXdRS9tdKJfGeXbsp5gdewblbB/YrUDTDZAwPBBj5c8CCYOdEgpjT6Mw/ixkZBXM5js41vEgwk07iYQ+x0pWxEjmzDAbarrJl3cGl3sqGUCDacKQAhXXRaZiI65r1YTyQqdPAOI6UVyXWJfwALlgK54tuZgoOJjM2x6gVFxULK+c13EZrqvg+lK9+oJzwH7PBwRQtjy13oX6iYIITORVchFB6ALQANcRcB3Xe+1a0AvpcFyFyVQJY1N5vHJ0DKaUSMR0dDRFeFVbHKva4uhojiBkasjkLOw9NIJTZ6YRNjXEojo+/ZGbIfz8Ci3CYKQk/OSZPmSzZehC4P0f2IZbdq6qPnL12iZUig6fOTGB4fMzWLOumXfs6saqNY2k+bmc2ck8v/j94yhkK/jQr94F4c+Dv9ZtI8GaGRoaQj6fRywWm+dYpJSYnZ1FpVKBaZp1WZN6COvGABBBAswK07kRSOFXuVwIDVDKQSyURDTcOhdNWhaZ5TxWEgp3UFfvx9HV+zHkZo9wauwnyE++CKc0Co0IQoar3eAgCaEnANLgWhnkRp5AduTHMOLrOdb5ABKd74QWaq1WcHH1PZe/ITU9fMV0dTkmF9QyVVzXlwshzFZsdMfCi0D2AjZRM30vqguK6gI9CR1AGAyg4iiUbMVlW6HiKJQdBddlnJ0q4dBQDpqYw1OHGRXLhVIKqxpDsGyFyXSpCvlEQKHsAEqBmfykvQtWCiFNXiAJT34joxQEoUsIQ3i/Z6BsOegfTuPMYBrEXmWY5ivv6pKQiOjI5SzsvqMH8ZhJAdNYGPKSkrBv/xDv3z8MAeD+d2zALTtXkeuq6vWolG3OZkoIRw1oUmDw7DSG+qfR3BLlhuYY2HEwOZSBVXLwnk/tRLwxQq/XEKpgnff391dFFGslLjRNQzqTRjab5dbW1jpy1AHkzQ8gwW68YOU4W55dfpgSEZRyQSRorqv8UvMRXigpmAnibvoc56b2IDv2NMozr8Kx0tCkAQQ5CnYB0iH1EBQkrOIwpk9/BbNDP0By1fu4sfvd0MymGvn25bvdpTSrWlZLgg3RinpAFjlSAMBUscgHJ71QhSTCeLGCHc1z7ISW4oA0L4o2D6AJQEgTCGmCsAADp/M2u4phCK9zXbkKMUPDutYIblubxPbuJKUKFv/Bd07CcVyvHNhSuHNzC+7c2AJdCpRtB+m8hf7RDPadmITlzgFNsezAsh0o15s9QiAYkqBJgkYCmhTQQtI7Z74Ao/Cvu4DHlExTw7FTk9i2qZXX9jRQyNQuaATct3+In3jyJAjAxo2t2P32deSxEg+MhCDs/dl5TE/mEY97O/hQ2ABYITNbwuxEAcp2EI2ZeMe/2IFV65vpjZhgePr06UXBxROBLCCVSqG1tbXOQN6k4ao6gCyI94MI2dIMynYBhpSLakYxGJrQUCjP4OTAD3lH74cvQzRxvvouAEg9QQ1d70JD17tg5c9zfvwnKIz9BG7uDAD2WYjwH08gYUBqBpRbwfS5R5EefQaR5ls53noXoo03QWrRmlyJuoCVSGlctIaTIFY8lbAWECbyOT6bnsVksQwmCSF0EAipioNz2RL3JsI0j3EsYCMLWcipWYsrjsK6pI64OT9pU7QVp4o2jo0VsP98FmFDoGK56EiYeHhrCzobTESMueeEdAlTF3BdF6WKi529TfjFh9ZfcCLu29GBlniI/+knpxELa8iXbNy6sQWbVzfAcRSm0yVMpUrIZMsolCxUKi7Klg3XUTA0DRFDeuUX5HWak/CcpKELpLIl/M3X9qMpGeamZAjNjRE0JEMgBoaG0zjXn0I4JFFxXey6o2c+SxYEx1E4c2oShumFA62KA9sf2WuYGpKNEXR0JbD9zh40tce9vMfrCB6Bgzlz5syiYVApJYrFImZmZq4sTFq3OgO5fhiI58Cy5Vk4yvYrsNwlsEZB18I4cPIfUCxN8k1r3o9EbNUCh32J6rvVSi4BI7aWmjZ8Do29n0V5+hXOjzyB0sx+KCsNaBFARgEKQlwaNN2EcsvIjL2I9MTL0EIdiDZt52TrLsQathD5IMCsqj0iFwtxeTtFCSn1xWNNizCHsmPj8PgQj+ZyYNKgSx2qZhKgJggHpwuoKPDauElhKS7KQo7PWPyT8wWAFfYKQtwg1gWBlYLrKswWbOTKDiq2C0MSWHnH9/6b29DTFKpWKAVhp3TRQrHieIOdBOHhnR1g9hoGg3yu6/ohKuGxh2LFwW2bWvF/feLWCz6uYkap7HC+ZCOVKWFwLIvXTkyi79wMTL9T3uu+VqhYNsAEIbyxulPTBUxN5SFJgOAxFl2TCJtatX+klp0EbKxcsrlSdqqNhb2bW9G7qRUNTRFEYyYiMaOaB3kjdveBXPv58+erY44XAozrupiamqp73DqA3CAMxN/y5isZHwBwUf0oKQwc7/8uzg0/g86mbby+5yF0d9xN5DMFuqTE9hwr8Zy8N2Aq3PZ2Cre9HXZhkPNjP0Z+/HlUCiOeNLseB5GAAgOkQeommCQcO4PZ0Z9idnwPzFgPN7S9Dc0db4OmR6t9kY5T8gFrqSQ6QwoNmjRoOfwIoKhsW3h54BRnLQuGZvqqv4szjIMzRZzM2Jw0dDSYGppNiQaDEJYenFZc5lRF4WzaxqmZCoQ/LyRnKeQqCpaj4NguoBgELyke1qWfTGfEQxpa43Md6DVCvJjKVGA5nqZWR2MYq5oiXid7DbEh6VVFHT03Cym9RPcHd68D4JX7CkFVsUdBhGhYp2hYR3tTBFvWNeNd96zDH315Lx/pm0Q8YqJSdtCYNHHLbd2IhnUUihYy2TLKJRvlsoNMpgxdCkiqlYj38jFnz85gw4aWC867IEK54uCeBzbgbff30lJ5iNcbPALAmp2d5ZHREei6viiAAHOlvHUGUg9h3QAA4odErNwl6A4yTCMOZhfDE3sxNrEHnS038/bNn0Vz403+7vdSSm3ncg8UDGT1by49upoaN/wSGtZ9GsWZ/Zwbew6F1BE4VgaQYUDT/ZkbgXx7GIokrOIkRvsfx/T4y2hsu4OTTTcBJDA5vh9C6EuwEc9patKAVtWzWh5Cjo2f51ylCFMPw2HGcnJbpiDYzBgvuRgp+2NewdDJ68qr2ApF2yu1lQSsSehoj0hEdUJLREPFUUiXHKSKDtJFG+NZC7MFCxIe0EQNCVMTi0bo0gWvMdJxGauavbnmgTpv7S5/Kl3i8+OemGRPWwzruhLEDOiaWDT6GZQOKOWNu5XS00ezbBedbTF84V/uQlMyfMEnqlgOTvZN8z9/8xCEJqGYUS47CJkaQmG9KpVSeyyaLogEsRnScfOubq+yWbEvREDV0blvhAUAMjw8jNmZ2UUBJLB6M+GNy0DegvNA/OoWu3RJVY5BA6GhRyHBmJw5jBf2/P+wfs37edOGT8A0GmpyJHyJYFIjmuiXA5M0EW3bTdG23bBL45yfegWZ8ZdQyp8HQ/ozQYJBSR6L0bQwbCuLscGnMT76IhgaGMJvIuRF31YpBVOPLJtED2qtilaZp/IpGFKvVlwtZB21h6z8/+qCPA0rAlwFlBx/2h8DhvBKays2Y01Sx47W+dpNPcn5o3n/6cAE90+VwGAkwpofXvIa+GodfbZk+wCp0NkUnk+jahzga2dmUCw7cF2FnRtbPRl5xUtIWfsH6Df/jUzk+OiZKYRMDbbt4rM/tx1NyTA5rqoClVd9pWAaGnTdS44H1Vj339eL7ds6EI+biMe94659X9PUYJgSggFd97TASF4fSejAcZw9e7Zapuu6i4eC63Im9RDWDcdAHBU4GFwykDAYuh6FYBdn+r+JifGXsG7NI9zV9UDNYKnahPslhLhqyoGrrCTcQY2rP4SGng+gMHOQ0+MvIJ86DtvKAzLsK/B6YOKxEh2eMIsESFt2sBSzQsiMz3OqS0FI0S7DVQqeKnfNhD9mOKygSEExAYIhiefNPq89POnjpfJBRvmSJj8dKmL/aJEbQwItYYmOqI7miERE9+BhPGdxquRACoLjeM8jWoRXETCbt/yudkJbMgTF7IHegnX/yolJSEnQpcTOja1zndTzD3HBrHNASuAnewZQsVw4cPHw7nXoXd1Irs9MAiDzZMwFJqby/OjjR7xKLUH41MdvwaaNS5e2BtcimQxjNJOG6yjWDUnXUtfqcgDk1KlTS4Y7gscEOZB6BdYNENZCvZHwkqhXUALL/ghYj2aoKtMwjCQsK42TJ/4aA/2PoqlpG3d03Ium1l0wfFZSDXGBLkl9F4uUA8dabqNYy22wShOcmdyPzPSrKBXG4LoVsAyBRNj7bDSv1mnZxRANN6zoExlSr3ECXuLZZUbcNLGlpRUMIGe7SFVcZCxG0QVsX4K9diY61QyYqgUAQxDKjsJA2sbZmQrADJMYIUmsFCNVtMHKC3cZmsDATAkvn03x1s4YIqZGuiQIIoyny9w/UYChCRQdF4WyA0EEsWD3vu/UFJ8bz8GQArokrGqN0jydq8UUd9lr/pvNlPjlQyMwdIl4WMcjD26A67Mq9tkZEarg8bdfO4Bi0YahC3z652/FxvUt86RNaJH3IgJaO+I4c2ISuWwZoYiOFYuVvU7x8ZMnT170sdMz03UAuXEQpA4gK1/MBEdZCEkdQgCVShoaEUwZ8jSuqvNAdOhGEq5bweT4S5gZewGRcAuaW27j1s4HkGy5jYQwasCAryDx7gWGjHA7ta55P1pXvw+F7FnOzBxGLtOPUmECWOHgHgZDkEAs3LiiXUfcjFAyFOXZcgm6pkHBA4eEaaIrFp13MIqBouNy0WVUXKCiCCWXkLUZs2VGusKo+KGs4EwQAZoAhEZQ/jwQ12FkK16DoC4ICnMNeQzgiSNTePbEDCIasal5CerxdAll2/WARpf4/r4RnBjKcEgXYD+ElC/ZOHZ+1pvTASBfsvGDlwf4gVu7EAnpRJ6wI/tVXWQa0kt8E2E2U+a/fPQgimUHuhQIhzSETY3kghLafNHig0fH8aNn+lApO2AAH37/Vmxc30K27ULT5EXZROeqJBxHYWw4g9aO+KLKusFM8tdTLiR4rzNnzlTDdItt0oQQSM2m5j2nbtcH+NdDWFdgQd/DUlVYBIKrKmiO9+Ch234TggQmU8cxOr4XUzMHUbFSMDUdUviNeuxAkISmx+bmgQw/ienhHyIaW8MtHfeiqesdCMd7aU7+5HIS72IBEAlEkxsomtzghQtGX+DB/u9B6lEsS7J8vSxdCyMabqSFYZpFg1hE2NG1HnuH+lCwbeh6CLoQGMvnMFtq5KZwyNNE9AEypkuK6Yu9lkTGYp4ouhgvuJgo2EiXgIrDIH/GebDZEX6iWPlsZ2G1V1gXcBxGquLAcV2vNJe8UmKlGIKAkuVi3+kZuL7ulaeGwjAkQfry67om8K3nz+JHrwwgYmislIJlO3AdBhFxPCQRjxoQTBibziGbqyAaMgBmTKdK+C9feolXdya9jUbZQbnsYCZVQDpdRtiQMHQJq+KiWLJhOwq6X3671KTBYI20dyYQjRk4d3oKN+/qvuAaBdP/AiXc1wNIguubTqd5cHAQhmEsyeillMhmsyiXy9Vxt68HE3Fdt/pe9UmIlwcg9SqsJV0hYEhz2dwAEcF2K1jTfgcSUU8JNRZpR++qB1EoTvDg6As4P/A0rMoQXOV480CEVg031c4DqRSHMXL6y5g8/yiSzbdw86r3It6+m4Qwa1gFLiG8NRdCCp4fOPjWrnspk+rjTPoshBZdttJMKRexWAN0LbR4zOZC5opEKEK71271+kDyOWjSBDPw8ugYtrW2cHcsSlqNA1Nc8/yaOegNBlGDoWFzgwaXTcyUXB7N2RjK2hjP2chXvI5uwewtiEU+miDy8ieCIKSAJgAlGY7jzuvskUSIhjSwEmBXwVWAcpUnZVJzgiIhDZajUC6XvVyXz1aUArI5BUzmAQUYOiFsal5VF7z3H5sqYHgsD0FeubEkAU0nRCN6kNGCaWp48uk+vHpwhLff1I633bkasZhJtSGvGnwHGIgnQtS5KslD52aRy5Y5nghV8yDBPPJ0Os3ZbBarV6+mgA1cSxAJHPPg4CCmpqaWrMAKwCyXy6FQKHAoFHpdYljBeVn4ed9Kdq2ro97yDCSkR1Z8IZgZCi6EvzeORtrppg3/AhvWfBCpzFEeG/spZqcOoFIagwDDkCEfTLxciRAGNBkCsYPs5M9QmHwRkXgvN656HxJd75rTuAI8KZMVj7ydYyXefA6vryXRdBPSqb6LAAJBsYNkrMM/zpU1RTKAsG7grp71dHZ2mo9NewlShxX2TUzjxGyeWyJhdETCaI/MjbHlBUBUm6SWBLRFJLVFJG5tD6FgKR7L2xjOWBhOVzCZtbyJgQsqrUqWW3X0CABBAbXzosiXh1euJx3P/rwRteAGE4JgOwpKeUOubNuF67hQisEKEPCaDqOmBinnSoK9L68xkEyAwPBqDAhCeCBYrni7YQmClISxsRzGx3I4dGgM73hgPe/0xRODQVRVxVP/PdZtaEH/qSmcOTGBnXetASsFBc9J/u7v/i7/5V/+JcrlMm6+5Rb+n3/0R9i5cyddSxAJXruvr++iFVhBN3oun0dzc/M1d+bBZ/v+97/Pe/bswdve9jY88sgj9Eb1y9QB5IYDEG8BhY24H3pa3s0WKymvcoZF1cGyXyGl6wbaWm6jtpbbYNs5npl+DVPjLyIzexB2aRIshAcmgbw6AOmHuKzCECZO/DHS5/4R8fZ7Od71HoSabp0bLlVNnF9iiKs2X3LR/IdEY6LrouxjMSbCANY3tVBjOMqvjI6gogBTSpRcF+eyJZzL2YjoOnfHQlgXN5D0K6mCFPAFWliYSxxHDUEbmkxsaDLBHMdI1uJnT6cxkq5A93tLDI3w9g3NaI7q0IRX6lCsuDg5msPB86nqe1guw7EZuvSaCBUTSpY7r62SGSiWbIQNiUTCRFsyjK6WKBJRvz/DD3llCxXsOTSKXKECXXr9J6WKA/bFD01dQ8ggv6yYUCzZSEZN7NzWie6uBKQgTE8XMDiUxvRUAVNTBXzjm4dx6tQkP/zwZjQ3Rxa9COs2tSL0zBmcOjKOW+5YDfbB4z/+x//Iv/d7v4dQKAQpJZ5/7jk89PDD+Onzz/O2bdvoWjORY8eOXeCUg4704IuIUKlUkM1kXpewlZQSv/mbv8l/+Id/WP39r/3ar/Ff/MVfkFcRVw9nXRUAuY6bQl8XBhIzk/5sD15ypy1IolCavsDBUrVCak55V9fj1NF5Hzo674NVmeWZyT2YHv0JCqkjsO2CN1xKGgBcL/EuTEgZAjsFZAe+ieLwd2EkbuJIxzsQaXs79Nhamp84X3mIq1y8eOOWUi7CoUbEI610OTsz8hdRUzhMuzq7+KWRMYDnFpYAkLUcHJot4XTORXfU4K1JDUmdaPGWxgXS7zwHNt1Jgz5yczN/+ZUJlCwHtqvwvptbcfOq2AUf+pY1SVi2y4fOpyGJ0ZYI4YN3rEJz3PS7zYHJdAl/+2QfimXbu86C8Avv2YIdvU1IxkwKGUs7mVzB4uf2DUIPEWyXsXVDC1a1xqBJwqn+WYxP5RDWJSzLxS1bO/CR996Elqb5wKAUY3AozcePT+DkyUkcOz6OocE0du3q5m3bOtDUHCVNE1UZ+KnxHDRNYHa6gInRDHd2N9C+ffv5i1/8IhLxBBjejOrGxkbMzszg87/+eTz7zLPXbLcdvO7x48fn7Xa92SZF6LpeHS4lhECxWEQ6nb6mO+MAPP7pn/6J//AP/xDJZLL6Of/qr/4Kt956K/+rf/WvKHhc3eoM5DL5h19RFGqELo1F5zQE21IhNORKk3BdyxMlXC4PUQMmhtlEnT3vQ2fP+5DPnOKZkaeRmXgRdnEYkhhCC/nz05Un424kQaxgZU6gkj6GVP/XYDbs4GjHA4i0vg2a2bysYOLcjS3AykEufRpCLF3ySURQro3GZA+EkJek6bUwB8HMaA1HKKxpnLFdtEYi6E0mENIkig5jqOBguMQ4m3MxWCTc2SR4XUwsO3udvAjQvOkrUUNSY0TjTMlGxJBY1+z1djDP5Q4UK0gSaE+GwAxYrsKDO9pxU09y3lu1JEyEDMnFso1yxcUnHlyPh3Z1Uy0j4ZqEfSD3n8pV+ODJSZiGBDPwhU/fjp03zU2R/NOv7OexiSws28W6nkb86qdvp4VOM+gDWbumkdauaYQQxK/sOQ/lKrzwfD/27RlAY0OYI2EDkgDbcpGZLUI3JKyyg+mJHDq7G/CHf/AHcBwHJAiu44WPLMtCMpnECz99AU8++SQ/8sgj18RhSinhui76+vqqjEMIgXK5jLvvuRuDA4OYmJiAruvV5H4qlbqm4Zrg/X/v936vmpNxHAdCCJimiS9+8Yv45Cc/yYlEgt7qqsBXg5XyWzaE5a+buNmAkB6FZWchF2EiDIYUOnLFSaTzw9yUWEdzHeZYAZh4TjmW3Eyx5Ga4G/8lZ6b2IDP6NEqpg3DsNCTJqtw6AJAWhiAdLoDi7H4Upg9AhtsRbt7F8Y4HEGm6ef5wqZpekeD9sulTXCqOQ+pxjw3QUjecRGtj7yWFrxZjaUSEc+k0pysVrGtoxB3tLVR7c66Lmzids/nArAsG8NKUiwaDuNGgi4II4KnwFi0XpyZLGM9akAREDIGwIUlc0EXogUmmZFXnh4d06QNNAHrAmdEcT2fLkAREwxru3NJaLQ0mcaFEiFIeS3n8mdNI5SoQYHz6Qzuq4JHNV/h/PXYIx/qmEY9ogKswOVPA//2lF9mQojpZUQpP2FHAF10s2SiXbZim5lWtxUywUshmSsjMlryOd0HQpYBtK1gVG61tDZianuAfPfUUwuHwBbmHAKy+8pWv4JFHHrlmOYbx8XEeGhqaNyjKdV389V/9NX7/938fX/va12CaZvU8XktF3iA09fTTT/Px48eRSCSq50UphVAohMHBQXz3u9/FZz7zGTiOA03TULc6A7lsBhIyopQIN/KElYJXJ4NFd/S2W0b/6ItoTvbCVQ7kimeU18q4M6Qeo6auh9DU9RCs4gjnp/agMPE8KqnDcO00hBYFSPebAKX3f+hQThHZsZ8gM/ECjNg6jre9HYmO3TDCHfOGSwU2OfL88p+PCK5yEIt2IB5tv6zwVW0u49TMFB+ZmkZbJIZdPnioBb0KG+M6lV3iwxmvKutkxsXdrdoSu0nvuZMFh5/pzyJdtFG2FMqW4/WBMBDRJaQ/EnYx2JnJWZCCYDmMQsWZ96FJEI4PpuE4DKERYiEdyahBQSc/+30poGComNc42DeY4mf3D0IThG3rW/Du3esIAPrOz/Lff+swxicLiEUMz5n5AogjYxVowutNYcVQjhdqIuXtmHXhzRgheJViwm9b1aQ3/13TJHTNG1YVjRrYfO9adPTE6atffZzT6RSSySQcx7ngxjZNEy+//DJSqRQ3NjZe1R13AABnzpxBKpVCNBoFM6NSqaCjowNbtmyhrq4uXriuXg9F3m9961tYbD0Hx//444/jM5/5TL0fpQ4gV5oAUhAk0BztxGiqD6SZi6ZCmBUMPYqTA09ibcdd3Nq4ecFMkIuDyaIy7pFV1LTmo2ha81FUcmc4P/wDFEefhFuegtCT3shbZjA8GXehh8AkYRVGMHn265gZ+iGizbdyU897EEmsr94uk8PPcD59BlKP+Z9RLAqgrGy0t2ypmZ0uLgs8jk+N87GpSYT1MHZ1tHshLczXpQoevy2p0UDR5VmXMVFScHnxMe3Ba4/mbJydrSChUbXfw1VeWElKWjTsSAAqtsJktgJdEiwAwzNF3E0t8y7Ta2dmoOsCgoBMwcKJgTTv6G0iuciLBiq9X/7OES/Wrhir2mIYGM3wT/cP4/l9g4BiRCN6te8EgJd8lwTHdqFcRiSkobktgraWKFqbo2hIhiCIYFsOKhUHrqMghUA4rCEeMxGLm4hGDeiGhK5LRHyQA4BnnnlmWQdvGAbGxsZw+PBh3H///biayeMAQI4cPVJlI0op2LaNNWvWgIjQ0NBwwfOuFYAEJbvFYpFffPFF6Lp+AStTSkHXdezduxfpdJobGhqoPtzq6qyDtySABGjRkVyLw0PPYjkFWoKAYgs/3vd7uG3TJ3ndqgdg6NH5EiUrqpSqrY6aC3GZ8Q1k3vQbSK79BOcGHkV+9Ck4ldS8eSCoGS6laQYUu0hP/AzZ2cOINt3KRrgNldIM0qkTEDK07DxspRyEQkm0Nm2k+QB36ZatlMAAbm5vR9wwlr0pBQG9McJMGcjbjIyluMm8MBcSOOAtLSYdHNM5XXAgfBFGZk+YcSJroW+iyG1xPciYwFVeFdbLp2dQKDvQhUDIkDhwdhYdyRBv6UnCdRWeOzyG8VQJhhS+o2F86TtHsb4zwa0NITTFTTTFQ2hOmmhKhOC6jEef7kPfYBqJiA6WwHMHhvDcvkHYtkIspFcnCJIfxiQwCkUb4ZDE1o2t2L6lDb1rmtDSHKFAJ+vydn0M13Vw4MABSCmX3AUGTv3gwYO4//77r8nNfvjQ4Xnvx8xYv349AMwDkOC9A0HFq+20A3A8fvw4BgYGqiG1hc7ONE2MjY3h0KFDVx1U32xW70S/SmGsroZeaNLwZnIsAzZSaHDdCl45+iWc6v82VrXdzqva70Rr0zaS0qzJQaxUnmShNAlDC3dQ45YvILHuU5wb+RFyYz9GJT8EJs2fByL9UlcXgNfxrkDITh2AAoGEedHGQSIB17XQ1XITNM287OR5cIS3dvTQpmaLm8IR4iUWZlCeCwJ6woQjEihYwFhRockUF0hzBHqHIU1gTdLEVNZGaEFfh+0yHt0/AU2g2gfiOi4s24XtKBiagOt6oShXMR59cRAhXcBxXS/n4M8UYT8voRTjSP+M1/uhFKAAQYyQLuE4CoWSjXhEh+0qSHjMRQiCGTEANVeuKgXBshxIAu7Z1Y137F6LVR0JWujMVuLP55c5k99PJHDy5Gk+e/YsQqHQvJs4SFbX2tFjR6/6vRM43WPHjmFh0+KmTZsAAI2NjRfsUq+VoGLwHvv374dt24hEIheE9WpBdd/+/dcMVG80hrDca7ylGUiwiNsSqykeauJyJQ0hlinpZa9nwtDjKJZn0Hfuu+gf+B4aYqt4ddcDWN3zMMJ+M+CljbzFPGkSsII0W6ih99NIrv0YClOvcG7sJyikjsGxMiAZBmQURNJPkBOkFvHKkUmDe5EJJ4pdGHoMnW03X5WbOaRpCGnakoOoqj0fNOf8CZ7e1cm0i01JCV3QvFoxIk+LeDBj8emZMgxtTvuqlqXoGsFxFZTr5S2YPfkSoQu4rpoHOJGQBsevVAobGhy/mVD4EinMDFOX0GXQF8RwHQXbdry5IyEduUIFjXEv1Om6gTCk14lO8AAlV7CwpiuOn//ANmzubaZawAgS8/PEGi/BXNdz1Hv27EGpVLog/6GUmmtA9H8+3Xd6ntO/Gg6FiDA9Pc39/f3V3X7gTDZv3gwAaG5ungdoQghMT097vUdXOf8QHPPevXtX5AwP7N9/TYDsrRZieot3ons7OlMLo6uhF6fG9sCQkYtMJvRuFCl0aNKAIIVCcQzHTv09Bga+i55V7+A1q9+PSHTVvPCWxzZWNj8dQTUYK5AwEGu/l2Lt98IqDHFu8mXkpg+gXBiG65R8MAn78vIE+CXBy7Eu16mgq+tt0PUIXS77WAoklvp93lY8VHSRsoHpCsFWgEZA1lb4yYjFd7frSPrt6hWXMZ53+MRUGadnymC/+1wRLpT5AEESgXwAYia4wAW7eyJvqJTyy7VVtWcBKFVcKOUiamheqFK5sGyFfNH2wMivyCoUbDx05xrs2tqOP//ng/MyS4GjzBUs3H/nanzyka0UMrV5XeVXw1cFDu+FF15Y9GY2DAO2bVfDSbqu4/z588jlchyPx69KzD8I+/T19WFqagqRSMRTafAT9xs3bgQAtLS0VPWxAi2q2dlZFAtFjkajVzX/EITyDh06VGUZy+VKDh8+DNu2q6W+b8U8yNUA8bqcu+9217fegpOjP6uGtVb0TH8miCYM6DIExymj/+yjGB36ITo67uFV3e9CsnEb1XaEX0quZA5IPDdsRHuoeV0Pmtd9DOVsP+dnjyA9uQeV0jQgV3K6yMt9hBvR2bHTL0e+OjfOcuBxLlfhV2fKyLsSLiR0qUMnwPVzGaN5hW/lS4hrYMGMguUiU3bgulwdvMQBw/Cro5gB22FYtgIrNU/KRMCT/lA14FG2FAwhYRoaciXLYwz+7zd2J/C+O3vQ1RyBLoU318RRmEqX8Mf/fBCFkgXbdrGqPYbbbmrD1394Ao6rYOoE+Bpctu1CAPjMh7fjnXevrUqSLCaQeCW7SSklLMvCvn37qk5T0zRkMhn8+3//7/Gud70L73jHOxAOe8OzdF3H+Pg4zp07h5tvvvmqOMtgV3vo0CG4rgshvDySbdtoaWnBmjVrAABNTU2IRCIol8uQUkLTNKTT6WrV1tXcBQshMDIywufOnZuX/6g91gDkQqEQzg8MoL+/nzdv3lxPpF/HDOe6B5Bg972+fSdiZiNct+xNzbtIGIhIVPMW3pwQF4IkdCMJZhsjg09icuRpxONruKn5VrS034tE0/bL6CqfP6WQ4c0DCSXWUyixHi1rPogz+3+HS8UJkBa+6O7VcS2s697tix9eHfaxHCMZyJd5z2QBChIdYYnmkIaiSxjMc3XIlBRASBAkMbIVhbylYEqCxYzephDWJHUYfpmrIQmmLx+SrzgYSVUwnqmgYjsAA5btYipTxkzOgiG9U1exFdZ3xPEv7lmNiCHx598/iZHpAsBASyKE/9dHtpOpX8jazgxnuFTxZohIIZDNW/if//gqiBhh3Su11qRAoWijpTGMX/vYrdiyvpmCSYZXEzxqQ0enTp3is2fPIuyr2gaNg7/+67+Orq4u6u3t5XPnziEcDkMIgUKhgBMnTuDmm2++KgKLgbM98Oqr835nWRbWrFmDpqYmCnIgiUQChUIBmqZBSolcLofJyUl0d3dftZ1/4MSOHz+OdDqNeDwO13WrPSnBMUspqyCczWZx8OBBbN68+ZqLTt7IVp8H4oexYmYDrW3dzieHX4RuROcc/BKsxXZKMIQECc0fLsWYK9El6EYCgl0UcudRTJ/ExPlvIp7YwC2dD6Cx8wGY4c4FFVxyJXdudXY6KxckNOTTJ9iuzHqijcsJepGA45TR0LARrS3bKJCAv5bgMVu2ee9EDpIkbmsOY2PCpMCnHtbAr84oEANJg/C+NSEyBMFyGd8/m+eBdAU3t5p457r4kh6mNaZjXfOFoFm0XH7m6CR+emIKIU3AshXu3tyCjgZPBbYtGeKBiTwkEcKmrI6ddVyFdN7igfEsXjw8hj1HxqFpBPLDXpbtwtCFPw/EO9BMtoJbNrfh1z5xK5qTYXKV1/R3rW5WIQRefOkllMtlJJNJMDMKhQLuve8+dHV1ETNj8+bNOH36dDXXAgCHDx/GJz7xiasW+mBmHDl8uBouCn63ZcsWP1fjIhaLUWNTY7XRUAgB27YxOjqK22677artXoPXOXz4cBWUAkDr7OzEl7/8ZXzuc5/DyMgIDMOonpM9e/ZctXNyI4ewlp9k+hZnILUOb2vXPTg58uJFHy2Fhsb4WpRLkyiXJyFJ+cq7hifO5w+YAhhCmtBkCIIdFDJ9KKaOYuLs19DQehc3d78XsZY75rOSizr1oFNaQ2r0GR47/TUwNJA0l+VMzApSGli79qHX5ZxarsKeiRRcBt7eEUN31PDCOn4ieUejpIG8y1MlBZe9PIZij2G0hCX6Z4FNzSEw/A5wWjzPwtVJ7YF2FiNiSHrktk6cnyrw4FQRpi7x4okpbOiMc7ZooW80C0PzlItHZ4r43a+9ysmIjnSugtlMGdlCBa6rEA55VVqqZpcdTE8slG0YmsBHH96Ej75rCwVVXNcKPGp3/s89++w8RsLMeOD++6uP2bx5M77//e/PS2AfOXrkqsS9A7AYHR3ls2fPXlAuu2PHjiqAGIaB9rb2ajI/+PyDg4NX1flUQfLI4Xk5EcuycM899+DBBx+kRx55hP/0T/8U4XC4+nn27dtXfWzdLupy6gCyJBL7TntNy3ZqinZxoTQJXWoXZGKJJGy7iM622/GOO/8TlSopnpo5jInJvZidPYxKcQKG1KDLEAjKk2T3w04AQ2hhaIiAlY3Z0R8iN/YUYo3buLHnw0h0vcsDEj/hjsVKgasjagkTfX/PM0NPgvQEQHLZng8iAdcpYe36DyIcbqbXI3R1eCbNM2ULuzub0R01yFOm9YHAB5GNCYmpkkK6onAyZfO2Jp087SqGFN5o27lhUksdnj9ACZ6cem2mOqRLKMUwdcLAZB5/8PgJWJaDkuVCCsB1PWXe0ZkiTg1WvIFPQiDiz/hwHVXNoUhBcBSjWLIhBGHnpjZ87F2bsaGnoVplJa4heAShl1wux3v27IFhGFVHKITAPffcU33s1q1b5z1P0zT0nZqTXL+S0FHgMI4dO4bZ2dlquCj4LDfffPO8x3d1dV2Qizh37txVPTdBaKrvVN8FJcW7du2CUgrvete78Gd/9mfVajHTNHHy5EmMjo5yV1cX1cNY9RzIle2sWEGXJjZ23IG9Z74FUxoA1AUnSwgNhfIUXGUhbDbS6q77sbrrflSsNI+Pv4TBge8jn+mDJjSYWshnJG4VADwHK6DpCRAUiuljKM28iuzAN7j1pi8g1HjLPHFzZjUPwMAKo8f/hNNjz0GaLVC8fL4mAL3m1pvR3rHrdQGP6VKZ+1I5bGpIoDcRJub5XemBL1kbE3RwGlxWhFcnLXRHJSdNQY4XDYRa5n143jx17wUtR2EyW+HRVBmnx3Lon/TmoSulYOgCxYoDsCcRohwvLOW6QDJq4DMPbYAA8LWn+rykfQAI7M1nL5UdhAwNu7a24z33rMOtm9vmJcqvdRI2qHzau3cvhoeHEYvFoJRCpVLBqlWrcOutt1Yfu3nz5mqCPXCWw8PDGBgY4E2bNtHVAJD9+/fPCxc5joOmpqZqCCuw1atXX/Dc/nP9F4DKlTgwIsLs7CwPDQ1Vq76qJcVbNkMIgdtvvx2tra3IZrPQNA2GYWBmZgYHDhxAV1fXDd8Pstg1rFdhXdVciL97W7UbB88/AQW1iACIgiZNpLMDmJg5yp0tO4nZAZGEaTTQmtXvR0/3uzA++hwPD34f+fRxCHZgaGE/x+HOy5V4Ia4IpIygnDmO0b3/Gsnu93O042EYiQ0QenJeBZdTmeGJE19CdvoANKMJSrnLluyCCMqtIBRpwdr1j+BqVl0tZ8dm0ghrEre2+Oq3iwo5AiFJ2JSUeHXKhmLgqcES3tkT4pKtAAKcmr6PC0HD+32h4vLZqSL6xgsYmikinbdQtryEui7nQIjZ05myXFXt2/A2DgxD0zA0WcDJgVS1l6NSUShXHACMzqYIbt/cjrff2oV1Xcm5lBeuLetY7Kb/4Q9/OI95WJaF22+/HclkkgJxwN7eXjQ2NqJQKEBKWU0aHzt2DJs2bboiZxk4nSD8UzvrY/v27ejs7KTacbpr165dsAETGDg/cMG0wCsFkKGhIczMzMAwjGoILRKJYH2v1xXf2dlJ27dv52effXZeHuSnL/wUH/jAB95SAHI1WUadgdSEeRiM1sQaWtW4mUdmjiKkmRc2FIDAYPQP/QRdrbcBkP6O3h9hK3R0dT9MXd0PY2ZqH08MPYn09F7YVhqa9CcSzrlEn5UwpBYBsYvcwDeQH/wOZLgDWnwjG4kNEEYLnMo0shM/g11JQdOTUH4n+nKQGAgyrt/0MWha5JqWKwbsI1Wu8Gi+iDs6WmHKCzvMa1kIA9jWqFFf2uGSzUhXFB7vy3vqs+TN2agF+OB1SrbC+Zkyn5oooH+qiHTB9spZ/RnoUb//wlmohcSMhqgBZoV0tlJtOswULPzk1WGAAcf2cletyRC2re3Eri1t2LquiQxdVm8YT+drpWoDVy9MY9s2nnrqqSq7CJz0vffeO+9mbmtro+7ubj58+DCi0Wj1mr/66qv4uZ/7ucu+6QMAKJVKOHLkCHRdrzIjpRRuvfXWKhsJ3nPdunUgP9Ee9KWMjI5iZmaGW1parnhNBscyODgIy7KqnfmWZaG7uxs9PT1VkHnggQfwzDPPzAtzvfjCi1cNzN5sdjUr4N7yAOKdDK9Edmv3fRicOghCaNHH6FoEo5P7kC+McSzaWQ0L1SrvEgk0t95Bza13oFQY5unRZ5AaexaVfD8EO9C0UHV2upcr8fbKQk8CYCgrjeLUKyhMvQyGBhYaoMUh9KiXoCdtWfAgApRbQe+WzyISW3VNQ1fVLT4RzmUyiBsaepMxX9ZkedQxJeH2Vh3PDJVhCsBRcwDTN1PBmqQOECFXcnk0a6F/uoSB2TJSRctrMAQhYkgwU1XKJAhB1YKV5Sh8YNcqbFvdgB+/NoqXUhNeeIsJtt/d3ZYMYfvaRty8vgkbuxsobGo1VL2mIfB1XpfBLI8DBw7w8ePHEYlEvNCBn9/YvXt3lR0Ej920aRMOHjw4L5F+4MCBKwpdBI749OnT8yTcA7vrrrsucE6rV69GPBarzuTQdR0z09MYHBxES0sLrhaADAwMVN83qPZau3YtotFolZm94x3vwH/+z/+5CmahUAjHjh1Df38/r1+//i2XB6kDyFU2Ue0JuQMN0XaUy2kYF0ibeDM0rEoGpwd+gJ1bf3lRNhMACQCEo93Us/Gz6F7/SWRnXuXU6NPIT78CpzwJXRqADM2VDQf5EtIhdN1LkEMDk4SC8F9TXiQcR7DsArrXfxSNLTuuPXj4i9FRCkO5PDY1NVaHTC2HIOQn1DcmNTqf0fhM2kKICLZiGIJwJlXB1BGbmRnZoo1ixa3KpId1b76G63izyyu2C8dWkPC61p1Ajt3HNl0KvHRiGs8cHkc6V0HIkKj48vDrO+J48NZO7NzQMm8KYe1skNcrVLXcTfqNb3yj6oiZGeVKBT09Pdi+ffs8AAG8aqhHH3202gthGAaOHz9+RR3pgYN99dVXUalUEAqF4DgOHMeBaZq48847q58jeO3Ozk5qb2/n8+fPwzTNKpM6efIkbrvttqvWg9HX13eBYwyKCYLzt3PnTqxbt64qtqjrOjKZDJ57/nn09vbeUP0gtXI2b9UQ1htwJQmKFUw9Qhs774bllhYXBmQFXYvi/PAzKJVnmIgWrZX2WImolvWS0JFsvYvW3vIfaPPuv0P3jt9CKLkFys5jsQJVsAKzW/26WHOjV70lYNs5dK15D9pX3fe6gEewiKZLRS47Dpr9LugVaXf4oax7u0xqCgmUXa7Ku2tESJUczBQdKPak3COGhCYIrgJKloui5UISsLY5jHfc1ILP7O7Bb7xnA3ZvakbZdquOnwhI5S0USg7ChkSh7CAW1vBL796I3/rkLXT31nYKGV7VlgrCVIGMyRscf9Y0DcVikb/9nW9XZcqDXfbOnTsRjUZpYansLbfcMhdy86XdR0ZGquNnr8SxvLxnz7ycSKVSQW9vLzZv3lydKxMwn1AohDVr1lTlVQI7cuTI1XES/mueOnXqAocWnIMARCORCO3evRu2bVcrtwDgh08+WWUubyW70Y/3DTm6wFls63kQphaBWqShMKjGKldmcfrcd30vyMttB+aFt5gV9FArNa3+MK296y+oad0noZzilTn6oBnOzqNr7SPoXPPe1wU8ai1bqXjlr75zWnhKeBEIDM53SCO8d20ETSGBgu3DMQG6JBh+57mjgKLlomi70CRhQ2sEj+xowa/c24NfensPPbSthTZ1xsjQCMOzJWj+Tr1KaSVB1wSyRRs3r2vEb33iZuze1u4NkVI1oEFvLGgsDF8BwBNPPomzZ84iHA7PC/s8/PDD8wAh+P327duRSCSq+QgpJRzHwSuvvHLZO0dN0+C6Lvbt3TuvgdBxHNx5150wDKPaAV772QNxxdpw2uHDh6/YiQU5mXw+z6dPn67mZJTrQtO0aklx7fu+5z3vmceoTNPEiy++iJmZGRYL1kvd6gzkMvywFyZqinXT6tZbYDmlRZ0wswtdi+Lc8I9QKk9zkIRfyet7z3f88IgGLdTq94rQZX5mCVYOlFtB94ZPoOMNAI9q8IyA89m8HxKcDxpLpZ2DQuSEIejDG2N0a7sJTRAqDqNkM0q2C9tlhHTC5rYIPritGf9q9yr8/K522rU2Qc0xnRzFGEmV+cdHp/hPfngW/ZMF6JqYB2JEhFzJxjtv7cT/+YGbqCFmUBCmeqOZxsV2iX/9V381jw3bto1EIlF1iMHjgu+rV6+mdevWoVKpzHOggQjj5YSvAODs2bPc19dXbcgL7B3veOeSDmW731wYvI6u6zhx4gQKhcIVOe3geadPn8bo6ChM0xupYNk2Ojo6qsAVyJgAwIMPPojW1lZUKhVv4xIKYXx8HD/+8Y/BzBcMobqh7Sos+HoOZBm7ec27MTD+ypJnWggd5UoaA8NPY8uGT/qNfkvlJ7iapPccu4BVHOGZc19HZug70LWYH6a6dPBw7CKE0YDVW34RieabX3/wCGTxIxGYUmI0X8RrUym+qSlJoZrBSTlbcc5hdIYvnPkXgIgpCff1RGhXR4gnCg4KFReCgLgp0RHTSffjW/mKy6cnizyetTCRKWMiU8HwbAmFoo2YKbwZHjXOQAhCNm/jXTu78HN39xD7ApVvZG5jJexDSomf/exn/OyzzyIajVZ/l8vl8P73vx9r1669IPkbPGbnzp1VdVrXdaHrOvbv3498Ps+xWOyS8iBBiOyFF15AoVCoyshbloVEIoH777vvAkYRvPaO7dvnDb4yTRMjIyPo6+vDzp07LzuRHhz33r17YVlWlZ1VKhXcdNNNaGhomHdulFJob2+nt7/97fz4448jkUjMyy994hOfqCfRbyDT3rgT67GJ7pYd1N64iafTpxHWwtVEt1eJ48mzK2Ujlx9eFjQCKfcgjJWfPcypkR8gN/482E5D12IALhE8fEVfx8og0nQzVm35JZjh9jeEeQTOvyEUog0NjXx8NoPT6RxGCjY3hkyABFIWUHQFdjSa6AwvLktSm/SO6ILWNRj/f/auOz6O6lp/d8r2XfVuS7LlgnvBxgYbDJgeqgMEQoAXIARISIE8CIGElkeAUAKBhOLQQ4AECMWATTFgg3vFNjZusmT1ru27M3PeH7t3PLvalVbNlmEvv0GytJpy597znfqdmN97giqtrfZiV5MfTZ4QfAEFYSUSK9I0DdNKXTiiyIENle3Yvr8Tkhi5s0iPjjBmj83FeUcPZxovgDtMNsI999wTw3rLNb8rr7wyRpDGa4Vz587Fc889p//MYrGgqroK69at63U3Ph4Y/+CDD/TzRVvIYvbs2SgrK4up/zCCyRFHHIH8/Hy0trZCluUYRuFp06b1OXit13J8/rl+T/xnPCMsHkAEQcCCBQvw5ptv6rERq9WKjz/+GDU1NVRSUpKuSk+7sAZk1sCYgKOPuAQgQlgN6PUeihpCKOxBINiGDGc5KsrPxgGRSNGgNweOiMUR9NdTfeXr9PWKn9E3q36J5qq3QRSGFE3b7Z2bSISm+ECagvyR52PE1JvZoQIPo/BXNQ3jsjLZhOxMmEUBnrCCSncQ+70huGQBJxdbMS5D6rGCgteIEEW4s7QovclbW1uxZEc79rQE4AupsMgi7CYRVlmALDJkWGW0ecNo6gxCFCNxKYExBEIqSvPs+MGxZSxSjDj0wUNRFIiiiHcXLaJFixbB4XDoIOLz+TBlyhScccYZjAfZE7m9jjnmGFitVr3hlCAI0FQNH374Ya82PweGlpYWWrZsGcxms26REBHOPPNM3fKJF/BEhJycHHbEEUcgFArFCOYvvviiz5owBzCv10srVqzQ4x8cRObOndvl3BwszzjjDBQVFSEQCAAATCYT2tra8Nprr8W4677t1ofAvt0gKR3aCY7EQopzJrKTj/wNrf76eQQCLTCJIlyOYchxjUBx7lSUFM6GLNmYsc8HtzSCgRZqb16HloZl8LRughpsgSzIkEQLRMkMRLOrWMr3JIJIhaL4Yc2ahIJRl8GaMZpzvR8y8OCbThQEQAAm5mWzEZlOagsqEAURGSYJNiniKyKk2uw38j/+WZPIcN7EHHQEFATCGr6q9eCrGg8kFgEYUWD4/JtWKGEVJhF6GrEWpS754XHl0boP0tl3h+rgGnBnZyf9+te/0qurOQgoioKbbr4ZJpMJvMYhHkCICGPGjGHjx4+njRs2wGa36+f98MMPcdddd6VsfXCX2AcffICGhgbdfRUOh2G323UASaS1q9GA9qxZs7B06VK9iE+SJKxevbrP/Fz8WVauXIl9+/bFULsUFBRgxowZCV1qqqoiOzubnXX22fTUk0/qhYeiKOLFF1/E9ddf/62yPga71iMdA+kRRAgji45mpfnT0eGpIVmywGHNZ4IgdRHuAODz1VFry0a0NK5CZ9tWKIEmSIzBJJohmzIhRAsHOZVJysABQA21Q7QVo3DU+cgadhpDFORSa1A1uCZupE/3dtq5cyeGDx+OqVOnMrssG5x5sW6qPhiEsJkEZjOZ4AuptF6lCMli9HeKSjBLAqwig6apUNQIULiDCs6eUYTibCs7HMCDZzEJgoCfXP0T7Nq5SxfYPPZx7LHH4gcXXsi4IE52Hl48t27dOh14rFYrNm/ejK1bt9LEiRNTctdw99VLL72kWxX8Xk488USMGTMm6Xm4AJs3bx7uu+++mKZOu3fvxldbvqIZR85gvXGnGc/93//+N4baJRgMYubMmcjJyen22a668ko8+8wzUFUVRAS73Y4NGzZg0aJFdM4557BEwJzKPuBWGL+ukZdLd38bjsF2O7Fu67C+3TGQIaEGRDZMhAMrJ2Mkc9mLY8BD0xS0deykb3a/RitW/Za+WP5zfLXpATTUL4eq+CCbMiDJDkSEvdptn5GuF49sKDXUAcYEZJV/H2UzH0TW8DMYog2t2CE2Q/nmvemmm2jSpEk4++yzMW3aNCxYsIBaWlspQsdO6C/xB2NAkydMS3a00VNf1mNbvReKRggqBJPEkGmT4bCIUDQNQYUgCEAgrKI834Z5E/KHLHhwgaqqqq7pC4KA6667jl579TUdPLjmLssyHnnkkZg6hu6Ew9nnnB0TwJYkCYFAAK+88op+zlQAbdOmTfTpp5/CZrPpPyMi/OCiH3Tr9uGCdNasWSgqKtKzwnhB4Scff9JrTZYDmMfjoUWLFum1MfyZeWZaonviczFz5kx24okn6nxhnGblD3/4A0KhUEpzY5wjfn1JkiBJEgRB0LO/+M94Uy1joaWmaXoxpqIo+rm4Oy6VY7Ctk8M1tVkaCps70gEwVkgHQ53U3L4DDU3r0dz6FXzeapDihyxIMIkyTKYMCNDAeCFgwgqIbpw30etp4Q4wyQnXsO8hc8SFMNmHR4n8ovd0iMGDC7wnnniC/vznP8d0fXvzzTfR3NyMDz/8MNp3uo++bkSqyj/b00lb63zwBhVoGiHTKmFUrhVjC2zIc5hgkQWmEaGpM0hvr6tDbZsfAHDmkcUQBab3QO/Vu4/bpMb77+lZEm066tIeICJIjedav349/e53v8PixYv1Og4u+Nvb2/Hggw9i2rRpjM99ssHfw6yjZrEJEybQtm3bYLVaoaoqLBYLnnnmGfzqV7+injR1fp8PPfSQ3sRKVVUEAgEUFhbivHPPi4kvJPpbVVWRlZXF5syZQ//+979jXFbvv/8+brrppl5ZH9y6+uCDD7B37164XC6oWqSlrtPp7JLanOw93HTTTVgSjQdpmgabzYbNmzfjN7/5DT366KNMB3VB7KL9aJqmu774vdfU1NDSpUuxatUqVFVVIRgMwuFwoLi4GKWlpSgtLcWwYcNQVFSE3NxcOJ1OxoFmIJQQDqyp7rM0gAwGaCBK5c3NzChtiNvXQHUtX6G2aT1a2rYj6G8CIwWyJMMkyBBNGRBBACl9AA2AZ2qBFGihTgimLDiHnYmMsgtgclZE+Re1aFHioTfOuNsqFArh/vvv17UqbsKbTCYsW7YMixYtogULFvQo8JK5rRgDPvimg7bV+yAxwGEWMb3EjqklDjjMIosHm2HZVnbsETn01Md7MWdMDkYVOnptfXCBejBMfE3TsG/fPvryyy/xxhtv4P3334ff748IxehcyrKM9vZ2/M///A9uuOGGlN0rPHX3Bz/4AW699VbdjWU2m1FfX48//vGPeOSRR7pUiccrCGvXrqVXX30VjiinlSRJ8Hg8OP/885Gbm9vju+UC6Nxzz8W///1vnXDRZrNh5cqV+Oqrr2jixIkprxF+rwsXLtTfkSiIcHvdOPnkkzFy5MhuQZFbISeeeCI783vfo3feeUe39FwuF/7617/C6XTS//3f/zHjXBj/3ij4P/vsM3rmmWfw/vvvo6mpqdt7F0URDocDWVlZyMvLo+LiYpSUlKCkpARFRUUoKChAbm4uMjMz4XQ6YbVaYTabGbdeEu1Broj0RuAPtgvtOwcgpKfoHhAcbZ4aqmragP2N69HSvhPhUDtEBsiiCSbZDhGAwFQ9GN4nnGaRznikBaCGvRDNWXCVXwhn2QWQHSNjgANDKGuCL969e/dSZWVll0XLN/DWrVuxYMGCXmsxvAHV1sYAbW8OwiwJyLaKOOOITOTa5QPdDbnRZgCc/S1+WE0iTplcwKG5188FABs3bqR9+/bpbg6TyQSLxQKHwwGbzQaz2QyTyQRJknSXjqIoCAaDCAQCCIfD+hEKheDz+eB2u9Ha2or6+nrs27cPu3fvxt69e9HW1gYAsNvtOnhwl0h7ezvOO+88PPXUU6y3qbcAcNlll+GBBx5AIBDQK9IdDgf+9re/4ayzz6aT5s9n4XAYsiFmxd03iqLg5z//uQ48qqrqwv9nP/tZSposv9/TTz8dxcXFaGlp0dN5vV4vnnzySTz22GMpWx+CIGDt2rUxtTGSJEVcaj/4Qcz66+ld33ffffjkk090N6GqqnA6nbjnnnuwbt06+u1vf4tj5hzDTHJsSvmuXbtoyZIleOWVV/Dll1/qll1GRkYXIc7nh1uzvK1v1b4qrKE1XQWfJMFiscBisXAAIeM6M74jURRhs9lQXFyM2bNn47LLLkNeXh4bSKshbYH0aG0Iulbf4W+iPQ3rsKd+NZrad0IJuyEzESbJBLPJCQEUdU1FaNipT7ARzdQiNZqOG4TFWoSM8h/ANfxcSLaSA8CBoQUc8YuqpaUlprVqPIj0dfEJDAiqhNU1XggAsmwSLpyczSySoLfFNdYAcheZN6jQ8m9aMG9cLrIdpl5ZH1zofPbZZ3TjjTdiy5YtesVyvPYmSRLEqFYoCoJeDMPdHoqixPixuxPyZrMZLpcrxp8uSRJCoRA8Hg+uuOIKPPnkk4wL4lQ1R143MmzYMHbxxRfT3/72t5iYiiiKuOzSS7F06VIaO3Ys4+4yLsQA4Nprr6VVq1bpfydJEjo6OvDjH/8YRxxxREpWg9GNde5559LfHv8bLBaL3rPjpZdewo033kiJiiKTne/+++/Xiwe5S624uBjnnntuty61+LkZN24cu+uuu+jGG29EZmYmwuFIawCXy4XFixfjo48+wvjx42ns2LFwOp3weDyorKzEN998g46ODgCA0+nUYyac7LI7DZ+/c7PZ3OVz3B2laRo8Hg863W5QD+uIr7P//Oc/eOKJJ/Dhhx9SaWkp62nvpl1Y/YptcOBgULQw9jRuom01y7G/ZQsCwXbIggCzaIZVj2eoOmiwPoOGEOkmogagKD7Iogm2zPHILD4FrqL5EE1ZUeBQAQhDEjjih8fjidmQffWzJrI+tjcFqNWvQhKBk0a5wMEjUfE4gSCA4dOvmyEyhhMm5EdShnuxSRhjaG5upksuuQQ1NTVwOp36Jjduopj4iKZBMQRbGZgeOE0UM4kXFvzgyQiSJEFRFHR0dCAzMxP3338/fv7zn7PebnrjdYkIN910E15++WW9FoPzQDU1NeGUU07Bc889RyeccAIzvFe68cYb8dRTT8VwaoXDYWRkZOD3v/99r1Jv+ed+evVP8cw/DmQ/cUbcP/zhD3jxxRd1C6O7mNsXX3xBb775ZheX2kUXXYTs7OyUXWGiKEJVVdxwww1szZo19MorryArKwuhUAiqqsLlckHTNGzbti2G/FEQBN3a4MKeZ4ERkW59JhRqhmB6fLyCrwUO7nweUpljxhgsFgt27dqF559/Hrfffju+62PAASTG4gBDZ6CNNu9fji37l6PFXQWRESyiGVaTC0K0p7lGKhioTxlELOqeAqnQtCA0NQASRNjsw5CZdxSyik6ALWuyYQWpEdBgh09zG16MNZCD75ddbSEQAXl2GSUuEyMkAY+o9dHmDdOyHa04a1oBbCaxV9YHd121tbWhrq4OTqczJtOpNwH0eK0t/vt4MOGuLz6XTqcTl19+OW699VaMHj26C9Nuryy5KLCXlZWxm2++mW655RZd0+YWQF1dHU4//XScc845dMwxx6CjowOvvfYatm7dGhOL4e60Bx54ACNGjOhVXIvfx+TJk9nZZ59Nr712IMPM6XTi5ZdfxiWXXEKnnXZawhgPn0NFUXDDDTfo70tVDwTPf/azn/W6noSD6TPPPMPa2tpo8eLFOjDw57bZbDEMw/zgYCdJku62ZIxh1KhRmDRpEkaMGAGn04lgKIjamlpUVVehvq4eLS0t6OzshNfrTQoyHGCMsbjunos0QmtrKwBg9uzZAxoDSSXb71sPIBppusXR4m2glZUfYUvNCngDrbCIMiyyHQIjsH6BBjOARhiK4gfTQpBFE+z2YcjKmYas/GPgzJ7EBNFieEHRIsRBBQ6jCcyiQrr/jXyMbo+BucvIXfkVDe2BCA+WSWQ9KgYCGN7f3IgchwmzR2VH+7D3XpCMGjWK/eY3v6H7779/AAGRdbsJZVlGdnY2Zs2ahZNPPhnfP/98HBGlRu9L8kEy4f2b3/yGLV68mD799NMYEOHFdK+99ppejS1JUsJA/mmnnYZf//rXrDtLoad1c9ttt+Htt9+OsVhlWcbVV1+NlStXUnFxcUxMhgtzWZZx88030+rVq5GRkQH+mY6ODvz85z/HyJEje52swQWg1WrFf//7X3bNNdfQ888/D0mSdOAwgoaR9j0cDusWeF5eHi66+CL86JIfYfbs2bDb7QkXXygUQmtrKzU1NaGurg61tbWoqalBbW0t6uvr0djUiNaWVnR0dMDj8cDn8yW1ZuKtqVGjRuEPf/gDTj311C4xrf4I/++0C4sDh8AEdATa6LPd72Nj9TL4Q52wSRbYTS4wKLp7qi/CgUEEA0HTQggrfjDSYDE5kZE1Edm5U5GdeyQcGWOYIJhiQSMa32CDBhx80fMKeZbg9xiQjnADPYIqUVgjyCJDky8MX1gjmyyweBeWFgWKymY/bazuxE+OK42m7fYeHrlguO+++9j5559Pn376Kerq6pDIAjC6nlRVhaKqUKLB8kAggEAgAJ/Pp8dQTCYT7HY7MjMzkZubi8LCQhQWFiIvLw/5+fkYPnw4CgoKYrJ+4rNr+gNg/FwvvfQS5s2bh927dyMrK0v3+QOAy+WKqU8wZtS1tbVh4sSJeOHFF1LSiLtzGU2aNIld97Pr6KEHH0JmZiZCoZBOsHjeeefh3Xffpby8PGZ8L4Ig4NFHH6X7778fTqdTjzWEQiHk5+fjd7/7XZ9JGY29S5577jl20skn0b1/uhdbt27tVglwuVw4/vjjcd6C83Deuedh+PDhzBiXMNaRcOAxmUwoLCxkhYWFmGRgKTaOcDgMt9tNHR0daG1tRWtrK5qbm9Hc3Iy2tjZ4vV4Eg0E9o6u4uBgTJkzAzJkzWW+q+tMWSIrgoWgKPt/7IX2+ezHcgVbY5QhwcBeV0EvgYFEXGEiBogShaAHIggyHNQ85mUcgL3casrInwmYviU0zJVUX5GyQ3VQH2uxGCiEDgRYKBd3QSIUoWmC2ZMJschref6okI4mF7kAPs8iYxEAqAwJhwqLt7ThtTAY5o6m7GhnfM+HNDfWYOMyJ0YX2fhcNEhFmzpzJZs6ceVAXvNEtMtA9urmFVVJSwhYvXkwLFiwA75nOM5jiBR7v/9HW1oYZM2bgzTffRF5uXr/IBvl93HH7HXj/vfexc+dO2Gw2PTNs9erVmDdvHv785z/T/PnzmcViQWVlJT344IN47LHH4HA4dIHGYx+PPPoIioqKWH+sNaOl8aNLfsQuOP8CLF68mD755BPs2LEDHR0dEEURWVlZGDlyJKZPn46jjz4ao0eP7pLqywEv0Rzxe09UCMhBJmqRsuzsbIwYMaJXz5HqHAxUGu+3EkC4JSEwATtbdtB/t72K6tZdsMlW2E1OMFJ6CRxMZ28lTUFY9UPRFFgkCzJdI1GYOxkFudOQlTkGknTAdI1Unh9o63rQKEeiVkco2E5N9avR3vYN/MHOSLCXidF2uXbYbIWUnzcB+Tlj2AFO3d4FRXtL99DzTEfuwioJyLZKqGoPwiQxVLUH8eK6JkwrttP0YXZmlg5szve3NFOTO4RLZhfzt9XvTRGvQfZ2MyXaoMmEBhc6XGgP1uDCu6Kigi1btoxuu+02PPvcs3o2kSzLMe4ZHmi/7rrrcN9998HhcPSbqZbPrdPpZM8//zzNmzdPp2nhdRg7d+7EWWedhXHjxpHD4cDu3bvR0tICp9OpA53JZEJ7ezsuuOAC/OSqn7CBcPXxd6aqKsxmM84++2x29tlnpwT6xoLCVNZIt/EMA8gY101P5x0MxeM7Z4FwqwMA3t7xFn248z0I0OAwu8BIi2jgKYqySGe6iGtKUQIQSYXN7EJu1mgU505Dcf50ZLlGMKPQjcQzeLqlmEAe903T783LZoyhqe5Lqq1agnDYDyZYwUQzJMkMMAlEIjRNQbt7P1o796OhZSeNKT8eZpOd9fb+eBOfgc07j2BuRZYJe9uCkBGJgwQVwtJd7dhc46GJRTbk2mXsbPBi+a42HFnqQr7TzHpow94rYTvQ1tVQ2GwcRFwuF3v00Ufxs5/9jP7973/js88+Q2VlpV7zUlBQgLlz5+Kyyy/D9GnTGXfLDMSccFfWzJkz2TPPPEMXX3yxbglxenUg0uecx2iM8Riz2Yy2tjYceeSRePrppwe8lzlPnuDpsXwtGN2WxjUy4EpUH12Eg3Wd7wyAcPDoDLrp2Y3P4Kv6Tcgw2aOMrQpS0f8ZEyBEQcMf9kNmDBm2PBRnjcXwvGkozBkPhzU/gWsKUep2EV5vDbk7dqCz7WsEfTVg0OB0jkDRiPNhMuewwQIR7raqq3yXavctgWByQpLt0EiM0qNT1D4jAAIkUQZBRGtHNTZ98x4mjzmDLL0EEQ4gAyrkopcem2tmq2u85Aup0aJNwG4S0BFQ8MnONp3r3SIJmDTMqVufh0+nj0MHIlwQjh07lt1222247bbb4PP5yOfzcVdNjGtmoAGVWxwXXXQRCwQCdPXVV+tFfPzeOJDwf3PtmrvUohXkepbaQAvXeG2eWyiHE1vvYLPxfmsAhINHdWcN/X3tk2j01CHD7AKRAo2oR6sjEi/QEFL8gKYgy5aL8uI5qCiciaLs8TDLxqyK+EZREWhqbFxNu3e9Aq97DzTFDZE0iEyAyAjt9Z+hrf4zTDj6rySbswccRDh4tDasorp970E2ZUGDGL3P5JxABA0myQJfoANf7/0cU8ecGqFLSfG6FotlwC0QboVYJAHHlTnw9vZ2mFnEslA1QBIYRFmAphFUhWCRGIZlWYb8gh5qgoVTehgqmpnNZtM/wwPVg+EW4e5PVVXxP//zP2zEiBH0q1/9Chs3bgQQyYridRKc+oRnPF122WX461//CpfLlW7+1EcASafxGoYaFdRft3xDj695CkHFB4fJATVqdfQ8yQJCig8CA8qzx2Pi8OMwMn86rCYnMwro6Iej3QjFGJeR21NNa9b/EUwNwiRZIMsuiCAIETEOUXbC765E9fYnMXLK76ICf6AmPxLzUMIeqt/7NkTJ1isuLqIIiLS561DbvJNK8sb22O6U/26wAIRF+kFhbK6FnTY6gz7e2QFNi1C4a4gG0SlCtOiyibBFc33T8NF3V118bGYw4zHx7qx58+axlStX4qWXXqKXX34ZmzdvRnt7u14omJWVhfknzcfPfvZznHbqqQPqUkuDS//Wz2ENIBw8Njduo0fXPAUGFRbJDDXFWAcDQyDsQXn2GBw7ZgFG5E3uAhqsWwLDiCXh9zdC08KwyC6AQjqhov4fKZDNmWit+RAFZeeQPXPCgHUQ5MK+vXE1QsEWiKasCPtsL9YHgSAKEmqad6I4d3TK92Ws1O6p3qGvIDKpwMqyrSK9s7UNnkCcUqB3GExDx1AQKH0FER6wv/LKK9mVV16J2tpaqqqqgtfrhd1uR3l5OQoLC3XgONxcSYfz+/7WWiBaFDy+atpOD695CgIAsyBHM6xSszyCYT+ml56AMyZdwURBjGPjFVJ+CZkZo2AxZ0MLuyEmnVQGIgX1u19GxZH/N+ALwd2yBYxJ0RfeuxdLRBAFEb5ABzz+dnLaslkq8QSz2az7swdnkUesjRKXiZ0/OZv+ua4JYSVS5KkBEATAH1YRCKuwmsRBTlFIj8G0hHisQxAEFBcXs+Li4tj9bqAsT4/+ywzWC464w9ECEXoCD4EJ2NW+jx5a8zQAQBZEaNS71EuNVJRkjYpw/uNACmjqenSkzsJkymAlxccjrPiS1ngQqRAlBzqbVsLv3k0s2hSqv+4rgEFTgwgGGhFpdtVXK4BB1VR4/O1IdRJkWR50V4cQjX3k2mU2q8yJoHLA/ScwBl9QRYdfoV6+uPQYgkKNxz14eiw/eDZUGjwO/uiJDPSwAxDOZ9Xka6U/r34aYU2BJMhJwYNXosdr00QaLLINi7e8gNfXPkz7mrcQ/zyLAgOlIOC5MBs58vuwWHKhaeHkejAToCp+tFQv0l1HAzFUNUCaEsRANHIMhv0pP3OiPgWDo6FGsGFSkQ0Os6g3iGIsEgdxB5Q0fnwLwSQR6WB69B8QBiqIftgBCE9CDalhPLjuObQFOmEWzQnBg0UD3v6wH76wFyqpMdTtRoDZXrcar676E/715e30VdXH5A91EjN8lkjtZiIZiFRYzNmsdPipUBRvcvcXaRAlKzobl0FTvBSxVvov9iK0KAOzyXoDagOZ3tld2J/3+7CbRHZEgRX+sAZJiNTqaASYooWFaTGTHumRtkCSAginqfjHljfp65bdsMs2aKQmtDrCahghNYgj8sZjeslsWGUbvCEPgmFf9DMHNBurbIcsmVHT9g0Wb3oCLy+7GUu/eopqWraQFi0OPMCemghMIjXUJcNOhiTZddqSRGKSCWYEfTVwN6/RLaH+QAcAiJKNiZINgNrviTfL1oO3OA2gYeybrlFXMGHR5Oc5IzKQa5fREVDR4VdQkW9DSaaFEQYMQ9MjPdLjMAcQqSt4RILmy2o20PuVy5BpdibMthKZCF/YjSJ7AS6aeBHG509kAOAOdtD2xs3YUrsK1a3fwBvqhEU0wSzK4MSCJskKEQR/qBObKz/A9qoPkecqoxEFMzGi8Chku0YwYwpvJIU2EnAn0mC3D2P5BbOpofYTSLIzeYyDgI76pcgoPL7/ejMRmCDBYitCwN8EJvT1fJFMLKc1G6mq8z01jUralzoOMEIqIaQRCWCwykx/BGNlOac5cZhF9qOZBbS5xgOzKGDKMAeTRJZ2X6VHeiTwwvRX+He3v4dyTEqKfwjGGFoDnbRwyxuwSpbIz7pMiojOUCemF0zCj6f+GE6zk3FB7zRnsJnDj8XM4cei0V1DW+tWY3vdGjR3ViEMBRbJHMliggaRSTCZnBCgoc29Dy3t32Dr7jeQn1lBZYWzMaxgJpy2Ip3GhMdLGGMYXnomGus+i1ZFJxS7ECQrvC3roYbaSTRl9quwkGdLuXIno615IwTG0Nts2khjLRV2aw4c1gymu8V6GD1xACVafMYn3d2p0q4OBa1+FUFFBTSCVQQNc0qYlGeByyywRCDiskhsbkVmAlssPdIjPVIBkO+UBRLp+SDg+a/fRUugA5kme5QS3ThZAjqDbpwy8gT8aNLFjIEZuLGYnqLLGEO+s4TlO8/DcaPOwt7mrbS1Zjn2NW2GL9gKkyBBlEy6BiyJZoiiBYzCaGjdhobmjdjyzb9QmD2eSovnoih/Bkyyk/G4R1b2JGa3l1LQtx+iKKOLNCeCIMhQAk3wNK9GRvEpUfAR+7VIMnKnMYttCQWDnYDYW4oRBlULozB7RNSaSo0SOiMjg2VkZFBnZ2eXOhDGGIyVzUZB71OIljUo2O9RAI3AKGIBairBH9ZQ7wljS4Mfx5XaaVyeJSGI8GsJab9VeqTHoIHLYW+BcBD4qnkXLd2/Fk6TPVJAGCeUvCEfLhx3Ls4ecwYzdh888BkW2+A+6rIZlT+Fjcqfgk5/C+2sX41var9AU8duhNQgrJJFt0oYAFmyQoAVRApqGlahruFLOG35KCmYRSWFc5GRMZqRFiYipWedmDF0NnyGjOJT+lkIF8kYE0QLCkecg71bF0IULb1qiaVpCmxmF4pyR7FUFhdnLTWZTDjqqKNQVVUFk8mEcDgcQzw3f/78GE2FAIQ04ON6BU1+DRaJQVMAVSO917nEGCSJIaxoeG9nJ8Iq0eRCaxcQSWfmpEd69B08vjMAEkmpJbz8zeKEgpYxAZ6QBz8cfw7OHn0a06jnFqA8Q4u7nwDAZc1hR444HUeOOB01rdtox/5lqG5cC5+/Se+RzjOuAAaTbIfAgFCoHbv3/hdV+xbBYcsjgTSEgy2QBDlpgJxIgyha4W/dBDXURpF+6H13Y/EYTGbuVFZYdgbV7nsfoikz+vOe/1ZVFRRnlUESTb1qSENEuP3227F48WK43e4oGEWe+dZbb8WUKVN0viL+dBvaVGoJAlYRULSuwXKKWn4CY7BIDB/v6US+XaJCpzxgbLvpkR7p0X8X1pAHEG59rK7fRlua98AhW3RmXe6+8Ia9OKl8rg4eBAIjlrIs1lN1DVZLSfZ4VpI9Hv7gRVTZsBq7az5HS9t2KKoPJskMSTBF/oJUCEyCyZQBERoC/maIjCAyCZFa6aSvBUyQoQSb4WvbBGfB8f1yYxlBpKj8DCZIVqqt+hCqEoYg2g+QPjIBgBCVwrETJAqSQYT3PHnc0pg4cSJbvnw5PfDAA9i5cyfy8/Nx6aWX4vzzz+8CHu4wUaWXYBaBntpt8KwqTSOsqHLjvAnZ6d2cHulxkEd3lehDHkC4Jvz2nmVRXzd1ETKSIKLN34GtTdvpiNzRTDRkSRF4d7oUNOoYF1dk0qzmDDau9GSMKz0ZTe07qbL2c+xvWAmftw4iALNsjgJQlGZBkBHJB0otNZdA8LWsh7Pg+AEyTQWACAXDTmDOzDFUX7MMHR17EVb8UMEAJkd6gjAJJACCaIq68kQ0te9DaeHkXgXGeIOgyZMnsxdeeKGL5hJ/rlq/hrBGkFPWfiI1HjUdIXQEVMqwiCxNV5Ie6TFEXFgHgXCzzwDCrY8dbVW0pWUPrJKlS80HEcEsmrC5aRu2Nm3FiIwSmlk0DTOKp6PAXqATk2ukxQBEb6wSRF06eZmjWV7maEwdczHVNq5FVc1naGndjGCoA7Joglm0gEFDyilQ0WC6v2NbxCIZqDa30Ta2NkcJGzn2IoSCHeTx7IfP14Kw4oemERRNgS/YCY+/DaIoQhBM8PjasL9pO5UWTGC96S/NmxRxwOAaSyLtpCPc+5oXBiCgEJq9YWRYRKTdWOmRHofWhaW3FR7KFgi/7Y/3r0dIU2BjJqiU+AEtkgUiNOzvrEFl+14s2f0BxuWMpVnDjsL4vInMLFli3FTGGEhPVgnirBJZsrGy4uNQVnwc3N79VFO3HHV1y+Bx74WqhWGSLGCCFAUTtVvrQxBMCHv3Qwk0k2TJG7A+IdwSAQCTOYNlmzOQnRM/byoamnfQ7qovQaRBEk2orN+CvMwyspodrDfNmYyWRvdpvX3CQxARgkq60iM90mMoAIgupIeyBSIyAQElhHWNO2CRTF3oSljcQxIIJtEEq2SCSgo21m/A5vp1KLIX0NSiaZhefBRKMsoY16wjVglSpi4/8DnSGW+d9mHsiFEXYczI89HSsolqaz9Ba9MahALNkAUJJsnSDYgQwCSo4U6EvPsgWfIwoOq1fp74nsq8laWIwrzxDEykHZXLIMkOBJUQdtVswKSRxw5K40Sr1NUNmYobizGGaMuP9EiP9OiFq6q/LqzuYiAcQIZiRqQEANvb9lG9rxVOyQyK0nSwqGWgkgaTIEQC6lFwISJoIAhgsMk2CCC0+Vvw4c538cXejzAyexRNL5mN8QVTYTM0jNIzt1KSmLHpwIAGQZCQl3cky8s7EoFAMzXVf4HG2k/gad8KQTR1+4I1LYyQpxK2nBndFB/2axklfcFEGgpzx7J2dz3VteyByeRAY3s16tv2UWFWWa9cWamMbJMQscx6ASKESE/0HLsUi4vpkR7p0ScA6Y3QP2wtEADY1LwHqqbqZHoRYU/QSIVDNqMz0AGZATbJDJEJYHRAOGnR7yVBhkWUANKwq3kbdjduRo4tB+MKptKUkmNQmj2GCTppohZnbaTygsSYv7VYctnw8nMwvPwc1FUtot3bHolmZSV/EWFv9SFcYISK4bNYu6eJgkoIkihjV80mZDkKYJYtGEhTJMPEYBIi9OypQR8QVgnDnDKyrFI6gJ4e6XGQRyoWyJAFkJ3t+yFFGz3xEmRVU/Hr6T/C+JyR2NT4Nb6sWYNdrXvQGfbBJsqwSDIEkO46IlAkHRiAWbJCAsEXcmPV3g+xsepTDM8cQROLZ2Ns0Uw4LdmxVkkfAu/GnulFpd9jnW2bqXn/+zDLzi7urIiGL0IJ1HNb4WBDCIgIsmTB6NLZ2LTrE8iiBcFwADv2b6LJI2YlrL3oopWk0Eed0DuyeQaAGINGhFnDHQdOkkaQ9EiPgza6C6LLsjxk71tyh3xU522BLEogAkRBQEfIg/NHnYi5JdMYAJxQOhsnlM7Gvo79tLp2HTbUb0SDpxYMGuyiGYIQdZnQAZeNBoLEJJhMDghE2N+2C/tbtmHFrjdRkTeZJpTMxfDcCUxgYpxVklo6MKI90yOsvRryi09C8/7FyS0QJkIJtkW/P/jcMrwgMCdjGCvJG0v7m3fBJDtR374f2S0FNCynnMWDaVKXmEH4JwKPXW4VIS2iHTBEmkVp7ACxon4wQAHgDak4ptSBsixzuogwPdJjiADIYWGB1Hhb0B7yQGIiQISQGkaBLQcXjDk52lCIdGFWljGMlWUMwzljTse2pq9pTe1q7Gj6Gp5gO8yCCJsoR9rURt1aEaskEm8wSWZIsCCk+LGl+lPsqPkcBa4yGlM0G6OLjkaGvTCuTzpLsVI7wsFld42CyZwFTfHGUKtw0cqYAE3xgEgB011dB1dSMhbhCqsomcZa3Y0UUEKQRRN21GyF1WSnHGeefkNBJYTOgI+84RAUjSCJEuwmCzLNFiZHM7AoAXjs9Si0tUOBSZCgaUBYi7inSI2kSoMiNO6qRgirBBmE48udmFliS4NHeqTHwGx0ULS4t79BdMbY0LZA9nsaEVIV2GQZAgBPOIjzR58Ih2xjapTa3YiSPAtrauEUNrVwClr9LbSpbgM21q1BTfseBNQQLKIJFlGKdXERQYMGgYmwmhxgIDS796Gp/Rts2v0mhudOojElx2FY3lQmRUkKjfUhyYV95OcmcyYzWfIo0NkRJTmMd/8IIDUAUkPEJOnQiUkCJNGEscNnYOPu5RAkBhBh4771yM8oJpNkgTvkR1vAj6CiQoMIYhI0iGCiDItkoXyHE+UZGci2mJlxFva6w7SiOQhJlKAhwodVaBWQZxGhqoTWgAp3UIWiAiYmIN8mYnyuGdlWMd3nIz3SY6DwA71vX5fMhSUIwlAHkGb95hXS4DLZceKwGQAAIU5oGzOoeLpvtjWHnTDyJJww8iRUtu2ijTWrsb1hA9p8DZAAWCUTBCZFKsejFo1GBAEEOcrAq1IYe+pWoLLuS+Q4SmhE0SyMKD4WWc5yFl8fkjDwHlWdZVMm/KQe6IoU91pJC4MofIiVk4grK9tVxHIziqmxswGiaIUGoKatFgQRJEhgogxZkEGCBGJiBEggIKQq2NPegUq3D8UOJ1VkOGGVRFR6QtjSFoIsSlAp4raamy9ipMPYuESO1O8TYMzWTVse6ZEeh3Yks0AEQRjaabz1vlYwxiAwAT7Fh9mF45Fvy2a8K2GyISTgtirPGsXKs0YhMPZc2tG4GV/VrsS+lm3whjphFqRI4J2xaDow6VaJCAazbIcAgttXj43fvIIde99GYdZ4Ki85DiX5R8FkSAcmUnXXFb8HBgZJtgPRupPEMK8lbz510A0RQrarEA0ddQfEuyQDiICFaqDG1/9jvFZDhAaGKrcX+70hiExGkBgkUYJCkXTceQUickyMcdLEaPMRMBwAD87MmwaP9EiPQywPklkgoji0LZCWQKcOFCppmFk4/oD7qA/cVgTAItvYlJLZmFIyGy3eetpWuxrb61ahubMSoEhTKUkQo1aJavjbSDqwIJoAUlDXtB71jWvgsuejJH8mDS8+HjnZEwzdCjkYaAAECKIt5fs+9GYugy/ojrlX3pSLGPW82BhgEgRojOk1HBHwAOYXyMgwMRaZleQAIbD+LHgY1glfB2lBkB5p4T+QFog41F1YnSEfRCZA0VQ4TTZMyhmV0H2VmntGiPr/olQmYMixF7JjR5+NOaPORFXLNvq69ktUNm6A198MWWAwi+ZI33RS+V9GGHMByLIdIoBgqAO7K9/Gvur3kZ0xioYVHovCwmNgtRVFb1LQOxV2p/ODiQPHhdWPxcYYg9vXRrXNeyCJMghc1PfWiokdCgFHZZuQYRKYRv0DiGTXo6hleqBnSOxFtOh7T4NJeqQBJBWZyRKegxOoSpI0tAHErwQhMAFBNYQxGcUotGen1OwodauEs/UKKM+dyMpzJ8IX7KA9DevwTd0XaGjdDn/YDbNohkmUI+IozioRmATZ5IJAGjrad6KzdQv27HoZuTmTqaj4eGTnHglJdrCgvzGS2puw0pzARBOYYDq0oo0BihrC1soVkboZIdoat593pREgC0CuRcRgBMRJd3dFTtzuC1OHN4ygokISBLisEnKcZiboFDbJXaADobGlR3p8W6yPZBYIEUEc6gASUsMQmYCwFsbYrDKdvkQcoFqJRE2lbOYMNrH0REwsPRFNnXtpd+2X2Fe/Ch2eaohQYZYsYEw0sO6S7h6TJCtEWEAURkP9MjTXfQaHowQZGaPJ0/41RNGSOM5BGgTRBiaYD0jyQ2R97KpZTx5/G0yyHeoALUDGIim7njDBIfU+C6QncBIYEFQ0rKvsoK+qOtDYEYAvqEBVNZBGkAUgy26i8cNcOOaIPGQ5TEnTgtNdDtPj2zIGai0nAxDJEAMZkkF0Jdq2loFhXHb5IE92fOCdIc81guW5RmDG6POxv2kj7an5HA0tGxEMtsEkmGCWTAbWWw4kkYI7WXZABCHob0KTtzrC0JtwkiMdDkVTRlSiHfw4CQePTm8T1TXvgkm26vGDATJsoBFQ7VVQaB04Nx0Hj52NPlq0uRGN7YFoPlikh4gmRHqsa6qGujY/Kus9+HxLI86YUUzzJhYkBBFVVQ+ZpvddEkrp+Rj49aRpEVc5r/FQVRWBQGDAAMR4X0QEWZaHtgUCRILnTpMNozOHAehb/KM/Li6AIIlmlBfOYuWFs+D2NVBV3ReorluOjo5dYBSGWTKD8UZSOqlj1MUlyJBEGSAFiXRvFu1nLppzdWsEhygWUtO0PWkL3v66mGQB2O9TMUUjmAYgAMLBY31VJ729sREAwW6WoKkqVFWDFjUQ+Zo3SQwmQUYorOKlT3ajsc1PFxxbzjQiaKoKSZLwl7/8hR5//HHYHXaoijroGz4tML8d8zHQc6KR1msznQBoqgrGGCRJgiRJCIVCqK2thdVqhaIoAwpoRDT0YyAiE+BXQxjlLEK+NYsd7AV8oEjwAB2601bAJlQswPiR56KpZQtV1S5FY9NqBP1NEfp2MdqhMBp470ql3lU9J9Ig20r0hcAOwSZQ1BDa3fUQBUnPpBrIITLAq2io92tUau9fV0GKgsfuJj+9s7kZJimSqKCplLTfCFGkNa4gABl2E95fux8ZdplOmV7ClOgf+Xw+7Nq1C2azGeFwOC0d0+OwBHeiAz2PTCZTTK8e/nP+mf64sEwm09CmMrFKZnQGPRibVRoxyQYw/tFru4QdqOtAtHd5fu5klp87GYFgG9XVf4na2qVwt38NVQ1Em0rJANTuuyjxXheOskMyyXxBeXytFAr5wCQLBlOfrg9oKLWL0SlJHIjoIV8NABAIa3hvawskgYEx6rG/uhFISCO4bDLeXlGNSeVZVJhlZUTAZZddhgcffBDBYBAmkykdUE+Pw350R0PSVwDhf2symSDL8pA1SaXhjjzsaq/GzIIjehQsBxFKdBeTTt9uzmIjyr6HEWXfQ1vbNqqr+QjNDV8i5G+ALMoQBVPSsxFUMNEGk2PkgfMfguELtEMjFd2TzvdjIQOQBYb9Pg2TMwkWMTViSo4zZAAAUQCW72mnVl8YVpFBUXvZoAqAwBgCIQUfrKnBj08dDUVRMGzYMHbHHXfQL37xC2RkZEBRlDSIpMe30lJRFCVl6yERCGmaNvQtkPNHHYdiezam5o5mkU0vDLEXEUvfzpiIrKzxLCtrPEKjL6XG+mVoqF6EgKcyCjrURdMmLQyTtQAm2zBOc3uwbRAADMGQd9CvJAAIqIRPmzQaafbBCg9MpgyIggRJkCAyQBIYkwRAZAwCi61S55NW3R6kddUeWGUBmtrHgCMRrGYRm/e2osMbogy7iSmKiuuvv55t2bqFnnryKTgcDoii2G0/hPT49rqC4i31wTjvQReqkoS2tjacdtppGDFiBNOixIo97ZVE82G2WHpk5z6kzzolt4JNya04HJYc4ivQTeYsNqzsbOTlz6Z1y66M8lzFUZkxAaSEYHaNARPN0QC6cNDvHQBUbfB9/gRAYkBrkNDgMwGqM+LK0hSAwhHXoEYkgCAxgsQYJBYJussMMIsMYVXDtnrvgNyMKAhw+0LYU+fGtFE5eoHUk088yXJzcun++++HoigJN9jBtkzSllB6DNQoLS3F3//+95SVI0rgwgIAq8Wir80hCSDGQr/DR3sR+KxHm2AJEAQJUEIJrQsiFbbsI7kdc8jcdAfTdSYxgAkCNJj0lsCqFoljKBpFvlc1qKoGRdOgqgRNUaFpBE3TIDMGkUUysfoLnapGaGjz6xuDsciG+L//+z925pln0l/+8hd8/fXXerYNF+TJsm8EQUgo7PuTrTMYmT6HoxWQ1LLtRoNmXTL+WJci0vj5ZUKU8JRFPqtRpJ6IC1utu0xFij1fou/jD7CIdyXZ7xljYEIk/5Q/a/xXxonjEiTAkBYJmAeDQWRnZ+P2229HeXl5StZHMhcWAFiGOoCk3qN86I3I+hMRCrZAUbyQmYRIJODAQo7Uf7hgz53Jl/ahulOYTLaDYuuEomy7JgEIEyGgEhSVIiChUSSTSos0BiFEXFmRFiORehsiBkXV0IdyjaT35A8pXQSXqqo4+uij2dFHH61vIGOWS6INk05xTY/DZaQCHnw9JwMQm802pK1j6fB+RZFJ9Xv3Q1WDkGVTXBU6g6b64cg5Mhr/OFS85ZFrOm25YEwYtAysCHgQSmwCpmSJsImMqUTkVQieMMEb0uBXNIQVDSGV4A+r8AY1+EIqAmEVYWIIhwmBsAqR9a2vQbK3ZJa71t1w856IIIpiGigO1104xF1/vb2//nJbceUnFcujp2tyABmqQ/o2LGB3xzeRxlOJ0F1T4Co8MfqStENCpsiFodOex6wWF/lDfjBhgO+DASoBLlnAsfkmdqDfB2M2iSHPAgCJr6lSpDthWCUKhDXUdoawfHcHPEGl37Ypz8bKy7B06xpJtoHSQPLtcoWl7y/xiC9C5Pdst9uHNEgLh/fCjdy+u30HBEGOowZh0LQwJEsBXIXHx3z+UGlBoiAhL7MMqqYM+KJmABQNmJRpiold8KbEFG1lyw9jBbnIGCySAKdZZHkOmU0ptrNzJuX0f9GySGGh3SJhRKGz282c1HedHunxLR6CIEDTNNTX1yesG3E4HEP7/g9jwxkAQ8DfSF7P3mgPkdjsK03xIqPoRIimTEakYihUueRkDIPAxAHVKDiRYp5FxHC7FE3HPgAsLPoZwXDEN5Iiw6FqwLBMMxudZ0NQ0fpMCy8whkBYxagSF3IzLCxi2qeFRnqkBwDdfbt3717auXMnLBZLF7ngdDrTADI4Gn0k1tHWsgGhYEckCwtGIjIVouxEbtn5UUF6iCUXi2SAyZIVwgC7r/hTT8+x9FnYM8PBzzG73BXhEevfi8LJ04tj7jM90iM9DhAzvv766/B4PJBlOQZAGGO6CysNIIMhkQE01X8RCUzHeK9EqGE3sktOh8k+jNEhqf3oKkgZGJraKqGq4QFxz3CBH1Q1TMuxIdcisYHg+eKZiiWZZjazzAlPUIXYy+mTBIYOXwhzJxZgXGkmox5aJKdHenzXwEMQBHR0dNDjjz8Ok8nUhaWaiFBcXJwGkIGXxQTGBPh9ddTWsgmiZMWB9F0G0sKQzdkoqLgUwKHPn44E7wU0t++jfXUbIYqmLmmqvb1HIRrnCKoaJmfbMSbDwgaSJJKDyElHZLMJRXZ0+lXd/dXT30kCQ4c3jHHDM/GDeSMY0eHRZjg90uNgDFVVI+0sBAG//NUvUVVVBbPZHFP/FAqFkJmZiXnz5kX2uzA0RfVhmoWlARBRu38JwuEOWEwZUSr3SF2IGmrH8CNuhGzJY1x4DyY4dGchRQBMQFPrbvq68nMwJkdJKwmKEoIgWqCRBkUJQRQtEITkwWPGJTQizZ1kUcSs/EyMcNnYoIhoFgmwX3BkAXOYRFq1tw3QCCYWiW/owRUGkMAAYgioGvyBMKaPysbl8yuYWRaRrLFUKnGggfpMfz7/bR7pRIWDt14YYxBFUU9Z/+1vf0vPP/c8nE5njPUhiiK8Xi8uuugiDBs2jKmq2iXNfcisn8NvM0XuNxz20splP4USaoMsiBBIg8gYEHYjI2cKxs56mEXoT4RBWlSpA1N90zbauW85mGgBBBmKqkGUrCgvmoRsZyHCahgNHQ2o62iEPxwCMQkQZTAmgZgMYgJUEqFChAYRkiSjyOHC+JwsOGSJDaZ+bzz3Nw1eWrajBVVNPgRCCjQtwk+mKpGqdgZCvsuCeRPyMGd8fqQ7cRQ8NE3Tzfahqk2lR3ocDOvjiy++oD/96U/44IMP4HA4EhYRqqqKlStXYurUqUMaQA47C4TXctRUvw+frxZWUwaIFIBFXFeSbEf5pP+N1HuQNlg3EY27aPB07iWPuwqhoDtiF0l2WKw5MFuyAQCNLTtR37wdgmgFYwJCSghmcwYmVRwPhzVTl/uZ9myMyK9AQ2cTNXna4Q4FEFIjvRfBRJglE2xmK3JtThQ5HHCaIr3dB9s5xA48MsYU2NmYAjtq2wK0t8mLxo4gfEEFIgOy7DLK8+yoKHQySWT63xBpkd4iCYCDiKCqKjRNg6qqUFWVol97PJToVy36vRb3e35O41fjoWoaKO5nRHTga5Rag/d0iD/4/VOUxlj/mWGNdOvnA+mJHboVwJiB2PIABQf/PtG/+3UIAgQ2QOdKQifCnykZ7QgMz9vl+1TVScP7iP+ayhH//hMdqqZGqX9i1uuB9ago0XWpQFUi/+ZHOByGoihobm7G5s2bsXnzZqiq2sXy4NaH2+3GaaedhqlTpzJN04YseByGABIR3OGwm/ZV/heSZIsASnTlqaoPo6fcBYt9+CC6riIqdXvTRqrf/zG8vsaIoIQMYiJUJgJMAhNMkX9DhChawRhDWA3DZsnE5NEnwWJyRO+R6XTqJsmE4dklbHh2CTQihFUFWjRWIotiTJ8WihPwg+/qOGBNFGdZWHGWJbmDMVpkYqww37BhA73++utYvXo1WltbI/MR3Vj8SAQUyQS+UdgnEu7pMbRcYwldZaxnqEjI2mvsPdANoCRyQR3q9cEYg81mgyAICVs78+c988wz9TU+lC32wwpAuPVRufdN+Hx1euyDMRHhYDNGjrsGOUUnMCJ1UCrOeeC7Zvd/qHH/p4BoiYAYk0CQQEyEEAUQjQSAiWCCBA2AqmmQZRsmj+LgQTrAMQMS8E6FAmMwS3LizcEOTVIyYweU65iiTTJq46T7egFg7dq1dO+99+Ldd99FMBjsIhS602ARp7nyr3xDGfsk9JU367sSAxhowdnf8w22IO/te0318/1dL7pClIRojs/LsGHDDot1JR0+GyBiUfh89bS38i1IsiP6MwmhYDPKKi7G8FGXDiJ4RK5fs+tVatz/MSRzDjRE21ZCix4syvWrIdLI6kDbS1ULY2TRVFjMzm6to+4WqDGIfmi1qAhocN9tIvfU9u3b6cGHHsJLL76IQCAAu92uZ5ok6//cW8EzlDTL9Ph2A+bBttzWr1+Pc845Z+jf7+Ey0Vzobtj0Z6qtXgKL2QWBVKjBNpSNPB+jJ1yvu4QG2rHDr93euJr2bX0SojkHGgCCGLE+mKhbIFrUAon8ToQW/b2qEaYccTZcjgJG9O3TfD0eD+3btw+bN2/G4sWL8dZbb6G9vR12ux2iKHbh+kmP9DiYFsbhAkS8k6HL5cK6deswfPhwNpTdWIeFBcIFeGvbNtpfsxQWUyY0LQg17MGosVdg5JjLBw08eNxFUwNo2PsWBMmC3oauGRg0UuELdMLlKNQtlMNZs2ttbaV//etf+PTTT7Ft2zZUV1fD4/HonxNFEXa7XQ+UC4KQEr8VdeO//rZomYMlIAfLbTNYwjdVSzTRzxJlLg1IR8OY1pxI6kpNNH+9mU9jDC/+5yaTCc3NzfjRj36EDz/8EJIkDd1+IIfTpqqseg8AIr0/RBMmTLsFxSUnDyJ4HIh7dDZvpKCvDqIpM1L70Us3GQPQ1rkfhbljcDgX1fFFbLPZ2Ny5c6mwsBD79u1DTU0NGhoa0NTUhJaWFrS2tsLj8cDv9yMUCiEcDg/IdZPGSgwbP5UCzYHcjMkEVyrCsDs3XHxWUU+fP9wBs6csM+Pn4hWSnmJoPb2/iCsaQFwGnp6JRQQtmtQxUPMuyzKsVqteWMgHz9D6/PPP8eMrfkz/fOmfzKiEpQGk14srYr61d+yGoniRlzcTEyZeB6dz5KDFPLq4aFq/OuD87yUAEAiCIKHT2wRFDUMS5cN+w1utVkydOpVNnTo14Yb0er3k8XjgdrvhdrvR2dmpf+/xeODxeOD1euHz+eD3++H3+xEIBBAIBBAMBhEMBhEOh3Xwic/WMqbuJkq77C5tM6lwjgqQVACnuyC/UYglEoJG4RefTst/Z/waf/DPi6IY83Pjv/n3oihCEEWI0e8T/btPhyRB6uEzgihAFOJ+Fn9vcfct8J8ZnzX6vPrzxwFIl/lMACSJ3mGidF9jhl98mm44HEYoHEYoGIz5eSgU0tdsMBhEKBSCoii6laRqGkRBgMlkgqZpaG9vxzfffIPly5dj3759MJvNXahMFEVBRkYGXv7ny6gYWUF33XXXkKwHOUxiIBGh3dSykVTFj8KCo5nRtXUwrr973d3k8+wHk2wAIrGOlGMgEIHo99PHnQObJYORoQ7gcB1GQc036GBqSfF5+qqqUjLwSJTqa9QeD2idsXUcqVpB3YEFB4n4OUl0xPzuAJCwpJ9J09x/a0ZraystXLgQf/7zn9Hc3AyXyxUTK+Tv2uv1YtmyZZgzZ86QAxF2uJrBB9MnSFoIO9fcTqFgGyBYomDQewBRIWDqEWfBacv5VgBIKi6d7jT+7rT8ZFpkevTsKuttltvBkAFDLT4zkK7KvsyhMdV99+7ddO211+LDDz9MSmsyZ84cfPbZZ4xzaKVdWH16gZru0jqoC4yJYIIIQv/6WTDGInUih3ADqCk2Ou9tQDBZTUZvfMbd+fcTaV38eVJtRJWsQrmn8ydKAEjkDks2RFFMeE8xvu9eVmAb+w3Hn9/Yc76/4GAUdH1dS93NUTJqG25NDhVA68v5uLBPJvD5HFZUVLDFixfjmmuuoaeeeioGRFRVhd1ux7Jly/DKK6/QxRdfPKSsEJbOn0/NfVa5+SFyt26DIDtBEHpngTARRAJEkx0zxi9gkmj6zmjGh0KTNLqyYsA7zcOVXieHYPSUhmusp7rs8svpxRdeiAERQRAQDAYxYsQIrF+/HlarlQ0VyzwNIClYPYwJaKp6j+p2/xuiKQsE1isAAZMQ1lTkZI7EhFEns4O9YYwxgD//+c+0d+9eCILQReNP9d+MMb0ZTjgcxpgxY3DTTTcxs9ms/54/48KFC2nlypV6t7WefPvxv1dVFYWFhbj++uuZLMsx2mwgEMDDDz9Ma9euhcfrhdfj0YPvqiFjhp9LFEVIkgSz2Qyr1QqHw4GMjAxcdNFFOOOMM/T3wjd8bW0t3XvvvdizZw88Hg9CoRA4N5Esy7BYLLBarbDZbLBarbBYLJAkKVI4qqpob29Hbm4ubr/9drhcLmacO0EQsOi99+j1//wHJnMkgEoAKCpMNC3SHzJCCwNopEVjNdqBbCGNcPTRR+OqK6+EzWbT759/ffbZZ2n58uWwWq0xGn93X/nci6KIcDiMyVMm47JLL+ty7ubmZvrjH/8In8/XZS2lqq1rmoZf/OIXmDhxol7rwL+uXLWSnnv2OTBRiElsSIWuhs9ddyohdWvdRAuESYs19eI+E3knCZJqWOT9mExmXHvtNZg+bTpLFUS8Xi9NmjQJdXV1etAdiLAudHZ24k/3/gm/vfm3Q8cKSZVw7Lt7RIRQyN9MW5ZdT18tu56+Wn4jbfrif2njF7+lDV/eRutX3E7rVtxNa1feQ6tX3UerVz9Iq1b/hVau+St9ufbv9OW6p+nT1X+n5rZK6o7DabAORVFARPjVr35l7Fw7oMf7779P/Fr8ei+88MKAnX/p0qX6+cPhMIgIv//977t8jjFGgiCQIAgkiqJ+8J8xxrr8jSzLtHPnTuIuBX6cdNJJA3Lvv/jFL/R7526lXbt3kdlqGZDzn37G6RQKhfT75m1SZVkekPP//e9/7zL3jz322ICce8SIEdTS0kJGig8iwrzj5w3aWj2Yh93hoJUrV5Kmafq+SHbwuX3ppZcIADmdTrLZbGSz2chut5PFYqHc3Fyqqakh41wdykNK2xg9O5uJNMiWHJZXejrV7v43ZEte92yrMdqWgJASQE7mCORklkaC5wfR+uCayqpVq+jRRx+Fy+UaUOWDa0bc3OYEiqqq4qGHHoIoiglZR1MdnJ3U7XbHnB8AlixZAkmS4HA4eu3v5+9AFEV0dHSgtrYWo0aNgqqqkGUZ27dvp88//zymJ3W8S6wnH7ggCHC73Vi3bp3+b65RdnZ0QtM0uDIzUrvvKOMmJbjm+x98gB07dtDEiRNZOByGIAioqqqCpmnIysrqcyxBkiS0tbVhyZIluOaaa2J+5/P5IElSl8yh3qwdWZaxd+9evPHGG7jqqqugKIqupTscjsi7dfV97RxqxVySJLS3tuHG//0Nln32eY/7XpIkqKqKH/7wh2zhwoX02WefxaxtXmD4xz/+EX/7298GNUaU8hpJA0RqIECkIb/0NBbw1lBrwxqI5mwQsR7+TkRYDcJidmF0+XFA/zuM9zm4d++99+oLbiBpRZSwApvNhvHjx8dc8+uvv6Zt27bBZDLpJIp9AsCoUCkrK4txsYTDYXR2duoWVl82E6eNcDqdGD58eMzvvvnmG4RCIZjN5j4LMA4YPp8vxn0HAMXFxcjIyEBnZ6deadzX9ysIAtwed8zPTSZT5PnC4YgbrI9CkN+/0TUJAGXl5TG1OX29d8YYVqxYgauuuiomXjB16jQseneRXmtxOA5VVeFwOfHlF1/go48/opNPOjkl1xNjDPfffz/mzJkTs64VJbLXnn32WVxzzTU0efLkQ+7KSkcUeymIhx/xY5ZdNBfhUCeIlGhGmNGXL+i1KaGwF1ZLJiaOOQtmk5MRDm5rV74Zv/76a1q8eDFsNtuAbkZRFBEIBjBt2jSMGDEiJp62evVqhEKhmOysvsx5MBTC8OHDMXr0aGZ8D4FAgLxeb6/9713OHwyirKwMw4cPjzl/U1NTUsuit9fgsZN4cBmogD5jDCbZFHO/OTk5sFgsUPuhpXItuqqqCqFQKOaeTzj+eAwbNgyBQKDPc8Rdubt27dLXEz/X3LlzI50uvwUxWiLgL488ktJ64tb7zJkz2eWXXw6v1xuzh0RRRCAQwG233TYkni0NIL1wZXGronTspaxs7I8gmzIQDnuhhH1QlAAUJQhF8SMc9gKMobhwOqaMu4DZrNmRIORBrvvgQuvZZ5+F3+/vlzBPJriICGeffbaucfGxdu3a/i/OaBD9qKOO0ikf+Kirq0NLS0u/tXdN0zBq1CjdfcBHc3PzgM1RopTUsBIeEDAXBAGqoqC5JfZ+8/PzkZWVpac59xVATCYT9u/fj+rqajKuq7y8PDZ9+nQdWPpz/t27d6O9vZ2MFtpRM2eioLAQwWDwsM7QUlUVVpsVH334ETZt2kTJ+oAk2ld33HEHcnJyEAqFYlKzHQ4H3n33XSxZsoQ44KQB5DACEYCQU3g0O2L6jWzkET9CftHRyMwei4ysUcjLm4IRI07D5EmXYWT5iUyKki/2dhMYg2593ZxR/z698sorXagSBmKEw2E4HA6ce+65+sIXRRGhUAifffZZjM+/P1Yf743AXSoA8NWWLV20s74Oi6Vrc6zOzs4B0j4pYW0KQ2ynvn7NEQGbNm2K+XlGRgYrLi6OET59tUDcbjd2Rq0EYxJIeXl5vwo9OYDU1dXhq6++ijl/dnY2mzVrFsL9AKihMiRJQigYxJNPPdUr12dJSQn7zW9+g0AgEOOm4m7L2267DYqixLgW0wBymAAJkQZRtCA7byorrTiHjR73IzZ23CWsYtTZrLDwSGYxZxpcOr0ryuPFazwltK+aD2MML774Iqqrq/U02oF0XwWDQRx77LGoqKhgXLAzxvDxxx/Ttm3bYLPZBgRAEgX+t23dOmAuJq/X2+VcA0nJIklSFyFot9th7ef88PUCBmzZcmA+uKJQUVGhP0t/38HGDRsOXC/68/Hjx/d7TXFh+fnnn3dREr53xhnRBqCHd42IqqqQzSa8/sbraGlpIVEUe5w3Pi/XX389xowZA7/fr79HXly4Zs0avPDiC9RfRS0NIIcCQpgAgKf5RrqDxP+7twufC31RFLFr1y76/e9/T1dddRX1RaiIooja2lp64IEHYLPZ9PMO1MEX7ZVXXhnjLgOAf//73zGEePFHKueXJEk/55lnntlFENbW1fX7GWRZBhHpFohxU5eXl+saeG/PK0mSfjDGdAAx1lK4XC42fPhwhENhmEymfj0HCOjs7IgFFQAnnniivhb6czDG8NFHH+nvgL+H+fPnw2QyQVGUPs0TP7fZbMYTTzwBj8dDkiSBCZF9873vnQlXZka/zn8wD6mbveJwONDU0Ih33n23i7u3OzeW3W5nd9xxB8LhcBeGB5PJhD/e/Ud0dnbSIbNC0nUeh/4wuqv2799P1113HWVlZREAuvrqq8mYI96buo9PPvlkUHPczz//fOJFe8brXnjhhQSAJEnq9zWuvfZavT7DOA+33377gDyDIAi0ePFiMmZzaZqGuro6Ki8vH5Br/P73vyfj/PCv//rXvwbsXbz0z5e6PIPH46Gp06YNyPlvvvnmmPfAn+F3v/vdgJz/iiuuSLiWfjmItUuH4vhbtKYm1f3Ma3vmzJlDgiDE1Ia4XC4CQHfeeWfM+jqYR7oS/RAPo0vgxRdfpJtvvhl1dXV6lfOGDRtQVlbW665k/LyffvopNTU1xQSbjedJRHnd07/5OebPn8+4Fm+sUm5sbKSdO3fGBJC7O+JZdjmgulwunHHGGSxeK2OMoa2tjbhW3FfXjKIoGD58OObMmRPDDsC/r6mpoc8//zzm572Zf03TYLPZcM455zDutoi/xsqVK2nX7l0wyaYI7T8TAAb9K2NCJFYi8K+R6AkTmD6/ebl5mD17dsJnqK+vpzVr1ugWXSpCwWhRhsNhlJeXJ5wjPo9Lly4lr9erv59UD+AAbfn3vvc9Zrxv/vtQKIT333+fQuGQbvX3ZfRUnd5vl7amxVXpxMa7NE2D0+nEKaecojM2pOqVEEURn376Kc2fPz/GLczdlTabDZs2bUJJScnBJ1tMWwCH7jBqW9dddx0BIJPJRDk5OV0qmNPzNbgWYCo/OxjXHarPkOxcB5tV4bt48P1/3oLzCAC5XK4uVgj3VBxsWZG2QA7R4BaFx+Ohiy++GO+++y5cLpfupnE4HNi0aROKiop6rVUY00YHM8UvWaDZGAg1Zo1omgZBFBOmFHBAFUUx2gFOA8MBJth4KyieDTa+J0l3Mab4zye7RqLmVMnYabs8NzuQZWX8vM7gG+1Dwq/BYws8gcIYM4k/v/HnRu6qRA2TjIqK8frxnzfygBkZjvl7TsYobDy/0bKIr3MxWhaJ3kNKc2qwnjknWU/rv9csx8kU7cjN9DthQ5ZlfZ5TLQDka2Lr1q101FFHJWRBUBQFq1atwpQpUw5qcWEaQA4heDQ3N9M555yDL7/8EhkZGQiHwzo1yN13343bbrut14vh285s+l2eg8PtudJrceAGlwPXXHMNPfnkkzEUMpzu56yzzsLbb7+dBpDvAni0trbSqaeeirVr1+rgIQgCQqEQiouLsWnTJjgcjl7RNvNzr1ixgl544QXdx2yMM8THNoysuElboibIpNK1QRZV8+K6AHKKi3A4jIaGBtTU1OCUU07BLbfcwozaLF9/fr+f7rzrLnz88UcoHzECxUXFACLaaigURijEW4aGEA6HdFOdMQEmkwyHwwGX04WzzzkbZ5x+RsJ4wN///nf6z3/+g6KiImRnZ+stc42tSOPbkQqiCIvZDJfLhZycHJxwwgk477zzEp7f6/PRiy+8gPUbNqCtrRVerw9hJdILXhQiWV8mswlmsxkWswUmswmyJIExATW1NaitqcWso47C1VdfjfHjx8fEvfg13nvvPXr//ffR0dEBt9sNv9+fkIGYvydJkmAymWC1WuF0OpGVlYUpU6bghz/8IRwORxeW3X379tEzzzyDqupqdLo74fV4EQgEEAoFETaQQRqtN5Msw2yxwGG3IzMzE8OHDcfFP7wYEydMZPExjc7OTvrTn/6EDRs2wGKxID8/H06nE7Is65lWRrZevpbC4TA8Hg8qKytBRPjhD3+IH//4xwnn6JNPPqHly5dD0zR9fvg5jC2Q49T4mLa5ybIF9Z8b2u/G7FEGvZ86D4DzlsydnZ345ptvoGkaLrroIlx33XWMs1r3tMf556qrq2nq1Knw+/0wpgMLggCv14vFixfj5JNPPnggkvYxHlxfu6qq8Pv9mDcvwjaakZHRxZ/59NNP99qfyTfGtm3b9PMMpYOz4P7jmWdislD4Jjvv+wsinxVYv64jyhJt37FDzxjic/jqq68O2LN8/PHHXTKe/H4/5veXvTf67MfMnUNGZl3+DIsXLx6wZzjrrLOIswMbmXDnzJ3b/d+yuCPJ55wuF23ctElnjY1n8U3Eitzb49FHH415D0SEhoYGPYNxqB+//e1ve7XP4zPfjLEQp9NJgiDQUbOOIiPrc5qN91tmhkqShGuuuYY+++wz3fIwahDTp0/H5Zdf3uusK64R/vKXv0RnZ6dOYzFUXBncNbdxwwbgxz+OMctXrV5Fb77+BhwZTmhqco2sJ2tZkiR0tLVjx/btGDtmTMznX3vtNQiCgIyMDL16N9Xz8iHLMtrb27Fz5069xoL7sjdt2kQff/QRnBmulHpWJIpXcE1z27ZtaGhspKLCQmY815IlS8AYQ3Z2NkKhUMpzE18gqaoqPv74Y9TU1FBpaamurXq9XqrcVwmz1QKTLMeQMPbmeWRZRntrGx559BE8s/AfMetQkiP1MU6ns8t7SHUtCYIAn8+Hvz72GK6++mq9b4YoimhubobP50NGRsaQ5NHilgQQITg97bTTaN68eSlZDDwudsMNN+C5556LofLhFCerV63GP//5T7rssssOihWSBpCDDB5PP/00Pf/883C5XDp48A2oqiruuusuyLKsB8564x99//33ifdV7g8D7qD4SqPPVxKlJTEKpeXLv9ApOfpTUatpGkRJRFFRUcymA4CGhgZomoZwONznayTjtQKA9vb2mCB0f4bP50NLczOKCgtjhKDf79eTLPp6jQN/RzpFvkHwM5vNRqFQSE9m6HPsQ2DYsH6Drjxwf31ZaZnu3uzPM5hMJlTu3YutW7fS9OnT9RTgkSNHslGjRtH27dthtVqHBOV5osHdYb/73e9gTBXvaf0pioKcnBx2ww030G9+8xuYzWZ9bjVNgyzLuPvuu7FgwQIyNhkbrJGuRD+IcY/Kykq66aabuhAD8iDYaaedhu9973u91hz4AvnzA38eskFLvpCPnD69i8a6ddvWAendHQqFUFxSgiOOOEIHD36dgQTURIJPGYD+4/yeQ8EQ2jvau2j+AyUMI3T4SowCAwBmsxkOh6OHbn0pvm8AwVAwpmYBiPCa9Yci37hnwuGwzgHGtXCLxYKTTjqpVwrYoVIo7XY7vvzyS7z77rspkSzy59Y0DT/96U9RUVERQ3GiaRqsVit27dqFxx9/HAeD4iQNIAdReN56661ob2/XKTSMgsFkMuH//u//+rQQBUHAl19+SZ9/9jnsdvuQa8DDGEMgEEBxcTFmzJihC0oOkpV79/bKlZRM8KphBZMmToLT6WTxG6cv7pLevuOBmitQV+sAiPT4GJBNLwgIh8Ooq6vrAk7FxcX95p/iJIk1NTWoqa2JmZiSkhLk5OQM2PtIxPo8Z86cAX0ngyqABQEPPfRQyhlr3Mp1OBzs5t/+Vk++McoDi8WChx56CPX19YPOk5UGkIOgaYiiiGXLltGrr74Kh8MRQ+MtSRK8Xi8uvexSTJ8+vc9+y8cff3zIal1cYB1zzDHIzMxkHOC41bC/pgZMFAZkw7syXF2Eh9frpfb2dqRCYpeq9tv1GQcWnHw+v67J82HsjthvkAJitHcuZMaPG9fFQuwLgMiyjM72DqxauUr/GREhKyuLlZWVdeF26otVL4oili5dimAwGEM8euSRR3bZZ0NVNthsNnzxxRf48ssve22FXHbppWzSpEl6X3ojeDc2NuLee+8ddKbeNIAcJBfWb3/72y5aBhegWVlZuP0Pt/c6b55vot27d9Pbb78Nq9U6JDcNf6aTTz45RpgAQH19PdXV1UE2mQbEjaUY3DL8fHV1dWhsauxi+fV1yLLc9WeSzC86IHMW4s9hOF9ubu6AaO18DjZGAcR4zsmTpwzoO//iyy+6uP6mTJmiKxb92VNWqxU7duzAunXryPhs5eXlbOwRRyAYDA55KngOGgsXLuzV3BIRzGYzfnfrrV2sOd65cOHChdixY8egWiFpADkI1seLL75IX375pd7f2KhJBAIB/PrXv8bw4cP7zHf1yCOPwOPxJBRsQ8WFJ4oipkfjH8YugmvXroXP44Xcj8ZQfC4jQdSKLiBVXV0Nv6//DbV43UOi/iEmsxmCKAADQC8vRHmv4kdxcTEGgutI5/rav19/H1wAjR41CkwUBuSdC4KAb77Z2QUs5syZMyDPwdeRkQqeW+GnnnoqNE0b8CZqgyEjzGYzlixZgqamJkrVSuZWyAXnn8+OOuoo+Hy+GMuYezbuuOOOQbVC0gAyiEKTMYaOjg6644479FRD4+L3+/2oqKjAr371K/QWPLj1UVtbSwsXLoTVao0pIEt2GAsIB/MwLnS3242SkhJUVFR0IcvbvWe3ntqYiA69u0Mv7hIFeDweMIHhoosu6iKweDaOEu2v3ptr8OvwIk9VVZGfn68Le/6sebm50FSti0/a+LlU6Ox5lldhYUEMoADQOzO2t7d3OY+UoOAtEdUMn2eeHWVcq0SEsrIyZGRkwNPpjrnfXr+TqNZrtphjlAgiwumnn46ioiJwt2Jf3ge/BhFhx44dXUDlyiuugN1uh8fjSUgeerCP7vajJEmora3Fnj179L3dG8Xs97//fRfXl6IosNvt+M9//oMVK1YMXufCdIHf4BKg3XrrrV2KfnjhDwB68cUX+0SCZqTsPuWUUyJFdKJIjDH9EAShx0MUxZhDkqQeD1mWYw6TydTlMJvNZDabyWQyUUVFBX366acxdOD8/tva2uiM751Bgij0vUhRFKiouIieefYZiudm0jQNoVAIV199NVmt1j4XsAmCQNnZ2fSLX/yC/H5/DE8WL/q76+67yWK1EhMFkkwyyWYTyWYTSSaZBEnstvAOhkK9c849N+E1iAivv/46VVRUpESVLwgCSZJEZrOZLBYLWSwWMplMJIoiZWZm0ptvvhmz9vg1ln76KU2ZMoUkue90/IIk0pgxY2jV6lV6MaHxGh9++CFVVFSQKIp9LxoVRcrNzaVXX3014XO89NJLlJWVpa9Fvh7jj/j1zI9U9kL8/uFHKnuPH2azmc4//3xyu92UiH8tlQLiefPmkcBYDN270+kkxhjNnz+fEnGWpckUh3DMgzGG3bt30/Tp03XN10gc5/V6cfTRR+Pzzz9nffUHc81RVVWsW7eO4tuXdtE+41wjCVutGn6WrB1rMg070d8QEUaNGgW73c4SkQPyf2/evJlq6+rQ0d4Oj8cDf8AfpRVRow26AEEQdWoOi8UCu80Gp9OJ7OxsjBs/HpkZGd3mve/du5f2VVUduEYcDYiRLFGWZZjNZthsNrhcLmRlZaGiogKFhYXd+qh27d5FjY1NsNttEARRd1OEwyGEgiEEggEEA0EEQ8FoTUrkmiaTCXabDQUFBZgyZQqLnx/jv/1+P7Zs2UItLS3wer16lbd+3xYLLBYLLGYzzGYzZFnW3RuKosDr9aKsvAwlxSVJ34mqqtj81WZqamqG290Jr9cLvz9goHo5kAghiRJkkwyL2QK73Q6ny4XcnBxMmjSJ8U6Yia7h8/noq6++QnNzMzo7O+HxeCK0KeEw1Gh1Odfeje/dFn3vWVlZGDt2LHJzc1miPSgIAhoaGqgu2nwsnhQzXolOpFgn+ll//tawaXT6n/z8fIwZM6ZPvk/uJv/444/p5JNPht1uj7FguAfgrbfewtlnnz3gxYVpABnE2McPfvADeu2112KIz4zuq6VLl+LYY489qORnhxJUE4HkQBY6dTePfans7+01BvI9JpuXgbxGsjkZqLnq7lyH23McDJd3IqWvN89/2mmn0ZIlS2JirVzWTJ48GatWrdJ56AZqz6UBZJDAY+nSpXTSSSd16QvOKT0uvvhivPzyy30Gj3gq85T8pqxvxNY90WZzX65Ry+Ian/H33W0eY4V3dzTxRrpzliDYHE9bbtyU3DUQ//uegtrxvvPuNno83XoyQdHd+0l1vpKdryfhwGNlyehUEv2bP1d3ayF+zvX5MlCqx88jxdGlGJuKJaOQNxYn9iQMU6GV4dT6YBHnWDydfjwV/UDsqXhFIZ76PtmzdydzvvjiC5o3b16XCnwucxYuXIgrr7yS8RbBaQAZgloEP+bMmUNr1qyJKezjRUCyLGP9hg0YOWLEwe8gNkDa8FCg6ubz+m233ob6WujP++M+fP4eh0om4eFGRc9B5LzzzqP//ve/MV4PngBSMmwYNm3cCLvdzgbKCklzYQ2wm0YURTz77LO0atWqLq4rHvu48cYbUTFyZJ+sD67RP/zww7Ry1Uo9u0szaNZEBAZAi/seOmEf6bpXl+8Z9M8pioKrrrwKCxYsSEib3djUSA8/9DDWrVsHn88HWZZht9shyzKqq6tht9uRk5PTpXVtPB24JEm6r764uBi///3vY2Im/PNut5v++Mc/Yu3atejo6EAwGNSb9FitVtjtdtgdDjjsdlitVoiiCEVR4Ha7EQqFDtDPx+tMEXc0yEB0l0h7Nf47UftX4/epfC7eaov/jP4z0nrsxtpdK+JkWUHJMoSICIFAIErjHpk3u92OH/zgB10o1Pn3a9eto0f+8hcoaqQVsSybwFhkDQWDIfgDfvh8Pvi8Xvh8fvgDB+JP/JllSYLNZofL5cKso47CLbfcApfLxYzKlyAIWLZsGf31r3+FPxAAE5huPQhM4EaB3v6WMW5p8e+j8xO9f709MBiINLicLtx3333IzMzsQnXf1NRE99xzD9weD8JKGOFQGIoShqKoUDUVmhpphMb33AELx5CJxxgYEyAIDEwQIBi+F6OZa7m5efjTPffA4XCw3rq1brvtNrz33nsxlpKmabBYLNi7Zw8ee+wx3HLLLQPnQkxnTA0sVXt7ezuVl5eTyWQih8OhZ0Q4HA4ymUxUXl5O7e3tZOzd0NvMrscef/ygUU4LkkhfffUVGTM+NE2D2+2mI488clCu+eSTT8ZQvvPnvuqqq7pQxA8ELXj6SP34xz/+0YXK3u120/Cy0t6di0Xo65kokCCJxESBmCjE0MRf+ZOrYrKr+H6ZNn36oD7jsccdS+0dHfoe5evw5X/966DN82WXX97r7Ez+2YsuuqhL5qfdbieLxUJ5eXlUV1cXkxmXpnMfQtbHww8/jMrKyoSB81AohDvvvBMZGRmst7QjxqySO+64HRarBXJcbcmgmMaKGnMNziq8aNEirFu3DpmZmYbmTqxfLgBZluF2u1FdXd3FH9za2kpvvfUWLBYLZFlO+NzJMtAOdzdtKvc/WO4Wfm1JkuB2u/Hkk0/iiiuu0NcuYxHtWRJFmCyRrK9E90u9oIbne+WDDz6A1+slbo3yPTZr1ix8tXkz7M6BpyuRZRnLPl+G//73TVx+2eUxVd7cqnW4nFAVNRI7GeDBLZUXnn8eP/7x/9Dx847vlaciWjqA//73vzFzwylOmpqacN999+Hhhx8eENmRLiQcIPAQBAFVVVX0yCOPJGTb9Xg8mDNnDn70ox/1yXXFBfKDDz+E5qZmSLKsp58OxkFE8Ho8mD59GiZPntwlVrN+/Xo93ZNrasa/j/93Kgen+S4rL+vi0tlbWYm2tjYIghCj/RqPROfitOGH85HoWbt79sG4Ni+QrKurQ2dnJ3Ghqqoq7DYbO3LGDIQCwZhulMneQU/PwtPem5qa9OI6o6vvuOOOHbT3yp/z/Q8+6OIKnDplCqx2G/x+P1RtcOZbMaQv/zFKrpqqcsALUSdOnMguvvhi+Hy+mGC5qqqwWq1YuHAhdu7cOSAUJ2kAGcCA2913352QbZf//t577+1TwJwDVE1tDS18eiFMFvOgM+4KggAQMH/+/IQB69WrV6eU4dIXK27SxEldNk5Lc7MuWNLj0KxxSZLQ0tKC+vr6LkJ94oSJA2oJSZKEUCCIr776KmYPAMARY4+AJEuDsgc0TQMTBSxfvhydnZ3Es6E0TUN5eTk78sgjEQqGBnUdqqoKi92GT5cuxYoVK1ImWeTzT0S45ZZb4HA4Yij7Ocmlx+PBXXfdNSAUJ+ndOAAvm3eke/HFF2G327uw7Xo8Hlx88cWYO3du3wLnUQB67LHH0NbaCtMAEA+mtJEEpgMI30SMMezZs4fWrFkDi8UyYJuYMYZgMIiioiKMizLCGosvW1pa+kypMZB574frSESr0R2lSjwliiAIkGUZgUAgBkD4KB0+fFDue/2GDV3cdBUVFSgoLER84exAAaXFYkHN/v1YsWKFvhe4pn7O2WcDByFDSxQEqIqKv/39771W/DRNw+jRo9lll10Gvz+WA45TnLz66qtYu3ZtvylO0gAyABsTAO644w4Eg8EYcIg07gkjMzMTd999d5/iAkQEKdKqk5555hnIZtOgWx+MMQQDAZSWlWHGkTNiFiYALFq0CF6vt9cpl0YBFg8GPJts8uTJcLlcjIMVn6/XX3+dJynA7Xb3eHg8Hng8Hni9XgSDwZiUxniOq28DuCSbV6MGHQ6HEQwG4fP54PV69Tnq7vB6vfB6vfD5fAgEAtA0LQZA+NwNLy2NuX5/C9Y4OHHKef6eNE2Dy+ViRxxxBNTw4FikvDsmd2MZ3bfnnHMO7E5Hl2Zcg6GYmixmvP32W6iqqqLedIjklsVNN92EzMzMLtT5vL3C7bff3n9LMQ0B/bc+vvjiC3rnnXe69CDgabu33HILysvL+2R98KD1c88/h8aGRjhczkGnbBdFEaqiYt68ebDb7fp98030gWFj9aTtGov4uI+5OwC85JJLDlhA0TRfzmc1Y8YMjB8/Hrm5uXraL6eEUBQFoVAIfr8fXq8XnZ2d6OjoQHt7O9rb29HR0QG32w2/39/lmrIsQ5blmILI+MK4oWhNGFNpFUVJ2q5XkiRYrVZkZmbCZrPph8VigclsgizJMeuSg00oFEIwGNTTeTXS4PP6YLVauyhQ4444AqIc6UmvCypJ7NYSjE91jk9nFiURO77ZAY/XSw5DIF0QBMyeNRsff/jRoIB/xPoW8PEnH4MX3fFrjxwxks2dO5eWfLAYNsfgNW/jQe/Ojk788+V/4pbf3pJyZT13eZWVlbErr7ySHnzwwZikHt4//f3338eHH35IJ598cp8LmtOFhP1caIIg4JxzzqG3334bTqczhkIgGAyirLwcG9avh81m61PxDk+fPXLmDNry1VewHIQ+z5IkwdPpxrvvvqu32OUCwOPx0KRJk1BbW6t3yDNWFXNhFgqFupzX6XQiNzcXxcXFGD58OEpLS1FaWgqn04m2tjaUlJTg/PPPH/A+zn6/Hx0dHdTS0oK6ujpUVVVhz5492LVrF/bu3Yvq6mo0NzfHaJWiKMJkMkESJYAhJuX6UAIGT2+Ob9HrcrlQWFiI4cOHo7y8HGXl5Rg+fDiKi4qQl5eHrKwsOJ1O2Gw2Zjabe625K4oCVVMBinRGTPR+PvjgA9q5cyd27tqFzV9tRnVVFdrb2+HxehEKdNNSWGC6xcTdZXw9KYqCFV98iWnTpjFjNfzST5fS/BPnw+qwQxsEIR7pLaNgzZo1mDJ5MuNuLEmS8MKLL9Dll10+6MqcIAjw+3yYedRRWPnlil7VhHAFrL6+niZPngyPxxPTUI0rt7NmzcLy5cv7zMeXBpB+WB+CIGDd+vU055hjugTOOYnZK6+8gh/84Af9KhrcsWMHTZ0+rXf0A9Hiqt6+X1EU0dnegXHjx2HD+g3M2EaVMQa3200jK0aiuak5Yc9lSZKQnZ2N4uJijBgxAqNHj8aYMWNQUVGB0tJSFBQUwG639wkdUrUMjAV0qWyK1tZW2rdvH7Zv347Nmzdj8+bN2L59O2pqamIEtclk0oVcfzv2JfqaCDS48A4EAjFgUVFRgUmTJmHatGmYOHEiKioqUFRUxBL1KunuPrqby55oW3oaPp+P2tvb0draiubmZjQ1NaGxqQmNjY1obGxAY2Mjmpqa0NzSgtbWVnR0dCDg88dYMJqi4pVXX8EPLvwBUxRF30PBYBCTpkymvXv2wp6iJcBdU6lwj8iyjLaWVvz+D3/AXXfeGXNtt8dDEyZOQH1dna7Q9VfhYRG+l6RAsG3LVpSWlvaqZxCXObfffjvdddddCQub+yuj0gDST/fVxRdfTK+88krMy+Ev5sQTT8THH33MVK1v5iFfLHsr99K48eMR9AcOyrPZ7Hb8979v4uSTYk1bfj/PPvssPfroo9Gq2VyUl5ejoqICo0ePxsiRIzF8+HDk5eWx7vigEgEBd1kNhjsgkauku+t5vV7avXs3Nm7ciNVrVmP9uvXYvXs3WltbB1TrjK8Kj+cFAwCbzYYJEybgmGOOwbHHHovp06ejvLycJeOJin/GREDQF0u4p7/l6d98D6R6DUVR0NHRQU1NTaitrcWOb3Zg+fLl2LJ1K8rLyrDw6YU64y5PHRdFEc+/8AL9z+WXD+peeOmf/8QlP/yhvg/4V26FHIxRUFSIdWvWori4uFfUR/xdtLa20uTJk9HS0qK747jFEQgEMHbsWKxdu5Zxy7I3ayMNIH0BD02FwARs27aNZsyY0aVhDd/AK1aswJQpU/rFtsvdOV+u+JK2bd0G2SSDc79F+nCzqBcgKoC4UDJ8f4DKgR2geACL0jnQAToHxqAoCiZPmoyxY8f26ErqSRsyCpRkDadSEVjJNPVEANSdBt0doBkJCo1uFONoamqiqqoqVFVXoa62Tqch9/l8erOpeDp4Tj9ut9v1r/Yo1YrVao2hW+drh1scPBlgypQpGDt2LEs2v6nObaJ57HH/xxEh9gaAuotxxINnX/fFJ598Qq2trRFrmDQADKRpepGfTtVDZKD70fROwZHPRkx2TTtAQRIKhTBu3DjMnTs3KeX96jVrqKO9HUxg0DSKnFeLriWK/psIpJFOK8Q/k/j3Wuxno/vr+OOPR2lpaZ9cuzyG8+c//5luuummLlYIJ1r829/+hmuvvbbXRItpAOmH9fHjK66g5559Nual8Bdyyy234J577mGKqkI6DMn+ugOHeKvEKByMAqEnRtpEgqU/QiXV54rX0JMx0xpBZSi0RuVV0d2BRTzVxGDOaaK4UCIOrr4AjTGBortnHewMumTXOFzIFvl8er1emjJ1KvZXV8NsNscoS6FQCEVFRdi0aROcTmevYrVpAOnDpmGMYfv27TRjxoyY3/GXUVhYiM2bN/f6ZfQEWoP2rgxaprEtbl83V7xQMNJvpyrIVFWFz+cjnnLKU0n9fr+eFaQ3ggJBYIIe+DabzTq5osPhgNPphMPh0MnpeprjnkAlWewgEUV5d7GFZD+LtxC6ex9GId7TewuHw/B6vRSfmsvTnI3vSZKkLlaU1WrlX2NiY71Zu8no8fu7Lwaj73dPLtW+ZmDp9PEpjlT3Y09WyJNPPknXXHNNUivk7rvvxm233dYrj0kaQPpofVx66aX00ksvxVofoohOtxsvvPACLr300u9Eo6hkGm93zx0MBtHS0kINDQ2ora3Vj7q6OjQ0NKClpQXt7e26e4gDBqevSHXz8/oSq9UKp9OJzMxM5Ofno6SkRI/bVFRUoLy8HAUFBUldRP3dwAM93xw0EllFHq+HqquqsWfPHuzevRt79+7F/v37UV9fj9bW1pg55QCcLKvPyJYsy7I+l3Z7hDE3MzMTOTk5yM/PR0FBAQoLC1FUVISCggLk5eUhOzu7x4SJRAzNAwUu6RG7R6Op8LRjxw5YLJYYhUFVVTidTmzevBkFBQUpx1rSANIH8Ni4cSPNnj07JiDFA+fz58/Hhx9+yDgtx7dpAcZ/351gJSI0NjZSdXU1du/ejZ07d2L37t3Yt28famtr0dzcrNOsJ9O6jJXQfYmhGAPSRo4v4xBFETk5OSgrK8OECRMwY8YMzJw5ExMmTIgRfjydujfB4YEGDX6/xrFnzx7asGEDVq9ejY0bN2LXrl2or6+Hz+frch4+n6IoRmJmcVZWsiZP8XPJ5zPZkGUZDocDWVlZyMvL09O2y8rKUFZWhuHDh6OoqAi5ubnMWFPSHbik2tArPbqXXa+88gpdfPHFMSUHRivkhhtuwIMPPpiy8psGkD68hO9///v0xhtvdDEFiQirVq3CpEmTDrn1wTd6/KZLJYXUqAn2pIUoioKGhgbat28fdu3ahe3bt2PHjh3Ys2cPamtrE2YtGaul411F8WCVyK3TWzdE/DMlq1sxMs+WlZVh5syZOPnkk3H88cdj5MiRLN51Mdjv11h7wIfb7aYVK1bgww8/xLJly7B9+3Z0dHTECAKTyRQDdInmtLfzmQhoEglzY2sDTqSYCMhcLhdycnJQVFSk166MGDECI0aMQFlZGYqKirq1Xnqqy+mOmTkVBejbOPicddfsTpIkbNi4MeVmd2kA6SV4LP/iCzo+rm0kR++bbroJ991335AAj4HcGB6Ph9ra2tDQ2ID91ftRWVmJ3bt3Y/fu3aiqqkJ9fT3a2tpiXCGMsYR1Ez3FERIJgVSzqZJpsd0BUyKwVFU1hv4kIyMDs2bNwrnnnouzzjoLw4YNY0ZhOdDvmmdz8fsJBAJYunQpvfHGG/j444+xd+9e/bNWqxWSJPVYQd9fDT6VeUz27hKBdncA43A4kJ+fj7KyMowePRpjx47FmDFjUF5ejqKiImRlZbGBWt/GItnvigz74IMP6PTTT09qhfz4xz/GM888k5IcSwNIL4XySSedRJ988oneuF4PnBcVYtPGTXC5XOxQ+m/5fX69fTu98/bb2L59O5qbmyMU1NEFIcsyLBYLrDYbbIZ0UlEUEQ6H4Xa70dHRgba2NrS2tqK1tVXnoIrf8IIgxABFb0AimVvKKGSM7qf+DB4T4dZP/L3GxwGMgfRQKKQX8uXm5uL000/HFVdcgeOPP57Fb86BFGbbt2+nf/7zn3j99dfx9ddf65vcYrHoRZzJsqCMz2cU1v1lMeBxkXiyxUQur74AjJHyJn6tWSwW3S1WUFCA3NxcZGVlweFwQJIknfLG7/frhx7r0VSIggibzYbs7GyMHj0aZ5xxBsaPH68rA98FEOHy4bTTTqMlS5bocix+Ha5avRqTJ03q0RWfBpBeIPe7775LZ511Vgxyc9R+4okn8NOf/nRAG9b39T7/+thf6ab/vSmmerm/AiMR+WCqWUmJhISRvynR38qyDJvNBofDAZfLhYyMDLhcLrhcLjidTr2Wwmw26/dGcZxYHo8HHZ0daG+LVEPz4Lzb7Y4RpKIo6udJBChGMAkGgwgGgxAEAccddxyuv/56LFiwgPHn4qDa201t/LvPPvuM/v73v+O9996D2+2GKIqwWq06x1EywDDSnBg/Y7FY9IB3bm4ucnJykJWdjQyXCw6HI8KJZaAniecVc7vdaO/oQHtbWwy3mNfrTbjGOHNvX61P/vn4vx1IIOTW27XXXot7772X8T37bQcRLiNWrVpFxx57bBdmby7PFixYgNdff71HKyQNICmY7nzxzp49mzZt2qS7rwRBgN/vx+Qpk7Fy5UomidIhyx7hL/rNN9+kBQsWwGq16gy3qfq2++Ky6EmT5CARP2w2m54VVVRUhGHDhmH48OEoKSlBUVER8vPzkZOTg4yMDNjt9l6ljSZ7HrfbTS0tLaitrcWePXvw9ddfY+vWrdi+fTuqqqp0YWgU2PFV4UZh7fF4AABz587FzTffjDPPPJP1xi0S7wL78MMP6aGHHsKSJUugaRqsVqvefTEZoIXD4RiCyOzsbIwcORITJkzAxIkTMXbsWJSXl6OwsBCZmZmstwzKie7Z5/NRR0cHWlpa0NDQgJqaGlRVVelHTU0NGhsb0d7e3uXdcws4mcWaimssVVdcT/ERVVXh9XqxYMECvPLKK8yYsPFdAJFLLrmEXn755YTdU/1+Pz777DPMmTOnWxBJA0iKk/3MM8/QlVdemZCyZNGiRTjjjDMOWeyDb77Ozk6aMmUK6uvrYTYPTNOpZIVhRmBN5G7gvuycnBwUFxejtLQUI0eOxIgRI1BeXo5hw4YhvyAfWZmp+bPj/fu9afPaU/2J3+/Hrl27aM2aNfj888+xYsUK7Nq1SxfuVqtVDzLG850BgNvtBgCcffbZuPPOOzF16lTWk1vL+LtVq1bRH//4RyxatAhEBIfDkfB6XMAZXWpOpxOTJ0/G3LlzMWfuXEyZPBmlpaXdBp85GKW691OdRz7C4TCaW1qotqYG+/btw+7du7Fr1y7s2bMH1dXVaGxsREdHR0IeNX4kWmvxis1ArG1ZltHe3o477rgDt99++3ci9Z4n1+zatYumT5+u/9uYUerxeGIySpPt0TSApCCYvV4vTZ06FdWGKk4+yaeccgo++OCDg77wjAVU4XAYFosFDzzwAP3v//4vMjIyEmp+STdSZDd1eW5j/CERQAiCAKfTiezsbBQVFaG0rAwjR4zQgWL48OEoKCiAy+XqsYAvXsPuT+C8p/cZn4kTPzc+n4/WrFmDRYsW4b333sPWrVt1V5DJZOriRjICic1mwy9/+UvceuutOhV+IjeOIAior6+nO++8E8888wxCoZAOHEbgT2TxZGVlYc6cOTjzzDNx4oknYvTo0Sz+ORPRnAzkPCayHPh1ultrgUBAT+/eu3cvdu3alTC9O5HlHB976SmtO1lv9vi1pmkaTCYTtmzZgqKiIn0vc+bfb6NFwp/xxhtvpIceeigp0eJ7772H008/Pal8SwNICpN8zz330K233hozydwfvnz5csyaNeugAkiigJ/b7aYJEyagoaEhYUvdRH0wuhuCIMBiscBut+v+88LCQpSUlOg07MOHD0dxcTHy8vK6rfI2CrTeuiEOpqLANTFjDCsQCODTTz+lf/7zn1i0aBHa2tpgMpn0bozxQBKtoMfEiRPx8MMP46STTtLjI1wgAcALL7xAv/vd71BTUwO73Y74tqVcEHMXlSiKmDVrFi666CKcedaZGFE+ghk1Sp65lSoD8WDPZyKASQTWxuH1enVCxerqalRVVaG6uho1NTWor69HS0sLOjo64PF4EAgE+kRqKQgCrFZrQr//z3/+c/z1r3/9ThSY8LXe1NREU6ZMQVtbW5e6Nq/Xi6OOOgrLly9nydZVGkB6mOBEfPp8wV144YV49dVXDyp4cHNy0aJF9NZbb4GIkJGRgbVr12L58uUx6cX88yaTCQ//5S8oLirSF4iqqggrCpRwWHdBqaoKWZaRmZmJ7OxsZGVlISsrCxkZGd0WfBldI9+GimKju8wYM9i9ezc999xzeO6557B//34dSIyCjAt+bi3ccMMNuPvuu3Wa9ZaWFrr++uvxr3/9K+Hfc4HGgcPhcOCcc87BT37yE8ybN4/FW22HWy1Dd1xdPT1HMBhEZ2cnJWoUFgwGdRDltTAmk0mPt2gUIUp84YUX8M4778Bms3Vx8TLGcMEFF8DhcKCzsxMjR47EL3/5S2RkZPSqF8fhpiA//PDDdMMNN/SN7j3+haYP0rN5iAg///nPCQC5XC6y2Wz6YbFYaPPmzcSziQ7mPb300kuECJtOzOFwOGLu0eVyEQD68Y9/TP29trFLHT+MwGOsUh6Mg7vSejoG47rhKMjyuWhoaKC7776bioqKCADZbDZyOp0xc+9wOMjhcBAAmjlzJm3bto02bNhAY8aM0ddT/PtyOp1kt9v1d/nTn/6UtmzZQsZ3wCldvi3zG38PPOkifp1xa6+/x9atW8lsNsfMu/GI31Pf+973iN/bt03G8efy+XwYP348SZIUsyYdDgdJkkQTJ02kQCCgv/8Y0su0BZLc+khEmMitj8svvxzPPffcQbM++HsKBAKYMmUKVVZWwm6369ZGfKaO8e/WrFmDI444IqYpTnd+4kRB84HWvnrK8oq3YFK9fqKNMlAWEZ9jbpXU1tbS/fffj6eeegp+vx8ul6uLW4uvl9zcXBARWltb4XQ6E1otnZ2dkCQJF110EW6++WZMnDiRAYhJ2hgord+o+Q/E/CaLXw3kukkWSE8lJZhr3GazGeeeey699dZbXQrp+Bwb04jb29vx+uuvY8GCBd/KADt/pjfeeIO+//3vJy0ufOqpp/CTn/ykS5lCGkC6mdQLL7yQ/v3vf3cx7QBg3bp1es+Mg+FC4C/u6aefpquvvjrhPSUCuosvvhgvv/xyQvA4FH7wVN0V8SNqBZDR4jESHUYJ/1iydqvJ3G29FaDGmA4HkrVr19JNN92EpUuXwm6364FZo1DiSQ284M34nnhtybHHHou77r4Lx887XgeO3sY04vmjUp1rbk0qikLxrshoHRDjGVJ9cQcmSpI42DEwVVUhSRI+/vhjOuWUU2Cz2bpNc+fprFOmTMHKlSsZD95/W+XdySefTB9//HFMcSFvzV1aWoqNGzd2ac2dBpAkk7l8+XI6/vjjE1KWXHnllVi4cOEhsT6mT59Ou3btimHTTKZ5hcNhrFixAtOnTx8UAOmvP7u9vZ1aW1sjrU4bGw+0OG1uRlu0aM3tdusFa8FgsAuDLL+ekS2WM+9GuJYKUVIyDGVlZSgtLUVJSQmys7NZIoDuLfOuEUiICPfccw/deeedICLYbLYuVobxXRqtjry8PNx+++247rrrGG/q1Rvg4AI6WZC6ra2NeK2GMSjd3Nwc6Vnu8cDv9+v0LUYrip+Tzy+ndnc6nXC5XMjKytILFPPy8pCXl6cXK2ZmZvZIoc/3nLEPyMGwfI899lhauXJlDB9UosHjAC+//DIuvvjiQ1ooPNgyb926dXRMgvbcXO498MADuPHGG2PmIA0gCTajIAiYP38+LV26tEupP2MM69atw5gxYw669bFw4UL6yU9+krL1cd555+GNN97QU0kHyk2QSsEVp2yvr6/H/v379SKz6upq1NbWorGpCW2trXC73QgEAj1SivPr8edgSdKOjf7q+GG1WnXm3fHjx2P69Ok48sgjMW7cuBhBx8+RKphwl6cgCFi6LdCQaAAAlWVJREFUdCldddVV2LNnT9L3JAiC3nVwwYIFeOCBBzBixAjG7z8VoDcSLRrnoqOjg77++musX78eGzZswLZt21BdXa3T2SQTkt2lxSZi4+1u8Oy9jIwM5OTkoKCgAMXFxXqx6LBhw/Ri0Z6KG5OleCdbez0BjqIokGUZL7/8Ml1yySUJ3ViJrJBJkyZh9erVLH6+v20gcs0119CTTz7ZJeNUURTk5ORg8+bNugLGGEsDSKJJfPudd+ics89OSFlyxRVX4B//+MdBtz6CwSCmT59OO3fuTMn6CAaD+Pzzz3H00Uf3aH0YBVdvgKa9vZ2amppQEy0Yq6ysRGVlpQ4SvOVrPGW7weWUkO4ingfL6HLqzu/OteVE9QJGmhMj1YfJZEJpaSmOPPJIzJ8/H8cff3xMXUVv4g9cODU0NNAVV1yB9957D3a7vQv48pqP++67D1dddRXjLqSeNFtjqrHxfrZu3UpLly7F0qVLsX79euzfvz9m85vNZp2hl9eUcOvJOMfGGolEsSMOqMYjnneLn9cYEE+0Vm02G1wuF3Jzc1FcXIyysrKY/izDhg1Dbm4u6622ryhKt8oNt3ICgQCmTZ9Ou1Ow5rkV8u9//xvnn3/+t9IK4euqsrKSZh99NDo7OvSsU6P8451W9dYGaQDpqmXNnj2bNm7c2CUl9lDGPv7xj3/QVVdd1aP1wVNITzvtNLz33ns9Wh/xz+H3+9HZ2Ukej0fvBOj2eNDU2Ijq6mr9qK2tRWNjI9ra2uD1ersEZ40plFxoGfmvkvEZcV4qi8Wi9w23Wq2wWCx6//D4c3L+J7/fj/gOhvEFlbzLnizLuiXg9/t1kMvIyMCMGTNwzjnn4KyzzkJ5eTkzuqt6sry4n52I8Itf/IKeeOIJXUAZNbmPPvoI48aNY6m4q/hzGoXWjh076L///S/eeecdbNy4EV6vFwD0ueP3wOcmHsQ5sPB5jp9fI90Hf2c8VsMPnhGWbB0aK8uNwWm+DoxrwThMJhMyMzNRUFCgU72Xl5frdUe5ubnIyMjQaV54urXL5dIpb7qTaxzoH3roIbrxxhtT2lNerxezZ8/GsmXLGFcEvq0K9H333Ue//e1vu1ghqqrCarVi8+bNKCkpYeksrJhFpUKSRLz40ot02aWXJbQ+DlXmVW+sD0EQ4PP58NFHH+GEE07o1vrg4LFhwwZauHAhtmzZgvr6enR0dOhClcccErnJOAsvFw5GgDD21zC6kFwuF7Kzs5GXl4fCwsKYg3exy8zM1AkTo9XfKWmiURp28vl86OzsREtLC+rr61FVVYU9e/boVc/79++P6aHBBSi33LgwzsnJwcknn4z/+Z//wSmnnMKMZIPdvX8uoFpaWmjUqFEIBAL6HHm9XkydOhXr169nxuK/7tYA/0wwGMS7775Lzz//PD799FO43W4wxmC323VSvGAwGNNMym63o6ioSO/AyBkCCgsLdUFss9n0eY4nzIwjviR+fq/Xi87OTrS3t6OlpQVNTU1oaGjQj8bGRp280uPxJORDk2VZByyjdcRTpxMBFC9wtdlsuoZsABCcdNJJ+NOf/gSbzZa0doPPaUtLC02aNKlLEV2yfeX1evHhhx9i/vz538qMLD6XgUCAjjrqKMR3LuRy8H//939x//33R5SfNIAcmDguqBMFqXk67Pjx44e89XHCCSfg448/7tb64JuooaGBpk+fjrq6Ot0FxDc0F3rxjKi8LiIeWEwmE1wZGcjLzdUJEuO70HG/t9ls7tM7SqRd9iY9NxQKYf/+/bR161asXr0aK1euxObNm9HY2Kg/g81mgyAICAQC8Pl8YIxh9uzZuPbaa3HRRRcxrvkmux6Po1VVVdG0adP0SnLGGHw+HyZMmIB169b1eB6+zhRFwTPPPEOPP/44Nm/erAMDp9Xx+Xy6hZGbm4uJEyfiqKOO0jsrlpaW9thaNpny0h9/v9vtptbWVjQ2NqK2thb79+/XK8xramr09sW8EDD+nfJiwHiL01h7FX+/gUBATzntzi3IQf6mm26iP//5zyntLbfbjbPPPhtvvfUWG+ieO0POjf/223TOOefEKNLcgs7IyMDWrVuRm5vL0gBiENSPPfYYXX/99bF9zqOo+6Mf/QgvvvjikLc+PB6Pzl/TnabMYx5r166lmTNnIjs7O8Z3HQqFulzL2KqU9xYvLS1FeXm5DhKFhYXIycnpESDiK9eNwipRBk6qbWzjv0+FSqO2tpZWrlyJ999/H5988gn27Nmj++l5pbjb7QYRYfr06bjvvvtw0kknJQVoPrfV1dU0ZcqULgAyceJErF+/nomimBRA+M/3799Pl1xyCZYvXw5RFOF0OvXALg+KV1RU4Pjjj8dpp52G2bNn6w2v4gVDorqYRHMbn6CQbJ67q99JhXQxEAigtbWVOKPv/v379Vgad5NygEnUitgYP+MV6B0dHXjooYfw61//ulsA4QCwZ88emjZtmp75lkqsYPXq1Zg4ceJBUyQPFYiccsop9NFHH8UkEnF5+Le//Q3XXnttGkCMTLaTJ09OyCWlaipWr1qNSSk0WDmU1ofX68XRRx+Nzz//PCXqBSJCKBTCBRdcQO+++64et8jMzERRURFGjBiBioqKGJr1goIC5OTk9Jie2R21SX8124F43/yITxzo6OigpUuX4pVXXsGSJUvQ1tamxwmiiQOwWq1Yv349xo4dm1ATHQgA4bGUq6++mp5++mnk5uZCURR4PB4oioLc3FycdtppuPDCCzFv3rwYwsp4bqxDRScTbzX2NtXb7XZTQ0ODDix79+7VrZempia0tbXFWF8AMHbsWLzxxhsoKipiPTWJ4nN82WWX0YsvvphyduN1112Hxx9//FvL3Muf6/PPP6cTTzwxJhbMXXnHHXccli5dmgYQPll33HEH3XnnnQmtD16Mdyisj2nTpqVU98FN7DfeeAPnnXceSzWrRxAEhMNhrF27lhhjyMrK4nn8rKe/7Q9BYry1MFA03X2pojdmHxnnbPfu3bRw4UL84x//QFNTE6xWK2w2G1paWvDpp59i3rx5Ca28gQSQ888/n15//XXIsoxwOIzx48fj8ssvx0UXXRRD287XbKp9SBIJ9v7Md2/nPJGV2BuA0TQNXq+XvF4vgsGgzhBQXFzMuLsrlb0vSRJWr15Nc+fO7dJcKdGzKooCp9OJLVu2oKCggH1bOxlyxWjevHm0bNmyGCuEz/XKlSu/21xYPH2xsrKSsrKyyGq1kt1uj+HGMZvNtHHjxoPKecV5l55++umEPFzxh8PhIFEUacaMGWQstEv1SIX7KhxHutgbXqP48xyseTRyiMXzKvWG+6qyspJuuOEGysvLIwB04oknEm8RnOhc/PmqqqooKyuLLBYL2e12cjgcJAgCTZ48mfhnkt0Lz1JbuXIljRs3jmbOnEnPPvsseb1eMq6T7p7H6I7kz57sfQ8Gz1I8r1V/1lBv1k9v1j8/16mnnkqMsS6cZvEH55e7//77ybhXv20Hf67HHnusiwxyOp0EgB599FH6Tlsg3KJIZMJyjX7BggX497//wxQlnLypSh81ECPnTqJgYKpV5/HVsqlYH8k08FRdHt1pj6n0UOAaZGdnJzo6OsAZVjs6OtDZ2alXoPMK6VAoFNPjwtjb3W63w+l0IiMjA9nZ2XoldE5ODrKyslgiq5EDS3cU6PHcV5WVlbRs2TKcddZZyMzMZN3FL/prgcRrvcY5DYfDCefYSGPSHXV6Z2cntbS0oKWlBc3NzYae953wen0IBAIxGVDx8221WmG32+FwOJCRkYGMDBcyMjKj32fwDDqWyrrrqxWbKN4VbwWlKgMkScKiRYvozDPPTKmwMBAIYNSoUdiwYYMe60uW7dVdLCmV36Vi+Q2WBcIYw/r16+mYY46B2WyOoXp3u924+OKLIX3XwWPlypX0r3/9C3a7PSH1xPnnnw9BiGSEHKz7kiQJL730Em3fvr1Hv6yxUvb73/9+n2I0yZoAdUdK2NM1Ojs7qbm5GQ0NDXoGTk1NDerq6tDQ0KDTaLjd7hgajYEYkiTpKcN5eXlUUlKCiooKjBs3DuPGjcPo0aNRXFwcUwGdqPqcf8+BpLy8nJWXl8dssIO1Hjjg8SLMZO43IxC63W7avXs3vv76a2zbtg3ffPMNqqqq0NDQkDB+0J/BwYXTnGRlZVFubi4KCgpQVFSEkpISlJSUoLi4OCaO1p2ik2ocrT/vgYP4qaeeyqZPnx7TsjrZPVmtVmzfvh1vv/02XXjhhUkLC3ubBDKUBl9HfO0ZOxZyGbNjx47vLoDwF3rPPfdAUZQu3EW8j8bjjz+OxsZGcjgcB6qbRRFiVLjwIKwgCMl/nuDfgiDAZDIhOztbz5rhQV2/348HH3xQL5Tq6UUrioJf/epXMJlMSNX6MBLuJWJp7akqPRQKoaWlherq6lBdXa1XoRt7Yre1telB30RC3niYTKYYmpJkFk58emkii4lrtu3t7WhsbMTGjRtjBEZOTg5GjRpF06ZNw5w5c3DUUUehoqJCb63LwcSYxsybRfHNc7AEAgcwfu/x788IGm63m9avX48vvvgCK1euxNatW1FbW6u3v+Xn4Kmx3HJLZlkmiy0Zv8a7jjweDzraO1BZWZnwvcuyzAEG+fn5VFJSoqd6l5WV6TQnOTk5zLgmkoFrvHITf8+pWCScz+y6667DVVddFTPn3b2Xxx9/HBdccEHSe/T5fGQsutRdeKoCVVFjaHeStWtOtM7j2QHiv0/2NRnDdfy75K69trY2/P73v084f6Iooqmp6btZB8IDRFu3bqUjjzwyecP4qMthoPovxy8EURRhs9kwadIkvPnmm8jMzGSCIODJJ5+ka665JiXrIxAIoKKiokdzOv7ZUxGAbrdb7xBXVVWFvXv3orKyUm8/2tTUhI6Oji45/NzdIctyTD2JUdD0VJFuPFc8LYlR8zb2qkg058bGQrzrH+8pzv8mMzMTEydOxCmnnIIzzzwT06ZNiwlO95YRdyBdWImEndFKam9vp6VLl+Kdd97BsmXLsHfvXv25eHU5Vyj4s3NB1t26StSGNz7tOpn1Z6w+Nyoi8e8+Ec2J2WxGRkaGTnHCK9F5mjjvgJmZmclStbR7mmOuSHm8XpoyebLeLKynwkK/349PPvkExx13XExKt6IouPDCC2ndunW8ME+vvE8We+zJlZXImkmWMJLs3z0lmFDkJvR3HQgEEA6Hu9Dx8Dl1uVzfTQuEC9F33nkHwWAwqaAmIr3daF/8lsk+Y+SXaWtrgyzLsNvtLFpFTg899FCvrI9f/OIXsFqtPVofxpaq27dvp46ODh2E6urqdNJD7m7irg63291FQHOA4DEIo/bONwtfgPH3zBlzEzG58k6IUSZX2Gw2WK1WncuJ378xQOzz+eB2u9He3o7m5mbU19ejtrYWNTU1ujXU3t4e4+qx2Wy6WzIYDOLLL7/E8uXL8ac//QkzZ86kCy+8EOeddx6Ki4v7TK0+GC5X/n5XrFhBL7/8MhYtWoS9e/fqwtflckEURX3+jRX3FosF2dnZKCwsQFFRMYqKivSK9KysLLhcLr1AkQOuMV7Egdfr88Lj9uhxq9bWVvCYSktLC9ra2mJazyayQjiljNGC4gqB1+tFe3s7tm3b1mXt2Gw2ZGZmIjc3lwoLC1FcXKy7xoqKipCbm6s/v8PhwLhx4xgvuOyOgFFRFDgdDnbllVfSbbfdBqvV2q3yxqk9/vKXv+C4447rIuC//vprVFdXw2KxdGEZ7i4+1VMspVeyhghaN3U8SZ8NDGCRolo+d0k/+120QPhmvOiii+jVV1/tMXA2GEOSJLjdbkycNAnLPv8cLpeLMcbwxBNP0LXXXpuS9REMBjF8+HBs3LgRDoej25RCrhVv3LiRrrvuOnz11VcIBoP6RkjUWMdINWEkJTRm2CSqSLdarcjMzER+fr6uRfI+6iUlJUYajV5XpPd2BAIB1NfX0969e7F161Zs2LABmzZtwq5du3ThyqvPJUlCKBSCx+OBpmkoLCzEeeedh5/85Ce6VZIKLf5AWyDGOMgbb7xBf//737Fs2TKEQiGYzWa9B0kwGNRb6UqShGHDhmH8+PGYNm0aJk2ahDFjxmDYsGHIyclhgwmEHo+HOjra0dzcooM551Dbv38/6urq0NTUhPb29hjaFb6u4ylOjOvOmCWXzDXKP88Yw5QpU/Dee+8hNze3x/0hCALq6+tp0qRJ8Hg8PdKb8HezatUqTJ48mRndinfeeSfdcccdyMzMjFGiDra87Y+rtbt71TQNNpvtu+3COvXUU2nJkiUHHUAkSYLP50N+fj6WLVuGkSNHMlVVEQgEaOrUqdi3b1+PyB/P0Z+K9UFEOP7442n58uXIzs5OGJw0AkQysjtBEPSK9MLCQgwbNgwjRozAiBEjdNK7goICZGdn98hhlYxltzcByGS+42S1BJqmYc+ePbR69WosXboUy5cvx86dO6GqKkwmk25R+Xw++P1+2Gw2nHvuubj11lsxfvx41pPQH0gA4RlVn3zyCd1+++1Yvnw5GGNwOp2QZRmhUAhutxsAkJ2djSOPPBInnngijj32WEyYMAGZmZkslfhBb4VNd4zIqVCqt7W16RXovEhw3759qK6uRn19PVpaWhJyaHErLN49aly7xnXa0tKCJUuW4OSTT+6RlZoD9fXXX0+PPfZYyoWFRo48Pgf79u2jyZMn65brt03OEhFMJlMaQLoDkL50zktldHZ2wuVyYcmSJZg1axYLBoMwm814/PHH6ec//3mPC5c3i8rPz8fmzZuRmZnZrXbFhVAgEMCYMWOourq6R4Cz2Wx6P4fCwkIYg52lpaV6Nk1GRgbrSUNL1lq2vxpSqgs9podzlPIiPti5evVqvPPOO3jvvfewfft2AIDL5YoR0g6HAw8++CCuvvrqbnnGBgpA+HnuvfdeuuWWWyAIgu6i4szDTqcTc+fOxXnnnYeTTz5ZZw82zj+/xsGadyICgQBCnzL4vF4vNTc369l7vJdMvPXCG431JMM2btyIKVOm9MhMzeXC119/TTNmzEi5GDJKSIpRo0YxI8PBrbfeSvfccw8yMzN7dEcfTMHfW1mZjM5GkqTvtgvr+9//Pr3xxhsJAYSX7A/G/EyaNAlPPfUUZs+erWtFXq+Xpk6diqqqqpStj7vvvhu33XZbSnUfXBi99dZb9J///Edf+BaLBU6nE9nZ2cjPz9cPHo/IyMjoNljZ34r0ZAu7Rz9tN/xNvQEVPp9G4fXRRx/hH//4BxYvXoxQKKQDSSAQgN/vx6pVqzBjxoxB58ISRRH79++n8ePH6zEOn88Hn8+H7OxsXHrppfjJT36CCRMmJKQxSTVZIn6+U13zfU1T7W+7Y6/XSzze1dTUpB/Nzc16vI67ZseMGY2f/eznrDfdHUVRxAUXXED/+c9/UrZCfvazn+Gxxx7T1wQnZz3vvPNo8eLFh62s5AkoidaEIAjfbQC58sor6ZlnnumySPhGnzFjBk455RQUFRXpWRnctaNXu5IWTV+I21z65o1oYoIgwG63o6KiAscedxwzGXoZiKKIRx99lH75y1+mZH0oioLMrEx8tfmrHn27A+ET7c6KSKXYK1mmSXxgsb+aldEd1tv7NPaV4OOLL76ghx9+GG+//baejeL1evHRRx9h/vz5SV0iAwEgHODr6upozJgxemwjKysLl1xyCX75y19i1KhRzGhldCd8E9X06EJggOY/UaZWX6llkqVu9yWFujcyjruxEvFAJdtPqqrCYrFg8+bNGDZsGDMGzVVVxaL33qOmxsakVfKqpkHTVGiq1qXzo34QgZKwSPT4M9JAGiX/dxIWACLCV199hebm5i4gwi2Q72QWFp8Im93W5XeclPDcc8/Fa6+9xgar85jRreB2u+kvf/kLZFmGqnUfi+H3d9WVVyEvL6/XVeecuiEZ42qiDZ+qUOqtqyJ+PqI58xRPO2KkkzZ2MzSbzbwpEuM+8e7Ob6x9MWrnxnvln2OMYc6cOWzOnDlYvnw5Pfzww1ixYgWuvPJKzJs3b9BJNXl/laKiIvbEE0/Q22+/jfHjx+NHP/oRKioqumSGxb8jozBP5LaLH+FwGH6/n/x+f0z/eSNtunHueYaOxWKBxWJhZrM5ZZdvKm7NnmJMiYCmOysh1cFB/dhjj2Vz586lZcuWdds3nccC2tvb8fjjj+Pee+/VwZwrh2efddZhS5a1evVqOuGEExLO8Xe2pS2vHP3Vr35FjzzySIzWz2MFK1aswIwZM1gwGBxQQWEUXvw+Hn74YbrhhhtSsj5UVcX/t/fe4VVV2fv4e26v6b2QAAk1gUSk2MGGwjDYwS52nBHrOI7jqJ/xqz/HMgojNkYRC+qoY1dEQEGkiCKEHlJJ78lNbi/794esM+ece25LggY563nOk3LvaXuvvdZe7V1msxllZWXIzMw8on0JBupq8Pv9sNlsrLOzE+3t7WhtbeUbDpHLoaurCzabDX19fXA4HLzwEubNC+MXJOxJkFH1c1xcHBITE5Gamsr336Z4TU5ODhITE2UhzsPt3KUFfN3d3YyC0pEyVAYzC0v6eaiUYlJ+cmmifr8fTU1NTFjPU1dXh6amJrS3t4vSbmn85QQ9jRUVf+r1ehiNRlgsFsTFxyPx5xRbpKWlISMjg08VTktPQ0ryz5l30fDNrxk3o14h7733Hrv44osjJtkI+2Ts2rULqampPNR7tPUzR5JiGS/pd1UqFSZNmsR++uknmM1mkfLW6/XHbiU6Map08DweD5KTk5GXl/e/TIMjwLC0O+np6WGLFy+GTqeLmAlG1sctt9yCrKwsLpqU0mhcBdLfo1UQbreb7+cgTdWklrcdHR2w2WxwOBwhlaOw8l1YxCZ1cdHzUXaY3W4Xmdxyrgaj0UjzycaOHYvS0lKUlpZi3LhxoviOECpECGUiFNgJCQkRA7FHSqAJ3VrSinRqdyusk2lvb2dlZWX44YcfsH37duzfvx8NDQ3o6uoKjvdxHNSCroA0/lqtVhZvS1gXYrPZRAWdofiWkjKofiMnJ4dP7xYWCIbCLotGwURjvcRihcyZM4crKipi+/fvD4tHR8K0tbUVL730Eu6//35R++OjFfKdxthqtcp6LY5ZF5bQbJcbtMNm+RFNvyNf60svvYTa2tqoM6/i4+Nx6623Itq4h9AlI7WCIglCr9eLzs5O1traioaGBhw6dAi1tbV8RgylW9pstqBqdHIdUD6/xWKRbZUqVQDCat1oLDkSqMJCQxIidN329nY0NDTgu+++458rJycHJSUlbMaMGZgxYwaKi4t5bCwSUFIB8Gt1oZMTQBQoFiqNsrIytmbNGqxbtw47duxAY2OjCABPr9fzDamE40PWhhCtNxoSjr9erw/aBAhdaTQPTU1NImgZIpPZhIT4BKSmpjJhN0tSMJmZmVFXoIeKxQjHIhJv+Xw+6PV63HTTTbj11lsjwptQCviLL76IP/7xjyw+Pv6oh3qnNSDFAaSx1Gq1x6YCEfa2DvX5kU5zVKvV6O7uZkuWLInJ+rjmmmuQn58flfVB9wn1vb6+PtbV1YW2tjY0NTXxSqKurg4NDQ1obm7m3RvhKoqpGj1UwRd1zpM7X6/X89Xm9JMUeLie65QRRf257XY7X5EuFYA6nQ4Wi4XfUft8PjQ1NaGmpgYffvghjEYjJk6cyGbPno3zzjsPRUVFnFCJSlv7/tpWs7Aivby8nH3wwQf46KOPsGPHDn6sDQYD4uPj+d20x+MJ6pfOcRyPCmC1WhEXFweLxcL3ohcqBZpTuo507Ak52eFwhIRJEVafCyFOhNhLra2t2LlzZ9B6lFagZ2dnIycnBzk5OaLi1MTERFgsloiZg9EobcYYLr/8cvzjH/9Aa2tryGwk4cazvr4ey5cvxx133IFQIIu/BTqmFUgkRqId2ZEUAhqNBs8//zzq6+sjWh8kyCwWC26//faorA/aPVRVVbElS5agrq6OF8i0UDs6OtDd3Y2+vj5ZVFZSEDqdDkajUeTTJQVBfcOli89sNiM1NRXJycm8P5wOShNOTEzk4b9NJhP0en1EAD05N5rT6WR9fX0gZdjY2MgXp1VVVaG2thbNzc18wR0JJLPZDOBnYMitW7diy5YteOyxx3DKKaewq6++GnPnzuWMRiPPE7+mK4JcIlQd/fnnn7OXX34Za9asgc1mA8dxMJvNfIGo2+1Gd3c3f35CYiJGFhRgVGEhRo0ahYKCAuTl5VELYlitVhiNRi5WJXkYMoU5HA7YbDZ0d3ejo6MDra2taG5uRlNTExobG9Hc3CxyaYbakFAhp1wFemdnJ1paWoIUjHCTcBjSn6WkpPCwJnaHHW6XGxqNBjNnzsRNN90U0TqgjUZiYiJ39dVXs0ceeYRvbxzO2tdqtVi6dCluvPFGZjKZjmorhJ6bMgCF8jIQCMBsNh/bWVhaGYh2tVr9M6JoTw/i4+Mx2AxAVkFHZwd79tlnodfrIyoryjVfsGABRo0aFdH6oGdua2tj55xzDg4ePCj7nuRioiConAUh3bWS68JsNiM9PZ3vjS4HV5KcnIy4uLiYYTOEgbpwzM1xHGVicQkJCcjJyQlpadXW1mLXrl28oti7dy86Ozv5OElCQgKvTFatWoVVq1ahqKiILViwAFdeeSVSU1O5XwrCXW48aCf79jvvsCWLF2Pz5s0/u35MJl5pOBwOfrEnJydj6tSpOPHEEzFt2jSMLypCbk4OFwmtQJoKHW7sKbvLYrFwFosFaWlpEZV9V1cXD9BZV1eH2tpa1NbW8kWCHR0d6OnpCdrQ0L0IdkZagU7WEaFDhxL0H374IbRaLbvuuusiriPaMN1www149tlnQQk1ocaGoN4rKyvx5ptv4sYbbzxqrRAeYLKvjzU2NoqsL0rmSU1NPbYtkMTDQkNklmm06OnpQVVVFXJycgbd700ZOs8tfQ6NjY1RWR8+nw9GoxF33nlnVAqN4KnXr1+PgwcPIiUlhXcr0GIj37dc7IJA62gXl5mZidzcXL4SXRj0jFSJLowphGr8M5C2qFJlI5cMYLFYuPHjx2P8+PGYP38+AODgwYNs/fr1+Pzzz7Fx40a0tbUBAA/REggEsH//ftx11114+umn8ac//YktWrSIEzbe+qUWslqtxtdff80efPBBfPvtt+A4jndPeTweXhFmZ2fjtOnTMXvWLJx88smilrdCXoqUPhvLuwnjAqHa5NI86PV6ZGRkcBkZGSguLpZVMB0dHay5uVkUc6urq+OTMjo7O9Hb2wubzRZSuWk0GhgMhqCEDIPBgI6ODnz55Ze47rrrIrqyCL05Ly+Pu+iii9jLL78ccb2Ssl+8eDGuvvpqvn7saLNCyOLduXMn6uvrRfUwpLRHjhx5bCuQYcOGyXDgzwvg9ddfx/Tp0/mME6nQCIfbFGoRklupra2NLV26NCbr49JLL0VRUVFUsQ/aOY0aNQopKSno6uriF7HQ4khISOBdTJmZmSJkU/IpE8R8pF2rUJBIxyAaBRxtT/RQFeiR6gaEqcgajQaFhYVcYWEhrr/+etTV1bHPP/8cb7/9NjZt2oS+vj4+NsAYQ3t7O2677TaUl5ezf/3rX9wv3UyKIG5IcRCUuMvlgtFoxLnnnotLL70UM2fORFpammxVen9qc8LNRyjlH62yl1P0er0eWVlZXFZWFo477rig810uF7q7u/lGZc3NzWhububTwskla7PZeJgTqmdxuVy80pkzZ07Mm4A//OEPeP311yNu9ghkcO/evXj33XfZFVdcwUWyQmJFYQi3HgbDM0P8rVKp8Mwzz4RMNZ8yZcqxXYn+3XffsVNPPRUmkylIAHo8Hjz1z3/i1j/+kRvIPeT+98ADD7CHH344KuuDsq82bdqESZMmxZS6y3Ec6uvrWX19PZ+WSR3jrFYrLBYLF23hVygFEUsl+kAA+ITXFILm9bdWIFTNxJYtW9grr7yC999/H52dnXwLVwDo7e1FWVkZxo4d+4tiYZ1wwgls27ZtSE5O5uMHaenpmD9vHhYsWICSkhJR/xKhUI5mLENZDOHmNxwIZn87Bg607kj4bC6Xi5EC8Xg86O7uRllZGfLy8nDqqafGtAmgeZgzZw779NNPo2p763Q6MXHiRGzdupULVehJG8qhSo8++ii7//77RfUfNB56vR7btm07NhWIsO94SUkJq6qqEuV5k4nmcDhw0kkn4eyzz0ZBQQHi4uJ45SIMtOsNephNZhiNRiQmJiI/Px8Wi4WTMiHHcWhpaWETJkxAb29vWH+q0Po4//zz8d///jfmGoRoGFTaNjQWYRxOAMUKOUE+bI/Hw4Q1BSTgDwf0uWhbCwuVXiRhSG49YWZQVVUVe+mll7B8+XK0trYC+Blccc+ePcjJyeGiEfwDVSBkgTz11FPs7rvvBgBkZGTg+uuvx0033cR3spSmHUeap1jm57AAZvScGo0Ger2ei8WSCdU5MFaXWbgNSSywOP1p4qXRaLB69Wp2zjnnBAlUOaK+4e+9/z4uvOCCsFaIEIGBqv+Jf/n35H7u0yFUpuE6ogrcdxzBKnEyPOH3+xllNjqdTvT29mLPnj145ZVXsGrVqqBmUiST5s6diw8//JA7JhUI7dQ0Gg0ef+Jx9ud7/oz4+HjZ5kcUlIx2x28wGJCcnIybbroJf/nLXziacBJQf/nLX9hjjz0WtfXhdruxfv16nHjiif0qHJT2lI7F7RBuRxiNAHI4HIyyctra2tDa2orW1la+Cr2zsxM9PT3o7e3l00DdHjff7lNYOKfTaaHX/6/ndoKg4pkqz7Ozs6leQDZYLESmDSVshT0dAKCuro69/vrraG5uxrnnnotzzz33F4dz5zgOL730EmtsbMSNN96I7OzssE2uhFZaKBiTw645Runb1ESsubkZbW1tPEKA3W6H2+3mXWGUkXc42wmpqakiAM7U1FSkpKQgKSkJCQkJsFqtXDSWUCQ36ECg5qWumf5k09GYnnTSSWzbtm1BXgs5BWK32zF16lRs3LiRI3lCVu/GjRvZAw88wCfsuFwuEXyPbDLDYSUgbVcrd4iKQlUcVJyY3wOMISCIhRJ8EKXHUzxQ+o5qtRq9fb3YsGEDTjn5lGNXgdDkOJ1ONmXKFOzfvx8Wi0W294WwCjqaaxIjbNu2Dccffzwv+BsbG9mECRPgcDgiWh+UDXbOOefg888/P2IV0KGA68J1TSPq7e1lVBwmhN1uaGhAc1Mz2trb+DThcLDbquBdU0iBKGcxEVE71IyMDOTn52PMmDEoKirC+PHjMXLkyKCAfzgoE6kiiWX3eiRa2gq/QxXz0picsOOiNHZQVVXF9u7di127dmHv3r2oqqpCU1MTurq6ZFNqpfMibElMsZVQRKm4hPKcmpqK9PR0PsZGcba0tDQkJycjPj6ei9bleKSqz6PZcGq1Wrz++uvsqquuimoDSFbIZ599hlmzZnFkiWk0Gr5tNXUelUNfiDVmKP1fpLiidPyEyofWh3Reu7u7cd111+Hf//73z/1PjlUFQkypUqnwww8/sOnTp8Pr9UbM9Y5modNi2759O0aNGsV5vV5otVrcdddd7J///GdUzEcNjdasWYMZM2bEZH2EYq5Y3UyHc+9ZS0sL6urqUFNTwx9Uid7Z2Ym+vj5ZxSvsKiftaS5VDKGAGaWMLtx5CRcaCRiv18srcKFiycrKwtixYzFlyhSceOKJKC0tRUpKSlDAWTomwp1xtLvXI6FAaHylz0eLXKg0XC4Xdu3axTZt2oRNmzahrKwMdXV1sNvtIuFGUN3SvuWhDrn4l7QZmbRXfajOgVQrRJZkZmYmnw6em5uLnJwcUfW5ECU5nBUTLiYjp4xjcZ9xHAen08lKSkpQXV0dFt5EuAmcPn061q1bxwldqk1NTWzixIlwOBwRq9wHStF4GUL9TedrNBr09PTghBNOwOrVq2EymbhjFkxRLrC9evVqdvHFF8NmsyE+Pl429TQaIi1NcQsS/IcOHWIlJSVwu90RLRpivBkzZmDt2rVRWx+hds3hqK+vj1EfcaGSqK2tFfUTl6sFEQogYeEXxRQIEDFSYSbtekIpG6FQEl5TjqgYTYjOS+nKtNPmOA7Z2dk8XP8ZZ5yBUaNGiQLRA4E4PxIKRC5mQ+1bAaCrq4tt2LCBT0uuqKjgaykILYAq8UmwSxWtHB+Gmgsh5Hc4wSXtHijtfx6q6yUAUVMzsmAIQ0vgrkRSUhKsVis3kFYF0WwMyAp54okn2D333NOvjSCtA7VajQcffJD9/e9/R2JiIjwez5DqWijcpJF7beY5M7HyzZVISkr6H2T9sa5AhEpk586d7Oabb8aWLVsA/FxgFkswOBAIwG63IyUlBZu3bMbIESM5Yrpo22QKTd/PP/8c5557btSwJbQ4XS4Xenp6mMvlgt/vh9Pp/BmupLkJDfU/+7upcKulpYW3IqTChCrRhbvUaFreAoDBaITFbEZ8fDwSExORnJzM15UkJyfzPvK4uDi+Ep3gM+h+QqElhDAhyBJqKkRgjg0NDXzVc1dXl+h9CCJFrVbjcPtg3tcbHx+PKVOm4IILLsDvf/97ZGVlcULroz9xpyOhQGjsaYMQCATwzTffsLfffhurV69GbW0tb42YTCbePULQI0IsqISEBF4oZ2VlITMzE+np6eRSgsVqhclo5JWxFHaE/OVUvEhV6J2dnejo6EB7ezsf5+rq6uLjXHKN24Q90OVwuggSR26zQFYMNURLT0/nEQ/S09P5mIzFYuE3KBqtBga9ARaLhbdsonVRU4FucXExenp6IvZNl7qiab7J1T179my2bt06GI1G0YYglhjPYFojxGOUKAT8XO5w++234/bbb+dobogfFAUiUSJ+vx+vvvoqW758OXbt2hWyYEmOjEYjSktL8cwzz2Dy5Mm84K+qqmKlpaW87zoSw9ntdkybNg3ffvstF42ZTQJr06ZN7Mknn8S+ffvQ3d0Nj8fDu3WcTqdsQIzgTWjxCgV2qMUrdD+kpqby7gfCJqJ2tykpKT8LI0lG2pH2Vbe1tbFDhw7hwIEDKCsrw86dO3HgwAE0NjbyAsxsNkOv1/OJCpQskZ6ejjlz5uCaa67BSSedxGc5xeJnPxIKROgn7+rqYitXrsSKFSvw448/8mmVBDfj8XhE3TTT0tIwevRoTJw4ERMnTsTYsWORl5eH1NRUTq/XH9H58Hq96O3tZZ2dnTzmWn19PY/c3NDQgJaWFnR0dKC3t1eW16SbGGH1udBNFsrK1ev1PNgmWbuEFZaTk4Obb74Zc+fOjcrSJ8vvzjvvZE8//XS/kmGESL29vb3s7rvvxrvvvouurq4hIQtNJhMyMzNRUlICwoejdgjSokhFgcgsUqLq6mpWW1uLrq4uuN1uUU8K/lBxUKt+hqvOHZaLcWPHcUJ3klqtxo033cSWvfRSTNbHf//7X5x//vkRG0bRM9XX17PjjjsO7e3tMBgMot2c0ByVsyDkdoUEXkcKgirRpeio0QRAw0Fk9NcfLd2JRSqU6+joYLt378amTZuwfv16bN++na8+NxqNfK2H0+mE0+mEVqvFKaecgttvvx1zDjcEikXYD6YCoet5vV4sXbqULVmyBNXV1VRlD41GQ4KaV47FxcU47bTTcOqpp6K0tBSZmZlcNMHpaOckUgvcWGo3HA4H6+jo+BmGpL4etYfdqJSQ0draiq6uLvT19QXdS9jkSorGHC7GRgkUHo8HWVlZOHDgACwWS1QZdiqVChUVFay0tDSq+aPU1/POOw8ffPABJ2w4Rec2NDSw3bt3o6mpiU86oeNwmi/cHjc8Hi+8Hg+85BqWS/kVjI0kDR66w03ATEYjn+hA4KVxcXFITU3lY1Fms5mTbrCD5llRIPLCbiDAecKga3l5OZs0aVJUZicVIJWUlGDLli1cNAV29Kw7d+5kJSUlMBqNfE1FqHsYDAYeeC4lJSXIvywNYA60Ev2XhHEQpmwKXTbSd6ivr2fr16/HJ59+gvXfrEdzczM4joPVaoVWq+UFMmMMl112GZ5//nnekvols7DoWgcPHmRXX301Nm/ezONBcRwHu90Oj8cDg8GAKVOmYO7cuZg5c6aoT7ow5iAXCP8l5iNUi9polEx3dzePn0XZfkLrpb29Hd3d3Xz1eTRuHHKZOZ1O5OXlYc+ePTAajVy0UEEajQaXX345W7lyZcwFwccddxzf0XIw5M2R9MrQJjjUmCgKJIqgdCS/ozRLSKixr7nmGrZixYqYrI8333wTl112WdTtaokJ7733Xvbhhx/CYrHwcCSJiYlISkri4w90JCUlIT4+HiaTKSosq/5Uog/FjYFcU6ampib22Wef4e2338bGjRvhdrt5RcIYQ1dXFy677DK8/vrrXLTFmYOhQEiYtbS0sOnTp+PAgQMi4ES3243hw4fjkksuwbx581BaWsoJzxUmAwzleQqXSh7Ns9vtdr7eiGIvBG1C8b34+Hj4fD60tLSgoaEBbW1tfFzooYcewrXXXht1piMpkM2bN7NTTz0Ver0+qra6NpsN8+fPx1tvvcVJd/TRgFiG2rxEm/IbTQwl1rWtKJAjpLlVKhV2797NpkyZEpUZT610x44di23btvHBvWgXPn3P4/HwZn0sgnWoWBC/tEKhxU30/fffs3/961/4z3/+A4/HA4vFAr/fD4vFgqqqqqjdHINZib5mzRp21llnwWKxwOFwIBAIICcnB7feeituuOEG3j9NVkYssB9Hy1zJpXdHq2Dk6HACADMajZzRaIw5ME1zfOZZZ7F1a9fyfBLNed9//z2Kioq4oWp5xEIqKHREiOM4PProo3C5XFEJc5VKBZ/Ph9tvv50HWYy1+pa6omk0mqAMFgo0CmEShHnpFEgXNvs5miyM/swP+cyFnRCnTJnCvf7669zatWtx0UUXwWQyweVy4brrroPFYuFinZeBED3bySefzN288GYYDAaUlJTg0UcfxY8//oh77rmHS0xM5CiORWmzvyXlIbTsKdVbmvQhTCuW8jzxvfD3QCAAg8GApKQkjly+/XVT3/rHP0atfDQaDVwuF5566qnfzLpSLJAjYH2o1Wps376dnXDCCWG7mEmtj5EjR2L79u0wGAxR+drDmau/VcH/S7ktSenX19czm82GcePGRb1WBjuITp+3t7ezpKQkTtirfai7p4ayVROtCyjSpm3KlCls165dIsjzcOeoVCr8+OOPGDVqFDfUARUVC+RXokceeQQejycqE5Wsj1tvvRUmk2lAu9yhaDXIQZFQCma0hxTG5EhtfGinS/fNycnhxo0bxx3JSuFoBVVKSgqnUqn43uWxAlb2x2Uk7V0f7gjlahqqVs1A1wr13Vm4cCGiLfbVarWw2+345z//KerTrlggCvHWx5YtW9gpp5wSVXCN0H1zcnKwc+dOWCyWo6YNZjToqEdyhyyXHjzYAf7+wG4fyULCgVqX4eBiBnu+wkG+Szc7R6MVReu0t7eXTZgwAY2NjXwDqXDrnYpBd+zYgeHDhx/VVsgx3VDqSOxqAODhhx/muwhG8q9Sfv8f/vAHWK3WqDOvfmnrYSD9GShg6XA4QAc1RKJ+DUIId4pPECwJ9eSgPuaHD46QAiLt3OXqd45Gi64/kCfCTMJoADJpvpxOJ+PrEDweeA9XJgsz2YQ1GDqdjtoLE6IAFwsfhwJLHMoKhvqmx8XFcQsWLGAPPfQQjEZj2GxLxhi0Wi1sNhueeeYZLFmyBL+mdatYIEPM+li/fj07/fTTI8I9EwN6vV6kpaWhrKwMCQkJv4r1IbQkhDUskZSEx+Phe1w3NzejqamJhxKh6uKuri709PT8D679cIc4EkbRLlRSKFRxTXUsycnJyMjIQFZWFt+TnVruJicnyw4ktXU9UhbSkcbCimbHL1f7AvxcKNnS0sKoZSy1i21ububrKXp7e+FwOHgMJJ/PB3/ADxYQ9xQR8ggpElIgBLsfFxeHxMREpKSk8NDv6enpPPz74XTyiGCJwlTyoaRcqCgwFqRtwiMzGAwoKytDTk7OUWuFKBbIYFsf/+/hqHsgE2zJTTfdBMqmOZLWh5w1Id2VSneodrudtba2oq6uDrW1taiqqkJNTQ3q6urQ1NSE9vZ2vkteuPcUZnbRjjWaxS91kxH+UltbW0gwv8MZNsjOzmYFBQUYP348iouLCcIjqCkVKZSjMf1ViNclnbv29nZWXl6O3bt3Y9euXThw4ABqa2vR0tKC3t5eWQUuzHgSKli1Sg1OzYWcH6/XC7fbzV9XmhouJZVKBaPRiLi4OCQlJ7H0tHS+ApoQD7Kzs5GRkYGkpCRO7v2GgnKhXj/Z2dncJZdcwl544YWINV+MMR50dcmSJXjiiSeOWitEsUAG0fqIpWMZmb8JCQnYtWsXUlJSBsX66I/Lye/3o62tjdXX16OqqgoHDx5ERUUFqqurUV9fj7a2Nr4qW6oYhGm/wuuH60kwGGBwoXpBCJGApdX4VqsVw4YNw4QJEzB16lRMnToVRUVFIqyuwailONIWCAlnKfheXV0d++GHH7Bp0yb88MMPOHDgAFpbW0VuVLLk6FwhYOVA5itSr3o5GH+/oKGR3HoxGAxISEhAWloasrOzkZ+fjxEjRmD48OHIy8tDVlYWUlNTw1ovct0QB7sIlqyQPXv2sMmTJ0fFN2SFmM1m7Nq1CxkZGUelFaJYIINkfTDG8PDDD0fNmGR9XHfddUhNTY26CnYgfaP7+vpYc3MzamtrUVFRgfLyclRUVKCmpuZnBNvOTnglOydyTZjN5iAFIZetE06w9KcBUChBFklBk5UjxEXy+XzYt28f9uzZg7feegsajQZ5eXmYPHkyO/3003Haaadh1KhRnBRUcihYJjTGZB2Q8tm+fTv7cvVqrF2zBjt37kRHRwd/DsWPiK+k8xVJEUTb2zxcU6NIVgi1BZDegxRMZ2cnWltbUVZWFjS/8fHxSEtLY7m5ucjPz8fIkSMxYsQIXrmkpKSEbYE8WHEX6udRVFTEnXvuueyDDz6I2DedrJCOjg48++yzeOSRR45KK0SxQAbJ+vj444/Z3LlzIzKO0PowGo0oKytDVlaWyPoYSIaMy+VCW1sba2hoQE1NDSoqKlBZWYmqqiqRNSFdAFI4baHQCrULDbXgIqWBxspzch3TpB3cQllfoZ6XdoBut5t3N1itVkycOBHnnHMOZs2aJYIGidR3/EhZIDRmwl12WVkZ++CDD/Dpp5+irKyMt7SEPT+kAXSpUuDhuGX6fPRnvqRzJNf4S84KkeP3cHMmfFYqEJQKXo1Wg4T4n4FAc3JyMGzYMAwfPhz5+fkYNmwY3w3RarVykdZ2qF7u0vGkFtlff/01O/PMM2XbwYaSA/Hx8di1axdSU1OPOitEUSCDYL4yxnDCCSew7du3w2QyRVQghIvz4IMP4qGHHorJjqY+40Jo7NraWtTW1vIB0ba2NthstqDnEKKWhupAF+0CFrofQr0vNTIitFs6CP1T2hFP2gfC4/HwjaCcTiefvSUM8Ia6L1UrCzsWhkIDpnsT7D3wMwT4pEmTcP755+O8885DQUEBP09erzdsDcZgKBBpQWNPTw/78MMP8cYbb+C7776D0+kEx3EwmUwiUD45tFqpFRayS6BGDaPByGe8GY1GGIxGGA7DoZO7Uihg6XqEGCtEkaUsu3DxgEhu0P7wphBtWsqbBJiZlJSEzMxMXrmMGDEC+fn5yMnJQVpaGhITE7n+CvKZM2eyr776Kip4E5IFDz30EB588EGOFJGiQI4h6+Odd95h8+fPj8r6EPo/ly5diqlTp8JkMvGQ3H19fejs7OSbJDU1NaG1tRVtbW1ob29HR0cH32fc7XbLLkipUI5kTYRTEiQgpOdQ//Hk5GSkZ6QjK/N/jYkyMjKEGTawWCykODhhc6JYyOv1wuV2MafDib6+PvT09PDuDWGPCWqU1d7eLmrhSs9M95cTuMJdtM/n47swxsfH4/TTT8eVV16Jc889lzMYDPyuU65z4UAUiLRpVHV1NXvllVfw5sqVqK6qAgCeX+TchkKFSJ0YhTxpNpuRnp6OnJwc5OXnIz8vTwTPn5iYyDf40uv1nE6ni2q+GGMEF8JI4dvtdlGjqfb2drS2tqKlpQUtLS1obW1Fe3s733BKLhGDrGMhVPtAlIuUr6VEbQzS0tKQkZHB92/PyMhAWno6kgU8TRafsJGbx+PBy6+8jCcefwJmszlqb0RycjLKysqQlJTUbxQKRYEcRUTM6PP5MGXKFLZnz56IPZLl3E3kpxa6VOQUg3DHQotJuqDkYiORFpXQFSBldspmCrVTS09PR2JiYkxNieQQjiP1CIk16NnT08MaGxtRXV2Nffv2YdfuXdi7Zy+qqqqCYgTUVEpOGNOOWGiZFBUV4aqrrsIVV1zB99iQKpL+KBASbOSqOnDgAPvXv/6Ft956C52dnbw1RxsXueek7oMkGPV6PfLy8lBUVITS0lJMmDABo0aNQnZ2dkT3jTTWFI6n5NyZ0ZLT6UR3dzffVpng2oUpxm1tbejp6QlaF0JY9liVi9wzEw+E2jRJ42tanQ4aQadOoUs0FpBEskIeffRR/OUvfzmqrBBFgfSTaJJfffVVtmDBgqjg2uUYWVp7IfUby5nz4WIScguD33V5ffB4xW4fTqVCQnw80tPTMWzYMIwcORKFhYUoKChAXn4esjJ/rqcIJxiEDW2kgj+Uz7g/Clv6u1ysKFT9A2Ps56Y9e3Zj65at2Lx5M3bu3Inm5mbecqOOftL0YKFycDgc8Pv9yMjIwOWXX46FCxdi5MiRvCIhYRKLAhFaHLW1teypp57CihUrYLPZeFef3DPR+U6nk1cq+fn5mDp1KqZPn46pU6di9OjRspD90jkLleAQK6Cn3N/hXIeRBK3H40FnZydrbm5GXV0dqqurUVVVherqatTV1fHti6XWSzTKJZrYXqg1KOcS7a8SJfdpeno6ysrKEB8ff9RYIYoCGYD14Xa7cdxxx7GKioqYrY9QO+5wwcRQglkYAA1lmhuNRiQnJyMrKwv5+fkoKChAYWEhRo4ciWHDhiEtLY2jrnxywkZOgIVK0TziTBtl32ihcJRTLM3Nzez777/H6tWr8fXXX2Pfvn18pXCk3T5ZiomJibjiiitw++23Y8SIERwJPZ1OF1GBkKvsMD4Se+aZZ7B48WK0tbXBaDRCq9UGjbv0/gAwZswYnD1zJmbPmoUpU6YgISGBk7r/pNX4kYT/YM5nLLU+9FNajxLKOuro6GBNTU04dOgQqqqqUFlZGaRc5Cx6qXKRS2CJZrM2WGNGVsiTTz6Ju+6666ixQhQFMgDr4/nnn2e33HJLv6yPcIpBKgSlAISh5sxisSAxMfF/Pu68PD5vftiwYcjMzERSUpIsxARdXy4t+GhFe2WMIcAY2OHdoiCriKnVapHrzel0YsuWLXx2U3V1NR8zIKtEOm9qtRoejwculwuJiYlYuHAh/vKXv/Cw7/X19ay0tFRWgWzbto1PMf3444/Zfffdhz179vBwINJ5pp06tXVNT0/HrFmzMG/ePJxyyim8lUEbmwBjTMVxnFwm1FCH6ZciI0h3/cLxl9bCCC2Xjo4O1tzcjEOHDqGmpgZV1dWoralBfX29qE2uHAndxFKPQKh1Gk3zuXBywOPxIDs7G2VlZVF3v1QUyFEolA4vZFZSUoK6ujro9fqYrA/hd4W9KOSI8uSpfzG1oR02LBfp6Rl8D+PRo0cjMzMTKSkp3GA0qaEMFp/PxyijRdhH/X9/e+H1ij/3+/3w+X3w+/6HpPs/l0kAgQAL6dYQpYOqVVCrDi9kjRoatSaoJwTf6/lwRpf+cMbQz/ENHafRaKOaD6F10t7ezj7/4gusePVVrF+/ni/4ojiDnCuJFElxcTGeeuopnHXWWVxzczMbO3YsXC6XSIGMHz8eZWVlXEtLC7vnnnvw2muvQa1Sw2wxh3SfUep1SUkJrrrqKlx00UXIzc3lhII22nn3eDzwer3s8E8erkTaMyYSIrLQWhA+K6dSQS1IuaZD2G9GOH/S/2k0mrCV54NBfX19rLW1FY2Njairqwtqk9vW1iZqkxttKrMwFT5WRU1WyOLFi7Fo0aKjwgpRFEg/rY+nnnqK3X333f2yPqhvAKVhJiUlISsr63B+eiqSkpKRlJSEhIQEJCUlwWq1IjU1FWazCVqtDgaDnmMMPOAdpVE6nU4epPDnlFcHHA4nnC4nXE4XXC4nnE4X/z36rsvlhMvlhsvlhsfjPgxw6IbH4w3ZkEpOqAQEO/3BhvSWKhehi0OtVkGt1oiA/Sh9mBRvQkI8EhN/bu1LeEwZGRlIT09HSkoyzGaLbNHZ+vXr2eLFi/Hxxx/z+EVarTakIiEL4bbbbsM111yDM888E729vfxO2W6347jjjsPf/vY33HbbbaitrYXVapUt7qOe3V6vFyeddBIWLVqECy64QNaCdDgcjLKcmpub0dzczGfvdXZ2oru7GzabDX12O5yH06CFuGTSRmNBO38wgMVWlS43X8JDqlikG4L/bQT+B9LIH0YDn3JMB6UeC1PF6VzhNQTfDVvF7nK5YLPZWG9vL/r6+tDX1we73Y7e3l709vaC2uhSFmBTUxPvMvP7/TxIaLRErslhw4Zh586dvFU5lK0QRYH0w/ro7u5mEyZMQGtra1QNo4QLy+/3Y9KkSSguLkZ8fBySkpJhtVp55unt7YXNZoMQudblcsFut4t+t9lsvHAJBPzw+wdWrBdOYIdyf4SrVO5P7+ZofMmh/NOhKuPDjYNKpYLZbEZCQgJv1RUXT8CYMWN+jg8VFCD5cFrlrl272FNPPYUPPviAD25TjEJ6TcYY7HY7UlNTeSEt9LHr9Xr02fvAAgxmszloA6JWq+Hz+eB0OpGbm4unnnoKF198MUdCrbKyklVUVGD//v3Yv38/qqur0djYiPb2dvT19cnGwOTiONLi1HBumn4JMcbAIsydXEHhYG5ApIqKLB5SVEJlRUCdpGAI/TmUYqLaGMrU6+vrg81mQ1VVFX748QfUVNeExYgLZ4U899xzWLhw4ZC3QhQFEgNR3cfDDz/MHnjggX7HPrxeL7+I/X5/2JTBoAkDB07FhazGDifgYxXS4b4Ty+dHlIFDvFs4SA6hsKKaAAJUJNLpdEhPT8eIESMwceJEXHjhhTjppJO4Q4cOsRdeeAHLli1DV1cXTCZTSB869aeX85kLCydlrAlYLBbcdtttWLhwITweD95//31s3LgRBw4cQH19fZDvXq4dcTSK99eYy3D8GM3Goz9BebnjSDTBEgblQ8VnIik8l8uFESNGYMeOHXzN0VC1QhQFEkPcguM4tLW1seLiYvT09ECj0fQ7YCb0H0fjK41GqCtz2X9hJpwHEi4UIyAqKirC7NmzMWvWLKjVavzr2X/how8/4nfycjGdWOeEMYY5c+bg0ksvRWdnJ1auXIlvv/2Wr7pXq9X8zjfaugeF+q+cYsVsG2gwXWiFLFu2DNdff/2QtkIUBRKj9XHvvfeyf/zjHwPKvFLo6BE4QkvB4XDwFsPxxx+PWbNmoaOjAytWrMBA2hDTvdxuNwoLC3Haaafhq6++QmVlJYD/VZ4riuLYILJCRo0ahe3bt3PRtj5QFMgQJbIUmpubWXFxMR8UVcbu2FvYlIlFLqSUlBS4XK4BKxChIrHb7VCr1TCZTGGRcxX67RJZIa+88goWLFgwZK0QlTJVkYkW8O7du9He3h6x77FCv10+oJRri8UCq9WK3t7eQVMetFkhnCVhoySFjs1N66effjpkrQ9A6QcSE1mtVn4nGgnmPBJzHAs0GEw/VMeK5v9I1CrEqjSO1kLPo3n+f4m1w3Ecj4gwZC0lRS1EJsIcKi0t5U477TS2fv166PX6oApVYRMdUaEVOICLLt01GmFA14tuBQLiZMrBWbjRfD5Yi3+wgp+hMpNCfX8oxRzCQZcLU4kH6zkHoztmVLw8CN8ZrPsMJcVMlu7s2bOHtCJVYiAxLqiWlhZ27733Ys2aNbDZbPB4PCIwRGHlLaX0UbpouII7OWH1S81NtF3ooqkBiaomhONk9V+4znaRUjKPpKCnegGKew1GnU00RDEXgkoX8prw2agnCD/WgvEdzLqOSJl/Uc3fz38Ifh4uUJT+P0Rr3XAtd6NpyztYrZWPNMXFxeHqq6/GP//5Ty4cHpiiQI5CJQL8DBne0dEBp9PJZ2NRsRKlWVKRkRDHSq6SW/ozFiUTaSctVyMS9lBxUHEq2crv/hwhhdhhIRdKKMkpjIAE00qKEeYPBOD/HwTLz/AcPh98hyvp/QE/wIIrpEnBe71eOBwO2Gw2tLe3o6mpiYe5aGpq4kH5qKo5VBvfwbB4GWOirK+EhATk5eWhoKAABQUFGD58OHJzc5GRkYHk5OSgHuexFnnKKZdYaoQiCfRoiwjD/e9IHDxWGgT/D8TwLAKFF+l9hQpU4CAQ/80Y4uLiUFRUxEPVDGlXm6JAYlciseAOKXT0U29vL6usrMT333+Pr776Chs2bEBrayuAn8EW1Wp1WJDLaK1AqkCnRlZFRUU4/fTTMWPGDJSUlGDYsGHc0dTuVKGBkRSjTVEgvzFFEmnsYoE5H8rzMNSDtEdqbEP1oW9sbGSffvop3nzzTWzatAk+n6/fVokUjNFsNuN3v/sdrr32Wpx22mlBzbqkiMmD0Wvl15yb39p9B+NZhBayYoEopNBvZLNAO0Lhwv7uu+/Y8uXL8dFHH6G9vV1klYSKywgr18kFmpiYiPnz5+OWW25BUVERrwmo2+HRAMOu0LFHigJRSKF+KBRCJiCBXl9fz95991385z//wU8//cTHSyguRkpHin02cuRIzJ8/H9deey3fkIqyqo7mXiwKKQpEIYUUikAU2BfGxLZv387Wrl2L7777DgcOHEB7ezvcbjcP35+VlYVJkybh3HPPxZlnnsn3J6eCRCXOoZCiQBRS6Bi0SqRwE9TT2263Q6VSwWq1IiUlRWRWKIpDIUWBKKSQQrxVIhcvkSoNQHFTKaQoEIUUUiiMZSL8qQTCFVIUiEIKKaSQQsc8KU5XhRRSSCGFFAWikEIKKaSQokAUUkghhRRSFIhCCimkkEKKAlFIIYUUUkghRYEopJBCCimkKBCFFFJIIYUUBaKQQgoppJCiQBRSSCGFFPoNk0YZAgy4l/ZgNPUJ1YRoKEFfhBqnowWeI9SzK6RQf+XDsQ5No0CZHAFmE6KrKgLq1yUCNpSi5CqkkEKKAhkUAdPS0sJiaUMqtwvR6/WwWCyyLUgj9U/3+XxoaWlh0msyxmAymZCYmDgktFBLSwvz+XxB/9dqtUhLS+OG4twK0XB7enoY9RvX6/UwmUycXq9XlPwxbkVEgtF3OBysq6uLX5PC9ZmYmAiTyXTMMtAxq0AYY+A4Dj09PWz8+PHo7OwUMUh/FEhcXByys7MxYcIEnHHGGZg5cyasVitH9wol4KqqqtjEiRN5y4UxBo1GA5/Ph8svvxzLli3jfD7fr7KLpmf3+/0oKSlhlZWVUKlU/LMHAgGMHz8e27Zt44bi/H7//ffs9ddfx6ZNm1BXVweHwwHGGMxmMwwGA+644w7ccccdXDSKXqFji2jNvfHGG+zGG2/k1yQA/vd///vfuOyyy3619flr0zFv1zPG4HA44HQ6B3Qdh8OBrq4u1NbWYtOmTXjhhReQn5+Pe+65hy1cuJAT7lzknqGvrw9yFojL5RoyYxVqnAY6dkdip8lxHG6//Xa2ePHikO8CAE1NTfx5Cv12yWazse7ubtnPUlNTOaPRGFaROJ1OWQtEziI/lkjJwgL43tYUs+jvoVKp+B7YarUaNTU1uOWWW3DttdcysjhCCSqNRiN6Bvp7KO2KaZzkfg41t9VNN93EFi9eLOpJLpwnnU7H/1Tot21FAMDy5ctRWFiIsWPHorCwEIWFhRgzZgwKCwvxzTffMOB/Tb7kPAzCNSn9/VgmJbIIsX9UKuCjCYQzxmSVg0qlgkajwfLly1FQUMDuu+8+WVeJUAjTzkatVoMxNqTanKrVaqjVan5MhD+HAtHYfvzxx2zZsmXQarXw+XxBzZwCgQA8Ho9ieRxjisTj8cDn8/EZj7TWQikO4Tom3id+GYrrU7FAhiAFAgH4/f6wRyAQkBWkgUAAXq8XarUajzzyCBobG5larQ5K2fX5fHC73fD5fPB6vfD5fHC5XPD5fOjp6RkyY9He3i5aiPSzs7NzSDwfKfonnniCVxRCBUGK3mQyITk5GQaDQQmgHyNEmweymqW/hyOn0ylak8Lfh5L7VrFAhhjDMcYwevRoDB8+HHKBcIpd1NfXo7a2lv+OVGhxHAeHw4GPPvoICxcu5N0sdL3U1FT8/e9/5xUR7WwCgQAmTpzI74J+TaGsUqnwwAMPiJIN6Gd6evqQcV01NjayH3/8kVcWwnfQ6/V47rnnMHPmTGi1WrjdbhgMBn5HqdCx4WkQehvCWaC05qZOnYqHHnqIX5P0WSAQwJQpU37V9akokCFKarUaPp8Pt9xyCxYtWsRF2qFs2rSJ3XXXXdi5c6eI0YSCeOvWrVi4cGGQcE5MTOT+9re/RTSjY10s0h24NGYTqyK59dZbuSO5uOWeN9paGjqvtrY2KOBJc3neeedhwYIFXDhFGeszxvqcgyEAhYqxP4pPep1Y3mGg8xSLkJcb5/7yb7/cM4fX3MSJEznayA3G+gz1joP5ftGM42AoPUWBRCCXywW/3w+5ND2aCKPRiDPOOIP76quv2MSJE9Hc3CwSYDSJlPEjnTjGGO+Tl1Nk0aYHkj83muA2+X1jEUAejydkNbdcMDqa6noSRuRjDmVd0HfCCUOy9IQWpJBGjx4Nr9cLv9/PjymNVSSBKUySCGcFkfUYrQCIZozoGYTJHqHGIZxwi+Y6wiJYuXEIN0/94SnhtaOZD+m9Qo21kC/CJa/QHNBB15LyqNfrlT1Xq9VGLYiFRa2R3tHn80GlUsUs5IVzEM049vc+igKJYedNAbNQC4MUQGpqKnfJJZewxYsXi3LGhQI41D2kBYixEikOEozNzc2ssrISjY2N6O3thVqtRmJiInJzc1FQUACr1cpJ3T+RKNaMpUjXFCYUdHd3s927d6O2thZutxtmsxlZWVkoLCxERkYGJ/1+qN2U2WwOeT+DwQCtVgutVhvTmNI9Ozo6WGVlJerr6/nYVHx8PHJycjBy5EgkJydzcu82kDGiuSEe3LlzJysvL4fdbsfo0aNx4oknclKBF4pHiZcDgQB2797NKioq0NPTA71ej5ycHIwbNw4pKSlBYy0ch+7ublZWVoa6ujp4PB5YrVbk5ORg1KhRSEpK4mJ5d7kx7uvrY4cOHUJdXR3a29vhdDqhVqthsViQnJyMrKws5Obmwmw2h+VfGg/i2XDry2QyhRWiKpVqQOtTqHxVKhX8fj8qKytZdXU12tra4PF4YDQakZ6ejuHDh2P48OEcrWO/3x+1ZSiUUR6PB7W1tay+vh5tbW2w2+0AAKvVioyMDOTl5SE3N1d0n/5Ys4oCGUQlEwgEkJubG/J7CQkJIncLLeqmpiZ27bXXimIgarUafr8fZ555Jv70pz+FLHQTMk5PTw9788038Z///Ac//fQTbDab7HNkZmbi5JNPZldeeSXmzJnDEVOHuj4FpK+77jrW2NgYFAMZPnw4XnjhBU74bhzH4Y9//CM7ePAg79IjV9KcOXPwhz/8gVOr1di9ezd75pln8Nlnn6G5uTno/vHx8Tj11FPZHXfcgRkzZnDCAkaVSoUffviB3XffffyzUEBfuOOkXdm///1vrFu3jgkFzrJly5CXlxdU7Enj4fF48N5777GVK1di69ataG9vlx3T5ORkTJkyhV166aW45JJLOL1eH3FMfT4frr76atbW1hYU+1qwYAHmz5/P0d8vvvgie+GFF1BWVsZfJy8vD5WVlVCr1di6dSt74IEH+PPpejqdDsuWLUNGRgbn9/vxwgsvsBdffBG7du2SfYezzjqL/elPf8Jxxx3HCXezbW1t7NFHH8Xbb78tO08pKSk47bTT2KJFi3Dqqady0WxKaHxcLhfef/999t5772Hbtm1obGwMiz2VnZ2NqVOnsksvvRQXXHABJ+QHuuYrr7zC3n77bX4camtrRbwgtP7uvvtupKamMhp/xhiefvppjB8/ngOAr776ij355JP8mqQx8fv9uOeee3DGGWeEXJ/CDcD+/fvZyy+/jM8++wwHDx6UrSHR6/UYPXo0O/fcc3HVVVdh3LhxXKRNnjCmum7dOvb6669jw4YNOHToUMg6FaPRiFGjRrFZs2ZhwYIFKCwsDFurFrWv7Fg5yKzt6upiycnJDADjOI4BYACYRqNhANjjjz/OGGPwer1hr+fxeBAIBHD//feLzhf+/re//U10Lb/fD8YYysvL+e9Kj4svvjjk/YXm+SuvvMKGDRsmOpfjOKZWq5lGo2EajYap1eqg65900knsu+++Y1QUFWqcfD4fMjIyZJ9x+PDhTO6ZxowZI/v9q666ih1epEyv18s+r1qtZiqVSnTefffdxz8njcenn34acuyiOXbt2sWEcyEch88++4xNmDAh4pgK+QYAGzduHPvggw/469J4SMfU7XYjKSlJ9rnoXaurq9kpp5wi+kyr1TKNRsPGjRvHaBw++OCDkO/Y1tbGmpqagq4Taqy1Wi174403+Dldv349y8vL4z9XqVSic6XvL5ynUOuFPvvoo4/Y2LFjg55ZpVLxYxyKHwCwE088ke3evZsfaxqPO++8c0B8sX79ev79ly1bFvJ7L7/8csj1STzlcDhw9913M4PBEPId5d5Pr9ezRYsWsd7e3iAeld6jsbGRzZ07t1/jaDKZ2P/93/+F5Ndwh6JAIiiQxx57jHm9XjidTni9XtmDYgOMMUyaNIlfnNJr/fTTTyJGoJ8VFRVMp9OJFqVer2dqtZpdc801sgxK6cU+nw8LFiwQ3YsWNQk7OlQqVdD/6VmfffbZkPehBT969GimVquZVqsV/SwpKZFVIFOmTGFqtZp/N3qnv/3tb+yNN97gx5yuQ88nJ7DpWZ944glGwpcxhtWrVzOdTseMRiPT6XQixS096FmEx969e0VzQoLtwQcfFJ0nHFMSoOHGFAC75557eEEqXJRCBTJ8+HDRWNIYPfHEE8xms7Hhw4fzQp0WPv0cM2YMEypS4VgTHyQmJrIdO3aw8ePH89cJN9Y0fiqVipWXl7NNmzbxSj6ac+n9H3744ZBKhP73+OOPy/KtdBzl5oDeEQBLSUnhlYjL5QJjDPfffz/T6XTMZDLxYxKKLzQaDc8PBoOB6XQ6flPFGMOKFStEcyP8/fXXX5ddN8RP9fX1bMqUKaJ7yfGMlL+EfFxSUsJqa2uDlAgJ+8bGRlZYWCjaFKhUKv46wrmiz+Xuc+WVV8ryq6JABqBA/vWvf7Forudyufhdj1DD63Q6BoBdffXVQQtKqECIwekZ6P60WxcyaCAQ4K9z8cUXBwkYqQKTLnTp9+jvJUuWBD2jUIEUFBSI3o9+FhcXyyoQqTKl70+ePJlZrVZecMoJeqmAIqGt1+tZTU0No3sMpgVC733vvfcGjY10XuV2etJdHwC2aNGikGPqdrtBViOdT+c9/fTT7KqrruLnVjh3JCDGjh3LK5BPPvlENNY0fnFxcbxw0el0QeMqxyf0v+OOO45lZWXxzyB3bihFolKp2J49e0Jad2+99ZZIMYQaY+nf0vvR2EyaNIkJN3ODaYG8+uqrIb0Kr732muz69Pv96Ojo4K0r6fiFWp/C/9PmCgAbO3Ys6+7uZkILgX4/++yzRbIm1oPjOP7cRx99NKL1KDyUGEiE7Jj169fDaDQyOR8kua4qKirw+eef48CBA0EpvB6PB6eeeiqWLl0adbA6mmdTq9V4+OGH2bvvvgudTicK0JMvWKPRoLS0FNnZ2fB6vaioqMCBAwdEvnbKutFoNFi0aBFKSkrYKaecckTABWlctm3bJvKD63Q6pKenQ6VSobW1VbY4i7JX3G43VqxYgQceeACMMaSmpuLkk0/m36enpyfIv09+8GHDhmHYsGGimh6LxcI/h1arxX/+8x/22GOPBVWxC+d1woQJyM/PBwDU1NSgrKxMBC5JC1yr1WLJkiUoKSlhCxYsiGpMyce+cuVKbN++XeTXJ0FMzxEOh4me22azwWazQaPR8DySkZEBnU6H1tZWWaw14ont27fz/n7KQkpPT4fRaERXV5dskSvdNxAIYOnSpSK+J77r6upiixYt4udFmJYcCAQwbdo0LFy4EBMmTIDJZILD4UBlZSVeeuklrF69WpRhR4W6P/74I7Zs2cJOPvlkjuq3hHzR0NCA6upqWTyrcePGISkpCcIYSGJiIkKBoEa7Pq+55hrs27cPWq1WlMVFc5qcnIyJEyciLi4OnZ2d2LVrF4Sov6SYtFot9u3bh9tuuw2vvvqqKA6zYcMGtnr1atH80vmpqan485//jBNOOAHx8fHwer2orKzERx99hDfeeEM0ZzSO/+///T9cffXVLCsrK6o4lmKBhLBA+nPQbol2iklJSeyuu+5ifX19TChYBmKB0K5j3759sj5ous55553H7wCFzPjFF1+wUaNGBe3u6Lxx48Yxt9vN32cwLRDpztVisbAnn3ySVVZWMgJqPHToEHvsscd4i0r6bhzHsVNOOSWkT3jjxo1B70Zj+cgjj7BQvBAIBNDZ2cnS09N58186r1OnTmWbNm0KusZhwSVriahUKpaQkMCam5uZMF00lAUSzrIxmUxs5MiRrLi4mI0YMYJNnjyZ0Y5baoFIXTQA2KxZs9jGjRsZwdrX1NSwv/71ryHvT1YfADZlyhT2zTffsJ6eHuZ0OtHS0sKWL1/OEhISeNeL8DyO49ioUaOYcGdOvy9fvpzfMdM9yM101llnsXC735kzZ8q6iDmOY08//XTIeMSSJUuCrAi6xurVq2X5gq4TiwVCz/6f//xHZCFJ73nXXXex1tZW0X2bm5vZn//8Z9n5oPN27NghctXdc889ItejcN7Wrl0b0nuycuVK3nVH8RGDwcDUajV7/vnno4r7Ki6sKBSINAgV6pAKDo7j2FlnncUOHjzIpApjIAqEGJTcG3IL4rLLLmNSd5dwUba0tLBRo0YFubPoWm+++SZ/z8FWIHRPnU7Hvvnmm5AM/thjj8meC4BlZmYyu93OhO4Cj8cDv9+PtWvXhlQgDz30EPP7/SAFSeNPY0v3lBvTE044gREUvHBMhcrg9NNPDxn/uv/++4PGNJwCIf84JSm89NJLrLa2llcYXq8XXV1dTJpMIKeshckYcsfNN98c9N5CPh4zZgyz2Wyy53/88cdBz0/zZDAYWH19fVCA+8ILLwy5CduyZQsvIIVwQS6XC4FAgE8WkBvjhx56SDTGwrl+8sknQ87tJ598woQ8JOWLWBQI8UZxcbFoDoX3+/vf/y6SCT6fTyQbyP1GmyhaLyqViv3pT38SKZALLrhA9Dw09haLhdntduZyuXiYJJoDela55AUA7He/+13UCkRxYcWQrRbKnJWmHJJJ/tVXX6GwsBBXXXUVe+qpp5CSksINxI1F6bptbW3sgw8+4Pt0CF0sWVlZePHFF3nmJGRgoUstLS2NW7ZsGZs+fbrsOy5btgyXXXbZEYFnoFTeK664Aqeddhrndruh1WpFRVsAcMstt+Dxxx+X7dPS0dGB1tZW5Ofni9wOkfLlqV5EmvNPz7R8+XLejSJ0qRgMBrz66qswGo28S0E4pl6vFzqdDq+++irGjRsHu90ucs9wHIfXXnsNf/3rX6OuJ6C5PfHEE/HRRx/x9Rl8/r1Gg4SEBC4cECA9f0JCApYuXco/qzD3n+M43HzzzXjxxReDQAXp/AcffBBWq5Vzu918XQW925w5c7iioiK2e/dungdprlwuFzo6OpCdnS2qeUlLS8O0adNEKceBQABxcXEYP348n34sLKSk76SlpcmuuVCFmDTX0fDFQMERydW4adMmtmvXLt5VJUz7nTZtGv72t79xcgV89N0HHngAL730kqjFA7mn1q5dKyrWpMJZYWmASqVCX18fVqxYIUK+oOuTbPjnP/+J/fv3B6V+UylCNC5sRYFEWT0a7aKXCiAAeO211/D9999j9erVLDc3t99KhBTC119/zRcHChWIz+fDtddeC4vFErLBjU6ng9/vx6mnnspNnTqVbdmyhb8OLf6tW7eisbGR94MOJmwEjc28efP4uIZUmAcCAVitVq60tJStXbuWX4i0SDweT8gal/48j0qlwu7du9mBAwdE80f3nT17NkaNGsX5fD7ZIkSKl+Tm5nIXXHABe+211/hCUrr+oUOH8OOPP7KTTjopYvMhWsjJycl4//33kZKSwgkFf7SNz0gxnnXWWUhNTQ16fqpWHjZsGBISEkT+d1JgJpMJ06dPDxLqQrDKoqIiCBWI8Bmlvn8AeO6557hwPE48LQQ8JMW1Zs2a2GsVfsFNJgB88sknvFKSyo67775bFFeTzlcgEEB8fDx38cUXs/Xr1/P/o/E0m818vAIAEhMTgxQkzd8tt9yCn376iV133XUoLS3lpIXA55xzDnfOOeeE5UNFgQygOJAxhuLiYowaNQrhBGlfXx+qqqpQUVEhClATo+h0Ouzfvx8XXXQRNm7cKIKF7g9t2bIlqPKYGPXss8+OGPwj8/Occ87hryW0cJxOJ3bs2IGsrKxBC/wLhY5Op8OYMWNkFxi9C8dxfKBa+C40L4MFx07v9/3334uErvC+s2bNiuo+jDHMmjULr732muj79J5bt27FSSedFPFaQhy2jIwMjqyeWBc3UVFRUdh76vV6PjAuLRLNzMxESkoKF0ppcRyHxMTEfglaYQBdCC0jpc7OTrZ371688cYbWLZsmcjyHkpE62TLli0iaBl63vj4eMyYMSMszBBZQi+//DIn7FAq5Q+qUD/rrLPw1ltvia4n/P6yZcuwbNkyFBYWsuOPPx6TJ0/G8ccfj3HjxomQE8jtF2uPH0WBRFjEN9xwQ1Qggh6PBxs3bmT33nsvtm3bJhKOHo8HWq0W33//PZYvX85uvPFG3oSNVQADwMGDB4MYNBAIwGg0YuTIkVEBpXEch/HjxwcxnPAegyGg5SgxMZEXOqEEIcdxiIuL+8Xmu7y8PGRG1JgxYyJChdAucPTo0SKFHukeoXbhHMfh/PPPH5SeExaLJeyzq9Vq7rAPPYji4uKg0WjCbkpihfkgfqXsP3J3VVZWsv3796O8vBxVVVVoaGhAU1MT6uvreQSA/rad/iWsD5VKBZfLherqatHaIUu2sLAQSUlJXKQNHn0Wzkoly+TSSy/lFi9ezHbu3AmdTsfHLaTfO3jwIA4ePIi33noLACgDjJ1++umYNWsWSktLY4ahURRIFORwOEKCKQonXKfT4fTTT+e+/vprNnXqVOzdu1ekRGjBPPfcc7j++uv7ZYWQIAkFp5GYmMjDpUSjiDIyMkIqidbW1iNm1en1euj1ei7a5/wlSPq+QkgZ8rtH8zwpKSkwGAxwuVxBwq6trS3ideiclJSUqDcDgzEvof4XjTCJdZ6EVu3HH3/M3nnnHWzevBk1NTUh1wSBFrrd7iEtL7q7u1lXV5fsuqJYEO30o7XUQrnKGWMwGAz48MMPMXfuXB7mhqwIoftdGG85XKOCdevWYd26dbj//vsxY8YM9te//hVnnHFGTC52paFUFEKbwN5CHTTYh0EAuT//+c9BO0fy4e/evRuVlZVM6GuPlaTuG1rAOp0OGo0m6tVMvTDkGPVI9mIfSv5repZQgkmj0UQFIimcA+n3aXyjEX50nbS0NB7w8rfU9IqE08GDB9mMGTPY3LlzsXLlSlRXV/+c1aPRwGAwQK/XByUqEMjmUOy9QXPsdDqDQFOlNUfRbhxDtc6Wukbz8/O57777Dn/729+QlZXFb3iF6NDCui8hsCZtir/++muceeaZ+L//+z8mDP4rFsgvSFqtFowxHH/88TxYnpzvsry8HIWFhf02xckfLhUshzsEMp1OF5XEIYEm5xYYKDrw0ULCOJUcUefFaK7DcRwPbRNKuUSrQIxGoyge8VtRHhzHoaamhk2fPh2NjY28e4w+o/RoGouUlBRkZmZi1KhRmDlzJlJSUnDhhRcOWVdWuIwv2pRFM5+hknek8QlSDBaLhfv73/+OO+64g61btw6rVq3Cpk2bcPDgwSAoeo1GwysToWwCgIceeggZGRnspptuiqrwVVEgR2BHazKZoNVq4fF4gipfgZ/TUGPZiQiZSq1WIzk5WVYIdnd3o6enByaTKSqh2dLSEpKhU1NTj6l5k76vMBOpra0NBQUFUc1Xe3u7rPuK3FvRzvtvrdUujSfHcbjpppvQ2NgoQlAgi1yv12PevHmYNWsWiouLkZWVhYSEBH4wqJCTgs1DzZK1Wq0wmUyiTQQ9Z6h+QKEUUaTvSVEJAoEAEhMTuQsvvBAXXnghAoEAqqqq2I4dO7Bx40Zs3LgRO3fu5BW00MVOQXmVSoW//OUvuOSSS1hiYmLEeI2iQI7ALstut8Pr9YbcJQ2kDgQACgoKghoOqVQqOBwOVFVVISMjI6KfldxpUmElvMdvUZCFolGjRgX9jxIp9u/fj2nTpkVsSsRxHA4cOMDPsXCHx3Gc7D2OpXWh1Wqxfft2tnr1ah4qX8hj6enp+OSTTzB58uQgpvN4PHyQeqhuHAEgISGBy8zMZN3d3UFQLQcPHkRfXx+zWCwhBTMJ8lWrVrE333yT91oI4fmfeuopJCYmcuGypchVWFBQwBUUFOCiiy4CAOzbt4999tlnePHFF1FRUSGSUSQzurq6sHHjRsyZMyeiHFEUSBRCO1RrSOl3fD4f9Ho9vvjiC96fK3RjESPRbre/wnnatGlYsmSJbKro6tWrI6aKkvL58ssvRUqDdtx6vR4lJSUDUnZHk8UIAMcffzy/gKX0xRdfYMGCBVFd64svvgiaWwE68TGllKUCDQC+/fbbIF8+KeqHHnoIkydP5txuN9+1j75Hscju7u4hO4bk8iktLRUV6NEGr62tDVu3bsWMGTPCdtjkOA4vv/wy3n///aDPk5KS8Pzzz3PAz+nNtGGRnl9aWsrp9Xpx1bhGg7Fjx3Jjx47FDTfcwGbPno1NmzYF1e9wHIdDhw5FZS0rQfQIRIxM1dJyBzG3Xq/H1q1b2SOPPBKUqy6sgRg7dmy/FgEx3IwZM2A2m0W1KcQAL7/8Mux2e8hAmMfjgVqtxnfffcc2b94s2imT/3by5MmggsffurAjV8jEiRM5srqE2Socx+GTTz5BZWUl02g0sq1NKUOvoaGBvf/++0EIAYwxZGdn8zvrwQapPJpIrlkUjdXkyZNF6AnEj8JY0M6dOwddgVDQmeA+hCCa/fEQ/P73v5ftdw4AixcvDiqMFfKRWq1Ge3s7W7duHTQaDbRaLZ9YoNVqMWvWLD4+uX37dpx44omi46STTsKJJ56I8vJyJmw2R8HyQCAAl8uF+Ph47o477pBNE48ldVxRIBGou7sbLS0trKGhgbW0tAQdTU1NrKKign311VfstttuY9OnT4dcGh8thilTpmDYsGH9qkYnwZSRkcHNnTtXtIshU7OhoQG33norr9QoWEYMq9Pp0NXVxW688UbZ6zPGcP3114uU0m/dAiEk3muuuUa0eIS5/ddddx0PY0JQEDSmFJS8/vrr0dvbK/LP0+9XXHEFjEZjWATdY8kSkbMC7XY7j6ggBJ6kcfd4PHj99dcHjTfpvtu2bYNGo+Ezv6LpWR5qg8cYw+zZs4OKcMk19cknn2DFihWMqvqJj4S90m+//XZ0dXXxlfzUQM3r9eL888/n71dQUACdTieCa6GN7gcffMArKikwKhHJKTklmJWVpSiQgRAt9H/84x8oKCjAmDFjUFBQEHQUFhZi7NixOPvss7FkyZKQAVQSJH/+85+jMg3DMT1jDPfddx/PsMTsZEIvX74c8+fPZ+Xl5UyYhhwIBLBu3Tp22mmnYe/evSLYDnIjjB49GvPnz+fC9YD/rRGN480334yUlBTZhb9+/XqceeaZ7IcffmDke6Z8++3bt7Ozzz6brVq1KsiiI4yn2267bVCKAo92ysjICBLO9Pdzzz0HjuOg1+tFmGUEFXPttdey6urqkAgG0SgLqeXBcRyefvpp3H///ez9999nn3zyCVu+fDmrr69n/d3gWSwW7t577w1yU9H8X3vttXjiiSdYb28vE5YC1NTUsGuvvZa9+eabsnw0YsQIzJo1i6O4yrBhw7hx48aJ5ILP5wPHcfjHP/6Bzz77jOl0Ov76JAsMBgMaGxvZE088IYv9ptPpcNxxx0XlwlZiIBHI4/FElcZJEyQ1TWlX4PF4cNlll+F3v/sdnx7Xn10U3WP8+PHcX//6V/b3v/9dlM1CAu+dd97Bhx9+iJKSEpaVlcX3Ati3b5+IKYV+TwB4/vnnIeznPRRTJY+UFZKcnMwtXryYXX755bylQYtVpVJhw4YNmDx5MkpKSlheXh4A4NChQ/jpp59kx5T6aDz55JPIzMzkjqUxleNbAHwygpD3SZC/8847cLlc7Nprr8XIkSOh1+ths9mwc+dOPPfcc/jhhx+g0WhC1iiEq7OhDDi53bbdbscjjzwi+uybb75BTk5OvzYjfr8fCxcu5N566y22efNmvh+I0K11zz33YPHixSguLmZGoxEtLS3YsWMHHA5HkIIkPnr88cdhMBhElu+tt96K6667DlqtViR7nE4nfve73+Gqq65iF1xwAQoLC/keMOvXr8fSpUvR0NAgu4mcPXs2cnNzlTTewRIu0ZrmwkknAUIw0WeffTb+/e9/c4OBLUVK5KGHHuJ2797N/vvf//IMRM+hVqvhdruxdevWoPeRMg3tXP75z39ixowZR6SZ1NFghfj9flx22WXcTz/9xJ588kneNUUHLewdO3Zgx44dsucLNxNerxe33HILbrjhhmNyTKU8GwgEMHXqVG7ixIls586doiQTsqQ/+ugjfPTRRz8LJ0kSirQxk5y7OZzikuJ8SdeqMD1YDjQzWnlBcdF33nkHJ598Mg4dOhTUoIzczQ0NDWH5iJTHwoULceGFF4r4KBAI4JprruHeeecdtnr1ahGUCb3ja6+9htdee012PIWKij6zWCx4/PHHo64/UlxYhyctXJV5NAf5TgldloSyWq3G3XffjU8//ZSj4jC5iYlU5S5lUrrH22+/zV155ZU8GBoFICkATs8lvBaZ1kJmfeaZZ3DHHXdw9MyxjlMooRHL92M5PxyGVixjKbd4n3jiCe6+++7jffF0Pi0q4XOR71n4PeoJceedd2Lp0qUcWYWxjGl/3CexjtVgPUc080QCTaPRYOnSpXysgyBKpHxJbmSVSgW9Xg9KYCguLkZGRoaIr+kgOBpptfZhHCrummuu4S1r6ZoQ9uOQegZiHVtam7m5udzatWtRXFzMC3aSD7QhET6LkI9o8+L1enHDDTfgueeeC+IjkiXvvPMOTjrpJL6dL8VSVCqVqJ0DubeECNj0Hj6fD0ajEe+++y4KCgq4qN2tx3pDqc7Ozn73Eg53pKens2uvvZZt376dSe8pbShVXl4e8jrUCEiuuYswMLZs2TKWm5sr25SIOhfKdas74YQT2IYNG0L2QRY2lMrIyJB9xhEjRsg2lBozZozs95OTk5nb7ZYdE+G73nLLLREbD9Ez08/Vq1eHPOe+++6LqlEOzcvHH3/MioqKZPtWUyMxuTEdO3Yse++995iwg6TcmLrdbiQlJck+69ixY1mo8REe9N4ffvhhyPf+//6//0/2vQf6HHS9hQsXhrz31q1bRd0j6ecHH3zApPeMNK5XXnkl6+7uZpMnT5a9V2ZmpixfUSKJ3W5nl1xyScw90ZctWxbyey+//HJInqJ3tdlsbNGiRUyv18t2fJTrLAqAZWVlsZdeeikkHwnf0+l04u6772YmkymkDKD7yHWfPPHEE9m2bdti6oeuNJQ6bBbPnj0bNput3/AIHMfBYDAgLS0NhYWFOO644zB58mQkJiZywrhEqOCh2WzGWWedJcL9p93wpEmTQrrShIVK119/PXfRRRext956C++99x5++ukndHV1yfqLc3JycPLJJ+Oyyy7DnDlzokLh5DgOZ599Nu83FZrJw4cPlz1nxowZyM7ODmpYQ/3PI7kNi4uLccYZZ4gsJbqGFM2XfqampuKMM84QzSWdP2bMmKjckrRrnTNnDjdz5ky899577O2338bWrVvR2toqO6ZpaWmYMmUK5s2bh4suuogjX3W4MVWpVJg5cyZaW1v556WxirboUAiMecYZZ4jcEvTekYpC+/sc0cwTgXsK6zn8fj/OO+88rrS0lD399NP4+OOPUV1dHTSuarUaw4cPx/Tp03H11Vfj5JNP5gDgsssuY3FxcUH3OxyTYrSDlvKGyWTi3nnnHdx4443sv//9L8rKytDc3Ay73c7PlcFgQFxcnAiiftiwYUHvR79T86VQlgj1tlm8eDFuvvlmtmLFCnz22Wc4cOCArDvOarViwoQJuPDCC3HllVdGbEInBFV84oknuBtvvJGtXLkSq1atwv79+9Hd3S3LrxqNBjk5OZg2bRrmz5+PuXPn9guNlzsWA3q/FFFw8JfIvJFOfFtbG6uurkZTUxP6+vqgUqmQmJiInJwcDB8+HGazmRPGb4717KBoxrS7u5tVV1ejoaGBb2gVFxeH7OxsDB8+XAS5cazHPGIZW5fLhfLycnbo0CHYbDYermfYsGEYPnw4R/EI2oT0twZECilEz+FyuZjAdcRFg1kW632FFd2MMVRVVbGamhp0dHTA6/XCZDIhPT0dw4cPR2ZmZsx8JL0HyYC6ujo0Nzejp6eHbxCWnJyM7Oxs5ObmckLMu36VFigKZHByyqX9OSK10Yz2GWJZMGTqRoOjI9xJDcY4yd0v1u+HG9Nozw93Tn+EjxDoL9IzC6uOY0m+GMgYDdZ7D+Q5+jNPwvGKxINSXh3omJFrKdr40GDxFLl3w/X5EK7jaJ9P7h7RnistIo7Z+6IokN8mCYOBQsUmBwutUGxCWrpZUMb06B3XUF0Wf433HYggj/Y+gz22igJRSCGFFFKoX6Q4vhVSSCGFFFIUiEIKKaSQQooCUUghhRRSSFEgCimkkEIKKQpEIYUUUkghhQ6TAqaI0HneSmqmQr82Hyo8+Ouv41B1J0Oh+DbUO/9Sz6ak8SqkkEIKKaRYIP3R3hzHoaKigrW2tkKj0fD/8/v9GDduHOLj47looY0VUijUjjXcLpj4q7y8nLW3t/OVyj6fD8nJyRg9erTCfFGs497eXrZ7924RcrLP50NKSgpGjRrVr3XMGENZWRlzOp38rp6qySdOnMj1F/b9SL0zfVZcXCyCK1IUyBGchE2bNmHjxo2wWCw8bIXL5cKiRYsQHx8PRYEoFAvF6j4g/tqwYQO2bdsGs9kM4OdGR5MmTcLo0aMVHoxi/FpbW/H6669Dr9fzkDJ2ux1TpkzBqFGj0F8F8sknn6CpqQmEkeX3+2EwGDB69Gim1Wp/lQ1mqHcmBXfPPffAbDYfcb5RYiAA9Ho9LBYLzGYzr0Cot4dCCsVCXq8X27dvZwSkyXEcvF4vxowZg9TU1LDChvjQZDLxVovBYFAGNUpSq9WwWCzQ6XQiTDIhYGB/yGQywWKx8E2mSIEMBYUufWdSIL9UDESRkICojwUpEGkDeoUUimZH6HK52Pvvvw+n08k39unr68P111+P1NRUxZL4hdaxUJgOdB0LZYPw76H8zr8UKQpEIYUGkTiOg9lsDupYF42v3O12w26383/b7fawfb4VUujXJkWBKKTQIJPQko3GmiUlM3LkSJHLxe12h2zWpZBCigJRSCGFeAUyY8YMbsaMGWG/o5BCigL5DVA0fQTk8P6l/Q6G2r0iPYe0r4Bcn4H+vlN/BGV/rxOqaE96zWjeSXitUM8TrgeE0HIRxkjo9/5kdUnfoT/vNRh8ONB7/lZkQ6y8Pdh9So6UfFAUyAB3jeEmX25ihH+HEiSx3itUFzy5xd0fRqGsjkjvI/1+f99psOYh1rkJdc1IrVSFn2m1Wg5A0OqngHqkHumDIbhCPWus7zUYPC/Hg791i2ow3m0wm0oNpixSFMggkM/nQ29vL5MufmnRYX19PTtw4ABaWlrgcrn4lLuMjAwUFhYiLS2NiyTYGWPo6elh0h1JXFwcJxTqHR0dbP/+/WhoaEBfXx84joPJZEJaWhpGjBiBvLw8jnaEsVgJtANmjOHQoUOstrYWra2tsNvtfNtNi8WCtLQ05OXlITc3lyNBGOpejDF0d3cz6f+0Wi2sVmtMHGyz2ZjP5xPtqkJdR6jYbDYb3y+6q6sLDocDPp8PKpUKRqMRSUlJyMrKQl5eHiwWS8h58vv9sNlsjD5zOByylo3NZkNXVxejZ2CMwWq1igrR+vr6mMfjEb2LTqfj7x/NXAFAe3s7q66uRlNTE2w2G+ia9F65ubnIz8/njEZj1Arf6/Wit7eXCZ9No9EgLi6OE55fW1vLysvL0draCrfbDbVaDavViqysLBQUFCAlJSUizx/t5HA4mMvlioon5cjlcsHhcDDh+KhUKsTFxXGxWjE0L3V1dezgwYNobm4WyaL09HSMGDECWVlZ/ZoXRYHEqM05jkNTUxNbunQpn1nj9XqRmpqKRYsWQavVorW1lX366afYv38/3G43v3sXmpEGgwHjx49ns2fPRlJSUlB9AP3tdrvx/PPPw2az8ZXyAPDHP/6RZWRkcE6nE6tWrWI//PADrziEjEtCKD8/n82aNQvDhw/nohEY9B2/348tW7awLVu2oLm5mRdGdAjdMjqdDllZWWzatGmYOnUqp1KpgoQTfXflypWora3lC6ACgQCMRiPuuOMOZrFYuEhK9bBQZk8//TQ/xiqVCn19fTj33HNx1llnie5Ni6m+vp598803KC8vR29vr2gXLn2fw5sCjBs3jp1xxhlITk7mhAqA4zh0dXWxxYsXi5QG9aWn6xgMBqxatQpffvklr1DcbjduuukmNnLkSI4U8YcffogdO3bwdSAOhwMTJ07ElVdeGVYZ07MfPHiQbdiwAZWVlXA4HKKdp/C91Go1EhMT2cSJE3HqqaciISEh5HjT+1ZXV7Nly5bxdSlutxvDhg3DH/7wB6hUKjQ0NLBPP/0UBw8ehNfrleV5k8mECRMmsNmzZ8Nqtf7mEB5orL766its2LCBL+SjZIiFCxeGFdB0/tatW9lHH30Ei8UCxhi8Xi+SkpJw5513Rl2bRnxdV1fHvvjiC1RUVISURXq9HsOHD2dnnHEGCgsLY5oQRYH0U5EId70+nw8ejwdarRYHDhxgr732GhwOB8xms6iISThpgUAAP/74IyoqKnDdddexYcOGhVxQPp8PPp+Pv0YgEIBer0dXVxd76aWX0NDQAIvFgri4uJB++crKSixduhTz5s1jkydP5qJh5IaGBvaf//wHNTU10Gq10Ol00Ol0QeawUDg1NDTg7bffxrZt29gll1yCjIwMTk6QFxUV4cCBA9BoNPz92tvbUV5ejuOOOw7RKJDy8nJ0dnbCbDbz46PValFcXCwyx+n6a9euZatWrYLP54Ner4fJZArpwqH3cTqd2LRpE3bu3Il58+axCRMmBOHH+f3+iLUGwtx8gsqRu45wrn0+H/x+f8QNjcfjwYcffsi2bt0Kxhj0ej3MZrPsu9F79fX1Ye3atfjhhx8wZ86ciDxBPC98Nq/XC47jsHPnTrZy5Up4vV6YTCaQZSN970AggM2bN6OiogI33HADS09P/03CBAnnkcYt3DyGG+v+nB8IBGA2m7Fz5072xhtvwO/3w2g0hpRFjDGUl5ejvLwc06dPZ3PmzOGidWcpcO4D8FEKF6jJZEJ1dTV75ZVX4Pf7YTKZ4HA4YLPZYLPZ0NvbC4/HI1pUFosFfX19WLFiBfr6+pjQFxnuXjqdDjabDa+88gpaWloQHx8Pt9uN3t5e/l4ul4u/F2MMRqMRGo0Gb7/9Nmpqahill4Yye8vLy9mzzz6L+vp6WK1WHsYhEAjA6XSK7uV0Ovnn1ul0sFqtqKmpwb/+9S9UVlYy2o0LmXLChAmwWq2i+I1KpcLu3bsj+mLps7179/LnqtVqeL1e5OfnIyMjgxdMdP1Vq1axDz/8EFqtllccZD329fWJ5qmvrw9er5d3HVgsFvh8Prz66qs4cOBA0Nj5/X7REWpRS78XCnlXeoRTHna7nb3wwgts48aNMBgMMBqN/DlCnrDZbOjr6+MVALmW3G433njjDXz55ZcheSIUH5rNZpSXl7MVK1ZApVLBYDDAbrfL3o+ua7Va0dHRgddee42vcTnSBbv0vP2JMfXnnFjmMZZrRKt8DAYDtm3bhpUrV0KtVsNsNgfxgsPh4HmV5IPRaMRXX32Ft956i0mVv2KBHEEizJ23334bjDE+NlBUVIT09HSo1Wp0d3ejsrISHR0dMBgM/PeMRiPa2tqwYcMGzJo1izc9I93vvffeQ3NzM7RaLVwuFwoKCpCdnQ2dTge73Y7a2lrU19fzu45AIAC1Wg2Px4NVq1bh5ptvDhlkb2hoYMuXL+chG8glQ77TkSNHIisrCwaDAS6XCw0NDaipqeF3v/ReXq8Xr7zyChYtWiTabTLGkJiYyI0cOZLt3r0bRqMRgUAAOp0OVVVVsNvtzGw2y+5OBbEGVl1dDZ1Ox1sYPp8PEyZMEO2wVCoVDhw4wFatWoW4uDjegqN4RXx8PMaMGYOUlBTodDq43W50dnaitrYW3d3dMBqN8Pv90Gg08Pv9+OCDD3DnnXfy7kuVSoXExETRbq6vr08kFBljMJvNPNwEWQ0DBeLz+XxYsWIFKisrERcXB4JP8fv98Hg8yMzMRH5+Pv9ZW1sbqqqq0Nvby78X+cI//fRTmEwmdsopp0Tl4tRoNOjs7MS7774LrVYLj8cDs9mMkpISpKamknsPFRUV6O7uFvG82WxGXV0dNm/ezKZPn85Fw/MDtQg8Hk+/ID6ONkQKmv+1a9dCrVbD7/fD4XAgOzsbubm5sFqt8Pv9aGlpobUGk8nEK4v4+Hhs3rwZKSkp7Oyzz47IC4oCGQR3llqtRk9PDy/EioqKMGfOHKSmpopWhdPpxMcff8y2bt3KC81AIACDwYBdu3bh7LPPjujjJOHT2toKxhhSUlJw4YUXYsSIEZyU8Tds2MA+/fRTXnDRvWpqatDS0iLrQvD5fHjnnXfg8XhgMBj4RedwOJCfn4+5c+ciPz8/aLVXVVWxjz76CHV1dfy7abVaOJ1OvPPOO7yvXKgESkpKUFZWJhJK3d3dKC8vR2lpKcIpEBJMxPx+vx9WqxXjx48X7ToZY1izZg2f/SSMLZ122mk4/fTTERcXF/Q+vb297PPPP8fWrVv5cdDr9WhpaUFFRQUbN24cFwgEkJiYyN11110ii+Dpp5/mle3h/+HCCy9ESUmJaEHSXMcqPOkaa9asYfv27UN8fDyvPHw+H3Q6Hc4//3xMmjQpCC22q6uLffXVV9i8eTOMRiMvHC0WCz755BOMGDGCZWdnR3RnaTQadHV1QaVSwev1YtKkSZg1axYSExNFJ/X19bH3338fO3fuFPG8TqfDjh07cNpppx0x5UFzVlFRgSeeeKLfWsDhcIiQuo8GmaTRaODxeBAfH4+5c+di/PjxnFQRtLW1sc8++ww7d+4MWkdfffUVioqKWFZWVlheUFxYg0QajQZutxvFxcW49tprudTUVE6IoUNB4nnz5nG5ublwu938blytVqOrqwvt7e0sGpOeXEIpKSlYuHAhRowYwcnheU2fPp2bPHkyhFDUhDRcV1cnch/QOVu3bmU1NTX8YlepVHA6nRgzZgxuueUWLj8/P+hejDGMGDGCu+WWW7iCggL+foFAACaTCZWVlfjxxx95Fwkx45gxY5CUlCSKJ3EcF5Uba+/evfyzq1QquN1ujBw5kg8I0/kdHR2ssbGRD/7SPE2bNg3nnXceFxcXF/Q+gUAAVquVmzdvHpefnw9hZlQgEEBTU5NIoVN8SKvVhrQqNBpN0Pf6I4zIquro6GAbNmyAxWIRWR56vR433XQTpk2bxmm12qD3SkxM5C655BJu9uzZ/DzRNb1eL1atWhWTkHK5XJgyZQouv/xyLjExMYjnLRYLd/nll3Opqal8zITObW9vR09PD5PWiwz2jtzr9aKnp6ffx1DBvYr1na1WKxYuXIji4mI+oUV4pKamctdccw134oknwuFwiGSEz+fDmjVrlBjILzVhPp8PFosFF110kWiXKDzI5zh58mR+MQmtip6enqh9wn6/HxdccAEsFgtHAkR4L3LhTJs2TdQrgK7f3d0dpJT8fj82b94MvV7PC3qv14uEhARcccUVIIEkvRcpBr1ejyuvvBJxcXG8UiBL5LvvvguC9zCZTNzo0aN5ZUo708rKStjtdlnBQsqisrJShEDKGENJSUlQgPDQoUNob2+H0+mE3W5Hb28vfD4fTj31VJE7S26uGGMoLCwUzRUAPpYlFajh5i7S57EoEADYsmUL7Ha7yL3g9Xpx8cUXIzc3lyNek74XvfOZZ57JlZSU8IKDNjiHU8EjCnXijeTkZJx//vm8EpYbR41Gg0mTJvHzLJxHm832i6xPQtfuz3E0ks/nw4UXXojk5OSQvECbvwsvvJAbNmyYaB0aDAbs378fnZ2dYXlBUSCDxKButxslJSWwWq0h/Yb0v5ycHGi12qCKXQrcRuPCysvL45vkyBWnkaBOS0vjhP5xOSFIO6xDhw4xiqsIXT0zZsyA2WwO6w8lhrRardxpp50GyoOn9N7GxkZZwVRaWgphkJ3cWAcPHgwSuPScNTU1rKOjQ9R4KSkpCaNHjxa9OwBkZ2fjyiuvxLx58zB//nzMmzcPV155JZKTkzlplTcpHmFF+FADMyShvHfvXlFMxe12o7CwEBMmTOAo3hWKf+hdzz33XH6zIOQtcitGUiButxvHH3+8qP9GqPvl5uYGPZOQ5490nEGKBBDLcTTKohEjRmD8+PFheUFofc6cOVMkIyiue+DAgbDzoyiQQZy4wsLCqBiO8Pv7YxrTzm/kyJFhJ5YYwWAwwGw2R5UGWFlZKdpt+/1+xMXFYeLEiVH5f0k5lJSU8K4VoWCqrKwUuWEAYMSIEVxGRkaQRRbOjbVnzx6RNeN2uzF69GiYTCZOWl2dnp7OnXzyydzUqVO5adOmcSeccAJ3/PHHc2q1OggGm85TqVTQaDRoaWlhZWVlfAB4KPi2AaClpYXvXCjsoDlp0qSolRAApKWlcSNHjuTrA8i1VFVVFdGFSN8tKCiIyBPAz9lXlLKt0JH3hlAySbS8MGrUKC4jI0Pkrj1c/xPeda8M+eAsbI1Gg/j4+Kj82mq1mt+x95cSEhKiZqhw8BlCamxsFPlBPR4Phg0bBooTRAu5kpiYyKWnpzMqFKTPhLEDsig0Gg2Kiorw5Zdf8jtqcmM5HA5mMpk4YUW8z+fDwYMHRVaSSqXi3VeRdp5Ct5vwfbxeLxwOB7Pb7Whvb0dlZSV++uknOJ1O0b1+bT7jOA4tLS1wu918AzRyP+Xn50cU/NJrjRgxglfWZAF2dHTA6XSKguxy5+t0Or72KNI9hdD2v6S15nQ6MW7cOPz+97/vdxbWihUrILR4h7os0mq1yMvLi5oXaB3m5+fznRfJs9Ha2ipSNIoCOYKa/5dcIIPZcYyuRZlkwl1tWlqaSOBEw4wqlQppaWmoqqoS1WJQjEdoaQDAxIkT8c033/BCntxYFRUVmDBhgkj4Hzp0iLW1tfF1KR6Ph+BaOLlxkasBaG9vZy0tLWhpaUFHRwc6OzvR1dUFu90Oj8cDr9fL59MLXY1Dhbq6ukRWk8/nQ0JCAhISEqIuACNKT08PKvJ0uVzo6+tjRqORi+TG+qU63w1EoBqNRh42qD+k0WjY0eDKoviF0WhEfHx8zOdnZGSIeJ3cWG63m3dTSnlLUSAK8QvN4/EExQQsFku/rkcwDEJmpOJGoQJhjCErK4vLzc1lNTU1vMXCGMOuXbuCTPG9e/fC6/Xy3/N4PBg/fjwf4JcKNGJ6n8+HrVu3sh9//JHHAyJ3l1qtFoEdqtVq+Hw+uFwu6PX6IZe6SeMoVNpGo5FXqrHOkzDJgsZKLlHgaCVyUx4LdSBkgej1eq4/vCAFwjy8oWKhrqcoEEVxiFw9g2XpRHse7XonTpyIiooKPt6g1+tRWVkpcqUEAgGUl5eLrAKdToeJEyeGddO0trayN954A7W1tXwKrclk4pUmwUZQ7EOn0yE9PR2lpaXYvn07WlpaBlz0N9gCMZT7MJbdajjL+bfWznkgFeFH23rurzdEunkMJxcUBaJQkItHzl3jdDr7dV2HwxGElyUnhOk7xcXF+PLLL/lMECpUq6ioYMXFxRxVyDc3N4vcVzk5OcjJyeGkWUD0Hr29vWzZsmVob2+H1Wrlha/dboder0dubi6ysrKQnp6O5ORkxMXFwWKx8O6gsrKyIee+EGIa0Ri63W6+sjwWQeNyueD3+/m5oXEcSgpTodiUwGGcMmYwGGLSIkLLltYlWeSKAlEoojCxWCyiQj+1Wo329vaYdmL0vY6ODlFA/nBRGb+DFn4mhDbZtWsXj0bLGMPu3bt5cMR9+/bB7XaLUJCLi4v5hASpwjqMgcXjhVFtitvtxqRJk3DGGWcgMzOTO9rcF3FxcaJ0XLVajb6+PvT29rJwyLpy1N7ezittIVoBzYFCv55XoL/r2Ol0wmazwWq1xnS+MLZG/H8YhDFkbE1J41WIZ5q0tDQRwJpGo0FTU5OoACwaBnY4HKJ6EmLG9PT0sPcn+BL6n16vx8GDB3kraN++fXzqKjG3FHlXqDzsdjvbvXs3TCYT76JyuVwoKSnBFVdcwZHykFbVCyvchyKlpaWJoP1JgTQ2NkZdu0DjVVNTI1Lmfr8f8fHxfA8SpZXu4K6xaMaToPj7a4F4PB7U19fHzAsNDQ28tUG8kJCQEFSIrCgQhWQpPz9flJGj1WrR0dGB8vLyqNA5ickOHDiArq6uICFHaaahGHjMmDEcQZsA/wPsO3ToELPZbKy+vp6Hk3e73cjPz0daWppsLxUAfOMraZYRVaHLVehK+6kMpboFeq6MjIyg4lDGGHbs2BG1oj/sxmMHDx4UFST6fD7k5uYiEjKvQqFJyPfEX5S0Ec38tra2DmjzQqjWsfBCT08Pq62tFdWn+f1+ZGdnh7WKFAWiEM9oh7GkeFcPCdyvv/46onkttDS++eYbUZEbNcQZPnx4yFRbsijGjBkjsngCgQAOHDiA/fv3izC9/H4/HzwP9UwEWS28lsFg4FMcQy1SskR6e3uZVBH+0m6JUOM0cuRIvuiL/rdz5040NTWxSDVG5O779ttv0d3dLXo/juNQVFSkLIoBkNlsFs07ga329vayUFYB/c/pdEIqyGMh4vF9+/ahpqaGCSGUwvHCxo0b0dfXJ4p3qNVqFBYWhldWynQrJMSmmjhxIlwuF18PYjAYUFVVhVWrVjEh7pUwNVKIJ/XZZ58xYadBwjwqLS0VwWaEIiG0Cd1/z549oH4XZDnExcUFIe/KvZe0LajL5UJPT4/oPaTvQu/5xRdfwG63ywYRw8VH5BCEhf1Z6BgITZs2LSjlUoikLATPk4JFqtVqVFRUsK+//poHzaSUzezsbBQWFnKhoEkUikwpKSmisVOr1ejt7cXevXt515B0XmiztW7dOkZKfaBr+p133oHdbmdS1AU5Xli/fn0QL6SlpYXc9CkKRKEghmOMYfr06UFgiCaTCatXr8ann37KqF2psAkU+V0/+ugjtm7dOh4aWgjGSG6jkLDQhxl0+PDhPLQJLb7u7m6+9wm5rwoKCkJWyNPfCQkJspll//3vf9HR0cEIEUD6Lna7nb377rsi2H0hUX8UOcWi1Wo54T0plrRz5054vV5oNBr+Pv11TzDGMHLkSK64uJhXcKRsDx06hGXLlrG2tjYm924qlQo7duxgy5cvF40VzeGZZ545KBbXsWzJ5+TkiPiG4nlffvklGhoaGPGAdF7Wr1/Pvv76a1F/jv5auzqdDq2trXjxxRfR2NgYkhfKysrYq6++Ktps0abvhBNO4OurQrrrlGlXSGiFJCQkcL///e/Z66+/zmdxkHBas2YN9u/fz0pLS5GTkwO9Xg+Xy4X6+nps374djY2NfH0FMaTL5cL8+fPDgkwKd/Vy0CbCPs7E5ELk3VAKJD09nUtNTWVUx0ELq6GhAUuWLEFJSQnLz8/nYWF6enpQU1ODXbt2obOzk38X6cLct28fRo8ezbRaLTIzMzlhsaFer4fFYkFXV5dIeFRVVWHx4sUsOzubD3qfffbZfApyf4TE3LlzUV1dDafTybs8jEYjqqqqsGTJEpSWlrLDipZvKLV7924+GYEUj0ajQU9PD6ZOnYqSkhLF+hjgJiwpKYnLy8tjBw4c4BWJWq2Gw+HA888/j6lTp7KCggIYjUZ4PB60tLRg586dqKqq4huyhQtcR0NerxdGoxENDQ149tlnUVJSwkaPHo34+HgEAgG0t7djz5492LNnD5+qS/PudDqRm5uLE044ISIvKApEIdHuNhAIYPLkyVxHRwf7/PPPYTab+f+bzWa0tLTg448/5nfS1J5Vp9Px2Ez0/b6+Pvz+979HaWlpVF3uhMrhm2++Ee18pPGU0aNHh80SokU7Y8YMvPrqq7z7jJSAy+XC+vXrsWHDBt6SIAh3UgIul4v/PrnUtFot2tra8MILL4DjONx9992MWujSYsvJyUFVVRWMRiPf11qn06GlpQX19fW8i+Dkk08esKC6+uqr2bJly+B2u/nukQaDAV6vFxs2bMC3337LWxSUiUa9UcjC6+npwZgxY3DRRRdxiuUxMCI+Pf3007Fv3z7RfGk0Gvh8PqxduxZff/01yLVETcCoV0xxcTG2bdsWsxtL6IY0m83Yu3cv4uLi4PF4sGnTJmzevDkkLwh7wmg0GsyfP1/ULiGkzFCmHLL9IGjXG07YSo9YhXW094v12QZ6r0AggHPOOYe76KKL+JaYwh221WrlcaKMRiOsVqsINNHhcIAxhnnz5uHMM8/kooWQoIWWmZnJDRs2DD6fD0I3E/U9HzNmDN8lMNx7MMYwadIk7vTTT0d3dzeE0NbUytVkMkGv1/NKw2q18u1Yc3JyMHPmTLjdbh4MkASBwWCAXq8XvRc9y0knnQSj0cj32qDxprGzWq08hEgkPgwX3wkEAhg5ciS3cOFCJCcno7e3V5T1Ru93GNqC/5usQ4/Hg76+PkyZMgXXX389J5zDaNfIkeL5X2od9/e5w7U1ONxHhjv77LPR09MDYcsFlUoFi8UCo9HIr5/4+Hi+6+cFF1yA8ePHw+v1ing/1P3k3lmtVuPiiy9Geno6enp6oFarYbVa+bbKer0eZrMZRqNRdA2n0wmNRoMFCxYgNzc3qlbDigUCwO12o6+vj9+5kuuF0knldrd2u12E0urz+aL2WzLGYLfb+TgDgZbJ3Y8xBofDAbvdzuM19fX1RdU7hMjhcKCvr4/Hderr6wuLdUSL4NRTT+VGjBjBvvzySxw4cIAXpMTYwuIzskQMBgMmTJiAmTNnIisri4sVf0gIbVJWVsYXNwp3WKGQd0MppPPPP59LTExka9euhc1mE2Ff0T1pJ0ixk5NOOglnnXUWd7jdMKuqqhJl1/j9fni9XlGGC90vIyODu+6669h///tftLS0BPViIZ6TZscQHwrSbMP2IyFln5eXx91+++1Ys2YN27ZtGw+KSXNFz0UJCBTDyc7Oxumnn47S0lIulDuQyOfzoa+vj7fShO8bDdGaEdacyI3BQMjv9/N8LqgFGnBPF1p/VMBKYxguiYIxhlmzZnEGg4GtWbMGvb29smvH7XbD5/MhNTUVl19+OSZMmMBt3ryZ2e12UZ8gEvbh3plcVz09PUhMTOT+8Ic/sA8++AC7du3iN2NSVGS/38/LocLCQpx//vnIzMyMftN3LJustGAqKytZa2trUH+FsWPHIj4+PihQ29fXx/bs2SNCrg0EAhg/fjxfgBVJYe3atYsJBaPP58Po0aORlJQkup/f70dZWRkTAh1SPxC5nuZytGvXLibMJvJ6vcjNzUVubm7Y84VMVF9fz/bs2YPa2lp0dXXxee1kBicmJiI/Px/jx49HdnY2Jz0/1jlxOBxs165dQWOs1WoxceJELlrIDuE1u7u72U8//YSDBw+io6MDbrebdy9ZrVZkZGRg5MiRGDVqFKxWKy9UbTYbW7VqFaqqqvhe51arFcOGDcM555wDs9nMyd3P7/ejoqKCNTY2wmazgRIQjEYjzGYzSkpKRIkABw8e5Pt8kNBOTk7GqFGjuGje73Ach+3ZswcVFRVoa2uDw+HgBYRWq0VcXByys7MxZswYjBkzhhOOb7hr9/T0sH379gUBLxYXF3NCl1gYIcyEsPE0PuPGjYu6XUCk9+/t7WXk0xdu7FJSUvjMsljvcRjUkwlTyMmimDBhAhcO8oXu19nZyX766SdUVVWhs7OTXzsajQYpKSkYP348Jk+ezBHmW3t7O6uoqOD5gLp9TpgwgRM+g/Sd6btGoxFFRUX8d2tqalhZWRnq6upgs9ng8Xh4a9xsNiMrKwsTJ07EuHHjYl63is9ToYgLSOrW8Pl8cLvdjJhQp9NxQn9tLFW3vxRJF4XX64Xb7WaHFQgnxZeSW0Rerxcul4up1WqYTKaohfovNU9yeGCHLWnGcRx0Oh0nReztj5JXqP98J1w7Wq1WpHwHey7k1qHX64XH4+HXrtFo5KRFuLHwraJAEBpxMhyqpZy7KpbJD4WoKne/WL470HtFGqNw6K0DQQKNdk5iHedYnlFoEUoXlVw2SjQLPpxbM9R9BmOewo0TPVOsrsWBojUPBh8O9jru73PH8u7h+E7us1jGOprvRsv3/VlXigJRaEC7m6FmaRzpdxiK1tVvfZ6UtTN0768oEIUUUkghhfpFivNTIYUUUkghRYEopJBCCimkKBCFFFJIIYUUBaKQQgoppJCiQBRSSCGFFFJIUSAKKaSQQgopCkQhhRRSSCFFgSikkEIKKaQoEIUUUkghhX7D9P8DMydEZ4Gi4a4AAAAASUVORK5CYII=" +_sample_name = SAMPLE_NAME or (Path(INDIR).resolve().name if INDIR else "—") +_now = datetime.now(timezone.utc).strftime("%Y-%m-%d") +display( + HTML( + f""" + +
+
+

Xenium Transcript QC Report

+
Date: {_now}
+
+
+ Bioinformatics Innovation Hub +
+
+
+Sample ID: {_sample_name} +
+""" + ) +) +``` + +## 1. At-a-glance Sample Health {#sec-1} + +::: {.callout-note collapse="true" title="📖 Metric explanation"} +QC results consolidated at sample level. One row per granularity (sample / field of view (FoV) / feature / cell), each carrying the worst sub-verdict from the corresponding section. + +- **Sample-level** — transcript yield, sample-wide quality value (QV), decoded gene fraction, unassigned codewords, negative control probes (see [Section 2](#sec-2)). +- **FoV-level** — per-FoV outlier detection within sample (see [Section 3](#sec-3)). +- **Feature-level** — gene-vs-control transcript-count separation (see [Section 4](#sec-4)). +- **Cell-level** — segmentation quality, nucleus-to-cell area ratio, nucleus RNA fraction, cell size (see [Section 5](#sec-5)). Most cell-level signals are advisory (PASS / WARN only): they flag cells for review rather than driving sample-level filtering. The exception is cells with no in-plane nucleus, which can reach FAIL at a high enough fraction. +::: + +```{python at-a-glance-scorecard} +#| echo: false + +# v5 4-row at-a-glance: one row per granularity, each aggregated worst-of +# from its section summary. Single source of truth for verdicts lives in +# the build_section_*_summary helpers; this cell composes them. + +_section_rows = { + "Sample-level": build_section_2_summary(qc_metrics, qc_cutoffs), + "FoV-level": build_section_3_summary(qc_metrics, qc_cutoffs), + "Feature-level": build_section_4_summary(qc_metrics, qc_cutoffs), + "Cell-level": build_section_5_summary(qc_metrics, qc_cutoffs), +} +_section_links = { + "Sample-level": ("sec-2", "Section 2"), + "FoV-level": ("sec-3", "Section 3"), + "Feature-level": ("sec-4", "Section 4"), + "Cell-level": ("sec-5", "Section 5"), +} + + +def _build_caveat_body(rows: list[dict]) -> str: + """For chevron expansion on WARN/FAIL §1 rows — list which sub-metric drove it.""" + bad = [r for r in rows if str(r["Status"]).upper() in ("WARN", "FAIL")] + if not bad: + return "" + bullets = [] + for r in bad: + bullets.append( + f"**{r['Status']}** — {r['Metric']}: {r['Value']} ({r['Cutoffs']})" + ) + return "\n\n".join(bullets) + + +_rows = [] +for label, rows in _section_rows.items(): + statuses = [r["Status"] for r in rows] + overall = _worst_status(statuses) + bad_summary = ", ".join( + f"{r['Metric']}" + for r in rows + if str(r["Status"]).upper() in ("WARN", "FAIL") + ) + if overall in ("WARN", "FAIL"): + detail = bad_summary or "see section summary" + elif overall == "N/A": + detail = "no eligible metrics" + else: + detail = "all metrics PASS" + caveat = _build_caveat_body(rows) + anchor, link_label = _section_links[label] + _rows.append({ + "Category": metric_cell_with_caveat(label, overall, caveat) if caveat else label, + "Status": overall, + "Detail": detail, + "Links": _link(anchor, link_label), + }) + +display_status_table( + pd.DataFrame(_rows), + status_cols=["Status"], + narrow_cols={"Detail": "28em", "Category": "14em"}, +) +``` + +## 2. Sample-level metrics {#sec-2} + + +```{python display-section-2-summary} +#| echo: false + +display_status_table( + pd.DataFrame(build_section_2_summary(qc_metrics, qc_cutoffs)), + status_cols=["Status"], + narrow_cols={"Cutoffs": "22em", "Detail": "10em"}, +) +``` + +### 2.1 Transcript yield summary {#sec-2-1} + +::: {.callout-note collapse="true" title="📖 Metric explanation"} +Counts and thresholds for the loaded transcripts and segmented cells. + +**Total transcripts** is the count of decoded transcripts in `transcripts.parquet`. + +**Total cells** comes from the segmentation in `cells.parquet`. + +**Min transcripts per cell** and **min genes per cell** are noise-derived per-cell filters: cells below these thresholds are likely segmentation artefacts. + +**Median transcripts / genes per cell** are sample-wide aggregates; the typical cell should sit well above the corresponding minimum. They carry an advisory PASS/WARN flag (never FAIL): PASS when the median sits at or above the transcripts- and genes-per-cell floor shown in the Cutoffs column, WARN below. A low median usually reflects panel size or panel/tissue compatibility; these are the first things to check when the flag reads WARN. +::: + +```{python display-yield-summary} +#| echo: false + + +def _fmt(v): + if v is None: + return "—" + if isinstance(v, bool): + return str(v) + if isinstance(v, (int, float)): + return f"{v:,}" if isinstance(v, int) else f"{v:,.3f}" + return str(v) + + +# Three logical groups so each metric breathes — rendered as separate +# status-styled tables (status_cols=[] gives consistent line-height / +# cell padding without colour pills). Tables share the same CSS as +# §2.2 / §3.1 / §5.3 for visual consistency across the report. + +# "Selected transcripts" only differs from "Total transcripts" when the +# script was run with --num-row-groups (a dev/CI subsampling flag). +# Suppress the redundant row in the common case. +_mol_total = qc_metrics.get("total_transcripts") +_mol_selected = qc_metrics.get("selected_transcripts") +_mol_rows = [ + {"Metric": "Total transcripts", "Value": _fmt(_mol_total)}, +] +if _mol_selected is not None and _mol_total is not None and _mol_selected != _mol_total: + _mol_rows.append( + {"Metric": "Selected after subsampling", "Value": _fmt(_mol_selected)} + ) +_mol_rows.append( + {"Metric": "Total features (genes + controls)", "Value": _fmt(qc_metrics.get("total_features"))} +) + +display(Markdown("**Transcript counts**")) +display_status_table(pd.DataFrame(_mol_rows), status_cols=[]) + +# "Analyzed genes" and "Retained genes (above noise bound)" are the same +# number by construction (the script filters AnnData to the retained +# genes: ad = ad[:, ad.var_names.isin(retained_genes.index)]). Drop the +# Analyzed row; keep the more descriptive Retained label. +display(Markdown("**Gene & cell counts**")) +display_status_table( + pd.DataFrame( + [ + {"Metric": "Retained genes (above noise bound)", "Value": _fmt(qc_metrics.get("retained_genes_count"))}, + {"Metric": "Total cells (segmented)", "Value": _fmt(qc_metrics.get("total_cells"))}, + ] + ), + status_cols=[], +) + +display(Markdown("**Per-cell aggregates**")) +# Median transcripts / genes per cell carry an advisory PASS/WARN pill (never +# FAIL): both are strongly panel-size- and tissue-driven, so the floors flag a +# sample whose typical cell carries very little signal rather than gatekeeping. +# The min (noise-floor) rows stay informational. Floors: per_cell_median in +# conf/transcript_qc_thresholds.yaml; warn_min=None gives WARN (not FAIL) below pass_min. +_pcm_cfg = (qc_cutoffs.get("transcript_qc", {}) or {}).get("per_cell_median", {}) or {} +_med_mol = qc_metrics.get("median_transcripts_per_cell") +_med_gene = qc_metrics.get("median_genes_per_cell") +_mol_pass_min = _pcm_cfg.get("median_transcripts_pass_min") +_gene_pass_min = _pcm_cfg.get("median_genes_pass_min") +display_status_table( + pd.DataFrame( + [ + {"Metric": "Min transcripts per cell (noise floor)", "Status": "—", + "Value": _fmt(qc_metrics.get("min_transcripts_per_cell")), "Cutoffs": "metric is informational only"}, + {"Metric": "Min genes per cell (noise floor)", "Status": "—", + "Value": _fmt(qc_metrics.get("min_genes_per_cell")), "Cutoffs": "metric is informational only"}, + {"Metric": "Median transcripts per cell", + "Status": _status_higher_better(_med_mol, _mol_pass_min, None), + "Value": _fmt(_med_mol), + "Cutoffs": (f"PASS ≥ {_mol_pass_min:,} (advisory; no FAIL)" if _mol_pass_min is not None else "—")}, + {"Metric": "Median genes per cell", + "Status": _status_higher_better(_med_gene, _gene_pass_min, None), + "Value": _fmt(_med_gene), + "Cutoffs": (f"PASS ≥ {_gene_pass_min:,} (advisory; no FAIL)" if _gene_pass_min is not None else "—")}, + ] + ), + status_cols=["Status"], +) +``` + +### 2.2 Codeword category breakdown {#sec-2-2} + +::: {.callout-note collapse="true" title="📖 Metric explanation"} +Every decoded transcript is classified into one of several codeword categories. + +**Decoded gene** transcripts are the real gene assignments. + +**Unassigned codeword** transcripts failed to match any feature — typically background, imperfect tissue permeabilization, or probe leakage. + +**Negative control probe** and **negative control codeword** are deliberate non-gene controls used to estimate background. + +**Deprecated codeword** are panel-version artefacts. + +::: + +::: {.callout-tip collapse="true" title="💡 Interpretation help"} +A high-quality experiment has: + +- Low percentage of unassigned transcripts (< 5 %) +- Minimal negative-control transcripts (< 2 %) + +Higher unassigned or negative-control fractions point to imperfect permeabilization, probe leakage, or low-quality regions. + +**Missing categories** (e.g. panels without `Negative Control Probe`) render as `N/A` rows rather than failing the report; the [Section 1](#sec-1) Decoded fraction row degrades to N/A in that case rather than asserting on a partial taxonomy. +::: + +```{python display-codeword-breakdown} +#| echo: false + +# Canonical category names are the snake_case strings Xenium's decoder +# emits. Order is set so PASS/WARN/FAIL-bearing rows come first (the +# rows a reader cares about for triage), informational rows next, then +# unknown categories appended alphabetically. +# +# The synthetic "decoded_gene_combined" row sums custom_gene + +# predesigned_gene (the two real-gene categories Xenium emits) — this +# matches v1 prose intent of a single "Decoded gene >= 90%" verdict. + +_mol_cfg = qc_cutoffs.get("transcript_qc", {}) or {} +_cw_thresh = _mol_cfg.get("codeword_categories", {}) or {} +_cw_info = set(_mol_cfg.get("codeword_informational", [])) +_cw_labels = _mol_cfg.get("category_display_labels", {}) or {} +_counts = qc_metrics.get("codeword_category_counts", {}) or {} +_total = int(sum(_counts.values())) if _counts else 0 + + +def _status_min_max(pct: float, cfg: dict) -> str: + if "pass_min" in cfg: + if pct >= cfg["pass_min"]: + return "PASS" + if pct >= cfg["warn_min"]: + return "WARN" + return "FAIL" + if "pass_max" in cfg: + if pct < cfg["pass_max"]: + return "PASS" + if pct < cfg["warn_max"]: + return "WARN" + return "FAIL" + return "N/A" + + +def _fmt_cutoffs(cfg: dict) -> str: + if "pass_min" in cfg: + return ( + f"PASS ≥ {cfg['pass_min']:.0f}%; " + f"WARN ≥ {cfg['warn_min']:.0f}%; FAIL otherwise" + ) + if "pass_max" in cfg: + return ( + f"PASS < {cfg['pass_max']:.0f}%; " + f"WARN < {cfg['warn_max']:.0f}%; FAIL otherwise" + ) + return "—" + + +_cw_rows = [] + +# 1) Synthetic "Decoded gene" combined row (pass_min on custom + predesigned sum) +_combined_cfg = _cw_thresh.get("decoded_gene_combined") +if _combined_cfg: + _contributors = _combined_cfg.get("contributors", []) + _combined_count = sum(int(_counts.get(c, 0)) for c in _contributors) + _combined_pct = 100.0 * _combined_count / _total if _total > 0 else 0.0 + _combined_status = ( + _status_min_max(_combined_pct, _combined_cfg) + if _combined_count > 0 + else "N/A" + ) + _combined_cutoffs = ( + _fmt_cutoffs(_combined_cfg) + if _combined_count > 0 + else "no contributor categories present on this panel" + ) + _cw_rows.append( + { + "Category": _combined_cfg.get("display_label", "Decoded gene (combined)"), + "Count": f"{_combined_count:,}", + "% of total": f"{_combined_pct:.3f}%", + "Status": _combined_status, + "Cutoffs": _combined_cutoffs, + } + ) + +# 2) Threshold-bearing single-category rows (unassigned_codeword, negative_control_probe) +for cat_key, cfg in _cw_thresh.items(): + if cat_key == "decoded_gene_combined": + continue + label = cfg.get("display_label") or _cw_labels.get(cat_key, cat_key) + if cat_key in _counts: + count = int(_counts[cat_key]) + pct = 100.0 * count / _total if _total > 0 else 0.0 + status = _status_min_max(pct, cfg) + cutoffs = _fmt_cutoffs(cfg) + else: + count = 0 + pct = 0.0 + status = "N/A" + cutoffs = "category absent on this panel" + _cw_rows.append( + { + "Category": label, + "Count": f"{count:,}", + "% of total": f"{pct:.3f}%", + "Status": status, + "Cutoffs": cutoffs, + } + ) + +# 3) Informational categories (no PASS/WARN/FAIL) +for cat_key in _mol_cfg.get("codeword_informational", []): + label = _cw_labels.get(cat_key, cat_key) + if cat_key in _counts: + count = int(_counts[cat_key]) + pct = 100.0 * count / _total if _total > 0 else 0.0 + _cw_rows.append( + { + "Category": label, + "Count": f"{count:,}", + "% of total": f"{pct:.3f}%", + "Status": "—", + "Cutoffs": "no thresholds, informational only", + } + ) + # Skip informational categories that aren't present (don't clutter the table + # with N/A rows for categories the panel never emits). + +# 4) Unknown categories (present in data but not in our taxonomy) — append alphabetically +_known = set(_cw_thresh.keys()) | set(_cw_info) | {"decoded_gene_combined"} +_unknown = sorted(set(_counts) - _known) +for cat_key in _unknown: + count = int(_counts[cat_key]) + pct = 100.0 * count / _total if _total > 0 else 0.0 + _cw_rows.append( + { + "Category": _cw_labels.get(cat_key, cat_key), + "Count": f"{count:,}", + "% of total": f"{pct:.3f}%", + "Status": "N/A", + "Cutoffs": "no thresholds defined", + } + ) + +# 5) Overall (sample) summary row — _summary_row_styles will pick up the prefix +_cw_rows.append( + { + "Category": "Overall (sample)", + "Count": f"{_total:,}", + "% of total": "100.000%", + "Status": "", + "Cutoffs": "", + } +) + +_df_cw = pd.DataFrame(_cw_rows) +display_status_table( + _df_cw, + status_cols=["Status"], + narrow_cols={"Cutoffs": "20em"}, +) +``` + +### 2.3 Transcript quality (sample-wide QV) {#sec-2-3} + +::: {.callout-note collapse="true" title="📖 Metric explanation"} +Every decoded transcript has a per-transcript **quality value (QV)** — a Phred-style confidence score from the decoder. QV measures how cleanly the observed optical signal matches the assigned codeword vs alternatives. QV > 20 is the standard "high confidence" threshold (~99% probability the call is correct). Modern Xenium chemistry caps QV at 40, so most high-confidence transcripts pile up at the right-hand end (QV = 40). What matters for QC is the low-QV **left tail**: transcripts below QV = 20 are lower-confidence calls, so a heavier left tail means worse quality. +::: + +::: {.callout-tip collapse="true" title="💡 Interpretation help"} +**How to read the violin plot:** + +- One violin per codeword category, stacked top to bottom +- Thickness at each QV = density of transcripts at that QV in that category +- Box plot inside shows median, quartiles, and outliers +- Good pattern: gene categories sit well to the right of QV = 20; control categories typically don't + +**Warning signs:** gene-category violins centred to the left of QV = 20, large bulges of mass left of QV = 20, or unusual bimodality. +::: + +```{python display-quality-violin} +#| echo: false +display(fig_html("figures/quality_distributions_comprehensive.png")) +``` + +### 2.4 Negative-control noise bound {#sec-2-4} + +::: {.callout-note collapse="true" title="📖 Metric explanation"} +**Noise bound** — the minimum number of transcripts a gene needs to count as real signal rather than background. Genes at or below it are dropped from downstream analysis. + +It's set from the **negative-control codewords** (probes with no matching gene, so their transcripts are pure noise): take each control's transcript count and put the threshold at about the 99th percentile of that control distribution (in log10: the mean plus 2.33 median absolute deviations). A gene must exceed almost all negative-control counts to be retained. + +One value per sample, applied to every gene. +::: + +```{python display-noise-bound} +#| echo: false + +_total_genes = qc_metrics.get("total_genes_count") +_retained = qc_metrics.get("retained_genes_count") +_filtered = qc_metrics.get("filtered_genes_count") +# Backwards-compat: if filtered_genes_count isn't in the JSON (e.g. an +# older render before this field was added), derive it from total minus +# retained when both are available. +if _filtered is None and _total_genes is not None and _retained is not None: + _filtered = max(0, _total_genes - _retained) + +_nb_rows = [ + {"Metric": "Noise bound (minimum transcript count per gene)", + "Value": f"{int(qc_metrics.get('neg_control_quantile', 0)):,}"}, + {"Metric": "Total genes evaluated against threshold", + "Value": f"{int(_total_genes):,}" if _total_genes is not None else "—"}, + {"Metric": "Genes retained (above threshold)", + "Value": f"{int(_retained):,}" if _retained is not None else "—"}, + {"Metric": "Genes filtered out (at or below threshold)", + "Value": f"{int(_filtered):,}" if _filtered is not None else "—"}, + {"Metric": "% genes above the noise bound", + "Value": ( + f"{100.0 * _retained / _total_genes:.1f}%" + if (_retained is not None and _total_genes) + else "—" + )}, +] +display_status_table(pd.DataFrame(_nb_rows), status_cols=[]) +``` + +## 3. FoV-level metrics {#sec-3} + +Metrics partitioned by field of view (imaging tile). Each FoV is imaged and decoded independently, so a localised defect (out-of-focus tile, dirty optic, illumination dropout) shows up here without dragging down the sample-wide numbers. + +### 3.1 FoV summary table {#sec-3-1} + +::: {.callout-note collapse="true" title="📖 Metric explanation"} +**Field of view (FoV)** — the rectangular imaging tile the Xenium instrument captures in one shot. The instrument images the slide as a grid of adjacent tiles, moving the stage between exposures so the tiles together cover the full slide surface. FoVs divide the slide across its **surface (the X-Y plane), not through its depth**. Each is addressed by a row-letter / column-number coordinate (e.g. `AD14`, `AB17`), and a typical sample carries dozens to hundreds. Because each tile is imaged separately, a problem confined to one region — a fold or tear, an out-of-focus tile, a patch of high background — shows up in that FoV's metrics without dragging down the sample-wide numbers. + +For each FoV with at least 100 decoded transcripts, this table reports: + +- `n_transcripts` — total decoded transcripts in this FoV +- `mean_qv` — mean quality value of those transcripts +- `pct_above_qv20` — fraction with QV > 20 (the high-confidence threshold) +- `dev_z` — how far this FoV's `pct_above_qv20` sits from the sample's median, in standard deviations of the per-FoV distribution (0 = typical for this sample; more negative = worse; below −2 is flagged) +- `Status` — PASS if `dev_z ≥ -2`; WARN if `-3 ≤ dev_z < -2`; FAIL if `dev_z < -3`; N/A if `n_transcripts < 100` +- `% unassigned` — fraction of this FoV's transcripts that segmentation did not assign to any cell +- `Transcript dev_z` — how far this FoV's transcript count sits from the sample's median, same within-sample standard-deviation scale as `dev_z` +- `Density status` — the transcript-count equivalent of `Status`: flags FoVs carrying far fewer transcripts than their neighbours (regional tissue damage, folds, detachment) + +Only low outliers flag, for both quality and density. FoVs with fewer than 100 transcripts are N/A — small FoVs at slide edges or partial tissue give noisy estimates. +::: + +::: {.callout-tip collapse="true" title="💡 Interpretation help"} +This table compares each FoV within the same sample. `pct_above_qv20` is tissue-dependent — typically lower in brain and liver, higher in pancreas — so an absolute floor would mark normal brain or liver as failing. Flagging FoVs relative to the sample's own per-FoV median, rather than a fixed cutoff, removes the tissue dependence: an outlier is defined by its distance from the other FoVs in the same sample. + +- The table is sorted by deviation, with the worst FoVs at the top. +- Cross-check with the image QC report's focus score: a blurry tile there often appears here as a low-`pct_above_qv20` outlier — same physical defect, two different signals. +- Cross-check with the QV by FoV boxplot in [Section 3.2](#sec-3-2): the outlier FoV's QV distribution should sit visibly below the rest. +- An outlier FoV is a region to consider excluding from downstream spatial analysis, not a reason to discard the whole sample. + +If the field-of-view row in [Section 1](#sec-1) reads PASS but sample quality still appears low, check sample-wide transcript quality ([Section 2.3](#sec-2-3)): a uniformly low `pct_above_qv20` reflects the tissue baseline rather than a single-FoV defect. +::: + +```{python display-fov-summary-table} +#| echo: false + +_fov_summary = qc_metrics.get("fov_summary") or [] +_fov_eligible = qc_metrics.get("fov_eligible_count") + +if not _fov_summary: + display(Markdown("*Per-FoV summary not available — `fov_summary` key missing from `transcript_qc_metrics.json`. Re-run `TRANSCRIPT_QC_PROCESSING` against this sample to populate the table.*")) +elif _fov_eligible is not None and _fov_eligible < 2: + display(Markdown(f"*Single-FoV sample (`fov_eligible_count = {_fov_eligible}`) — consistency table not rendered. The field-of-view row in [Section 1](#sec-1) reads N/A.*")) +else: + # Sort by dev_z ascending so worst (most-negative) outliers appear first. + _fov_rows = sorted(_fov_summary, key=lambda r: r.get("dev_z", 0.0)) + _df_fov = pd.DataFrame( + [ + { + "FoV": r.get("fov_id", "—"), + "Transcripts": f"{int(r.get('n_transcripts', 0)):,}", + "Mean QV": f"{r.get('mean_qv', 0):.2f}" if r.get("mean_qv") is not None else "—", + "% above QV 20": f"{r.get('pct_above_qv20', 0):.2f}%" + if r.get("pct_above_qv20") is not None + else "—", + "dev_z (within sample)": f"{r.get('dev_z', 0):+.2f}" + if r.get("dev_z") is not None + else "—", + "Status": r.get("status", "N/A"), + "% unassigned": f"{r.get('pct_unassigned'):.1f}%" + if r.get("pct_unassigned") is not None + else "—", + "Transcript dev_z": f"{r.get('transcript_dev_z'):+.2f}" + if r.get("transcript_dev_z") is not None + else "—", + "Density status": r.get("transcript_status", "N/A"), + } + for r in _fov_rows + ] + ) + + # Scroll-cap at ~20 rows (≈30 px per row × 20 = 600 px); the rest is + # scrollable. Outliers come first (table is sorted by dev_z ascending). + # The scroll_max_height kwarg embeds the scroll wrapper inside the + # same HTML blob as the table — separate display() calls for the + # opening / closing
get auto-closed by Quarto before the + # table arrives, breaking the nesting. + display_status_table( + _df_fov, + status_cols=["Status", "Density status"], + narrow_cols={"FoV": "6em"}, + scroll_max_height="600px", + ) + + # Cross-link to the full per-FoV CSV (figures_source/quality_by_fov.csv) + display( + Markdown( + f"*Sorted by `dev_z` ascending — most-negative deviation first. " + f"Total: {len(_fov_summary)} FoVs ({_fov_eligible} eligible, " + f"n_transcripts ≥ {(qc_cutoffs.get('transcript_qc', {}) or {}).get('fov_consistency', {}).get('n_gate', 100)}).*" + ) + ) +``` + +### 3.2 QV by FoV (boxplot) {#sec-3-2} + +::: {.callout-tip collapse="true" title="💡 Interpretation help"} +**How to read FoV quality:** + +- One violin per FoV, showing the QV distribution of its gene transcripts only (control codewords are excluded). +- Per-FoV verdicts (PASS / WARN / FAIL) are reported in the [Section 3.1](#sec-3-1) table, sorted worst-first. The verdict is a within-sample relative flag: it marks FoVs whose fraction of high-QV transcripts is unusually low relative to the rest of the same sample, so a flagged FoV in an otherwise strong sample is a relative outlier rather than an absolute failure. +- A flagged FoV's distribution sits to the left (lower QV) of the others. Such FoVs warrant inspection before exclusion from secondary analysis. + +**Warning signs:** several FoVs whose distributions sit well left of QV = 20, or a wide spread in per-FoV QV. +::: + +```{python display-fov-quality} +#| echo: false +display(fig_html("figures/quality_by_fov.png")) +``` + +## 4. Feature-level metrics {#sec-4} + +Metrics partitioned by feature (gene or codeword). The noise-bound *threshold* itself is sample-level and lives in [Section 2.4](#sec-2-4); this section is the *distribution* of transcript counts across features. + +```{python display-section-4-summary} +#| echo: false + +# §4 has only one threshold-bearing row (gene/control shift); all rows +# link to the same subsection (§4.1), so the Detail column is dropped. +_s4_df = pd.DataFrame(build_section_4_summary(qc_metrics, qc_cutoffs)) +_s4_df = _s4_df.drop(columns=["Detail"], errors="ignore") +display_status_table( + _s4_df, + status_cols=["Status"], + narrow_cols={"Cutoffs": "26em"}, +) +``` + +### 4.1 Transcript count per feature {#sec-4-1} + +::: {.callout-note collapse="true" title="📖 Metric explanation"} +For each feature (gene or codeword), the total number of decoded transcripts across the sample. This distinguishes informative genes (above the noise bound from [Section 2.4](#sec-2-4)) from features indistinguishable from background. + +**Important note:** the noise bound is calculated from negative-control features only, but the plot shows all features (gene, genomic, negative-control, deprecated). +::: + +::: {.callout-tip collapse="true" title="💡 Interpretation help"} +**How to read the plot:** + +- X-axis: number of transcripts per feature (log scale) +- Y-axis: smoothed density of features at each transcript count (kernel density estimate) +- Different colours: gene vs non-gene categories +- Gray dashed line: noise bound from [Section 2.4](#sec-2-4) +- Good pattern: clear bimodal separation, most genes above the noise bound. + +**Problem signs:** uniform distribution, unclear separation, or many genes below the noise line — suggests poor signal-to-noise. +::: + +```{python display-transcripts-per-feature} +#| echo: false +display(fig_html("figures/num_transcripts_per_feature.png")) +``` + +### 4.2 Transcripts not assigned to a cell, per gene {#sec-4-2} + +::: {.callout-note collapse="true" title="📖 Metric explanation"} +For each gene, the fraction of its decoded transcripts that segmentation did not assign to any cell (`cell_id == "UNASSIGNED"`). The sample-wide rate is the row "Transcripts not assigned to a cell" in [Section 2](#sec-2); this plot breaks it down across genes. The dashed line marks the sample-wide rate. + +This is the segmentation cell-assignment rate, not the decoding "Unassigned codeword" metric in [Section 2.2](#sec-2-2) — the two measure different stages and differ by orders of magnitude. + +The sample-wide rate is **informational, not a pass/fail gate**. It is strongly tissue-driven: tissues with abundant extracellular or low-density transcript signal sit much higher than dense epithelial tissue, and across the calibration cohort the rate reaches ~59% at the 95th percentile for biological reasons. The value provides context for this tissue and is not a quality threshold. + +Genes with fewer than 10 observed transcripts are dropped (too sparse for a stable fraction). +::: + +::: {.callout-tip collapse="true" title="💡 Interpretation help"} +- An even spread, with most genes clustered near the sample-wide line, points to a uniform process (e.g. permeabilisation or diffusion affecting all transcripts). +- A long tail of genes far above the line points to specific genes leaking out of cells, which can be genuine extracellular biology or a probe issue. +- Inspect the high-unassigned genes before acting — some will be biologically real. +::: + +```{python display-unassigned-per-gene} +#| echo: false +display(fig_html("figures/unassigned_per_gene.png")) +``` + +## 5. Cell-level metrics {#sec-5} + +Per-cell metrics derived from the segmentation in `cells.parquet` and the feature-by-cell count matrix in `cell_feature_matrix.h5`. Most verdicts in this section are **advisory (PASS / WARN only)** — candidates for review, not sample-level filtering decisions. Cells with no in-plane nucleus is the exception: a high fraction (≥ 15%) reaches FAIL, because it points to a segmentation problem rather than the usual thin-section z-plane effect. + +::: {.callout-note collapse="true" title="📖 Metric explanation"} +The summary table above reports one advisory verdict per cell-level metric (PASS / WARN, plus a FAIL tier for cells with no in-plane nucleus). The metrics: + +- **Cell yield** — the total number of segmented cells. A small biopsy or partial tissue section can legitimately fall below the ~50k WARN floor, so treat it as advisory and read it against your tissue area. +- **Median cell size (px²)** — the typical segmented cell area. Too small suggests under-detection; too large suggests merged cells. +- **Nucleus RNA fraction** — the median share of a cell's assigned transcripts that overlap its nucleus, and the % of cells above 0.6. +- **Nucleus-to-cell area ratio** — the median ratio across cells, and the % of cells above 0.8. Flags over- or under-segmentation of nuclei relative to whole cells. +- **Cells with no in-plane nucleus** — % of cells that carry transcripts but have no nucleus in this section's image plane. In thin 2D sections the nucleus often sits in an adjacent plane, so a nonzero fraction is expected; a high fraction, with those cells carrying few transcripts, points to sectioning geometry rather than a segmentation fault. +- **Median transcripts per cell** — the typical decoded-transcript count per cell. +- **Median genes per cell** — the typical number of distinct genes detected per cell. +- **Cells segmented by stain** — % of cells whose boundary was drawn from a membrane/boundary stain rather than by expanding outward from the nucleus ("nucleus expansion"). Higher is better; PASS at ≥ 70% when stain segmentation was used. A nucleus-expansion run reports 0% and is shown as informational. + +Cell size, nucleus RNA fraction, nucleus-to-cell ratio, transcripts per cell and genes per cell each have a distribution plot in the subsections below, with its own explanation and reading guide. Cell yield, cells with no in-plane nucleus, and cells segmented by stain appear only in the table above. +::: + +```{python display-section-5-summary} +#| echo: false + +# Drop the Detail column — the master 📖 callout above already links +# each metric to its subsection, so per-row links in the summary table +# are redundant. +_s5_df = pd.DataFrame(build_section_5_summary(qc_metrics, qc_cutoffs)) +_s5_df = _s5_df.drop(columns=["Detail"], errors="ignore") +display_status_table( + _s5_df, + status_cols=["Status"], + narrow_cols={"Cutoffs": "22em"}, +) +``` + + +### 5.1 Cell size distribution {#sec-5-1} + +::: {.callout-note collapse="true" title="📖 Metric explanation"} +Each cell's segmented area (px²), one value per cell. The figure overlays vertical lines at the sample mean and median. +::: + +::: {.callout-tip collapse="true" title="💡 Interpretation help"} +The distribution is typically log-normal in shape, with a typical body of 50–500 px² in many tissues. Tails on either side carry diagnostic signal: + +- Very small cells (left tail) may be segmentation artefacts — incomplete cell boundaries. +- Very large cells (right tail) may be merged-cell artefacts — over-segmentation. + +Cross-check with the image QC report's Cell-level metrics section if outliers are abundant — they may indicate upstream segmentation issues. +::: + +::: {.panel-tabset} + +#### Linear scale + +```{python display-cell-size-linear} +#| echo: false +display(fig_html("figures/cell_size_distribution.png")) +``` + +#### Log scale + +```{python display-cell-size-log} +#| echo: false +display(fig_html("figures/cell_size_distribution_log.png")) +``` + +::: + +### 5.2 Nucleus RNA fraction per cell {#sec-5-2} + +::: {.callout-note collapse="true" title="📖 Metric explanation"} +For each cell, the fraction of its assigned transcripts that fall within the nucleus (over cells with at least one assigned transcript). +::: + +::: {.callout-tip collapse="true" title="💡 Interpretation help"} +Most mRNA is cytoplasmic, so the nuclear fraction is expected to be relatively low — typically 0.1–0.4. Deviations from that range have specific meanings: + +- Very high nuclear fractions (> 0.6) may indicate segmentation issues — nuclei boundaries too generous, or cytoplasm under-segmented. +- Very low nuclear fractions (< 0.05) may suggest poor nuclear detection. + +Tissue-type variation is real: lymphocytes and other small cells have proportionally more nuclear RNA than large epithelial cells. +::: + +```{python display-nucleus-rna-fraction} +#| echo: false +display(fig_html("figures/nucleus_transcript_fraction_per_cell_distribution.png")) +``` + +### 5.3 Nucleus-to-cell area ratio {#sec-5-3} + +::: {.callout-note collapse="true" title="📖 Metric explanation"} +Each cell's nucleus area divided by its total cell area. The figure shows the full per-cell distribution; the summary table above carries the per-sample median. +::: + +::: {.callout-tip collapse="true" title="💡 Interpretation help"} +A consistent ratio across cells indicates clean separation between nuclei and cytoplasm. The expected range is roughly 10–40 % of cell area, with the upper end common in tissues with small cells (brain neurons, lymphocytes, lung epithelium). + +- Ratios > 0.8 suggest over-segmentation of nuclei or under-segmentation of cells. +- Ratios < 0.05 suggest under-segmentation of nuclei or over-segmentation of cells. + +**Tissue-type caveat** — smaller cells have inherently higher nucleus-to-cell area ratios for biological reasons, not segmentation artefact. Cross-check with [Section 5.1](#sec-5-1) cell size distribution before flagging this as a quality issue. +::: + + +```{python display-nucleus-area-fraction} +#| echo: false +display(fig_html("figures/nucleus_to_cell_size_fraction_per_cell_distribution.png")) +``` + +### 5.4 Transcripts per cell {#sec-5-4} + +::: {.callout-note collapse="true" title="📖 Metric explanation"} +Total decoded transcripts per cell (summed across all genes). The histogram is log-scale. + +The grey dashed line is the **minimum-transcripts-per-cell threshold** (also reported in [Section 2.1](#sec-2-1)), derived from this sample's own distribution: in log10 counts, find the peak (mode) where most cells sit, measure how far the 99th percentile lies above that peak, and place the threshold the same distance *below* the peak — floored at 10 transcripts. Cells to the left of the line carry far fewer transcripts than the healthy-cell peak and are likely segmentation artefacts or debris. +::: + +::: {.callout-tip collapse="true" title="💡 Interpretation help"} +Distribution patterns to look for: + +- **Single peak** (good): one clear peak of healthy cells with hundreds to thousands of transcripts. +- **Two peaks** (warning): a main peak of healthy cells plus a left peak of poorly-segmented "cells" with very few transcripts. Indicates segmentation issues requiring refinement. + +Concerns: + +- Broad distributions without clear peaks → poor segmentation. +- Many cells with < 10 transcripts → likely artefacts dominate the population. +- No clear high-confidence cell population. +::: + +```{python display-transcripts-per-cell} +#| echo: false +display(fig_html("figures/num_transcripts_per_cell.png")) +``` + +### 5.5 Detected genes per cell {#sec-5-5} + +::: {.callout-note collapse="true" title="📖 Metric explanation"} +Number of distinct genes with at least one detected transcript in the cell. Log-scale histogram. + +The grey dashed line is the **minimum-genes-per-cell threshold** (also reported in [Section 2.1](#sec-2-1)), derived the same way as the transcripts threshold in [Section 5.4](#sec-5-4): from the peak of this sample's log10 gene-count distribution, reflected below the peak by the mode-to-99th-percentile distance, floored at 10 genes. Cells to the left of the line detect very few genes and are likely artefacts. +::: + +::: {.callout-tip collapse="true" title="💡 Interpretation help"} +This view complements [Section 5.4](#sec-5-4) (transcripts per cell): a cell can have many transcripts but few genes (over-expressed transcript on a small panel) or vice versa. + +- High-quality cells detect hundreds of different genes (50–500+ depending on panel). +- Low-quality cells detect very few genes (< 10–20), often segmentation artefacts. +- Biological variation is real: different cell types naturally express different numbers of genes — a small population of cells with low gene counts may be a real biological signal. + +Concerns: + +- Most cells detecting < 20 genes → poor sensitivity. +- Very broad distribution without a clear peak → segmentation issues. +- Large population of cells with < 5 genes → extensive artefacts. +::: + +```{python display-genes-per-cell} +#| echo: false +display(fig_html("figures/num_genes_per_cell.png")) +``` + +## 6. Metadata {#sec-6} + +```{python display-metadata} +#| echo: false + +_xb = XENIUM_BUNDLE if XENIUM_BUNDLE else "—" +_outdir = SAMPLE_PUBLISHED_OUTDIR if SAMPLE_PUBLISHED_OUTDIR else INDIR +_seg_sw = qc_metrics.get("segmentation_software") or "—" +# XOA (onboard analysis) version, emitted by transcript_qc_processing.py from the +# bundle's experiment.xenium. Strip the "xenium-" prefix; blank on bundles +# processed before this field was added (reprocess to populate). +_xoa_raw = qc_metrics.get("xoa_version") +_xoa_version = str(_xoa_raw).split("-", 1)[-1] if _xoa_raw else "— (not recorded; reprocess to populate)" +with open(versions_path, "r") as f: + _versions_text = f.read() + +_meta_html = f""" + + + + + + + + + + + + + + + + + + + + + + + + + +
Sample{_sample_name}
Xenium bundle{_xb}
Xenium Onboard Analysis (XOA) version{_xoa_version}
Segmentation software{_seg_sw}
Transcript QC output{_outdir}
Software versions
{_versions_text}
+""" +display(HTML(_meta_html)) +``` + +```{python authors-footer} +#| echo: false + +display( + HTML( + """ +
+Generated by nf-xenium-processing v0.1.0 — +pipeline maintained by Altos Labs Spatial Bioinformatics. +Authors: Malwina Prater, Dongze He, Nell Yu Nie & Felix Krueger. +
+""" + ) +) +``` +```{python versions} +#| echo: false + +# The nf-core quarto/notebook module requires the notebook to export the +# versions of the packages it uses to versions.csv (package,version — no +# header), which the module folds into the pipeline's versions topic. Read at +# runtime rather than hardcoded, per the pipeline's version-reporting rules. +from importlib.metadata import PackageNotFoundError, version as _pkg_version + +with open("versions.csv", "w", encoding="utf-8") as _fh: + for _pkg in ("numpy", "pandas", "ipython"): + try: + _fh.write(f"{_pkg},{_pkg_version(_pkg)}\n") + except PackageNotFoundError: + continue +``` diff --git a/bin/transcript_qc_processing.py b/bin/transcript_qc_processing.py new file mode 100755 index 00000000..baf1828e --- /dev/null +++ b/bin/transcript_qc_processing.py @@ -0,0 +1,1616 @@ +#!/usr/bin/env python3 + +""" +Transcript QC Processing Module +Performs all analysis from notebooks/1_qc_molecule.ipynb and generates figures and metrics. +Authors: Malwina Prater, mprater@altoslabs.com; Dongze He, dhe@altoslabs.com; Felix Krueger, fkrueger@altoslabs.com +""" + +import argparse +import json +import os +import sys +from pathlib import Path +from typing import Any, Dict, List, Optional + +import matplotlib.pyplot as plt +import numpy as np +import pandas as pd +import pyarrow.parquet as pq + +import transcript_stream +import scanpy as sc +import seaborn as sns + +# =========================================================================== +# Vendored VERBATIM from upstream ``xenium_helpers.utils`` +# (bin/xenium_helpers/src/xenium_helpers/utils.py) at nf-xenium-processing dev +# HEAD 5e35cae. This pipeline does not ship the xenium_helpers package, so the +# helpers this script needs are inlined here with their bodies unchanged. +# Re-sync: re-copy these definitions from that file; do NOT rename symbols +# inside this block (including the "mols"/"molecule" names) so it stays a +# mechanical copy. +# =========================================================================== + + +def calculate_noise_bound( + n_molecules_non_gene_prefix: pd.Series, quant: float = 0.99 +) -> tuple[float, float]: + """Calculate the noise bounds based on the non-gene molecules.""" + from scipy.stats import median_abs_deviation, norm + + if n_molecules_non_gene_prefix.empty: + return 0, 0 + + quant_val = norm.ppf(quant) + n_mols_log = np.log10(n_molecules_non_gene_prefix.values) + std = median_abs_deviation(n_mols_log, scale="normal") + noise_lb = np.mean(n_mols_log) - quant_val * std + noise_ub = np.mean(n_mols_log) + quant_val * std + return 10**noise_lb, 10**noise_ub + + +def estimate_min_mols_per_cell(n_mols_per_cell: List[int], min_value: int = 10): + n_mols_per_cell = np.log10(np.asarray(n_mols_per_cell) + 1) + nm_hist = np.histogram(n_mols_per_cell, bins=100) + mode = nm_hist[1][nm_hist[0].argmax()] + ci = np.quantile(n_mols_per_cell[n_mols_per_cell > mode], 0.99) - mode + return max(min_value, int(round(10 ** (mode - ci)))) + + +def format_yaml_like(data: dict, indent: int = 0) -> str: + # Borrowed from nf-core + """Formats a dictionary to a YAML-like string. + Args: + data (dict): The dictionary to format. + indent (int): The current indentation level. + Returns: + str: A string formatted as YAML. + """ + yaml_str = "" + for key, value in data.items(): + spaces = " " * indent + if isinstance(value, dict): + yaml_str += f"{spaces}{key}:\n{format_yaml_like(value, indent + 1)}" + else: + yaml_str += f"{spaces}{key}: {value}\n" + return yaml_str + + +def dump_versions( + file_path: str, + packages: List[str], + task_name: Optional[str] = None, + show: bool = True, + include_python: bool = True, +): + # Inspired by nf-core + from importlib.metadata import version + import platform + + versions: Dict[str, str] = {k: version(k) for k in packages} + + if include_python: + versions["python"] = platform.python_version() + + nested: Dict[str, Any] = ( + {task_name: versions} if task_name is not None else versions + ) + versions_yaml: str = format_yaml_like(nested) + + if show: + print(versions_yaml) + + with open(file_path, "w") as f: + f.write(versions_yaml) + + +# --------------------------------------------------------------------------- +# Segmentation-software provenance for QC reports. +# --------------------------------------------------------------------------- + +# Display names for pipeline resegmentation tools (raw params.segmentation +# value -> human-readable). Used as the name-only fallback when no parsed tool +# version is available. +SEGMENTATION_PRETTY = { + "cellpose": "Cellpose", + "cellpose_baysor": "Cellpose + Baysor", + "proseg": "Proseg", + "segger": "Segger", +} + +# Per-method component tools as (display name, versions.yml key) pairs. The key +# is the tool name as it appears inside the segmentation modules' versions.yml +# (e.g. ``cellpose: 3.0.6``). Order defines how multi-tool labels read. +SEGMENTATION_TOOL_KEYS = { + "cellpose": [("Cellpose", "cellpose")], + "cellpose_baysor": [("Cellpose", "cellpose"), ("Baysor", "baysor")], + "proseg": [("Proseg", "proseg")], + "segger": [("Segger", "segger")], +} + + +def parse_tool_versions(version_files) -> Dict[str, str]: + """Union a list of nf-core ``versions.yml`` files into a flat + ``{tool: version}`` map. Each file maps ``process -> {tool: version}``; we + flatten across processes (later entries win). Safe-fails per file. Versions + are run-global (one tool version per pipeline run), so collecting across all + samples/processes and flattening is correct.""" + import yaml # local import — pyyaml is a runtime dep, not needed at import + + out: Dict[str, str] = {} + for path in version_files or []: + try: + with open(path, encoding="utf-8") as f: + data = yaml.safe_load(f) or {} + except (OSError, yaml.YAMLError): + continue + if not isinstance(data, dict): + continue + for tools in data.values(): + if isinstance(tools, dict): + for tool, version in tools.items(): + if version is not None: + out[str(tool)] = str(version).strip() + return out + + +def _tool_label(display: str, key: str, tool_versions: Optional[Dict[str, str]]) -> str: + """``"Cellpose"`` + version -> ``"Cellpose v3.0.6"`` (name only if absent).""" + version = (tool_versions or {}).get(key) + return f"{display} v{version}" if version else display + + +def read_xenium_analysis_sw_version(bundle_dir) -> Optional[str]: + """Read ``analysis_sw_version`` from ``experiment.xenium`` (e.g. + ``"xenium-4.0.1.0"``). Returns ``None`` on missing file, missing key, or + malformed JSON. Mirrors ``read_xenium_pixel_size_um`` in bin/snr_metrics.py. + """ + exp = Path(bundle_dir) / "experiment.xenium" + if not exp.is_file(): + return None + try: + with open(exp, encoding="utf-8") as f: + meta = json.load(f) + version = meta.get("analysis_sw_version") + except (OSError, ValueError, TypeError, json.JSONDecodeError): + return None + if not version or not isinstance(version, str): + return None + return version + + +def _parse_xenium_version(analysis_sw_version: Optional[str]) -> Optional[str]: + """``"xenium-4.0.1.0"`` -> ``"4.0.1"`` (major.minor.patch). Returns ``None`` + if no leading numeric components can be parsed.""" + if not analysis_sw_version: + return None + tail = analysis_sw_version.split("-", 1)[-1] # drop a 'xenium-' style prefix + nums = [] + for part in tail.split("."): + if part.isdigit(): + nums.append(part) + else: + break + if not nums: + return None + return ".".join(nums[:3]) + + +def resolve_segmentation_software( + bundle_dir, + pipeline_segmentation: str = "skip", + is_resegmented: bool = False, + tool_versions: Optional[Dict[str, str]] = None, +) -> str: + """Human-readable label for the segmentation software that produced the + bundle a QC report describes. + + - Un-resegmented / pre-seg / ``skip``: the onboard analysis version from the + bundle's ``experiment.xenium`` -> ``"Xenium Onboard Analysis v4.0.1"``. + - Pipeline ``xr`` resegmentation: the reseg bundle's own + ``analysis_sw_version`` -> ``"Xenium Ranger v4.0.1 (resegmentation)"``. + - Other pipeline tools (cellpose / cellpose_baysor / proseg / segger): the + tool name plus its version from ``tool_versions`` (parsed from the + segmentation ``versions.yml``), e.g. ``"Cellpose v3.0.6"`` or + ``"Cellpose v3.0.6 + Baysor v0.6.2"``. Falls back to name-only when the + version is unavailable. Their reseg bundle is packaged via ``xeniumranger + import-segmentation``, so its ``experiment.xenium`` would mislabel them as + Xenium Ranger; the pipeline tool name is authoritative here. + """ + seg = (pipeline_segmentation or "skip").strip() + parsed = _parse_xenium_version(read_xenium_analysis_sw_version(bundle_dir)) + + if not is_resegmented or seg == "skip": + if parsed: + return f"Xenium Onboard Analysis v{parsed}" + return "Xenium Onboard Analysis (version unknown)" + + if seg == "xr": + if parsed: + return f"Xenium Ranger v{parsed} (resegmentation)" + return "Xenium Ranger (resegmentation)" + + components = SEGMENTATION_TOOL_KEYS.get(seg) + if components: + return " + ".join( + _tool_label(display, key, tool_versions) for display, key in components + ) + return SEGMENTATION_PRETTY.get(seg, seg) + + +# =========================================================================== +# End vendored xenium_helpers.utils block. +# =========================================================================== + +# Set plotting style +sns.set_theme(style="whitegrid") +plt.rcParams["figure.dpi"] = 300 + +# Transcript-to-cell assignment sentinel: the cell_id value for transcripts +# that decoding placed but segmentation did not assign to any cell. Distinct +# from the "unassigned" codeword decoding category. Confirmed as the literal +# string "UNASSIGNED" on real Xenium bundles. +BACKGROUND_CELL_ID = "UNASSIGNED" + +# Minimum observed transcript count for a gene to enter the per-gene +# unassigned-fraction breakdown (genes below this are too sparse for a stable +# fraction). Provisional — calibrate in the Phase B triage spike. +MIN_GENE_COUNT_FOR_UNASSIGNED = 10 + +# Cap on the per-FoV quality figure height, in inches. +_FOV_FIG_MAX_HEIGHT_IN = 40 + + +def _min_count_threshold(counts) -> int | None: + """Call-site guard for the vendored ``estimate_min_mols_per_cell``. + + That function is inside the vendored block and must stay byte-identical to + upstream, so the degenerate-input check lives here instead. It does + ``np.quantile(x[x > mode], 0.99)`` where ``mode`` is the left edge of the + modal histogram bin; for a constant (or empty) input the modal bin's left + edge equals the only value, the upper-tail slice is empty and numpy raises. + Since the bin edges are strictly increasing, ``max > mode`` always holds for + any non-constant input — so ``size == 0 or min == max`` is the exact + degenerate condition, not a heuristic. + + Reachable with zero cells, one cell, or zero retained genes (all counts 0). + Returns None in that case, meaning "threshold not computable": callers skip + the plot cutoff line and emit JSON ``null``, which the report already + renders as absent. Deliberately not a numeric fallback — any number here + would read as a real noise floor that empty data had passed. + """ + arr = np.asarray(counts, dtype=float).ravel() + if arr.size == 0 or arr.min() == arr.max(): + return None + return estimate_min_mols_per_cell(arr) + + +def _csv_float(value): + """Parse a metrics_summary.csv cell to float, or None when blank/unparseable. + Keeps "absent / blank" (None) distinct from a real 0.0 value. pandas reads a + blank cell as float NaN, so guard both NaN and the literal string 'nan'.""" + if value is None: + return None + if isinstance(value, float) and pd.isna(value): + return None + s = str(value).strip() + if s == "" or s.lower() == "nan": + return None + try: + return float(s) + except (ValueError, TypeError): + return None + + +def _csv_str(value): + """Parse a metrics_summary.csv cell to a non-empty string, or None. Treats a + blank cell (pandas NaN → 'nan') as absent, so an empty stain_definition is + correctly None rather than the string 'nan'.""" + if value is None: + return None + if isinstance(value, float) and pd.isna(value): + return None + s = str(value).strip() + return s if s and s.lower() != "nan" else None + + +def read_bundle_metrics(bundle_dir, is_resegmented): + """Read the three 10x-defined QC metrics from the bundle's + ``metrics_summary.csv`` (single data row). Returns a dict with float-or-None + values plus a ``source`` tag. + + Fail-soft to None when the file is missing, the column is absent or blank. + On resegmented bundles the CSV reflects the ONBOARD segmentation, not the + pipeline's resegmentation, so the segmentation-dependent metrics are not + meaningful — everything is set to None (N/A) in that case. + + ``stain_definition`` is carried through so the report can tell apart a run + that used stain segmentation (gate stain_frac) from a nucleus-expansion run + where stain_frac == 0 by design (informational, not a WARN). + """ + out = { + "nuclear_transcripts_per_100um2": None, + "fraction_empty_cells": None, + "segmented_cell_stain_frac": None, + "stain_definition": None, + "source": None, + } + if is_resegmented: + out["source"] = ( + "n/a (resegmented bundle; metrics_summary.csv reflects onboard segmentation)" + ) + return out + csv_path = Path(bundle_dir) / "metrics_summary.csv" + if not csv_path.is_file(): + return out + try: + row = pd.read_csv(csv_path, nrows=1).iloc[0].to_dict() + except (OSError, ValueError, pd.errors.ParserError, IndexError): + return out + out["nuclear_transcripts_per_100um2"] = _csv_float( + row.get("nuclear_transcripts_per_100um2") + ) + out["fraction_empty_cells"] = _csv_float(row.get("fraction_empty_cells")) + out["segmented_cell_stain_frac"] = _csv_float(row.get("segmented_cell_stain_frac")) + out["stain_definition"] = _csv_str(row.get("stain_definition")) + out["source"] = "metrics_summary.csv" + return out + + +# --------------------------------------------------------------------------- +# Host memory instrumentation +# +# This module reports progress with print(), which Python block-buffers when +# stdout is not a tty. An OOM SIGKILL therefore discards every buffered line: +# a killed task's log contains the kill message and nothing else, which is why +# this OOM had been "addressed" by climbing the memory ladder rather than +# diagnosed. The module .nf now exports PYTHONUNBUFFERED=1, and the stage +# markers below report where memory actually goes. +# +# Duplicated from image_qc.py rather than shared via xenium_helpers on purpose: +# module binaries ship from the repo through moduleBinaries, but xenium_helpers +# is pip-installed into the container image, so a helper placed there would not +# take effect until the image is rebuilt. Consolidate when the image is bumped. +# --------------------------------------------------------------------------- + +_GIB = float(1024**3) +_MEM_PEAK = {"working_set": 0.0, "rss": 0.0} + +# (usage file, stat file, inactive key, active key) for cgroup v2 then v1. +_CGROUP_SOURCES = ( + ( + "/sys/fs/cgroup/memory.current", + "/sys/fs/cgroup/memory.stat", + "inactive_file", + "active_file", + ), + ( + "/sys/fs/cgroup/memory/memory.usage_in_bytes", + "/sys/fs/cgroup/memory/memory.stat", + "total_inactive_file", + "total_active_file", + ), +) + + +def _cgroup_memory(): + """``(working_set_bytes, page_cache_bytes)`` for this cgroup, or None. + + The usage counter includes reclaimable page cache; ``usage - inactive_file`` + is what actually drives an OOM kill. Report both so they are never confused + when sizing the `processing` label. + """ + for usage_path, stat_path, inactive_key, active_key in _CGROUP_SOURCES: + try: + with open(usage_path) as fh: + usage = float(fh.read().strip()) + inactive_file = 0.0 + active_file = 0.0 + with open(stat_path) as fh: + for line in fh: + key, _, value = line.partition(" ") + if key == inactive_key: + inactive_file = float(value) + elif key == active_key: + active_file = float(value) + except (OSError, ValueError): + continue + return max(usage - inactive_file, 0.0), inactive_file + active_file + return None + + +def _process_rss(): + """Peak RSS of this process in bytes (VmHWM), excluding children.""" + try: + with open("/proc/self/status") as fh: + for line in fh: + if line.startswith("VmHWM:"): + return float(line.split()[1]) * 1024.0 + except (OSError, ValueError, IndexError): + return None + return None + + +def _log_mem(stage, extra=""): + """Print host memory at a stage boundary and track the high-water mark.""" + parts = [] + cgroup = _cgroup_memory() + if cgroup is not None: + working_set, page_cache = cgroup + _MEM_PEAK["working_set"] = max(_MEM_PEAK["working_set"], working_set) + parts.append( + f"working_set={working_set / _GIB:.1f}GB " + f"(+{page_cache / _GIB:.1f}GB reclaimable page cache)" + ) + rss = _process_rss() + if rss is not None: + _MEM_PEAK["rss"] = max(_MEM_PEAK["rss"], rss) + parts.append(f"peak_rss={rss / _GIB:.1f}GB") + if extra: + parts.append(extra) + if parts: + print(f"[MEM] {stage}: {' '.join(parts)}", flush=True) + + +def _log_mem_summary(): + """Final high-water summary — the number to size the memory request from.""" + print( + f"[MEM] PEAK working_set={_MEM_PEAK['working_set'] / _GIB:.1f}GB " + f"peak_rss={_MEM_PEAK['rss'] / _GIB:.1f}GB " + "(working_set excludes reclaimable page cache)", + flush=True, + ) + + +def _frame_gb(df): + """Deep memory of a DataFrame in GB, for the [MEM] extra field.""" + try: + return f"frame={df.memory_usage(deep=True).sum() / _GIB:.1f}GB rows={len(df):,}" + except Exception: + return "" + + +def read_random_parquet_row_groups( + parquet_file, num_row_groups=4, random_seed=42, columns=None +): + """ + Read a random subset of row groups from a parquet file. Each row group is a set of rows that are contiguous in the file. For 10x transcripts.parquet file, each row group has about 262,000 rows. + """ + np.random.seed(random_seed) + selected_row_groups = np.random.choice( + parquet_file.metadata.num_row_groups, size=num_row_groups, replace=False + ) + # Project columns here too: without `columns` this read pulled all ~20. + return parquet_file.read_row_groups( + selected_row_groups, columns=columns + ).to_pandas() + + +def scaled_noise_threshold( + nongene_feature_names, + n_nongene_total: int, + non_gene_prefixes: tuple[str, ...] = ("NegControl",), +) -> tuple[float, float, int, int]: + """Noise threshold on the FULL-table scale, from a sampled non-gene column. + + Returns ``(threshold, scale, n_sampled, n_total)``. + + `calculate_noise_bound` is fed negative-control counts drawn from a *sample* of at + most ``transcript_stream.DEFAULT_SAMPLE_ROWS`` non-gene rows, while the per-gene + counts it is compared against are exact over the whole table. The threshold therefore + has to be scaled up by how much of the non-gene population was sampled. + + The bug this exists to prevent: the scaling used to be ``num_transcripts / + num_selected_transcripts``, and once the streaming aggregation made + ``num_selected_transcripts`` the *full* row count, that ratio became exactly 1.0 and the + threshold silently stayed sample-scale. Measured on the reference sample: threshold + 323 where the scaled value is ~1,400, with 5,005 of 5,006 genes retained. The factor + must be over non-gene transcripts specifically, because that is the sampled population + -- scaling by total rows would overshoot by the gene fraction. + + Extracted from ``main()`` so the scaling is testable; a test against + ``transcript_stream`` alone cannot see it, and did not. + """ + sampled = nongene_feature_names + n_sampled = int(len(sampled)) + n_total = int(max(0, n_nongene_total)) + scale = (n_total / n_sampled) if n_sampled and n_total else 1.0 + # Only negative-control features feed the noise model — the report states this + # explicitly. The prefix set is a parameter rather than a literal so + # --non-gene-prefix actually takes effect; its default reproduces the + # previously hard-coded "NegControl". + _, threshold = calculate_noise_bound( + sampled[sampled.str.startswith(non_gene_prefixes)].value_counts() + ) + return float(threshold) * scale, scale, n_sampled, n_total + + +def main(): + parser = argparse.ArgumentParser(description="Transcript QC Processing") + parser.add_argument( + "--xenium-bundle-dir", required=True, help="Path to Xenium bundle directory" + ) + parser.add_argument("--outdir", required=True, help="Output directory") + # nargs="+" on both: modules/local/transcript_qc/main.nf splits the ";"- + # separated config value (documented in nextflow_schema.json) into separate + # bare tokens, e.g. `--non-gene-prefix 'A' 'B'`. Without nargs that form + # makes argparse exit 2. Neither value is read anywhere in this script, so + # the str-default/list-from-argv asymmetry is harmless. + parser.add_argument( + "--non-gene-prefix", + nargs="+", + # "NegControl" (not "NegControlProbe") so the default reproduces the + # previously hard-coded selection and matches the pipeline default. + default=["NegControl"], + help="Prefix(es) identifying non-gene features for the noise model", + ) + parser.add_argument( + "--stain-names", + nargs="+", + help="Stain names (unused but kept for compatibility)", + ) + parser.add_argument( + "--task-process", default="TRANSCRIPT_QC", help="Task process name" + ) + parser.add_argument( + "--num-row-groups", + type=int, + default=None, + help="Number of row groups to process", + ) + parser.add_argument( + "--threads", type=int, default=1, help="Number of threads (for compatibility)" + ) + parser.add_argument( + "--pipeline-segmentation", + default="skip", + help="Pipeline segmentation method (params.segmentation); 'skip' for none", + ) + parser.add_argument( + "--is-resegmented", + action="store_true", + help="Set when this run analyses a pipeline-resegmented bundle (post-seg)", + ) + parser.add_argument( + "--seg-versions-file", + nargs="*", + default=None, + help="Segmentation step versions.yml file(s); tool version parsed into " + "the segmentation-software label (post-seg pipeline tools)", + ) + + args = parser.parse_args() + + # argparse yields a list with nargs="+", so normalise to the tuple + # pandas' str.startswith expects for a multi-prefix test. + non_gene_prefixes = tuple( + args.non_gene_prefix + if isinstance(args.non_gene_prefix, (list, tuple)) + else [args.non_gene_prefix] + ) + + # Validate parameters + if args.xenium_bundle_dir is None or not os.path.exists(args.xenium_bundle_dir): + raise FileNotFoundError( + f'The given XENIUM_BUNDLE_DIR, "{args.xenium_bundle_dir}" doesn\'t exist' + ) + + XENIUM_BUNDLE_DIR = Path(args.xenium_bundle_dir) + # Create output directories + outdir = Path(args.outdir) + outdir.mkdir(parents=True, exist_ok=True) + output_fig_dir = outdir / "figures" + output_fig_dir.mkdir(parents=True, exist_ok=True) + figures_source_dir = outdir / "figures_source" + figures_source_dir.mkdir(parents=True, exist_ok=True) + output_metrics_path = outdir / "transcript_qc_metrics.json" + + NUM_ROW_GROUPS = args.num_row_groups + TASK_PROCESS = args.task_process + + transcripts_parquet_path = XENIUM_BUNDLE_DIR / "transcripts.parquet" + morphology_focus_dir = XENIUM_BUNDLE_DIR / "morphology_focus" + cells_parquet_path = XENIUM_BUNDLE_DIR / "cells.parquet" + cell_feature_matrix_h5_path = XENIUM_BUNDLE_DIR / "cell_feature_matrix.h5" + + # EXACT CODE FROM ORIGINAL NOTEBOOK - check if all required files are present + required_files = [ + transcripts_parquet_path, + morphology_focus_dir, + cell_feature_matrix_h5_path, + cells_parquet_path, + ] + for file in required_files: + if not os.path.exists(file): + print(f"Required file not found: {file}") + sys.exit(1) + + print("=== Transcript QC Processing ===") + print(f"Input directory: {XENIUM_BUNDLE_DIR}") + print(f"Output directory: {outdir}") + + # EXACT CODE FROM ORIGINAL NOTEBOOK - check if all required columns are present in the transcripts parquet file + transcripts_parquet = pq.ParquetFile(transcripts_parquet_path) + transcripts_parquet_columns = transcripts_parquet.schema.names + num_transcripts = transcripts_parquet.metadata.num_rows + print(f"Total number of transcripts: {num_transcripts:,}") + + # EXACT CODE FROM ORIGINAL NOTEBOOK - required columns + required_columns = [ + "cell_id", + "qv", + "fov_name", + "codeword_category", + "is_gene", + "feature_name", + # Per-transcript nuclear-overlap flag — drives the §5.2 nucleus RNA + # fraction (fraction of a cell's transcripts that overlap its nucleus). + "overlaps_nucleus", + ] + missing_columns = [ + col for col in required_columns if col not in transcripts_parquet_columns + ] + if missing_columns: + print(f"Missing required columns in transcripts.parquet: {missing_columns}") + sys.exit(1) + + # Stream the table rather than hold it resident. The eager read cost ~145 GB + # of RSS on the 1.33 B-row reference sample (117 B/row as object dtype) and + # drove the 120 -> 240 -> 480 GB retry ladder, yet every consumer below is a + # reduction: the largest output is per-cell at ~530 k rows. See + # transcript_stream for the measurements. + if NUM_ROW_GROUPS is not None: + print( + f"--num-row-groups ({NUM_ROW_GROUPS}) ignored: the streaming " + "aggregation reads the whole file at bounded memory, so subsampling " + "is no longer needed to avoid running out of it." + ) + NUM_ROW_GROUPS = None + + _tx = transcript_stream.aggregate_transcripts( + transcripts_parquet_path, BACKGROUND_CELL_ID + ) + _log_mem("after streaming aggregation") + + # The two QV violin plots are the only row-level consumers, and the previous + # code already capped them at 1e6 rows each -- it just got that cap by + # materialising every row and sampling down. + _SAMPLE_COLUMNS = ["qv", "codeword_category", "fov_name", "feature_name"] + df_spatial_gene, df_spatial_nongene = transcript_stream.sample_rows( + transcripts_parquet_path, + _SAMPLE_COLUMNS, + k=transcript_stream.DEFAULT_SAMPLE_ROWS, + ) + _log_mem("after violin sampling") + + num_selected_transcripts = _tx.n_rows + + if num_selected_transcripts != num_transcripts: + print( + f"Number of random transcripts selected for analysis: {num_selected_transcripts:,} (out of {num_transcripts:,})" + ) + + codeword_category_counts = _tx.codeword_counts + + print(f"Features: {_tx.n_features:,}") + print("\nFeature categories:") + for cc in codeword_category_counts.sort_values(ascending=False).keys(): + count = codeword_category_counts[cc] + percentage = count / num_selected_transcripts * 100 + print(f" {cc:<26} - {count:>12,} transcripts ({percentage:>6.3f}%)") + + # EXACT CODE FROM ORIGINAL NOTEBOOK - quality distribution plots + _log_mem("before quality plots") + print("\nGenerating quality distribution plots...") + + # df_spatial_gene / df_spatial_nongene were sampled during the streaming pass. + + # define hue order: genes first, then non-genes + codeword_categories_order = codeword_category_counts.index + codeword_categories_order = ( + codeword_categories_order[codeword_categories_order.str.endswith("gene")] + .sort_values(ascending=False) + .tolist() + + codeword_categories_order[~codeword_categories_order.str.endswith("gene")] + .sort_values() + .tolist() + ) + + df_spatial_quality = pd.concat([df_spatial_nongene, df_spatial_gene]) + + print( + f"Using {len(df_spatial_nongene):,} non-gene transcripts and {len(df_spatial_gene):,} gene transcripts for quality values (qv) distribution" + ) + + # Print total number of rows and rows per category in df_spatial_quality + _log_mem("after quality sampling") + print("\ntranscript per category:") + print(df_spatial_quality["codeword_category"].value_counts()) + + # Quality distribution density plot (Figure 1) — REMOVED 2026-05-22 + # per transcript QC v5 redesign feedback: redundant with the violin plot + # below, which shows the same per-category QV distribution but is + # easier to read (one violin per category, side-by-side, with median + # and quartiles marked inside each violin). + # The underlying df_spatial_quality data is still saved by the violin + # plot block below — figures_source/quality_distributions_comprehensive.csv. + + # Figure 2: Quality distribution by codeword category (violin plot) — + # axes swapped (was x=codeword_category/y=qv, now y=codeword_category/x=qv) + # so QV reads left-to-right. The low-QV tail then sits on the LEFT (matches + # the report prose) and the layout is consistent with the per-FoV QV violin + # below, which is already horizontal. Category labels move to the y-axis and + # read horizontally instead of rotated 45°. + fig = plt.figure(figsize=(10, 6)) + sns.violinplot( + data=df_spatial_quality, + y="codeword_category", + x="qv", + hue="codeword_category", + split=False, + inner="box", + palette="husl", + density_norm="width", + legend=False, + order=codeword_categories_order, + ) + plt.ylabel("Codeword Category", fontsize=12) + plt.xlabel("Quality Value (qv)", fontsize=12) + plt.title( + "Distribution of Transcript Quality by Codeword Category", fontsize=14, pad=20 + ) + plt.tight_layout() + # vertical reference line at qv=20 (matches horizontal-violin layout) + plt.axvline(x=20, color="grey", linestyle="--") + plt.savefig( + output_fig_dir / "quality_distributions_comprehensive.pdf", + dpi=300, + bbox_inches="tight", + ) + plt.savefig( + output_fig_dir / "quality_distributions_comprehensive.png", + dpi=300, + bbox_inches="tight", + ) + plt.close(fig) + + # Save data for quality distributions comprehensive + df_spatial_quality.to_csv( + figures_source_dir / "quality_distributions_comprehensive.csv", index=False + ) + + # Save df_spatial_quality to CSV + output_file = outdir / "df_spatial_quality.csv" + df_spatial_quality.to_csv(output_file, index=False) + print(f"Saved df_spatial_quality to: {output_file}") + + # ===== Per-FoV QV-consistency stats (computed here so the §3.2 figure can + # colour each FoV by its verdict; reused later for fov_summary — one + # groupby over df_spatial, single source of truth). ===== + # Spike-validated design at plans/2026-05-22_SPIKE_fov_zscore_denominator.py + # against 11 calibration samples (brain/lung/liver/skin/pancreas). + # + # Per-FoV stats — computed from df_spatial (full, not subsampled gene view). + # Within-sample z-score on pct_above_qv20, one-sided (low outliers only). + # Per-FoV median QV is deliberately NOT computed: Xenium QV caps at 40 so + # the median saturates in every healthy FoV and carries no signal. + _N_GATE_FOV = 100 # FoVs with fewer transcripts → N/A (sampling noise dominates) + _Z_FAIL = 3.0 + _Z_WARN = 2.0 + + # From the streaming pass. qv is accumulated in float64 there, so these means + # are marginally more accurate than a pandas float32 column mean. + _fov_n = _tx.fov["n"] + _fov_mean = _tx.fov["mean_qv"] + _fov_pct_above = _tx.fov["pct_above_qv20"] + + _eligible_mask = _fov_n >= _N_GATE_FOV + _eligible_pct = _fov_pct_above[_eligible_mask] + + # _baseline_ok: a within-sample median/stdev can actually be computed + # from the eligible FoVs (need ≥2 eligible FoVs with non-zero stdev). + # When False, dev_z is meaningless and per-FoV status falls through + # to N/A — see _per_fov_status below. + _baseline_ok = len(_eligible_pct) >= 2 and _eligible_pct.std() > 0 + if _baseline_ok: + _ref_median = float(_eligible_pct.median()) + _ref_stdev = float(_eligible_pct.std()) + _dev_z = (_fov_pct_above - _ref_median) / _ref_stdev + else: + _dev_z = pd.Series(np.zeros(len(_fov_pct_above)), index=_fov_pct_above.index) + + def _per_fov_status(n: int, dev: float) -> str: + if n < _N_GATE_FOV: + return "N/A" + # If there's no within-sample baseline (single FoV, all-equal pct, + # etc.), the dev_z above is 0.0 by fallback — meaningless. Don't + # report PASS in that case; emit N/A so downstream consumers of + # fov_summary[i].status see honest data. + if not _baseline_ok: + return "N/A" + # One-sided: only low outliers (negative dev_z) flag. + # Unusually high QV in a FoV is not a quality concern. + if dev < -_Z_FAIL: + return "FAIL" + if dev < -_Z_WARN: + return "WARN" + return "PASS" + + _fov_qv_status = { + str(_fov): _per_fov_status(int(_fov_n.loc[_fov]), float(_dev_z.loc[_fov])) + for _fov in _fov_n.index + } + + # Figure 3: Quality distribution by Field of View — axes swapped (was + # x=fov_name/y=qv, now y=fov_name/x=qv). Xenium slides can carry 100+ + # FoVs; vertical violins crammed the labels and made the QV + # distributions unreadable. Horizontal layout gives each FoV one row + # and the figure height scales with FoV count. + _n_fov = max(1, df_spatial_gene["fov_name"].nunique()) + # Bounded: this is the only artefact whose size grows without limit with the + # sample. At dpi 300 an unbounded 0.22*n_fov height renders a 3000 x 59,400 px + # canvas on a 900-FoV slide, and bbox_inches="tight" renders it twice; the PNG + # is then base64-embedded by the report step. + _fov_fig_height = min(_FOV_FIG_MAX_HEIGHT_IN, max(6, 0.22 * _n_fov)) + fig = plt.figure(figsize=(10, _fov_fig_height)) + # Single-colour violins. Per-FoV verdicts are reported in the §3.1 table + # (sorted worst-first); colouring the violins by verdict was dropped — the + # flagged FoVs are a handful among 100+ and were not visible on this tall, + # thin-violin plot. + sns.violinplot( + data=df_spatial_gene, + y="fov_name", + x="qv", + color="steelblue", + inner="box", + density_norm="width", + ) + plt.ylabel("Field of View", fontsize=12) + plt.xlabel("Quality Value (qv)", fontsize=12) + plt.title( + "Distribution of Transcript Quality by Field of View\n(gene transcripts only)", + fontsize=14, + pad=20, + ) + plt.tight_layout() + # vertical reference line at qv=20 (matches horizontal-violin layout) + plt.axvline(x=20, color="grey", linestyle="--") + plt.savefig(output_fig_dir / "quality_by_fov.pdf", dpi=300, bbox_inches="tight") + plt.savefig(output_fig_dir / "quality_by_fov.png", dpi=300, bbox_inches="tight") + plt.close(fig) + + # Save data for quality by fov + df_spatial_gene.to_csv(figures_source_dir / "quality_by_fov.csv", index=False) + + del df_spatial_quality, df_spatial_gene + + # Noise threshold from the negative-control features. + # + # SCALE MISMATCH, fixed here. The threshold is derived from `df_spatial_nongene`, + # which is a *sample* of at most transcript_stream.DEFAULT_SAMPLE_ROWS non-gene rows, + # but it is compared against `_tx.gene["n_molecules"]`, which the streaming + # aggregation computes exactly over the whole table. The two must be brought onto one + # scale. + # + # The original line multiplied the threshold by `num_transcripts / + # num_selected_transcripts`. That worked when both sides came from the same subsampled + # frame, but `num_selected_transcripts` is now `_tx.n_rows` -- the full row count -- so + # the ratio is exactly 1.0 and the threshold stayed sample-scale. Observed on the + # reference sample: total_transcripts == selected_transcripts == 1,325,798,498, threshold + # 323 where the correctly scaled value is ~1,400, and 5,005 of 5,006 genes retained. + # + # The right factor is over *non-gene* transcripts specifically, since that is the + # population sampled. + n_mols_threshold, nongene_scale, n_nongene_sampled, n_nongene_total = ( + scaled_noise_threshold( + df_spatial_nongene["feature_name"], + n_nongene_total=max(0, _tx.n_rows - _tx.n_gene_rows), + non_gene_prefixes=non_gene_prefixes, + ) + ) + print( + f"Noise threshold for genes' transcript count: {n_mols_threshold:.0f} transcripts " + f"(from {n_nongene_sampled:,} of {n_nongene_total:,} non-gene transcripts, " + f"scale x{nongene_scale:.2f})" + ) + + # No rescaling: the streaming aggregation counts every row, so these are already + # full-scale. Multiplying by num_transcripts / num_selected_transcripts here was a no-op + # (that ratio is 1.0) and would double-count if the aggregation ever became partial. + n_mols_per_gene_df = _tx.gene[["n_molecules", "is_gene"]].copy() + + # Figure 4: Distribution of transcripts per feature — smooth density (KDE) + # by gene vs non-gene. The binned histogram was dropped; only the density + # curve is shown, so the y-axis is density rather than a feature count. + fig = plt.figure(figsize=(8, 4)) + sns.kdeplot( + data=n_mols_per_gene_df, + x="n_molecules", + hue="is_gene", + log_scale=True, + fill=True, + alpha=0.4, + ax=plt.gca(), + ) + plt.xlabel("Num. transcripts") + plt.ylabel("Density") + plt.axvline(x=n_mols_threshold, color="grey", linestyle="--") + plt.title("Distribution of transcripts per feature", fontsize=14, pad=20) + plt.tight_layout() + plt.savefig( + f"{output_fig_dir}/num_transcripts_per_feature.pdf", + dpi=300, + bbox_inches="tight", + ) + plt.savefig( + f"{output_fig_dir}/num_transcripts_per_feature.png", + dpi=300, + bbox_inches="tight", + ) + plt.close(fig) + + # Save data for transcripts per feature + df_transcripts_per_feature = n_mols_per_gene_df.copy() + df_transcripts_per_feature["n_mols_threshold"] = n_mols_threshold + df_transcripts_per_feature.to_csv( + figures_source_dir / "num_transcripts_per_feature.csv", index=True + ) + + retained_genes = n_mols_per_gene_df.query( + "n_molecules > @n_mols_threshold and is_gene == True" + ) + print( + f"Number of genes with a total transcript count higher than the threshold : {len(retained_genes):,}" + ) + retained_genes.to_csv(outdir / "retained_genes.csv", index=True) + + # EXACT CODE FROM ORIGINAL NOTEBOOK - Cell size distribution + _log_mem("before cell statistics") + print("\nGenerating cell statistics plots...") + + # Check available columns and use appropriate cell area column + available_columns = pq.ParquetFile(cells_parquet_path).schema.names + cell_area_column = "cell_area" if "cell_area" in available_columns else "volume" + cells_parquet = pd.read_parquet( + cells_parquet_path, columns=["cell_id", cell_area_column] + ) + # Rename the column for consistency + cells_parquet.rename(columns={cell_area_column: "cell_size"}, inplace=True) + + # Figure 5: Cell size distribution. Rendered on both a linear and a log + # x-axis so the §5.1 report tabset can flip between them: linear shows the + # raw spread (a few very large cells stretch the axis and compress the + # body), while log makes the log-normal body a readable bell and exposes + # both tails (small = segmentation artefacts, large = merged cells). + # Figure size + title style aligned with §5.4 / §5.5 (figsize=(8,4), + # title fontsize=14, pad=20) for uniform appearance across cell-level plots. + _mean_cs = cells_parquet["cell_size"].mean() + _median_cs = cells_parquet["cell_size"].median() + for _scale, _suffix in (("linear", ""), ("log", "_log")): + fig = plt.figure(figsize=(8, 4)) + sns.kdeplot( + data=cells_parquet, + x="cell_size", + fill=True, + color="skyblue", + alpha=0.5, + log_scale=(_scale == "log"), + ) + plt.xlabel("Cell size (pixels²)") + plt.ylabel("Density") + _scale_label = " (log scale)" if _scale == "log" else "" + plt.title(f"Distribution of cell size{_scale_label}", fontsize=14, pad=20) + + # Add vertical lines for mean and median + plt.axvline(_mean_cs, color="red", linestyle="--", label="Mean") + plt.axvline(_median_cs, color="green", linestyle="--", label="Median") + plt.legend() + plt.tight_layout() + + # Save the plot (filename matches actual content — cell size; renamed + # from genes_per_cell_distribution.* on 2026-05-22 per transcript QC v5 + # filename-content alignment pass). The "_log" suffix is the log-x twin. + plt.savefig( + os.path.join(outdir, "figures", f"cell_size_distribution{_suffix}.pdf"), + dpi=300, + bbox_inches="tight", + ) + plt.savefig( + os.path.join(outdir, "figures", f"cell_size_distribution{_suffix}.png"), + dpi=300, + bbox_inches="tight", + ) + plt.close(fig) + + # Save data for cell size distribution + df_cell_size = cells_parquet[["cell_id", "cell_size"]].copy() + df_cell_size["mean_cell_size"] = cells_parquet["cell_size"].mean() + df_cell_size["median_cell_size"] = cells_parquet["cell_size"].median() + df_cell_size.to_csv(figures_source_dir / "cell_size_distribution.csv", index=False) + # Capture per-sample aggregates for §5.1 status row (advisory PASS/WARN). + median_cell_size = ( + float(cells_parquet["cell_size"].median()) if len(cells_parquet) > 0 else None + ) + del cells_parquet + + # Cells read — nucleus_count drives the §5 "cells with no in-plane nucleus" + # metric (a count of nuclei == 0). The total_counts > 0 filter sets that + # denominator (cells carrying transcripts). Kept even though the nucleus RNA + # fraction below no longer comes from this read. + cells_parquet = pd.read_parquet( + cells_parquet_path, + columns=["nucleus_count", "total_counts"], + filters=[("total_counts", ">", 0)], + ) + + # Metric 2 (SegTraQ Phase A): fraction of cells with no in-plane nucleus. + # Uses nucleus_count == 0, NOT nucleus_area == 0 — no-nucleus cells still + # carry nonzero nucleus_area, so the area test silently returns ~0% on + # every sample. Denominator is the total_counts > 0 population this read is + # filtered to (i.e. cells carrying transcripts). + if "nucleus_count" in cells_parquet.columns: + pct_cells_no_nucleus = float((cells_parquet["nucleus_count"] == 0).mean() * 100) + else: + pct_cells_no_nucleus = None + del cells_parquet + + # 10x-defined gate: % of segmented cells with zero gene transcripts. + # Derived in-house from cells.parquet transcript_counts over ALL cells + # (single matched cell base — numerator and denominator from the same + # module). Verified to reproduce 10x's metrics_summary.csv + # fraction_empty_cells exactly on the XOA-3.2 and 4.0 example bundles, so it + # sidesteps the cross-module derivation that yields impossible fractions + # (qc_threshold_refinement report §3.3). Works on resegmented bundles too, + # since it uses this run's actual cells.parquet rather than the onboard CSV. + _cell_cols = pq.ParquetFile(cells_parquet_path).schema.names + if "transcript_counts" in _cell_cols: + _zt = pd.read_parquet(cells_parquet_path, columns=["transcript_counts"]) + pct_cells_zero_transcripts = ( + float((_zt["transcript_counts"] == 0).mean() * 100) + if len(_zt) > 0 + else None + ) + del _zt + else: + pct_cells_zero_transcripts = None + + # Figure 6 / §5.2: Nucleus RNA fraction per cell — the fraction of a cell's + # assigned transcripts that overlap its nucleus. Derived from transcripts + # (overlaps_nucleus), NOT from cells.parquet nucleus_count: nucleus_count is + # a count of nuclei (0/1/2), so the old nucleus_count/(total_counts+1) + # collapsed to ~1/total_counts — a latent bug, not a nuclear fraction. + # Matches the MultiQC report definition: sum(overlaps_nucleus)/count per + # cell_id over all assigned transcripts (cell_id != UNASSIGNED; no QV or + # is_gene filter). Distribution is over cells with >= 1 assigned transcript. + # Full-read only — a cell's transcripts span FoVs/row groups, so per-cell + # fractions are unreliable under --num-row-groups; emit N/A there. + # Streamed: the sentinel is filtered before aggregation, so it never becomes + # a group, and the fraction is sum(overlaps)/count per cell. + nucleus_count_fraction = _tx.cell_nucleus_fraction + + # Figure + source CSV + summary are full-read only — skip entirely under + # subsampling (a misleading KDE next to an N/A table would be worse than + # nothing). Figure 6 size/title style aligned with §5.1 / §5.4 / §5.5. + _none_summary = { + "mean": None, + "median": None, + "stdev": None, + "pct_cells_above_60pct": None, + "pct_cells_below_5pct": None, + } + if nucleus_count_fraction is not None and len(nucleus_count_fraction) > 0: + fig = plt.figure(figsize=(8, 4)) + sns.kdeplot(x=nucleus_count_fraction, fill=True, color="skyblue", alpha=0.5) + plt.xlabel("Nucleus RNA fraction per cell") + plt.ylabel("Density") + plt.title( + "Empty plot — no nuclear transcripts detected" + if not (nucleus_count_fraction > 0).any() + else "Distribution of nucleus RNA fraction per cell", + fontsize=14, + pad=20, + ) + plt.tight_layout() + plt.savefig( + os.path.join( + outdir, + "figures", + "nucleus_transcript_fraction_per_cell_distribution.pdf", + ), + dpi=300, + bbox_inches="tight", + ) + plt.savefig( + os.path.join( + outdir, + "figures", + "nucleus_transcript_fraction_per_cell_distribution.png", + ), + dpi=300, + bbox_inches="tight", + ) + plt.close(fig) + + # Save data for nucleus RNA fraction (one row per cell). + pd.DataFrame( + { + "cell_id": nucleus_count_fraction.index, + "nucleus_fraction": nucleus_count_fraction.values, + } + ).to_csv( + figures_source_dir + / "nucleus_transcript_fraction_per_cell_distribution.csv", + index=False, + ) + + # Per-sample aggregates for the §5.2 status row (advisory PASS/WARN). + _ncf = nucleus_count_fraction.dropna() + if len(_ncf) > 0: + nucleus_transcript_fraction_summary = { + "mean": float(_ncf.mean()), + "median": float(_ncf.median()), + "stdev": float(_ncf.std()), + "pct_cells_above_60pct": float((_ncf > 0.60).mean() * 100), + "pct_cells_below_5pct": float((_ncf < 0.05).mean() * 100), + } + else: + nucleus_transcript_fraction_summary = dict(_none_summary) + else: + # Subsampled run (or no assigned cells): full-read-only metric → N/A. + nucleus_transcript_fraction_summary = dict(_none_summary) + + # EXACT CODE FROM ORIGINAL NOTEBOOK - Figure 7: Nucleus-to-cell area fraction + cells_parquet = pd.read_parquet( + cells_parquet_path, columns=["nucleus_area", "cell_area"] + ) + nucleus_size_fraction = cells_parquet["nucleus_area"] / ( + cells_parquet["cell_area"] + 1 + ) + + # Figure 7 size + title style aligned with §5.1 / §5.2 / §5.4 / §5.5 + # for uniform appearance. + fig = plt.figure(figsize=(8, 4)) + sns.kdeplot(x=nucleus_size_fraction, fill=True, color="skyblue", alpha=0.5) + plt.xlabel("Nucleus to cell area ratio") + plt.ylabel("Density") + plt.title( + # Same fix as §5.2: show the distribution unless NO cell has a + # positive nucleus-to-cell ratio. `.all() != 0` was inverted — it read + # False (→ "Empty") whenever any single cell had a zero ratio. + "Distribution of nucleus-to-cell area ratio" + if (nucleus_size_fraction.fillna(0) > 0).any() + else "Empty plot — no nucleus transcripts detected", + fontsize=14, + pad=20, + ) + plt.tight_layout() + + # Save the plot + plt.savefig( + os.path.join( + outdir, "figures", "nucleus_to_cell_size_fraction_per_cell_distribution.pdf" + ), + dpi=300, + bbox_inches="tight", + ) + plt.savefig( + os.path.join( + outdir, "figures", "nucleus_to_cell_size_fraction_per_cell_distribution.png" + ), + dpi=300, + bbox_inches="tight", + ) + plt.close(fig) + + # Save data for nucleus to cell size fraction + df_nucleus_size_fraction = pd.DataFrame( + { + "nucleus_area": cells_parquet["nucleus_area"], + "cell_area": cells_parquet["cell_area"], + "nucleus_size_fraction": nucleus_size_fraction, + } + ) + df_nucleus_size_fraction.to_csv( + figures_source_dir / "nucleus_to_cell_size_fraction_per_cell_distribution.csv", + index=False, + ) + del cells_parquet + + # EXACT CODE FROM ORIGINAL NOTEBOOK - Load cell feature matrix + ad = sc.read_10x_h5(cell_feature_matrix_h5_path) + # filter for retained genes + ad = ad[:, ad.var_names.isin(retained_genes.index)] + print(f"AnnData object with n_obs × n_vars = {ad.shape[0]} × {ad.shape[1]}") + + # EXACT CODE FROM ORIGINAL NOTEBOOK - Figure 8: Distribution of transcripts per cell + n_mols_per_cell = ad.X.sum(axis=1).A1 + n_mols_threshold_cell = _min_count_threshold(n_mols_per_cell) + + fig = plt.figure(figsize=(8, 4)) + sns.histplot(n_mols_per_cell, log_scale=True, bins=50, ax=plt.gca()) + plt.xlabel("Num. transcripts") + plt.ylabel("Num. cells") + if n_mols_threshold_cell is not None: + plt.axvline(x=n_mols_threshold_cell, color="grey", linestyle="--") + plt.title("Distribution of transcripts per Cell", fontsize=14, pad=20) + plt.tight_layout() + plt.savefig( + f"{output_fig_dir}/num_transcripts_per_cell.pdf", dpi=300, bbox_inches="tight" + ) + plt.savefig( + f"{output_fig_dir}/num_transcripts_per_cell.png", dpi=300, bbox_inches="tight" + ) + plt.close(fig) + print(f"Threshold for transcripts per cell: {n_mols_threshold_cell}") + + # Save data for transcripts per cell + df_transcripts_per_cell = pd.DataFrame( + { + "n_transcripts_per_cell": n_mols_per_cell, + "n_mols_threshold_cell": n_mols_threshold_cell, + } + ) + df_transcripts_per_cell.to_csv( + figures_source_dir / "num_transcripts_per_cell.csv", index=False + ) + + # Convert numpy array to pandas DataFrame + n_mols_per_cell_df = pd.DataFrame(n_mols_per_cell, columns=["num_of_transcripts"]) + # Filename matches actual content — transcripts; renamed from + # num_transcripts_per_cell.csv on 2026-05-22 (CSV holds num_of_transcripts + # column, not transcripts/genes). + output_file = os.path.join(outdir, "num_transcripts_per_cell.csv") + n_mols_per_cell_df.to_csv(output_file, index=False) + print(f"Saved transcripts-per-cell distribution to: {output_file}") + + # EXACT CODE FROM ORIGINAL NOTEBOOK - Figure 9: Distribution of genes per cell + n_genes_per_cell = (ad.X != 0).sum(axis=1).A1 + n_genes_threshold = _min_count_threshold(n_genes_per_cell) + + fig = plt.figure(figsize=(8, 4)) + sns.histplot(n_genes_per_cell, log_scale=True, bins=50, ax=plt.gca()) + plt.xlabel("Num. genes") + plt.ylabel("Num. cells") + if n_genes_threshold is not None: + plt.axvline(x=n_genes_threshold, color="grey", linestyle="--") + plt.title("Distribution of number of detected genes per Cell", fontsize=14, pad=20) + plt.tight_layout() + # Filenames match actual content — genes per cell; renamed from + # num_transcripts_per_cell.* on 2026-05-22 (xlabel reads "Num. genes", + # title is "Distribution of number of detected genes per Cell"). + plt.savefig( + f"{output_fig_dir}/num_genes_per_cell.pdf", dpi=300, bbox_inches="tight" + ) + plt.savefig( + f"{output_fig_dir}/num_genes_per_cell.png", dpi=300, bbox_inches="tight" + ) + plt.close(fig) + + # Save data for genes per cell — figures_source/ version is the + # richer pandas DataFrame (data + threshold). The bare numpy dump + # at outdir-root (num_genes_per_cell.csv, line ~642 below) keeps + # the legacy "single-column raw" name. + df_genes_per_cell = pd.DataFrame( + {"n_genes_per_cell": n_genes_per_cell, "n_genes_threshold": n_genes_threshold} + ) + df_genes_per_cell.to_csv(figures_source_dir / "num_genes_per_cell.csv", index=False) + + # Convert numpy array to pandas DataFrame + n_genes_per_cell_df = pd.DataFrame(n_genes_per_cell, columns=["num_of_genes"]) + # Save to CSV using os.path.join for path handling + output_file = os.path.join(outdir, "num_genes_per_cell.csv") + n_genes_per_cell_df.to_csv(output_file, index=False) + print(f"Saved gene distribution to: {output_file}") + + # ===== transcript QC v5 redesign — Phase 0 schema extension (2026-05-22) ===== + # Spike-validated design at plans/2026-05-22_SPIKE_fov_zscore_denominator.py + # against 11 calibration samples (brain/lung/liver/skin/pancreas). + + # Sample-wide QV summary, from the streaming pass. + pct_transcripts_above_qv20 = float(_tx.pct_qv_above_threshold) + mean_qv = float(_tx.mean_qv) + + # Per-FoV QV-consistency stats (_N_GATE_FOV, _Z_FAIL, _Z_WARN, _fov_n, + # _fov_mean, _fov_pct_above, _eligible_mask, _baseline_ok, _dev_z, + # _per_fov_status, _fov_qv_status) are computed earlier, just before the + # §3.2 QV-by-FoV figure, so that plot can colour each FoV by its verdict. + # They are reused here — single source of truth, one groupby over df_spatial. + + # Metric 3 (SegTraQ Phase A): per-FoV transcript density + unassigned%. + # Full-read only — row groups are FoV-ordered, so under --num-row-groups + # whole FoVs drop and per-FoV counts are no longer comparable; emit N/A. + # _fov_n is the transcript count per FoV. Same one-sided z machinery as QV. + # The _N_GATE_FOV gate marks FoVs below 100 transcripts N/A: those are + # empty/edge tissue, not dropouts. An in-tissue dropout still carries + # hundreds of transcripts, just fewer than its neighbours, so it stays + # eligible and can flag. + _fov_unassigned_pct = _tx.fov["pct_unassigned"] + + _mol_eligible = _fov_n[_eligible_mask] + _mol_baseline_ok = ( + NUM_ROW_GROUPS is None and len(_mol_eligible) >= 2 and _mol_eligible.std() > 0 + ) + if _mol_baseline_ok: + _mol_ref_median = float(_mol_eligible.median()) + _mol_ref_stdev = float(_mol_eligible.std()) + _mol_z = (_fov_n - _mol_ref_median) / _mol_ref_stdev + else: + _mol_z = pd.Series(np.zeros(len(_fov_n)), index=_fov_n.index) + + def _per_fov_transcript_status(n: int, dev: float) -> str: + if NUM_ROW_GROUPS is not None: + return "N/A" + if n < _N_GATE_FOV: + return "N/A" + if not _mol_baseline_ok: + return "N/A" + # One-sided: only low-density FoVs (dropouts) flag. + if dev < -_Z_FAIL: + return "FAIL" + if dev < -_Z_WARN: + return "WARN" + return "PASS" + + _subsampled = NUM_ROW_GROUPS is not None + fov_summary = [ + { + "fov_id": str(_fov), + "n_transcripts": int(_fov_n.loc[_fov]), + "mean_qv": float(_fov_mean.loc[_fov]), + "pct_above_qv20": float(_fov_pct_above.loc[_fov]), + "dev_z": float(_dev_z.loc[_fov]), + "status": _fov_qv_status[str(_fov)], + # SegTraQ Phase A per-FoV additions (N/A under subsampling) + "transcript_dev_z": (None if _subsampled else float(_mol_z.loc[_fov])), + "transcript_status": _per_fov_transcript_status( + int(_fov_n.loc[_fov]), float(_mol_z.loc[_fov]) + ), + "pct_unassigned": ( + None if _subsampled else float(_fov_unassigned_pct.loc[_fov]) + ), + } + for _fov in _fov_n.index + ] + fov_eligible_count = int(_eligible_mask.sum()) + if fov_eligible_count >= 2: + fov_outlier_count = sum(1 for r in fov_summary if r["status"] == "FAIL") + fov_warn_count = sum(1 for r in fov_summary if r["status"] == "WARN") + fov_pct_fail = 100.0 * fov_outlier_count / fov_eligible_count + fov_pct_warn = 100.0 * fov_warn_count / fov_eligible_count + else: + fov_outlier_count = None + fov_warn_count = None + fov_pct_fail = None + fov_pct_warn = None + + # Per-FoV transcript-density dropout rollup (SegTraQ Phase A). Counts FoVs + # flagged WARN or FAIL on low density as the sample-level dropout fraction + # (advisory). N/A under subsampling or with too few eligible FoVs. + if not _subsampled and fov_eligible_count >= 2: + fov_density_dropout_count = sum( + 1 for r in fov_summary if r["transcript_status"] in ("FAIL", "WARN") + ) + fov_density_pct_dropout = 100.0 * fov_density_dropout_count / fov_eligible_count + else: + fov_density_dropout_count = None + fov_density_pct_dropout = None + + # Per-cell median aggregates (n_mols_per_cell + n_genes_per_cell are + # numpy arrays defined earlier; reuse them). + median_transcripts_per_cell = ( + int(np.median(n_mols_per_cell)) if len(n_mols_per_cell) > 0 else 0 + ) + median_genes_per_cell = ( + int(np.median(n_genes_per_cell)) if len(n_genes_per_cell) > 0 else 0 + ) + + # Nucleus-to-cell area ratio summary (reuses nucleus_size_fraction Series + # from the §5.3 figure block above). + _nf = nucleus_size_fraction.dropna() + if len(_nf) > 0: + nucleus_to_cell_area_ratio = { + "mean": float(_nf.mean()), + "median": float(_nf.median()), + "stdev": float(_nf.std()), + "pct_cells_below_5pct": float((_nf < 0.05).mean() * 100), + "pct_cells_above_60pct": float((_nf > 0.60).mean() * 100), + "pct_cells_above_80pct": float((_nf > 0.80).mean() * 100), + } + else: + nucleus_to_cell_area_ratio = { + "mean": None, + "median": None, + "stdev": None, + "pct_cells_below_5pct": None, + "pct_cells_above_60pct": None, + "pct_cells_above_80pct": None, + } + # ===== end Phase 0 schema extension ===== + + # ===== SegTraQ-derived metrics — Phase A (2026-06-16) ===== + # Metric definitions mirrored from SegTraQ (github.com/LazDaria/SegTraQ, + # MIT), reimplemented in pandas on the parquet transcript QC already loads. + + # Metric 1: transcripts not assigned to a cell. cell_id == "UNASSIGNED" + # marks transcripts that decoding placed but segmentation left unassigned. + # Distinct from the "unassigned" codeword decoding category. The headline + # ratio is unbiased under row-group subsampling, so it is always computed. + pct_transcripts_unassigned_to_cell = float(_tx.pct_unassigned) + + # Per-gene breakdown — full-read only (FoV-ordered row groups). CV across + # genes is the Phase B triage signal (even vs gene-driven); surfaced here + # only as a value + figure. + # NUM_ROW_GROUPS is forced to None above (streaming reads the whole file), so + # this guard is now always taken; kept to preserve the original intent. + if NUM_ROW_GROUPS is None: + # From the streaming pass: per-gene totals and unassigned counts, real + # genes only. Previously this materialised _genes = df_spatial[is_gene], + # a 7-column copy of ~80% of the rows, held for the rest of the block. + _genes_only = _tx.gene[_tx.gene["is_gene"]] + _per_gene = pd.DataFrame( + { + "total": _genes_only["n_molecules"], + "unassigned": _genes_only["n_unassigned"], + } + ) + _per_gene["perc_unassigned"] = ( + _per_gene["unassigned"] / _per_gene["total"] * 100 + ) + _gated = _per_gene[_per_gene["total"] >= MIN_GENE_COUNT_FOR_UNASSIGNED] + if len(_gated) > 1 and _gated["perc_unassigned"].mean() > 0: + unassigned_per_gene_cv = float( + _gated["perc_unassigned"].std() / _gated["perc_unassigned"].mean() + ) + else: + unassigned_per_gene_cv = None + + # Source data for the figure (genes that passed the count gate). + _per_gene_out = _gated.sort_values( + "perc_unassigned", ascending=False + ).reset_index() + _per_gene_out.to_csv( + figures_source_dir / "unassigned_per_gene.csv", index=False + ) + + # Figure: per-gene distribution of unassigned fraction. The dashed line + # marks the sample-wide rate, so genes leaking far above it stand out. + fig = plt.figure(figsize=(8, 4)) + sns.histplot(_gated["perc_unassigned"], bins=50, ax=plt.gca()) + plt.axvline(x=pct_transcripts_unassigned_to_cell, color="grey", linestyle="--") + plt.xlabel("Transcripts not assigned to a cell, per gene (%)") + plt.ylabel("Num. genes") + plt.title("Per-gene transcript-to-cell assignment", fontsize=14, pad=20) + plt.tight_layout() + plt.savefig( + output_fig_dir / "unassigned_per_gene.pdf", dpi=300, bbox_inches="tight" + ) + plt.savefig( + output_fig_dir / "unassigned_per_gene.png", dpi=300, bbox_inches="tight" + ) + plt.close(fig) + else: + unassigned_per_gene_cv = None + # ===== end SegTraQ Phase A metrics ===== + + # EXACT CODE FROM ORIGINAL NOTEBOOK - Save metrics + # Gene-only counts for §2.4 (the noise bound is applied to genes, + # not to all features — non-gene categories like negative_control_* + # don't go through this filter). Computed from the per-feature + # rollup so we get the same denominator the histogram in §4.1 uses. + _gene_feat = n_mols_per_gene_df.query("is_gene == True") + _ctrl_feat = n_mols_per_gene_df.query("is_gene == False") + total_genes_count = len(_gene_feat) + filtered_genes_count = int((_gene_feat["n_molecules"] <= n_mols_threshold).sum()) + + # §4 feature-level gene-vs-control transcript-count shift. A healthy + # panel should have gene features carrying much higher per-feature + # transcript counts than control features. log2(median_gene / + # median_control) > 2 means genes carry at least 4× control medians. + if len(_gene_feat) > 0 and len(_ctrl_feat) > 0: + _median_gene = float(_gene_feat["n_molecules"].median()) + _median_ctrl = float(_ctrl_feat["n_molecules"].median()) + median_transcripts_per_gene_feature = _median_gene + median_transcripts_per_control_feature = _median_ctrl + log2_gene_vs_control_median_ratio = ( + float(np.log2(_median_gene / _median_ctrl)) if _median_ctrl > 0 else None + ) + else: + median_transcripts_per_gene_feature = None + median_transcripts_per_control_feature = None + log2_gene_vs_control_median_ratio = None + + metrics = { + "total_transcripts": int(num_transcripts), + "selected_transcripts": int(num_selected_transcripts), + "total_features": int(_tx.n_features), + "total_genes_count": total_genes_count, + "filtered_genes_count": filtered_genes_count, + "codeword_category_counts": { + str(k): int(v) for k, v in codeword_category_counts.items() + }, + "neg_control_quantile": int(n_mols_threshold), + # None ⇒ not computable on degenerate counts (see _min_count_threshold); + # emitted as JSON null, which the report already treats as absent. + "min_transcripts_per_cell": ( + int(n_mols_threshold_cell) if n_mols_threshold_cell is not None else None + ), + "min_genes_per_cell": ( + int(n_genes_threshold) if n_genes_threshold is not None else None + ), + "retained_genes_count": len(retained_genes), + "total_cells": int(ad.shape[0]), + "analyzed_genes": int(ad.shape[1]), + # ----- Phase 0 (v5 redesign) additions ----- + "pct_transcripts_above_qv20": pct_transcripts_above_qv20, + "mean_qv": mean_qv, + "fov_summary": fov_summary, + "fov_eligible_count": fov_eligible_count, + "fov_outlier_count": fov_outlier_count, + "fov_warn_count": fov_warn_count, + "fov_pct_fail": fov_pct_fail, + "fov_pct_warn": fov_pct_warn, + "median_transcripts_per_cell": median_transcripts_per_cell, + "median_genes_per_cell": median_genes_per_cell, + "nucleus_to_cell_area_ratio": nucleus_to_cell_area_ratio, + # Section-summary additions (2026-05-22) for §5 cell-level summary + "median_cell_size": median_cell_size, + "nucleus_transcript_fraction_summary": nucleus_transcript_fraction_summary, + # §4 feature-level shift (drives §1 row 3 in the v5 4-row scorecard) + "median_transcripts_per_gene_feature": median_transcripts_per_gene_feature, + "median_transcripts_per_control_feature": median_transcripts_per_control_feature, + "log2_gene_vs_control_median_ratio": log2_gene_vs_control_median_ratio, + # ----- SegTraQ-derived metrics (Phase A) ----- + "transcript_assignment": { + "pct_transcripts_unassigned_to_cell": pct_transcripts_unassigned_to_cell, + "unassigned_per_gene_cv": unassigned_per_gene_cv, + "per_gene_min_count": MIN_GENE_COUNT_FOR_UNASSIGNED, + "subsampled": NUM_ROW_GROUPS is not None, + }, + "cells_without_nucleus": { + "pct_cells_no_nucleus": pct_cells_no_nucleus, + "denominator": "cells_with_total_counts_gt_0", + }, + # % cells with zero gene transcripts (in-house, cells.parquet). Gated at + # warn 5 / fail 10. None when transcript_counts is unavailable. + "pct_cells_zero_transcripts": pct_cells_zero_transcripts, + "fov_density_dropout_count": fov_density_dropout_count, + "fov_density_pct_dropout": fov_density_pct_dropout, + # 10x-defined bundle metrics read from metrics_summary.csv (None when + # absent / blank / resegmented). See read_bundle_metrics(). + "bundle_metrics": read_bundle_metrics(XENIUM_BUNDLE_DIR, args.is_resegmented), + # XOA (onboard analysis) version of the bundle, e.g. "xenium-4.0.1.0". + "xoa_version": read_xenium_analysis_sw_version(XENIUM_BUNDLE_DIR), + # Segmentation software that produced the bundle this report describes. + "segmentation_software": resolve_segmentation_software( + XENIUM_BUNDLE_DIR, + args.pipeline_segmentation, + args.is_resegmented, + parse_tool_versions(args.seg_versions_file), + ), + } + + # Save metrics + with open(output_metrics_path, "w") as f: + json.dump(metrics, f, indent=2) + + # EXACT CODE FROM ORIGINAL NOTEBOOK - Save versions + dump_versions( + str(outdir / "versions.yml"), + ["matplotlib", "numpy", "pandas", "scipy", "seaborn", "scanpy", "pyarrow"], + task_name=TASK_PROCESS, + ) + + _log_mem("processing complete") + _log_mem_summary() + print("\n=== Processing Complete ===") + print(f"Generated {len(list(output_fig_dir.glob('*.pdf')))} figures") + print(f"Generated {len(list(figures_source_dir.glob('*.csv')))} CSV data files") + print(f"Saved metrics to: {output_metrics_path}") + print(f"Output directory: {outdir}") + print(f"Figures directory: {output_fig_dir}") + print(f"Source data directory: {figures_source_dir}") + + +if __name__ == "__main__": + main() diff --git a/bin/transcript_qc_thresholds.yaml b/bin/transcript_qc_thresholds.yaml new file mode 100644 index 00000000..254fe586 --- /dev/null +++ b/bin/transcript_qc_thresholds.yaml @@ -0,0 +1,299 @@ +# Transcript QC thresholds — drives PASS/WARN/FAIL verdicts in the transcript QC +# report (notebooks/transcript_qc.qmd) §1 at-a-glance scorecard, §2.2 codeword +# table, §3.1 FoV summary, §5.3 segmentation quality. +# +# Calibrated 2026-05-22 against 12 calibration samples (skin × 2, lung × 3, +# pancreas × 2, liver × 2, brain × 2, control × 1). See +# plans/2026-05-22_CALIBRATION_v0_thresholds.py for the per-metric analysis. +# +# Provenance tags: +# calibrated: empirical p5/p25 quantile cutoff from the 2026-05-22 cohort +# prose: cutoff appeared in the v1 (pre-2026-05) report prose at the line +# cited; the analysis team has used these values in practice and +# the cohort confirms. + +transcript_qc: + # ========================================================================= + # Codeword category thresholds — drives §2.2 codeword table + §1 row 3. + # Category names are the canonical snake_case strings Xenium's decoder + # emits in transcripts.parquet `codeword_category`. "decoded_gene_combined" + # is a synthetic computed at render-time as custom_gene + predesigned_gene. + # ========================================================================= + codeword_categories: + decoded_gene_combined: + display_label: "Decoded gene (custom + predesigned)" + contributors: ["custom_gene", "predesigned_gene"] + pass_min: 95.0 # %: PASS if combined decoded gene fraction >= 95% (loosened per "lenient — spot very bad samples only" guidance) + warn_min: 90.0 # %: WARN if 90% <= fraction < 95% + # FAIL if combined fraction < 90% (no cohort sample triggers this) + + unassigned_codeword: + display_label: "Unassigned codeword" + pass_max: 5.0 # %: PASS if unassigned fraction < 5% (prose; cohort max 1.97% — well below) + warn_max: 10.0 # %: WARN if 5% <= unassigned fraction < 10% (prose extension; nothing in cohort triggers) + # FAIL if unassigned fraction >= 10% + + negative_control_probe: + display_label: "Negative control probe" + pass_max: 2.0 # %: PASS if neg-ctrl-probe fraction < 2% (prose; cohort max 0.078% — well below) + warn_max: 5.0 # %: WARN if 2% <= fraction < 5% (prose extension; nothing in cohort triggers) + # FAIL if neg-ctrl-probe fraction >= 5% + + # Informational categories — shown in §2.2 table but no PASS/WARN/FAIL. + codeword_informational: + - custom_gene + - predesigned_gene + - genomic_control_probe + - negative_control_codeword + - deprecated_codeword + + # Pretty-print labels for display (snake_case → human-readable). + category_display_labels: + custom_gene: "Custom gene" + predesigned_gene: "Predesigned gene" + genomic_control_probe: "Genomic control probe" + negative_control_probe: "Negative control probe" + negative_control_codeword: "Negative control codeword" + unassigned_codeword: "Unassigned codeword" + deprecated_codeword: "Deprecated codeword" + + # ========================================================================= + # % of panel genes that survive the negative-control noise filter. + # Computed as filtered_genes_count / total_genes_count × 100. A healthy + # sample retains most of its panel genes above the noise floor; a sample + # where the majority of genes drop below noise indicates a broken + # decoder, weak signal, or wildly off-target panel. + # Drives the §2 sample-summary "% genes filtered out (noise)" row. + # Lenient — only spots clearly broken samples. + # ========================================================================= + noise_filter: + pct_filtered_pass_max: 50.0 # PASS if < 50% of panel genes filtered out + pct_filtered_warn_max: 80.0 # WARN if 50% ≤ pct < 80% + # FAIL if ≥ 80% of panel genes filter out (catastrophic signal loss) + + # ========================================================================= + # Transcript yield — drives §1 row 1. + # Sample-level total transcript count. Cohort range 300k–1.1B (skin_bad + # at 300k correctly fires WARN under current thresholds). + # ========================================================================= + yield: + total_transcripts_pass_min: 1_000_000 # PASS if total_transcripts >= 1M (validated by cohort) + total_transcripts_warn_min: 100_000 # WARN if 100k <= total < 1M (validated by cohort) + # FAIL if total_transcripts < 100k + + # ========================================================================= + # Per-cell median yield floors — drive the §2.1 "Per-cell aggregates" + # status pills. ADVISORY (PASS/WARN only, no FAIL): median transcripts and + # genes per cell are strongly panel-size- and tissue-driven, so these are + # advisory floors that flag a sample whose typical cell carries very little + # signal, not hard gates. A small or tissue-mismatched panel can sit below + # these for biological reasons. warn_min is null so sub-floor medians get + # WARN, never FAIL (_status_higher_better). + # ========================================================================= + per_cell_median: + median_transcripts_pass_min: 100 # PASS if median transcripts/cell >= 100; WARN below (advisory, no FAIL) + median_genes_pass_min: 50 # PASS if median genes/cell >= 50; WARN below (advisory, no FAIL) + + # ========================================================================= + # Sample-wide QV summary — drives §1 row 2 (Transcript quality). + # ADVISORY tier (PASS/WARN only, no FAIL) — calibration showed this metric + # is heavily tissue-dependent: brain 74-76%, liver ~80%, pancreas 90-94%. + # A FAIL tier on absolute thresholds would mark normal brain as failing. + # Move to tissue-stratified thresholds in a future calibration with more + # samples per tissue. + # ========================================================================= + qv: + pct_above_qv20_pass_min: 75.0 # %: PASS if >= 75% (2026-06-29: raised from 70 = pooled p05 of the 164-sample deduped cohort; brain ~74% now WARNs, advisory only) + pct_above_qv20_warn_min: 60.0 # %: WARN if 60% <= pct < 75% + # FAIL if pct_above_qv20 < 60% (catastrophic decoder failure; no cohort + # sample triggers — added per user request to give pct_above_qv20 a FAIL tier). + + # ========================================================================= + # Transcripts not assigned to a cell — drives §2 sample-summary row. + # cell_id == "UNASSIGNED": transcripts decoding placed but segmentation did + # not assign to any cell (~tens of % in healthy Xenium). Distinct from the + # "unassigned_codeword" decoding metric above. SegTraQ-derived (Phase A). + # ADVISORY (PASS/WARN, no FAIL). + # + # 2026-06-23: made INFORMATIONAL — that read used the 260-sample NON-deduped + # cohort (p95 ≈ 59%, replicate-inflated) and judged any numeric warn would fire + # on most samples; 10x treats it as informational too. + # + # 2026-06-29: re-gated as ADVISORY WARN-only. Deduping to 164 physical samples + # (one row per slide) drops pooled p95 to 39 / p99 to 47 — tighter than the 59% + # above implied. pass_max = 30 sits just above the pancreas median (29): PASS + # below 30, WARN at/above 30, never FAIL (warn_max null). Strongly tissue-driven + # (pancreas highest) so expect occasional pancreas WARNs by design; per-tissue + # cuts are a future refinement. + # ========================================================================= + transcript_assignment: + pct_unassigned_pass_max: 30.0 # %: PASS if < 30%; WARN at/above (advisory, never FAIL) + pct_unassigned_warn_max: null # advisory: >= pass_max → WARN, never FAIL + + # ========================================================================= + # FoV consistency — drives §1 row 4 and §3.1 per-FoV table. + # Spike-validated design (plans/2026-05-22_SPIKE_fov_zscore_denominator.py): + # within-sample z-score on per-FoV pct_above_qv20, one-sided, n-gate. + # ========================================================================= + fov_consistency: + # Per-FoV thresholds — empirically calibrated against 1,129 FoVs. + # z=-3 sits at p1.24 (close to 1% target for FAIL); z=-2 sits at p3.45 + # (close to 5% target for WARN). Keep as is. + n_gate: 100 # FoVs with fewer transcripts → status = N/A + z_warn: 2.0 # WARN if dev_z < -2.0 (calibrated: empirical p3.45) + z_fail: 3.0 # FAIL if dev_z < -3.0 (calibrated: empirical p1.24) + # Sample-level rollup — calibrated tighter than v0: + # cohort max fov_pct_fail was 3.85% (lung_low_signal), max fov_pct_warn 5.13%. + # Original v0 (5%/10%) never fired. Tightened so lung_low_signal, + # skin_bad, pancreas_0060257 fire WARN appropriately. + sample_pct_fail_warn: 2.0 # %: §1 WARN if fov_pct_fail > 2% (calibrated) + sample_pct_fail_fail: 5.0 # %: §1 FAIL if fov_pct_fail > 5% (calibrated) + sample_pct_warn_warn: 5.0 # %: §1 WARN if fov_pct_warn > 5% (calibrated: cohort max 5.13%) + + # ========================================================================= + # Per-FoV transcript-density dropout — drives §3 FoV table + a §3 rollup row. + # Same one-sided within-sample z-score machinery as fov_consistency, on + # per-FoV transcript counts instead of QV. Flags in-tissue FoVs carrying far + # fewer transcripts than their neighbours (regional sectioning damage, folds). + # SegTraQ-derived (Phase A). ADVISORY (PASS/WARN, no FAIL). PROVISIONAL — + # not yet cohort-calibrated; calibrate in the Phase B triage spike. + # ========================================================================= + fov_density: + z_warn: 2.0 # per-FoV WARN if transcript dev_z < -2.0 + z_fail: 3.0 # per-FoV FAIL if transcript dev_z < -3.0 + sample_pct_dropout_pass_max: 10.0 # PROVISIONAL: PASS if < 10% of FoVs flagged + sample_pct_dropout_warn_max: null # advisory: >= pass_max → WARN, never FAIL + + # ========================================================================= + # Per-cell aggregates — RESERVED (not yet wired to any §1 verdict). + # + # §1 row 5 (Cell yield) is informational only: PASS if total_cells > 0 + # else N/A. Calibration shows median/min ratios range from 0.20 (skin_bad + # — sample-median below noise floor) to 58.7 (pancreas). The "5x" v0 + # placeholder bears no relation to real data; thresholds below are + # placeholders pending a row-7 design. + # ========================================================================= + cells: + median_transcripts_pass_x: 5.0 # UNUSED (no row 7 yet) + median_transcripts_warn_x: 2.0 # UNUSED + median_genes_pass_x: 5.0 # UNUSED + median_genes_warn_x: 2.0 # UNUSED + + # ========================================================================= + # Cell-level section summary thresholds — drive the §5 section summary + # table (advisory PASS/WARN only; no FAIL). Designed lenient: the goal + # is to draw attention to clearly-bad samples, not to gatekeep marginal + # ones. All values are wide-band placeholders pending cohort calibration. + # ========================================================================= + + # Cell yield — total segmented cells (advisory PASS/WARN, no FAIL). + # A standard Xenium section yields ~100k–700k cells depending on tissue + # area and cell density. Flag implausibly low counts, but keep it advisory: + # a small biopsy or TMA core can legitimately fall below the floor. + cell_yield: + warn_below: 50000 # WARN if fewer than 50k cells; PASS at or above + + # §5.1 Cell size distribution — median pixels². + cell_size: + median_pass_min: 30.0 # too small → segmentation under-detects + median_pass_max: 500.0 # too large → merged-cell artefacts; lenient + # Outside this band → WARN. No FAIL tier. + + # §5.2 Nucleus RNA fraction — median, plus % cells with fraction > 0.6. + # §5.2 Nucleus RNA fraction — median per-cell fraction of assigned transcripts + # overlapping the nucleus, plus % cells above 0.6. PROVISIONAL / not + # cohort-calibrated: the old band was set against a broken metric + # (nucleus_count/total_counts ≈ 1/total_counts); the metric is now correctly + # derived from transcripts.parquet overlaps_nucleus, so these are lenient + # advisory placeholders (one calibration sample showed median ~0.51). Advisory + # PASS/WARN only, no FAIL. Recalibrate from a cohort spike. + nucleus_transcript_fraction: + median_pass_min: 0.05 # PROVISIONAL + median_pass_max: 0.70 # PROVISIONAL (widened from 0.60 — corrected metric centres ~0.5) + pct_above_60_pass_max: 50.0 # PROVISIONAL: < 50% of cells above 0.6 → PASS + + # §5.4 / §5.5 per-cell median multipliers — NOT WIRED. + # Dropped 2026-05-22 per user feedback: ratio of median transcripts/genes + # per cell to the noise-floor min is a real quality signal (skin_bad + # median=2 vs min=10 → 0.2× catches it) but the formulation is + # confusing for a scientific reader. Same diagnostic is captured more + # legibly by §5.3 nucleus-to-cell ratio + §5.2 nucleus RNA fraction. + # Median transcripts/cell and median genes/cell still appear in §5 + # summary as INFORMATIONAL rows (no status pill, alongside the + # min reference value); they just don't drive PASS/WARN today. + # Kept here as YAML config for a possible future revisit. + cells_transcripts: + median_pass_x_min: 2.0 # NOT WIRED — kept for potential future use + cells_genes: + median_pass_x_min: 2.0 # NOT WIRED + + # ========================================================================= + # Segmentation quality — drives §1 row 6 (advisory PASS/WARN, no FAIL). + # Calibrated: cohort median nucleus_to_cell ratio range 0.32 (liver) to + # 0.57 (brain). Brain/skin sit at ~0.54 — biologically real, not artefact. + # Widened PASS band from v0 [0.10, 0.40] → [0.10, 0.60] so normal brain/ + # skin PASS; only pathological segmentation flags. pct_above_80pct PASS + # widened from 5% → 10% so only skin_bad (18.3%) fires WARN. + # ========================================================================= + segmentation: + nucleus_to_cell_median_pass_min: 0.10 # calibrated: cohort min 0.32, no sample below 0.10 + nucleus_to_cell_median_pass_max: 0.60 # calibrated: cohort max 0.565 (brain); was 0.40 in v0 + nucleus_to_cell_pct_above_80_pass_max: 10.0 # calibrated: only skin_bad (18.3%) above; was 5.0 in v0 + nucleus_to_cell_median_warn_min: 0.05 # outside this — catastrophic over-segmentation + nucleus_to_cell_median_warn_max: 0.70 # outside this — catastrophic under-segmentation; was 0.50 in v0 + nucleus_to_cell_pct_above_80_warn_max: 20.0 # was 15.0 in v0 + # No FAIL tier — advisory only (matches image QC §4 convention). + + # ========================================================================= + # Cells with no in-plane nucleus — drives a §5 cell-level row. + # nucleus_count == 0: cells the segmentation found but with no nucleus in + # this z-plane (inherent to thin 2D sections — the nucleus sits in an + # adjacent plane). Denominator is cells with total_counts > 0. SegTraQ- + # derived (Phase A). + # + # 2026-06-23: warn 10 / fail 15, matching the qc_threshold_refinement report's + # deduped cut (one row per physical sample, replicate-inflation removed: p95/ + # p99 = 10% / 15%). No-nucleus is segmentation-method-dependent and is recorded + # only for onboard segmentation in this cohort, so it is calibrated on onboard + # and applied as one method-agnostic cut (revisit per-method later; + # cells.parquet carries segmentation_method). This ADDS a FAIL tier where the + # row was previously advisory — the one §5 cell-level row that can FAIL, so it + # can turn the §1 "Cell-level" scorecard line to FAIL (see transcript_qc.qmd). + # ========================================================================= + nucleus_coverage: + pct_cells_no_nucleus_pass_max: 10.0 # PASS if < 10% of cells lack a nucleus (buffer above deduped p95 = 7%) + pct_cells_no_nucleus_warn_max: 15.0 # WARN 10-15%, FAIL >= 15% (deduped p99 = 9%) + + # ========================================================================= + # 10x-defined bundle metrics — read from the Xenium bundle's + # metrics_summary.csv (bin/transcript_qc_processing.py: read_bundle_metrics). + # All N/A when the CSV is absent, the column is blank, or the bundle was + # resegmented (the CSV then reflects onboard, not the new segmentation). + # Added 2026-06-22 per the qc_threshold_refinement report. + # ========================================================================= + + # Nuclear transcripts per 100 um² — capture density in nuclei. Catastrophic + # floor only (higher is better). Report: warn 10 / fail 1; cohort min ~12. + nuclear_density: + per_100um2_pass_min: 10.0 # PASS if >= 10 + per_100um2_warn_min: 1.0 # WARN 1-10, FAIL < 1 (no usable data) + + # % cells with zero gene transcripts — derived IN-HOUSE from cells.parquet + # (transcript_counts == 0 over all cells; verified to reproduce 10x's + # fraction_empty_cells exactly on the example bundles). warn 5 / fail 10: + # _status_lower_better(pct, pass_max=5, warn_max=10) → PASS < 5, WARN 5-10, + # FAIL >= 10 (10x's >10% error floor, plus a WARN tier per user). Works on + # resegmented bundles (uses this run's cells.parquet, not the onboard CSV). + empty_cells: + pct_pass_max: 5.0 # PASS if < 5% zero-transcript cells + pct_warn_max: 10.0 # WARN 5-10%, FAIL >= 10% (10x error floor) + + # Fraction of cells segmented using a stain (vs nucleus expansion). ADVISORY + # (WARN only, no FAIL). Gated ONLY when stain segmentation was actually used + # (stain_definition present); a nucleus-expansion run reports stain_frac = 0 + # by design and is shown as informational, NOT WARN. stain_frac is available + # in both XOA 3.x and 4.0 — the 0-vs-version pattern is a segmentation-config + # confound, not drift. Report value: warn 70% (fail tier dropped per advisory). + segmentation_stain: + stain_frac_pass_min: 70.0 # PASS if >= 70% (×100 of the 0-1 fraction) + stain_frac_warn_min: null # advisory: below pass_min → WARN, never FAIL diff --git a/bin/transcript_stream.py b/bin/transcript_stream.py new file mode 100755 index 00000000..15a5a7a7 --- /dev/null +++ b/bin/transcript_stream.py @@ -0,0 +1,318 @@ +"""Streaming aggregation over a Xenium ``transcripts.parquet``. + +``molecule_qc_processing.main()`` used to open with + + df_spatial = pd.read_parquet(path, columns=required_columns) + +and hold the whole table to the end of the function, then make ~13 passes over +it. On the 1.33 billion-row reference sample that frame costs ~127 GB of RSS +under *any* encoding — measured 117 B/row as ``object`` and 96 B/row +dictionary-encoded — which is what drove the 120 -> 240 -> 480 GB retry ladder. +Note that ``DataFrame.memory_usage(deep=True)`` is badly misleading here: it +over-reports ``object`` (it counts shared ``str`` objects once per reference) and +under-reports ``category`` (it ignores the Arrow dictionary buffers), so only RSS +measured in a subprocess means anything. + +Every one of those passes is a reduction. The largest output is per-cell +(~530 k rows); the rest are per-FoV (~900), per-gene (~13.8 k), or scalars. The +two QV violin plots are the only row-level consumers and the previous code +*already* capped them at 10^6 rows each — it just obtained that cap by +materialising 1.33 billion rows and sampling down, i.e. reading 127 GB to keep +0.08 % of it. + +Aggregation runs on **pyarrow Acero**, the streaming execution engine already +shipped in the container's pyarrow 21. Acero does the hash aggregation +out-of-core off the parquet scan, so there is no hand-rolled accumulation to get +wrong. Measured on a 20 M-row synthetic file with the production cardinalities +(700 k cells, 13.8 k genes, 900 FoVs): peak RSS 2.0 GB, per-cell aggregate in +1.9 s. + +**One deliberate behaviour change.** The violin samples are no longer the same +rows as before. The previous code sampled the complete frame with +``random_state=42``, which is not reproducible without the complete frame. +Sampling here uses the A-Res scheme — draw an independent uniform key per row, +keep the ``k`` smallest — which is an exactly uniform sample of the whole stream, +deterministic for a given seed, and vectorised. The plotted distribution is +equivalent; individual rows differ. Every scalar and aggregate metric is exact. +""" + +from __future__ import annotations + +from dataclasses import dataclass + +import numpy as np +import pandas as pd +import pyarrow as pa +import pyarrow.compute as pc +import pyarrow.dataset as ds +from pyarrow import acero + +# Rows per record batch for the sampling pass. Large enough that per-batch +# overhead is negligible over ~660 batches, small enough never to be the peak. +DEFAULT_BATCH_ROWS = 2_000_000 + +# Matches the previous per-violin cap. +DEFAULT_SAMPLE_ROWS = 1_000_000 + +QV_THRESHOLD = 20.0 + +# Cap how far the scan may run ahead of the aggregate. Arrow's parallel scan will +# otherwise keep an unbounded number of batches in flight, and the pool grows with +# how far the scan outruns the consumer: measured on identical-schema files the +# default gave 2.24 GB at 20 M rows and 3.80 GB at 60 M (39 B/row and climbing), +# while capping the window gave 1.04 GB and 1.35 GB (7.8 B/row, so roughly 11 GB +# projected at 1.33 B rows against ~145 GB for the eager frame). +# +# Threads stay on: use_threads=False is marginally lower (1.02 GB) but discards +# scan parallelism on CPU-bound work. malloc_trim does not reclaim any of it, +# which confirms the memory is live Arrow buffers, not heap fragmentation. +_SCAN_READAHEAD = {"fragment_readahead": 1, "batch_readahead": 2} + + +def _aggregate(dataset, scan_columns, project_exprs, project_names, aggregates, keys): + """Run one streaming hash aggregation off the parquet scan. + + ``scan_columns`` must list the *input* columns the projections reference, not + the projection output names: a projection over a column the scan did not read + yields nulls, which surface as NaN in the aggregate rather than as an error. + """ + return acero.Declaration.from_sequence( + [ + acero.Declaration( + "scan", + acero.ScanNodeOptions( + dataset, columns=list(scan_columns), **_SCAN_READAHEAD + ), + ), + acero.Declaration( + "project", acero.ProjectNodeOptions(project_exprs, project_names) + ), + acero.Declaration( + "aggregate", acero.AggregateNodeOptions(aggregates, keys=keys) + ), + ] + ).to_table(use_threads=True) + + +@dataclass +class TranscriptStats: + """Small tables and scalars; nothing here scales with transcript count.""" + + n_rows: int + n_gene_rows: int + n_unassigned: int + qv_sum: float + qv_above_threshold: int + codeword_counts: pd.Series + fov: pd.DataFrame # index fov_name; n, mean_qv, pct_above_qv20, pct_unassigned + gene: pd.DataFrame # index feature_name; n_molecules, n_unassigned, is_gene + cell_nucleus_fraction: pd.Series # index cell_id (assigned cells only) + + @property + def mean_qv(self) -> float: + return self.qv_sum / self.n_rows if self.n_rows else float("nan") + + @property + def pct_qv_above_threshold(self) -> float: + if not self.n_rows: + return float("nan") + return 100.0 * self.qv_above_threshold / self.n_rows + + @property + def pct_unassigned(self) -> float: + if not self.n_rows: + return float("nan") + return 100.0 * self.n_unassigned / self.n_rows + + @property + def n_features(self) -> int: + return len(self.gene) + + +def aggregate_transcripts(path, background_cell_id: str) -> TranscriptStats: + """Stream the transcript table, returning only small aggregates.""" + dataset = ds.dataset(str(path), format="parquet") + + unassigned = (pc.field("cell_id") == background_cell_id).cast(pa.int64()) + above = (pc.field("qv") > QV_THRESHOLD).cast(pa.int64()) + + fov_tbl = _aggregate( + dataset, + ["fov_name", "qv", "cell_id"], + # qv is float32 on disk; cast to float64 so the sum accumulates in double + # precision, matching pandas .mean() which upcasts. In float32 the FoV + # means differ from pandas in the 7th significant digit. + [pc.field("fov_name"), pc.field("qv").cast(pa.float64()), above, unassigned], + ["fov_name", "qv", "above", "unassigned"], + [ + ("qv", "hash_count", None, "n"), + ("qv", "hash_sum", None, "qv_sum"), + ("above", "hash_sum", None, "n_above"), + ("unassigned", "hash_sum", None, "n_unassigned"), + ], + ["fov_name"], + ) + fov = fov_tbl.to_pandas().set_index("fov_name").sort_index() + fov["mean_qv"] = fov["qv_sum"] / fov["n"] + fov["pct_above_qv20"] = 100.0 * fov["n_above"] / fov["n"] + fov["pct_unassigned"] = 100.0 * fov["n_unassigned"] / fov["n"] + + gene_tbl = _aggregate( + dataset, + ["feature_name", "is_gene", "cell_id"], + [ + pc.field("feature_name"), + pc.field("is_gene").cast(pa.int64()), + unassigned, + ], + ["feature_name", "is_gene", "unassigned"], + [ + ("is_gene", "hash_count", None, "n_molecules"), + # is_gene is a property of the feature, so max == the value. + ("is_gene", "hash_max", None, "is_gene"), + ("unassigned", "hash_sum", None, "n_unassigned"), + ], + ["feature_name"], + ) + gene = gene_tbl.to_pandas().set_index("feature_name").sort_index() + gene["is_gene"] = gene["is_gene"].astype(bool) + + cw_tbl = _aggregate( + dataset, + ["codeword_category"], + [pc.field("codeword_category")], + ["codeword_category"], + [("codeword_category", "hash_count", None, "n")], + ["codeword_category"], + ) + codeword = ( + cw_tbl.to_pandas() + .set_index("codeword_category")["n"] + .sort_values(ascending=False) + ) + + # Per-cell nucleus fraction, assigned cells only. Filter before aggregating so + # the sentinel never becomes a group. + cell_tbl = acero.Declaration.from_sequence( + [ + acero.Declaration( + "scan", + acero.ScanNodeOptions( + dataset, + columns=["cell_id", "overlaps_nucleus"], + **_SCAN_READAHEAD, + ), + ), + acero.Declaration( + "filter", + acero.FilterNodeOptions(pc.field("cell_id") != background_cell_id), + ), + acero.Declaration( + "project", + acero.ProjectNodeOptions( + [ + pc.field("cell_id"), + pc.field("overlaps_nucleus").cast(pa.int64()), + ], + ["cell_id", "ov"], + ), + ), + acero.Declaration( + "aggregate", + acero.AggregateNodeOptions( + [ + ("ov", "hash_sum", None, "ov_sum"), + ("ov", "hash_count", None, "n"), + ], + keys=["cell_id"], + ), + ), + ] + ).to_table(use_threads=True) + cell = cell_tbl.to_pandas().set_index("cell_id").sort_index() + fraction = (cell["ov_sum"] / cell["n"]).astype(float) + + # Count gene rows directly. Deriving it from the per-gene table would assume + # is_gene is a function of feature_name; true in Xenium data, but a scalar + # aggregate is exact either way. + gene_rows_tbl = acero.Declaration.from_sequence( + [ + acero.Declaration( + "scan", + acero.ScanNodeOptions(dataset, columns=["is_gene"], **_SCAN_READAHEAD), + ), + acero.Declaration( + "project", + acero.ProjectNodeOptions( + [pc.field("is_gene").cast(pa.int64())], ["is_gene"] + ), + ), + acero.Declaration( + "aggregate", + acero.AggregateNodeOptions([("is_gene", "sum", None, "n_gene")]), + ), + ] + ).to_table() + + n_rows = int(fov["n"].sum()) + return TranscriptStats( + n_rows=n_rows, + n_gene_rows=int(gene_rows_tbl.column("n_gene")[0].as_py()), + n_unassigned=int(fov["n_unassigned"].sum()), + qv_sum=float(fov["qv_sum"].sum()), + qv_above_threshold=int(fov["n_above"].sum()), + codeword_counts=codeword, + fov=fov[["n", "mean_qv", "pct_above_qv20", "pct_unassigned"]], + gene=gene[["n_molecules", "n_unassigned", "is_gene"]], + cell_nucleus_fraction=fraction, + ) + + +def sample_rows( + path, + columns: list[str], + predicate_column: str = "is_gene", + k: int = DEFAULT_SAMPLE_ROWS, + seed: int = 42, + batch_rows: int = DEFAULT_BATCH_ROWS, +) -> tuple[pd.DataFrame, pd.DataFrame]: + """Uniform samples of up to *k* rows where *predicate_column* is True / False. + + A-Res: give every row an independent uniform key and keep the ``k`` smallest. + That is an exactly uniform sample of the stream, needs only ``k`` rows + resident per group, and is deterministic for a given seed. + """ + import pyarrow.parquet as pq + + rng = np.random.default_rng(seed) + keep = {True: (None, None), False: (None, None)} # group -> (keys, frame) + wanted = list(dict.fromkeys([*columns, predicate_column])) + + handle = pq.ParquetFile(str(path)) + for batch in handle.iter_batches(batch_size=batch_rows, columns=wanted): + df = batch.to_pandas() + del batch + flag = df[predicate_column].to_numpy(dtype=bool, copy=False) + for group in (True, False): + sel = df.loc[flag if group else ~flag, columns] + if sel.empty: + continue + new_keys = rng.random(len(sel)) + keys, frame = keep[group] + if keys is None: + keys, frame = new_keys, sel.reset_index(drop=True) + else: + keys = np.concatenate([keys, new_keys]) + frame = pd.concat([frame, sel], ignore_index=True) + if len(keys) > k: + take = np.argpartition(keys, k)[:k] + keys = keys[take] + frame = frame.iloc[take].reset_index(drop=True) + keep[group] = (keys, frame) + del df + + out = [] + for group in (True, False): + _, frame = keep[group] + out.append(frame if frame is not None else pd.DataFrame(columns=columns)) + return out[0], out[1] diff --git a/bin/utility_downscale_morphology.py b/bin/utility_downscale_morphology.py index 8544ecf3..f1f41568 100755 --- a/bin/utility_downscale_morphology.py +++ b/bin/utility_downscale_morphology.py @@ -42,6 +42,7 @@ def downscale_image( print(f"Original: {img.shape}, dtype={img.dtype}, ndim={img.ndim}") # Handle multichannel OME-TIFFs: shape can be (H, W), (C, H, W), or (Z, C, H, W) + output_shape: tuple[int, ...] if img.ndim == 2: orig_h, orig_w = img.shape new_h = max(int(orig_h * scale), MIN_DIM) @@ -87,8 +88,12 @@ def parse_args() -> argparse.Namespace: description="Pre-downscale a morphology image for Cellpose." ) parser.add_argument("--image", required=True, help="Morphology TIFF input") - parser.add_argument("--diameter", type=float, required=True, help="Target object diameter") - parser.add_argument("--diam-mean", type=float, required=True, help="Cellpose model diam_mean") + parser.add_argument( + "--diameter", type=float, required=True, help="Target object diameter" + ) + parser.add_argument( + "--diam-mean", type=float, required=True, help="Cellpose model diam_mean" + ) parser.add_argument("--prefix", required=True, help="Output directory") return parser.parse_args() diff --git a/bin/utility_upscale_mask.py b/bin/utility_upscale_mask.py index 6cc1694e..e38ed2a3 100755 --- a/bin/utility_upscale_mask.py +++ b/bin/utility_upscale_mask.py @@ -39,7 +39,7 @@ def upscale_mask(mask_path: str, scale_info_path: str, prefix: str) -> None: print(f"Upscaling to ({orig_h}, {orig_w})") pil_mask = Image.fromarray(mask) - pil_mask = pil_mask.resize((orig_w, orig_h), Image.NEAREST) + pil_mask = pil_mask.resize((orig_w, orig_h), Image.Resampling.NEAREST) mask_up = np.array(pil_mask, dtype=mask.dtype) out_dir = Path(prefix) @@ -47,9 +47,7 @@ def upscale_mask(mask_path: str, scale_info_path: str, prefix: str) -> None: base = Path(mask_path).stem out_name = out_dir / f"upscaled_{base}.tif" tifffile.imwrite(str(out_name), mask_up, compression="zlib") - print( - f"Done: {out_name}, unique cells: {len(np.unique(mask_up)) - 1}" - ) + print(f"Done: {out_name}, unique cells: {len(np.unique(mask_up)) - 1}") def parse_args() -> argparse.Namespace: @@ -58,7 +56,9 @@ def parse_args() -> argparse.Namespace: description="Upscale a Cellpose mask back to original resolution." ) parser.add_argument("--mask", required=True, help="Downscaled mask TIFF") - parser.add_argument("--scale-info", required=True, help="scale_info.json from downscale step") + parser.add_argument( + "--scale-info", required=True, help="scale_info.json from downscale step" + ) parser.add_argument("--prefix", required=True, help="Output directory") return parser.parse_args() diff --git a/bin/xenium_image_qc_report.qmd b/bin/xenium_image_qc_report.qmd new file mode 100644 index 00000000..8d257502 --- /dev/null +++ b/bin/xenium_image_qc_report.qmd @@ -0,0 +1,3252 @@ +--- +title: "Xenium Image QC" +author: + - name: "Malwina Prater, Hanneke Okkenaug & Nell Yu Nie" + affiliation: "Data Science/Bioinformatics & Imaging" +date: last-modified +format: + html: + embed-resources: true + standalone: true + toc: true + toc-depth: 2 + fontsize: 0.9rem + title-block-style: none +jupyter: python3 +--- + +```{python title-banner} +#| echo: false +from IPython.display import HTML, display +from datetime import datetime, timezone +_LOGO_B64 = "iVBORw0KGgoAAAANSUhEUgAAAZAAAAHDCAYAAAAKpUyVAAEAAElEQVR42uz9d3xd53UmjD7rfXc5/aA3AixgFYskSpRkiVazJVe5xB7bsf3ZmcQp43wzzr3fzZ07+c1kJpPMxPPNJJl8KU6bxLHjeBLJtlwlW7JVLMmiWCT2ApIg0TtOb7u86/6x9z44AAEQbBJFncUffgCB0/be717P+6zyLGJmvGWNXYAkrPw5nj75Fyhl+tB9xx/ATGwgsAJIoG51q1vd6ra4aW918ChOvsSTx/4HXLsMJgHlluqrom51q1vd6gCyPHgURp/iySP/FSzDIKFDhlpgxjcQAICovjrqVre61W0Ze+vFaALmMfFTnj78n0EyBJImXCePaMudEFoYzC6AOoDUrW51q1sdQKrgobycR/YUTx/6bUCYIJKAciCEgXjHAx75qINH3epWt7rVAaQGPQAAysnz1MHfBisLJHQADOWWYSY3I5TcTADXk+d1q1vd6lYHkFr88IBh5uSfwsqdgdCiXjgLAqwsxNrvBUiAWdVXRd3qVre61QEkAA+vJLc4/QpnBx+HNBrAyvH/5kJoUURadgGoh6/qVre61a0OIHPoAYDAysL0yT8DCVnzN4JSNvRQO/RIt199VQ9f1a1udatbHUAALyRFhMzwD7icOQ4hox4jAbxSXXYhzaZqPqRudatb3epWBxAfIwTYrWB24DEIGQbjwhxHNZxVt7rVrW51qwOIxz68fo7c1M+4kjsLkqE59uE9ACQN2MUhuFaGAfKfU7e61a1udXtLA0iQEM+M/miJ5DiDSINrZTB98s8BVl5fCPzQVz2kVbe61a1uS9oNLGXile3a5Wkuzh6C0MLz2UeVhCgILYrc2DNwrBQ3rvsUIs23EgXJ9Gq+pJ5cr1vd6la3twSAsM8mirOvwbFmYehJgO0lHq0g9BhKqaMopv8TzOQWjrXvRrxtN/RQC81/TUJd5qRudatb3d4CYor52ddWiDgKpEUAkijnzqCQOYXpge8h1nonN3Tcj0jD5iorYVZeSKwuuFi3utWtDiA3nnm5DEY5exokjEWrrxYDEUCApAlN6lBuBemRZ5Ae/xlCyU3c0L4bidbboOnxNw0r4Tpfqlvd6lYHkEt3m04lzVZpHGLZHg8CSAJc42aZwXAB0iF1EwoCxcwZ5NNnoA8+iUTLTm5ouwvRxNp5rAQgH0yuIyCtr/G61a1udQC5FPxggAhWeQKunYMm9UUT6N5jXSi7CIgIoEUWJMu52ogoNBMCOhy7gOmR5zE9/goiifXc0HY7Gpq3Q9dj1x0rsRUwWVa8KiLqOFK3utWtDiAr4x8MAmCXxqHcCkgai+/NWUHocURaHkQldw7l/HmwMEBaDETSf5VaVqJApEHTQ3AhkM/2I5M5h7GhZ5Fo2spNLbcgkVwzx0rAXjHY68xKgrDVa7MuH5i28fY2jXc0aVQPZ9WtbnWrA8gKzSpPAlC+ZMkC+CAB1y4gtuq9aNn2/yXlllGc3svZsWdRSJ+AY2dBMgrIsNfNvhgrkSYE6XCcEqbG9mFq8jDCsVXc1LwNTc2bETIbqu0nwez51wNMyGcfQ0WFmE44knKwISER1mjJnAgzw3XrTZR1q9s1uy/JC3ELceO0BNzQAOJa6Yvu1UmYHhMRBmLt91Gs/T5U8oOcnXgB2cl9qJQmwCRBWtQDElqalSgIFArjyOXHMTK6F4nkWm5uvgmNydUka1gQM3upl2vABwKAyNjMZRfQBFC0GEMFlzcll0YQIoKmafW7vG51u8amlAIzQwhx3eVM6wBSCyBO8WJbAiin4OU92PVl3wlmbDW1xj6N5rUfRX7mEKcn96CQ7oNt5wARgtAi8MJU5DOTGlYidEihQ7GLmdnTmEqdgxlq4qbGXrQ2rkc81ka1i4aZ/Wrgq7uQMjbgMGCS98ojeRebktoFb+O9P2F6Zob/nz/+Y1iWNY8xeeFAmkdvgv8TUfX/3s/+XxY+ZgFQLQZelwyUzCv+f7Dzq/2M1f8v8vO8z0RzR7+SY7ks0Ge+5N8t9jP7m5ol/77U90Wet+Lvfpj2wu/BBuvir7308XA1clD7OvM3TDwvurBMqcyi123hdV5sjRARSBAEiSqDWPglpax+aZoGTdOg6zoMw4BpmgiFQ4jH4ti2bRt6enqqH8B1XUgp6wByPZonkkjLBnvYLdV4RlHDLBhChpBou4sSbXfBKk1yduYQMjNHUMqPwLHzABlzzKRaxeWDCSQ0zYQiDZaVx8j4IYxMnUI03MrNDavR0rAasXBDFUyudr4kbQU3KSAFYabMcBUgF7Bn13WhaRqefOIJ/Jf/8l/q28O61e0aWlNTE+677z7+7Gc/iw9+8IMkpYRSat4mpg4g14stKz/iSZ0op1DdgdQyE28Pwt4OHQQj3EYt3Q+jpfthlIvjnEv3ITN7AvncEBy7AJJRkNDmARb7QEQkoUsdLiTypVmki7M4P3kKiUgLtzT0oDXZhbARmZcvudLFlLEZwt+RCQIKDiNnK24wxaLJ9MnJSWiahkQiAcepqxPXrW5XdTPrs6disYhvf/vb+Pa3v4077riD/92/+3f4yEc+Qm9WNnJDA4iQJpYTRCSScCszYHarIooLGcqcI/fBhARCkQ4KRTrQ2nUfSsVxnp06hJnpoyiVs4A0ITWtJrxVG+ISkEKDIAkFQrowjZn8LM6On0ZjvI07G7vREm8hKeTcoqNLy5QQvNBVzmEI8iuaATiKkbMZDcEpoQuZiOM41a+61a1uV9+klIjH4wCA/fv346Mf/Sg+9rGP8Z/8yZ+go6OD3mwgckMDiNRiy+4ISBiwi6Owi8NsRHouMpFwMTAhhCMdtGpNB9pX3YeZqSM8MXkQxdIMGBqEFl4Q3vJCVUESXRO6ByasMJEZx3h2GrFQgjsautDd2EEh3Qy4kic9fxFWEuBC1mYuOh7zUD6AKAaKDi+7sKkuzVK3ul1zJhJUO0ajUQgh8Nhjj2H//v149NFHedeuXW8qELmhAUQzGi4SwpJQThHZgW+iZev/VZNIv1iZ3YVgomlhtHfeSW0dtyOd7ufJqWNIZ4dh20UIGYIQGmgeK6kBEwC6poNJQ9Eq4dTEWZyfHeeuhnb0NLYjboYJ1VyJ/9GX8fXTFQWHGfqC31suzwOaeWxNiGplyKWUGXqaYJd6FwVH/8aEEa7X13ujbCUbh+t9c3El1+JynrvU+Vg0QV/zs1IKSql5PyeTSQwMDOChhx7CE088wffcc8+bBkRuaADRQ61eaGqpBcIupB5DZvgHkOEublz381T1cMwrlHAPwCRgJRKNjRupsXEjiqUZnpw+ialUP0qVAlh4iXkiusB5BvkSQRKG1GArB2enRzCYnkFzrJG7Ek1oicbIlHOVVLwgdBXYWElVczjzwlS89ILftWsXNE1DJpN58znHpUqiFwv/LfbYFf7uUirILqfa7Eqd9FLXbLkKtUuq1lqkeup1RLkLruWyVXELrt/FKugu9dzXnouFPwcAsdQ1jsVi1VJeALBtG7FYDMViER/+8Ifx0ksv8caNG0kpdd33jNCNsouaf3U9FlHOneWzP/sVSCEgmCGhIMCo4jppUKSBScJ1yoi03o3GdZ9AuHFHDZDgklV3F5bmOm4FM6nzPD57Fun8NFwGhAxDSB2K/U8lNDBpACQUeV9EGhQkbAYUNJh6CC2RODrjcbRGImQuskPJO4p/MGJ5x6b8xawYRUvhzjYdt7TopNgLby2048eP89TUFIQQ1QXONaWYvERZ5kod2QU36UInvVjJ7CK/u6zvwXvVlmtewfeV/m25n68EbJY715cKChf7CtZC7ffg65JBZQEQXHCdaY7VEhYvr15pOfbCx6302i91/pc6Z4oZrBRc161+2bYN27ZRLpdRLBaRy+UwPj6OH/zgB9i7dy8ikUj1Pqvu5jUN2WwWd955J55//nnSdf267xW5MQHED9K4do5Pv/hZsJWGFAKSlwIQASYDyi2DhYlQ4y1oXP0hxFpupyogAZcxVGou8R5YtjDJYzPnMJUdQ8WugMmAEAYgdSjIeQACSA9UhAbFEi4EHAgokgjrIbSEI+iIRtAcMhDVJVVcxs+mKjxWBnSpwXXnA8juDh1bm5YGkLrVrW7X1lzXxZ/8yZ/wb/3Wb3lREl2fByK6riOTyeC3f/u38bu/+7vXfSjrBgWQORDp3/PrXE4dhqaFINldGkCggYXusRHXAkMi2noXWtZ+FOHEeqplN1zdoazcCy8szbXsMiYzIzyeGka6mIUDgpQhCKFDkQcWAYAwSSj/ZxIeuLgsYbMAkwZDGojoOsqKUHAFNJ/N1DKQkq3wji4DvUmNeIkcSu2O8i1p10gd4E1597yF8jvX+nzV/i0oVvnRj37EH/vYx+A4TjX/GHzewFe8+uqr2LhxIwW5yTqAvK43gFeaO3rsDzk18CgMPQnB9vIAQr7DFjoYEq5TAckIYm1vQ2Pn/QgneskrDa59n0tT3l2sYTBdmOWx9BimcjMoWpYXvtJMCNLAASMJ2IjPToJwFwsNzBocEJg0CNI89lIDIKwYlqPwvtUmOiKyLqpYt7q9gcBs2zYMw8B3vvMd/shHPoJIJLJoKOuXf/mX8Td/8zfXNQu54QEkPfIkjxz6nUsDEJLeo4QBxeQxEmlCD3cgFFuDaMMWxBo2IxRpp4W7jEvZ8SxkJbZrYzo3y2PZKUwX86g4LkgYkFKH8vMhtQDCAahAA0jAJVk9jloAUS6DmPFz60KI6lQHkLrV7Q0227ah6zq+8IUv8J/+6Z/Oa+AlIriui3A4jCNHj2JVV9d1m1C/cUNYfiK9Uhjgcy/9UhU4hJ9pWBJAhOE5YPi7fghA6FAQUIrher3dID2OcGwNki23oKF5GwyzccE8kJVf7MVYScmuYDKX4pFsGjOlElwISGmAhAYX4gIAYfJyIwsBhJWC5TAadOBDa0P1Vo+61e06sIBxpFIp3rFjB2ZmZqDrenUjGrCQL33pS/j85z9PjuNcl2Kn4oa9Qr4DNyI9ZERXQylrBftuhnKKUE4Bjp0Ds1Mj5c4goUPTItB0r0Exn+nH4NnHcfzQn6O/7zHOpM8yMJc09zSxLg7QtdUmQTlvWDexpqmD7lm7he5ZsxHdiQYwAMtvQroUIHCZ0RwSnqo912/eutXtDXe8fgVWc3Mz/R+f+T9gWda8MFUAJE899VT18dflcdzIF8kLYwlEm3aC3cpFvS6zQtumX8aqW34bjaveDU1PwLXzUL6qb5DgYr8qS2gmdD0GVjZmpg7h1Il/xNEjX+aJiVfZcUpzir0+KKwI96o6XHOLqDkSo12r1tD9q9dhXbIBggiWoy7pXHRERP2urVvdrqc9ru9PPvqRj0LTtHl5EKUUNE3DkSNHUCwWuTbRfl0dw41cdROEkgrTe3lo329A1yIQ7C7aB8JMgBbGmt1/B6nHPXEzO8/5mdeQntiDQuY0HNcBaREIGYZLAuxXQXkhIwNMArZyoZhghprR0roD7W07YBrxebmSS5VvZ7+iLHhGwba5P5PHmUwBChqIlg5huUqBFOPDa8x6/qNudbuu/JOXAy0UCrx9+3YMDw/DNM3q75VS0HUdhw4dwtq1a6/LPMgN3YkehIXCjTtIj3Qzl8YBqS0Sx2GAdCg7j0q2D+GmWwFWkHqMkh33ItlxL0r5QU5PvIz09CFY5bQ3+lZG/BAXgVmBQRDSgCQdtl3E0MjLGJs6juamjdzechPi0dpZIOyPbl+BjARqZEwARHWddrQ0Im4YvHcyA10sHijzRBSB7rCog0fd6nadMpBoNEqdnZ18/vx5hEKhOabxJtjb3+Aj6AjMLoQMI9ZyJzLnH4UmEwDcxa4mmF3kJ19GpPl2fx56jWhibDWFY6vRtvoRTs8cxOzEfuRzw1CuBdIiIKEjSDIEUwp1acJxbYxOHMb49GnEY13c0tSL5kQ3Qmasih0rnQUS/FX5fRymlMuvMT/n0ZuQcwtykbdwXXdRCfmLsdPlPu/VfK1r8bwr3Tm+GV7zWr5u3ZZej5fTPW6a5gXrmcHVYVV1AHnDIMS7IPHOdyI7+K2lcxE+0OSnXkHz+s+wNJJUmxAPhkxJPUrNHbvR3LEb+Ww/z0y+hkyqD5VK3mMlWqj6HMU+kGhhMATS+XHM5iag6VHEo23ckuxGS7ILoQWzQC7W0ObhFOPoTBpimce5CkgahJ6opOB5i9mbeSLaG8ls61a3Je89110RkAQbt6UqrC5V3LQOIFf9bvfGKoUbbyYzvpGd/BlAmouWI5HQ4ZSnkB1/Do2rP+TnUGTVa88bMkUCsUQvxRK9sO08Z2ZPYWb2JHK5Udh2ESADwgcTVSPfzkJCsYvZ3DhmchM4O3EKjbF27mjsRnO8tToLpHZx1VogQ3JwapZnyxYMzfAOZcE6FQTYCtjRIKohLlpk4QLAt771LX7++ecxOTkJ27YhhKiO5NQ0rQowgRS167o1ekguXNf72XVdzM7OVhd9bWUZEUFKWW2kchwHyWSyqidERAiFQtXpbMHvA+nrgCXVajEBi3fPB5pddFlRAK5St6WmxC28oYPH8YLnrGRsbvBz8JpL/W6xxyz8qn1M9bFCzNOCEvO+z1UAzmlHAYLEij5T9TjEXOFH7bmp/X31PZc5jwtf90qZ18VEJIPrtVDvbanHK6XgKgXlunCVC+UquMoF2GMQLS0tuOmmm7B79250dHRc0kEsBiCLrac6gLwBxj4TiHc9jJkTRwEZXjyMxQokQ0iP/AjJVe+BkMYicZ+aBe6zEl2PUUv77Whpvx3lcoozmXOYSZ1FLj8B2y6CpAkpNDAFi1JAkzpAEo5yMJ4ewVhmAtFwktuTnehIdiAeitKizouAM6k0n0lnYUodaolQl8tATAc2J+QFKftaxdBPfvKT/Nhjj121c/3Zz34W73//+xGPx6HrnqC8N6TKBZHn8HO5HJ566il85Stfmffc9evXI5VKVROIwc0agFMAFq5SYKVQt7pdyzAUCBBCQgpRvW8c14VSLlgtvTVpaWnBr/7ar/Lv/KffoQAYLub8l2Ig1c9SZyBv5ILwLkC8693I9H8VUBUAcpG9J0NIE5X8EDJjz3Bj93tpTqpk8VgSLciVhEKNFAo1or39NhRLMzw9ewZTqXMolDJgUj4rISh/JgZBQJcaWHizQM5M9OPc9AgaY83c1dCOlmiSTM1zxLZy0Z9K87GZFIwlwCMIVVkKuLlRwpR0AQQqpSClxBe/+EV+7LHHqjv/pVjPSnZPxWKx2vS0kud86lOfwrZt2/jf/tt/i3A4DNu20d/fD9M0IISc9zlqKf7FFFPrIbK6XQ5bWe53tSoTwaZoKVbAzCgUCvj9//r7mJyY5L/5m7+hgFUv995LhZGv91np2ltjiRDAClqojaJt93Fh+NuQesOi4Q1mhtBCmBn4NhId97LUorRk9nkxVlIDJpFwM61e1Yzuzl1IZYd5fOYMZnPjsJwKhCRIIUEgTw6avVkgUtPggjCRn8VEPgNDD3PEiICEhoKtUHBcSKkvGZohP3TVbACbExfOPw9KAUdHR/mLX/wiNE2DbduXnWw1DAPFYhH/5t/8G3z+85+nIATGDGiaB9yDE3lmMFa3xchjP97x/uZv/iZ95zvf4ZdeeqlavmhZNkxzftx3qbBC3er2egL6YmGxpTZUuq7jueeeq845v9jGbFkAEXUAuW4sseajKI4+iaVBgSGEAbs0gZnzj6Ntw2ewLAtZAZgIIdHcsIaaG9agWM7w+Ox5TKaHUah4fRxSmtUhU0ESXZeeWKLjukiVCnAhQUKHLrTFgm8LjgC4vUlC0oU5gKBB6Y//+I+Ry+VgGMYlzUCvjYUTEcrlMj74wQ/ij//4j8lxHEhN88BDEmZzFf7aT87i5FAalu1i18Zm/uX3bSHvhvIYz1e/+lXceeedmJmZgaZpPohYME2zGp9+Kzinul0/7ONyHrPw8Uop6IaOlQohLgsgqAPIdXCnCoAVzORWCrfcyaWpn4H0hiUWgILUYpgd+iES7fdyKL6WVjbqdmkwCYAhEkpSb9ctWNOxDbO5CR5Pj2I6OwXLsSG1hVIGfoUGCQgIqEUmGdaaAFB0GVsbJDpCF/Z9MDOklJienua//du/rcopBI77YjdK7VChILn9i7/4i/jLv/xLL88iRDU5OzRV4L984hQmZkuIGBLCAH56eBx3bGrl2ze3EsPLc/T29tITTzzBn/jEJzAwMFDNf1QqlSqIXI8OxxtPzJf9GrWFAtfC6V3N5y3xYq9bm8KlroErmRK53P+X+rm2qKP277ZlLzud8GIAwssUctQB5I25/T0WsvZTKE29vPzjiMDKxsTpf8Ca2/4DrrQJb2EDoRQaWpOrqDW5CoVKns+M92EiNwNBctFPzRd9faCigLaQxK2N+qJNg67rQtM0/K//9b8wOzsLwzBgWdYlh6za2tpwxx134Fd+5Vfw3ve+d97b5EsOv9Y/i+/uGUSxbCMe1uE4LgQBuiT0DWdw++ZWgL2bxnEc3HnnnfS9732Pd+/ejVKpVE2aF4vFJUNZAQO6LDfGl+BkaOn93/IjVS+cvLfQISxWzXWxn6/W31by/4s64xXcEITlHWAVRGnpa3UlI3mX+/mikxj9SYMXA14p5bwGwOC7X7G4ItchpFiagdQB5HphId589FDLHRRq3sXl2YMgPbm4C2IFqUWQTx1FavQ5bux6kC5VZXdpVjKfZUTNGN2y5ja82PciFy1rrnR4xeAE2AyEJGF3q7lo6CpgH8Vikb/0pS9V2Uc4HMbnPvc5PPjgg2hubq4mCYOxnAHT0HUd0WgUTU1N6OjoQDQa9eRelIIgwnTO4u/vG8G5iTxSuQokAYYm4boMRzEIDCEIs9lycBqq8WKlFLZu3UqHDx/mSqVyQanwwhv7auyol7oxa+PNizm/eWWqgRPlC1/vYqW9gVepfc4Fr32R11rJ35b67JcCIFcycnclj1sJ850XhmVVXeCLleauBCgWG89b+xWsQcdx5o2otW0blUoFpVIJ+XweuVwOjz32GA4dOgTTNKuMI5BkD+6fizIQ8ebsxXrL5UDgy7Enen8BpZnXlt0aMCtIGcbkuW8h3nIba0ZiBQn1SwgnEEC+nuVoaoTLdtkbCHVJcAQoX1/r7e0RxHSxLPv4+te/jqGhIYTDYZRKJfzGb/wG/uAP/uCSDyi4MQImULIcDE4XMJOrQPigRgBcxWiIGbBtF8WSjXTBYzxiEce5Zs2aekKgbm86Gx4e5v379yMcDl8AICsNYS3s+bmSEF4dQK4pC/FyIeGWOynStpsLU68ARnLJGBEJHXZlBlODP0Dnhk9dYkJ9IXTxBbvVfCnD56fOYjQ9DhLGJcm0B/0eDODtHTG0hrRFwSMYiWnbNv7oj/5oXnJ6y5Yt1R1WwD5WsnNfGLNd3RKl//jxHTg2mOb9p2dwdCCFUtlB2VL4F/euwZq2KH7nKweQylZQrDgcMbVFK8RWcg4JVye5/lZLYL9VjvdqrI2VjKgNZnQUi8UlN1krZSBL9Xpc79fsLchA5iy54ZdQmNmPRVu559wapBZBZuIVtKx+H+tGwyWxkLlkuPCqKQheJ3p2nMdmBzCdm4atGFILgy9hsQgAts9idncm0BkxlhRLDNjHY489xidOnICu6/M6uYMywyuRNAnmrG9f00Db1zRgKlPmb754Hi8enUS+ZKO3M0F3bmnjFw+NYmgij009DV7YRdBFb6K61e16BUpN05ZkDUEobCVA9WZd+2/NO7ZakXUTxbreDWVnANIW3+OT8FnILKzixAp2ON68kKBM1duxe6c5V5zmc6Ov8v4TT/Dh/p9iMj0MANCljksR3RBEsJSCIQXu72pCZ8Sk5fKQQbL6i1/84gWL/WrpYAUvq/w+j9ZkiH7lvZtpdVsUh/pnAQAfuHs1hCD8aN/QgjxQ3ep249lKAGQlm6d6COv63KYAYDRt/GUUpvfBtfIgI+lhqj8fhJUF5VagGIi33I5wbPV8gcUqYABBye380l2FfHGKU5lBzGZGkCulvYSyNKDJEEDSC0Et5/0XhKwAQtlx0RSO4s7ONsR0PxS0xPMDmv29732PDx06BF3Xq4zkagJILbiBANdlSEnYva0N//uZfpwdzfL6rgQ9dHs3/+Bn57G2PcYffPs68m40L8Fet7rdSCyoNgdysUbCNysDeQuHsLzudGm2UPst/4nHD/4unMosWOhQkCAZhRlbi1ByM6JNNyPechuRL8xY7TaHJ0xXOyDKtvOcz48hnRlEJjeKQiXrVSoJEyRD0HUNimU1tHUpjtkGw1Uu1jc04ea2VpJEFw2mBQvzS1/60oqEAa8qPgNojJlwFeOpAyP4fFcCH3twPU4NpvAPT53C0GSOf/6hTWhOhKgOInW70axWc+5KGMj1zNLf0jmQakK98WbquecvOT/xIlw7B2m2IJTYCDPeS/Pk3NmtltgGzth1KygVxjiXHUQ2N4RCYQqWU4ZiAskQhDQhNBOAhAtcMnB43QSEsusiZOjY0daJ7niCgItnYgLZknPnzvELL7xQ/d3rufOpWC7CpoYj52YxNJnnnrYY/fIjW/m/f/0Ann11BCfOpfAv37+Fd21pJxXEg+vd2XW7ARjIpYSwLqaVVQeQ6xxENLOFGlZ/eOHVAysHJDS/MUwCYJQKo5zPnEU+cw6FwhgqVt7TdxIGIE1omgmGBgUB5YeycBn9I0TkjcglYE2yBVvbOimk6XM9BBd5fgAgzz33HCqVyjzZktcreTedq0AKQtlysefEJHraYtjYnaTf/Pmd/OffOoKxqQL+5/9+DZ94aBN/8N5eL6TFQbVaXeqjbm9eU1eJgdQB5M0AIvAYRjAnJJgwSKRBuWUUs/2cSx1HPn0a5eIUHGWBSQfJMIQwIaQOJgEXwp8voIDLcH5B85piBdd10BhtwKbWbrTFAtbBl6yNs3fv3td94QaOf2iyAEEAScKJgZTf4QtsXt1Iv/0Ld/BffecoDvZN4cvfP4YT52f4E+/cjLVdCarP3q3bDRDDuipJ9DqAXPcXOqiY0qrbetfOcTF9AvmZQ8inT6FSnvUk2GUIQoah6TEwJBQkOKi8Al12j2Ew8MdVLmzHRSycxLrmHvQ0tFIg2UG4NGG1YFH29fUtSYevxTTCoKR3NlfhoakCdE3AcRnpvIVC2eFYWCdXMVobw/Tvf+EO/HDPeX70x314+cgYDp2axi0bmnnX1nbcuqkNzQ1hcv38SB1T6vbmwA2uhrDqOZAbGjTYkzcJRtDaeS6mDiE3tReF9EnY5RQYAtDCEFoIAjoUguEyAWBc2c7Ba+oDHNeBgkIs3ICeph50N3ZUpxNeDuuoXZRTk1NLLsRrsfMJbqCD/bPIlWxETc0bxOOX+AJensOLVDHee/daamsM83//h/3QJGH/8XG8dHAEuhT4D79yN+/c0ka1wFS3ur1ZgGSlAFLPgbw5Limqela+42dlo5Q6xPnxF1CcfQ1WeQZMEpARCD0KQEJdYVjqgiCVzzYUK9iuBRIGmmKtWNW8Bu3JdgpGilb1jS4DPILn2raNfCF/UZC5mrkGIoKrGC+fmISuCTAYSgHRkI5ISJs3n10pAASUKg4UA/mSjbAu0dvdgFs3taGnI47XTkzwTb3NFDK1OojU7U1jdQZyI4FG0BToV1HZubNcmHgOhYkXUS4MeKkPLQahxwCSUCxqWMaVeazqe4Og2IXjOoBQCJsN6EquQntjD5KRRlro/AOnXrsQF5MRWc4cx4Ft25e1cC/vpvHCTQdOz/DAZAERQ4IVw3Jc9HYloElvRrzw5VSkIPxozwB/5YnjcF2Fd97Rg3e/bS3WdSVJSoLjKPzbfzzAqzsT/G8/9zaETY3qIFK3N4tdSQ7kSiT/6wByxfQxcLhzoOGUxrk49RJK48+gkj4K5ZTBWhRCiwDQoUBzoHFl4u3zWI6rbCjlAMKEaTagLd6FloYeNMbbSQoNSwFHAB5CLD6h72qwhquZA2H/M1VsF0/sG4ahCT9MBWiCcP8tndUHKngg8pUnT/D3XzwHVgq/+Mg2fMCvxPLAT0HTBD7wwAb83bcO46vfPoJ/9fM7/eOvI0jdrl+7VOdfT6JfVyEqWe0Wd+0cF6ZfQW78GVRmDoAr09CEBpJhSKPBr5pSYLhePuQqAIZSNly2wSQhtBhi0TYkEz1oSPQgEW0lKfX5YECLS4cH4DE4OMhPP/00stksbr/9dtx3331UCzgXW5hLzRq42gtXKY9RPLF/hMdSJUQNCWYgnbNw/44ObOhKeL0e5OVA/uFHp/gHL52HIMJ7dq/DB+7tJVd5ysKCCFIKMID33tdL+46O87N7B3HH9k6+fXtHvfGwbm+CTSxflT6QOgN5XYHDA4H87CFOjz2NwuTLcEsjEAB0aUIaDSAwwMor26XLYBtVfSvPwSlle18kIfQIwuEWRGOrEE+sRjzWhXCogRYuimDRLLVwAvD4xje+wZ/73OeQzWarf/voRz/KX/3qVxEKhehiTERKCV3TrzmAuD54HBvM8DOHJhA1NRAY+ZKDdR0xfOKBXmKGP0yL8PKxcf7+z84jbOpoShj41Lu2UBDaCg6HyAOlkKHh/jt6cPr8LJ5+6Rx2bmuv94fU7U0BILVSJnUAuS6Bg6thKtvK8OzYs5gd+RFKmZMgVYaumdD1pNfSx64HGpfLMOAnhF0LLlfA0CD0OMKRNkRiPYgm1iIWW4VQuJkWDp5aCWjUggcRYXh4mD/72c+iVCpB1/Uq4/jmN7+JTZs24fd///erWlfLAchif7+ajYQBeIylSvy1585BkwQpgHzJQVsyhP/zg1sRNjUoP+dh2S4ef/4swqZEueTg4+/cBtOQUIovLGrzT9X6ngbEogYGRjMYnypwV1uMVsLA6la3NxJAVur8lwQQ1JPo1+jiuD7jIJSLYzw2+D3MjP4YdnEUutChaSakZoJ80GBcioTIXMKdGWBlwXXLYOiQRgLhaBfC8XWIJnsRjvXADLXQwgVQO9McuLSxlMGc8n/+53+ugkeQCBdCQEqJr3/96/id3/kdGIaxaCirdkpdPB5fmoFIcQXXwCvDlYIwni7xXz91BmXLhaFJFEoWWhImvvChrWhOmBR0l4OAY+dTPDZbBAHYvKYRd23rIG9mCS12JQAAsYiBkCFRKtmYmimgqy1Wr8iq25uGgVzMltzIMeP1mz7/FgCQoAyXSKJcmuKhc9/A5MiP4FZmYWgh6EbSCyyx65XdXiJoeNImrqfCCwGhxWFGuxFObkKkYQsi8XUwwq20OBNamAi/Mu92+PDhecOfahfl2NgYxsbGeM2aNUvuxINZH729vXj11VerY2xrF+2llggHoEH+aF4C4ehQhh99aRCFkg1Tl6hYDuIRHf/mg1vRmpwTSnT94zh0ZhrMHnN58PZu77MyIJf5KMI/p0oxsgWr7p3qdl1bbQXlFTGQYONVB5CrxzqUsnH+3Ld46Nw3YJenYeoR6EYDiJ1LD1EFTEO5UHYZTBIy1IpwYgsizbci2rANRrR7fkiKlV9xJGpCXHN5mKu1+PL5/AX0Nfi/ZVnIZDIX3QEBwMc+9jE8+uijIKJ54SzHcdDY2Lhi0AjyEwHojMyW+KcnpvDq2RQIDEMTUIpRcRR+8eH1aGsIVZkFA1VNrINnphEyJIolG8mY4ZX0glGbQF9otquqTYiZXAX1KSJ1u9FDWAsrua7HkO2bBECCHb5EKn2Sjx/9c2TTx2BoERhGsia3cYlsAwA7Ba8vIdSBeNOtiLbejXDTLdDMJlrIfHyo8Hs6gHJxgnOZfuRzAyiVUnCUQjSxGqtXvxO6Hrni+emlUmkeECxcWJa1/E5c0zQopfCRj3yEPvWpT/HXv/71eX9/97vfje3bt1PAVC4EDXg6VjWgMZ2z+OxkAceGs+ifyKNsOTB1AVYKrqvADGiScGo4g5a4yY0xg0xDAgzM5iv8v398BqlcBSFdQNME/veP+gAG37yhhWRNCCsIebmKoWuEmXQR5YoDKQgTMwUQUAeRul33djUYyEpfow4gi0NHtcT17Llv88lTfwehbBhGQzW/QSt1JQFouGWwW4YwGhBuuRPRjnci0nInZC1o+LHHuZkfwSwQQmb6IE8MP4tScRK2awNCAygECB358UmUSils3fpJEGmXFaMPFlOgnHupi26xx/zjP/4jfehDH+If/ehHsCwLu3fvxi/+4i9SkENZHDQAy2WMpMp8dqqIc1NFTKTLKJRsEBiaIEQMDY6jAg1KMAO6FHjm0DhePDqBqKmzqRPAjJlMCbmChZChwXVd6JrA6EwB/+0f9mNNW4zv3NqO27a0Y3VHnDQpvHJfQXBdhadeOg8QYOoaTpyZRqlsIxzS68Oo6nZDA8hiG8g6gKyYAnr5DqUcvHb0T3lg8AmE9Tg0TbsM4GAoJw+wDSO6GtHOhxHpejeMWC9dGJryQWMee/DAwypP88DJv/cqhrQIdD0KJukNiSIJ00ggmx3E6Og+7u6+h6rSKddw8a10YX784x+nj3/844s9CMFUwwA0BmbL3DdZxPnpElIFG7bt5ZQMIkRNDa5y4bqMeTM8hE+6mBANaXBsF5mCBcfxJrNJYoQNDY7jVsHG0AV0SRiYyOHUYArf+MkZdLVEuLsthoaYCWbG+ZE0BkdzCBsSkgipTBlf+dYR/tWf30lCUB1E6vaG2tUowa2X8V4j8LDsPL/86u9jamofwkZDtX9jRe6iFjgARBt2oKHng4h2PAihRakKDMG8Dj80tfyFlpDSBATA8zrX/fdiF7oWxtj4AbS2bmfTTFxxKOtqmeu68xajkNLLa/j/H8/ZfHyiiNNTJczmLTjKAwwCICVBFxLliuvpD/P8cFex4sB1FZTrATGBoZHHRgR5Wlhck8eofa5iRkiXMDUBx2GMTOYxNJb1Z4IwTE0gbBq+FDwjEtax/8g4CoVX+Oc/sBWrOhIrbqysW91eLwC5FPZQB5BrAB7lSop/uvd3kMqcQtRsBCvnUi4rlFOAABBvvRtNaz6KWOs9NWzDxVzllVzR6zEzdLOR4k3beGZiH6SeWJQDEQlYTgkjY/vQu/adl1xuGiyYpfo7gr9fqgxJ8Hh/hHv1M/WnKvzaaBFDqTIqtscUTF0gxIRC2UVrTMf9mxrRHjdw4Hwazx2fhi49SAxyHu++tRPxsIZUzsJkuoSpTBljMwU4rvKICc9JnQhBYPayGEFz4VxfCkHTJYQpIfziBPL/TlVm5oHI6fOz+B9/9TIefnsvv+eB9SSlqLORul1nvuzqJdHrAHIJ4FGqpPjZV/4jstl+hI0klLKx8kAQQXEF8YYd6Nz4S4i17JoPHCsGjYWv6llbz7uQnj6y5KRBBkOTJqamT6Cr43YOhRoui4UYhrH8xdMu/fJVPwUBQ1mb944UcWamDIM8vaqILuC4CmBG0VJY3WTi47e1U0j3jvNt6xt5z+lZ2I6CEECl7OKjd3Xj7k3NtPB9hibz/Fc/OIVUrlItbS5ZDioVG4IATQhI6QGKVWEUSg5YAVFTQApRBRaxyA2mFCMU0kCK8b2n+3Di9BR/8sPbsaojUZc5qdt1Y/Uk+uuM1kQCFSvLP9n7O8hkzyOix6HYwcrdPYHZga4nseFtf+4NY2LXC6uQvAK9K3i5AlYIRTqoofU2nh5/BcJILvFQAdutYHTiEHrX3H9J4ZVgx5FIJC7oXA92JaFQCA0NDZd4fj3WUXIYL48U+fhUGSXLxdu6o7i5PYxvHJlBruxAEKHiKDRGNHzMBw9XcfX3Qb7EdhitCRO71jeSYo9NEOaYzeq2GK3tiPNEqoSQIZAve7Imuza1Ym1HHA0xA7omvc9UcTAwlsX+4xM4dGoCmYKFkCYRMsWS5VZKMSSAeMzAwHAa/8/f7MG/eGQr37mzmwLBxXpEq25vVgZSD2Fd0s7Y6wFw3DKe2f/7mM2crYLHpW0mvXJf185j6Oj/zW1rP4ZQfD3NAw52AYjLHjkLAG3dD2B26qCfP5GLMikpDUzN9qG76w42LqOs96abbvJ24DVdqsHPa9asQWdnJy38+5I7IfZy3BNFl398vojZknde37M+jh3tYSraii3XAwkoL9707puaENYFHBXIsACzBRtlWyGkEYquwppWT6I9kGavZSCuYkxlypDSky/5+AO9eM8dPbQUO1jdHse9t67CxEyBD5yYwJ7DY+gfTiGse7kagRqAqrmnXNdjI6wY//iNQ5iYzPMH3r2lnhep2xtmwZq7GgByPTOQ60RD2AuEMzOee/WPeGzmCEJGAuqydKvmGMDM0Hdx9uVfw8DeL/DM2a9yOXPc94RyzgtdUv9IwEIYoUgnNbTsgOuWlqyyEiRhWQVMTp+6pMUkhOeQP/nJTyIUCsGyLEgpIaWsKnx+4QtfgJRyRWqfAXgM5x3+wbki8raCJoB3ro1hR3uYGMBwxkLRVpAEOIrRGNGwviVMgBfa0nynv+dMqip2rxSjLWHO5Sb8JLervGs5Ml3k0ZkiXJfxyNtW4313rSYiD1iUf71rcyDK/317c5Te9/Ze+p3P30Nb1jYhX7RRrjgoVmxULAeOo/zzRNWmQ6UYUgrEYyZ+8vxZ/P3XX+Vyxbmgk79udXszMJBa+ZI6A7mog1MQJPHikb/ic2MvIRZqgKvsi4atyK+aIq8eyvNENSde0xMAOyjMHEBx6mXoMoRwYiPH2u9HpP1+6LG1c8ykNqm+EsADobXz7ZidPgIscYEZDCE0TM72YVXHLSsu5w0kR3p7e+nxxx/nf/2v/zXOnj3rHZOm4Qtf+AJ+/dd/nQLNrJWBh8tPDc51cO9eFcGWZrM6KvbVsSKkz5GkAHJlF986NMU9DSY0QShZLk6N53FusgBTF3CVgq4J9I3lccuaJDfFzLl7wP/+8slJFMsO2htCeNft3eS6DBKe0OKi3I7mQm2O671+YyKEZNzExp5Gr8u94qBQtJHJVpAv29AFIWTo0CShXHbAroJpaDh8dByFgsW/8KnbEIsa9SFUdXtjfNvl5kB4LlytuJ4DWcbBuRAkcfDs43z03PcRMxuglLMsNSJfr8pxCn5ynaHIKzMVwqj+PZA1EVoUElEQO6hkTsBOHUTm7N8j1Hw7RzrfjUjb3RBarFrWOxf2oKXfH4xoYi1FE+s4nx2E0KIX8hhmCKmjUJxBNj/GyfiqFSvIBiDynve8hw4fPsyvvPIKstkstmzZgs2bN180dMVVFgQM5l1+dsSqki5dELrjGnKW4qmCg9fGihjN2tAlQblcBYHj4wUcHclDuQxXKQgoT67Ebxw0NIHzUwX88Q9OozGqcXsyhLWtUXQ2hjE0lcfLxycRCWmwXQVXMYcM78CVqlUmXpTkQdMEShUH5YqD//GbDyIeMaqPrFgOJqeLfGZgFidOT6F/II1svoJ/+bFbkMtXcPzkJGZmijjVN4V/+Pqr+Nwv3AFdF2CuBam5UmSv7aeOLnV740NeizEYVnUGssTJ8ZhH//grvOfEPyBkJvzxs8udaAnHyUMTEs0ttyORWA9JElZpHKXsaVQKA3DcCjQtDBIaANdvEPTVcbUwJCJgdlGc/BnyU3ugRVYj0nYvx7veCTOxiebil6raib7YhSUiNLfehlzm3NJg47/O1Gw/kvFVlxZf9EEkEonQgw8+OG9Xsxh4cM17BmmCo2mXD0zZ1XilC8Blxrf6cgAzChUXzAqGRlDO/IUa0gUgGcplKEVwFeYAxmcKukZwbMbobAnnJ/J4+fg0BDFsx4Xhh75yRRt/9u1jeP/benjjqiSZupz3Gioo0fUbGV2XoWkCR05P8eh0AfGIMS8pbhoaeroS1NOVwIN3r8XUTIF//EI/0pky3vOOjfTQfesxOVXgF18+j5++eA4/fOoUf/CRreS6QXUWLwperDyGVLe6vd4MZLkQWL0Ka4kTQySQLU7wc4f/ApoMYum8LHhYVgYtjTdh69ZfQ0PjVpp/sWwU0ic4NfYsMhM/hVMagS4NkAyh2jDod5sD8Oef63CsaaTPP4bMyJMwG7ZzvPOdiLXeBakvzUoCkEk2b4U++DRc5QBkLHqcQmhIZUfgKge1I2xXCiK1i8jroxDVXg7UOF6qCVuNlJiPpV2MFV1o/h/cGoBx2XOYpkYeOLh8AQtAVQjUb5jkxa6jx3IMTUAXBKUBrnKhS4LruFDK6zY/PZrFHz12BG3JEG/qTmD7uiZs7G5AY9wkucCTaxrBcRUef+YMJmcKmEoVuaVhrgiB/TcOwlKtzVH65Id34OjJSX7tyBjv3NFJ7W0x+uiHtmNoKM0vvXwe69c387ab2oNRVZgYz/GpExPQNIHungZ09zSSkHXwqNv1x0BUPQeyeJCFGXj2yF+jZOUQ0aNgdpYFj4qVxtruh3DLzf9vksLwBQ7nvKgQOuJNN1O86WY4G3+BU6NPIT38A9jZPggCpAz74S01ByhwQaRDGlEoEIqpIyikjkALr0Ks7W5OtN+LcKJ3bt5HjRIvM0PXYxRPruOZ6eMQurkECEiUK1nkClPcEO9cURiLMV/Guco4/OcR5pMeSzHSlstjJYWREmPW8hLupgBcdSEsCwBM8DrEF1mflsOwHQVWHgORxBAL4F0QLSqYH7CK2v+bugRLwnSmhOGJHH6yfwiJiIGe1hhv6E5iXVcCrckwhCCMT+fx5EvnMDiehXIVBseyaG2MwHYYUs6N/63NmSilsH1LGx0+Ps4HDo3w7bd44cKbd3RicCiN733/OIjBXV0JHHxtBK+8PIBy0YIkwNAk2trjfOuubtxye089mlW36yyEVWcgF4SuiASODDzN5yYOIG4mwewsE7oiWHYWm9Z/DDtu+lWqfY0LQckLVWlGA7Wu/ThaVv8ccpMvcmboO6jMHIDrFiG0KCD0moQ5++NtJYQWAUiDY6UxM/g9zI48g3DDFk603Y14y63QjYbqpWa/HDiWWIeZqaPLfHqCYgfp7Aga4p1YSTkvBbRiEXOUQsF2OGs5SFku0pZCxmYUXILDEkJoMIQHDrXEguDLVZHHRmzX6/JeaC4DqxtNNEc0mJLguIyTY3nM5CuQ/qdnAGXLhe4J7cKyFWzHhXIBU1s85MfM0DUBLaJDKYWS5eBI/wwOnJoAMUOTAsSAZTnQBCEW1lEuKzzx07PYsbEVxrzQ1xwDIX9+ulKMm7d20N/94wFmBd61cxXdfedq6u+f4b5T03j0sUOIhDWUijY0IdDYFIGhCyiHMT2ZxxPfPoKxoTS/+4PbicSbo4ekdtJl3a7f67MCBFkSVOohrIU7ayIUyil+ue9RmHp02bwHkYDj5LFl/cexY8vnfHFCWqKiqbbpzp/PIXQkOh6kRMeDKKePcW74eyhPPAe3PAWSEUCP+5MHaxmGCyINmh6BYkIhdQK51Alo51sRbdzKydbbEWvYTFIL++zAuOgxC5LI5Ceqx3Qxy5TznK+UUXEc2MxwFMNWQMlRyNgOSi7BZoIiDQI6SEhoUkIHQfnBHtSo6jIAhz1mwa7HTFpjOnIVF9myXxZLQNllvHNjA3auis67JNs6o/w3LwzO0Q0GPrSrExs7YnAVI1eykcpb6B/PY3/f9AU3QjVhXSPErkkBGSKEDQHXVZ6qrwKiYd3L17gM09TQd34W//WvXuZ7bu3C+p5GrGqPUzg0X+lYqbmk+H13r8OX/nYPWpoivHZNI33207fTT1/o5/37h1Aq2dCkwK5dPdj99nWQmiBWzNlMGU9+5ygOHxjG5q0d3Lu59bruaA8AOWCm9X6XNzkDWeY61wFkQXyDSOCVM99CvjyLmBFfJnRFcF0b4VAztm/+lzQ3LEqs6JIEw50C0Ak1bKNQwza4Gz7HhfGfoDD2Y5SzZ7wyOS0OiLDPSmiOlUBCaGEIknCdElITryA19Sr0cDtHExugmQ1IzZ6EkCaWE70SQqJYzsCySzD08LIs5PDwSR7KTIEhoSDBpFW/k9AAoYGEDoMEWAiACUxeGMtlhiKvJVwpP0SlAJ2AppBAa0iiKyLRGpGIG4IOjZf4x/05hKT3/OaoVgUPxcEhMWIhCUMKOI5C2Xaxe2MT7qmRL+loCAEA7tzYjHLF4VdOTSHsd7CXLQchXcBxPNFF23aqEiWan9SuFVlUyvv8wtcfC5kaBkYzODvoNRU2NYS5pyOBjWsasbm3CWtWJavNiUoxVncnqakxzP/46EH8+q+8jRsbwvTQOzfSzltX8fe+dxRnTk9j0+Y2xOJmEE6kSNQAwCyEQCF//U88DBQKRkdHub29nYIeoTqIvDkZSL0T/RJCV7P5UT429DxCehSK3WX6PRhCSFh2DmMTr3BXxz3V8NVKd/LzHuc/T4ZaKbH255FY+wmUZl/j/NgzKEwfgF2ZAZMO0qIgoYPJl9HwFXeJdGi6CZAG28pgZmIfmCRIC4MotPTsYmaQELCcMgrlNBt6eNG+hGD+iDedz0XYCMOF8OTifQABBFxffp0xJ1LoMqPBlEjoOkxNQBMCAgydBGIakDQE4vqFq7Rgzw3KclzGqqTHpoL+kUAAMVV0ULYVdOFpWN26JolAviSYKOiNpaV5+Q+lGJ9+x3rsWNcEy3YxPlvEwHgOYzMFjM8UMDpVQKFkQfMv0Zw6MFd3ZgwgHNK9PhXFyOYqODg7gYPHxhA2JFa1J3jn9nbcccsqNDWEyTC8FbXrtm6MjWdh6JKjUYOamyPU2Zngk8cnMTiQwrrepurn/OlPTvP0ZAHRmIHutY1LRRWuG+ZhWRY+//nP87e//W1s3LiR/+mf/gnr1q2jpSr06vbmBZCAgVyPQPKG5ED29X8fFaeEqBED+GIKu1590b6D/x29q9/HG9Z9GOFwG80LU1VZw8Veai7nEUiQhJtuo3DTbWiy0lyY3o/81B6UMqdhWRlAhDwwIb/qyX8/QPkhLs1nB+Kig+8JniZXoZhCY7xzyccAwPauTWQp5rFsCoYfJmP/32KHyWDc3ZbAqqhx0SrU2jVIBEwUHH8WufeHjri+KKidnynBUR5raIzqaEuaJGguT0MgCAIsR2FoughTEyhbLtZ3xvHALZ3VT9XZHMHOjS3V159MFfkvHz+CY2dnYOoC+aINjQDTBwHHZSjXRdHvDTKkgGlIRCO6BzqKMTqew8BwCs//7Dxu3drBmiRMThWQTISwdUs7/c3f7eVNG1p4ZrqAEyfGEU+Y2L9vEL3rm7mlNUqv7h3k/XsGQAA2b+9AY3OErtfdfDA98r/9t//Gf//3f49kMol9+/bh3/3Wb+HRf/7n6zrcUbfLv+ZveQYSlO1mSlN8amwPTD1y0Z6PwIURSRCAM/3fwOjoM+juvJd7ut+FRHIj1YapVs5KaE6/KmAlRgMluh5CoushOOVpzk3vR2biZyhl+6G4AtJjNUAyByYMtaKtagAwhXL6oo+VQmLX6q10amKQz8xMAIJAQi4uHe+Dgik9B+7yInpRVAPFNR+14jJSJRdSeKxBk4S2mD4vJks+2zk9UYAmCZatsKojDF2KeRE7xZ6O1sBkgaezFYQ0gULZxuq2mL+LAoSoATD/87U1RigZNVkxo2Q5uOfmTrx/dy9iER1SeBpauYKF6VQJY5N5jIxnMTFdQDZnQbmu34muIRw2YdsKL+0dgCRCOKRj36tDGJ/I8vmBFM71z0IKr8s+mQjDdRT+4e/3oiER4ny2AkMTiCZDuPv+9XMJpOtwNyulRC6X4y9/+csIhUJgZkQiETz33HOYmpri1tZWqoey3ny2nJx7HUC8KDcIEkeHfoqilUPciK+Afcw9GwAMIwnXKeP8uccxOvQkmhq3cWfXA2hpexsMs5HmbjT3MlmJn9wNtVBj93vQ2P0eFFLHODX6LLIzR+A4eQgtBhK+ECNf6iIRKFdyy1LWWmDY0r6amqJJPjw+jLzteJMYFzkzgoAXJ3K4szXO3VF9jk/UFHLVirwo9rJI6bLLgf6VrRhRQ6IhrFXrhAOAmM7bPJ6xYEgBy3KxtjU8j51U34CAvtEsHJdBvvx7Z1PYj9fPb95T7CWoB8ZzfODUJKQgrG5P4P/z6dvpYucmX7B4ZCKH88Np9A+kMDyaRTpbQkiXiEW8EBwxMDaew/BQBtGIAUlecPDhhzbippvaoRTj+WfO4OihEcRjIRRzFdz/8EZEosYlJ89fL4cdsI/XXnsNw8PDiEajcF0XUkrMzsygr68Pra2tWGzGfd1eX+d/qSGs5Z5fBxB4woKOsnFibA90afqAsryzJT8jQKT8xjFP9kQ3kiB2MDvzKtJTexEJt6K59XZu63oIyebbaliJC1qx6m5txzlXGVO0cRtFG7ehnB/imdHnkJk+BNvOg4UJCvpKSMxVcS2DgYIEKnbR1/4SK1o8bbEk3bs2ilfHhnisUIChmYucW4LLwAsTRWxKhnhHo0HGAgdY+7+gX+70bAUOM0zhdZknIhKhIBnBXl5FgjA4W0LZUQhrBFMXWNMcmRdyq72Jzo7noUlvzKyhCXS3RC/8ADWA881nz8CyXTiuws89sAFEXhNhoOw7r2HSf59Y1KDNvc3Y3NtcBZQTZ6bx9PNnkJotQtcEBAlIKWBoEkQEy3KwvrcZd9yxuvpJdt/byyeOjsG2HCQbw9h6cxfBz+lcb+BR64xOnz5dVSNQyisQcV0Xw8PD122svA4gl/8ab3kACRzm0MxJns4NI6KHwcso7RIJ2E4BjnIgwdCEgCG9yiMvZOP1jGhaFBKAbecwPvQEpod/iERyE7d1vxdNnQ9CM5JUG6bCimeT15QD+88NxXpo1abPoG3NBzgzfRCp6cMoFsZg2wWAJEhGLzJrxAMkx7XguBU/t7F8P0hAYU1Nw9t61tG+0WEezuWhS1ntKmd4eQIhAUMCJzM2RkrgjQkdq8KEkCRiZtgKbCmG4zIqLmMkZ+PoVAWmFGDlTQ0s2gq5istxUxIRoPnn4MRYAZrwHHtjREdTzBMnDFiK8qXcx1MlHpgsIKRLVCwHbQ1hdDb5BQM1x+kq7/EnB1J84NQUpCCsWdWAO7a2kzfhUMwHvoXFBjzXaEkExKIG3XFLF86dn+WXJnIwdYli0UY04pX6lss2ykUb5bLtqQSPZHhyIo/TpyYhBMG2FTo6EzBN7ZLnthARsjMF1k0N4Zj5uiDJuXPnFnU+4+PjdW9+g1mdgdRY38Q+qEBfahnnbTtFdDXfjLbGLWC3gnxhGLlcPyqlSSgwTM30ejfg+npaGjQ9CQkXhUwfzqeOYuLs19Dc+SA39bwPodi6CycSrjTGXa3g8pLYutlALaseQMuqB1AuTnAuew6uU8bU5CGUrRxA5rLH5ioHjmOxoXky6hefwU5VmLm9cxXlrQHO2A4kSdjM0AWhKawj5zBy/jyPvMM4MKtwmBgamF1XeZ3lroLjKjiOC8dlGDQX2tIEIVt28LUDU4ibgqO6QGtMR6Hs4PxMGYZGsGxf/6pW8oO8yitXMb79yjAcV3klu65Cc9xEre5V4Pyl8IDx60/3QQhCpaLwzjt6IIXwwOhi4T2aa7RUypsjc34ozXsPjiAc1lGuOHjH/b248/ZuSCJMTRdw/nwKzX447ekf9eHs6WnEohpMQ4NyXLiuqn6+leBHoJt17ug4v/p0H971C7su6flXstMNAGTh7nZycrLucW9AFvOWBxBBAq6yMTB9Aro0PEayFPOwc9i58ePYufnT886oZed4ZvYoxkafx/TUXlhWGro0oUnDK3z1E9pCC0NDBI6VxkT/PyA19DiSrXdxY/cHEG25k+Ynz3nlYEJUbYILwluhSDuFIu0AAD3UxGf7vgUhQ0umRogIrnLhuva8MM4KgmteApUIt7S34/mhEf+phLs7GtAaNqnsKowUHR4ouJi1gAoDZdd3dAwI9kNXAhCSoBPm6V+xHwqrOIxc2YLjKhwd9UKHuvBCUpogzBYsfPfVcd61rgGxkITlKAzPlPDC8Un0j+UQ0iUcV8HUJc5N5PD84XG+aXUDklGdTN2bPpjKVfh//6QPZ0YyCOsSYA27bmoH+WDk6f8ESf+lO8IDBlAoWvzVbxwCM1AqOfjI+2/C/ffMbRqamiLYvKm1+ryW1ihGhtMwDB1EDMOUGBpIYXQozV09DcT+my9ZWumDx8jZGX7usUPY/cFtiDdd+8qtoDx3YGBg0TknExMTdU98A9pbGkCCMtvJ7DDPFsZhSh2AWiJsVURbwybs3Pxp4qBE17t1YOhx6my/G53td6NQGOax0WcxNfZTlHLnIODCkGGQkACUnyvRII0GgB1kx55BfvxZRBObObHq3Yh2PAgt1F7DStTc1vbinGBBt7t3Ezc1b6PRyEtcLKcgZGiZQBZ7wouXsUNhAM3hMHXGojycLyFq6GgKeTLnISmwPm7Q+jhQcJinKozJEmOqzEiXGWXFHmCwL1/Cc7ImNA/sAUMj6EJASc9ZOg5XQUaTAq+cSWF//6yXVLdd5Es2iBmmIeE6c3LwtqPwtWf6EdYFoiHJUVOCGBifLSCVLSNialWp+O+92I+H7ljNLckwmYbEQl1DtUDSOigeIAL+/tFDmJotghRj952rcf8963zlXe+DBNfIq2LymiGDIValsg1T1+DaLv75q/vx3g9t4y3bO6tzUgLl3oXgkUsV+dlHD2LLHT3YdHs3XWsl3ypYFgo8OjrqFVTUHFctA6n3gdQZyI0DIPCkuodTp2G7FYQ0c85hXxDesbC++4Eq8Ih5OYU5Zx2NdtOGjZ9B7/qfx+z0qzw18mOkp/fCrsxAEzqkDPk7NM9RSz0BgkIlewrT6cPInv0yQk23c6TjnQi37ILQk5dXweWDCfvz1mOxVSgUp4DlVF2vZECMz1i643EM58uwFaPouBzXtXmSVlGNKKoR1kYBQCJnM0+XFCaKDiYKLlIlBwVHwXX8c+p3I9KCt6r9XmthQ8JxXVRsj6FETQ2uOxcGmgMjQtTUYLsuZrIVTLou2PXyLRFTg+N6Mvu6JvC9F/rx9CsDaIia3NMWw6Y1jdjQ3Yju9hgaE6Elx+D+8/eO8bG+KWiSsHVzKz72gW3VKqrgngykVIgIxaLN58/PQtclbNvBPff2YstN7ahUHBx5dRjff+wwTh2Z4Nvetho965qq8FrVnBIEq+Lgx/90EMnWKO5675bXpWw2eI/x8XFMT09D1/UL/jY9PV0HkBsQVN7iISzvJIykz1YHMS1+g7gwtCi6mrf7zxJL7vy9m1lBCB0tbXdRS9tdKJfGeXbsp5gdewblbB/YrUDTDZAwPBBj5c8CCYOdEgpjT6Mw/ixkZBXM5js41vEgwk07iYQ+x0pWxEjmzDAbarrJl3cGl3sqGUCDacKQAhXXRaZiI65r1YTyQqdPAOI6UVyXWJfwALlgK54tuZgoOJjM2x6gVFxULK+c13EZrqvg+lK9+oJzwH7PBwRQtjy13oX6iYIITORVchFB6ALQANcRcB3Xe+1a0AvpcFyFyVQJY1N5vHJ0DKaUSMR0dDRFeFVbHKva4uhojiBkasjkLOw9NIJTZ6YRNjXEojo+/ZGbIfz8Ci3CYKQk/OSZPmSzZehC4P0f2IZbdq6qPnL12iZUig6fOTGB4fMzWLOumXfs6saqNY2k+bmc2ck8v/j94yhkK/jQr94F4c+Dv9ZtI8GaGRoaQj6fRywWm+dYpJSYnZ1FpVKBaZp1WZN6COvGABBBAswK07kRSOFXuVwIDVDKQSyURDTcOhdNWhaZ5TxWEgp3UFfvx9HV+zHkZo9wauwnyE++CKc0Co0IQoar3eAgCaEnANLgWhnkRp5AduTHMOLrOdb5ABKd74QWaq1WcHH1PZe/ITU9fMV0dTkmF9QyVVzXlwshzFZsdMfCi0D2AjZRM30vqguK6gI9CR1AGAyg4iiUbMVlW6HiKJQdBddlnJ0q4dBQDpqYw1OHGRXLhVIKqxpDsGyFyXSpCvlEQKHsAEqBmfykvQtWCiFNXiAJT34joxQEoUsIQ3i/Z6BsOegfTuPMYBrEXmWY5ivv6pKQiOjI5SzsvqMH8ZhJAdNYGPKSkrBv/xDv3z8MAeD+d2zALTtXkeuq6vWolG3OZkoIRw1oUmDw7DSG+qfR3BLlhuYY2HEwOZSBVXLwnk/tRLwxQq/XEKpgnff391dFFGslLjRNQzqTRjab5dbW1jpy1AHkzQ8gwW68YOU4W55dfpgSEZRyQSRorqv8UvMRXigpmAnibvoc56b2IDv2NMozr8Kx0tCkAQQ5CnYB0iH1EBQkrOIwpk9/BbNDP0By1fu4sfvd0MymGvn25bvdpTSrWlZLgg3RinpAFjlSAMBUscgHJ71QhSTCeLGCHc1z7ISW4oA0L4o2D6AJQEgTCGmCsAADp/M2u4phCK9zXbkKMUPDutYIblubxPbuJKUKFv/Bd07CcVyvHNhSuHNzC+7c2AJdCpRtB+m8hf7RDPadmITlzgFNsezAsh0o15s9QiAYkqBJgkYCmhTQQtI7Z74Ao/Cvu4DHlExTw7FTk9i2qZXX9jRQyNQuaATct3+In3jyJAjAxo2t2P32deSxEg+MhCDs/dl5TE/mEY97O/hQ2ABYITNbwuxEAcp2EI2ZeMe/2IFV65vpjZhgePr06UXBxROBLCCVSqG1tbXOQN6k4ao6gCyI94MI2dIMynYBhpSLakYxGJrQUCjP4OTAD3lH74cvQzRxvvouAEg9QQ1d70JD17tg5c9zfvwnKIz9BG7uDAD2WYjwH08gYUBqBpRbwfS5R5EefQaR5ls53noXoo03QWrRmlyJuoCVSGlctIaTIFY8lbAWECbyOT6bnsVksQwmCSF0EAipioNz2RL3JsI0j3EsYCMLWcipWYsrjsK6pI64OT9pU7QVp4o2jo0VsP98FmFDoGK56EiYeHhrCzobTESMueeEdAlTF3BdF6WKi529TfjFh9ZfcCLu29GBlniI/+knpxELa8iXbNy6sQWbVzfAcRSm0yVMpUrIZMsolCxUKi7Klg3XUTA0DRFDeuUX5HWak/CcpKELpLIl/M3X9qMpGeamZAjNjRE0JEMgBoaG0zjXn0I4JFFxXey6o2c+SxYEx1E4c2oShumFA62KA9sf2WuYGpKNEXR0JbD9zh40tce9vMfrCB6Bgzlz5syiYVApJYrFImZmZq4sTFq3OgO5fhiI58Cy5Vk4yvYrsNwlsEZB18I4cPIfUCxN8k1r3o9EbNUCh32J6rvVSi4BI7aWmjZ8Do29n0V5+hXOjzyB0sx+KCsNaBFARgEKQlwaNN2EcsvIjL2I9MTL0EIdiDZt52TrLsQathD5IMCsqj0iFwtxeTtFCSn1xWNNizCHsmPj8PgQj+ZyYNKgSx2qZhKgJggHpwuoKPDauElhKS7KQo7PWPyT8wWAFfYKQtwg1gWBlYLrKswWbOTKDiq2C0MSWHnH9/6b29DTFKpWKAVhp3TRQrHieIOdBOHhnR1g9hoGg3yu6/ohKuGxh2LFwW2bWvF/feLWCz6uYkap7HC+ZCOVKWFwLIvXTkyi79wMTL9T3uu+VqhYNsAEIbyxulPTBUxN5SFJgOAxFl2TCJtatX+klp0EbKxcsrlSdqqNhb2bW9G7qRUNTRFEYyYiMaOaB3kjdveBXPv58+erY44XAozrupiamqp73DqA3CAMxN/y5isZHwBwUf0oKQwc7/8uzg0/g86mbby+5yF0d9xN5DMFuqTE9hwr8Zy8N2Aq3PZ2Cre9HXZhkPNjP0Z+/HlUCiOeNLseB5GAAgOkQeommCQcO4PZ0Z9idnwPzFgPN7S9Dc0db4OmR6t9kY5T8gFrqSQ6QwoNmjRoOfwIoKhsW3h54BRnLQuGZvqqv4szjIMzRZzM2Jw0dDSYGppNiQaDEJYenFZc5lRF4WzaxqmZCoQ/LyRnKeQqCpaj4NguoBgELyke1qWfTGfEQxpa43Md6DVCvJjKVGA5nqZWR2MYq5oiXid7DbEh6VVFHT03Cym9RPcHd68D4JX7CkFVsUdBhGhYp2hYR3tTBFvWNeNd96zDH315Lx/pm0Q8YqJSdtCYNHHLbd2IhnUUihYy2TLKJRvlsoNMpgxdCkiqlYj38jFnz85gw4aWC867IEK54uCeBzbgbff30lJ5iNcbPALAmp2d5ZHREei6viiAAHOlvHUGUg9h3QAA4odErNwl6A4yTCMOZhfDE3sxNrEHnS038/bNn0Vz403+7vdSSm3ncg8UDGT1by49upoaN/wSGtZ9GsWZ/Zwbew6F1BE4VgaQYUDT/ZkbgXx7GIokrOIkRvsfx/T4y2hsu4OTTTcBJDA5vh9C6EuwEc9patKAVtWzWh5Cjo2f51ylCFMPw2HGcnJbpiDYzBgvuRgp+2NewdDJ68qr2ApF2yu1lQSsSehoj0hEdUJLREPFUUiXHKSKDtJFG+NZC7MFCxIe0EQNCVMTi0bo0gWvMdJxGauavbnmgTpv7S5/Kl3i8+OemGRPWwzruhLEDOiaWDT6GZQOKOWNu5XS00ezbBedbTF84V/uQlMyfMEnqlgOTvZN8z9/8xCEJqGYUS47CJkaQmG9KpVSeyyaLogEsRnScfOubq+yWbEvREDV0blvhAUAMjw8jNmZ2UUBJLB6M+GNy0DegvNA/OoWu3RJVY5BA6GhRyHBmJw5jBf2/P+wfs37edOGT8A0GmpyJHyJYFIjmuiXA5M0EW3bTdG23bBL45yfegWZ8ZdQyp8HQ/ozQYJBSR6L0bQwbCuLscGnMT76IhgaGMJvIuRF31YpBVOPLJtED2qtilaZp/IpGFKvVlwtZB21h6z8/+qCPA0rAlwFlBx/2h8DhvBKays2Y01Sx47W+dpNPcn5o3n/6cAE90+VwGAkwpofXvIa+GodfbZk+wCp0NkUnk+jahzga2dmUCw7cF2FnRtbPRl5xUtIWfsH6Df/jUzk+OiZKYRMDbbt4rM/tx1NyTA5rqoClVd9pWAaGnTdS44H1Vj339eL7ds6EI+biMe94659X9PUYJgSggFd97TASF4fSejAcZw9e7Zapuu6i4eC63Im9RDWDcdAHBU4GFwykDAYuh6FYBdn+r+JifGXsG7NI9zV9UDNYKnahPslhLhqyoGrrCTcQY2rP4SGng+gMHOQ0+MvIJ86DtvKAzLsK/B6YOKxEh2eMIsESFt2sBSzQsiMz3OqS0FI0S7DVQqeKnfNhD9mOKygSEExAYIhiefNPq89POnjpfJBRvmSJj8dKmL/aJEbQwItYYmOqI7miERE9+BhPGdxquRACoLjeM8jWoRXETCbt/yudkJbMgTF7IHegnX/yolJSEnQpcTOja1zndTzD3HBrHNASuAnewZQsVw4cPHw7nXoXd1Irs9MAiDzZMwFJqby/OjjR7xKLUH41MdvwaaNS5e2BtcimQxjNJOG6yjWDUnXUtfqcgDk1KlTS4Y7gscEOZB6BdYNENZCvZHwkqhXUALL/ghYj2aoKtMwjCQsK42TJ/4aA/2PoqlpG3d03Ium1l0wfFZSDXGBLkl9F4uUA8dabqNYy22wShOcmdyPzPSrKBXG4LoVsAyBRNj7bDSv1mnZxRANN6zoExlSr3ECXuLZZUbcNLGlpRUMIGe7SFVcZCxG0QVsX4K9diY61QyYqgUAQxDKjsJA2sbZmQrADJMYIUmsFCNVtMHKC3cZmsDATAkvn03x1s4YIqZGuiQIIoyny9w/UYChCRQdF4WyA0EEsWD3vu/UFJ8bz8GQArokrGqN0jydq8UUd9lr/pvNlPjlQyMwdIl4WMcjD26A67Mq9tkZEarg8bdfO4Bi0YahC3z652/FxvUt86RNaJH3IgJaO+I4c2ISuWwZoYiOFYuVvU7x8ZMnT170sdMz03UAuXEQpA4gK1/MBEdZCEkdQgCVShoaEUwZ8jSuqvNAdOhGEq5bweT4S5gZewGRcAuaW27j1s4HkGy5jYQwasCAryDx7gWGjHA7ta55P1pXvw+F7FnOzBxGLtOPUmECWOHgHgZDkEAs3LiiXUfcjFAyFOXZcgm6pkHBA4eEaaIrFp13MIqBouNy0WVUXKCiCCWXkLUZs2VGusKo+KGs4EwQAZoAhEZQ/jwQ12FkK16DoC4ICnMNeQzgiSNTePbEDCIasal5CerxdAll2/WARpf4/r4RnBjKcEgXYD+ElC/ZOHZ+1pvTASBfsvGDlwf4gVu7EAnpRJ6wI/tVXWQa0kt8E2E2U+a/fPQgimUHuhQIhzSETY3kghLafNHig0fH8aNn+lApO2AAH37/Vmxc30K27ULT5EXZROeqJBxHYWw4g9aO+KLKusFM8tdTLiR4rzNnzlTDdItt0oQQSM2m5j2nbtcH+NdDWFdgQd/DUlVYBIKrKmiO9+Ch234TggQmU8cxOr4XUzMHUbFSMDUdUviNeuxAkISmx+bmgQw/ienhHyIaW8MtHfeiqesdCMd7aU7+5HIS72IBEAlEkxsomtzghQtGX+DB/u9B6lEsS7J8vSxdCyMabqSFYZpFg1hE2NG1HnuH+lCwbeh6CLoQGMvnMFtq5KZwyNNE9AEypkuK6Yu9lkTGYp4ouhgvuJgo2EiXgIrDIH/GebDZEX6iWPlsZ2G1V1gXcBxGquLAcV2vNJe8UmKlGIKAkuVi3+kZuL7ulaeGwjAkQfry67om8K3nz+JHrwwgYmislIJlO3AdBhFxPCQRjxoQTBibziGbqyAaMgBmTKdK+C9feolXdya9jUbZQbnsYCZVQDpdRtiQMHQJq+KiWLJhOwq6X3671KTBYI20dyYQjRk4d3oKN+/qvuAaBdP/AiXc1wNIguubTqd5cHAQhmEsyeillMhmsyiXy9Vxt68HE3Fdt/pe9UmIlwcg9SqsJV0hYEhz2dwAEcF2K1jTfgcSUU8JNRZpR++qB1EoTvDg6As4P/A0rMoQXOV480CEVg031c4DqRSHMXL6y5g8/yiSzbdw86r3It6+m4Qwa1gFLiG8NRdCCp4fOPjWrnspk+rjTPoshBZdttJMKRexWAN0LbR4zOZC5opEKEK71271+kDyOWjSBDPw8ugYtrW2cHcsSlqNA1Nc8/yaOegNBlGDoWFzgwaXTcyUXB7N2RjK2hjP2chXvI5uwewtiEU+miDy8ieCIKSAJgAlGY7jzuvskUSIhjSwEmBXwVWAcpUnZVJzgiIhDZajUC6XvVyXz1aUArI5BUzmAQUYOiFsal5VF7z3H5sqYHgsD0FeubEkAU0nRCN6kNGCaWp48uk+vHpwhLff1I633bkasZhJtSGvGnwHGIgnQtS5KslD52aRy5Y5nghV8yDBPPJ0Os3ZbBarV6+mgA1cSxAJHPPg4CCmpqaWrMAKwCyXy6FQKHAoFHpdYljBeVn4ed9Kdq2ro97yDCSkR1Z8IZgZCi6EvzeORtrppg3/AhvWfBCpzFEeG/spZqcOoFIagwDDkCEfTLxciRAGNBkCsYPs5M9QmHwRkXgvN656HxJd75rTuAI8KZMVj7ydYyXefA6vryXRdBPSqb6LAAJBsYNkrMM/zpU1RTKAsG7grp71dHZ2mo9NewlShxX2TUzjxGyeWyJhdETCaI/MjbHlBUBUm6SWBLRFJLVFJG5tD6FgKR7L2xjOWBhOVzCZtbyJgQsqrUqWW3X0CABBAbXzosiXh1euJx3P/rwRteAGE4JgOwpKeUOubNuF67hQisEKEPCaDqOmBinnSoK9L68xkEyAwPBqDAhCeCBYrni7YQmClISxsRzGx3I4dGgM73hgPe/0xRODQVRVxVP/PdZtaEH/qSmcOTGBnXetASsFBc9J/u7v/i7/5V/+JcrlMm6+5Rb+n3/0R9i5cyddSxAJXruvr++iFVhBN3oun0dzc/M1d+bBZ/v+97/Pe/bswdve9jY88sgj9Eb1y9QB5IYDEG8BhY24H3pa3s0WKymvcoZF1cGyXyGl6wbaWm6jtpbbYNs5npl+DVPjLyIzexB2aRIshAcmgbw6AOmHuKzCECZO/DHS5/4R8fZ7Od71HoSabp0bLlVNnF9iiKs2X3LR/IdEY6LrouxjMSbCANY3tVBjOMqvjI6gogBTSpRcF+eyJZzL2YjoOnfHQlgXN5D0K6mCFPAFWliYSxxHDUEbmkxsaDLBHMdI1uJnT6cxkq5A93tLDI3w9g3NaI7q0IRX6lCsuDg5msPB86nqe1guw7EZuvSaCBUTSpY7r62SGSiWbIQNiUTCRFsyjK6WKBJRvz/DD3llCxXsOTSKXKECXXr9J6WKA/bFD01dQ8ggv6yYUCzZSEZN7NzWie6uBKQgTE8XMDiUxvRUAVNTBXzjm4dx6tQkP/zwZjQ3Rxa9COs2tSL0zBmcOjKOW+5YDfbB4z/+x//Iv/d7v4dQKAQpJZ5/7jk89PDD+Onzz/O2bdvoWjORY8eOXeCUg4704IuIUKlUkM1kXpewlZQSv/mbv8l/+Id/WP39r/3ar/Ff/MVfkFcRVw9nXRUAuY6bQl8XBhIzk/5sD15ypy1IolCavsDBUrVCak55V9fj1NF5Hzo674NVmeWZyT2YHv0JCqkjsO2CN1xKGgBcL/EuTEgZAjsFZAe+ieLwd2EkbuJIxzsQaXs79Nhamp84X3mIq1y8eOOWUi7CoUbEI610OTsz8hdRUzhMuzq7+KWRMYDnFpYAkLUcHJot4XTORXfU4K1JDUmdaPGWxgXS7zwHNt1Jgz5yczN/+ZUJlCwHtqvwvptbcfOq2AUf+pY1SVi2y4fOpyGJ0ZYI4YN3rEJz3PS7zYHJdAl/+2QfimXbu86C8Avv2YIdvU1IxkwKGUs7mVzB4uf2DUIPEWyXsXVDC1a1xqBJwqn+WYxP5RDWJSzLxS1bO/CR996Elqb5wKAUY3AozcePT+DkyUkcOz6OocE0du3q5m3bOtDUHCVNE1UZ+KnxHDRNYHa6gInRDHd2N9C+ffv5i1/8IhLxBBjejOrGxkbMzszg87/+eTz7zLPXbLcdvO7x48fn7Xa92SZF6LpeHS4lhECxWEQ6nb6mO+MAPP7pn/6J//AP/xDJZLL6Of/qr/4Kt956K/+rf/WvKHhc3eoM5DL5h19RFGqELo1F5zQE21IhNORKk3BdyxMlXC4PUQMmhtlEnT3vQ2fP+5DPnOKZkaeRmXgRdnEYkhhCC/nz05Un424kQaxgZU6gkj6GVP/XYDbs4GjHA4i0vg2a2bysYOLcjS3AykEufRpCLF3ySURQro3GZA+EkJek6bUwB8HMaA1HKKxpnLFdtEYi6E0mENIkig5jqOBguMQ4m3MxWCTc2SR4XUwsO3udvAjQvOkrUUNSY0TjTMlGxJBY1+z1djDP5Q4UK0gSaE+GwAxYrsKDO9pxU09y3lu1JEyEDMnFso1yxcUnHlyPh3Z1Uy0j4ZqEfSD3n8pV+ODJSZiGBDPwhU/fjp03zU2R/NOv7OexiSws28W6nkb86qdvp4VOM+gDWbumkdauaYQQxK/sOQ/lKrzwfD/27RlAY0OYI2EDkgDbcpGZLUI3JKyyg+mJHDq7G/CHf/AHcBwHJAiu44WPLMtCMpnECz99AU8++SQ/8sgj18RhSinhui76+vqqjEMIgXK5jLvvuRuDA4OYmJiAruvV5H4qlbqm4Zrg/X/v936vmpNxHAdCCJimiS9+8Yv45Cc/yYlEgt7qqsBXg5XyWzaE5a+buNmAkB6FZWchF2EiDIYUOnLFSaTzw9yUWEdzHeZYAZh4TjmW3Eyx5Ga4G/8lZ6b2IDP6NEqpg3DsNCTJqtw6AJAWhiAdLoDi7H4Upg9AhtsRbt7F8Y4HEGm6ef5wqZpekeD9sulTXCqOQ+pxjw3QUjecRGtj7yWFrxZjaUSEc+k0pysVrGtoxB3tLVR7c66Lmzids/nArAsG8NKUiwaDuNGgi4II4KnwFi0XpyZLGM9akAREDIGwIUlc0EXogUmmZFXnh4d06QNNAHrAmdEcT2fLkAREwxru3NJaLQ0mcaFEiFIeS3n8mdNI5SoQYHz6Qzuq4JHNV/h/PXYIx/qmEY9ogKswOVPA//2lF9mQojpZUQpP2FHAF10s2SiXbZim5lWtxUywUshmSsjMlryOd0HQpYBtK1gVG61tDZianuAfPfUUwuHwBbmHAKy+8pWv4JFHHrlmOYbx8XEeGhqaNyjKdV389V/9NX7/938fX/va12CaZvU8XktF3iA09fTTT/Px48eRSCSq50UphVAohMHBQXz3u9/FZz7zGTiOA03TULc6A7lsBhIyopQIN/KElYJXJ4NFd/S2W0b/6ItoTvbCVQ7kimeU18q4M6Qeo6auh9DU9RCs4gjnp/agMPE8KqnDcO00hBYFSPebAKX3f+hQThHZsZ8gM/ECjNg6jre9HYmO3TDCHfOGSwU2OfL88p+PCK5yEIt2IB5tv6zwVW0u49TMFB+ZmkZbJIZdPnioBb0KG+M6lV3iwxmvKutkxsXdrdoSu0nvuZMFh5/pzyJdtFG2FMqW4/WBMBDRJaQ/EnYx2JnJWZCCYDmMQsWZ96FJEI4PpuE4DKERYiEdyahBQSc/+30poGComNc42DeY4mf3D0IThG3rW/Du3esIAPrOz/Lff+swxicLiEUMz5n5AogjYxVowutNYcVQjhdqIuXtmHXhzRgheJViwm9b1aQ3/13TJHTNG1YVjRrYfO9adPTE6atffZzT6RSSySQcx7ngxjZNEy+//DJSqRQ3NjZe1R13AABnzpxBKpVCNBoFM6NSqaCjowNbtmyhrq4uXriuXg9F3m9961tYbD0Hx//444/jM5/5TL0fpQ4gV5oAUhAk0BztxGiqD6SZi6ZCmBUMPYqTA09ibcdd3Nq4ecFMkIuDyaIy7pFV1LTmo2ha81FUcmc4P/wDFEefhFuegtCT3shbZjA8GXehh8AkYRVGMHn265gZ+iGizbdyU897EEmsr94uk8PPcD59BlKP+Z9RLAqgrGy0t2ypmZ0uLgs8jk+N87GpSYT1MHZ1tHshLczXpQoevy2p0UDR5VmXMVFScHnxMe3Ba4/mbJydrSChUbXfw1VeWElKWjTsSAAqtsJktgJdEiwAwzNF3E0t8y7Ta2dmoOsCgoBMwcKJgTTv6G0iuciLBiq9X/7OES/Wrhir2mIYGM3wT/cP4/l9g4BiRCN6te8EgJd8lwTHdqFcRiSkobktgraWKFqbo2hIhiCIYFsOKhUHrqMghUA4rCEeMxGLm4hGDeiGhK5LRHyQA4BnnnlmWQdvGAbGxsZw+PBh3H///biayeMAQI4cPVJlI0op2LaNNWvWgIjQ0NBwwfOuFYAEJbvFYpFffPFF6Lp+AStTSkHXdezduxfpdJobGhqoPtzq6qyDtySABGjRkVyLw0PPYjkFWoKAYgs/3vd7uG3TJ3ndqgdg6NH5EiUrqpSqrY6aC3GZ8Q1k3vQbSK79BOcGHkV+9Ck4ldS8eSCoGS6laQYUu0hP/AzZ2cOINt3KRrgNldIM0qkTEDK07DxspRyEQkm0Nm2k+QB36ZatlMAAbm5vR9wwlr0pBQG9McJMGcjbjIyluMm8MBcSOOAtLSYdHNM5XXAgfBFGZk+YcSJroW+iyG1xPciYwFVeFdbLp2dQKDvQhUDIkDhwdhYdyRBv6UnCdRWeOzyG8VQJhhS+o2F86TtHsb4zwa0NITTFTTTFQ2hOmmhKhOC6jEef7kPfYBqJiA6WwHMHhvDcvkHYtkIspFcnCJIfxiQwCkUb4ZDE1o2t2L6lDb1rmtDSHKFAJ+vydn0M13Vw4MABSCmX3AUGTv3gwYO4//77r8nNfvjQ4Xnvx8xYv349AMwDkOC9A0HFq+20A3A8fvw4BgYGqiG1hc7ONE2MjY3h0KFDVx1U32xW70S/SmGsroZeaNLwZnIsAzZSaHDdCl45+iWc6v82VrXdzqva70Rr0zaS0qzJQaxUnmShNAlDC3dQ45YvILHuU5wb+RFyYz9GJT8EJs2fByL9UlcXgNfxrkDITh2AAoGEedHGQSIB17XQ1XITNM287OR5cIS3dvTQpmaLm8IR4iUWZlCeCwJ6woQjEihYwFhRockUF0hzBHqHIU1gTdLEVNZGaEFfh+0yHt0/AU2g2gfiOi4s24XtKBiagOt6oShXMR59cRAhXcBxXS/n4M8UYT8voRTjSP+M1/uhFKAAQYyQLuE4CoWSjXhEh+0qSHjMRQiCGTEANVeuKgXBshxIAu7Z1Y137F6LVR0JWujMVuLP55c5k99PJHDy5Gk+e/YsQqHQvJs4SFbX2tFjR6/6vRM43WPHjmFh0+KmTZsAAI2NjRfsUq+VoGLwHvv374dt24hEIheE9WpBdd/+/dcMVG80hrDca7ylGUiwiNsSqykeauJyJQ0hlinpZa9nwtDjKJZn0Hfuu+gf+B4aYqt4ddcDWN3zMMJ+M+CljbzFPGkSsII0W6ih99NIrv0YClOvcG7sJyikjsGxMiAZBmQURNJPkBOkFvHKkUmDe5EJJ4pdGHoMnW03X5WbOaRpCGnakoOoqj0fNOf8CZ7e1cm0i01JCV3QvFoxIk+LeDBj8emZMgxtTvuqlqXoGsFxFZTr5S2YPfkSoQu4rpoHOJGQBsevVAobGhy/mVD4EinMDFOX0GXQF8RwHQXbdry5IyEduUIFjXEv1Om6gTCk14lO8AAlV7CwpiuOn//ANmzubaZawAgS8/PEGi/BXNdz1Hv27EGpVLog/6GUmmtA9H8+3Xd6ntO/Gg6FiDA9Pc39/f3V3X7gTDZv3gwAaG5ungdoQghMT097vUdXOf8QHPPevXtX5AwP7N9/TYDsrRZieot3ons7OlMLo6uhF6fG9sCQkYtMJvRuFCl0aNKAIIVCcQzHTv09Bga+i55V7+A1q9+PSHTVvPCWxzZWNj8dQTUYK5AwEGu/l2Lt98IqDHFu8mXkpg+gXBiG65R8MAn78vIE+CXBy7Eu16mgq+tt0PUIXS77WAoklvp93lY8VHSRsoHpCsFWgEZA1lb4yYjFd7frSPrt6hWXMZ53+MRUGadnymC/+1wRLpT5AEESgXwAYia4wAW7eyJvqJTyy7VVtWcBKFVcKOUiamheqFK5sGyFfNH2wMivyCoUbDx05xrs2tqOP//ng/MyS4GjzBUs3H/nanzyka0UMrV5XeVXw1cFDu+FF15Y9GY2DAO2bVfDSbqu4/z588jlchyPx69KzD8I+/T19WFqagqRSMRTafAT9xs3bgQAtLS0VPWxAi2q2dlZFAtFjkajVzX/EITyDh06VGUZy+VKDh8+DNu2q6W+b8U8yNUA8bqcu+9217fegpOjP6uGtVb0TH8miCYM6DIExymj/+yjGB36ITo67uFV3e9CsnEb1XaEX0quZA5IPDdsRHuoeV0Pmtd9DOVsP+dnjyA9uQeV0jQgV3K6yMt9hBvR2bHTL0e+OjfOcuBxLlfhV2fKyLsSLiR0qUMnwPVzGaN5hW/lS4hrYMGMguUiU3bgulwdvMQBw/Cro5gB22FYtgIrNU/KRMCT/lA14FG2FAwhYRoaciXLYwz+7zd2J/C+O3vQ1RyBLoU318RRmEqX8Mf/fBCFkgXbdrGqPYbbbmrD1394Ao6rYOoE+Bpctu1CAPjMh7fjnXevrUqSLCaQeCW7SSklLMvCvn37qk5T0zRkMhn8+3//7/Gud70L73jHOxAOe8OzdF3H+Pg4zp07h5tvvvmqOMtgV3vo0CG4rgshvDySbdtoaWnBmjVrAABNTU2IRCIol8uQUkLTNKTT6WrV1tXcBQshMDIywufOnZuX/6g91gDkQqEQzg8MoL+/nzdv3lxPpF/HDOe6B5Bg972+fSdiZiNct+xNzbtIGIhIVPMW3pwQF4IkdCMJZhsjg09icuRpxONruKn5VrS034tE0/bL6CqfP6WQ4c0DCSXWUyixHi1rPogz+3+HS8UJkBa+6O7VcS2s697tix9eHfaxHCMZyJd5z2QBChIdYYnmkIaiSxjMc3XIlBRASBAkMbIVhbylYEqCxYzephDWJHUYfpmrIQmmLx+SrzgYSVUwnqmgYjsAA5btYipTxkzOgiG9U1exFdZ3xPEv7lmNiCHx598/iZHpAsBASyKE/9dHtpOpX8jazgxnuFTxZohIIZDNW/if//gqiBhh3Su11qRAoWijpTGMX/vYrdiyvpmCSYZXEzxqQ0enTp3is2fPIuyr2gaNg7/+67+Orq4u6u3t5XPnziEcDkMIgUKhgBMnTuDmm2++KgKLgbM98Oqr835nWRbWrFmDpqYmCnIgiUQChUIBmqZBSolcLofJyUl0d3dftZ1/4MSOHz+OdDqNeDwO13WrPSnBMUspqyCczWZx8OBBbN68+ZqLTt7IVp8H4oexYmYDrW3dzieHX4RuROcc/BKsxXZKMIQECc0fLsWYK9El6EYCgl0UcudRTJ/ExPlvIp7YwC2dD6Cx8wGY4c4FFVxyJXdudXY6KxckNOTTJ9iuzHqijcsJepGA45TR0LARrS3bKJCAv5bgMVu2ee9EDpIkbmsOY2PCpMCnHtbAr84oEANJg/C+NSEyBMFyGd8/m+eBdAU3t5p457r4kh6mNaZjXfOFoFm0XH7m6CR+emIKIU3AshXu3tyCjgZPBbYtGeKBiTwkEcKmrI6ddVyFdN7igfEsXjw8hj1HxqFpBPLDXpbtwtCFPw/EO9BMtoJbNrfh1z5xK5qTYXKV1/R3rW5WIQRefOkllMtlJJNJMDMKhQLuve8+dHV1ETNj8+bNOH36dDXXAgCHDx/GJz7xiasW+mBmHDl8uBouCn63ZcsWP1fjIhaLUWNTY7XRUAgB27YxOjqK22677artXoPXOXz4cBWUAkDr7OzEl7/8ZXzuc5/DyMgIDMOonpM9e/ZctXNyI4ewlp9k+hZnILUOb2vXPTg58uJFHy2Fhsb4WpRLkyiXJyFJ+cq7hifO5w+YAhhCmtBkCIIdFDJ9KKaOYuLs19DQehc3d78XsZY75rOSizr1oFNaQ2r0GR47/TUwNJA0l+VMzApSGli79qHX5ZxarsKeiRRcBt7eEUN31PDCOn4ieUejpIG8y1MlBZe9PIZij2G0hCX6Z4FNzSEw/A5wWjzPwtVJ7YF2FiNiSHrktk6cnyrw4FQRpi7x4okpbOiMc7ZooW80C0PzlItHZ4r43a+9ysmIjnSugtlMGdlCBa6rEA55VVqqZpcdTE8slG0YmsBHH96Ej75rCwVVXNcKPGp3/s89++w8RsLMeOD++6uP2bx5M77//e/PS2AfOXrkqsS9A7AYHR3ls2fPXlAuu2PHjiqAGIaB9rb2ajI/+PyDg4NX1flUQfLI4Xk5EcuycM899+DBBx+kRx55hP/0T/8U4XC4+nn27dtXfWzdLupy6gCyJBL7TntNy3ZqinZxoTQJXWoXZGKJJGy7iM622/GOO/8TlSopnpo5jInJvZidPYxKcQKG1KDLEAjKk2T3w04AQ2hhaIiAlY3Z0R8iN/YUYo3buLHnw0h0vcsDEj/hjsVKgasjagkTfX/PM0NPgvQEQHLZng8iAdcpYe36DyIcbqbXI3R1eCbNM2ULuzub0R01yFOm9YHAB5GNCYmpkkK6onAyZfO2Jp087SqGFN5o27lhUksdnj9ACZ6cem2mOqRLKMUwdcLAZB5/8PgJWJaDkuVCCsB1PWXe0ZkiTg1WvIFPQiDiz/hwHVXNoUhBcBSjWLIhBGHnpjZ87F2bsaGnoVplJa4heAShl1wux3v27IFhGFVHKITAPffcU33s1q1b5z1P0zT0nZqTXL+S0FHgMI4dO4bZ2dlquCj4LDfffPO8x3d1dV2Qizh37txVPTdBaKrvVN8FJcW7du2CUgrvete78Gd/9mfVajHTNHHy5EmMjo5yV1cX1cNY9RzIle2sWEGXJjZ23IG9Z74FUxoA1AUnSwgNhfIUXGUhbDbS6q77sbrrflSsNI+Pv4TBge8jn+mDJjSYWshnJG4VADwHK6DpCRAUiuljKM28iuzAN7j1pi8g1HjLPHFzZjUPwMAKo8f/hNNjz0GaLVC8fL4mAL3m1pvR3rHrdQGP6VKZ+1I5bGpIoDcRJub5XemBL1kbE3RwGlxWhFcnLXRHJSdNQY4XDYRa5n143jx17wUtR2EyW+HRVBmnx3Lon/TmoSulYOgCxYoDsCcRohwvLOW6QDJq4DMPbYAA8LWn+rykfQAI7M1nL5UdhAwNu7a24z33rMOtm9vmJcqvdRI2qHzau3cvhoeHEYvFoJRCpVLBqlWrcOutt1Yfu3nz5mqCPXCWw8PDGBgY4E2bNtHVAJD9+/fPCxc5joOmpqZqCCuw1atXX/Dc/nP9F4DKlTgwIsLs7CwPDQ1Vq76qJcVbNkMIgdtvvx2tra3IZrPQNA2GYWBmZgYHDhxAV1fXDd8Pstg1rFdhXdVciL97W7UbB88/AQW1iACIgiZNpLMDmJg5yp0tO4nZAZGEaTTQmtXvR0/3uzA++hwPD34f+fRxCHZgaGE/x+HOy5V4Ia4IpIygnDmO0b3/Gsnu93O042EYiQ0QenJeBZdTmeGJE19CdvoANKMJSrnLluyCCMqtIBRpwdr1j+BqVl0tZ8dm0ghrEre2+Oq3iwo5AiFJ2JSUeHXKhmLgqcES3tkT4pKtAAKcmr6PC0HD+32h4vLZqSL6xgsYmikinbdQtryEui7nQIjZ05myXFXt2/A2DgxD0zA0WcDJgVS1l6NSUShXHACMzqYIbt/cjrff2oV1Xcm5lBeuLetY7Kb/4Q9/OI95WJaF22+/HclkkgJxwN7eXjQ2NqJQKEBKWU0aHzt2DJs2bboiZxk4nSD8UzvrY/v27ejs7KTacbpr165dsAETGDg/cMG0wCsFkKGhIczMzMAwjGoILRKJYH2v1xXf2dlJ27dv52effXZeHuSnL/wUH/jAB95SAHI1WUadgdSEeRiM1sQaWtW4mUdmjiKkmRc2FIDAYPQP/QRdrbcBkP6O3h9hK3R0dT9MXd0PY2ZqH08MPYn09F7YVhqa9CcSzrlEn5UwpBYBsYvcwDeQH/wOZLgDWnwjG4kNEEYLnMo0shM/g11JQdOTUH4n+nKQGAgyrt/0MWha5JqWKwbsI1Wu8Gi+iDs6WmHKCzvMa1kIA9jWqFFf2uGSzUhXFB7vy3vqs+TN2agF+OB1SrbC+Zkyn5oooH+qiHTB9spZ/RnoUb//wlmohcSMhqgBZoV0tlJtOswULPzk1WGAAcf2cletyRC2re3Eri1t2LquiQxdVm8YT+drpWoDVy9MY9s2nnrqqSq7CJz0vffeO+9mbmtro+7ubj58+DCi0Wj1mr/66qv4uZ/7ucu+6QMAKJVKOHLkCHRdrzIjpRRuvfXWKhsJ3nPdunUgP9Ee9KWMjI5iZmaGW1parnhNBscyODgIy7KqnfmWZaG7uxs9PT1VkHnggQfwzDPPzAtzvfjCi1cNzN5sdjUr4N7yAOKdDK9Edmv3fRicOghCaNHH6FoEo5P7kC+McSzaWQ0L1SrvEgk0t95Bza13oFQY5unRZ5AaexaVfD8EO9C0UHV2upcr8fbKQk8CYCgrjeLUKyhMvQyGBhYaoMUh9KiXoCdtWfAgApRbQe+WzyISW3VNQ1fVLT4RzmUyiBsaepMxX9ZkedQxJeH2Vh3PDJVhCsBRcwDTN1PBmqQOECFXcnk0a6F/uoSB2TJSRctrMAQhYkgwU1XKJAhB1YKV5Sh8YNcqbFvdgB+/NoqXUhNeeIsJtt/d3ZYMYfvaRty8vgkbuxsobGo1VL2mIfB1XpfBLI8DBw7w8ePHEYlEvNCBn9/YvXt3lR0Ej920aRMOHjw4L5F+4MCBKwpdBI749OnT8yTcA7vrrrsucE6rV69GPBarzuTQdR0z09MYHBxES0sLrhaADAwMVN83qPZau3YtotFolZm94x3vwH/+z/+5CmahUAjHjh1Df38/r1+//i2XB6kDyFU2Ue0JuQMN0XaUy2kYF0ibeDM0rEoGpwd+gJ1bf3lRNhMACQCEo93Us/Gz6F7/SWRnXuXU6NPIT78CpzwJXRqADM2VDQf5EtIhdN1LkEMDk4SC8F9TXiQcR7DsArrXfxSNLTuuPXj4i9FRCkO5PDY1NVaHTC2HIOQn1DcmNTqf0fhM2kKICLZiGIJwJlXB1BGbmRnZoo1ixa3KpId1b76G63izyyu2C8dWkPC61p1Ajt3HNl0KvHRiGs8cHkc6V0HIkKj48vDrO+J48NZO7NzQMm8KYe1skNcrVLXcTfqNb3yj6oiZGeVKBT09Pdi+ffs8AAG8aqhHH3202gthGAaOHz9+RR3pgYN99dVXUalUEAqF4DgOHMeBaZq48847q58jeO3Ozk5qb2/n8+fPwzTNKpM6efIkbrvttqvWg9HX13eBYwyKCYLzt3PnTqxbt64qtqjrOjKZDJ57/nn09vbeUP0gtXI2b9UQ1htwJQmKFUw9Qhs774bllhYXBmQFXYvi/PAzKJVnmIgWrZX2WImolvWS0JFsvYvW3vIfaPPuv0P3jt9CKLkFys5jsQJVsAKzW/26WHOjV70lYNs5dK15D9pX3fe6gEewiKZLRS47Dpr9LugVaXf4oax7u0xqCgmUXa7Ku2tESJUczBQdKPak3COGhCYIrgJKloui5UISsLY5jHfc1ILP7O7Bb7xnA3ZvakbZdquOnwhI5S0USg7ChkSh7CAW1vBL796I3/rkLXT31nYKGV7VlgrCVIGMyRscf9Y0DcVikb/9nW9XZcqDXfbOnTsRjUZpYansLbfcMhdy86XdR0ZGquNnr8SxvLxnz7ycSKVSQW9vLzZv3lydKxMwn1AohDVr1lTlVQI7cuTI1XES/mueOnXqAocWnIMARCORCO3evRu2bVcrtwDgh08+WWUubyW70Y/3DTm6wFls63kQphaBWqShMKjGKldmcfrcd30vyMttB+aFt5gV9FArNa3+MK296y+oad0noZzilTn6oBnOzqNr7SPoXPPe1wU8ai1bqXjlr75zWnhKeBEIDM53SCO8d20ETSGBgu3DMQG6JBh+57mjgKLlomi70CRhQ2sEj+xowa/c24NfensPPbSthTZ1xsjQCMOzJWj+Tr1KaSVB1wSyRRs3r2vEb33iZuze1u4NkVI1oEFvLGgsDF8BwBNPPomzZ84iHA7PC/s8/PDD8wAh+P327duRSCSq+QgpJRzHwSuvvHLZO0dN0+C6Lvbt3TuvgdBxHNx5150wDKPaAV772QNxxdpw2uHDh6/YiQU5mXw+z6dPn67mZJTrQtO0aklx7fu+5z3vmceoTNPEiy++iJmZGRYL1kvd6gzkMvywFyZqinXT6tZbYDmlRZ0wswtdi+Lc8I9QKk9zkIRfyet7z3f88IgGLdTq94rQZX5mCVYOlFtB94ZPoOMNAI9q8IyA89m8HxKcDxpLpZ2DQuSEIejDG2N0a7sJTRAqDqNkM0q2C9tlhHTC5rYIPritGf9q9yr8/K522rU2Qc0xnRzFGEmV+cdHp/hPfngW/ZMF6JqYB2JEhFzJxjtv7cT/+YGbqCFmUBCmeqOZxsV2iX/9V381jw3bto1EIlF1iMHjgu+rV6+mdevWoVKpzHOggQjj5YSvAODs2bPc19dXbcgL7B3veOeSDmW731wYvI6u6zhx4gQKhcIVOe3geadPn8bo6ChM0xupYNk2Ojo6qsAVyJgAwIMPPojW1lZUKhVv4xIKYXx8HD/+8Y/BzBcMobqh7Sos+HoOZBm7ec27MTD+ypJnWggd5UoaA8NPY8uGT/qNfkvlJ7iapPccu4BVHOGZc19HZug70LWYH6a6dPBw7CKE0YDVW34RieabX3/wCGTxIxGYUmI0X8RrUym+qSlJoZrBSTlbcc5hdIYvnPkXgIgpCff1RGhXR4gnCg4KFReCgLgp0RHTSffjW/mKy6cnizyetTCRKWMiU8HwbAmFoo2YKbwZHjXOQAhCNm/jXTu78HN39xD7ApVvZG5jJexDSomf/exn/OyzzyIajVZ/l8vl8P73vx9r1669IPkbPGbnzp1VdVrXdaHrOvbv3498Ps+xWOyS8iBBiOyFF15AoVCoyshbloVEIoH777vvAkYRvPaO7dvnDb4yTRMjIyPo6+vDzp07LzuRHhz33r17YVlWlZ1VKhXcdNNNaGhomHdulFJob2+nt7/97fz4448jkUjMyy994hOfqCfRbyDT3rgT67GJ7pYd1N64iafTpxHWwtVEt1eJ48mzK2Ujlx9eFjQCKfcgjJWfPcypkR8gN/482E5D12IALhE8fEVfx8og0nQzVm35JZjh9jeEeQTOvyEUog0NjXx8NoPT6RxGCjY3hkyABFIWUHQFdjSa6AwvLktSm/SO6ILWNRj/f/auOz6O6lp/d8r2XfVuS7LlgnvBxgYbDJgeqgMEQoAXIARISIE8CIGElkeAUAKBhOLQQ4AECMWATTFgg3vFNjZusmT1ru27M3PeH7t3PLvalVbNlmEvv0GytJpy597znfqdmN97giqtrfZiV5MfTZ4QfAEFYSUSK9I0DdNKXTiiyIENle3Yvr8Tkhi5s0iPjjBmj83FeUcPZxovgDtMNsI999wTw3rLNb8rr7wyRpDGa4Vz587Fc889p//MYrGgqroK69at63U3Ph4Y/+CDD/TzRVvIYvbs2SgrK4up/zCCyRFHHIH8/Hy0trZCluUYRuFp06b1OXit13J8/rl+T/xnPCMsHkAEQcCCBQvw5ptv6rERq9WKjz/+GDU1NVRSUpKuSk+7sAZk1sCYgKOPuAQgQlgN6PUeihpCKOxBINiGDGc5KsrPxgGRSNGgNweOiMUR9NdTfeXr9PWKn9E3q36J5qq3QRSGFE3b7Z2bSISm+ECagvyR52PE1JvZoQIPo/BXNQ3jsjLZhOxMmEUBnrCCSncQ+70huGQBJxdbMS5D6rGCgteIEEW4s7QovclbW1uxZEc79rQE4AupsMgi7CYRVlmALDJkWGW0ecNo6gxCFCNxKYExBEIqSvPs+MGxZSxSjDj0wUNRFIiiiHcXLaJFixbB4XDoIOLz+TBlyhScccYZjAfZE7m9jjnmGFitVr3hlCAI0FQNH374Ya82PweGlpYWWrZsGcxms26REBHOPPNM3fKJF/BEhJycHHbEEUcgFArFCOYvvviiz5owBzCv10srVqzQ4x8cRObOndvl3BwszzjjDBQVFSEQCAAATCYT2tra8Nprr8W4677t1ofAvt0gKR3aCY7EQopzJrKTj/wNrf76eQQCLTCJIlyOYchxjUBx7lSUFM6GLNmYsc8HtzSCgRZqb16HloZl8LRughpsgSzIkEQLRMkMRLOrWMr3JIJIhaL4Yc2ahIJRl8GaMZpzvR8y8OCbThQEQAAm5mWzEZlOagsqEAURGSYJNiniKyKk2uw38j/+WZPIcN7EHHQEFATCGr6q9eCrGg8kFgEYUWD4/JtWKGEVJhF6GrEWpS754XHl0boP0tl3h+rgGnBnZyf9+te/0qurOQgoioKbbr4ZJpMJvMYhHkCICGPGjGHjx4+njRs2wGa36+f98MMPcdddd6VsfXCX2AcffICGhgbdfRUOh2G323UASaS1q9GA9qxZs7B06VK9iE+SJKxevbrP/Fz8WVauXIl9+/bFULsUFBRgxowZCV1qqqoiOzubnXX22fTUk0/qhYeiKOLFF1/E9ddf/62yPga71iMdA+kRRAgji45mpfnT0eGpIVmywGHNZ4IgdRHuAODz1VFry0a0NK5CZ9tWKIEmSIzBJJohmzIhRAsHOZVJysABQA21Q7QVo3DU+cgadhpDFORSa1A1uCZupE/3dtq5cyeGDx+OqVOnMrssG5x5sW6qPhiEsJkEZjOZ4AuptF6lCMli9HeKSjBLAqwig6apUNQIULiDCs6eUYTibCs7HMCDZzEJgoCfXP0T7Nq5SxfYPPZx7LHH4gcXXsi4IE52Hl48t27dOh14rFYrNm/ejK1bt9LEiRNTctdw99VLL72kWxX8Xk488USMGTMm6Xm4AJs3bx7uu+++mKZOu3fvxldbvqIZR85gvXGnGc/93//+N4baJRgMYubMmcjJyen22a668ko8+8wzUFUVRAS73Y4NGzZg0aJFdM4557BEwJzKPuBWGL+ukZdLd38bjsF2O7Fu67C+3TGQIaEGRDZMhAMrJ2Mkc9mLY8BD0xS0deykb3a/RitW/Za+WP5zfLXpATTUL4eq+CCbMiDJDkSEvdptn5GuF49sKDXUAcYEZJV/H2UzH0TW8DMYog2t2CE2Q/nmvemmm2jSpEk4++yzMW3aNCxYsIBaWlspQsdO6C/xB2NAkydMS3a00VNf1mNbvReKRggqBJPEkGmT4bCIUDQNQYUgCEAgrKI834Z5E/KHLHhwgaqqqq7pC4KA6667jl579TUdPLjmLssyHnnkkZg6hu6Ew9nnnB0TwJYkCYFAAK+88op+zlQAbdOmTfTpp5/CZrPpPyMi/OCiH3Tr9uGCdNasWSgqKtKzwnhB4Scff9JrTZYDmMfjoUWLFum1MfyZeWZaonviczFz5kx24okn6nxhnGblD3/4A0KhUEpzY5wjfn1JkiBJEgRB0LO/+M94Uy1joaWmaXoxpqIo+rm4Oy6VY7Ctk8M1tVkaCps70gEwVkgHQ53U3L4DDU3r0dz6FXzeapDihyxIMIkyTKYMCNDAeCFgwgqIbpw30etp4Q4wyQnXsO8hc8SFMNmHR4n8ovd0iMGDC7wnnniC/vznP8d0fXvzzTfR3NyMDz/8MNp3uo++bkSqyj/b00lb63zwBhVoGiHTKmFUrhVjC2zIc5hgkQWmEaGpM0hvr6tDbZsfAHDmkcUQBab3QO/Vu4/bpMb77+lZEm066tIeICJIjedav349/e53v8PixYv1Og4u+Nvb2/Hggw9i2rRpjM99ssHfw6yjZrEJEybQtm3bYLVaoaoqLBYLnnnmGfzqV7+injR1fp8PPfSQ3sRKVVUEAgEUFhbivHPPi4kvJPpbVVWRlZXF5syZQ//+979jXFbvv/8+brrppl5ZH9y6+uCDD7B37164XC6oWqSlrtPp7JLanOw93HTTTVgSjQdpmgabzYbNmzfjN7/5DT366KNMB3VB7KL9aJqmu774vdfU1NDSpUuxatUqVFVVIRgMwuFwoLi4GKWlpSgtLcWwYcNQVFSE3NxcOJ1OxoFmIJQQDqyp7rM0gAwGaCBK5c3NzChtiNvXQHUtX6G2aT1a2rYj6G8CIwWyJMMkyBBNGRBBACl9AA2AZ2qBFGihTgimLDiHnYmMsgtgclZE+Re1aFHioTfOuNsqFArh/vvv17UqbsKbTCYsW7YMixYtogULFvQo8JK5rRgDPvimg7bV+yAxwGEWMb3EjqklDjjMIosHm2HZVnbsETn01Md7MWdMDkYVOnptfXCBejBMfE3TsG/fPvryyy/xxhtv4P3334ff748IxehcyrKM9vZ2/M///A9uuOGGlN0rPHX3Bz/4AW699VbdjWU2m1FfX48//vGPeOSRR7pUiccrCGvXrqVXX30VjiinlSRJ8Hg8OP/885Gbm9vju+UC6Nxzz8W///1vnXDRZrNh5cqV+Oqrr2jixIkprxF+rwsXLtTfkSiIcHvdOPnkkzFy5MhuQZFbISeeeCI783vfo3feeUe39FwuF/7617/C6XTS//3f/zHjXBj/3ij4P/vsM3rmmWfw/vvvo6mpqdt7F0URDocDWVlZyMvLo+LiYpSUlKCkpARFRUUoKChAbm4uMjMz4XQ6YbVaYTabGbdeEu1Broj0RuAPtgvtOwcgpKfoHhAcbZ4aqmragP2N69HSvhPhUDtEBsiiCSbZDhGAwFQ9GN4nnGaRznikBaCGvRDNWXCVXwhn2QWQHSNjgANDKGuCL969e/dSZWVll0XLN/DWrVuxYMGCXmsxvAHV1sYAbW8OwiwJyLaKOOOITOTa5QPdDbnRZgCc/S1+WE0iTplcwKG5188FABs3bqR9+/bpbg6TyQSLxQKHwwGbzQaz2QyTyQRJknSXjqIoCAaDCAQCCIfD+hEKheDz+eB2u9Ha2or6+nrs27cPu3fvxt69e9HW1gYAsNvtOnhwl0h7ezvOO+88PPXUU6y3qbcAcNlll+GBBx5AIBDQK9IdDgf+9re/4ayzz6aT5s9n4XAYsiFmxd03iqLg5z//uQ48qqrqwv9nP/tZSposv9/TTz8dxcXFaGlp0dN5vV4vnnzySTz22GMpWx+CIGDt2rUxtTGSJEVcaj/4Qcz66+ld33ffffjkk090N6GqqnA6nbjnnnuwbt06+u1vf4tj5hzDTHJsSvmuXbtoyZIleOWVV/Dll1/qll1GRkYXIc7nh1uzvK1v1b4qrKE1XQWfJMFiscBisXAAIeM6M74jURRhs9lQXFyM2bNn47LLLkNeXh4bSKshbYH0aG0Iulbf4W+iPQ3rsKd+NZrad0IJuyEzESbJBLPJCQEUdU1FaNipT7ARzdQiNZqOG4TFWoSM8h/ANfxcSLaSA8CBoQUc8YuqpaUlprVqPIj0dfEJDAiqhNU1XggAsmwSLpyczSySoLfFNdYAcheZN6jQ8m9aMG9cLrIdpl5ZH1zofPbZZ3TjjTdiy5YtesVyvPYmSRLEqFYoCoJeDMPdHoqixPixuxPyZrMZLpcrxp8uSRJCoRA8Hg+uuOIKPPnkk4wL4lQ1R143MmzYMHbxxRfT3/72t5iYiiiKuOzSS7F06VIaO3Ys4+4yLsQA4Nprr6VVq1bpfydJEjo6OvDjH/8YRxxxREpWg9GNde5559LfHv8bLBaL3rPjpZdewo033kiJiiKTne/+++/Xiwe5S624uBjnnntuty61+LkZN24cu+uuu+jGG29EZmYmwuFIawCXy4XFixfjo48+wvjx42ns2LFwOp3weDyorKzEN998g46ODgCA0+nUYyac7LI7DZ+/c7PZ3OVz3B2laRo8Hg863W5QD+uIr7P//Oc/eOKJJ/Dhhx9SaWkp62nvpl1Y/YptcOBgULQw9jRuom01y7G/ZQsCwXbIggCzaIZVj2eoOmiwPoOGEOkmogagKD7Iogm2zPHILD4FrqL5EE1ZUeBQAQhDEjjih8fjidmQffWzJrI+tjcFqNWvQhKBk0a5wMEjUfE4gSCA4dOvmyEyhhMm5EdShnuxSRhjaG5upksuuQQ1NTVwOp36Jjduopj4iKZBMQRbGZgeOE0UM4kXFvzgyQiSJEFRFHR0dCAzMxP3338/fv7zn7PebnrjdYkIN910E15++WW9FoPzQDU1NeGUU07Bc889RyeccAIzvFe68cYb8dRTT8VwaoXDYWRkZOD3v/99r1Jv+ed+evVP8cw/DmQ/cUbcP/zhD3jxxRd1C6O7mNsXX3xBb775ZheX2kUXXYTs7OyUXWGiKEJVVdxwww1szZo19MorryArKwuhUAiqqsLlckHTNGzbti2G/FEQBN3a4MKeZ4ERkW59JhRqhmB6fLyCrwUO7nweUpljxhgsFgt27dqF559/Hrfffju+62PAASTG4gBDZ6CNNu9fji37l6PFXQWRESyiGVaTC0K0p7lGKhioTxlELOqeAqnQtCA0NQASRNjsw5CZdxSyik6ALWuyYQWpEdBgh09zG16MNZCD75ddbSEQAXl2GSUuEyMkAY+o9dHmDdOyHa04a1oBbCaxV9YHd121tbWhrq4OTqczJtOpNwH0eK0t/vt4MOGuLz6XTqcTl19+OW699VaMHj26C9Nuryy5KLCXlZWxm2++mW655RZd0+YWQF1dHU4//XScc845dMwxx6CjowOvvfYatm7dGhOL4e60Bx54ACNGjOhVXIvfx+TJk9nZZ59Nr712IMPM6XTi5ZdfxiWXXEKnnXZawhgPn0NFUXDDDTfo70tVDwTPf/azn/W6noSD6TPPPMPa2tpo8eLFOjDw57bZbDEMw/zgYCdJku62ZIxh1KhRmDRpEkaMGAGn04lgKIjamlpUVVehvq4eLS0t6OzshNfrTQoyHGCMsbjunos0QmtrKwBg9uzZAxoDSSXb71sPIBppusXR4m2glZUfYUvNCngDrbCIMiyyHQIjsH6BBjOARhiK4gfTQpBFE+z2YcjKmYas/GPgzJ7EBNFieEHRIsRBBQ6jCcyiQrr/jXyMbo+BucvIXfkVDe2BCA+WSWQ9KgYCGN7f3IgchwmzR2VH+7D3XpCMGjWK/eY3v6H7779/AAGRdbsJZVlGdnY2Zs2ahZNPPhnfP/98HBGlRu9L8kEy4f2b3/yGLV68mD799NMYEOHFdK+99ppejS1JUsJA/mmnnYZf//rXrDtLoad1c9ttt+Htt9+OsVhlWcbVV1+NlStXUnFxcUxMhgtzWZZx88030+rVq5GRkQH+mY6ODvz85z/HyJEje52swQWg1WrFf//7X3bNNdfQ888/D0mSdOAwgoaR9j0cDusWeF5eHi66+CL86JIfYfbs2bDb7QkXXygUQmtrKzU1NaGurg61tbWoqalBbW0t6uvr0djUiNaWVnR0dMDj8cDn8yW1ZuKtqVGjRuEPf/gDTj311C4xrf4I/++0C4sDh8AEdATa6LPd72Nj9TL4Q52wSRbYTS4wKLp7qi/CgUEEA0HTQggrfjDSYDE5kZE1Edm5U5GdeyQcGWOYIJhiQSMa32CDBhx80fMKeZbg9xiQjnADPYIqUVgjyCJDky8MX1gjmyyweBeWFgWKymY/bazuxE+OK42m7fYeHrlguO+++9j5559Pn376Kerq6pDIAjC6nlRVhaKqUKLB8kAggEAgAJ/Pp8dQTCYT7HY7MjMzkZubi8LCQhQWFiIvLw/5+fkYPnw4CgoKYrJ+4rNr+gNg/FwvvfQS5s2bh927dyMrK0v3+QOAy+WKqU8wZtS1tbVh4sSJeOHFF1LSiLtzGU2aNIld97Pr6KEHH0JmZiZCoZBOsHjeeefh3Xffpby8PGZ8L4Ig4NFHH6X7778fTqdTjzWEQiHk5+fjd7/7XZ9JGY29S5577jl20skn0b1/uhdbt27tVglwuVw4/vjjcd6C83Deuedh+PDhzBiXMNaRcOAxmUwoLCxkhYWFmGRgKTaOcDgMt9tNHR0daG1tRWtrK5qbm9Hc3Iy2tjZ4vV4Eg0E9o6u4uBgTJkzAzJkzWW+q+tMWSIrgoWgKPt/7IX2+ezHcgVbY5QhwcBeV0EvgYFEXGEiBogShaAHIggyHNQ85mUcgL3casrInwmYviU0zJVUX5GyQ3VQH2uxGCiEDgRYKBd3QSIUoWmC2ZMJschref6okI4mF7kAPs8iYxEAqAwJhwqLt7ThtTAY5o6m7GhnfM+HNDfWYOMyJ0YX2fhcNEhFmzpzJZs6ceVAXvNEtMtA9urmFVVJSwhYvXkwLFiwA75nOM5jiBR7v/9HW1oYZM2bgzTffRF5uXr/IBvl93HH7HXj/vfexc+dO2Gw2PTNs9erVmDdvHv785z/T/PnzmcViQWVlJT344IN47LHH4HA4dIHGYx+PPPoIioqKWH+sNaOl8aNLfsQuOP8CLF68mD755BPs2LEDHR0dEEURWVlZGDlyJKZPn46jjz4ao0eP7pLqywEv0Rzxe09UCMhBJmqRsuzsbIwYMaJXz5HqHAxUGu+3EkC4JSEwATtbdtB/t72K6tZdsMlW2E1OMFJ6CRxMZ28lTUFY9UPRFFgkCzJdI1GYOxkFudOQlTkGknTAdI1Unh9o63rQKEeiVkco2E5N9avR3vYN/MHOSLCXidF2uXbYbIWUnzcB+Tlj2AFO3d4FRXtL99DzTEfuwioJyLZKqGoPwiQxVLUH8eK6JkwrttP0YXZmlg5szve3NFOTO4RLZhfzt9XvTRGvQfZ2MyXaoMmEBhc6XGgP1uDCu6Kigi1btoxuu+02PPvcs3o2kSzLMe4ZHmi/7rrrcN9998HhcPSbqZbPrdPpZM8//zzNmzdPp2nhdRg7d+7EWWedhXHjxpHD4cDu3bvR0tICp9OpA53JZEJ7ezsuuOAC/OSqn7CBcPXxd6aqKsxmM84++2x29tlnpwT6xoLCVNZIt/EMA8gY101P5x0MxeM7Z4FwqwMA3t7xFn248z0I0OAwu8BIi2jgKYqySGe6iGtKUQIQSYXN7EJu1mgU505Dcf50ZLlGMKPQjcQzeLqlmEAe903T783LZoyhqe5Lqq1agnDYDyZYwUQzJMkMMAlEIjRNQbt7P1o796OhZSeNKT8eZpOd9fb+eBOfgc07j2BuRZYJe9uCkBGJgwQVwtJd7dhc46GJRTbk2mXsbPBi+a42HFnqQr7TzHpow94rYTvQ1tVQ2GwcRFwuF3v00Ufxs5/9jP7973/js88+Q2VlpV7zUlBQgLlz5+Kyyy/D9GnTGXfLDMSccFfWzJkz2TPPPEMXX3yxbglxenUg0uecx2iM8Riz2Yy2tjYceeSRePrppwe8lzlPnuDpsXwtGN2WxjUy4EpUH12Eg3Wd7wyAcPDoDLrp2Y3P4Kv6Tcgw2aOMrQpS0f8ZEyBEQcMf9kNmDBm2PBRnjcXwvGkozBkPhzU/gWsKUep2EV5vDbk7dqCz7WsEfTVg0OB0jkDRiPNhMuewwQIR7raqq3yXavctgWByQpLt0EiM0qNT1D4jAAIkUQZBRGtHNTZ98x4mjzmDLL0EEQ4gAyrkopcem2tmq2u85Aup0aJNwG4S0BFQ8MnONp3r3SIJmDTMqVufh0+nj0MHIlwQjh07lt1222247bbb4PP5yOfzcVdNjGtmoAGVWxwXXXQRCwQCdPXVV+tFfPzeOJDwf3PtmrvUohXkepbaQAvXeG2eWyiHE1vvYLPxfmsAhINHdWcN/X3tk2j01CHD7AKRAo2oR6sjEi/QEFL8gKYgy5aL8uI5qCiciaLs8TDLxqyK+EZREWhqbFxNu3e9Aq97DzTFDZE0iEyAyAjt9Z+hrf4zTDj6rySbswccRDh4tDasorp970E2ZUGDGL3P5JxABA0myQJfoANf7/0cU8ecGqFLSfG6FotlwC0QboVYJAHHlTnw9vZ2mFnEslA1QBIYRFmAphFUhWCRGIZlWYb8gh5qgoVTehgqmpnNZtM/wwPVg+EW4e5PVVXxP//zP2zEiBH0q1/9Chs3bgQQyYridRKc+oRnPF122WX461//CpfLlW7+1EcASafxGoYaFdRft3xDj695CkHFB4fJATVqdfQ8yQJCig8CA8qzx2Pi8OMwMn86rCYnMwro6Iej3QjFGJeR21NNa9b/EUwNwiRZIMsuiCAIETEOUXbC765E9fYnMXLK76ICf6AmPxLzUMIeqt/7NkTJ1isuLqIIiLS561DbvJNK8sb22O6U/26wAIRF+kFhbK6FnTY6gz7e2QFNi1C4a4gG0SlCtOiyibBFc33T8NF3V118bGYw4zHx7qx58+axlStX4qWXXqKXX34ZmzdvRnt7u14omJWVhfknzcfPfvZznHbqqQPqUkuDS//Wz2ENIBw8Njduo0fXPAUGFRbJDDXFWAcDQyDsQXn2GBw7ZgFG5E3uAhqsWwLDiCXh9zdC08KwyC6AQjqhov4fKZDNmWit+RAFZeeQPXPCgHUQ5MK+vXE1QsEWiKasCPtsL9YHgSAKEmqad6I4d3TK92Ws1O6p3qGvIDKpwMqyrSK9s7UNnkCcUqB3GExDx1AQKH0FER6wv/LKK9mVV16J2tpaqqqqgtfrhd1uR3l5OQoLC3XgONxcSYfz+/7WWiBaFDy+atpOD695CgIAsyBHM6xSszyCYT+ml56AMyZdwURBjGPjFVJ+CZkZo2AxZ0MLuyEmnVQGIgX1u19GxZH/N+ALwd2yBYxJ0RfeuxdLRBAFEb5ABzz+dnLaslkq8QSz2az7swdnkUesjRKXiZ0/OZv+ua4JYSVS5KkBEATAH1YRCKuwmsRBTlFIj8G0hHisQxAEFBcXs+Li4tj9bqAsT4/+ywzWC464w9ECEXoCD4EJ2NW+jx5a8zQAQBZEaNS71EuNVJRkjYpw/uNACmjqenSkzsJkymAlxccjrPiS1ngQqRAlBzqbVsLv3k0s2hSqv+4rgEFTgwgGGhFpdtVXK4BB1VR4/O1IdRJkWR50V4cQjX3k2mU2q8yJoHLA/ScwBl9QRYdfoV6+uPQYgkKNxz14eiw/eDZUGjwO/uiJDPSwAxDOZ9Xka6U/r34aYU2BJMhJwYNXosdr00QaLLINi7e8gNfXPkz7mrcQ/zyLAgOlIOC5MBs58vuwWHKhaeHkejAToCp+tFQv0l1HAzFUNUCaEsRANHIMhv0pP3OiPgWDo6FGsGFSkQ0Os6g3iGIsEgdxB5Q0fnwLwSQR6WB69B8QBiqIftgBCE9CDalhPLjuObQFOmEWzQnBg0UD3v6wH76wFyqpMdTtRoDZXrcar676E/715e30VdXH5A91EjN8lkjtZiIZiFRYzNmsdPipUBRvcvcXaRAlKzobl0FTvBSxVvov9iK0KAOzyXoDagOZ3tld2J/3+7CbRHZEgRX+sAZJiNTqaASYooWFaTGTHumRtkCSAginqfjHljfp65bdsMs2aKQmtDrCahghNYgj8sZjeslsWGUbvCEPgmFf9DMHNBurbIcsmVHT9g0Wb3oCLy+7GUu/eopqWraQFi0OPMCemghMIjXUJcNOhiTZddqSRGKSCWYEfTVwN6/RLaH+QAcAiJKNiZINgNrviTfL1oO3OA2gYeybrlFXMGHR5Oc5IzKQa5fREVDR4VdQkW9DSaaFEQYMQ9MjPdLjMAcQqSt4RILmy2o20PuVy5BpdibMthKZCF/YjSJ7AS6aeBHG509kAOAOdtD2xs3YUrsK1a3fwBvqhEU0wSzK4MSCJskKEQR/qBObKz/A9qoPkecqoxEFMzGi8Chku0YwYwpvJIU2EnAn0mC3D2P5BbOpofYTSLIzeYyDgI76pcgoPL7/ejMRmCDBYitCwN8EJvT1fJFMLKc1G6mq8z01jUralzoOMEIqIaQRCWCwykx/BGNlOac5cZhF9qOZBbS5xgOzKGDKMAeTRJZ2X6VHeiTwwvRX+He3v4dyTEqKfwjGGFoDnbRwyxuwSpbIz7pMiojOUCemF0zCj6f+GE6zk3FB7zRnsJnDj8XM4cei0V1DW+tWY3vdGjR3ViEMBRbJHMliggaRSTCZnBCgoc29Dy3t32Dr7jeQn1lBZYWzMaxgJpy2Ip3GhMdLGGMYXnomGus+i1ZFJxS7ECQrvC3roYbaSTRl9quwkGdLuXIno615IwTG0Nts2khjLRV2aw4c1gymu8V6GD1xACVafMYn3d2p0q4OBa1+FUFFBTSCVQQNc0qYlGeByyywRCDiskhsbkVmAlssPdIjPVIBkO+UBRLp+SDg+a/fRUugA5kme5QS3ThZAjqDbpwy8gT8aNLFjIEZuLGYnqLLGEO+s4TlO8/DcaPOwt7mrbS1Zjn2NW2GL9gKkyBBlEy6BiyJZoiiBYzCaGjdhobmjdjyzb9QmD2eSovnoih/Bkyyk/G4R1b2JGa3l1LQtx+iKKOLNCeCIMhQAk3wNK9GRvEpUfAR+7VIMnKnMYttCQWDnYDYW4oRBlULozB7RNSaSo0SOiMjg2VkZFBnZ2eXOhDGGIyVzUZB71OIljUo2O9RAI3AKGIBairBH9ZQ7wljS4Mfx5XaaVyeJSGI8GsJab9VeqTHoIHLYW+BcBD4qnkXLd2/Fk6TPVJAGCeUvCEfLhx3Ls4ecwYzdh888BkW2+A+6rIZlT+Fjcqfgk5/C+2sX41var9AU8duhNQgrJJFt0oYAFmyQoAVRApqGlahruFLOG35KCmYRSWFc5GRMZqRFiYipWedmDF0NnyGjOJT+lkIF8kYE0QLCkecg71bF0IULb1qiaVpCmxmF4pyR7FUFhdnLTWZTDjqqKNQVVUFk8mEcDgcQzw3f/78GE2FAIQ04ON6BU1+DRaJQVMAVSO917nEGCSJIaxoeG9nJ8Iq0eRCaxcQSWfmpEd69B08vjMAEkmpJbz8zeKEgpYxAZ6QBz8cfw7OHn0a06jnFqA8Q4u7nwDAZc1hR444HUeOOB01rdtox/5lqG5cC5+/Se+RzjOuAAaTbIfAgFCoHbv3/hdV+xbBYcsjgTSEgy2QBDlpgJxIgyha4W/dBDXURpF+6H13Y/EYTGbuVFZYdgbV7nsfoikz+vOe/1ZVFRRnlUESTb1qSENEuP3227F48WK43e4oGEWe+dZbb8WUKVN0viL+dBvaVGoJAlYRULSuwXKKWn4CY7BIDB/v6US+XaJCpzxgbLvpkR7p0X8X1pAHEG59rK7fRlua98AhW3RmXe6+8Ia9OKl8rg4eBAIjlrIs1lN1DVZLSfZ4VpI9Hv7gRVTZsBq7az5HS9t2KKoPJskMSTBF/oJUCEyCyZQBERoC/maIjCAyCZFa6aSvBUyQoQSb4WvbBGfB8f1yYxlBpKj8DCZIVqqt+hCqEoYg2g+QPjIBgBCVwrETJAqSQYT3PHnc0pg4cSJbvnw5PfDAA9i5cyfy8/Nx6aWX4vzzz+8CHu4wUaWXYBaBntpt8KwqTSOsqHLjvAnZ6d2cHulxkEd3lehDHkC4Jvz2nmVRXzd1ETKSIKLN34GtTdvpiNzRTDRkSRF4d7oUNOoYF1dk0qzmDDau9GSMKz0ZTe07qbL2c+xvWAmftw4iALNsjgJQlGZBkBHJB0otNZdA8LWsh7Pg+AEyTQWACAXDTmDOzDFUX7MMHR17EVb8UMEAJkd6gjAJJACCaIq68kQ0te9DaeHkXgXGeIOgyZMnsxdeeKGL5hJ/rlq/hrBGkFPWfiI1HjUdIXQEVMqwiCxNV5Ie6TFEXFgHgXCzzwDCrY8dbVW0pWUPrJKlS80HEcEsmrC5aRu2Nm3FiIwSmlk0DTOKp6PAXqATk2ukxQBEb6wSRF06eZmjWV7maEwdczHVNq5FVc1naGndjGCoA7Joglm0gEFDyilQ0WC6v2NbxCIZqDa30Ta2NkcJGzn2IoSCHeTx7IfP14Kw4oemERRNgS/YCY+/DaIoQhBM8PjasL9pO5UWTGC96S/NmxRxwOAaSyLtpCPc+5oXBiCgEJq9YWRYRKTdWOmRHofWhaW3FR7KFgi/7Y/3r0dIU2BjJqiU+AEtkgUiNOzvrEFl+14s2f0BxuWMpVnDjsL4vInMLFli3FTGGEhPVgnirBJZsrGy4uNQVnwc3N79VFO3HHV1y+Bx74WqhWGSLGCCFAUTtVvrQxBMCHv3Qwk0k2TJG7A+IdwSAQCTOYNlmzOQnRM/byoamnfQ7qovQaRBEk2orN+CvMwyspodrDfNmYyWRvdpvX3CQxARgkq60iM90mMoAIgupIeyBSIyAQElhHWNO2CRTF3oSljcQxIIJtEEq2SCSgo21m/A5vp1KLIX0NSiaZhefBRKMsoY16wjVglSpi4/8DnSGW+d9mHsiFEXYczI89HSsolqaz9Ba9MahALNkAUJJsnSDYgQwCSo4U6EvPsgWfIwoOq1fp74nsq8laWIwrzxDEykHZXLIMkOBJUQdtVswKSRxw5K40Sr1NUNmYobizGGaMuP9EiP9OiFq6q/LqzuYiAcQIZiRqQEANvb9lG9rxVOyQyK0nSwqGWgkgaTIEQC6lFwISJoIAhgsMk2CCC0+Vvw4c538cXejzAyexRNL5mN8QVTYTM0jNIzt1KSmLHpwIAGQZCQl3cky8s7EoFAMzXVf4HG2k/gad8KQTR1+4I1LYyQpxK2nBndFB/2axklfcFEGgpzx7J2dz3VteyByeRAY3s16tv2UWFWWa9cWamMbJMQscx6ASKESE/0HLsUi4vpkR7p0ScA6Y3QP2wtEADY1LwHqqbqZHoRYU/QSIVDNqMz0AGZATbJDJEJYHRAOGnR7yVBhkWUANKwq3kbdjduRo4tB+MKptKUkmNQmj2GCTppohZnbaTygsSYv7VYctnw8nMwvPwc1FUtot3bHolmZSV/EWFv9SFcYISK4bNYu6eJgkoIkihjV80mZDkKYJYtGEhTJMPEYBIi9OypQR8QVgnDnDKyrFI6gJ4e6XGQRyoWyJAFkJ3t+yFFGz3xEmRVU/Hr6T/C+JyR2NT4Nb6sWYNdrXvQGfbBJsqwSDIEkO46IlAkHRiAWbJCAsEXcmPV3g+xsepTDM8cQROLZ2Ns0Uw4LdmxVkkfAu/GnulFpd9jnW2bqXn/+zDLzi7urIiGL0IJ1HNb4WBDCIgIsmTB6NLZ2LTrE8iiBcFwADv2b6LJI2YlrL3oopWk0Eed0DuyeQaAGINGhFnDHQdOkkaQ9EiPgza6C6LLsjxk71tyh3xU522BLEogAkRBQEfIg/NHnYi5JdMYAJxQOhsnlM7Gvo79tLp2HTbUb0SDpxYMGuyiGYIQdZnQAZeNBoLEJJhMDghE2N+2C/tbtmHFrjdRkTeZJpTMxfDcCUxgYpxVklo6MKI90yOsvRryi09C8/7FyS0QJkIJtkW/P/jcMrwgMCdjGCvJG0v7m3fBJDtR374f2S0FNCynnMWDaVKXmEH4JwKPXW4VIS2iHTBEmkVp7ACxon4wQAHgDak4ptSBsixzuogwPdJjiADIYWGB1Hhb0B7yQGIiQISQGkaBLQcXjDk52lCIdGFWljGMlWUMwzljTse2pq9pTe1q7Gj6Gp5gO8yCCJsoR9rURt1aEaskEm8wSWZIsCCk+LGl+lPsqPkcBa4yGlM0G6OLjkaGvTCuTzpLsVI7wsFld42CyZwFTfHGUKtw0cqYAE3xgEgB011dB1dSMhbhCqsomcZa3Y0UUEKQRRN21GyF1WSnHGeefkNBJYTOgI+84RAUjSCJEuwmCzLNFiZHM7AoAXjs9Si0tUOBSZCgaUBYi7inSI2kSoMiNO6qRgirBBmE48udmFliS4NHeqTHwGx0ULS4t79BdMbY0LZA9nsaEVIV2GQZAgBPOIjzR58Ih2xjapTa3YiSPAtrauEUNrVwClr9LbSpbgM21q1BTfseBNQQLKIJFlGKdXERQYMGgYmwmhxgIDS796Gp/Rts2v0mhudOojElx2FY3lQmRUkKjfUhyYV95OcmcyYzWfIo0NkRJTmMd/8IIDUAUkPEJOnQiUkCJNGEscNnYOPu5RAkBhBh4771yM8oJpNkgTvkR1vAj6CiQoMIYhI0iGCiDItkoXyHE+UZGci2mJlxFva6w7SiOQhJlKAhwodVaBWQZxGhqoTWgAp3UIWiAiYmIN8mYnyuGdlWMd3nIz3SY6DwA71vX5fMhSUIwlAHkGb95hXS4DLZceKwGQAAIU5oGzOoeLpvtjWHnTDyJJww8iRUtu2ijTWrsb1hA9p8DZAAWCUTBCZFKsejFo1GBAEEOcrAq1IYe+pWoLLuS+Q4SmhE0SyMKD4WWc5yFl8fkjDwHlWdZVMm/KQe6IoU91pJC4MofIiVk4grK9tVxHIziqmxswGiaIUGoKatFgQRJEhgogxZkEGCBGJiBEggIKQq2NPegUq3D8UOJ1VkOGGVRFR6QtjSFoIsSlAp4raamy9ipMPYuESO1O8TYMzWTVse6ZEeh3Yks0AEQRjaabz1vlYwxiAwAT7Fh9mF45Fvy2a8K2GyISTgtirPGsXKs0YhMPZc2tG4GV/VrsS+lm3whjphFqRI4J2xaDow6VaJCAazbIcAgttXj43fvIIde99GYdZ4Ki85DiX5R8FkSAcmUnXXFb8HBgZJtgPRupPEMK8lbz510A0RQrarEA0ddQfEuyQDiICFaqDG1/9jvFZDhAaGKrcX+70hiExGkBgkUYJCkXTceQUickyMcdLEaPMRMBwAD87MmwaP9EiPQywPklkgoji0LZCWQKcOFCppmFk4/oD7qA/cVgTAItvYlJLZmFIyGy3eetpWuxrb61ahubMSoEhTKUkQo1aJavjbSDqwIJoAUlDXtB71jWvgsuejJH8mDS8+HjnZEwzdCjkYaAAECKIt5fs+9GYugy/ojrlX3pSLGPW82BhgEgRojOk1HBHwAOYXyMgwMRaZleQAIbD+LHgY1glfB2lBkB5p4T+QFog41F1YnSEfRCZA0VQ4TTZMyhmV0H2VmntGiPr/olQmYMixF7JjR5+NOaPORFXLNvq69ktUNm6A198MWWAwi+ZI33RS+V9GGHMByLIdIoBgqAO7K9/Gvur3kZ0xioYVHovCwmNgtRVFb1LQOxV2p/ODiQPHhdWPxcYYg9vXRrXNeyCJMghc1PfWiokdCgFHZZuQYRKYRv0DiGTXo6hleqBnSOxFtOh7T4NJeqQBJBWZyRKegxOoSpI0tAHErwQhMAFBNYQxGcUotGen1OwodauEs/UKKM+dyMpzJ8IX7KA9DevwTd0XaGjdDn/YDbNohkmUI+IozioRmATZ5IJAGjrad6KzdQv27HoZuTmTqaj4eGTnHglJdrCgvzGS2puw0pzARBOYYDq0oo0BihrC1soVkboZIdoat593pREgC0CuRcRgBMRJd3dFTtzuC1OHN4ygokISBLisEnKcZiboFDbJXaADobGlR3p8W6yPZBYIEUEc6gASUsMQmYCwFsbYrDKdvkQcoFqJRE2lbOYMNrH0REwsPRFNnXtpd+2X2Fe/Ch2eaohQYZYsYEw0sO6S7h6TJCtEWEAURkP9MjTXfQaHowQZGaPJ0/41RNGSOM5BGgTRBiaYD0jyQ2R97KpZTx5/G0yyHeoALUDGIim7njDBIfU+C6QncBIYEFQ0rKvsoK+qOtDYEYAvqEBVNZBGkAUgy26i8cNcOOaIPGQ5TEnTgtNdDtPj2zIGai0nAxDJEAMZkkF0Jdq2loFhXHb5IE92fOCdIc81guW5RmDG6POxv2kj7an5HA0tGxEMtsEkmGCWTAbWWw4kkYI7WXZABCHob0KTtzrC0JtwkiMdDkVTRlSiHfw4CQePTm8T1TXvgkm26vGDATJsoBFQ7VVQaB04Nx0Hj52NPlq0uRGN7YFoPlikh4gmRHqsa6qGujY/Kus9+HxLI86YUUzzJhYkBBFVVQ+ZpvddEkrp+Rj49aRpEVc5r/FQVRWBQGDAAMR4X0QEWZaHtgUCRILnTpMNozOHAehb/KM/Li6AIIlmlBfOYuWFs+D2NVBV3ReorluOjo5dYBSGWTKD8UZSOqlj1MUlyJBEGSAFiXRvFu1nLppzdWsEhygWUtO0PWkL3v66mGQB2O9TMUUjmAYgAMLBY31VJ729sREAwW6WoKkqVFWDFjUQ+Zo3SQwmQUYorOKlT3ajsc1PFxxbzjQiaKoKSZLwl7/8hR5//HHYHXaoijroGz4tML8d8zHQc6KR1msznQBoqgrGGCRJgiRJCIVCqK2thdVqhaIoAwpoRDT0YyAiE+BXQxjlLEK+NYsd7AV8oEjwAB2601bAJlQswPiR56KpZQtV1S5FY9NqBP1NEfp2MdqhMBp470ql3lU9J9Ig20r0hcAOwSZQ1BDa3fUQBUnPpBrIITLAq2io92tUau9fV0GKgsfuJj+9s7kZJimSqKCplLTfCFGkNa4gABl2E95fux8ZdplOmV7ClOgf+Xw+7Nq1C2azGeFwOC0d0+OwBHeiAz2PTCZTTK8e/nP+mf64sEwm09CmMrFKZnQGPRibVRoxyQYw/tFru4QdqOtAtHd5fu5klp87GYFgG9XVf4na2qVwt38NVQ1Em0rJANTuuyjxXheOskMyyXxBeXytFAr5wCQLBlOfrg9oKLWL0SlJHIjoIV8NABAIa3hvawskgYEx6rG/uhFISCO4bDLeXlGNSeVZVJhlZUTAZZddhgcffBDBYBAmkykdUE+Pw350R0PSVwDhf2symSDL8pA1SaXhjjzsaq/GzIIjehQsBxFKdBeTTt9uzmIjyr6HEWXfQ1vbNqqr+QjNDV8i5G+ALMoQBVPSsxFUMNEGk2PkgfMfguELtEMjFd2TzvdjIQOQBYb9Pg2TMwkWMTViSo4zZAAAUQCW72mnVl8YVpFBUXvZoAqAwBgCIQUfrKnBj08dDUVRMGzYMHbHHXfQL37xC2RkZEBRlDSIpMe30lJRFCVl6yERCGmaNvQtkPNHHYdiezam5o5mkU0vDLEXEUvfzpiIrKzxLCtrPEKjL6XG+mVoqF6EgKcyCjrURdMmLQyTtQAm2zBOc3uwbRAADMGQd9CvJAAIqIRPmzQaafbBCg9MpgyIggRJkCAyQBIYkwRAZAwCi61S55NW3R6kddUeWGUBmtrHgCMRrGYRm/e2osMbogy7iSmKiuuvv55t2bqFnnryKTgcDoii2G0/hPT49rqC4i31wTjvQReqkoS2tjacdtppGDFiBNOixIo97ZVE82G2WHpk5z6kzzolt4JNya04HJYc4ivQTeYsNqzsbOTlz6Z1y66M8lzFUZkxAaSEYHaNARPN0QC6cNDvHQBUbfB9/gRAYkBrkNDgMwGqM+LK0hSAwhHXoEYkgCAxgsQYJBYJussMMIsMYVXDtnrvgNyMKAhw+0LYU+fGtFE5eoHUk088yXJzcun++++HoigJN9jBtkzSllB6DNQoLS3F3//+95SVI0rgwgIAq8Wir80hCSDGQr/DR3sR+KxHm2AJEAQJUEIJrQsiFbbsI7kdc8jcdAfTdSYxgAkCNJj0lsCqFoljKBpFvlc1qKoGRdOgqgRNUaFpBE3TIDMGkUUysfoLnapGaGjz6xuDsciG+L//+z925pln0l/+8hd8/fXXerYNF+TJsm8EQUgo7PuTrTMYmT6HoxWQ1LLtRoNmXTL+WJci0vj5ZUKU8JRFPqtRpJ6IC1utu0xFij1fou/jD7CIdyXZ7xljYEIk/5Q/a/xXxonjEiTAkBYJmAeDQWRnZ+P2229HeXl5StZHMhcWAFiGOoCk3qN86I3I+hMRCrZAUbyQmYRIJODAQo7Uf7hgz53Jl/ahulOYTLaDYuuEomy7JgEIEyGgEhSVIiChUSSTSos0BiFEXFmRFiORehsiBkXV0IdyjaT35A8pXQSXqqo4+uij2dFHH61vIGOWS6INk05xTY/DZaQCHnw9JwMQm802pK1j6fB+RZFJ9Xv3Q1WDkGVTXBU6g6b64cg5Mhr/OFS85ZFrOm25YEwYtAysCHgQSmwCpmSJsImMqUTkVQieMMEb0uBXNIQVDSGV4A+r8AY1+EIqAmEVYWIIhwmBsAqR9a2vQbK3ZJa71t1w856IIIpiGigO1104xF1/vb2//nJbceUnFcujp2tyABmqQ/o2LGB3xzeRxlOJ0F1T4Co8MfqStENCpsiFodOex6wWF/lDfjBhgO+DASoBLlnAsfkmdqDfB2M2iSHPAgCJr6lSpDthWCUKhDXUdoawfHcHPEGl37Ypz8bKy7B06xpJtoHSQPLtcoWl7y/xiC9C5Pdst9uHNEgLh/fCjdy+u30HBEGOowZh0LQwJEsBXIXHx3z+UGlBoiAhL7MMqqYM+KJmABQNmJRpiold8KbEFG1lyw9jBbnIGCySAKdZZHkOmU0ptrNzJuX0f9GySGGh3SJhRKGz282c1HedHunxLR6CIEDTNNTX1yesG3E4HEP7/g9jwxkAQ8DfSF7P3mgPkdjsK03xIqPoRIimTEakYihUueRkDIPAxAHVKDiRYp5FxHC7FE3HPgAsLPoZwXDEN5Iiw6FqwLBMMxudZ0NQ0fpMCy8whkBYxagSF3IzLCxi2qeFRnqkBwDdfbt3717auXMnLBZLF7ngdDrTADI4Gn0k1tHWsgGhYEckCwtGIjIVouxEbtn5UUF6iCUXi2SAyZIVwgC7r/hTT8+x9FnYM8PBzzG73BXhEevfi8LJ04tj7jM90iM9DhAzvv766/B4PJBlOQZAGGO6CysNIIMhkQE01X8RCUzHeK9EqGE3sktOh8k+jNEhqf3oKkgZGJraKqGq4QFxz3CBH1Q1TMuxIdcisYHg+eKZiiWZZjazzAlPUIXYy+mTBIYOXwhzJxZgXGkmox5aJKdHenzXwEMQBHR0dNDjjz8Ok8nUhaWaiFBcXJwGkIGXxQTGBPh9ddTWsgmiZMWB9F0G0sKQzdkoqLgUwKHPn44E7wU0t++jfXUbIYqmLmmqvb1HIRrnCKoaJmfbMSbDwgaSJJKDyElHZLMJRXZ0+lXd/dXT30kCQ4c3jHHDM/GDeSMY0eHRZjg90uNgDFVVI+0sBAG//NUvUVVVBbPZHFP/FAqFkJmZiXnz5kX2uzA0RfVhmoWlARBRu38JwuEOWEwZUSr3SF2IGmrH8CNuhGzJY1x4DyY4dGchRQBMQFPrbvq68nMwJkdJKwmKEoIgWqCRBkUJQRQtEITkwWPGJTQizZ1kUcSs/EyMcNnYoIhoFgmwX3BkAXOYRFq1tw3QCCYWiW/owRUGkMAAYgioGvyBMKaPysbl8yuYWRaRrLFUKnGggfpMfz7/bR7pRIWDt14YYxBFUU9Z/+1vf0vPP/c8nE5njPUhiiK8Xi8uuugiDBs2jKmq2iXNfcisn8NvM0XuNxz20splP4USaoMsiBBIg8gYEHYjI2cKxs56mEXoT4RBWlSpA1N90zbauW85mGgBBBmKqkGUrCgvmoRsZyHCahgNHQ2o62iEPxwCMQkQZTAmgZgMYgJUEqFChAYRkiSjyOHC+JwsOGSJDaZ+bzz3Nw1eWrajBVVNPgRCCjQtwk+mKpGqdgZCvsuCeRPyMGd8fqQ7cRQ8NE3Tzfahqk2lR3ocDOvjiy++oD/96U/44IMP4HA4EhYRqqqKlStXYurUqUMaQA47C4TXctRUvw+frxZWUwaIFIBFXFeSbEf5pP+N1HuQNlg3EY27aPB07iWPuwqhoDtiF0l2WKw5MFuyAQCNLTtR37wdgmgFYwJCSghmcwYmVRwPhzVTl/uZ9myMyK9AQ2cTNXna4Q4FEFIjvRfBRJglE2xmK3JtThQ5HHCaIr3dB9s5xA48MsYU2NmYAjtq2wK0t8mLxo4gfEEFIgOy7DLK8+yoKHQySWT63xBpkd4iCYCDiKCqKjRNg6qqUFWVol97PJToVy36vRb3e35O41fjoWoaKO5nRHTga5Rag/d0iD/4/VOUxlj/mWGNdOvnA+mJHboVwJiB2PIABQf/PtG/+3UIAgQ2QOdKQifCnykZ7QgMz9vl+1TVScP7iP+ayhH//hMdqqZGqX9i1uuB9ago0XWpQFUi/+ZHOByGoihobm7G5s2bsXnzZqiq2sXy4NaH2+3GaaedhqlTpzJN04YseByGABIR3OGwm/ZV/heSZIsASnTlqaoPo6fcBYt9+CC6riIqdXvTRqrf/zG8vsaIoIQMYiJUJgJMAhNMkX9DhChawRhDWA3DZsnE5NEnwWJyRO+R6XTqJsmE4dklbHh2CTQihFUFWjRWIotiTJ8WihPwg+/qOGBNFGdZWHGWJbmDMVpkYqww37BhA73++utYvXo1WltbI/MR3Vj8SAQUyQS+UdgnEu7pMbRcYwldZaxnqEjI2mvsPdANoCRyQR3q9cEYg81mgyAICVs78+c988wz9TU+lC32wwpAuPVRufdN+Hx1euyDMRHhYDNGjrsGOUUnMCJ1UCrOeeC7Zvd/qHH/p4BoiYAYk0CQQEyEEAUQjQSAiWCCBA2AqmmQZRsmj+LgQTrAMQMS8E6FAmMwS3LizcEOTVIyYweU65iiTTJq46T7egFg7dq1dO+99+Ldd99FMBjsIhS602ARp7nyr3xDGfsk9JU367sSAxhowdnf8w22IO/te0318/1dL7pClIRojs/LsGHDDot1JR0+GyBiUfh89bS38i1IsiP6MwmhYDPKKi7G8FGXDiJ4RK5fs+tVatz/MSRzDjRE21ZCix4syvWrIdLI6kDbS1ULY2TRVFjMzm6to+4WqDGIfmi1qAhocN9tIvfU9u3b6cGHHsJLL76IQCAAu92uZ5ok6//cW8EzlDTL9Ph2A+bBttzWr1+Pc845Z+jf7+Ey0Vzobtj0Z6qtXgKL2QWBVKjBNpSNPB+jJ1yvu4QG2rHDr93euJr2bX0SojkHGgCCGLE+mKhbIFrUAon8ToQW/b2qEaYccTZcjgJG9O3TfD0eD+3btw+bN2/G4sWL8dZbb6G9vR12ux2iKHbh+kmP9DiYFsbhAkS8k6HL5cK6deswfPhwNpTdWIeFBcIFeGvbNtpfsxQWUyY0LQg17MGosVdg5JjLBw08eNxFUwNo2PsWBMmC3oauGRg0UuELdMLlKNQtlMNZs2ttbaV//etf+PTTT7Ft2zZUV1fD4/HonxNFEXa7XQ+UC4KQEr8VdeO//rZomYMlIAfLbTNYwjdVSzTRzxJlLg1IR8OY1pxI6kpNNH+9mU9jDC/+5yaTCc3NzfjRj36EDz/8EJIkDd1+IIfTpqqseg8AIr0/RBMmTLsFxSUnDyJ4HIh7dDZvpKCvDqIpM1L70Us3GQPQ1rkfhbljcDgX1fFFbLPZ2Ny5c6mwsBD79u1DTU0NGhoa0NTUhJaWFrS2tsLj8cDv9yMUCiEcDg/IdZPGSgwbP5UCzYHcjMkEVyrCsDs3XHxWUU+fP9wBs6csM+Pn4hWSnmJoPb2/iCsaQFwGnp6JRQQtmtQxUPMuyzKsVqteWMgHz9D6/PPP8eMrfkz/fOmfzKiEpQGk14srYr61d+yGoniRlzcTEyZeB6dz5KDFPLq4aFq/OuD87yUAEAiCIKHT2wRFDUMS5cN+w1utVkydOpVNnTo14Yb0er3k8XjgdrvhdrvR2dmpf+/xeODxeOD1euHz+eD3++H3+xEIBBAIBBAMBhEMBhEOh3Xwic/WMqbuJkq77C5tM6lwjgqQVACnuyC/UYglEoJG4RefTst/Z/waf/DPi6IY83Pjv/n3oihCEEWI0e8T/btPhyRB6uEzgihAFOJ+Fn9vcfct8J8ZnzX6vPrzxwFIl/lMACSJ3mGidF9jhl98mm44HEYoHEYoGIz5eSgU0tdsMBhEKBSCoii6laRqGkRBgMlkgqZpaG9vxzfffIPly5dj3759MJvNXahMFEVBRkYGXv7ny6gYWUF33XXXkKwHOUxiIBGh3dSykVTFj8KCo5nRtXUwrr973d3k8+wHk2wAIrGOlGMgEIHo99PHnQObJYORoQ7gcB1GQc036GBqSfF5+qqqUjLwSJTqa9QeD2idsXUcqVpB3YEFB4n4OUl0xPzuAJCwpJ9J09x/a0ZraystXLgQf/7zn9Hc3AyXyxUTK+Tv2uv1YtmyZZgzZ86QAxF2uJrBB9MnSFoIO9fcTqFgGyBYomDQewBRIWDqEWfBacv5VgBIKi6d7jT+7rT8ZFpkevTsKuttltvBkAFDLT4zkK7KvsyhMdV99+7ddO211+LDDz9MSmsyZ84cfPbZZ4xzaKVdWH16gZru0jqoC4yJYIIIQv/6WTDGInUih3ADqCk2Ou9tQDBZTUZvfMbd+fcTaV38eVJtRJWsQrmn8ydKAEjkDks2RFFMeE8xvu9eVmAb+w3Hn9/Yc76/4GAUdH1dS93NUTJqG25NDhVA68v5uLBPJvD5HFZUVLDFixfjmmuuoaeeeioGRFRVhd1ux7Jly/DKK6/QxRdfPKSsEJbOn0/NfVa5+SFyt26DIDtBEHpngTARRAJEkx0zxi9gkmj6zmjGh0KTNLqyYsA7zcOVXieHYPSUhmusp7rs8svpxRdeiAERQRAQDAYxYsQIrF+/HlarlQ0VyzwNIClYPYwJaKp6j+p2/xuiKQsE1isAAZMQ1lTkZI7EhFEns4O9YYwxgD//+c+0d+9eCILQReNP9d+MMb0ZTjgcxpgxY3DTTTcxs9ms/54/48KFC2nlypV6t7WefPvxv1dVFYWFhbj++uuZLMsx2mwgEMDDDz9Ma9euhcfrhdfj0YPvqiFjhp9LFEVIkgSz2Qyr1QqHw4GMjAxcdNFFOOOMM/T3wjd8bW0t3XvvvdizZw88Hg9CoRA4N5Esy7BYLLBarbDZbLBarbBYLJAkKVI4qqpob29Hbm4ubr/9drhcLmacO0EQsOi99+j1//wHJnMkgEoAKCpMNC3SHzJCCwNopEVjNdqBbCGNcPTRR+OqK6+EzWbT759/ffbZZ2n58uWwWq0xGn93X/nci6KIcDiMyVMm47JLL+ty7ubmZvrjH/8In8/XZS2lqq1rmoZf/OIXmDhxol7rwL+uXLWSnnv2OTBRiElsSIWuhs9ddyohdWvdRAuESYs19eI+E3knCZJqWOT9mExmXHvtNZg+bTpLFUS8Xi9NmjQJdXV1etAdiLAudHZ24k/3/gm/vfm3Q8cKSZVw7Lt7RIRQyN9MW5ZdT18tu56+Wn4jbfrif2njF7+lDV/eRutX3E7rVtxNa1feQ6tX3UerVz9Iq1b/hVau+St9ufbv9OW6p+nT1X+n5rZK6o7DabAORVFARPjVr35l7Fw7oMf7779P/Fr8ei+88MKAnX/p0qX6+cPhMIgIv//977t8jjFGgiCQIAgkiqJ+8J8xxrr8jSzLtHPnTuIuBX6cdNJJA3Lvv/jFL/R7526lXbt3kdlqGZDzn37G6RQKhfT75m1SZVkekPP//e9/7zL3jz322ICce8SIEdTS0kJGig8iwrzj5w3aWj2Yh93hoJUrV5Kmafq+SHbwuX3ppZcIADmdTrLZbGSz2chut5PFYqHc3Fyqqakh41wdykNK2xg9O5uJNMiWHJZXejrV7v43ZEte92yrMdqWgJASQE7mCORklkaC5wfR+uCayqpVq+jRRx+Fy+UaUOWDa0bc3OYEiqqq4qGHHoIoiglZR1MdnJ3U7XbHnB8AlixZAkmS4HA4eu3v5+9AFEV0dHSgtrYWo0aNgqqqkGUZ27dvp88//zymJ3W8S6wnH7ggCHC73Vi3bp3+b65RdnZ0QtM0uDIzUrvvKOMmJbjm+x98gB07dtDEiRNZOByGIAioqqqCpmnIysrqcyxBkiS0tbVhyZIluOaaa2J+5/P5IElSl8yh3qwdWZaxd+9evPHGG7jqqqugKIqupTscjsi7dfV97RxqxVySJLS3tuHG//0Nln32eY/7XpIkqKqKH/7wh2zhwoX02WefxaxtXmD4xz/+EX/7298GNUaU8hpJA0RqIECkIb/0NBbw1lBrwxqI5mwQsR7+TkRYDcJidmF0+XFA/zuM9zm4d++99+oLbiBpRZSwApvNhvHjx8dc8+uvv6Zt27bBZDLpJIp9AsCoUCkrK4txsYTDYXR2duoWVl82E6eNcDqdGD58eMzvvvnmG4RCIZjN5j4LMA4YPp8vxn0HAMXFxcjIyEBnZ6deadzX9ysIAtwed8zPTSZT5PnC4YgbrI9CkN+/0TUJAGXl5TG1OX29d8YYVqxYgauuuiomXjB16jQseneRXmtxOA5VVeFwOfHlF1/go48/opNPOjkl1xNjDPfffz/mzJkTs64VJbLXnn32WVxzzTU0efLkQ+7KSkcUeymIhx/xY5ZdNBfhUCeIlGhGmNGXL+i1KaGwF1ZLJiaOOQtmk5MRDm5rV74Zv/76a1q8eDFsNtuAbkZRFBEIBjBt2jSMGDEiJp62evVqhEKhmOysvsx5MBTC8OHDMXr0aGZ8D4FAgLxeb6/9713OHwyirKwMw4cPjzl/U1NTUsuit9fgsZN4cBmogD5jDCbZFHO/OTk5sFgsUPuhpXItuqqqCqFQKOaeTzj+eAwbNgyBQKDPc8Rdubt27dLXEz/X3LlzI50uvwUxWiLgL488ktJ64tb7zJkz2eWXXw6v1xuzh0RRRCAQwG233TYkni0NIL1wZXGronTspaxs7I8gmzIQDnuhhH1QlAAUJQhF8SMc9gKMobhwOqaMu4DZrNmRIORBrvvgQuvZZ5+F3+/vlzBPJriICGeffbaucfGxdu3a/i/OaBD9qKOO0ikf+Kirq0NLS0u/tXdN0zBq1CjdfcBHc3PzgM1RopTUsBIeEDAXBAGqoqC5JfZ+8/PzkZWVpac59xVATCYT9u/fj+rqajKuq7y8PDZ9+nQdWPpz/t27d6O9vZ2MFtpRM2eioLAQwWDwsM7QUlUVVpsVH334ETZt2kTJ+oAk2ld33HEHcnJyEAqFYlKzHQ4H3n33XSxZsoQ44KQB5DACEYCQU3g0O2L6jWzkET9CftHRyMwei4ysUcjLm4IRI07D5EmXYWT5iUyKki/2dhMYg2593ZxR/z698sorXagSBmKEw2E4HA6ce+65+sIXRRGhUAifffZZjM+/P1Yf743AXSoA8NWWLV20s74Oi6Vrc6zOzs4B0j4pYW0KQ2ynvn7NEQGbNm2K+XlGRgYrLi6OET59tUDcbjd2Rq0EYxJIeXl5vwo9OYDU1dXhq6++ijl/dnY2mzVrFsL9AKihMiRJQigYxJNPPdUr12dJSQn7zW9+g0AgEOOm4m7L2267DYqixLgW0wBymAAJkQZRtCA7byorrTiHjR73IzZ23CWsYtTZrLDwSGYxZxpcOr0ryuPFazwltK+aD2MML774Iqqrq/U02oF0XwWDQRx77LGoqKhgXLAzxvDxxx/Ttm3bYLPZBgRAEgX+t23dOmAuJq/X2+VcA0nJIklSFyFot9th7ef88PUCBmzZcmA+uKJQUVGhP0t/38HGDRsOXC/68/Hjx/d7TXFh+fnnn3dREr53xhnRBqCHd42IqqqQzSa8/sbraGlpIVEUe5w3Pi/XX389xowZA7/fr79HXly4Zs0avPDiC9RfRS0NIIcCQpgAgKf5RrqDxP+7twufC31RFLFr1y76/e9/T1dddRX1RaiIooja2lp64IEHYLPZ9PMO1MEX7ZVXXhnjLgOAf//73zGEePFHKueXJEk/55lnntlFENbW1fX7GWRZBhHpFohxU5eXl+saeG/PK0mSfjDGdAAx1lK4XC42fPhwhENhmEymfj0HCOjs7IgFFQAnnniivhb6czDG8NFHH+nvgL+H+fPnw2QyQVGUPs0TP7fZbMYTTzwBj8dDkiSBCZF9873vnQlXZka/zn8wD6mbveJwONDU0Ih33n23i7u3OzeW3W5nd9xxB8LhcBeGB5PJhD/e/Ud0dnbSIbNC0nUeh/4wuqv2799P1113HWVlZREAuvrqq8mYI96buo9PPvlkUHPczz//fOJFe8brXnjhhQSAJEnq9zWuvfZavT7DOA+33377gDyDIAi0ePFiMmZzaZqGuro6Ki8vH5Br/P73vyfj/PCv//rXvwbsXbz0z5e6PIPH46Gp06YNyPlvvvnmmPfAn+F3v/vdgJz/iiuuSLiWfjmItUuH4vhbtKYm1f3Ma3vmzJlDgiDE1Ia4XC4CQHfeeWfM+jqYR7oS/RAPo0vgxRdfpJtvvhl1dXV6lfOGDRtQVlbW665k/LyffvopNTU1xQSbjedJRHnd07/5OebPn8+4Fm+sUm5sbKSdO3fGBJC7O+JZdjmgulwunHHGGSxeK2OMoa2tjbhW3FfXjKIoGD58OObMmRPDDsC/r6mpoc8//zzm572Zf03TYLPZcM455zDutoi/xsqVK2nX7l0wyaYI7T8TAAb9K2NCJFYi8K+R6AkTmD6/ebl5mD17dsJnqK+vpzVr1ugWXSpCwWhRhsNhlJeXJ5wjPo9Lly4lr9erv59UD+AAbfn3vvc9Zrxv/vtQKIT333+fQuGQbvX3ZfRUnd5vl7amxVXpxMa7NE2D0+nEKaecojM2pOqVEEURn376Kc2fPz/GLczdlTabDZs2bUJJScnBJ1tMWwCH7jBqW9dddx0BIJPJRDk5OV0qmNPzNbgWYCo/OxjXHarPkOxcB5tV4bt48P1/3oLzCAC5XK4uVgj3VBxsWZG2QA7R4BaFx+Ohiy++GO+++y5cLpfupnE4HNi0aROKiop6rVUY00YHM8UvWaDZGAg1Zo1omgZBFBOmFHBAFUUx2gFOA8MBJth4KyieDTa+J0l3Mab4zye7RqLmVMnYabs8NzuQZWX8vM7gG+1Dwq/BYws8gcIYM4k/v/HnRu6qRA2TjIqK8frxnzfygBkZjvl7TsYobDy/0bKIr3MxWhaJ3kNKc2qwnjknWU/rv9csx8kU7cjN9DthQ5ZlfZ5TLQDka2Lr1q101FFHJWRBUBQFq1atwpQpUw5qcWEaQA4heDQ3N9M555yDL7/8EhkZGQiHwzo1yN13343bbrut14vh285s+l2eg8PtudJrceAGlwPXXHMNPfnkkzEUMpzu56yzzsLbb7+dBpDvAni0trbSqaeeirVr1+rgIQgCQqEQiouLsWnTJjgcjl7RNvNzr1ixgl544QXdx2yMM8THNoysuElboibIpNK1QRZV8+K6AHKKi3A4jIaGBtTU1OCUU07BLbfcwozaLF9/fr+f7rzrLnz88UcoHzECxUXFACLaaigURijEW4aGEA6HdFOdMQEmkwyHwwGX04WzzzkbZ5x+RsJ4wN///nf6z3/+g6KiImRnZ+stc42tSOPbkQqiCIvZDJfLhZycHJxwwgk477zzEp7f6/PRiy+8gPUbNqCtrRVerw9hJdILXhQiWV8mswlmsxkWswUmswmyJIExATW1NaitqcWso47C1VdfjfHjx8fEvfg13nvvPXr//ffR0dEBt9sNv9+fkIGYvydJkmAymWC1WuF0OpGVlYUpU6bghz/8IRwORxeW3X379tEzzzyDqupqdLo74fV4EQgEEAoFETaQQRqtN5Msw2yxwGG3IzMzE8OHDcfFP7wYEydMZPExjc7OTvrTn/6EDRs2wGKxID8/H06nE7Is65lWRrZevpbC4TA8Hg8qKytBRPjhD3+IH//4xwnn6JNPPqHly5dD0zR9fvg5jC2Q49T4mLa5ybIF9Z8b2u/G7FEGvZ86D4DzlsydnZ345ptvoGkaLrroIlx33XWMs1r3tMf556qrq2nq1Knw+/0wpgMLggCv14vFixfj5JNPPnggkvYxHlxfu6qq8Pv9mDcvwjaakZHRxZ/59NNP99qfyTfGtm3b9PMMpYOz4P7jmWdislD4Jjvv+wsinxVYv64jyhJt37FDzxjic/jqq68O2LN8/PHHXTKe/H4/5veXvTf67MfMnUNGZl3+DIsXLx6wZzjrrLOIswMbmXDnzJ3b/d+yuCPJ55wuF23ctElnjY1n8U3Eitzb49FHH415D0SEhoYGPYNxqB+//e1ve7XP4zPfjLEQp9NJgiDQUbOOIiPrc5qN91tmhkqShGuuuYY+++wz3fIwahDTp0/H5Zdf3uusK64R/vKXv0RnZ6dOYzFUXBncNbdxwwbgxz+OMctXrV5Fb77+BhwZTmhqco2sJ2tZkiR0tLVjx/btGDtmTMznX3vtNQiCgIyMDL16N9Xz8iHLMtrb27Fz5069xoL7sjdt2kQff/QRnBmulHpWJIpXcE1z27ZtaGhspKLCQmY815IlS8AYQ3Z2NkKhUMpzE18gqaoqPv74Y9TU1FBpaamurXq9XqrcVwmz1QKTLMeQMPbmeWRZRntrGx559BE8s/AfMetQkiP1MU6ns8t7SHUtCYIAn8+Hvz72GK6++mq9b4YoimhubobP50NGRsaQ5NHilgQQITg97bTTaN68eSlZDDwudsMNN+C5556LofLhFCerV63GP//5T7rssssOihWSBpCDDB5PP/00Pf/883C5XDp48A2oqiruuusuyLKsB8564x99//33ifdV7g8D7qD4SqPPVxKlJTEKpeXLv9ApOfpTUatpGkRJRFFRUcymA4CGhgZomoZwONznayTjtQKA9vb2mCB0f4bP50NLczOKCgtjhKDf79eTLPp6jQN/RzpFvkHwM5vNRqFQSE9m6HPsQ2DYsH6Drjxwf31ZaZnu3uzPM5hMJlTu3YutW7fS9OnT9RTgkSNHslGjRtH27dthtVqHBOV5osHdYb/73e9gTBXvaf0pioKcnBx2ww030G9+8xuYzWZ9bjVNgyzLuPvuu7FgwQIyNhkbrJGuRD+IcY/Kykq66aabuhAD8iDYaaedhu9973u91hz4AvnzA38eskFLvpCPnD69i8a6ddvWAendHQqFUFxSgiOOOEIHD36dgQTURIJPGYD+4/yeQ8EQ2jvau2j+AyUMI3T4SowCAwBmsxkOh6OHbn0pvm8AwVAwpmYBiPCa9Yci37hnwuGwzgHGtXCLxYKTTjqpVwrYoVIo7XY7vvzyS7z77rspkSzy59Y0DT/96U9RUVERQ3GiaRqsVit27dqFxx9/HAeD4iQNIAdReN56661ob2/XKTSMgsFkMuH//u//+rQQBUHAl19+SZ9/9jnsdvuQa8DDGEMgEEBxcTFmzJihC0oOkpV79/bKlZRM8KphBZMmToLT6WTxG6cv7pLevuOBmitQV+sAiPT4GJBNLwgIh8Ooq6vrAk7FxcX95p/iJIk1NTWoqa2JmZiSkhLk5OQM2PtIxPo8Z86cAX0ngyqABQEPPfRQyhlr3Mp1OBzs5t/+Vk++McoDi8WChx56CPX19YPOk5UGkIOgaYiiiGXLltGrr74Kh8MRQ+MtSRK8Xi8uvexSTJ8+vc9+y8cff3zIal1cYB1zzDHIzMxkHOC41bC/pgZMFAZkw7syXF2Eh9frpfb2dqRCYpeq9tv1GQcWnHw+v67J82HsjthvkAJitHcuZMaPG9fFQuwLgMiyjM72DqxauUr/GREhKyuLlZWVdeF26otVL4oili5dimAwGEM8euSRR3bZZ0NVNthsNnzxxRf48ssve22FXHbppWzSpEl6X3ojeDc2NuLee+8ddKbeNIAcJBfWb3/72y5aBhegWVlZuP0Pt/c6b55vot27d9Pbb78Nq9U6JDcNf6aTTz45RpgAQH19PdXV1UE2mQbEjaUY3DL8fHV1dWhsauxi+fV1yLLc9WeSzC86IHMW4s9hOF9ubu6AaO18DjZGAcR4zsmTpwzoO//iyy+6uP6mTJmiKxb92VNWqxU7duzAunXryPhs5eXlbOwRRyAYDA55KngOGgsXLuzV3BIRzGYzfnfrrV2sOd65cOHChdixY8egWiFpADkI1seLL75IX375pd7f2KhJBAIB/PrXv8bw4cP7zHf1yCOPwOPxJBRsQ8WFJ4oipkfjH8YugmvXroXP44Xcj8ZQfC4jQdSKLiBVXV0Nv6//DbV43UOi/iEmsxmCKAADQC8vRHmv4kdxcTEGgutI5/rav19/H1wAjR41CkwUBuSdC4KAb77Z2QUs5syZMyDPwdeRkQqeW+GnnnoqNE0b8CZqgyEjzGYzlixZgqamJkrVSuZWyAXnn8+OOuoo+Hy+GMuYezbuuOOOQbVC0gAyiEKTMYaOjg6644479FRD4+L3+/2oqKjAr371K/QWPLj1UVtbSwsXLoTVao0pIEt2GAsIB/MwLnS3242SkhJUVFR0IcvbvWe3ntqYiA69u0Mv7hIFeDweMIHhoosu6iKweDaOEu2v3ptr8OvwIk9VVZGfn68Le/6sebm50FSti0/a+LlU6Ox5lldhYUEMoADQOzO2t7d3OY+UoOAtEdUMn2eeHWVcq0SEsrIyZGRkwNPpjrnfXr+TqNZrtphjlAgiwumnn46ioiJwt2Jf3ge/BhFhx44dXUDlyiuugN1uh8fjSUgeerCP7vajJEmora3Fnj179L3dG8Xs97//fRfXl6IosNvt+M9//oMVK1YMXufCdIHf4BKg3XrrrV2KfnjhDwB68cUX+0SCZqTsPuWUUyJFdKJIjDH9EAShx0MUxZhDkqQeD1mWYw6TydTlMJvNZDabyWQyUUVFBX366acxdOD8/tva2uiM751Bgij0vUhRFKiouIieefYZiudm0jQNoVAIV199NVmt1j4XsAmCQNnZ2fSLX/yC/H5/DE8WL/q76+67yWK1EhMFkkwyyWYTyWYTSSaZBEnstvAOhkK9c849N+E1iAivv/46VVRUpESVLwgCSZJEZrOZLBYLWSwWMplMJIoiZWZm0ptvvhmz9vg1ln76KU2ZMoUkue90/IIk0pgxY2jV6lV6MaHxGh9++CFVVFSQKIp9LxoVRcrNzaVXX3014XO89NJLlJWVpa9Fvh7jj/j1zI9U9kL8/uFHKnuPH2azmc4//3xyu92UiH8tlQLiefPmkcBYDN270+kkxhjNnz+fEnGWpckUh3DMgzGG3bt30/Tp03XN10gc5/V6cfTRR+Pzzz9nffUHc81RVVWsW7eO4tuXdtE+41wjCVutGn6WrB1rMg070d8QEUaNGgW73c4SkQPyf2/evJlq6+rQ0d4Oj8cDf8AfpRVRow26AEEQdWoOi8UCu80Gp9OJ7OxsjBs/HpkZGd3mve/du5f2VVUduEYcDYiRLFGWZZjNZthsNrhcLmRlZaGiogKFhYXd+qh27d5FjY1NsNttEARRd1OEwyGEgiEEggEEA0EEQ8FoTUrkmiaTCXabDQUFBZgyZQqLnx/jv/1+P7Zs2UItLS3wer16lbd+3xYLLBYLLGYzzGYzZFnW3RuKosDr9aKsvAwlxSVJ34mqqtj81WZqamqG290Jr9cLvz9goHo5kAghiRJkkwyL2QK73Q6ny4XcnBxMmjSJ8U6Yia7h8/noq6++QnNzMzo7O+HxeCK0KeEw1Gh1Odfeje/dFn3vWVlZGDt2LHJzc1miPSgIAhoaGqgu2nwsnhQzXolOpFgn+ll//tawaXT6n/z8fIwZM6ZPvk/uJv/444/p5JNPht1uj7FguAfgrbfewtlnnz3gxYVpABnE2McPfvADeu2112KIz4zuq6VLl+LYY489qORnhxJUE4HkQBY6dTePfans7+01BvI9JpuXgbxGsjkZqLnq7lyH23McDJd3IqWvN89/2mmn0ZIlS2JirVzWTJ48GatWrdJ56AZqz6UBZJDAY+nSpXTSSSd16QvOKT0uvvhivPzyy30Gj3gq85T8pqxvxNY90WZzX65Ry+Ian/H33W0eY4V3dzTxRrpzliDYHE9bbtyU3DUQ//uegtrxvvPuNno83XoyQdHd+0l1vpKdryfhwGNlyehUEv2bP1d3ayF+zvX5MlCqx88jxdGlGJuKJaOQNxYn9iQMU6GV4dT6YBHnWDydfjwV/UDsqXhFIZ76PtmzdydzvvjiC5o3b16XCnwucxYuXIgrr7yS8RbBaQAZgloEP+bMmUNr1qyJKezjRUCyLGP9hg0YOWLEwe8gNkDa8FCg6ubz+m233ob6WujP++M+fP4eh0om4eFGRc9B5LzzzqP//ve/MV4PngBSMmwYNm3cCLvdzgbKCklzYQ2wm0YURTz77LO0atWqLq4rHvu48cYbUTFyZJ+sD67RP/zww7Ry1Uo9u0szaNZEBAZAi/seOmEf6bpXl+8Z9M8pioKrrrwKCxYsSEib3djUSA8/9DDWrVsHn88HWZZht9shyzKqq6tht9uRk5PTpXVtPB24JEm6r764uBi///3vY2Im/PNut5v++Mc/Yu3atejo6EAwGNSb9FitVtjtdtgdDjjsdlitVoiiCEVR4Ha7EQqFDtDPx+tMEXc0yEB0l0h7Nf47UftX4/epfC7eaov/jP4z0nrsxtpdK+JkWUHJMoSICIFAIErjHpk3u92OH/zgB10o1Pn3a9eto0f+8hcoaqQVsSybwFhkDQWDIfgDfvh8Pvi8Xvh8fvgDB+JP/JllSYLNZofL5cKso47CLbfcApfLxYzKlyAIWLZsGf31r3+FPxAAE5huPQhM4EaB3v6WMW5p8e+j8xO9f709MBiINLicLtx3333IzMzsQnXf1NRE99xzD9weD8JKGOFQGIoShqKoUDUVmhpphMb33AELx5CJxxgYEyAIDEwQIBi+F6OZa7m5efjTPffA4XCw3rq1brvtNrz33nsxlpKmabBYLNi7Zw8ee+wx3HLLLQPnQkxnTA0sVXt7ezuVl5eTyWQih8OhZ0Q4HA4ymUxUXl5O7e3tZOzd0NvMrscef/ygUU4LkkhfffUVGTM+NE2D2+2mI488clCu+eSTT8ZQvvPnvuqqq7pQxA8ELXj6SP34xz/+0YXK3u120/Cy0t6di0Xo65kokCCJxESBmCjE0MRf+ZOrYrKr+H6ZNn36oD7jsccdS+0dHfoe5evw5X/966DN82WXX97r7Ez+2YsuuqhL5qfdbieLxUJ5eXlUV1cXkxmXpnMfQtbHww8/jMrKyoSB81AohDvvvBMZGRmst7QjxqySO+64HRarBXJcbcmgmMaKGnMNziq8aNEirFu3DpmZmYbmTqxfLgBZluF2u1FdXd3FH9za2kpvvfUWLBYLZFlO+NzJMtAOdzdtKvc/WO4Wfm1JkuB2u/Hkk0/iiiuu0NcuYxHtWRJFmCyRrK9E90u9oIbne+WDDz6A1+slbo3yPTZr1ix8tXkz7M6BpyuRZRnLPl+G//73TVx+2eUxVd7cqnW4nFAVNRI7GeDBLZUXnn8eP/7x/9Dx847vlaciWjqA//73vzFzwylOmpqacN999+Hhhx8eENmRLiQcIPAQBAFVVVX0yCOPJGTb9Xg8mDNnDn70ox/1yXXFBfKDDz+E5qZmSLKsp58OxkFE8Ho8mD59GiZPntwlVrN+/Xo93ZNrasa/j/93Kgen+S4rL+vi0tlbWYm2tjYIghCj/RqPROfitOGH85HoWbt79sG4Ni+QrKurQ2dnJ3Ghqqoq7DYbO3LGDIQCwZhulMneQU/PwtPem5qa9OI6o6vvuOOOHbT3yp/z/Q8+6OIKnDplCqx2G/x+P1RtcOZbMaQv/zFKrpqqcsALUSdOnMguvvhi+Hy+mGC5qqqwWq1YuHAhdu7cOSAUJ2kAGcCA2913352QbZf//t577+1TwJwDVE1tDS18eiFMFvOgM+4KggAQMH/+/IQB69WrV6eU4dIXK27SxEldNk5Lc7MuWNLj0KxxSZLQ0tKC+vr6LkJ94oSJA2oJSZKEUCCIr776KmYPAMARY4+AJEuDsgc0TQMTBSxfvhydnZ3Es6E0TUN5eTk78sgjEQqGBnUdqqoKi92GT5cuxYoVK1ImWeTzT0S45ZZb4HA4Yij7Ocmlx+PBXXfdNSAUJ+ndOAAvm3eke/HFF2G327uw7Xo8Hlx88cWYO3du3wLnUQB67LHH0NbaCtMAEA+mtJEEpgMI30SMMezZs4fWrFkDi8UyYJuYMYZgMIiioiKMizLCGosvW1pa+kypMZB574frSESr0R2lSjwliiAIkGUZgUAgBkD4KB0+fFDue/2GDV3cdBUVFSgoLER84exAAaXFYkHN/v1YsWKFvhe4pn7O2WcDByFDSxQEqIqKv/39771W/DRNw+jRo9lll10Gvz+WA45TnLz66qtYu3ZtvylO0gAyABsTAO644w4Eg8EYcIg07gkjMzMTd999d5/iAkQEKdKqk5555hnIZtOgWx+MMQQDAZSWlWHGkTNiFiYALFq0CF6vt9cpl0YBFg8GPJts8uTJcLlcjIMVn6/XX3+dJynA7Xb3eHg8Hng8Hni9XgSDwZiUxniOq28DuCSbV6MGHQ6HEQwG4fP54PV69Tnq7vB6vfB6vfD5fAgEAtA0LQZA+NwNLy2NuX5/C9Y4OHHKef6eNE2Dy+ViRxxxBNTw4FikvDsmd2MZ3bfnnHMO7E5Hl2Zcg6GYmixmvP32W6iqqqLedIjklsVNN92EzMzMLtT5vL3C7bff3n9LMQ0B/bc+vvjiC3rnnXe69CDgabu33HILysvL+2R98KD1c88/h8aGRjhczkGnbBdFEaqiYt68ebDb7fp98030gWFj9aTtGov4uI+5OwC85JJLDlhA0TRfzmc1Y8YMjB8/Hrm5uXraL6eEUBQFoVAIfr8fXq8XnZ2d6OjoQHt7O9rb29HR0QG32w2/39/lmrIsQ5blmILI+MK4oWhNGFNpFUVJ2q5XkiRYrVZkZmbCZrPph8VigclsgizJMeuSg00oFEIwGNTTeTXS4PP6YLVauyhQ4444AqIc6UmvCypJ7NYSjE91jk9nFiURO77ZAY/XSw5DIF0QBMyeNRsff/jRoIB/xPoW8PEnH4MX3fFrjxwxks2dO5eWfLAYNsfgNW/jQe/Ojk788+V/4pbf3pJyZT13eZWVlbErr7ySHnzwwZikHt4//f3338eHH35IJ598cp8LmtOFhP1caIIg4JxzzqG3334bTqczhkIgGAyirLwcG9avh81m61PxDk+fPXLmDNry1VewHIQ+z5IkwdPpxrvvvqu32OUCwOPx0KRJk1BbW6t3yDNWFXNhFgqFupzX6XQiNzcXxcXFGD58OEpLS1FaWgqn04m2tjaUlJTg/PPPH/A+zn6/Hx0dHdTS0oK6ujpUVVVhz5492LVrF/bu3Yvq6mo0NzfHaJWiKMJkMkESJYAhJuX6UAIGT2+Ob9HrcrlQWFiI4cOHo7y8HGXl5Rg+fDiKi4qQl5eHrKwsOJ1O2Gw2Zjabe625K4oCVVMBinRGTPR+PvjgA9q5cyd27tqFzV9tRnVVFdrb2+HxehEKdNNSWGC6xcTdZXw9KYqCFV98iWnTpjFjNfzST5fS/BPnw+qwQxsEIR7pLaNgzZo1mDJ5MuNuLEmS8MKLL9Dll10+6MqcIAjw+3yYedRRWPnlil7VhHAFrL6+niZPngyPxxPTUI0rt7NmzcLy5cv7zMeXBpB+WB+CIGDd+vU055hjugTOOYnZK6+8gh/84Af9KhrcsWMHTZ0+rXf0A9Hiqt6+X1EU0dnegXHjx2HD+g3M2EaVMQa3200jK0aiuak5Yc9lSZKQnZ2N4uJijBgxAqNHj8aYMWNQUVGB0tJSFBQUwG639wkdUrUMjAV0qWyK1tZW2rdvH7Zv347Nmzdj8+bN2L59O2pqamIEtclk0oVcfzv2JfqaCDS48A4EAjFgUVFRgUmTJmHatGmYOHEiKioqUFRUxBL1KunuPrqby55oW3oaPp+P2tvb0draiubmZjQ1NaGxqQmNjY1obGxAY2Mjmpqa0NzSgtbWVnR0dCDg88dYMJqi4pVXX8EPLvwBUxRF30PBYBCTpkymvXv2wp6iJcBdU6lwj8iyjLaWVvz+D3/AXXfeGXNtt8dDEyZOQH1dna7Q9VfhYRG+l6RAsG3LVpSWlvaqZxCXObfffjvdddddCQub+yuj0gDST/fVxRdfTK+88krMy+Ev5sQTT8THH33MVK1v5iFfLHsr99K48eMR9AcOyrPZ7Hb8979v4uSTYk1bfj/PPvssPfroo9Gq2VyUl5ejoqICo0ePxsiRIzF8+HDk5eWx7vigEgEBd1kNhjsgkauku+t5vV7avXs3Nm7ciNVrVmP9uvXYvXs3WltbB1TrjK8Kj+cFAwCbzYYJEybgmGOOwbHHHovp06ejvLycJeOJin/GREDQF0u4p7/l6d98D6R6DUVR0NHRQU1NTaitrcWOb3Zg+fLl2LJ1K8rLyrDw6YU64y5PHRdFEc+/8AL9z+WXD+peeOmf/8QlP/yhvg/4V26FHIxRUFSIdWvWori4uFfUR/xdtLa20uTJk9HS0qK747jFEQgEMHbsWKxdu5Zxy7I3ayMNIH0BD02FwARs27aNZsyY0aVhDd/AK1aswJQpU/rFtsvdOV+u+JK2bd0G2SSDc79F+nCzqBcgKoC4UDJ8f4DKgR2geACL0jnQAToHxqAoCiZPmoyxY8f26ErqSRsyCpRkDadSEVjJNPVEANSdBt0doBkJCo1uFONoamqiqqoqVFVXoa62Tqch9/l8erOpeDp4Tj9ut9v1r/Yo1YrVao2hW+drh1scPBlgypQpGDt2LEs2v6nObaJ57HH/xxEh9gaAuotxxINnX/fFJ598Qq2trRFrmDQADKRpepGfTtVDZKD70fROwZHPRkx2TTtAQRIKhTBu3DjMnTs3KeX96jVrqKO9HUxg0DSKnFeLriWK/psIpJFOK8Q/k/j3Wuxno/vr+OOPR2lpaZ9cuzyG8+c//5luuummLlYIJ1r829/+hmuvvbbXRItpAOmH9fHjK66g5559Nual8Bdyyy234J577mGKqkI6DMn+ugOHeKvEKByMAqEnRtpEgqU/QiXV54rX0JMx0xpBZSi0RuVV0d2BRTzVxGDOaaK4UCIOrr4AjTGBortnHewMumTXOFzIFvl8er1emjJ1KvZXV8NsNscoS6FQCEVFRdi0aROcTmevYrVpAOnDpmGMYfv27TRjxoyY3/GXUVhYiM2bN/f6ZfQEWoP2rgxaprEtbl83V7xQMNJvpyrIVFWFz+cjnnLKU0n9fr+eFaQ3ggJBYIIe+DabzTq5osPhgNPphMPh0MnpeprjnkAlWewgEUV5d7GFZD+LtxC6ex9GId7TewuHw/B6vRSfmsvTnI3vSZKkLlaU1WrlX2NiY71Zu8no8fu7Lwaj73dPLtW+ZmDp9PEpjlT3Y09WyJNPPknXXHNNUivk7rvvxm233dYrj0kaQPpofVx66aX00ksvxVofoohOtxsvvPACLr300u9Eo6hkGm93zx0MBtHS0kINDQ2ora3Vj7q6OjQ0NKClpQXt7e26e4gDBqevSHXz8/oSq9UKp9OJzMxM5Ofno6SkRI/bVFRUoLy8HAUFBUldRP3dwAM93xw0EllFHq+HqquqsWfPHuzevRt79+7F/v37UV9fj9bW1pg55QCcLKvPyJYsy7I+l3Z7hDE3MzMTOTk5yM/PR0FBAQoLC1FUVISCggLk5eUhOzu7x4SJRAzNAwUu6RG7R6Op8LRjxw5YLJYYhUFVVTidTmzevBkFBQUpx1rSANIH8Ni4cSPNnj07JiDFA+fz58/Hhx9+yDgtx7dpAcZ/351gJSI0NjZSdXU1du/ejZ07d2L37t3Yt28famtr0dzcrNOsJ9O6jJXQfYmhGAPSRo4v4xBFETk5OSgrK8OECRMwY8YMzJw5ExMmTIgRfjydujfB4YEGDX6/xrFnzx7asGEDVq9ejY0bN2LXrl2or6+Hz+frch4+n6IoRmJmcVZWsiZP8XPJ5zPZkGUZDocDWVlZyMvL09O2y8rKUFZWhuHDh6OoqAi5ubnMWFPSHbik2tArPbqXXa+88gpdfPHFMSUHRivkhhtuwIMPPpiy8psGkD68hO9///v0xhtvdDEFiQirVq3CpEmTDrn1wTd6/KZLJYXUqAn2pIUoioKGhgbat28fdu3ahe3bt2PHjh3Ys2cPamtrE2YtGaul411F8WCVyK3TWzdE/DMlq1sxMs+WlZVh5syZOPnkk3H88cdj5MiRLN51Mdjv11h7wIfb7aYVK1bgww8/xLJly7B9+3Z0dHTECAKTyRQDdInmtLfzmQhoEglzY2sDTqSYCMhcLhdycnJQVFSk166MGDECI0aMQFlZGYqKirq1Xnqqy+mOmTkVBejbOPicddfsTpIkbNi4MeVmd2kA6SV4LP/iCzo+rm0kR++bbroJ991335AAj4HcGB6Ph9ra2tDQ2ID91ftRWVmJ3bt3Y/fu3aiqqkJ9fT3a2tpiXCGMsYR1Ez3FERIJgVSzqZJpsd0BUyKwVFU1hv4kIyMDs2bNwrnnnouzzjoLw4YNY0ZhOdDvmmdz8fsJBAJYunQpvfHGG/j444+xd+9e/bNWqxWSJPVYQd9fDT6VeUz27hKBdncA43A4kJ+fj7KyMowePRpjx47FmDFjUF5ejqKiImRlZbGBWt/GItnvigz74IMP6PTTT09qhfz4xz/GM888k5IcSwNIL4XySSedRJ988oneuF4PnBcVYtPGTXC5XOxQ+m/5fX69fTu98/bb2L59O5qbmyMU1NEFIcsyLBYLrDYbbIZ0UlEUEQ6H4Xa70dHRgba2NrS2tqK1tVXnoIrf8IIgxABFb0AimVvKKGSM7qf+DB4T4dZP/L3GxwGMgfRQKKQX8uXm5uL000/HFVdcgeOPP57Fb86BFGbbt2+nf/7zn3j99dfx9ddf65vcYrHoRZzJsqCMz2cU1v1lMeBxkXiyxUQur74AjJHyJn6tWSwW3S1WUFCA3NxcZGVlweFwQJIknfLG7/frhx7r0VSIggibzYbs7GyMHj0aZ5xxBsaPH68rA98FEOHy4bTTTqMlS5bocix+Ha5avRqTJ03q0RWfBpBeIPe7775LZ511Vgxyc9R+4okn8NOf/nRAG9b39T7/+thf6ab/vSmmerm/AiMR+WCqWUmJhISRvynR38qyDJvNBofDAZfLhYyMDLhcLrhcLjidTr2Wwmw26/dGcZxYHo8HHZ0daG+LVEPz4Lzb7Y4RpKIo6udJBChGMAkGgwgGgxAEAccddxyuv/56LFiwgPHn4qDa201t/LvPPvuM/v73v+O9996D2+2GKIqwWq06x1EywDDSnBg/Y7FY9IB3bm4ucnJykJWdjQyXCw6HI8KJZaAniecVc7vdaO/oQHtbWwy3mNfrTbjGOHNvX61P/vn4vx1IIOTW27XXXot7772X8T37bQcRLiNWrVpFxx57bBdmby7PFixYgNdff71HKyQNICmY7nzxzp49mzZt2qS7rwRBgN/vx+Qpk7Fy5UomidIhyx7hL/rNN9+kBQsWwGq16gy3qfq2++Ky6EmT5CARP2w2m54VVVRUhGHDhmH48OEoKSlBUVER8vPzkZOTg4yMDNjt9l6ljSZ7HrfbTS0tLaitrcWePXvw9ddfY+vWrdi+fTuqqqp0YWgU2PFV4UZh7fF4AABz587FzTffjDPPPJP1xi0S7wL78MMP6aGHHsKSJUugaRqsVqvefTEZoIXD4RiCyOzsbIwcORITJkzAxIkTMXbsWJSXl6OwsBCZmZmstwzKie7Z5/NRR0cHWlpa0NDQgJqaGlRVVelHTU0NGhsb0d7e3uXdcws4mcWaimssVVdcT/ERVVXh9XqxYMECvPLKK8yYsPFdAJFLLrmEXn755YTdU/1+Pz777DPMmTOnWxBJA0iKk/3MM8/QlVdemZCyZNGiRTjjjDMOWeyDb77Ozk6aMmUK6uvrYTYPTNOpZIVhRmBN5G7gvuycnBwUFxejtLQUI0eOxIgRI1BeXo5hw4YhvyAfWZmp+bPj/fu9afPaU/2J3+/Hrl27aM2aNfj888+xYsUK7Nq1SxfuVqtVDzLG850BgNvtBgCcffbZuPPOOzF16lTWk1vL+LtVq1bRH//4RyxatAhEBIfDkfB6XMAZXWpOpxOTJ0/G3LlzMWfuXEyZPBmlpaXdBp85GKW691OdRz7C4TCaW1qotqYG+/btw+7du7Fr1y7s2bMH1dXVaGxsREdHR0IeNX4kWmvxis1ArG1ZltHe3o477rgDt99++3ci9Z4n1+zatYumT5+u/9uYUerxeGIySpPt0TSApCCYvV4vTZ06FdWGKk4+yaeccgo++OCDg77wjAVU4XAYFosFDzzwAP3v//4vMjIyEmp+STdSZDd1eW5j/CERQAiCAKfTiezsbBQVFaG0rAwjR4zQgWL48OEoKCiAy+XqsYAvXsPuT+C8p/cZn4kTPzc+n4/WrFmDRYsW4b333sPWrVt1V5DJZOriRjICic1mwy9/+UvceuutOhV+IjeOIAior6+nO++8E8888wxCoZAOHEbgT2TxZGVlYc6cOTjzzDNx4oknYvTo0Sz+ORPRnAzkPCayHPh1ultrgUBAT+/eu3cvdu3alTC9O5HlHB976SmtO1lv9vi1pmkaTCYTtmzZgqKiIn0vc+bfb6NFwp/xxhtvpIceeigp0eJ7772H008/Pal8SwNICpN8zz330K233hozydwfvnz5csyaNeugAkiigJ/b7aYJEyagoaEhYUvdRH0wuhuCIMBiscBut+v+88LCQpSUlOg07MOHD0dxcTHy8vK6rfI2CrTeuiEOpqLANTFjDCsQCODTTz+lf/7zn1i0aBHa2tpgMpn0bozxQBKtoMfEiRPx8MMP46STTtLjI1wgAcALL7xAv/vd71BTUwO73Y74tqVcEHMXlSiKmDVrFi666CKcedaZGFE+ghk1Sp65lSoD8WDPZyKASQTWxuH1enVCxerqalRVVaG6uho1NTWor69HS0sLOjo64PF4EAgE+kRqKQgCrFZrQr//z3/+c/z1r3/9ThSY8LXe1NREU6ZMQVtbW5e6Nq/Xi6OOOgrLly9nydZVGkB6mOBEfPp8wV144YV49dVXDyp4cHNy0aJF9NZbb4GIkJGRgbVr12L58uUx6cX88yaTCQ//5S8oLirSF4iqqggrCpRwWHdBqaoKWZaRmZmJ7OxsZGVlISsrCxkZGd0WfBldI9+GimKju8wYM9i9ezc999xzeO6557B//34dSIyCjAt+bi3ccMMNuPvuu3Wa9ZaWFrr++uvxr3/9K+Hfc4HGgcPhcOCcc87BT37yE8ybN4/FW22HWy1Dd1xdPT1HMBhEZ2cnJWoUFgwGdRDltTAmk0mPt2gUIUp84YUX8M4778Bms3Vx8TLGcMEFF8DhcKCzsxMjR47EL3/5S2RkZPSqF8fhpiA//PDDdMMNN/SN7j3+haYP0rN5iAg///nPCQC5XC6y2Wz6YbFYaPPmzcSziQ7mPb300kuECJtOzOFwOGLu0eVyEQD68Y9/TP29trFLHT+MwGOsUh6Mg7vSejoG47rhKMjyuWhoaKC7776bioqKCADZbDZyOp0xc+9wOMjhcBAAmjlzJm3bto02bNhAY8aM0ddT/PtyOp1kt9v1d/nTn/6UtmzZQsZ3wCldvi3zG38PPOkifp1xa6+/x9atW8lsNsfMu/GI31Pf+973iN/bt03G8efy+XwYP348SZIUsyYdDgdJkkQTJ02kQCCgv/8Y0su0BZLc+khEmMitj8svvxzPPffcQbM++HsKBAKYMmUKVVZWwm6369ZGfKaO8e/WrFmDI444IqYpTnd+4kRB84HWvnrK8oq3YFK9fqKNMlAWEZ9jbpXU1tbS/fffj6eeegp+vx8ul6uLW4uvl9zcXBARWltb4XQ6E1otnZ2dkCQJF110EW6++WZMnDiRAYhJ2hgord+o+Q/E/CaLXw3kukkWSE8lJZhr3GazGeeeey699dZbXQrp+Bwb04jb29vx+uuvY8GCBd/KADt/pjfeeIO+//3vJy0ufOqpp/CTn/ykS5lCGkC6mdQLL7yQ/v3vf3cx7QBg3bp1es+Mg+FC4C/u6aefpquvvjrhPSUCuosvvhgvv/xyQvA4FH7wVN0V8SNqBZDR4jESHUYJ/1iydqvJ3G29FaDGmA4HkrVr19JNN92EpUuXwm6364FZo1DiSQ284M34nnhtybHHHou77r4Lx887XgeO3sY04vmjUp1rbk0qikLxrshoHRDjGVJ9cQcmSpI42DEwVVUhSRI+/vhjOuWUU2Cz2bpNc+fprFOmTMHKlSsZD95/W+XdySefTB9//HFMcSFvzV1aWoqNGzd2ac2dBpAkk7l8+XI6/vjjE1KWXHnllVi4cOEhsT6mT59Ou3btimHTTKZ5hcNhrFixAtOnTx8UAOmvP7u9vZ1aW1sjrU4bGw+0OG1uRlu0aM3tdusFa8FgsAuDLL+ekS2WM+9GuJYKUVIyDGVlZSgtLUVJSQmys7NZIoDuLfOuEUiICPfccw/deeedICLYbLYuVobxXRqtjry8PNx+++247rrrGG/q1Rvg4AI6WZC6ra2NeK2GMSjd3Nwc6Vnu8cDv9+v0LUYrip+Tzy+ndnc6nXC5XMjKytILFPPy8pCXl6cXK2ZmZvZIoc/3nLEPyMGwfI899lhauXJlDB9UosHjAC+//DIuvvjiQ1ooPNgyb926dXRMgvbcXO498MADuPHGG2PmIA0gCTajIAiYP38+LV26tEupP2MM69atw5gxYw669bFw4UL6yU9+krL1cd555+GNN97QU0kHyk2QSsEVp2yvr6/H/v379SKz6upq1NbWorGpCW2trXC73QgEAj1SivPr8edgSdKOjf7q+GG1WnXm3fHjx2P69Ok48sgjMW7cuBhBx8+RKphwl6cgCFi6LdCQaAAAlWVJREFUdCldddVV2LNnT9L3JAiC3nVwwYIFeOCBBzBixAjG7z8VoDcSLRrnoqOjg77++musX78eGzZswLZt21BdXa3T2SQTkt2lxSZi4+1u8Oy9jIwM5OTkoKCgAMXFxXqx6LBhw/Ri0Z6KG5OleCdbez0BjqIokGUZL7/8Ml1yySUJ3ViJrJBJkyZh9erVLH6+v20gcs0119CTTz7ZJeNUURTk5ORg8+bNugLGGEsDSKJJfPudd+ics89OSFlyxRVX4B//+MdBtz6CwSCmT59OO3fuTMn6CAaD+Pzzz3H00Uf3aH0YBVdvgKa9vZ2amppQEy0Yq6ysRGVlpQ4SvOVrPGW7weWUkO4ingfL6HLqzu/OteVE9QJGmhMj1YfJZEJpaSmOPPJIzJ8/H8cff3xMXUVv4g9cODU0NNAVV1yB9957D3a7vQv48pqP++67D1dddRXjLqSeNFtjqrHxfrZu3UpLly7F0qVLsX79euzfvz9m85vNZp2hl9eUcOvJOMfGGolEsSMOqMYjnneLn9cYEE+0Vm02G1wuF3Jzc1FcXIyysrKY/izDhg1Dbm4u6622ryhKt8oNt3ICgQCmTZ9Ou1Ow5rkV8u9//xvnn3/+t9IK4euqsrKSZh99NDo7OvSsU6P8451W9dYGaQDpqmXNnj2bNm7c2CUl9lDGPv7xj3/QVVdd1aP1wVNITzvtNLz33ns9Wh/xz+H3+9HZ2Ukej0fvBOj2eNDU2Ijq6mr9qK2tRWNjI9ra2uD1ersEZ40plFxoGfmvkvEZcV4qi8Wi9w23Wq2wWCx6//D4c3L+J7/fj/gOhvEFlbzLnizLuiXg9/t1kMvIyMCMGTNwzjnn4KyzzkJ5eTkzuqt6sry4n52I8Itf/IKeeOIJXUAZNbmPPvoI48aNY6m4q/hzGoXWjh076L///S/eeecdbNy4EV6vFwD0ueP3wOcmHsQ5sPB5jp9fI90Hf2c8VsMPnhGWbB0aK8uNwWm+DoxrwThMJhMyMzNRUFCgU72Xl5frdUe5ubnIyMjQaV54urXL5dIpb7qTaxzoH3roIbrxxhtT2lNerxezZ8/GsmXLGFcEvq0K9H333Ue//e1vu1ghqqrCarVi8+bNKCkpYeksrJhFpUKSRLz40ot02aWXJbQ+DlXmVW+sD0EQ4PP58NFHH+GEE07o1vrg4LFhwwZauHAhtmzZgvr6enR0dOhClcccErnJOAsvFw5GgDD21zC6kFwuF7Kzs5GXl4fCwsKYg3exy8zM1AkTo9XfKWmiURp28vl86OzsREtLC+rr61FVVYU9e/boVc/79++P6aHBBSi33LgwzsnJwcknn4z/+Z//wSmnnMKMZIPdvX8uoFpaWmjUqFEIBAL6HHm9XkydOhXr169nxuK/7tYA/0wwGMS7775Lzz//PD799FO43W4wxmC323VSvGAwGNNMym63o6ioSO/AyBkCCgsLdUFss9n0eY4nzIwjviR+fq/Xi87OTrS3t6OlpQVNTU1oaGjQj8bGRp280uPxJORDk2VZByyjdcRTpxMBFC9wtdlsuoZsABCcdNJJ+NOf/gSbzZa0doPPaUtLC02aNKlLEV2yfeX1evHhhx9i/vz538qMLD6XgUCAjjrqKMR3LuRy8H//939x//33R5SfNIAcmDguqBMFqXk67Pjx44e89XHCCSfg448/7tb64JuooaGBpk+fjrq6Ot0FxDc0F3rxjKi8LiIeWEwmE1wZGcjLzdUJEuO70HG/t9ls7tM7SqRd9iY9NxQKYf/+/bR161asXr0aK1euxObNm9HY2Kg/g81mgyAICAQC8Pl8YIxh9uzZuPbaa3HRRRcxrvkmux6Po1VVVdG0adP0SnLGGHw+HyZMmIB169b1eB6+zhRFwTPPPEOPP/44Nm/erAMDp9Xx+Xy6hZGbm4uJEyfiqKOO0jsrlpaW9thaNpny0h9/v9vtptbWVjQ2NqK2thb79+/XK8xramr09sW8EDD+nfJiwHiL01h7FX+/gUBATzntzi3IQf6mm26iP//5zyntLbfbjbPPPhtvvfUWG+ieO0POjf/223TOOefEKNLcgs7IyMDWrVuRm5vL0gBiENSPPfYYXX/99bF9zqOo+6Mf/QgvvvjikLc+PB6Pzl/TnabMYx5r166lmTNnIjs7O8Z3HQqFulzL2KqU9xYvLS1FeXm5DhKFhYXIycnpESDiK9eNwipRBk6qbWzjv0+FSqO2tpZWrlyJ999/H5988gn27Nmj++l5pbjb7QYRYfr06bjvvvtw0kknJQVoPrfV1dU0ZcqULgAyceJErF+/nomimBRA+M/3799Pl1xyCZYvXw5RFOF0OvXALg+KV1RU4Pjjj8dpp52G2bNn6w2v4gVDorqYRHMbn6CQbJ67q99JhXQxEAigtbWVOKPv/v379Vgad5NygEnUitgYP+MV6B0dHXjooYfw61//ulsA4QCwZ88emjZtmp75lkqsYPXq1Zg4ceJBUyQPFYiccsop9NFHH8UkEnF5+Le//Q3XXnttGkCMTLaTJ09OyCWlaipWr1qNSSk0WDmU1ofX68XRRx+Nzz//PCXqBSJCKBTCBRdcQO+++64et8jMzERRURFGjBiBioqKGJr1goIC5OTk9Jie2R21SX8124F43/yITxzo6OigpUuX4pVXXsGSJUvQ1tamxwmiiQOwWq1Yv349xo4dm1ATHQgA4bGUq6++mp5++mnk5uZCURR4PB4oioLc3FycdtppuPDCCzFv3rwYwsp4bqxDRScTbzX2NtXb7XZTQ0ODDix79+7VrZempia0tbXFWF8AMHbsWLzxxhsoKipiPTWJ4nN82WWX0YsvvphyduN1112Hxx9//FvL3Muf6/PPP6cTTzwxJhbMXXnHHXccli5dmgYQPll33HEH3XnnnQmtD16Mdyisj2nTpqVU98FN7DfeeAPnnXceSzWrRxAEhMNhrF27lhhjyMrK4nn8rKe/7Q9BYry1MFA03X2pojdmHxnnbPfu3bRw4UL84x//QFNTE6xWK2w2G1paWvDpp59i3rx5Ca28gQSQ888/n15//XXIsoxwOIzx48fj8ssvx0UXXRRD287XbKp9SBIJ9v7Md2/nPJGV2BuA0TQNXq+XvF4vgsGgzhBQXFzMuLsrlb0vSRJWr15Nc+fO7dJcKdGzKooCp9OJLVu2oKCggH1bOxlyxWjevHm0bNmyGCuEz/XKlSu/21xYPH2xsrKSsrKyyGq1kt1uj+HGMZvNtHHjxoPKecV5l55++umEPFzxh8PhIFEUacaMGWQstEv1SIX7KhxHutgbXqP48xyseTRyiMXzKvWG+6qyspJuuOEGysvLIwB04oknEm8RnOhc/PmqqqooKyuLLBYL2e12cjgcJAgCTZ48mfhnkt0Lz1JbuXIljRs3jmbOnEnPPvsseb1eMq6T7p7H6I7kz57sfQ8Gz1I8r1V/1lBv1k9v1j8/16mnnkqMsS6cZvEH55e7//77ybhXv20Hf67HHnusiwxyOp0EgB599FH6Tlsg3KJIZMJyjX7BggX497//wxQlnLypSh81ECPnTqJgYKpV5/HVsqlYH8k08FRdHt1pj6n0UOAaZGdnJzo6OsAZVjs6OtDZ2alXoPMK6VAoFNPjwtjb3W63w+l0IiMjA9nZ2XoldE5ODrKyslgiq5EDS3cU6PHcV5WVlbRs2TKcddZZyMzMZN3FL/prgcRrvcY5DYfDCefYSGPSHXV6Z2cntbS0oKWlBc3NzYae953wen0IBAIxGVDx8221WmG32+FwOJCRkYGMDBcyMjKj32fwDDqWyrrrqxWbKN4VbwWlKgMkScKiRYvozDPPTKmwMBAIYNSoUdiwYYMe60uW7dVdLCmV36Vi+Q2WBcIYw/r16+mYY46B2WyOoXp3u924+OKLIX3XwWPlypX0r3/9C3a7PSH1xPnnnw9BiGSEHKz7kiQJL730Em3fvr1Hv6yxUvb73/9+n2I0yZoAdUdK2NM1Ojs7qbm5GQ0NDXoGTk1NDerq6tDQ0KDTaLjd7hgajYEYkiTpKcN5eXlUUlKCiooKjBs3DuPGjcPo0aNRXFwcUwGdqPqcf8+BpLy8nJWXl8dssIO1Hjjg8SLMZO43IxC63W7avXs3vv76a2zbtg3ffPMNqqqq0NDQkDB+0J/BwYXTnGRlZVFubi4KCgpQVFSEkpISlJSUoLi4OCaO1p2ik2ocrT/vgYP4qaeeyqZPnx7TsjrZPVmtVmzfvh1vv/02XXjhhUkLC3ubBDKUBl9HfO0ZOxZyGbNjx47vLoDwF3rPPfdAUZQu3EW8j8bjjz+OxsZGcjgcB6qbRRFiVLjwIKwgCMl/nuDfgiDAZDIhOztbz5rhQV2/348HH3xQL5Tq6UUrioJf/epXMJlMSNX6MBLuJWJp7akqPRQKoaWlherq6lBdXa1XoRt7Yre1telB30RC3niYTKYYmpJkFk58emkii4lrtu3t7WhsbMTGjRtjBEZOTg5GjRpF06ZNw5w5c3DUUUehoqJCb63LwcSYxsybRfHNc7AEAgcwfu/x788IGm63m9avX48vvvgCK1euxNatW1FbW6u3v+Xn4Kmx3HJLZlkmiy0Zv8a7jjweDzraO1BZWZnwvcuyzAEG+fn5VFJSoqd6l5WV6TQnOTk5zLgmkoFrvHITf8+pWCScz+y6667DVVddFTPn3b2Xxx9/HBdccEHSe/T5fGQsutRdeKoCVVFjaHeStWtOtM7j2QHiv0/2NRnDdfy75K69trY2/P73v084f6Iooqmp6btZB8IDRFu3bqUjjzwyecP4qMthoPovxy8EURRhs9kwadIkvPnmm8jMzGSCIODJJ5+ka665JiXrIxAIoKKiokdzOv7ZUxGAbrdb7xBXVVWFvXv3orKyUm8/2tTUhI6Oji45/NzdIctyTD2JUdD0VJFuPFc8LYlR8zb2qkg058bGQrzrH+8pzv8mMzMTEydOxCmnnIIzzzwT06ZNiwlO95YRdyBdWImEndFKam9vp6VLl+Kdd97BsmXLsHfvXv25eHU5Vyj4s3NB1t26StSGNz7tOpn1Z6w+Nyoi8e8+Ec2J2WxGRkaGTnHCK9F5mjjvgJmZmclStbR7mmOuSHm8XpoyebLeLKynwkK/349PPvkExx13XExKt6IouPDCC2ndunW8ME+vvE8We+zJlZXImkmWMJLs3z0lmFDkJvR3HQgEEA6Hu9Dx8Dl1uVzfTQuEC9F33nkHwWAwqaAmIr3daF/8lsk+Y+SXaWtrgyzLsNvtLFpFTg899FCvrI9f/OIXsFqtPVofxpaq27dvp46ODh2E6urqdNJD7m7irg63291FQHOA4DEIo/bONwtfgPH3zBlzEzG58k6IUSZX2Gw2WK1WncuJ378xQOzz+eB2u9He3o7m5mbU19ejtrYWNTU1ujXU3t4e4+qx2Wy6WzIYDOLLL7/E8uXL8ac//QkzZ86kCy+8EOeddx6Ki4v7TK0+GC5X/n5XrFhBL7/8MhYtWoS9e/fqwtflckEURX3+jRX3FosF2dnZKCwsQFFRMYqKivSK9KysLLhcLr1AkQOuMV7Egdfr88Lj9uhxq9bWVvCYSktLC9ra2mJazyayQjiljNGC4gqB1+tFe3s7tm3b1mXt2Gw2ZGZmIjc3lwoLC1FcXKy7xoqKipCbm6s/v8PhwLhx4xgvuOyOgFFRFDgdDnbllVfSbbfdBqvV2q3yxqk9/vKXv+C4447rIuC//vprVFdXw2KxdGEZ7i4+1VMspVeyhghaN3U8SZ8NDGCRolo+d0k/+120QPhmvOiii+jVV1/tMXA2GEOSJLjdbkycNAnLPv8cLpeLMcbwxBNP0LXXXpuS9REMBjF8+HBs3LgRDoej25RCrhVv3LiRrrvuOnz11VcIBoP6RkjUWMdINWEkJTRm2CSqSLdarcjMzER+fr6uRfI+6iUlJUYajV5XpPd2BAIB1NfX0969e7F161Zs2LABmzZtwq5du3ThyqvPJUlCKBSCx+OBpmkoLCzEeeedh5/85Ce6VZIKLf5AWyDGOMgbb7xBf//737Fs2TKEQiGYzWa9B0kwGNRb6UqShGHDhmH8+PGYNm0aJk2ahDFjxmDYsGHIyclhgwmEHo+HOjra0dzcooM551Dbv38/6urq0NTUhPb29hjaFb6u4ylOjOvOmCWXzDXKP88Yw5QpU/Dee+8hNze3x/0hCALq6+tp0qRJ8Hg8PdKb8HezatUqTJ48mRndinfeeSfdcccdyMzMjFGiDra87Y+rtbt71TQNNpvtu+3COvXUU2nJkiUHHUAkSYLP50N+fj6WLVuGkSNHMlVVEQgEaOrUqdi3b1+PyB/P0Z+K9UFEOP7442n58uXIzs5OGJw0AkQysjtBEPSK9MLCQgwbNgwjRozAiBEjdNK7goICZGdn98hhlYxltzcByGS+42S1BJqmYc+ePbR69WosXboUy5cvx86dO6GqKkwmk25R+Xw++P1+2Gw2nHvuubj11lsxfvx41pPQH0gA4RlVn3zyCd1+++1Yvnw5GGNwOp2QZRmhUAhutxsAkJ2djSOPPBInnngijj32WEyYMAGZmZkslfhBb4VNd4zIqVCqt7W16RXovEhw3759qK6uRn19PVpaWhJyaHErLN49aly7xnXa0tKCJUuW4OSTT+6RlZoD9fXXX0+PPfZYyoWFRo48Pgf79u2jyZMn65brt03OEhFMJlMaQLoDkL50zktldHZ2wuVyYcmSJZg1axYLBoMwm814/PHH6ec//3mPC5c3i8rPz8fmzZuRmZnZrXbFhVAgEMCYMWOourq6R4Cz2Wx6P4fCwkIYg52lpaV6Nk1GRgbrSUNL1lq2vxpSqgs9podzlPIiPti5evVqvPPOO3jvvfewfft2AIDL5YoR0g6HAw8++CCuvvrqbnnGBgpA+HnuvfdeuuWWWyAIgu6i4szDTqcTc+fOxXnnnYeTTz5ZZw82zj+/xsGadyICgQBCnzL4vF4vNTc369l7vJdMvPXCG431JMM2btyIKVOm9MhMzeXC119/TTNmzEi5GDJKSIpRo0YxI8PBrbfeSvfccw8yMzN7dEcfTMHfW1mZjM5GkqTvtgvr+9//Pr3xxhsJAYSX7A/G/EyaNAlPPfUUZs+erWtFXq+Xpk6diqqqqpStj7vvvhu33XZbSnUfXBi99dZb9J///Edf+BaLBU6nE9nZ2cjPz9cPHo/IyMjoNljZ34r0ZAu7Rz9tN/xNvQEVPp9G4fXRRx/hH//4BxYvXoxQKKQDSSAQgN/vx6pVqzBjxoxB58ISRRH79++n8ePH6zEOn88Hn8+H7OxsXHrppfjJT36CCRMmJKQxSTVZIn6+U13zfU1T7W+7Y6/XSzze1dTUpB/Nzc16vI67ZseMGY2f/eznrDfdHUVRxAUXXED/+c9/UrZCfvazn+Gxxx7T1wQnZz3vvPNo8eLFh62s5AkoidaEIAjfbQC58sor6ZlnnumySPhGnzFjBk455RQUFRXpWRnctaNXu5IWTV+I21z65o1oYoIgwG63o6KiAscedxwzGXoZiKKIRx99lH75y1+mZH0oioLMrEx8tfmrHn27A+ET7c6KSKXYK1mmSXxgsb+aldEd1tv7NPaV4OOLL76ghx9+GG+//baejeL1evHRRx9h/vz5SV0iAwEgHODr6upozJgxemwjKysLl1xyCX75y19i1KhRzGhldCd8E9X06EJggOY/UaZWX6llkqVu9yWFujcyjruxEvFAJdtPqqrCYrFg8+bNGDZsGDMGzVVVxaL33qOmxsakVfKqpkHTVGiq1qXzo34QgZKwSPT4M9JAGiX/dxIWACLCV199hebm5i4gwi2Q72QWFp8Im93W5XeclPDcc8/Fa6+9xgar85jRreB2u+kvf/kLZFmGqnUfi+H3d9WVVyEvL6/XVeecuiEZ42qiDZ+qUOqtqyJ+PqI58xRPO2KkkzZ2MzSbzbwpEuM+8e7Ob6x9MWrnxnvln2OMYc6cOWzOnDlYvnw5Pfzww1ixYgWuvPJKzJs3b9BJNXl/laKiIvbEE0/Q22+/jfHjx+NHP/oRKioqumSGxb8jozBP5LaLH+FwGH6/n/x+f0z/eSNtunHueYaOxWKBxWJhZrM5ZZdvKm7NnmJMiYCmOysh1cFB/dhjj2Vz586lZcuWdds3nccC2tvb8fjjj+Pee+/VwZwrh2efddZhS5a1evVqOuGEExLO8Xe2pS2vHP3Vr35FjzzySIzWz2MFK1aswIwZM1gwGBxQQWEUXvw+Hn74YbrhhhtSsj5UVcX/t/fe4VVV2fv4e26v6b2QAAk1gUSk2MGGwjDYwS52nBHrOI7jqJ/xqz/HMgojNkYRC+qoY1dEQEGkiCKEHlJJ78lNbi/794esM+ece25LggY563nOk3LvaXuvvdZe7V1msxllZWXIzMw8on0JBupq8Pv9sNlsrLOzE+3t7WhtbeUbDpHLoaurCzabDX19fXA4HLzwEubNC+MXJOxJkFH1c1xcHBITE5Gamsr336Z4TU5ODhITE2UhzsPt3KUFfN3d3YyC0pEyVAYzC0v6eaiUYlJ+cmmifr8fTU1NTFjPU1dXh6amJrS3t4vSbmn85QQ9jRUVf+r1ehiNRlgsFsTFxyPx5xRbpKWlISMjg08VTktPQ0ryz5l30fDNrxk3o14h7733Hrv44osjJtkI+2Ts2rULqampPNR7tPUzR5JiGS/pd1UqFSZNmsR++uknmM1mkfLW6/XHbiU6Map08DweD5KTk5GXl/e/TIMjwLC0O+np6WGLFy+GTqeLmAlG1sctt9yCrKwsLpqU0mhcBdLfo1UQbreb7+cgTdWklrcdHR2w2WxwOBwhlaOw8l1YxCZ1cdHzUXaY3W4Xmdxyrgaj0UjzycaOHYvS0lKUlpZi3LhxoviOECpECGUiFNgJCQkRA7FHSqAJ3VrSinRqdyusk2lvb2dlZWX44YcfsH37duzfvx8NDQ3o6uoKjvdxHNSCroA0/lqtVhZvS1gXYrPZRAWdofiWkjKofiMnJ4dP7xYWCIbCLotGwURjvcRihcyZM4crKipi+/fvD4tHR8K0tbUVL730Eu6//35R++OjFfKdxthqtcp6LY5ZF5bQbJcbtMNm+RFNvyNf60svvYTa2tqoM6/i4+Nx6623Itq4h9AlI7WCIglCr9eLzs5O1traioaGBhw6dAi1tbV8RgylW9pstqBqdHIdUD6/xWKRbZUqVQDCat1oLDkSqMJCQxIidN329nY0NDTgu+++458rJycHJSUlbMaMGZgxYwaKi4t5bCwSUFIB8Gt1oZMTQBQoFiqNsrIytmbNGqxbtw47duxAY2OjCABPr9fzDamE40PWhhCtNxoSjr9erw/aBAhdaTQPTU1NImgZIpPZhIT4BKSmpjJhN0tSMJmZmVFXoIeKxQjHIhJv+Xw+6PV63HTTTbj11lsjwptQCviLL76IP/7xjyw+Pv6oh3qnNSDFAaSx1Gq1x6YCEfa2DvX5kU5zVKvV6O7uZkuWLInJ+rjmmmuQn58flfVB9wn1vb6+PtbV1YW2tjY0NTXxSqKurg4NDQ1obm7m3RvhKoqpGj1UwRd1zpM7X6/X89Xm9JMUeLie65QRRf257XY7X5EuFYA6nQ4Wi4XfUft8PjQ1NaGmpgYffvghjEYjJk6cyGbPno3zzjsPRUVFnFCJSlv7/tpWs7Aivby8nH3wwQf46KOPsGPHDn6sDQYD4uPj+d20x+MJ6pfOcRyPCmC1WhEXFweLxcL3ohcqBZpTuo507Ak52eFwhIRJEVafCyFOhNhLra2t2LlzZ9B6lFagZ2dnIycnBzk5OaLi1MTERFgsloiZg9EobcYYLr/8cvzjH/9Aa2tryGwk4cazvr4ey5cvxx133IFQIIu/BTqmFUgkRqId2ZEUAhqNBs8//zzq6+sjWh8kyCwWC26//faorA/aPVRVVbElS5agrq6OF8i0UDs6OtDd3Y2+vj5ZVFZSEDqdDkajUeTTJQVBfcOli89sNiM1NRXJycm8P5wOShNOTEzk4b9NJhP0en1EAD05N5rT6WR9fX0gZdjY2MgXp1VVVaG2thbNzc18wR0JJLPZDOBnYMitW7diy5YteOyxx3DKKaewq6++GnPnzuWMRiPPE7+mK4JcIlQd/fnnn7OXX34Za9asgc1mA8dxMJvNfIGo2+1Gd3c3f35CYiJGFhRgVGEhRo0ahYKCAuTl5VELYlitVhiNRi5WJXkYMoU5HA7YbDZ0d3ejo6MDra2taG5uRlNTExobG9Hc3CxyaYbakFAhp1wFemdnJ1paWoIUjHCTcBjSn6WkpPCwJnaHHW6XGxqNBjNnzsRNN90U0TqgjUZiYiJ39dVXs0ceeYRvbxzO2tdqtVi6dCluvPFGZjKZjmorhJ6bMgCF8jIQCMBsNh/bWVhaGYh2tVr9M6JoTw/i4+Mx2AxAVkFHZwd79tlnodfrIyoryjVfsGABRo0aFdH6oGdua2tj55xzDg4ePCj7nuRioiConAUh3bWS68JsNiM9PZ3vjS4HV5KcnIy4uLiYYTOEgbpwzM1xHGVicQkJCcjJyQlpadXW1mLXrl28oti7dy86Ozv5OElCQgKvTFatWoVVq1ahqKiILViwAFdeeSVSU1O5XwrCXW48aCf79jvvsCWLF2Pz5s0/u35MJl5pOBwOfrEnJydj6tSpOPHEEzFt2jSMLypCbk4OFwmtQJoKHW7sKbvLYrFwFosFaWlpEZV9V1cXD9BZV1eH2tpa1NbW8kWCHR0d6OnpCdrQ0L0IdkZagU7WEaFDhxL0H374IbRaLbvuuusiriPaMN1www149tlnQQk1ocaGoN4rKyvx5ptv4sYbbzxqrRAeYLKvjzU2NoqsL0rmSU1NPbYtkMTDQkNklmm06OnpQVVVFXJycgbd700ZOs8tfQ6NjY1RWR8+nw9GoxF33nlnVAqN4KnXr1+PgwcPIiUlhXcr0GIj37dc7IJA62gXl5mZidzcXL4SXRj0jFSJLowphGr8M5C2qFJlI5cMYLFYuPHjx2P8+PGYP38+AODgwYNs/fr1+Pzzz7Fx40a0tbUBAA/REggEsH//ftx11114+umn8ac//YktWrSIEzbe+qUWslqtxtdff80efPBBfPvtt+A4jndPeTweXhFmZ2fjtOnTMXvWLJx88smilrdCXoqUPhvLuwnjAqHa5NI86PV6ZGRkcBkZGSguLpZVMB0dHay5uVkUc6urq+OTMjo7O9Hb2wubzRZSuWk0GhgMhqCEDIPBgI6ODnz55Ze47rrrIrqyCL05Ly+Pu+iii9jLL78ccb2Ssl+8eDGuvvpqvn7saLNCyOLduXMn6uvrRfUwpLRHjhx5bCuQYcOGyXDgzwvg9ddfx/Tp0/mME6nQCIfbFGoRklupra2NLV26NCbr49JLL0VRUVFUsQ/aOY0aNQopKSno6uriF7HQ4khISOBdTJmZmSJkU/IpE8R8pF2rUJBIxyAaBRxtT/RQFeiR6gaEqcgajQaFhYVcYWEhrr/+etTV1bHPP/8cb7/9NjZt2oS+vj4+NsAYQ3t7O2677TaUl5ezf/3rX9wv3UyKIG5IcRCUuMvlgtFoxLnnnotLL70UM2fORFpammxVen9qc8LNRyjlH62yl1P0er0eWVlZXFZWFo477rig810uF7q7u/lGZc3NzWhububTwskla7PZeJgTqmdxuVy80pkzZ07Mm4A//OEPeP311yNu9ghkcO/evXj33XfZFVdcwUWyQmJFYQi3HgbDM0P8rVKp8Mwzz4RMNZ8yZcqxXYn+3XffsVNPPRUmkylIAHo8Hjz1z3/i1j/+kRvIPeT+98ADD7CHH344KuuDsq82bdqESZMmxZS6y3Ec6uvrWX19PZ+WSR3jrFYrLBYLF23hVygFEUsl+kAA+ITXFILm9bdWIFTNxJYtW9grr7yC999/H52dnXwLVwDo7e1FWVkZxo4d+4tiYZ1wwgls27ZtSE5O5uMHaenpmD9vHhYsWICSkhJR/xKhUI5mLENZDOHmNxwIZn87Bg607kj4bC6Xi5EC8Xg86O7uRllZGfLy8nDqqafGtAmgeZgzZw779NNPo2p763Q6MXHiRGzdupULVehJG8qhSo8++ii7//77RfUfNB56vR7btm07NhWIsO94SUkJq6qqEuV5k4nmcDhw0kkn4eyzz0ZBQQHi4uJ45SIMtOsNephNZhiNRiQmJiI/Px8Wi4WTMiHHcWhpaWETJkxAb29vWH+q0Po4//zz8d///jfmGoRoGFTaNjQWYRxOAMUKOUE+bI/Hw4Q1BSTgDwf0uWhbCwuVXiRhSG49YWZQVVUVe+mll7B8+XK0trYC+Blccc+ePcjJyeGiEfwDVSBkgTz11FPs7rvvBgBkZGTg+uuvx0033cR3spSmHUeap1jm57AAZvScGo0Ger2ei8WSCdU5MFaXWbgNSSywOP1p4qXRaLB69Wp2zjnnBAlUOaK+4e+9/z4uvOCCsFaIEIGBqv+Jf/n35H7u0yFUpuE6ogrcdxzBKnEyPOH3+xllNjqdTvT29mLPnj145ZVXsGrVqqBmUiST5s6diw8//JA7JhUI7dQ0Gg0ef+Jx9ud7/oz4+HjZ5kcUlIx2x28wGJCcnIybbroJf/nLXziacBJQf/nLX9hjjz0WtfXhdruxfv16nHjiif0qHJT2lI7F7RBuRxiNAHI4HIyyctra2tDa2orW1la+Cr2zsxM9PT3o7e3l00DdHjff7lNYOKfTaaHX/6/ndoKg4pkqz7Ozs6leQDZYLESmDSVshT0dAKCuro69/vrraG5uxrnnnotzzz33F4dz5zgOL730EmtsbMSNN96I7OzssE2uhFZaKBiTw645Runb1ESsubkZbW1tPEKA3W6H2+3mXWGUkXc42wmpqakiAM7U1FSkpKQgKSkJCQkJsFqtXDSWUCQ36ECg5qWumf5k09GYnnTSSWzbtm1BXgs5BWK32zF16lRs3LiRI3lCVu/GjRvZAw88wCfsuFwuEXyPbDLDYSUgbVcrd4iKQlUcVJyY3wOMISCIhRJ8EKXHUzxQ+o5qtRq9fb3YsGEDTjn5lGNXgdDkOJ1ONmXKFOzfvx8Wi0W294WwCjqaaxIjbNu2Dccffzwv+BsbG9mECRPgcDgiWh+UDXbOOefg888/P2IV0KGA68J1TSPq7e1lVBwmhN1uaGhAc1Mz2trb+DThcLDbquBdU0iBKGcxEVE71IyMDOTn52PMmDEoKirC+PHjMXLkyKCAfzgoE6kiiWX3eiRa2gq/QxXz0picsOOiNHZQVVXF9u7di127dmHv3r2oqqpCU1MTurq6ZFNqpfMibElMsZVQRKm4hPKcmpqK9PR0PsZGcba0tDQkJycjPj6ei9bleKSqz6PZcGq1Wrz++uvsqquuimoDSFbIZ599hlmzZnFkiWk0Gr5tNXUelUNfiDVmKP1fpLiidPyEyofWh3Reu7u7cd111+Hf//73z/1PjlUFQkypUqnwww8/sOnTp8Pr9UbM9Y5modNi2759O0aNGsV5vV5otVrcdddd7J///GdUzEcNjdasWYMZM2bEZH2EYq5Y3UyHc+9ZS0sL6urqUFNTwx9Uid7Z2Ym+vj5ZxSvsKiftaS5VDKGAGaWMLtx5CRcaCRiv18srcKFiycrKwtixYzFlyhSceOKJKC0tRUpKSlDAWTomwp1xtLvXI6FAaHylz0eLXKg0XC4Xdu3axTZt2oRNmzahrKwMdXV1sNvtIuFGUN3SvuWhDrn4l7QZmbRXfajOgVQrRJZkZmYmnw6em5uLnJwcUfW5ECU5nBUTLiYjp4xjcZ9xHAen08lKSkpQXV0dFt5EuAmcPn061q1bxwldqk1NTWzixIlwOBwRq9wHStF4GUL9TedrNBr09PTghBNOwOrVq2EymbhjFkxRLrC9evVqdvHFF8NmsyE+Pl429TQaIi1NcQsS/IcOHWIlJSVwu90RLRpivBkzZmDt2rVRWx+hds3hqK+vj1EfcaGSqK2tFfUTl6sFEQogYeEXxRQIEDFSYSbtekIpG6FQEl5TjqgYTYjOS+nKtNPmOA7Z2dk8XP8ZZ5yBUaNGiQLRA4E4PxIKRC5mQ+1bAaCrq4tt2LCBT0uuqKjgaykILYAq8UmwSxWtHB+Gmgsh5Hc4wSXtHijtfx6q6yUAUVMzsmAIQ0vgrkRSUhKsVis3kFYF0WwMyAp54okn2D333NOvjSCtA7VajQcffJD9/e9/R2JiIjwez5DqWijcpJF7beY5M7HyzZVISkr6H2T9sa5AhEpk586d7Oabb8aWLVsA/FxgFkswOBAIwG63IyUlBZu3bMbIESM5Yrpo22QKTd/PP/8c5557btSwJbQ4XS4Xenp6mMvlgt/vh9Pp/BmupLkJDfU/+7upcKulpYW3IqTChCrRhbvUaFreAoDBaITFbEZ8fDwSExORnJzM15UkJyfzPvK4uDi+Ep3gM+h+QqElhDAhyBJqKkRgjg0NDXzVc1dXl+h9CCJFrVbjcPtg3tcbHx+PKVOm4IILLsDvf/97ZGVlcULroz9xpyOhQGjsaYMQCATwzTffsLfffhurV69GbW0tb42YTCbePULQI0IsqISEBF4oZ2VlITMzE+np6eRSgsVqhclo5JWxFHaE/OVUvEhV6J2dnejo6EB7ezsf5+rq6uLjXHKN24Q90OVwuggSR26zQFYMNURLT0/nEQ/S09P5mIzFYuE3KBqtBga9ARaLhbdsonVRU4FucXExenp6IvZNl7qiab7J1T179my2bt06GI1G0YYglhjPYFojxGOUKAT8XO5w++234/bbb+dobogfFAUiUSJ+vx+vvvoqW758OXbt2hWyYEmOjEYjSktL8cwzz2Dy5Mm84K+qqmKlpaW87zoSw9ntdkybNg3ffvstF42ZTQJr06ZN7Mknn8S+ffvQ3d0Nj8fDu3WcTqdsQIzgTWjxCgV2qMUrdD+kpqby7gfCJqJ2tykpKT8LI0lG2pH2Vbe1tbFDhw7hwIEDKCsrw86dO3HgwAE0NjbyAsxsNkOv1/OJCpQskZ6ejjlz5uCaa67BSSedxGc5xeJnPxIKROgn7+rqYitXrsSKFSvw448/8mmVBDfj8XhE3TTT0tIwevRoTJw4ERMnTsTYsWORl5eH1NRUTq/XH9H58Hq96O3tZZ2dnTzmWn19PY/c3NDQgJaWFnR0dKC3t1eW16SbGGH1udBNFsrK1ev1PNgmWbuEFZaTk4Obb74Zc+fOjcrSJ8vvzjvvZE8//XS/kmGESL29vb3s7rvvxrvvvouurq4hIQtNJhMyMzNRUlICwoejdgjSokhFgcgsUqLq6mpWW1uLrq4uuN1uUU8K/lBxUKt+hqvOHZaLcWPHcUJ3klqtxo033cSWvfRSTNbHf//7X5x//vkRG0bRM9XX17PjjjsO7e3tMBgMot2c0ByVsyDkdoUEXkcKgirRpeio0QRAw0Fk9NcfLd2JRSqU6+joYLt378amTZuwfv16bN++na8+NxqNfK2H0+mE0+mEVqvFKaecgttvvx1zDjcEikXYD6YCoet5vV4sXbqULVmyBNXV1VRlD41GQ4KaV47FxcU47bTTcOqpp6K0tBSZmZlcNMHpaOckUgvcWGo3HA4H6+jo+BmGpL4etYfdqJSQ0draiq6uLvT19QXdS9jkSorGHC7GRgkUHo8HWVlZOHDgACwWS1QZdiqVChUVFay0tDSq+aPU1/POOw8ffPABJ2w4Rec2NDSw3bt3o6mpiU86oeNwmi/cHjc8Hi+8Hg+85BqWS/kVjI0kDR66w03ATEYjn+hA4KVxcXFITU3lY1Fms5mTbrCD5llRIPLCbiDAecKga3l5OZs0aVJUZicVIJWUlGDLli1cNAV29Kw7d+5kJSUlMBqNfE1FqHsYDAYeeC4lJSXIvywNYA60Ev2XhHEQpmwKXTbSd6ivr2fr16/HJ59+gvXfrEdzczM4joPVaoVWq+UFMmMMl112GZ5//nnekvols7DoWgcPHmRXX301Nm/ezONBcRwHu90Oj8cDg8GAKVOmYO7cuZg5c6aoT7ow5iAXCP8l5iNUi9polEx3dzePn0XZfkLrpb29Hd3d3Xz1eTRuHHKZOZ1O5OXlYc+ePTAajVy0UEEajQaXX345W7lyZcwFwccddxzf0XIw5M2R9MrQJjjUmCgKJIqgdCS/ozRLSKixr7nmGrZixYqYrI8333wTl112WdTtaokJ7733Xvbhhx/CYrHwcCSJiYlISkri4w90JCUlIT4+HiaTKSosq/5Uog/FjYFcU6ampib22Wef4e2338bGjRvhdrt5RcIYQ1dXFy677DK8/vrrXLTFmYOhQEiYtbS0sOnTp+PAgQMi4ES3243hw4fjkksuwbx581BaWsoJzxUmAwzleQqXSh7Ns9vtdr7eiGIvBG1C8b34+Hj4fD60tLSgoaEBbW1tfFzooYcewrXXXht1piMpkM2bN7NTTz0Ver0+qra6NpsN8+fPx1tvvcVJd/TRgFiG2rxEm/IbTQwl1rWtKJAjpLlVKhV2797NpkyZEpUZT610x44di23btvHBvWgXPn3P4/HwZn0sgnWoWBC/tEKhxU30/fffs3/961/4z3/+A4/HA4vFAr/fD4vFgqqqqqjdHINZib5mzRp21llnwWKxwOFwIBAIICcnB7feeituuOEG3j9NVkYssB9Hy1zJpXdHq2Dk6HACADMajZzRaIw5ME1zfOZZZ7F1a9fyfBLNed9//z2Kioq4oWp5xEIqKHREiOM4PProo3C5XFEJc5VKBZ/Ph9tvv50HWYy1+pa6omk0mqAMFgo0CmEShHnpFEgXNvs5miyM/swP+cyFnRCnTJnCvf7669zatWtx0UUXwWQyweVy4brrroPFYuFinZeBED3bySefzN288GYYDAaUlJTg0UcfxY8//oh77rmHS0xM5CiORWmzvyXlIbTsKdVbmvQhTCuW8jzxvfD3QCAAg8GApKQkjly+/XVT3/rHP0atfDQaDVwuF5566qnfzLpSLJAjYH2o1Wps376dnXDCCWG7mEmtj5EjR2L79u0wGAxR+drDmau/VcH/S7ktSenX19czm82GcePGRb1WBjuITp+3t7ezpKQkTtirfai7p4ayVROtCyjSpm3KlCls165dIsjzcOeoVCr8+OOPGDVqFDfUARUVC+RXokceeQQejycqE5Wsj1tvvRUmk2lAu9yhaDXIQZFQCma0hxTG5EhtfGinS/fNycnhxo0bxx3JSuFoBVVKSgqnUqn43uWxAlb2x2Uk7V0f7gjlahqqVs1A1wr13Vm4cCGiLfbVarWw2+345z//KerTrlggCvHWx5YtW9gpp5wSVXCN0H1zcnKwc+dOWCyWo6YNZjToqEdyhyyXHjzYAf7+wG4fyULCgVqX4eBiBnu+wkG+Szc7R6MVReu0t7eXTZgwAY2NjXwDqXDrnYpBd+zYgeHDhx/VVsgx3VDqSOxqAODhhx/muwhG8q9Sfv8f/vAHWK3WqDOvfmnrYSD9GShg6XA4QAc1RKJ+DUIId4pPECwJ9eSgPuaHD46QAiLt3OXqd45Gi64/kCfCTMJoADJpvpxOJ+PrEDweeA9XJgsz2YQ1GDqdjtoLE6IAFwsfhwJLHMoKhvqmx8XFcQsWLGAPPfQQjEZj2GxLxhi0Wi1sNhueeeYZLFmyBL+mdatYIEPM+li/fj07/fTTI8I9EwN6vV6kpaWhrKwMCQkJv4r1IbQkhDUskZSEx+Phe1w3NzejqamJhxKh6uKuri709PT8D679cIc4EkbRLlRSKFRxTXUsycnJyMjIQFZWFt+TnVruJicnyw4ktXU9UhbSkcbCimbHL1f7AvxcKNnS0sKoZSy1i21ububrKXp7e+FwOHgMJJ/PB3/ADxYQ9xQR8ggpElIgBLsfFxeHxMREpKSk8NDv6enpPPz74XTyiGCJwlTyoaRcqCgwFqRtwiMzGAwoKytDTk7OUWuFKBbIYFsf/+/hqHsgE2zJTTfdBMqmOZLWh5w1Id2VSneodrudtba2oq6uDrW1taiqqkJNTQ3q6urQ1NSE9vZ2vkteuPcUZnbRjjWaxS91kxH+UltbW0gwv8MZNsjOzmYFBQUYP348iouLCcIjqCkVKZSjMf1ViNclnbv29nZWXl6O3bt3Y9euXThw4ABqa2vR0tKC3t5eWQUuzHgSKli1Sg1OzYWcH6/XC7fbzV9XmhouJZVKBaPRiLi4OCQlJ7H0tHS+ApoQD7Kzs5GRkYGkpCRO7v2GgnKhXj/Z2dncJZdcwl544YWINV+MMR50dcmSJXjiiSeOWitEsUAG0fqIpWMZmb8JCQnYtWsXUlJSBsX66I/Lye/3o62tjdXX16OqqgoHDx5ERUUFqqurUV9fj7a2Nr4qW6oYhGm/wuuH60kwGGBwoXpBCJGApdX4VqsVw4YNw4QJEzB16lRMnToVRUVFIqyuwailONIWCAlnKfheXV0d++GHH7Bp0yb88MMPOHDgAFpbW0VuVLLk6FwhYOVA5itSr3o5GH+/oKGR3HoxGAxISEhAWloasrOzkZ+fjxEjRmD48OHIy8tDVlYWUlNTw1ovct0QB7sIlqyQPXv2sMmTJ0fFN2SFmM1m7Nq1CxkZGUelFaJYIINkfTDG8PDDD0fNmGR9XHfddUhNTY26CnYgfaP7+vpYc3MzamtrUVFRgfLyclRUVKCmpuZnBNvOTnglOydyTZjN5iAFIZetE06w9KcBUChBFklBk5UjxEXy+XzYt28f9uzZg7feegsajQZ5eXmYPHkyO/3003Haaadh1KhRnBRUcihYJjTGZB2Q8tm+fTv7cvVqrF2zBjt37kRHRwd/DsWPiK+k8xVJEUTb2zxcU6NIVgi1BZDegxRMZ2cnWltbUVZWFjS/8fHxSEtLY7m5ucjPz8fIkSMxYsQIXrmkpKSEbYE8WHEX6udRVFTEnXvuueyDDz6I2DedrJCOjg48++yzeOSRR45KK0SxQAbJ+vj444/Z3LlzIzKO0PowGo0oKytDVlaWyPoYSIaMy+VCW1sba2hoQE1NDSoqKlBZWYmqqiqRNSFdAFI4baHQCrULDbXgIqWBxspzch3TpB3cQllfoZ6XdoBut5t3N1itVkycOBHnnHMOZs2aJYIGidR3/EhZIDRmwl12WVkZ++CDD/Dpp5+irKyMt7SEPT+kAXSpUuDhuGX6fPRnvqRzJNf4S84KkeP3cHMmfFYqEJQKXo1Wg4T4n4FAc3JyMGzYMAwfPhz5+fkYNmwY3w3RarVykdZ2qF7u0vGkFtlff/01O/PMM2XbwYaSA/Hx8di1axdSU1OPOitEUSCDYL4yxnDCCSew7du3w2QyRVQghIvz4IMP4qGHHorJjqY+40Jo7NraWtTW1vIB0ba2NthstqDnEKKWhupAF+0CFrofQr0vNTIitFs6CP1T2hFP2gfC4/HwjaCcTiefvSUM8Ia6L1UrCzsWhkIDpnsT7D3wMwT4pEmTcP755+O8885DQUEBP09erzdsDcZgKBBpQWNPTw/78MMP8cYbb+C7776D0+kEx3EwmUwiUD45tFqpFRayS6BGDaPByGe8GY1GGIxGGA7DoZO7Uihg6XqEGCtEkaUsu3DxgEhu0P7wphBtWsqbBJiZlJSEzMxMXrmMGDEC+fn5yMnJQVpaGhITE7n+CvKZM2eyr776Kip4E5IFDz30EB588EGOFJGiQI4h6+Odd95h8+fPj8r6EPo/ly5diqlTp8JkMvGQ3H19fejs7OSbJDU1NaG1tRVtbW1ob29HR0cH32fc7XbLLkipUI5kTYRTEiQgpOdQ//Hk5GSkZ6QjK/N/jYkyMjKEGTawWCykODhhc6JYyOv1wuV2MafDib6+PvT09PDuDWGPCWqU1d7eLmrhSs9M95cTuMJdtM/n47swxsfH4/TTT8eVV16Jc889lzMYDPyuU65z4UAUiLRpVHV1NXvllVfw5sqVqK6qAgCeX+TchkKFSJ0YhTxpNpuRnp6OnJwc5OXnIz8vTwTPn5iYyDf40uv1nE6ni2q+GGMEF8JI4dvtdlGjqfb2drS2tqKlpQUtLS1obW1Fe3s733BKLhGDrGMhVPtAlIuUr6VEbQzS0tKQkZHB92/PyMhAWno6kgU8TRafsJGbx+PBy6+8jCcefwJmszlqb0RycjLKysqQlJTUbxQKRYEcRUTM6PP5MGXKFLZnz56IPZLl3E3kpxa6VOQUg3DHQotJuqDkYiORFpXQFSBldspmCrVTS09PR2JiYkxNieQQjiP1CIk16NnT08MaGxtRXV2Nffv2YdfuXdi7Zy+qqqqCYgTUVEpOGNOOWGiZFBUV4aqrrsIVV1zB99iQKpL+KBASbOSqOnDgAPvXv/6Ft956C52dnbw1RxsXueek7oMkGPV6PfLy8lBUVITS0lJMmDABo0aNQnZ2dkT3jTTWFI6n5NyZ0ZLT6UR3dzffVpng2oUpxm1tbejp6QlaF0JY9liVi9wzEw+E2jRJ42tanQ4aQadOoUs0FpBEskIeffRR/OUvfzmqrBBFgfSTaJJfffVVtmDBgqjg2uUYWVp7IfUby5nz4WIScguD33V5ffB4xW4fTqVCQnw80tPTMWzYMIwcORKFhYUoKChAXn4esjJ/rqcIJxiEDW2kgj+Uz7g/Clv6u1ysKFT9A2Ps56Y9e3Zj65at2Lx5M3bu3Inm5mbecqOOftL0YKFycDgc8Pv9yMjIwOWXX46FCxdi5MiRvCIhYRKLAhFaHLW1teypp57CihUrYLPZeFef3DPR+U6nk1cq+fn5mDp1KqZPn46pU6di9OjRspD90jkLleAQK6Cn3N/hXIeRBK3H40FnZydrbm5GXV0dqqurUVVVherqatTV1fHti6XWSzTKJZrYXqg1KOcS7a8SJfdpeno6ysrKEB8ff9RYIYoCGYD14Xa7cdxxx7GKioqYrY9QO+5wwcRQglkYAA1lmhuNRiQnJyMrKwv5+fkoKChAYWEhRo4ciWHDhiEtLY2jrnxywkZOgIVK0TziTBtl32ihcJRTLM3Nzez777/H6tWr8fXXX2Pfvn18pXCk3T5ZiomJibjiiitw++23Y8SIERwJPZ1OF1GBkKvsMD4Se+aZZ7B48WK0tbXBaDRCq9UGjbv0/gAwZswYnD1zJmbPmoUpU6YgISGBk7r/pNX4kYT/YM5nLLU+9FNajxLKOuro6GBNTU04dOgQqqqqUFlZGaRc5Cx6qXKRS2CJZrM2WGNGVsiTTz6Ju+6666ixQhQFMgDr4/nnn2e33HJLv6yPcIpBKgSlAISh5sxisSAxMfF/Pu68PD5vftiwYcjMzERSUpIsxARdXy4t+GhFe2WMIcAY2OHdoiCriKnVapHrzel0YsuWLXx2U3V1NR8zIKtEOm9qtRoejwculwuJiYlYuHAh/vKXv/Cw7/X19ay0tFRWgWzbto1PMf3444/Zfffdhz179vBwINJ5pp06tXVNT0/HrFmzMG/ePJxyyim8lUEbmwBjTMVxnFwm1FCH6ZciI0h3/cLxl9bCCC2Xjo4O1tzcjEOHDqGmpgZV1dWoralBfX29qE2uHAndxFKPQKh1Gk3zuXBywOPxIDs7G2VlZVF3v1QUyFEolA4vZFZSUoK6ujro9fqYrA/hd4W9KOSI8uSpfzG1oR02LBfp6Rl8D+PRo0cjMzMTKSkp3GA0qaEMFp/PxyijRdhH/X9/e+H1ij/3+/3w+X3w+/6HpPs/l0kAgQAL6dYQpYOqVVCrDi9kjRoatSaoJwTf6/lwRpf+cMbQz/ENHafRaKOaD6F10t7ezj7/4gusePVVrF+/ni/4ojiDnCuJFElxcTGeeuopnHXWWVxzczMbO3YsXC6XSIGMHz8eZWVlXEtLC7vnnnvw2muvQa1Sw2wxh3SfUep1SUkJrrrqKlx00UXIzc3lhII22nn3eDzwer3s8E8erkTaMyYSIrLQWhA+K6dSQS1IuaZD2G9GOH/S/2k0mrCV54NBfX19rLW1FY2Njairqwtqk9vW1iZqkxttKrMwFT5WRU1WyOLFi7Fo0aKjwgpRFEg/rY+nnnqK3X333f2yPqhvAKVhJiUlISsr63B+eiqSkpKRlJSEhIQEJCUlwWq1IjU1FWazCVqtDgaDnmMMPOAdpVE6nU4epPDnlFcHHA4nnC4nXE4XXC4nnE4X/z36rsvlhMvlhsvlhsfjPgxw6IbH4w3ZkEpOqAQEO/3BhvSWKhehi0OtVkGt1oiA/Sh9mBRvQkI8EhN/bu1LeEwZGRlIT09HSkoyzGaLbNHZ+vXr2eLFi/Hxxx/z+EVarTakIiEL4bbbbsM111yDM888E729vfxO2W6347jjjsPf/vY33HbbbaitrYXVapUt7qOe3V6vFyeddBIWLVqECy64QNaCdDgcjLKcmpub0dzczGfvdXZ2oru7GzabDX12O5yH06CFuGTSRmNBO38wgMVWlS43X8JDqlikG4L/bQT+B9LIH0YDn3JMB6UeC1PF6VzhNQTfDVvF7nK5YLPZWG9vL/r6+tDX1we73Y7e3l709vaC2uhSFmBTUxPvMvP7/TxIaLRErslhw4Zh586dvFU5lK0QRYH0w/ro7u5mEyZMQGtra1QNo4QLy+/3Y9KkSSguLkZ8fBySkpJhtVp55unt7YXNZoMQudblcsFut4t+t9lsvHAJBPzw+wdWrBdOYIdyf4SrVO5P7+ZofMmh/NOhKuPDjYNKpYLZbEZCQgJv1RUXT8CYMWN+jg8VFCD5cFrlrl272FNPPYUPPviAD25TjEJ6TcYY7HY7UlNTeSEt9LHr9Xr02fvAAgxmszloA6JWq+Hz+eB0OpGbm4unnnoKF198MUdCrbKyklVUVGD//v3Yv38/qqur0djYiPb2dvT19cnGwOTiONLi1HBumn4JMcbAIsydXEHhYG5ApIqKLB5SVEJlRUCdpGAI/TmUYqLaGMrU6+vrg81mQ1VVFX748QfUVNeExYgLZ4U899xzWLhw4ZC3QhQFEgNR3cfDDz/MHnjggX7HPrxeL7+I/X5/2JTBoAkDB07FhazGDifgYxXS4b4Ty+dHlIFDvFs4SA6hsKKaAAJUJNLpdEhPT8eIESMwceJEXHjhhTjppJO4Q4cOsRdeeAHLli1DV1cXTCZTSB869aeX85kLCydlrAlYLBbcdtttWLhwITweD95//31s3LgRBw4cQH19fZDvXq4dcTSK99eYy3D8GM3Goz9BebnjSDTBEgblQ8VnIik8l8uFESNGYMeOHXzN0VC1QhQFEkPcguM4tLW1seLiYvT09ECj0fQ7YCb0H0fjK41GqCtz2X9hJpwHEi4UIyAqKirC7NmzMWvWLKjVavzr2X/how8/4nfycjGdWOeEMYY5c+bg0ksvRWdnJ1auXIlvv/2Wr7pXq9X8zjfaugeF+q+cYsVsG2gwXWiFLFu2DNdff/2QtkIUBRKj9XHvvfeyf/zjHwPKvFLo6BE4QkvB4XDwFsPxxx+PWbNmoaOjAytWrMBA2hDTvdxuNwoLC3Haaafhq6++QmVlJYD/VZ4riuLYILJCRo0ahe3bt3PRtj5QFMgQJbIUmpubWXFxMR8UVcbu2FvYlIlFLqSUlBS4XK4BKxChIrHb7VCr1TCZTGGRcxX67RJZIa+88goWLFgwZK0QlTJVkYkW8O7du9He3h6x77FCv10+oJRri8UCq9WK3t7eQVMetFkhnCVhoySFjs1N66effjpkrQ9A6QcSE1mtVn4nGgnmPBJzHAs0GEw/VMeK5v9I1CrEqjSO1kLPo3n+f4m1w3Ecj4gwZC0lRS1EJsIcKi0t5U477TS2fv166PX6oApVYRMdUaEVOICLLt01GmFA14tuBQLiZMrBWbjRfD5Yi3+wgp+hMpNCfX8oxRzCQZcLU4kH6zkHoztmVLw8CN8ZrPsMJcVMlu7s2bOHtCJVYiAxLqiWlhZ27733Ys2aNbDZbPB4PCIwRGHlLaX0UbpouII7OWH1S81NtF3ooqkBiaomhONk9V+4znaRUjKPpKCnegGKew1GnU00RDEXgkoX8prw2agnCD/WgvEdzLqOSJl/Uc3fz38Ifh4uUJT+P0Rr3XAtd6NpyztYrZWPNMXFxeHqq6/GP//5Ty4cHpiiQI5CJQL8DBne0dEBp9PJZ2NRsRKlWVKRkRDHSq6SW/ozFiUTaSctVyMS9lBxUHEq2crv/hwhhdhhIRdKKMkpjIAE00qKEeYPBOD/HwTLz/AcPh98hyvp/QE/wIIrpEnBe71eOBwO2Gw2tLe3o6mpiYe5aGpq4kH5qKo5VBvfwbB4GWOirK+EhATk5eWhoKAABQUFGD58OHJzc5GRkYHk5OSgHuexFnnKKZdYaoQiCfRoiwjD/e9IHDxWGgT/D8TwLAKFF+l9hQpU4CAQ/80Y4uLiUFRUxEPVDGlXm6JAYlciseAOKXT0U29vL6usrMT333+Pr776Chs2bEBrayuAn8EW1Wp1WJDLaK1AqkCnRlZFRUU4/fTTMWPGDJSUlGDYsGHc0dTuVKGBkRSjTVEgvzFFEmnsYoE5H8rzMNSDtEdqbEP1oW9sbGSffvop3nzzTWzatAk+n6/fVokUjNFsNuN3v/sdrr32Wpx22mlBzbqkiMmD0Wvl15yb39p9B+NZhBayYoEopNBvZLNAO0Lhwv7uu+/Y8uXL8dFHH6G9vV1klYSKywgr18kFmpiYiPnz5+OWW25BUVERrwmo2+HRAMOu0LFHigJRSKF+KBRCJiCBXl9fz95991385z//wU8//cTHSyguRkpHin02cuRIzJ8/H9deey3fkIqyqo7mXiwKKQpEIYUUikAU2BfGxLZv387Wrl2L7777DgcOHEB7ezvcbjcP35+VlYVJkybh3HPPxZlnnsn3J6eCRCXOoZCiQBRS6Bi0SqRwE9TT2263Q6VSwWq1IiUlRWRWKIpDIUWBKKSQQrxVIhcvkSoNQHFTKaQoEIUUUiiMZSL8qQTCFVIUiEIKKaSQQsc8KU5XhRRSSCGFFAWikEIKKaSQokAUUkghhRRSFIhCCimkkEKKAlFIIYUUUkghRYEopJBCCimkKBCFFFJIIYUUBaKQQgoppJCiQBRSSCGFFPoNk0YZAgy4l/ZgNPUJ1YRoKEFfhBqnowWeI9SzK6RQf+XDsQ5No0CZHAFmE6KrKgLq1yUCNpSi5CqkkEKKAhkUAdPS0sJiaUMqtwvR6/WwWCyyLUgj9U/3+XxoaWlh0msyxmAymZCYmDgktFBLSwvz+XxB/9dqtUhLS+OG4twK0XB7enoY9RvX6/UwmUycXq9XlPwxbkVEgtF3OBysq6uLX5PC9ZmYmAiTyXTMMtAxq0AYY+A4Dj09PWz8+PHo7OwUMUh/FEhcXByys7MxYcIEnHHGGZg5cyasVitH9wol4KqqqtjEiRN5y4UxBo1GA5/Ph8svvxzLli3jfD7fr7KLpmf3+/0oKSlhlZWVUKlU/LMHAgGMHz8e27Zt44bi/H7//ffs9ddfx6ZNm1BXVweHwwHGGMxmMwwGA+644w7ccccdXDSKXqFji2jNvfHGG+zGG2/k1yQA/vd///vfuOyyy3619flr0zFv1zPG4HA44HQ6B3Qdh8OBrq4u1NbWYtOmTXjhhReQn5+Pe+65hy1cuJAT7lzknqGvrw9yFojL5RoyYxVqnAY6dkdip8lxHG6//Xa2ePHikO8CAE1NTfx5Cv12yWazse7ubtnPUlNTOaPRGFaROJ1OWQtEziI/lkjJwgL43tYUs+jvoVKp+B7YarUaNTU1uOWWW3DttdcysjhCCSqNRiN6Bvp7KO2KaZzkfg41t9VNN93EFi9eLOpJLpwnnU7H/1Tot21FAMDy5ctRWFiIsWPHorCwEIWFhRgzZgwKCwvxzTffMOB/Tb7kPAzCNSn9/VgmJbIIsX9UKuCjCYQzxmSVg0qlgkajwfLly1FQUMDuu+8+WVeJUAjTzkatVoMxNqTanKrVaqjVan5MhD+HAtHYfvzxx2zZsmXQarXw+XxBzZwCgQA8Ho9ieRxjisTj8cDn8/EZj7TWQikO4Tom3id+GYrrU7FAhiAFAgH4/f6wRyAQkBWkgUAAXq8XarUajzzyCBobG5larQ5K2fX5fHC73fD5fPB6vfD5fHC5XPD5fOjp6RkyY9He3i5aiPSzs7NzSDwfKfonnniCVxRCBUGK3mQyITk5GQaDQQmgHyNEmweymqW/hyOn0ylak8Lfh5L7VrFAhhjDMcYwevRoDB8+HHKBcIpd1NfXo7a2lv+OVGhxHAeHw4GPPvoICxcu5N0sdL3U1FT8/e9/5xUR7WwCgQAmTpzI74J+TaGsUqnwwAMPiJIN6Gd6evqQcV01NjayH3/8kVcWwnfQ6/V47rnnMHPmTGi1WrjdbhgMBn5HqdCx4WkQehvCWaC05qZOnYqHHnqIX5P0WSAQwJQpU37V9akokCFKarUaPp8Pt9xyCxYtWsRF2qFs2rSJ3XXXXdi5c6eI0YSCeOvWrVi4cGGQcE5MTOT+9re/RTSjY10s0h24NGYTqyK59dZbuSO5uOWeN9paGjqvtrY2KOBJc3neeedhwYIFXDhFGeszxvqcgyEAhYqxP4pPep1Y3mGg8xSLkJcb5/7yb7/cM4fX3MSJEznayA3G+gz1joP5ftGM42AoPUWBRCCXywW/3w+5ND2aCKPRiDPOOIP76quv2MSJE9Hc3CwSYDSJlPEjnTjGGO+Tl1Nk0aYHkj83muA2+X1jEUAejydkNbdcMDqa6noSRuRjDmVd0HfCCUOy9IQWpJBGjx4Nr9cLv9/PjymNVSSBKUySCGcFkfUYrQCIZozoGYTJHqHGIZxwi+Y6wiJYuXEIN0/94SnhtaOZD+m9Qo21kC/CJa/QHNBB15LyqNfrlT1Xq9VGLYiFRa2R3tHn80GlUsUs5IVzEM049vc+igKJYedNAbNQC4MUQGpqKnfJJZewxYsXi3LGhQI41D2kBYixEikOEozNzc2ssrISjY2N6O3thVqtRmJiInJzc1FQUACr1cpJ3T+RKNaMpUjXFCYUdHd3s927d6O2thZutxtmsxlZWVkoLCxERkYGJ/1+qN2U2WwOeT+DwQCtVgutVhvTmNI9Ozo6WGVlJerr6/nYVHx8PHJycjBy5EgkJydzcu82kDGiuSEe3LlzJysvL4fdbsfo0aNx4oknclKBF4pHiZcDgQB2797NKioq0NPTA71ej5ycHIwbNw4pKSlBYy0ch+7ublZWVoa6ujp4PB5YrVbk5ORg1KhRSEpK4mJ5d7kx7uvrY4cOHUJdXR3a29vhdDqhVqthsViQnJyMrKws5Obmwmw2h+VfGg/i2XDry2QyhRWiKpVqQOtTqHxVKhX8fj8qKytZdXU12tra4PF4YDQakZ6ejuHDh2P48OEcrWO/3x+1ZSiUUR6PB7W1tay+vh5tbW2w2+0AAKvVioyMDOTl5SE3N1d0n/5Ys4oCGUQlEwgEkJubG/J7CQkJIncLLeqmpiZ27bXXimIgarUafr8fZ555Jv70pz+FLHQTMk5PTw9788038Z///Ac//fQTbDab7HNkZmbi5JNPZldeeSXmzJnDEVOHuj4FpK+77jrW2NgYFAMZPnw4XnjhBU74bhzH4Y9//CM7ePAg79IjV9KcOXPwhz/8gVOr1di9ezd75pln8Nlnn6G5uTno/vHx8Tj11FPZHXfcgRkzZnDCAkaVSoUffviB3XffffyzUEBfuOOkXdm///1vrFu3jgkFzrJly5CXlxdU7Enj4fF48N5777GVK1di69ataG9vlx3T5ORkTJkyhV166aW45JJLOL1eH3FMfT4frr76atbW1hYU+1qwYAHmz5/P0d8vvvgie+GFF1BWVsZfJy8vD5WVlVCr1di6dSt74IEH+PPpejqdDsuWLUNGRgbn9/vxwgsvsBdffBG7du2SfYezzjqL/elPf8Jxxx3HCXezbW1t7NFHH8Xbb78tO08pKSk47bTT2KJFi3Dqqady0WxKaHxcLhfef/999t5772Hbtm1obGwMiz2VnZ2NqVOnsksvvRQXXHABJ+QHuuYrr7zC3n77bX4camtrRbwgtP7uvvtupKamMhp/xhiefvppjB8/ngOAr776ij355JP8mqQx8fv9uOeee3DGGWeEXJ/CDcD+/fvZyy+/jM8++wwHDx6UrSHR6/UYPXo0O/fcc3HVVVdh3LhxXKRNnjCmum7dOvb6669jw4YNOHToUMg6FaPRiFGjRrFZs2ZhwYIFKCwsDFurFrWv7Fg5yKzt6upiycnJDADjOI4BYACYRqNhANjjjz/OGGPwer1hr+fxeBAIBHD//feLzhf+/re//U10Lb/fD8YYysvL+e9Kj4svvjjk/YXm+SuvvMKGDRsmOpfjOKZWq5lGo2EajYap1eqg65900knsu+++Y1QUFWqcfD4fMjIyZJ9x+PDhTO6ZxowZI/v9q666ih1epEyv18s+r1qtZiqVSnTefffdxz8njcenn34acuyiOXbt2sWEcyEch88++4xNmDAh4pgK+QYAGzduHPvggw/469J4SMfU7XYjKSlJ9rnoXaurq9kpp5wi+kyr1TKNRsPGjRvHaBw++OCDkO/Y1tbGmpqagq4Taqy1Wi174403+Dldv349y8vL4z9XqVSic6XvL5ynUOuFPvvoo4/Y2LFjg55ZpVLxYxyKHwCwE088ke3evZsfaxqPO++8c0B8sX79ev79ly1bFvJ7L7/8csj1STzlcDhw9913M4PBEPId5d5Pr9ezRYsWsd7e3iAeld6jsbGRzZ07t1/jaDKZ2P/93/+F5Ndwh6JAIiiQxx57jHm9XjidTni9XtmDYgOMMUyaNIlfnNJr/fTTTyJGoJ8VFRVMp9OJFqVer2dqtZpdc801sgxK6cU+nw8LFiwQ3YsWNQk7OlQqVdD/6VmfffbZkPehBT969GimVquZVqsV/SwpKZFVIFOmTGFqtZp/N3qnv/3tb+yNN97gx5yuQ88nJ7DpWZ944glGwpcxhtWrVzOdTseMRiPT6XQixS096FmEx969e0VzQoLtwQcfFJ0nHFMSoOHGFAC75557eEEqXJRCBTJ8+HDRWNIYPfHEE8xms7Hhw4fzQp0WPv0cM2YMEypS4VgTHyQmJrIdO3aw8ePH89cJN9Y0fiqVipWXl7NNmzbxSj6ac+n9H3744ZBKhP73+OOPy/KtdBzl5oDeEQBLSUnhlYjL5QJjDPfffz/T6XTMZDLxYxKKLzQaDc8PBoOB6XQ6flPFGMOKFStEcyP8/fXXX5ddN8RP9fX1bMqUKaJ7yfGMlL+EfFxSUsJqa2uDlAgJ+8bGRlZYWCjaFKhUKv46wrmiz+Xuc+WVV8ryq6JABqBA/vWvf7Forudyufhdj1DD63Q6BoBdffXVQQtKqECIwekZ6P60WxcyaCAQ4K9z8cUXBwkYqQKTLnTp9+jvJUuWBD2jUIEUFBSI3o9+FhcXyyoQqTKl70+ePJlZrVZecMoJeqmAIqGt1+tZTU0No3sMpgVC733vvfcGjY10XuV2etJdHwC2aNGikGPqdrtBViOdT+c9/fTT7KqrruLnVjh3JCDGjh3LK5BPPvlENNY0fnFxcbxw0el0QeMqxyf0v+OOO45lZWXxzyB3bihFolKp2J49e0Jad2+99ZZIMYQaY+nf0vvR2EyaNIkJN3ODaYG8+uqrIb0Kr732muz69Pv96Ojo4K0r6fiFWp/C/9PmCgAbO3Ys6+7uZkILgX4/++yzRbIm1oPjOP7cRx99NKL1KDyUGEiE7Jj169fDaDQyOR8kua4qKirw+eef48CBA0EpvB6PB6eeeiqWLl0adbA6mmdTq9V4+OGH2bvvvgudTicK0JMvWKPRoLS0FNnZ2fB6vaioqMCBAwdEvnbKutFoNFi0aBFKSkrYKaecckTABWlctm3bJvKD63Q6pKenQ6VSobW1VbY4i7JX3G43VqxYgQceeACMMaSmpuLkk0/m36enpyfIv09+8GHDhmHYsGGimh6LxcI/h1arxX/+8x/22GOPBVWxC+d1woQJyM/PBwDU1NSgrKxMBC5JC1yr1WLJkiUoKSlhCxYsiGpMyce+cuVKbN++XeTXJ0FMzxEOh4me22azwWazQaPR8DySkZEBnU6H1tZWWaw14ont27fz/n7KQkpPT4fRaERXV5dskSvdNxAIYOnSpSK+J77r6upiixYt4udFmJYcCAQwbdo0LFy4EBMmTIDJZILD4UBlZSVeeuklrF69WpRhR4W6P/74I7Zs2cJOPvlkjuq3hHzR0NCA6upqWTyrcePGISkpCcIYSGJiIkKBoEa7Pq+55hrs27cPWq1WlMVFc5qcnIyJEyciLi4OnZ2d2LVrF4Sov6SYtFot9u3bh9tuuw2vvvqqKA6zYcMGtnr1atH80vmpqan485//jBNOOAHx8fHwer2orKzERx99hDfeeEM0ZzSO/+///T9cffXVLCsrK6o4lmKBhLBA+nPQbol2iklJSeyuu+5ifX19TChYBmKB0K5j3759sj5ous55553H7wCFzPjFF1+wUaNGBe3u6Lxx48Yxt9vN32cwLRDpztVisbAnn3ySVVZWMgJqPHToEHvsscd4i0r6bhzHsVNOOSWkT3jjxo1B70Zj+cgjj7BQvBAIBNDZ2cnS09N58186r1OnTmWbNm0KusZhwSVriahUKpaQkMCam5uZMF00lAUSzrIxmUxs5MiRrLi4mI0YMYJNnjyZ0Y5baoFIXTQA2KxZs9jGjRsZwdrX1NSwv/71ryHvT1YfADZlyhT2zTffsJ6eHuZ0OtHS0sKWL1/OEhISeNeL8DyO49ioUaOYcGdOvy9fvpzfMdM9yM101llnsXC735kzZ8q6iDmOY08//XTIeMSSJUuCrAi6xurVq2X5gq4TiwVCz/6f//xHZCFJ73nXXXex1tZW0X2bm5vZn//8Z9n5oPN27NghctXdc889ItejcN7Wrl0b0nuycuVK3nVH8RGDwcDUajV7/vnno4r7Ki6sKBSINAgV6pAKDo7j2FlnncUOHjzIpApjIAqEGJTcG3IL4rLLLmNSd5dwUba0tLBRo0YFubPoWm+++SZ/z8FWIHRPnU7Hvvnmm5AM/thjj8meC4BlZmYyu93OhO4Cj8cDv9+PtWvXhlQgDz30EPP7/SAFSeNPY0v3lBvTE044gREUvHBMhcrg9NNPDxn/uv/++4PGNJwCIf84JSm89NJLrLa2llcYXq8XXV1dTJpMIKeshckYcsfNN98c9N5CPh4zZgyz2Wyy53/88cdBz0/zZDAYWH19fVCA+8ILLwy5CduyZQsvIIVwQS6XC4FAgE8WkBvjhx56SDTGwrl+8sknQ87tJ598woQ8JOWLWBQI8UZxcbFoDoX3+/vf/y6SCT6fTyQbyP1GmyhaLyqViv3pT38SKZALLrhA9Dw09haLhdntduZyuXiYJJoDela55AUA7He/+13UCkRxYcWQrRbKnJWmHJJJ/tVXX6GwsBBXXXUVe+qpp5CSksINxI1F6bptbW3sgw8+4Pt0CF0sWVlZePHFF3nmJGRgoUstLS2NW7ZsGZs+fbrsOy5btgyXXXbZEYFnoFTeK664Aqeddhrndruh1WpFRVsAcMstt+Dxxx+X7dPS0dGB1tZW5Ofni9wOkfLlqV5EmvNPz7R8+XLejSJ0qRgMBrz66qswGo28S0E4pl6vFzqdDq+++irGjRsHu90ucs9wHIfXXnsNf/3rX6OuJ6C5PfHEE/HRRx/x9Rl8/r1Gg4SEBC4cECA9f0JCApYuXco/qzD3n+M43HzzzXjxxReDQAXp/AcffBBWq5Vzu918XQW925w5c7iioiK2e/dungdprlwuFzo6OpCdnS2qeUlLS8O0adNEKceBQABxcXEYP348n34sLKSk76SlpcmuuVCFmDTX0fDFQMERydW4adMmtmvXLt5VJUz7nTZtGv72t79xcgV89N0HHngAL730kqjFA7mn1q5dKyrWpMJZYWmASqVCX18fVqxYIUK+oOuTbPjnP/+J/fv3B6V+UylCNC5sRYFEWT0a7aKXCiAAeO211/D9999j9erVLDc3t99KhBTC119/zRcHChWIz+fDtddeC4vFErLBjU6ng9/vx6mnnspNnTqVbdmyhb8OLf6tW7eisbGR94MOJmwEjc28efP4uIZUmAcCAVitVq60tJStXbuWX4i0SDweT8gal/48j0qlwu7du9mBAwdE80f3nT17NkaNGsX5fD7ZIkSKl+Tm5nIXXHABe+211/hCUrr+oUOH8OOPP7KTTjopYvMhWsjJycl4//33kZKSwgkFf7SNz0gxnnXWWUhNTQ16fqpWHjZsGBISEkT+d1JgJpMJ06dPDxLqQrDKoqIiCBWI8Bmlvn8AeO6557hwPE48LQQ8JMW1Zs2a2GsVfsFNJgB88sknvFKSyo67775bFFeTzlcgEEB8fDx38cUXs/Xr1/P/o/E0m818vAIAEhMTgxQkzd8tt9yCn376iV133XUoLS3lpIXA55xzDnfOOeeE5UNFgQygOJAxhuLiYowaNQrhBGlfXx+qqqpQUVEhClATo+h0Ouzfvx8XXXQRNm7cKIKF7g9t2bIlqPKYGPXss8+OGPwj8/Occ87hryW0cJxOJ3bs2IGsrKxBC/wLhY5Op8OYMWNkFxi9C8dxfKBa+C40L4MFx07v9/3334uErvC+s2bNiuo+jDHMmjULr732muj79J5bt27FSSedFPFaQhy2jIwMjqyeWBc3UVFRUdh76vV6PjAuLRLNzMxESkoKF0ppcRyHxMTEfglaYQBdCC0jpc7OTrZ371688cYbWLZsmcjyHkpE62TLli0iaBl63vj4eMyYMSMszBBZQi+//DIn7FAq5Q+qUD/rrLPw1ltvia4n/P6yZcuwbNkyFBYWsuOPPx6TJ0/G8ccfj3HjxomQE8jtF2uPH0WBRFjEN9xwQ1Qggh6PBxs3bmT33nsvtm3bJhKOHo8HWq0W33//PZYvX85uvPFG3oSNVQADwMGDB4MYNBAIwGg0YuTIkVEBpXEch/HjxwcxnPAegyGg5SgxMZEXOqEEIcdxiIuL+8Xmu7y8PGRG1JgxYyJChdAucPTo0SKFHukeoXbhHMfh/PPPH5SeExaLJeyzq9Vq7rAPPYji4uKg0WjCbkpihfkgfqXsP3J3VVZWsv3796O8vBxVVVVoaGhAU1MT6uvreQSA/rad/iWsD5VKBZfLherqatHaIUu2sLAQSUlJXKQNHn0Wzkoly+TSSy/lFi9ezHbu3AmdTsfHLaTfO3jwIA4ePIi33noLACgDjJ1++umYNWsWSktLY4ahURRIFORwOEKCKQonXKfT4fTTT+e+/vprNnXqVOzdu1ekRGjBPPfcc7j++uv7ZYWQIAkFp5GYmMjDpUSjiDIyMkIqidbW1iNm1en1euj1ei7a5/wlSPq+QkgZ8rtH8zwpKSkwGAxwuVxBwq6trS3ideiclJSUqDcDgzEvof4XjTCJdZ6EVu3HH3/M3nnnHWzevBk1NTUh1wSBFrrd7iEtL7q7u1lXV5fsuqJYEO30o7XUQrnKGWMwGAz48MMPMXfuXB7mhqwIoftdGG85XKOCdevWYd26dbj//vsxY8YM9te//hVnnHFGTC52paFUFEKbwN5CHTTYh0EAuT//+c9BO0fy4e/evRuVlZVM6GuPlaTuG1rAOp0OGo0m6tVMvTDkGPVI9mIfSv5repZQgkmj0UQFIimcA+n3aXyjEX50nbS0NB7w8rfU9IqE08GDB9mMGTPY3LlzsXLlSlRXV/+c1aPRwGAwQK/XByUqEMjmUOy9QXPsdDqDQFOlNUfRbhxDtc6Wukbz8/O57777Dn/729+QlZXFb3iF6NDCui8hsCZtir/++muceeaZ+L//+z8mDP4rFsgvSFqtFowxHH/88TxYnpzvsry8HIWFhf02xckfLhUshzsEMp1OF5XEIYEm5xYYKDrw0ULCOJUcUefFaK7DcRwPbRNKuUSrQIxGoyge8VtRHhzHoaamhk2fPh2NjY28e4w+o/RoGouUlBRkZmZi1KhRmDlzJlJSUnDhhRcOWVdWuIwv2pRFM5+hknek8QlSDBaLhfv73/+OO+64g61btw6rVq3Cpk2bcPDgwSAoeo1GwysToWwCgIceeggZGRnspptuiqrwVVEgR2BHazKZoNVq4fF4gipfgZ/TUGPZiQiZSq1WIzk5WVYIdnd3o6enByaTKSqh2dLSEpKhU1NTj6l5k76vMBOpra0NBQUFUc1Xe3u7rPuK3FvRzvtvrdUujSfHcbjpppvQ2NgoQlAgi1yv12PevHmYNWsWiouLkZWVhYSEBH4wqJCTgs1DzZK1Wq0wmUyiTQQ9Z6h+QKEUUaTvSVEJAoEAEhMTuQsvvBAXXnghAoEAqqqq2I4dO7Bx40Zs3LgRO3fu5BW00MVOQXmVSoW//OUvuOSSS1hiYmLEeI2iQI7ALstut8Pr9YbcJQ2kDgQACgoKghoOqVQqOBwOVFVVISMjI6KfldxpUmElvMdvUZCFolGjRgX9jxIp9u/fj2nTpkVsSsRxHA4cOMDPsXCHx3Gc7D2OpXWh1Wqxfft2tnr1ah4qX8hj6enp+OSTTzB58uQgpvN4PHyQeqhuHAEgISGBy8zMZN3d3UFQLQcPHkRfXx+zWCwhBTMJ8lWrVrE333yT91oI4fmfeuopJCYmcuGypchVWFBQwBUUFOCiiy4CAOzbt4999tlnePHFF1FRUSGSUSQzurq6sHHjRsyZMyeiHFEUSBRCO1RrSOl3fD4f9Ho9vvjiC96fK3RjESPRbre/wnnatGlYsmSJbKro6tWrI6aKkvL58ssvRUqDdtx6vR4lJSUDUnZHk8UIAMcffzy/gKX0xRdfYMGCBVFd64svvgiaWwE68TGllKUCDQC+/fbbIF8+KeqHHnoIkydP5txuN9+1j75Hscju7u4hO4bk8iktLRUV6NEGr62tDVu3bsWMGTPCdtjkOA4vv/wy3n///aDPk5KS8Pzzz3PAz+nNtGGRnl9aWsrp9Xpx1bhGg7Fjx3Jjx47FDTfcwGbPno1NmzYF1e9wHIdDhw5FZS0rQfQIRIxM1dJyBzG3Xq/H1q1b2SOPPBKUqy6sgRg7dmy/FgEx3IwZM2A2m0W1KcQAL7/8Mux2e8hAmMfjgVqtxnfffcc2b94s2imT/3by5MmggsffurAjV8jEiRM5srqE2Socx+GTTz5BZWUl02g0sq1NKUOvoaGBvf/++0EIAYwxZGdn8zvrwQapPJpIrlkUjdXkyZNF6AnEj8JY0M6dOwddgVDQmeA+hCCa/fEQ/P73v5ftdw4AixcvDiqMFfKRWq1Ge3s7W7duHTQaDbRaLZ9YoNVqMWvWLD4+uX37dpx44omi46STTsKJJ56I8vJyJmw2R8HyQCAAl8uF+Ph47o477pBNE48ldVxRIBGou7sbLS0trKGhgbW0tAQdTU1NrKKign311VfstttuY9OnT4dcGh8thilTpmDYsGH9qkYnwZSRkcHNnTtXtIshU7OhoQG33norr9QoWEYMq9Pp0NXVxW688UbZ6zPGcP3114uU0m/dAiEk3muuuUa0eIS5/ddddx0PY0JQEDSmFJS8/vrr0dvbK/LP0+9XXHEFjEZjWATdY8kSkbMC7XY7j6ggBJ6kcfd4PHj99dcHjTfpvtu2bYNGo+Ezv6LpWR5qg8cYw+zZs4OKcMk19cknn2DFihWMqvqJj4S90m+//XZ0dXXxlfzUQM3r9eL888/n71dQUACdTieCa6GN7gcffMArKikwKhHJKTklmJWVpSiQgRAt9H/84x8oKCjAmDFjUFBQEHQUFhZi7NixOPvss7FkyZKQAVQSJH/+85+jMg3DMT1jDPfddx/PsMTsZEIvX74c8+fPZ+Xl5UyYhhwIBLBu3Tp22mmnYe/evSLYDnIjjB49GvPnz+fC9YD/rRGN480334yUlBTZhb9+/XqceeaZ7IcffmDke6Z8++3bt7Ozzz6brVq1KsiiI4yn2267bVCKAo92ysjICBLO9Pdzzz0HjuOg1+tFmGUEFXPttdey6urqkAgG0SgLqeXBcRyefvpp3H///ez9999nn3zyCVu+fDmrr69n/d3gWSwW7t577w1yU9H8X3vttXjiiSdYb28vE5YC1NTUsGuvvZa9+eabsnw0YsQIzJo1i6O4yrBhw7hx48aJ5ILP5wPHcfjHP/6Bzz77jOl0Ov76JAsMBgMaGxvZE088IYv9ptPpcNxxx0XlwlZiIBHI4/FElcZJEyQ1TWlX4PF4cNlll+F3v/sdnx7Xn10U3WP8+PHcX//6V/b3v/9dlM1CAu+dd97Bhx9+iJKSEpaVlcX3Ati3b5+IKYV+TwB4/vnnIeznPRRTJY+UFZKcnMwtXryYXX755bylQYtVpVJhw4YNmDx5MkpKSlheXh4A4NChQ/jpp59kx5T6aDz55JPIzMzkjqUxleNbAHwygpD3SZC/8847cLlc7Nprr8XIkSOh1+ths9mwc+dOPPfcc/jhhx+g0WhC1iiEq7OhDDi53bbdbscjjzwi+uybb75BTk5OvzYjfr8fCxcu5N566y22efNmvh+I0K11zz33YPHixSguLmZGoxEtLS3YsWMHHA5HkIIkPnr88cdhMBhElu+tt96K6667DlqtViR7nE4nfve73+Gqq65iF1xwAQoLC/keMOvXr8fSpUvR0NAgu4mcPXs2cnNzlTTewRIu0ZrmwkknAUIw0WeffTb+/e9/c4OBLUVK5KGHHuJ2797N/vvf//IMRM+hVqvhdruxdevWoPeRMg3tXP75z39ixowZR6SZ1NFghfj9flx22WXcTz/9xJ588kneNUUHLewdO3Zgx44dsucLNxNerxe33HILbrjhhmNyTKU8GwgEMHXqVG7ixIls586doiQTsqQ/+ugjfPTRRz8LJ0kSirQxk5y7OZzikuJ8SdeqMD1YDjQzWnlBcdF33nkHJ598Mg4dOhTUoIzczQ0NDWH5iJTHwoULceGFF4r4KBAI4JprruHeeecdtnr1ahGUCb3ja6+9htdee012PIWKij6zWCx4/PHHo64/UlxYhyctXJV5NAf5TgldloSyWq3G3XffjU8//ZSj4jC5iYlU5S5lUrrH22+/zV155ZU8GBoFICkATs8lvBaZ1kJmfeaZZ3DHHXdw9MyxjlMooRHL92M5PxyGVixjKbd4n3jiCe6+++7jffF0Pi0q4XOR71n4PeoJceedd2Lp0qUcWYWxjGl/3CexjtVgPUc080QCTaPRYOnSpXysgyBKpHxJbmSVSgW9Xg9KYCguLkZGRoaIr+kgOBpptfZhHCrummuu4S1r6ZoQ9uOQegZiHVtam7m5udzatWtRXFzMC3aSD7QhET6LkI9o8+L1enHDDTfgueeeC+IjkiXvvPMOTjrpJL6dL8VSVCqVqJ0DubeECNj0Hj6fD0ajEe+++y4KCgq4qN2tx3pDqc7Ozn73Eg53pKens2uvvZZt376dSe8pbShVXl4e8jrUCEiuuYswMLZs2TKWm5sr25SIOhfKdas74YQT2IYNG0L2QRY2lMrIyJB9xhEjRsg2lBozZozs95OTk5nb7ZYdE+G73nLLLREbD9Ez08/Vq1eHPOe+++6LqlEOzcvHH3/MioqKZPtWUyMxuTEdO3Yse++995iwg6TcmLrdbiQlJck+69ixY1mo8REe9N4ffvhhyPf+//6//0/2vQf6HHS9hQsXhrz31q1bRd0j6ecHH3zApPeMNK5XXnkl6+7uZpMnT5a9V2ZmpixfUSKJ3W5nl1xyScw90ZctWxbyey+//HJInqJ3tdlsbNGiRUyv18t2fJTrLAqAZWVlsZdeeikkHwnf0+l04u6772YmkymkDKD7yHWfPPHEE9m2bdti6oeuNJQ6bBbPnj0bNput3/AIHMfBYDAgLS0NhYWFOO644zB58mQkJiZywrhEqOCh2WzGWWedJcL9p93wpEmTQrrShIVK119/PXfRRRext956C++99x5++ukndHV1yfqLc3JycPLJJ+Oyyy7DnDlzokLh5DgOZ599Nu83FZrJw4cPlz1nxowZyM7ODmpYQ/3PI7kNi4uLccYZZ4gsJbqGFM2XfqampuKMM84QzSWdP2bMmKjckrRrnTNnDjdz5ky899577O2338bWrVvR2toqO6ZpaWmYMmUK5s2bh4suuogjX3W4MVWpVJg5cyZaW1v556WxirboUAiMecYZZ4jcEvTekYpC+/sc0cwTgXsK6zn8fj/OO+88rrS0lD399NP4+OOPUV1dHTSuarUaw4cPx/Tp03H11Vfj5JNP5gDgsssuY3FxcUH3OxyTYrSDlvKGyWTi3nnnHdx4443sv//9L8rKytDc3Ay73c7PlcFgQFxcnAiiftiwYUHvR79T86VQlgj1tlm8eDFuvvlmtmLFCnz22Wc4cOCArDvOarViwoQJuPDCC3HllVdGbEInBFV84oknuBtvvJGtXLkSq1atwv79+9Hd3S3LrxqNBjk5OZg2bRrmz5+PuXPn9guNlzsWA3q/FFFw8JfIvJFOfFtbG6uurkZTUxP6+vqgUqmQmJiInJwcDB8+HGazmRPGb4717KBoxrS7u5tVV1ejoaGBb2gVFxeH7OxsDB8+XAS5cazHPGIZW5fLhfLycnbo0CHYbDYermfYsGEYPnw4R/EI2oT0twZECilEz+FyuZjAdcRFg1kW632FFd2MMVRVVbGamhp0dHTA6/XCZDIhPT0dw4cPR2ZmZsx8JL0HyYC6ujo0Nzejp6eHbxCWnJyM7Oxs5ObmckLMu36VFigKZHByyqX9OSK10Yz2GWJZMGTqRoOjI9xJDcY4yd0v1u+HG9Nozw93Tn+EjxDoL9IzC6uOY0m+GMgYDdZ7D+Q5+jNPwvGKxINSXh3omJFrKdr40GDxFLl3w/X5EK7jaJ9P7h7RnistIo7Z+6IokN8mCYOBQsUmBwutUGxCWrpZUMb06B3XUF0Wf433HYggj/Y+gz22igJRSCGFFFKoX6Q4vhVSSCGFFFIUiEIKKaSQQooCUUghhRRSSFEgCimkkEIKKQpEIYUUUkghhQ6TAqaI0HneSmqmQr82Hyo8+Ouv41B1J0Oh+DbUO/9Sz6ak8SqkkEIKKaRYIP3R3hzHoaKigrW2tkKj0fD/8/v9GDduHOLj47looY0VUijUjjXcLpj4q7y8nLW3t/OVyj6fD8nJyRg9erTCfFGs497eXrZ7924RcrLP50NKSgpGjRrVr3XMGENZWRlzOp38rp6qySdOnMj1F/b9SL0zfVZcXCyCK1IUyBGchE2bNmHjxo2wWCw8bIXL5cKiRYsQHx8PRYEoFAvF6j4g/tqwYQO2bdsGs9kM4OdGR5MmTcLo0aMVHoxi/FpbW/H6669Dr9fzkDJ2ux1TpkzBqFGj0F8F8sknn6CpqQmEkeX3+2EwGDB69Gim1Wp/lQ1mqHcmBXfPPffAbDYfcb5RYiAA9Ho9LBYLzGYzr0Cot4dCCsVCXq8X27dvZwSkyXEcvF4vxowZg9TU1LDChvjQZDLxVovBYFAGNUpSq9WwWCzQ6XQiTDIhYGB/yGQywWKx8E2mSIEMBYUufWdSIL9UDESRkICojwUpEGkDeoUUimZH6HK52Pvvvw+n08k39unr68P111+P1NRUxZL4hdaxUJgOdB0LZYPw76H8zr8UKQpEIYUGkTiOg9lsDupYF42v3O12w26383/b7fawfb4VUujXJkWBKKTQIJPQko3GmiUlM3LkSJHLxe12h2zWpZBCigJRSCGFeAUyY8YMbsaMGWG/o5BCigL5DVA0fQTk8P6l/Q6G2r0iPYe0r4Bcn4H+vlN/BGV/rxOqaE96zWjeSXitUM8TrgeE0HIRxkjo9/5kdUnfoT/vNRh8ONB7/lZkQ6y8Pdh9So6UfFAUyAB3jeEmX25ihH+HEiSx3itUFzy5xd0fRqGsjkjvI/1+f99psOYh1rkJdc1IrVSFn2m1Wg5A0OqngHqkHumDIbhCPWus7zUYPC/Hg791i2ow3m0wm0oNpixSFMggkM/nQ29vL5MufmnRYX19PTtw4ABaWlrgcrn4lLuMjAwUFhYiLS2NiyTYGWPo6elh0h1JXFwcJxTqHR0dbP/+/WhoaEBfXx84joPJZEJaWhpGjBiBvLw8jnaEsVgJtANmjOHQoUOstrYWra2tsNvtfNtNi8WCtLQ05OXlITc3lyNBGOpejDF0d3cz6f+0Wi2sVmtMHGyz2ZjP5xPtqkJdR6jYbDYb3y+6q6sLDocDPp8PKpUKRqMRSUlJyMrKQl5eHiwWS8h58vv9sNlsjD5zOByylo3NZkNXVxejZ2CMwWq1igrR+vr6mMfjEb2LTqfj7x/NXAFAe3s7q66uRlNTE2w2G+ia9F65ubnIz8/njEZj1Arf6/Wit7eXCZ9No9EgLi6OE55fW1vLysvL0draCrfbDbVaDavViqysLBQUFCAlJSUizx/t5HA4mMvlioon5cjlcsHhcDDh+KhUKsTFxXGxWjE0L3V1dezgwYNobm4WyaL09HSMGDECWVlZ/ZoXRYHEqM05jkNTUxNbunQpn1nj9XqRmpqKRYsWQavVorW1lX366afYv38/3G43v3sXmpEGgwHjx49ns2fPRlJSUlB9AP3tdrvx/PPPw2az8ZXyAPDHP/6RZWRkcE6nE6tWrWI//PADrziEjEtCKD8/n82aNQvDhw/nohEY9B2/348tW7awLVu2oLm5mRdGdAjdMjqdDllZWWzatGmYOnUqp1KpgoQTfXflypWora3lC6ACgQCMRiPuuOMOZrFYuEhK9bBQZk8//TQ/xiqVCn19fTj33HNx1llnie5Ni6m+vp598803KC8vR29vr2gXLn2fw5sCjBs3jp1xxhlITk7mhAqA4zh0dXWxxYsXi5QG9aWn6xgMBqxatQpffvklr1DcbjduuukmNnLkSI4U8YcffogdO3bwdSAOhwMTJ07ElVdeGVYZ07MfPHiQbdiwAZWVlXA4HKKdp/C91Go1EhMT2cSJE3HqqaciISEh5HjT+1ZXV7Nly5bxdSlutxvDhg3DH/7wB6hUKjQ0NLBPP/0UBw8ehNfrleV5k8mECRMmsNmzZ8Nqtf7mEB5orL766its2LCBL+SjZIiFCxeGFdB0/tatW9lHH30Ei8UCxhi8Xi+SkpJw5513Rl2bRnxdV1fHvvjiC1RUVISURXq9HsOHD2dnnHEGCgsLY5oQRYH0U5EId70+nw8ejwdarRYHDhxgr732GhwOB8xms6iISThpgUAAP/74IyoqKnDdddexYcOGhVxQPp8PPp+Pv0YgEIBer0dXVxd76aWX0NDQAIvFgri4uJB++crKSixduhTz5s1jkydP5qJh5IaGBvaf//wHNTU10Gq10Ol00Ol0QeawUDg1NDTg7bffxrZt29gll1yCjIwMTk6QFxUV4cCBA9BoNPz92tvbUV5ejuOOOw7RKJDy8nJ0dnbCbDbz46PValFcXCwyx+n6a9euZatWrYLP54Ner4fJZArpwqH3cTqd2LRpE3bu3Il58+axCRMmBOHH+f3+iLUGwtx8gsqRu45wrn0+H/x+f8QNjcfjwYcffsi2bt0Kxhj0ej3MZrPsu9F79fX1Ye3atfjhhx8wZ86ciDxBPC98Nq/XC47jsHPnTrZy5Up4vV6YTCaQZSN970AggM2bN6OiogI33HADS09P/03CBAnnkcYt3DyGG+v+nB8IBGA2m7Fz5072xhtvwO/3w2g0hpRFjDGUl5ejvLwc06dPZ3PmzOGidWcpcO4D8FEKF6jJZEJ1dTV75ZVX4Pf7YTKZ4HA4YLPZYLPZ0NvbC4/HI1pUFosFfX19WLFiBfr6+pjQFxnuXjqdDjabDa+88gpaWloQHx8Pt9uN3t5e/l4ul4u/F2MMRqMRGo0Gb7/9Nmpqahill4Yye8vLy9mzzz6L+vp6WK1WHsYhEAjA6XSK7uV0Ovnn1ul0sFqtqKmpwb/+9S9UVlYy2o0LmXLChAmwWq2i+I1KpcLu3bsj+mLps7179/LnqtVqeL1e5OfnIyMjgxdMdP1Vq1axDz/8EFqtllccZD329fWJ5qmvrw9er5d3HVgsFvh8Prz66qs4cOBA0Nj5/X7REWpRS78XCnlXeoRTHna7nb3wwgts48aNMBgMMBqN/DlCnrDZbOjr6+MVALmW3G433njjDXz55ZcheSIUH5rNZpSXl7MVK1ZApVLBYDDAbrfL3o+ua7Va0dHRgddee42vcTnSBbv0vP2JMfXnnFjmMZZrRKt8DAYDtm3bhpUrV0KtVsNsNgfxgsPh4HmV5IPRaMRXX32Ft956i0mVv2KBHEEizJ23334bjDE+NlBUVIT09HSo1Wp0d3ejsrISHR0dMBgM/PeMRiPa2tqwYcMGzJo1izc9I93vvffeQ3NzM7RaLVwuFwoKCpCdnQ2dTge73Y7a2lrU19fzu45AIAC1Wg2Px4NVq1bh5ptvDhlkb2hoYMuXL+chG8glQ77TkSNHIisrCwaDAS6XCw0NDaipqeF3v/ReXq8Xr7zyChYtWiTabTLGkJiYyI0cOZLt3r0bRqMRgUAAOp0OVVVVsNvtzGw2y+5OBbEGVl1dDZ1Ox1sYPp8PEyZMEO2wVCoVDhw4wFatWoW4uDjegqN4RXx8PMaMGYOUlBTodDq43W50dnaitrYW3d3dMBqN8Pv90Gg08Pv9+OCDD3DnnXfy7kuVSoXExETRbq6vr08kFBljMJvNPNwEWQ0DBeLz+XxYsWIFKisrERcXB4JP8fv98Hg8yMzMRH5+Pv9ZW1sbqqqq0Nvby78X+cI//fRTmEwmdsopp0Tl4tRoNOjs7MS7774LrVYLj8cDs9mMkpISpKamknsPFRUV6O7uFvG82WxGXV0dNm/ezKZPn85Fw/MDtQg8Hk+/ID6ONkQKmv+1a9dCrVbD7/fD4XAgOzsbubm5sFqt8Pv9aGlpobUGk8nEK4v4+Hhs3rwZKSkp7Oyzz47IC4oCGQR3llqtRk9PDy/EioqKMGfOHKSmpopWhdPpxMcff8y2bt3KC81AIACDwYBdu3bh7LPPjujjJOHT2toKxhhSUlJw4YUXYsSIEZyU8Tds2MA+/fRTXnDRvWpqatDS0iLrQvD5fHjnnXfg8XhgMBj4RedwOJCfn4+5c+ciPz8/aLVXVVWxjz76CHV1dfy7abVaOJ1OvPPOO7yvXKgESkpKUFZWJhJK3d3dKC8vR2lpKcIpEBJMxPx+vx9WqxXjx48X7ToZY1izZg2f/SSMLZ122mk4/fTTERcXF/Q+vb297PPPP8fWrVv5cdDr9WhpaUFFRQUbN24cFwgEkJiYyN11110ii+Dpp5/mle3h/+HCCy9ESUmJaEHSXMcqPOkaa9asYfv27UN8fDyvPHw+H3Q6Hc4//3xMmjQpCC22q6uLffXVV9i8eTOMRiMvHC0WCz755BOMGDGCZWdnR3RnaTQadHV1QaVSwev1YtKkSZg1axYSExNFJ/X19bH3338fO3fuFPG8TqfDjh07cNpppx0x5UFzVlFRgSeeeKLfWsDhcIiQuo8GmaTRaODxeBAfH4+5c+di/PjxnFQRtLW1sc8++ww7d+4MWkdfffUVioqKWFZWVlheUFxYg0QajQZutxvFxcW49tprudTUVE6IoUNB4nnz5nG5ublwu938blytVqOrqwvt7e0sGpOeXEIpKSlYuHAhRowYwcnheU2fPp2bPHkyhFDUhDRcV1cnch/QOVu3bmU1NTX8YlepVHA6nRgzZgxuueUWLj8/P+hejDGMGDGCu+WWW7iCggL+foFAACaTCZWVlfjxxx95Fwkx45gxY5CUlCSKJ3EcF5Uba+/evfyzq1QquN1ujBw5kg8I0/kdHR2ssbGRD/7SPE2bNg3nnXceFxcXF/Q+gUAAVquVmzdvHpefnw9hZlQgEEBTU5NIoVN8SKvVhrQqNBpN0Pf6I4zIquro6GAbNmyAxWIRWR56vR433XQTpk2bxmm12qD3SkxM5C655BJu9uzZ/DzRNb1eL1atWhWTkHK5XJgyZQouv/xyLjExMYjnLRYLd/nll3Opqal8zITObW9vR09PD5PWiwz2jtzr9aKnp6ffx1DBvYr1na1WKxYuXIji4mI+oUV4pKamctdccw134oknwuFwiGSEz+fDmjVrlBjILzVhPp8PFosFF110kWiXKDzI5zh58mR+MQmtip6enqh9wn6/HxdccAEsFgtHAkR4L3LhTJs2TdQrgK7f3d0dpJT8fj82b94MvV7PC3qv14uEhARcccUVIIEkvRcpBr1ejyuvvBJxcXG8UiBL5LvvvguC9zCZTNzo0aN5ZUo708rKStjtdlnBQsqisrJShEDKGENJSUlQgPDQoUNob2+H0+mE3W5Hb28vfD4fTj31VJE7S26uGGMoLCwUzRUAPpYlFajh5i7S57EoEADYsmUL7Ha7yL3g9Xpx8cUXIzc3lyNek74XvfOZZ57JlZSU8IKDNjiHU8EjCnXijeTkZJx//vm8EpYbR41Gg0mTJvHzLJxHm832i6xPQtfuz3E0ks/nw4UXXojk5OSQvECbvwsvvJAbNmyYaB0aDAbs378fnZ2dYXlBUSCDxKButxslJSWwWq0h/Yb0v5ycHGi12qCKXQrcRuPCysvL45vkyBWnkaBOS0vjhP5xOSFIO6xDhw4xiqsIXT0zZsyA2WwO6w8lhrRardxpp50GyoOn9N7GxkZZwVRaWgphkJ3cWAcPHgwSuPScNTU1rKOjQ9R4KSkpCaNHjxa9OwBkZ2fjyiuvxLx58zB//nzMmzcPV155JZKTkzlplTcpHmFF+FADMyShvHfvXlFMxe12o7CwEBMmTOAo3hWKf+hdzz33XH6zIOQtcitGUiButxvHH3+8qP9GqPvl5uYGPZOQ5490nEGKBBDLcTTKohEjRmD8+PFheUFofc6cOVMkIyiue+DAgbDzoyiQQZy4wsLCqBiO8Pv7YxrTzm/kyJFhJ5YYwWAwwGw2R5UGWFlZKdpt+/1+xMXFYeLEiVH5f0k5lJSU8K4VoWCqrKwUuWEAYMSIEVxGRkaQRRbOjbVnzx6RNeN2uzF69GiYTCZOWl2dnp7OnXzyydzUqVO5adOmcSeccAJ3/PHHc2q1OggGm85TqVTQaDRoaWlhZWVlfAB4KPi2AaClpYXvXCjsoDlp0qSolRAApKWlcSNHjuTrA8i1VFVVFdGFSN8tKCiIyBPAz9lXlLKt0JH3hlAySbS8MGrUKC4jI0Pkrj1c/xPeda8M+eAsbI1Gg/j4+Kj82mq1mt+x95cSEhKiZqhw8BlCamxsFPlBPR4Phg0bBooTRAu5kpiYyKWnpzMqFKTPhLEDsig0Gg2Kiorw5Zdf8jtqcmM5HA5mMpk4YUW8z+fDwYMHRVaSSqXi3VeRdp5Ct5vwfbxeLxwOB7Pb7Whvb0dlZSV++uknOJ1O0b1+bT7jOA4tLS1wu918AzRyP+Xn50cU/NJrjRgxglfWZAF2dHTA6XSKguxy5+t0Or72KNI9hdD2v6S15nQ6MW7cOPz+97/vdxbWihUrILR4h7os0mq1yMvLi5oXaB3m5+fznRfJs9Ha2ipSNIoCOYKa/5dcIIPZcYyuRZlkwl1tWlqaSOBEw4wqlQppaWmoqqoS1WJQjEdoaQDAxIkT8c033/BCntxYFRUVmDBhgkj4Hzp0iLW1tfF1KR6Ph+BaOLlxkasBaG9vZy0tLWhpaUFHRwc6OzvR1dUFu90Oj8cDr9fL59MLXY1Dhbq6ukRWk8/nQ0JCAhISEqIuACNKT08PKvJ0uVzo6+tjRqORi+TG+qU63w1EoBqNRh42qD+k0WjY0eDKoviF0WhEfHx8zOdnZGSIeJ3cWG63m3dTSnlLUSAK8QvN4/EExQQsFku/rkcwDEJmpOJGoQJhjCErK4vLzc1lNTU1vMXCGMOuXbuCTPG9e/fC6/Xy3/N4PBg/fjwf4JcKNGJ6n8+HrVu3sh9//JHHAyJ3l1qtFoEdqtVq+Hw+uFwu6PX6IZe6SeMoVNpGo5FXqrHOkzDJgsZKLlHgaCVyUx4LdSBkgej1eq4/vCAFwjy8oWKhrqcoEEVxiFw9g2XpRHse7XonTpyIiooKPt6g1+tRWVkpcqUEAgGUl5eLrAKdToeJEyeGddO0trayN954A7W1tXwKrclk4pUmwUZQ7EOn0yE9PR2lpaXYvn07WlpaBlz0N9gCMZT7MJbdajjL+bfWznkgFeFH23rurzdEunkMJxcUBaJQkItHzl3jdDr7dV2HwxGElyUnhOk7xcXF+PLLL/lMECpUq6ioYMXFxRxVyDc3N4vcVzk5OcjJyeGkWUD0Hr29vWzZsmVob2+H1Wrlha/dboder0dubi6ysrKQnp6O5ORkxMXFwWKx8O6gsrKyIee+EGIa0Ri63W6+sjwWQeNyueD3+/m5oXEcSgpTodiUwGGcMmYwGGLSIkLLltYlWeSKAlEoojCxWCyiQj+1Wo329vaYdmL0vY6ODlFA/nBRGb+DFn4mhDbZtWsXj0bLGMPu3bt5cMR9+/bB7XaLUJCLi4v5hASpwjqMgcXjhVFtitvtxqRJk3DGGWcgMzOTO9rcF3FxcaJ0XLVajb6+PvT29rJwyLpy1N7ezittIVoBzYFCv55XoL/r2Ol0wmazwWq1xnS+MLZG/H8YhDFkbE1J41WIZ5q0tDQRwJpGo0FTU5OoACwaBnY4HKJ6EmLG9PT0sPcn+BL6n16vx8GDB3kraN++fXzqKjG3FHlXqDzsdjvbvXs3TCYT76JyuVwoKSnBFVdcwZHykFbVCyvchyKlpaWJoP1JgTQ2NkZdu0DjVVNTI1Lmfr8f8fHxfA8SpZXu4K6xaMaToPj7a4F4PB7U19fHzAsNDQ28tUG8kJCQEFSIrCgQhWQpPz9flJGj1WrR0dGB8vLyqNA5ickOHDiArq6uICFHaaahGHjMmDEcQZsA/wPsO3ToELPZbKy+vp6Hk3e73cjPz0daWppsLxUAfOMraZYRVaHLVehK+6kMpboFeq6MjIyg4lDGGHbs2BG1oj/sxmMHDx4UFST6fD7k5uYiEjKvQqFJyPfEX5S0Ec38tra2DmjzQqjWsfBCT08Pq62tFdWn+f1+ZGdnh7WKFAWiEM9oh7GkeFcPCdyvv/46onkttDS++eYbUZEbNcQZPnx4yFRbsijGjBkjsngCgQAOHDiA/fv3izC9/H4/HzwP9UwEWS28lsFg4FMcQy1SskR6e3uZVBH+0m6JUOM0cuRIvuiL/rdz5040NTWxSDVG5O779ttv0d3dLXo/juNQVFSkLIoBkNlsFs07ga329vayUFYB/c/pdEIqyGMh4vF9+/ahpqaGCSGUwvHCxo0b0dfXJ4p3qNVqFBYWhldWynQrJMSmmjhxIlwuF18PYjAYUFVVhVWrVjEh7pUwNVKIJ/XZZ58xYadBwjwqLS0VwWaEIiG0Cd1/z549oH4XZDnExcUFIe/KvZe0LajL5UJPT4/oPaTvQu/5xRdfwG63ywYRw8VH5BCEhf1Z6BgITZs2LSjlUoikLATPk4JFqtVqVFRUsK+//poHzaSUzezsbBQWFnKhoEkUikwpKSmisVOr1ejt7cXevXt515B0XmiztW7dOkZKfaBr+p133oHdbmdS1AU5Xli/fn0QL6SlpYXc9CkKRKEghmOMYfr06UFgiCaTCatXr8ann37KqF2psAkU+V0/+ugjtm7dOh4aWgjGSG6jkLDQhxl0+PDhPLQJLb7u7m6+9wm5rwoKCkJWyNPfCQkJspll//3vf9HR0cEIEUD6Lna7nb377rsi2H0hUX8UOcWi1Wo54T0plrRz5054vV5oNBr+Pv11TzDGMHLkSK64uJhXcKRsDx06hGXLlrG2tjYm924qlQo7duxgy5cvF40VzeGZZ545KBbXsWzJ5+TkiPiG4nlffvklGhoaGPGAdF7Wr1/Pvv76a1F/jv5auzqdDq2trXjxxRfR2NgYkhfKysrYq6++Ktps0abvhBNO4OurQrrrlGlXSGiFJCQkcL///e/Z66+/zmdxkHBas2YN9u/fz0pLS5GTkwO9Xg+Xy4X6+nps374djY2NfH0FMaTL5cL8+fPDgkwKd/Vy0CbCPs7E5ELk3VAKJD09nUtNTWVUx0ELq6GhAUuWLEFJSQnLz8/nYWF6enpQU1ODXbt2obOzk38X6cLct28fRo8ezbRaLTIzMzlhsaFer4fFYkFXV5dIeFRVVWHx4sUsOzubD3qfffbZfApyf4TE3LlzUV1dDafTybs8jEYjqqqqsGTJEpSWlrLDipZvKLV7924+GYEUj0ajQU9PD6ZOnYqSkhLF+hjgJiwpKYnLy8tjBw4c4BWJWq2Gw+HA888/j6lTp7KCggIYjUZ4PB60tLRg586dqKqq4huyhQtcR0NerxdGoxENDQ149tlnUVJSwkaPHo34+HgEAgG0t7djz5492LNnD5+qS/PudDqRm5uLE044ISIvKApEIdHuNhAIYPLkyVxHRwf7/PPPYTab+f+bzWa0tLTg448/5nfS1J5Vp9Px2Ez0/b6+Pvz+979HaWlpVF3uhMrhm2++Ee18pPGU0aNHh80SokU7Y8YMvPrqq7z7jJSAy+XC+vXrsWHDBt6SIAh3UgIul4v/PrnUtFot2tra8MILL4DjONx9992MWujSYsvJyUFVVRWMRiPf11qn06GlpQX19fW8i+Dkk08esKC6+uqr2bJly+B2u/nukQaDAV6vFxs2bMC3337LWxSUiUa9UcjC6+npwZgxY3DRRRdxiuUxMCI+Pf3007Fv3z7RfGk0Gvh8PqxduxZff/01yLVETcCoV0xxcTG2bdsWsxtL6IY0m83Yu3cv4uLi4PF4sGnTJmzevDkkLwh7wmg0GsyfP1/ULiGkzFCmHLL9IGjXG07YSo9YhXW094v12QZ6r0AggHPOOYe76KKL+JaYwh221WrlcaKMRiOsVqsINNHhcIAxhnnz5uHMM8/kooWQoIWWmZnJDRs2DD6fD0I3E/U9HzNmDN8lMNx7MMYwadIk7vTTT0d3dzeE0NbUytVkMkGv1/NKw2q18u1Yc3JyMHPmTLjdbh4MkASBwWCAXq8XvRc9y0knnQSj0cj32qDxprGzWq08hEgkPgwX3wkEAhg5ciS3cOFCJCcno7e3V5T1Ru93GNqC/5usQ4/Hg76+PkyZMgXXX389J5zDaNfIkeL5X2od9/e5w7U1ONxHhjv77LPR09MDYcsFlUoFi8UCo9HIr5/4+Hi+6+cFF1yA8ePHw+v1ing/1P3k3lmtVuPiiy9Geno6enp6oFarYbVa+bbKer0eZrMZRqNRdA2n0wmNRoMFCxYgNzc3qlbDigUCwO12o6+vj9+5kuuF0knldrd2u12E0urz+aL2WzLGYLfb+TgDgZbJ3Y8xBofDAbvdzuM19fX1RdU7hMjhcKCvr4/Hderr6wuLdUSL4NRTT+VGjBjBvvzySxw4cIAXpMTYwuIzskQMBgMmTJiAmTNnIisri4sVf0gIbVJWVsYXNwp3WKGQd0MppPPPP59LTExka9euhc1mE2Ff0T1pJ0ixk5NOOglnnXUWd7jdMKuqqhJl1/j9fni9XlGGC90vIyODu+6669h///tftLS0BPViIZ6TZscQHwrSbMP2IyFln5eXx91+++1Ys2YN27ZtGw+KSXNFz0UJCBTDyc7Oxumnn47S0lIulDuQyOfzoa+vj7fShO8bDdGaEdacyI3BQMjv9/N8LqgFGnBPF1p/VMBKYxguiYIxhlmzZnEGg4GtWbMGvb29smvH7XbD5/MhNTUVl19+OSZMmMBt3ryZ2e12UZ8gEvbh3plcVz09PUhMTOT+8Ic/sA8++AC7du3iN2NSVGS/38/LocLCQpx//vnIzMyMftN3LJustGAqKytZa2trUH+FsWPHIj4+PihQ29fXx/bs2SNCrg0EAhg/fjxfgBVJYe3atYsJBaPP58Po0aORlJQkup/f70dZWRkTAh1SPxC5nuZytGvXLibMJvJ6vcjNzUVubm7Y84VMVF9fz/bs2YPa2lp0dXXxee1kBicmJiI/Px/jx49HdnY2Jz0/1jlxOBxs165dQWOs1WoxceJELlrIDuE1u7u72U8//YSDBw+io6MDbrebdy9ZrVZkZGRg5MiRGDVqFKxWKy9UbTYbW7VqFaqqqvhe51arFcOGDcM555wDs9nMyd3P7/ejoqKCNTY2wmazgRIQjEYjzGYzSkpKRIkABw8e5Pt8kNBOTk7GqFGjuGje73Ach+3ZswcVFRVoa2uDw+HgBYRWq0VcXByys7MxZswYjBkzhhOOb7hr9/T0sH379gUBLxYXF3NCl1gYIcyEsPE0PuPGjYu6XUCk9+/t7WXk0xdu7FJSUvjMsljvcRjUkwlTyMmimDBhAhcO8oXu19nZyX766SdUVVWhs7OTXzsajQYpKSkYP348Jk+ezBHmW3t7O6uoqOD5gLp9TpgwgRM+g/Sd6btGoxFFRUX8d2tqalhZWRnq6upgs9ng8Xh4a9xsNiMrKwsTJ07EuHHjYl63is9ToYgLSOrW8Pl8cLvdjJhQp9NxQn9tLFW3vxRJF4XX64Xb7WaHFQgnxZeSW0Rerxcul4up1WqYTKaohfovNU9yeGCHLWnGcRx0Oh0nReztj5JXqP98J1w7Wq1WpHwHey7k1qHX64XH4+HXrtFo5KRFuLHwraJAEBpxMhyqpZy7KpbJD4WoKne/WL470HtFGqNw6K0DQQKNdk5iHedYnlFoEUoXlVw2SjQLPpxbM9R9BmOewo0TPVOsrsWBojUPBh8O9jru73PH8u7h+E7us1jGOprvRsv3/VlXigJRaEC7m6FmaRzpdxiK1tVvfZ6UtTN0768oEIUUUkghhfpFivNTIYUUUkghRYEopJBCCimkKBCFFFJIIYUUBaKQQgoppJCiQBRSSCGFFFJIUSAKKaSQQgopCkQhhRRSSCFFgSikkEIKKaQoEIUUUkghhX7D9P8DMydEZ4Gi4a4AAAAASUVORK5CYII=" +_now = datetime.now(timezone.utc).strftime("%Y-%m-%d") +display(HTML(f""" + +
+
+

Xenium Image QC

+
Date: {_now}
+
+
+ Bioinformatics Innovation Hub +
+
+""")) +display(HTML(""" + + +""")) +``` + + + + +```{python parameters} +#| tags: [parameters] +#| echo: false + +# Nextflow QUARTO passes INDIR plus samplesheet-derived paths (see modules/quarto.nf). +INDIR = "image_qc" +SAMPLE_NAME = "" # samplesheet `id`; if empty, inferred from INDIR (parent of folder named image_qc) +XENIUM_BUNDLE = "" # samplesheet `xenium_bundle` path string (local or s3://) +SAMPLE_PUBLISHED_OUTDIR = "" # expected publish dir: params.outdir / id / image_qc (set by Nextflow) +ROI_THRESHOLDS_YAML = "" # optional path override; defaults to conf/roi_image_qc_thresholds.yaml if found +PIPELINE_VERSION = "" # nf-xenium-processing version from workflow.manifest.version; empty for local renders +``` + +```{python bootstrap} +#| echo: false + +from __future__ import annotations + +import json +from pathlib import Path + +import numpy as np +import pandas as pd +from IPython.display import HTML, Markdown, display + +base = Path(INDIR) + +# Analysis status marker written by modules/image_qc.nf. On a hard failure of the +# Python analysis (e.g. a very dim sample where no tissue tiles clear the intensity +# gate), the analysis exits without writing roi_qc_metrics.json. Rather than raise +# and produce no report at all, we render the report with a prominent QC-FAILED +# banner and let downstream sections fall back to their empty-input defaults. +status_path = base / "image_qc_status.json" +analysis_status: dict = {} +if status_path.exists(): + try: + with open(status_path, encoding="utf-8") as f: + analysis_status = json.load(f) + except Exception: + analysis_status = {} + +roi_json = base / "roi_qc_metrics.json" +ANALYSIS_FAILED = analysis_status.get("status") == "failed" or not roi_json.exists() + +if ANALYSIS_FAILED: + roi_metrics: dict = {} + _sample_label = SAMPLE_NAME or base.resolve().parent.name or str(INDIR) + _exit_code = analysis_status.get("exit_code", "unknown") + display( + HTML( + f'
' + f'

⚠ Image QC FAILED — {_sample_label}

' + f'

The image QC analysis did not complete ' + f'(exit code {_exit_code}), so no metrics are available for this sample. ' + f'The most common cause is a very dim image where no tissue tiles clear ' + f'the intensity gate, so the focus/blur model cannot be fit.

' + f'

What to check: DAPI/stain intensity and ' + f'tissue extent for this sample, and whether the correct morphology ' + f'image was supplied. Sections below render empty because the metrics ' + f'were never produced.

' + f"
" + ) + ) +else: + with open(roi_json, encoding="utf-8") as f: + roi_metrics = json.load(f) + +intensity_path = base / "intensity_assessment.json" +intensity_stats: dict = {} +if intensity_path.exists(): + with open(intensity_path, encoding="utf-8") as f: + intensity_stats = json.load(f) + +threshold_path = base / "roi_blur_threshold.json" +threshold_cfg: dict = {} +if threshold_path.exists(): + with open(threshold_path, encoding="utf-8") as f: + threshold_cfg = json.load(f) + +cell_metrics_path = base / "image_qc_metrics.json" +cell_metrics: dict | None = None +if cell_metrics_path.exists(): + with open(cell_metrics_path, encoding="utf-8") as f: + cell_metrics = json.load(f) + +versions_path = base / "versions.yml" +versions_text: str | None = None +if versions_path.exists(): + versions_text = versions_path.read_text(encoding="utf-8") + + +def load_snr_summary(indir: Path) -> dict | None: + """Prefer nested snr under roi_qc_metrics; else standalone snr_metrics.json.""" + if "snr" in roi_metrics: + return roi_metrics["snr"] + alt = indir / "snr_metrics.json" + if alt.exists(): + with open(alt, encoding="utf-8") as f: + return json.load(f) + return None + + +snr_summary = load_snr_summary(base) + + +def load_roi_cutoffs() -> dict: + """ + Load optional report cutoffs from conf/roi_image_qc_thresholds.yaml. + Falls back to empty dict when unavailable. + """ + try: + import yaml # type: ignore + except Exception: + return {} + candidates = [] + # Optional explicit override from Quarto parameter if provided + try: + c = str(ROI_THRESHOLDS_YAML).strip() # type: ignore[name-defined] + if c: + candidates.append(Path(c)) + except Exception: + pass + # Common repo-relative locations for local and Nextflow runs + candidates.extend( + [ + Path("conf/roi_image_qc_thresholds.yaml"), + Path("../conf/roi_image_qc_thresholds.yaml"), + Path("../../conf/roi_image_qc_thresholds.yaml"), + ] + ) + for p in candidates: + try: + if p.exists(): + with open(p, encoding="utf-8") as f: + d = yaml.safe_load(f) or {} + if isinstance(d, dict): + return d + except Exception: + continue + return {} + + +roi_cutoffs = load_roi_cutoffs() + + +def fig_html(rel: str, *, max_width: str = "88%") -> HTML | str: + import base64 + + p = base / rel + if not p.exists(): + return HTML( + f"

Figure not found (optional): {rel}

" + ) + b64 = base64.b64encode(p.read_bytes()).decode("ascii") + return HTML( + f'
' + f'{rel}' + f"
" + ) + + +def display_table(df: pd.DataFrame) -> None: + """Render table without row index for consistent report UX.""" + display(HTML(df.to_html(index=False, border=0))) + + +def display_kv_table(d: dict) -> None: + """Render dict as key/value table without row index.""" + kv = pd.DataFrame( + [{"Metric": str(k), "Value": v} for k, v in d.items()], + columns=["Metric", "Value"], + ) + display_table(kv) + + +def _status_theme(status: str) -> tuple[str, str, str]: + """Return (container_css, badge_css, label) for report note styling.""" + s = str(status).strip().upper() + if s in {"FAIL", "CRITICAL"}: + return ( + "background:#FEF2F2;border:1px solid #FECACA;color:#7F1D1D;", + "background:#DC2626;color:#FFFFFF;", + s, + ) + if s in {"WARN", "WARNING"}: + return ( + "background:#FFF7ED;border:1px solid #FED7AA;color:#7C2D12;", + "background:#EA580C;color:#FFFFFF;", + "WARN", + ) + if s in {"PASS", "GOOD"}: + return ( + "background:#F0FDF4;border:1px solid #BBF7D0;color:#14532D;", + "background:#16A34A;color:#FFFFFF;", + "PASS", + ) + return ( + "background:#F8FAFC;border:1px solid #E2E8F0;color:#334155;", + "background:#64748B;color:#FFFFFF;", + s if s else "N/A", + ) + + +def _status_cell_css(status: str) -> str: + """Cell-level status color styling for PASS/WARN/FAIL-like values.""" + s = str(status).strip().upper() + if s in {"FAIL", "CRITICAL"}: + return "background-color:#FEE2E2;color:#991B1B;font-weight:700;" + if s in {"WARN", "WARNING"}: + return "background-color:#FFEDD5;color:#9A3412;font-weight:700;" + if s in {"PASS", "GOOD"}: + return "background-color:#DCFCE7;color:#166534;font-weight:700;" + return "" + + +def _intensity_display_status(raw: str) -> str: + """Map intensity_quality verdict (pass/warn/fail/not_available) to PASS/WARN/FAIL/N/A. + + Also accepts the legacy good/warning/critical values for backwards + compatibility with pre-v5 JSON outputs. + """ + if not raw: + return "—" + sl = str(raw).strip().lower() + if sl == "pass" or sl == "good": + return "PASS" + if sl == "warn" or sl == "warning": + return "WARN" + if sl == "fail" or sl == "critical": + return "FAIL" + if sl == "not_available": + return "N/A" + return str(raw).upper() + + +def metric_cell_with_caveat(name: str, status: str, body_md: str) -> str: + """Wrap a metric label in an inline
when status is WARN/FAIL. + + The icon is a coloured pill with a chevron (▾) to make it visually + obvious that the row is clickable; clicking expands a coloured panel + with the explanation. Sample-specific notes that would otherwise sit + below the table are anchored to the row that triggered them. + """ + import re as _re + s = str(status).strip().upper() + if s not in ("WARN", "WARNING", "FAIL", "CRITICAL"): + return name + is_fail = s in ("FAIL", "CRITICAL") + icon = "✕" if is_fail else "⚠" + fg = "#7F1D1D" if is_fail else "#9A3412" + bg = "#FEE2E2" if is_fail else "#FFF7ED" + border = "#DC2626" if is_fail else "#F97316" + chip_bg = "#FCA5A5" if is_fail else "#FED7AA" + body_html = _re.sub(r"\*\*(.+?)\*\*", r"\1", body_md) + chip = ( + f"" + f"{icon}▾" + f"" + ) + return ( + f"
" + f"" + f"{name}{chip}" + f"" + f"
" + f"{body_html}" + f"
" + ) + + +_SUMMARY_ROW_CSS = ( + "border-top:2px solid #94A3B8; background:#F1F5F9; font-weight:600;" +) + + +def _summary_row_styles(row) -> list[str]: + """Apply a 'summary row' visual treatment (top divider, tinted bg, bold) + to rows whose first column value contains 'Overall' (aggregated rows in + [Section 3.4](#sec-3-4) SNR). The 'Tier ' trigger was removed under v5 Phase 8 — the + former 8.A tier-divider rows no longer exist now that 8.A split into + [Section 4.1](#sec-4-1)/4.2/4.3 separate tables.""" + label = str(row.iloc[0]) if len(row) else "" + if "Overall" in label: + return [_SUMMARY_ROW_CSS] * len(row) + return [""] * len(row) + + +def display_status_table(df: pd.DataFrame, status_cols: list[str], *, narrow_cols: dict | None = None) -> None: + """Render DataFrame without index and with status-colored cells. + + Rows whose first-column value contains 'Overall' get a summary-row visual + treatment (top divider line + tinted background + bolder text) so readers + can tell aggregated rows apart from per-component rows at a glance. + + narrow_cols (optional) — mapping {column_name: css_width} to apply an + explicit width constraint to specific columns. The default cell style is + already wrap-friendly (`white-space: normal`, `vertical-align: top`, + `overflow-wrap: anywhere`) — `narrow_cols` adds explicit `width` / + `max-width` so a column doesn't grow beyond the budget allocated to it.""" + sty = ( + df.style + .set_table_styles( + [ + {"selector": "th", "props": "text-align:left; vertical-align:top; background-color:#F8FAFC; white-space:normal;"}, + # Wrap-friendly default (matches the at-a-glance scorecard pattern): + # cells wrap on whitespace + break long tokens (e.g. + # `low_texture_cell_warn/fail`) at any character so they don't + # force horizontal scroll. Previously `white-space:nowrap` here + # forced side-scroll in long-content tables (8.A, SNR Summary). + {"selector": "td", "props": "text-align:left; vertical-align:top; white-space:normal; line-height:1.5; overflow-wrap:anywhere; word-break:break-word;"}, + {"selector": "td details[open]", "props": "white-space:normal;"}, + {"selector": "td details[open] > div", "props": "white-space:normal;"}, + ] + ) + .hide(axis="index") + ) + for col in status_cols: + if col in df.columns: + styles = [_status_cell_css(v) for v in df[col]] + sty = sty.apply(lambda _col, s=styles: s, subset=[col]) + sty = sty.apply(_summary_row_styles, axis=1) + if narrow_cols: + for _col, _width in narrow_cols.items(): + if _col in df.columns: + sty = sty.set_properties( + subset=[_col], + **{ + "width": _width, + "max-width": _width, + "white-space": "normal", + # Force long backticked identifiers (e.g. `low_texture_cell_warn/fail`) + # to break mid-token rather than overflow into the next column. + "overflow-wrap": "anywhere", + "word-break": "break-word", + }, + ) + display(HTML(sty.to_html())) + + +def render_status_note(title: str, status: str, subtitle: str = "") -> None: + """Render a compact status note with themed background + badge.""" + box_css, badge_css, label = _status_theme(status) + sub_html = ( + f"
{subtitle}
" + if subtitle + else "" + ) + display( + HTML( + f"
" + f"
" + f"{title}" + f"{label}" + f"
" + f"{sub_html}" + f"
" + ) + ) +``` + +```{python sample-id-banner} +#| echo: false + +# Resolve sample id (prefer parameter; fall back to INDIR parent dir name). +_resolved_out = base.resolve() +_sample_name = str(SAMPLE_NAME).strip() if SAMPLE_NAME else "" +if not _sample_name: + _sample_name = (_resolved_out.parent.name + if _resolved_out.name == "image_qc" + else _resolved_out.name) + +display(HTML( + "
" + f"Sample ID: {_sample_name}" + "
" +)) +``` + +## 1. At-a-glance Sample Health {#sec-1} + + +::: {.callout-note collapse="true" title="Metric explanation"} +QC results consolidated at sample level; each reflects the worst sub-verdict in the corresponding subsection. + +- **Morphology** — edge / hole burden, usable tissue, contiguous low-quality zones (see [Section 2.1](#sec-2-1)). +- **Focus** — Tile grid summary, tile focus and blurry (GMM) classification (see [Section 3.2](#sec-3-2)). +- **Stain intensity** — per-channel mean signal vs the intensity floor (see [Section 3.3](#sec-3-3)). +- **Signal-to-noise** — image- and transcript-derived signal-to-noise across tiles (see [Section 3.4](#sec-3-4)). +- **Cell quality** — per-cell focus / blurriness and nuclear texture metrics (see [Section 4](#sec-4), when segmentation is available). +::: + +::: {.callout-tip collapse="true" title="Interpretation help"} +**Status logic** — each row is the *worst* PASS / WARN / FAIL across its underlying components. Components that aren't available (typically optional dependencies or missing inputs) display as N/A and don't affect the consolidated status. The scorecard is a triage tool — always cross-check with the relevant section before acting. + +- An all-PASS scorecard indicates no flagged metrics; still review the per-section figures and tissue-specific caveats before treating the slide as passed. +- One or two WARN/FAIL rows: consult the linked section to determine whether the failure reflects sample quality, which warrants action, or tissue context, which can be noted without further action. +- Many WARN/FAIL rows: likely a systemic issue (registration, mounting, optical setup); review pipeline logs and inputs before re-running downstream steps. +- A FAIL on Morphology with a WARN on Intensity but a PASS on Focus is often a tissue / sample issue rather than imaging — check tissue-type caveats in [Section 2.1](#sec-2-1) first. (Intensity is advisory and never FAILs on its own.) + +A FAIL status doesn't always mean a failed experiment — many verdicts are tissue-context dependent. For example, lung tissue naturally produces a high hole-area burden and lower focus scores around airways and bronchioles, which reflect biological structure (lumens, sparse parenchyma) rather than imaging defects. Always cross-check failing rows against the tissue-specific caveats in the linked section before treating the slide as bad. +::: + +```{python at-a-glance-scorecard} +#| echo: false + + +iq = intensity_stats if intensity_stats else roi_metrics.get("intensity_quality") or {} +rows: list[dict[str, str]] = [] +# Per-category override for the Category-column expandable "Why" note. When a +# category sets an entry, the caveat shows that (e.g. only the WARN/FAIL +# drivers) instead of the full Detail column. The Detail column itself always +# shows the complete component list. +_caveat_overrides: dict[str, str] = {} + +def _worst_status(labels: list[str]) -> str: + rank = {"PASS": 0, "WARN": 1, "FAIL": 2} + vals = [str(x).upper() for x in labels if str(x).upper() in rank] + if not vals: + return "N/A" + return max(vals, key=lambda u: rank[u]) + +def _component_display_verdict(payload: dict) -> str: + """Map optional-dependency skips to N/A instead of FAIL.""" + if not isinstance(payload, dict): + return "N/A" + reason = str(payload.get("reason", "") or payload.get("moran_note", "")).lower() + if "modulenotfounderror" in reason or "no module named" in reason: + return "N/A" + return str(payload.get("verdict", "")).upper() + +# Detect tissue mask failure once — reused by Intensity and Focus rows +_tm_aag = roi_metrics.get("tissue_mask_qc") if isinstance(roi_metrics.get("tissue_mask_qc"), dict) else {} +_mask_ok_aag = _tm_aag.get("tissue_mask_generated", True) +_rit_aag = _tm_aag.get("rois_in_tissue", -1) +_mask_ok_aag = _mask_ok_aag and (isinstance(_rit_aag, (int, float)) and _rit_aag > 0) + +# Spatial morphology category +morph_statuses = [] +morph_parts = [] +tm = roi_metrics.get("tissue_mask_qc", {}) +if isinstance(tm, dict): + s = str(tm.get("status", "")).upper() + if s in {"PASS", "WARN", "FAIL"}: + morph_statuses.append(s) + rin, tr = tm.get("rois_in_tissue"), tm.get("total_rois") + if isinstance(rin, (int, float)) and isinstance(tr, (int, float)) and float(tr) > 0: + morph_parts.append(f"tissue tiles={int(rin):,}/{int(tr):,} ({100.0*float(rin)/float(tr):.1f}%)") + +# Aggregate the §2.1 morphology sub-verdicts into this category, so the +# at-a-glance reflects "the worst sub-verdict in the subsection" (its stated +# contract) rather than only the tissue-mask status (2026-06-23, fixes the +# "§2.1 shows FAIL but §1 Morphology shows PASS" mismatch). Self-contained +# because the §2.1 helpers/cutoffs live in a later chunk — KEEP THE THRESHOLDS +# AND METRIC KEYS IN SYNC WITH §2.1. Tissue-coverage, edge and hole are WARN-only +# (geometry/biology-confounded); usable-tissue is now WARN-only too (2026-06-26), so only +# cluster-zone (localized physical artefact) can FAIL. +_mc = (roi_cutoffs.get("image_qc") or {}).get("morphology") or {} +_sc = (roi_cutoffs.get("image_qc") or {}).get("slide_level") or {} +_mq = roi_metrics.get("morphology") or {} + +def _m_find(*names): + for _src in (_mq, roi_metrics): + if isinstance(_src, dict): + for _n in names: + _v = _src.get(_n) + if isinstance(_v, (int, float)) and np.isfinite(float(_v)): + return float(_v) + return None + +def _m_status(val, warn, fail, higher_is_better): + if val is None: + return None + if higher_is_better: + if fail is not None and val < float(fail): + return "FAIL" + if warn is not None and val < float(warn): + return "WARN" + return "PASS" + if fail is not None and val > float(fail): + return "FAIL" + if warn is not None and val > float(warn): + return "WARN" + return "PASS" + +_cov_mean = (roi_metrics.get("tissue_coverage") or {}).get("mean") +_cov_mean = ( + float(_cov_mean) + if isinstance(_cov_mean, (int, float)) and np.isfinite(float(_cov_mean)) + else None +) +_morph_subs = [ + ("tissue coverage", _cov_mean, _mc.get("tissue_coverage_warn"), None, True), # WARN-only (geometry/biology) + ("edge-zone burden", _m_find("edge_zone_frac", "edge_zone_fraction", "edge_zone", "tissue_edge_fraction"), _mc.get("edge_zone_warn"), None, False), + ("hole-area burden", _m_find("hole_area_frac", "hole_area_fraction", "hole_area", "internal_hole_fraction"), _mc.get("hole_area_warn"), None, False), + ("usable tissue", _m_find("usable_tissue_frac", "usable_tissue_fraction", "usable_tissue"), _sc.get("usable_tissue_warn"), _sc.get("usable_tissue_fail"), True), + ("largest low-quality zone", _m_find("cluster_zone_frac", "cluster_zone_fraction", "cluster_zone_bad_fraction"), None, _sc.get("cluster_zone_fail"), False), +] +_morph_driver_parts = [] +for _lbl, _val, _w, _f, _hib in _morph_subs: + _st = _m_status(_val, _w, _f, _hib) + if _st in ("PASS", "WARN", "FAIL"): + morph_statuses.append(_st) + if _st in ("WARN", "FAIL") and _val is not None: + _morph_driver_parts.append(f"{_lbl}={100.0*_val:.1f}% ({_st})") + +_morph_status = _worst_status(morph_statuses) +# Drivers-only caveat note (same pattern as Cell quality), so the "Why" doesn't +# list PASS sub-metrics. Detail column keeps the tissue-tile context. +if _morph_status in ("WARN", "FAIL") and _morph_driver_parts: + _caveat_overrides["Morphology"] = "
".join(_morph_driver_parts) +rows.append( + { + "Category": "Morphology", + "Status": _morph_status, + "Detail": ("
".join(morph_parts) if morph_parts else "See section 2.1 for morphology QC summary."), + "Links": 'Section 2.1', + } +) + +# Tile focus category — composite verdict +# Strategy: use GMM blur % when component separation is reliable (Cohen's d >= warn); +# fall back to median Laplacian variance + CCFS when GMM separation is poor. +roi_focus_status = "N/A" +roi_detail = "See section 3.2 for tile focus/blurriness metrics." +_roi_focus_path = "" # which path determined the verdict + +_iq_cut = roi_cutoffs.get("image_qc") or {} +_focus_cut_aag = _iq_cut.get("focus") or {} +_dapi_cut_aag = ((_iq_cut.get("channels") or {}).get("DAPI") or {}) + +# Extract component separation and thresholds +_lap_aag = roi_metrics.get("laplacian_sharpness") or {} +_comp_sep = _lap_aag.get("component_separation") +_cs_warn = _focus_cut_aag.get("gmm_2d_laplacian_component_separation_warn", 1.0) +_cs_fail = _focus_cut_aag.get("gmm_2d_laplacian_component_separation_fail", 0.5) +_gmm_reliable = ( + isinstance(_comp_sep, (int, float)) + and np.isfinite(float(_comp_sep)) + and float(_comp_sep) >= float(_cs_warn) +) + +# 2D first, matching :1119 / :1176 and the cutoffs this is graded against: +# conf/roi_image_qc_thresholds.yaml focus_warn/focus_fail were "empirically calibrated +# 2026-05-14" on GMM-2D, and the reliability gate just above reads +# gmm_2d_laplacian_component_separation. blur_gmm_1d is effectively always written, so +# with 1d first the 2D branch was dead code and a 1D number was graded against a 2D +# yardstick. Measured on 11 calibration samples: 2 FAIL -> WARN, 2 WARN -> PASS, nothing +# degrades; direction is sample-dependent (5 up, 6 down). blur_gmm_2d is genuinely +# optional (image_qc.py:8611), so the 1d fallback stays. See +# docs/plans/2026-08-24_SPIKE_blur-precedence.md. +blur = roi_metrics.get("blur_gmm_2d") or roi_metrics.get("blur_gmm_1d") + +if not _mask_ok_aag: + roi_focus_status = "N/A" + roi_detail = "no tissue mask — focus QC not performed" + _roi_focus_path = "no_mask" +elif _gmm_reliable and isinstance(blur, dict): + # --- Path A: GMM separation is reliable → use GMM blur % --- + pct_all = blur.get("pct_blurred_gmm") + pct_tissue = blur.get("pct_blurred_gmm_tissue_filtered") + pct = pct_tissue if isinstance(pct_tissue, (int, float)) else pct_all + fw, ff = _dapi_cut_aag.get("focus_warn"), _dapi_cut_aag.get("focus_fail") + if isinstance(pct, (int, float)) and isinstance(fw, (int, float)) and isinstance(ff, (int, float)): + if float(pct) >= 100.0 * float(ff): + roi_focus_status = "FAIL" + elif float(pct) >= 100.0 * float(fw): + roi_focus_status = "WARN" + else: + roi_focus_status = "PASS" + if isinstance(pct, (int, float)): + roi_detail = f"blurry (GMM)={float(pct):.1f}% (component separation d={float(_comp_sep):.1f})" + _roi_focus_path = "GMM" +else: + # --- Path B: GMM unreliable → honest fallback (2026-06-22) --- + # When the 2D GMM cannot separate blur from focus (component separation + # below the WARN cutoff), we refuse to manufacture a FAIL from a confounded + # proxy. The old fallback hard-failed on the absolute Laplacian floor, which + # scales with brightness² and is least valid on exactly the dim/degraded + # slides that land here (qc_drift_analysis). New behaviour: WARN only when + # CCFS low-nuclear-texture is high (a real per-cell signal worth inspecting); + # otherwise N/A with the blur % shown for reference. Never FAIL. + _cm_aag = cell_metrics or {} + _pct_ccfs_aag = _cm_aag.get("pct_low_nuclear_texture") + if _pct_ccfs_aag is None and _cm_aag.get("cells_low_nuclear_texture") and _cm_aag.get("total_cells"): + _pct_ccfs_aag = round(100.0 * _cm_aag["cells_low_nuclear_texture"] / _cm_aag["total_cells"], 2) + _ccfs_warn_thr = _focus_cut_aag.get("low_texture_cell_warn", 0.10) + _ccfs_bad = isinstance(_pct_ccfs_aag, (int, float)) and _pct_ccfs_aag >= _ccfs_warn_thr * 100 + + _d_str = f"d={float(_comp_sep):.1f}" if isinstance(_comp_sep, (int, float)) and np.isfinite(float(_comp_sep)) else "d=N/A" + _pct_blur_fb = blur.get("pct_blurred_gmm_tissue_filtered") if isinstance(blur, dict) else None + if not isinstance(_pct_blur_fb, (int, float)): + _pct_blur_fb = blur.get("pct_blurred_gmm") if isinstance(blur, dict) else None + _blur_str = f"blurry (GMM)={float(_pct_blur_fb):.1f}% (informational)" if isinstance(_pct_blur_fb, (int, float)) else "blur % unavailable" + + if _ccfs_bad: + roi_focus_status = "WARN" + _ccfs_str = f"{_pct_ccfs_aag:.1f}%" if isinstance(_pct_ccfs_aag, (int, float)) else "N/A" + roi_detail = f"GMM unreliable ({_d_str}); low nuclear texture={_ccfs_str} — inspect focus manually" + else: + roi_focus_status = "N/A" + roi_detail = f"GMM unreliable ({_d_str}); {_blur_str} — inspect focus manually" + _roi_focus_path = "fallback" + +rows.append( + { + "Category": "Focus", + "Status": roi_focus_status, + "Detail": roi_detail, + "Links": 'Section 3.2', + } +) + +# Intensity category +int_statuses = [] +int_detail_parts = [] +if not _mask_ok_aag: + # Tissue-mask generation failed → the sample is unusable and intensity + # cannot be measured. FAIL the Intensity category (2026-06-23): a mask + # failure is a hard QC FAIL, surfaced in both Morphology and Intensity. + # The mask-independent stain p99 in §3.3 explains why it failed. + int_statuses = ["FAIL"] + int_detail_parts.append( + "tissue mask failed — sample unusable; intensity not measurable " + "(see the mask-independent stain p99 in Section 3.3)" + ) +elif isinstance(iq, dict): + oq = iq.get("overall_quality") + if oq is not None and str(oq).strip(): + ov = _intensity_display_status(str(oq)) + int_statuses.append(ov) + int_detail_parts.append(f"overall={str(oq)}") + dapi = iq.get("dapi") if isinstance(iq.get("dapi"), dict) else {} + if dapi: + pbc = dapi.get("pct_tissue_roi_below_critical") + if isinstance(pbc, (int, float)): + int_detail_parts.append(f"DAPI percentage of tissue tiles below intensity floor={float(pbc):.1f}%") +rows.append( + { + "Category": "Stain intensity", + "Status": _worst_status(int_statuses), + "Detail": ("
".join(int_detail_parts) if int_detail_parts else "See section 3.3 for channel-level details."), + "Links": 'Section 3.3', + } +) + +# SNR category (split into tile-level and cell/matrix tracks) +if snr_summary and isinstance(snr_summary.get("components", {}), dict): + comp = snr_summary.get("components", {}) + roi_keys = ("SNR_image_roi_quartile_db", "SNR_image_otsu", "SNR_roi_tx", "SNR_roi_neg_spatial") + matrix_keys = () # Plummer/SpatialQM moved to molecule QC (v5 restructure) + roi_vals = [_component_display_verdict(comp.get(k) or {}) for k in roi_keys if isinstance(comp.get(k), dict)] + mat_vals = [_component_display_verdict(comp.get(k) or {}) for k in matrix_keys if isinstance(comp.get(k), dict)] + rows.append( + { + "Category": "Signal-to-noise", + "Status": _worst_status(roi_vals), + "Detail": "See section 3.4 for tile/image/transcript SNR components.", + "Links": 'Section 3.4', + } + ) +else: + rows.append( + { + "Category": "Signal-to-noise", + "Status": "NOT_COMPUTED", + "Detail": "No SNR data found for this run.", + "Links": 'Section 3.4', + } + ) + +# Cell-level category — composite of CCFS nuclear texture, GMM-ROI blur, and cluster outliers +cell_status = "N/A" +cell_detail = "See section 4 for per-cell QC summary and figures." +if cell_metrics: + nc = cell_metrics.get("total_cells") + # §1 Cell quality reads the same per-metric advisory cutoffs that [Section 4.1](#sec-4-1) + # and [Section 4.2](#sec-4-2) use, so the at-a-glance pill agrees with the §4 advisory + # pills. PASS/WARN only (no FAIL) — matches §4.x's advisory tier. + _cl_cut = ((roi_cutoffs.get("image_qc") or {}).get("cell_level") or {}) + _nuc_warn = float(_cl_cut.get("pct_low_nuclear_texture_warn", 5.0)) + _gmm_warn = float(_cl_cut.get("pct_blurred_gmm_2d_roi_warn", 20.0)) + _cell_verdicts = [] + _cell_parts = [] + + # (a) Nuclear texture (per-cell contrast — distinct from optical blurriness) + _pct_ccfs = cell_metrics.get("pct_low_nuclear_texture") + if _pct_ccfs is None: + nb = cell_metrics.get("cells_low_nuclear_texture") + if isinstance(nb, (int, float)) and isinstance(nc, (int, float)) and float(nc) > 0: + _pct_ccfs = 100.0 * float(nb) / float(nc) + if isinstance(_pct_ccfs, (int, float)): + _cell_parts.append(f"low nuclear texture={_pct_ccfs:.1f}%") + _cell_verdicts.append("WARN" if _pct_ccfs >= _nuc_warn else "PASS") + + # (b) GMM-ROI blur (tile-level classification propagated to cells) + _pct_gmm_roi = cell_metrics.get("pct_blurred_gmm_2d_roi") + if _pct_gmm_roi is None: + _n_gmm_roi = cell_metrics.get("cells_blurred_gmm_2d_roi") + if isinstance(_n_gmm_roi, (int, float)) and isinstance(nc, (int, float)) and float(nc) > 0: + _pct_gmm_roi = 100.0 * float(_n_gmm_roi) / float(nc) + if isinstance(_pct_gmm_roi, (int, float)): + _cell_parts.append(f"blurry cells (GMM)={_pct_gmm_roi:.1f}%") + _cell_verdicts.append("WARN" if _pct_gmm_roi >= _gmm_warn else "PASS") + + # (c) Cluster-outlier blur — any cluster with blur >> sample average + _cluster_outliers = cell_metrics.get("cluster_blur_outliers") + if isinstance(_cluster_outliers, dict) and _cluster_outliers: + worst_cl = max(_cluster_outliers, key=lambda k: _cluster_outliers[k]["pct_blurred"]) + worst_info = _cluster_outliers[worst_cl] + _cell_parts.append( + f"cluster outlier: C{worst_cl} " + f"({worst_info['pct_blurred']:.0f}% blurry cells (GMM), " + f"n={worst_info['n_cells']:,})" + ) + # Outlier cluster is a WARN — spatially biased quality loss + _cell_verdicts.append("WARN") + + cell_status = _worst_status(_cell_verdicts) + # Detail column keeps ALL components (full context). The Category-column + # expandable caveat (the "Why" note) is built from only the WARN/FAIL + # drivers, so it doesn't list PASS components (e.g. a blurry-cell % below + # its cutoff) alongside the real trigger and make them look causal. + cell_detail = ( + "
".join(_cell_parts) + if _cell_parts + else "See section 4 for per-cell QC summary and figures." + ) + if cell_status in ("WARN", "FAIL"): + _drivers = [ + _p for _p, _v in zip(_cell_parts, _cell_verdicts) if _v in ("WARN", "FAIL") + ] + if _drivers: + _why = "
".join(_drivers) + # Explain why a cell-level number can differ from the tile-level one. + if any("blurry cells (GMM)" in _d for _d in _drivers): + _why += ( + "
Note: blurry cells (GMM) is measured over solid-tissue " + "cells (tile coverage ≥ 0.2) to match the tile-level 'tiles in " + "focus' in Section 3.2; it is per-cell rather than per-tile, so " + "small differences from the tile figure are expected." + ) + _caveat_overrides["Cell quality"] = _why +rows.append( + { + "Category": "Cell quality", + "Status": cell_status, + "Detail": cell_detail, + "Links": 'Section 4', + } +) + +df_glance = pd.DataFrame(rows) + +# Anchor a sample-specific caveat to the Category cell when WARN/FAIL — the +# Detail column already names the offending components; the expandable wraps +# that in the same icon/colour pattern used by the other section tables. +for _i, _r in df_glance.iterrows(): + _s = str(_r["Status"]).strip().upper() + if _s not in ("WARN", "FAIL"): + continue + _cat = str(_r["Category"]) + # Caveat shows the per-category override (e.g. only the WARN/FAIL drivers) + # when set; otherwise the full Detail. The Detail column is unchanged. + _detail = (_caveat_overrides.get(_cat) or str(_r["Detail"])).strip() + _links = str(_r["Links"]).strip() + _body_parts = [f"**{_cat}** is currently **{_s}**."] + if _detail and not _detail.lower().startswith("see section"): + _body_parts.append(f"Why: {_detail}.") + if _links: + _body_parts.append(f"Drill into the underlying components in {_links}.") + df_glance.at[_i, "Category"] = metric_cell_with_caveat(_cat, _s, " ".join(_body_parts)) + +_df = df_glance.copy() +_status = _df["Status"].astype(str).str.upper() +def _status_css(s: str) -> str: + su = str(s).upper() + if su == "FAIL": + return "background-color:#FEE2E2;color:#991B1B;font-weight:700;" + if su == "WARN": + return "background-color:#FFEDD5;color:#9A3412;font-weight:700;" + if su == "PASS": + return "background-color:#DCFCE7;color:#166534;font-weight:700;" + return "" +_status_style = [_status_css(s) for s in _status] +_sty = ( + _df.style + .set_table_styles( + [ + {"selector": "th", "props": "text-align:left; background-color:#F8FAFC;"}, + {"selector": "td", "props": "text-align:left; vertical-align:top;"}, + {"selector": "td.col0, th.col_heading.level0.col0", "props": "text-align:left;"}, + ] + ) + .set_properties(subset=["Category"], **{"min-width": "12em", "text-align": "left"}) + .set_properties(subset=["Detail"], **{"min-width": "26em"}) + .set_properties(subset=["Links"], **{"min-width": "20em"}) + .apply(lambda _col: _status_style, subset=["Status"]) + .hide(axis="index") +) +display(HTML(_sty.to_html())) + +``` + +## 2. Sample-level metrics {#sec-2} + + +The tissue analysis generates whole-sample masks and identifies problematic regions that could affect data quality. + +```{python morphology-metric-explanation} +#| echo: false + +# Pull edge/hole band thresholds from the morphology JSON so the metric +# explanation tracks the actual values used by the pipeline. Both +# `edge_distance_threshold_px_ds` and `hole_distance_threshold_px_ds` are +# negative downsampled-pixel signed-distance thresholds; |value| × 8 × 0.2125 +# converts to µm of inward-band depth from the boundary. Defaults match +# bin/image_qc.py:4443-4444 (-25.0). +# +# TODO (future): `_XENIUM_PX_UM = 0.2125` is hardcoded here, in +# `bin/image_qc.py:70`, and at `qmd:1167`. The authoritative value for this +# sample's instrument lives in the bundle's `experiment.xenium` `pixel_size` +# field. When ready, emit `pixel_size_um` in roi_qc_metrics.json from +# bin/image_qc.py and read it here instead of hardcoding. +_morph = roi_metrics.get("morphology") or {} +_DS_FACTOR = 8 +_XENIUM_PX_UM = 0.2125 +_edge_thr_px = abs(float(_morph.get("edge_distance_threshold_px_ds", -25.0))) +_hole_thr_px = abs(float(_morph.get("hole_distance_threshold_px_ds", -25.0))) +_edge_band_um = _edge_thr_px * _DS_FACTOR * _XENIUM_PX_UM +_hole_band_um = _hole_thr_px * _DS_FACTOR * _XENIUM_PX_UM + +# Usable-tissue cutoffs read from YAML so the prose tracks any threshold +# change without a parallel prose edit. img_qc_cut / slide_cut are initialised +# later in the doc — read directly from roi_cutoffs here. +_slide_cut_2_1 = ( + ((roi_cutoffs.get("image_qc") or {}).get("slide_level") or {}) + if isinstance(roi_cutoffs, dict) + else {} +) +_usable_warn_pct = float(_slide_cut_2_1.get("usable_tissue_warn", 0.60)) * 100 + +display(Markdown(f"""::: {{.callout-note collapse="true" title="Metric explanation"}} +**Summary table metrics** + +- **Edge-zone burden** — fraction of tissue tiles whose centroid lies within ~{_edge_band_um:.1f} µm of the outer tissue boundary. High values indicate edge-dominated samples. **Advisory only: PASS / WARN, never FAIL.** Edge fraction is highly tissue-variable, sparse or branching tissues (lung, intestine, lymph node) legitimately sit high, so a WARN here flags the sample for review rather than indicating poor quality. (WARN calibrated on the 160+ sample cohort, just above the per-tissue p90.) +- **Hole-area burden** — fraction of tissue tiles within ~{_hole_band_um:.1f} µm of an internal hole/void; high values suggest tears, gaps, or many small voids spread through the tissue. **Advisory only: PASS / WARN, never FAIL.** Hole burden is highly tissue-variable, lumen-rich tissues (lung airways, blood vessels, glandular ducts) produce high hole-adjacency by design, so a WARN is biologically expected for them, not an artefact call. (WARN calibrated on the cohort to clear the lung tissue tail.) +- **Usable tissue** — fraction of tissue tiles (≥ 20% tissue coverage) that are neither blurry (GMM) nor below the tile DAPI-intensity floor. Dominated by blurry-tile rate — most tissue tiles pass the intensity floor, so usable ≈ 1 − blurry fraction. **Advisory (WARN only, no FAIL):** usable is partly intensity-confounded for dim tissue, so a low value flags the sample for review rather than failing it. Thresholds: PASS ≥ {_usable_warn_pct:.0f}%, WARN < {_usable_warn_pct:.0f}%. +- **Largest contiguous low-quality zone** — fraction of tissue tiles occupied by the largest single connected region of low-quality tiles (out-of-focus or below intensity floor), 4-neighbour connectivity. High values indicate a spatially concentrated artefact (fold, chip, shadow, coverslip defect) rather than diffuse degradation. Threshold: FAIL > 15%. + +**Visualisations** + +- **Stainings** — view of DAPI, Boundary, and IntRNA channels (the optically dense region map lives in the Masks panel). +- **Masks** — view of the three mask products used by the pipeline: tissue extent, holes in sample, and optically dense regions. +- **Distance to edge and holes** — two-panel map showing per-location distance to the nearest outer tissue boundary (left) and to the nearest internal void/hole (right). Both panels rendered in µm and masked to the tissue extent from all stains (DAPI, Boundary, IntRNA), the same mask shown in the Masks panel. +::: +""")) +``` + +::: {.callout-tip collapse="true" title="Interpretation help"} +**Stainings** + +This is a **whole-slide overview** at low resolution; individual cells, nuclei or membrane outlines are *not* resolvable here. It shows where signal is present across the slide and how evenly it is distributed. + +- *DAPI:* total tissue extent and overall staining uniformity — expect signal that traces the full tissue with broadly even intensity. Watch for missing regions, large dim patches, or hard intensity gradients across the slide. +- *Boundary:* regional membrane-stain coverage and intensity — expect signal co-located with DAPI tissue extent. A blank or near-blank panel can reflect (a) the channel missing or mis-aligned, (b) a staining failure (antibody cocktail didn't bind / wash properly), or (c) genuinely sparse target biology — Boundary targets epithelial / immune membranes and is expected to be dim in brain and other tissues lacking these cell types (see [Section 3.3](#sec-3-3) for the tissue-specific caveat). +- *IntRNA:* slide-scale IntRNA intensity — expect coverage that broadly tracks the DAPI footprint with some biologically expected heterogeneity (white matter, adipose and necrotic regions are dim; epithelium and pancreas are bright). + +**Masks** + +A three-panel view of the labelled mask products. In each panel the **coloured regions are the labelled features**; the dark background is everything *not* in that mask. + +- *Tissue mask:* coloured regions are detected tissue, found using all available stains (DAPI, plus Boundary and Interior when the slide has them), so the mask reflects tissue extent. Tissue extent metrics (coverage, edge and hole burden) use this combined mask. Focus, blur and usable-tissue metrics are computed on the DAPI nuclei mask instead, because they judge nuclear sharpness and would be misled by tissue that has few nuclei. On single-stain slides both masks are the DAPI mask. +- *Holes in sample:* coloured regions are internal voids inside the tissue mask — tears, gaps, folds, or detached patches excluded from analysis. Multiple separate holes appear as separate colours. +- *Optically dense regions:* coloured regions are unusually bright zones (folds, edge bleed, debris, coverslip contamination) — regions of unusually high pixel intensity not attributable to tissue staining. + +Unexpected mask shapes — missing tissue, spurious holes, oversized artefact regions — often account for anomalous downstream values; check the masks before interpreting the metrics. + +**Distance to edge and holes** + +- Yellow / bright regions are far from edges or holes — generally reliable tissue with good signal support. +- Dark purple regions on the **left panel** are near edges (partial tissue, sectioning artefacts, weak signal); on the **right panel** they are near holes (potential environmental effects on nearby cells). The thin black line in both panels marks the tissue mask boundary. +- Cells within ~40–85 µm of the edge are often excluded from downstream analysis. +- Biological holes (blood vessels, airways, glandular lumens) may be expected; technical holes (tears, folds, processing damage) should be flagged for exclusion. +- If poor focus / intensity failures align with dark regions in either panel, this usually reflects edge effects or local structural artefacts rather than whole-slide failure. +::: + +### 2.1 Morphology summary {#sec-2-1} + +```{python spatial-morphology-qc} +#| echo: false + +img_qc_cut = (roi_cutoffs.get("image_qc") or {}) if isinstance(roi_cutoffs, dict) else {} +morph_cut = (img_qc_cut.get("morphology") or {}) if isinstance(img_qc_cut, dict) else {} +slide_cut = (img_qc_cut.get("slide_level") or {}) if isinstance(img_qc_cut, dict) else {} + + +def _find_metric(d: dict, names: tuple[str, ...]) -> float | None: + if not isinstance(d, dict): + return None + for n in names: + v = d.get(n) + if isinstance(v, (int, float)) and np.isfinite(float(v)): + return float(v) + return None + + +def _status_from_value( + value: float | None, + *, + warn: float | None = None, + fail: float | None = None, + higher_is_better: bool = True, +) -> str: + if value is None or fail is None: + return "N/A" + if warn is None: + if higher_is_better: + return "FAIL" if value < float(fail) else "PASS" + return "FAIL" if value > float(fail) else "PASS" + if higher_is_better: + if value < float(fail): + return "FAIL" + if value < float(warn): + return "WARN" + return "PASS" + if value > float(fail): + return "FAIL" + if value > float(warn): + return "WARN" + return "PASS" + + +def _advisory_pass_warn( + value: float | None, + warn: float | None, + *, + higher_is_better: bool = False, +) -> str: + """PASS / WARN-only verdict for advisory tiers ([Section 4.1](#sec-4-1) cell-level). + + Returns "N/A" if either value or warn is missing. Two-tier by design — + these metrics surface flags-to-investigate, not strict FAIL gates. + """ + if value is None or warn is None: + return "N/A" + try: + v, w = float(value), float(warn) + except (TypeError, ValueError): + return "N/A" + if higher_is_better: + return "PASS" if v >= w else "WARN" + return "WARN" if v >= w else "PASS" + + +rows = [] + +# Tissue coverage moved to [Section 3.1](#sec-3-1) Tile grid summary (2026-05-15) — was a duplicate +# of the [Section 3.1](#sec-3-1) "Mean tissue coverage per tile" row. The verdict + cutoffs travel +# with it. Note: §1 At-a-glance Morphology pill is independent of this row +# (it consumes `roi_metrics["tissue_mask_qc"]["status"]` directly). + +edge_zone = _find_metric( + roi_metrics.get("morphology", {}), + ("edge_zone_frac", "edge_zone_fraction", "edge_zone", "tissue_edge_fraction"), +) +if edge_zone is None: + edge_zone = _find_metric( + roi_metrics, + ("edge_zone_frac", "edge_zone_fraction", "edge_zone", "tissue_edge_fraction"), + ) +edge_warn = morph_cut.get("edge_zone_warn") +# 2026-06-23: Edge-zone burden is WARN-only (no FAIL). Edge-dominance is +# biology-driven — sparse/branching tissues (lung, intestine, lymph node) +# legitimately have high edge fractions — so it flags for review, never fails. +rows.append( + { + "Metric": "Edge-zone burden", + "Status": _advisory_pass_warn( + edge_zone, + edge_warn if isinstance(edge_warn, (int, float)) else None, + higher_is_better=False, + ), + "Value": f"{100.0*edge_zone:.1f}%" if edge_zone is not None else "not available", + "Cutoffs": ( + f"PASS <= {100.0*float(edge_warn):.1f}%; WARN > {100.0*float(edge_warn):.1f}% (advisory, no FAIL)" + if isinstance(edge_warn, (int, float)) + else "not configured" + ), + } +) + +hole_area = _find_metric( + roi_metrics.get("morphology", {}), + ("hole_area_frac", "hole_area_fraction", "hole_area", "internal_hole_fraction"), +) +if hole_area is None: + hole_area = _find_metric( + roi_metrics, + ("hole_area_frac", "hole_area_fraction", "hole_area", "internal_hole_fraction"), + ) +hole_warn = morph_cut.get("hole_area_warn") +# 2026-06-23: Hole-area burden is WARN-only (no FAIL). The hole mask does not yet +# separate anatomical lumens (lung, intestine, glandular epithelia) from +# artefactual tears, so it flags for review, never fails. +rows.append( + { + "Metric": "Hole-area burden", + "Status": _advisory_pass_warn( + hole_area, + hole_warn if isinstance(hole_warn, (int, float)) else None, + higher_is_better=False, + ), + "Value": f"{100.0*hole_area:.1f}%" if hole_area is not None else "not available", + "Cutoffs": ( + f"PASS <= {100.0*float(hole_warn):.1f}%; WARN > {100.0*float(hole_warn):.1f}% (advisory, no FAIL)" + if isinstance(hole_warn, (int, float)) + else "not configured" + ), + } +) + +usable_tissue = _find_metric( + roi_metrics.get("morphology", {}), + ("usable_tissue_frac", "usable_tissue_fraction", "usable_tissue"), +) +if usable_tissue is None: + usable_tissue = _find_metric( + roi_metrics, + ("usable_tissue_frac", "usable_tissue_fraction", "usable_tissue"), + ) +usable_warn = slide_cut.get("usable_tissue_warn") +# usable_tissue is WARN-only (2026-06-26): no usable_tissue_fail key; row uses _advisory_pass_warn. + +# Build a breakdown string showing what drives the usable tissue score +_usable_detail = "" +if usable_tissue is not None: + _blur_gmm = roi_metrics.get("blur_gmm_2d") or roi_metrics.get("blur_gmm_1d") or {} + _pct_blur_tissue = _blur_gmm.get("pct_blurred_gmm_tissue_filtered") + if isinstance(_pct_blur_tissue, (int, float)): + _usable_detail = f" (out-of-focus {_pct_blur_tissue:.1f}%)" + +rows.append( + { + "Metric": "Usable tissue", + # 2026-06-26: WARN-only (advisory). usable is partly intensity-confounded for dim + # tissue and the FAIL cutoff was overfit; reintroduce FAIL only with a larger panel. + "Status": _advisory_pass_warn( + usable_tissue, + usable_warn if isinstance(usable_warn, (int, float)) else None, + higher_is_better=True, + ), + "Value": f"{100.0*usable_tissue:.1f}%{_usable_detail}" if usable_tissue is not None else "not available", + "Cutoffs": ( + f"PASS >= {100.0*float(usable_warn):.1f}%; WARN < {100.0*float(usable_warn):.1f}% (advisory, no FAIL)" + if isinstance(usable_warn, (int, float)) + else "not configured" + ), + } +) + +cluster_bad = _find_metric( + roi_metrics.get("morphology", {}), + ("cluster_zone_frac", "cluster_zone_fraction", "cluster_zone_bad_fraction"), +) +if cluster_bad is None: + cluster_bad = _find_metric( + roi_metrics, + ("cluster_zone_frac", "cluster_zone_fraction", "cluster_zone_bad_fraction"), + ) +cluster_fail = slide_cut.get("cluster_zone_fail") +rows.append( + { + "Metric": "Largest contiguous low-quality zone", + "Status": _status_from_value( + cluster_bad, + warn=None, + fail=cluster_fail if isinstance(cluster_fail, (int, float)) else None, + higher_is_better=False, + ), + "Value": f"{100.0*cluster_bad:.1f}%" if cluster_bad is not None else "not available", + "Cutoffs": ( + f"PASS <= {100.0*float(cluster_fail):.1f}%; FAIL > {100.0*float(cluster_fail):.1f}%" + if isinstance(cluster_fail, (int, float)) + else "not configured" + ), + } +) + +df_morph = pd.DataFrame(rows) + +# Usable tissue: collapsible caveat anchored to its own row when WARN/FAIL +_usable_status = df_morph.loc[df_morph["Metric"] == "Usable tissue", "Status"].values +if len(_usable_status) > 0 and str(_usable_status[0]).upper() in ("WARN", "FAIL"): + _blur_gmm = roi_metrics.get("blur_gmm_2d") or roi_metrics.get("blur_gmm_1d") or {} + _pct_blur_tissue = _blur_gmm.get("pct_blurred_gmm_tissue_filtered") + _blur_note = "" + if isinstance(_pct_blur_tissue, (int, float)) and isinstance(usable_tissue, (int, float)): + _combined_fail_pct = (1.0 - float(usable_tissue)) * 100.0 + if _combined_fail_pct > 30: + _blur_note = ( + f" In this sample, **{_combined_fail_pct:.1f}%** of tissue tiles failed the " + f"combined blurriness-or-intensity gate; **{_pct_blur_tissue:.1f}%** of tissue tiles are " + "GMM-blurry (the primary driver). The remaining failing tiles are below the intensity floor." + ) + elif isinstance(_pct_blur_tissue, (int, float)) and _pct_blur_tissue > 30: + # Fallback when usable_tissue isn't numeric — surface the GMM rate only. + _blur_note = ( + f" In this sample, **{_pct_blur_tissue:.1f}%** of tissue tiles were classified " + "as blurry by the GMM." + ) + _usable_body = ( + "**Usable tissue** is computed as the fraction of DAPI tissue tiles (those at least " + "20% covered by the DAPI nuclei mask) that are neither blurry (GMM) nor below the " + "intensity floor. It uses the DAPI mask, not the multi-stain extent mask, because it " + "judges nuclear focus quality. " + "In most samples, **out-of-focus tiles** dominate — the intensity gate " + "rarely excludes additional tiles beyond the blurriness classification." + f"{_blur_note} " + "See [Section 3.2](#sec-3-2) for spatial blurriness distribution and [Section 4](#sec-4) for cell-level blurriness assessment." + ) + df_morph.loc[df_morph["Metric"] == "Usable tissue", "Metric"] = metric_cell_with_caveat( + "Usable tissue", str(_usable_status[0]), _usable_body + ) + +# Edge-zone burden: tissue-type caveat anchored to its row +_edge_status = df_morph.loc[df_morph["Metric"] == "Edge-zone burden", "Status"].values +if len(_edge_status) > 0 and str(_edge_status[0]).upper() in ("WARN", "FAIL"): + _edge_body = ( + "**Tissue-type caveat — Edge-zone burden is elevated.** This is often tissue-specific " + "rather than a quality problem. Sparse or branching tissues (lung alveoli, intestinal villi, " + "lymph node sinuses) have proportionally more internal and external edges than compact " + "tissues (liver, brain cortex). High edge-zone burden in these contexts does not indicate " + "poor sample quality." + ) + df_morph.loc[df_morph["Metric"] == "Edge-zone burden", "Metric"] = metric_cell_with_caveat( + "Edge-zone burden", str(_edge_status[0]), _edge_body + ) + +# Hole-area burden: tissue-type caveat anchored to its row +_hole_status = df_morph.loc[df_morph["Metric"] == "Hole-area burden", "Status"].values +if len(_hole_status) > 0 and str(_hole_status[0]).upper() in ("WARN", "FAIL"): + _hole_body = ( + "**Tissue-type caveat — Hole-area burden is elevated.** Distinguish biological lumens " + "(airways, blood vessels, glandular ducts) from technical artefacts (tissue tears, " + "detachment, processing damage). Lumen-rich tissues (lung, kidney, intestine) routinely " + "score high here without any quality concern." + ) + df_morph.loc[df_morph["Metric"] == "Hole-area burden", "Metric"] = metric_cell_with_caveat( + "Hole-area burden", str(_hole_status[0]), _hole_body + ) + +display_status_table(df_morph, ["Status"]) +``` + +### 2.2 Stainings {#sec-2-2} + +A compact view of the morphology channels used for quality assessment (DAPI, Boundary, IntRNA) plus optically dense region map. + +```{python fig-morphology-overview} +#| echo: false +display(fig_html("figures/morphology_overview.png", max_width="95%")) +``` + +### 2.3 Masks {#sec-2-3} + +Shows the mask products used by the pipeline: tissue mask, holes mask, and optically dense region map. + +```{python fig-imageqc-masks} +#| echo: false +if _mask_ok_aag: + display(fig_html("figures/imageqc_masks.png", max_width="95%")) +else: + _mask_lines = [ + "> **Tissue mask was not generated** for this sample. The following mask products are unavailable:\n", + "> - **Tissue mask:** not generated — automatic tissue segmentation could not find a foreground population", + "> - **Hole mask:** not generated (depends on tissue mask)", + "> - **Artefact / optically dense mask:** may still be available but cannot be interpreted without tissue context", + ">\n> This typically occurs with tissue types that have very low or diffuse DAPI signal " + "(e.g. skin, adipose, decalcified bone). Consider lowering the tile intensity floor " + "in the thresholds configuration if tissue is genuinely present but below the default gate.", + ] + display(Markdown("\n".join(_mask_lines))) +``` + +### 2.4 Distance to edge and holes {#sec-2-4} + +Two-panel map showing per-location distance to the nearest **outer tissue boundary** (left) and to the nearest **internal void/hole** (tears, gaps, folds, detached regions; right). Distances are measured over the tissue extent from all stains (DAPI, Boundary, IntRNA), so dim nuclei-sparse regions (e.g. muscle, brain) are included — the same mask shown in §2.3. + +```{python fig-distance-combined} +#| echo: false +if _mask_ok_aag: + display(fig_html("figures/distance_maps.png")) +else: + display(Markdown( + "> **Not available:** tissue mask was not generated for this sample, so " + "distance-to-edge and distance-to-holes cannot be computed. This typically " + "occurs when none of the available stains (DAPI, Boundary, IntRNA) produce " + "a usable tissue mask." + )) +``` + +## 3. Tile-level metrics {#sec-3} + +### 3.1 Tile grid summary {#sec-3-1} + +This section assesses image quality at the tile level: a regular grid of square tiles laid over the capture area, where each tile receives focus score, stain intensity and signal-to-noise ratio (SNR) scores independently of cell segmentation. + +::: {.callout-note collapse="true" title="Metric explanation"} +**Tile grid parameters** + +- **Tile size (px / µm)** — square tile edge length in pixels, plus the µm equivalent (the Xenium morphology base resolution is 0.2125 µm/pixel). +- **Total tiles** — count of tiles laid over the capture area before any tissue masking. + +**Tissue coverage statistics** + +- **Tissue tiles** — count of tiles overlapping the tissue mask. Focus and blur metrics are computed on the DAPI nuclei tiles (those at least 20% covered by DAPI tissue); tissue extent (coverage below) uses all available stains. +- **Tissue mask generation status** — PASS or FAIL from the mask-generation pipeline step. FAIL means no tissue mask was produced and downstream tissue-filtered metrics are unavailable. +- **Mean tissue coverage per tile** — average fractional coverage across all tiles, continuous in [0, 1], computed from all available stains (DAPI plus Boundary and Interior when present), so it reflects true tissue extent including nuclei-sparse tissue. **Advisory (WARN only, no FAIL):** low values indicate small / partial or sparse sections, which is tissue geometry / biology rather than a data-quality failure (a genuinely empty slide is caught by the tissue-mask gate). A WARN carries an expandable biology caveat. + +::: + +::: {.callout-tip collapse="true" title="Interpretation help"} +**Tile grid sanity checks** + +- Tile size much smaller than expected cell diameter → metrics may be noisy at the tile level; cell-level metrics in [Section 4](#sec-4) carry more weight. +- Mean tissue coverage per tile gives a rough sense of how much of the capture area is empty / off-sample — low values indicate small or partial tissue sections relative to the capture window. + +::: + +#### Tile grid parameters + +```{python roi-grid-parameters} +#| echo: false + +# Xenium morphology base pixel size — see XENIUM_PIXEL_SIZE_UM in bin/image_qc.py +_XENIUM_PIXEL_SIZE_UM = 0.2125 + +rows = [] +if "roi_size_pixels" in roi_metrics: + _roi_px = roi_metrics["roi_size_pixels"] + rows.append({"Metric": "Tile size (px)", "Value": _roi_px}) + if isinstance(_roi_px, (int, float)): + rows.append( + {"Metric": "Tile size (µm)", "Value": f"{float(_roi_px) * _XENIUM_PIXEL_SIZE_UM:.1f}"} + ) +if "total_rois" in roi_metrics: + rows.append({"Metric": "Total tiles", "Value": f"{roi_metrics['total_rois']:,}"}) + +df_grid = pd.DataFrame(rows, columns=["Metric", "Value"]) +display_table(df_grid) +``` + +#### Tissue coverage statistics + +```{python roi-tissue-coverage-stats} +#| echo: false + +rows = [] +total = roi_metrics.get("total_rois") +in_tissue = roi_metrics.get("rois_in_tissue") +tissue_cov = (roi_metrics.get("tissue_coverage") or {}).get("mean") +tm = roi_metrics.get("tissue_mask_qc") if isinstance(roi_metrics.get("tissue_mask_qc"), dict) else {} + +# Tissue-coverage verdict moved here from [Section 2.1](#sec-2-1) (2026-05-15). The Mean tissue +# coverage per tile row carries the PASS/WARN/FAIL it used to carry in [Section 2.1](#sec-2-1); +# informational rows display "—" in Status / Cutoffs. +tissue_cov_warn = morph_cut.get("tissue_coverage_warn") +# 2026-06-24: Mean tissue coverage is WARN-only (no FAIL). Low coverage is +# tissue geometry / biology (small or sparse sections), not a data-quality +# failure; the genuinely-empty case is caught by the tissue-mask gate. See the +# biology-driven caveat attached to the row below. +_tc_cutoffs = ( + f"PASS >= {100.0*float(tissue_cov_warn):.1f}%; " + f"WARN < {100.0*float(tissue_cov_warn):.1f}% (advisory, no FAIL)" + if isinstance(tissue_cov_warn, (int, float)) + else "not configured" +) + +if isinstance(in_tissue, (int, float)): + rows.append({ + "Metric": "Tissue tiles", + "Status": "—", + "Value": f"{int(in_tissue):,}", + "Cutoffs": "—", + }) +# Fraction of tiles overlapping tissue (≥ 50%) row dropped 2026-05-15: +# decided "Mean tissue coverage per tile" is more interpretable for the +# biology-facing audience (continuous % of capture-area tissue) than the +# binary tile-count fraction. The mean carries the PASS/WARN/FAIL verdict; +# the reader can still compute (Tissue tiles / Total tiles) from the +# values shown above and in the Tile grid parameters table if needed. +if isinstance(tm, dict): + _tm_status = str(tm.get("status", "N/A")).upper() + rows.append({ + "Metric": "Tissue mask generation status", + "Status": _tm_status if _tm_status in {"PASS", "WARN", "FAIL"} else "N/A", + "Value": "—", + # On FAIL the sample is unusable; point the reader to the + # mask-independent stain p99 diagnostic in §3.3 (dim vs structural cause). + "Cutoffs": ( + "FAIL → sample unusable; see the mask-independent stain p99 in Section 3.3" + if _tm_status == "FAIL" + else "—" + ), + }) +if isinstance(tissue_cov, (int, float)): + rows.append({ + "Metric": "Mean tissue coverage per tile", + "Status": _advisory_pass_warn( + tissue_cov, + tissue_cov_warn if isinstance(tissue_cov_warn, (int, float)) else None, + higher_is_better=True, + ), + "Value": f"{100.0*float(tissue_cov):.1f}%", + "Cutoffs": _tc_cutoffs, + }) + +df_cov = pd.DataFrame(rows, columns=["Metric", "Status", "Value", "Cutoffs"]) +# Biology-driven caveat: on WARN, attach an expandable note explaining that low +# coverage is usually small/sparse tissue, not a quality failure. +_cov_st = df_cov.loc[df_cov["Metric"] == "Mean tissue coverage per tile", "Status"].values +if len(_cov_st) and str(_cov_st[0]).upper() in ("WARN", "FAIL"): + _cov_body = ( + "Low mean tissue coverage is usually **tissue geometry / biology**, not a " + "quality problem: small or partial sections (biopsy, needle core, TMA core) " + "and sparse or branching tissues (lung, intestine, lymph node, adipose) " + "naturally occupy little of the capture window. The data on the tissue that " + "is present is judged by the focus / intensity / SNR metrics over tissue " + "tiles; a genuinely empty slide is caught separately by the tissue-mask " + "generation gate. Read this as a 'small / sparse section' flag, not a " + "sample failure." + ) + df_cov.loc[df_cov["Metric"] == "Mean tissue coverage per tile", "Metric"] = ( + metric_cell_with_caveat( + "Mean tissue coverage per tile", str(_cov_st[0]), _cov_body + ) + ) +_cov_status_styles = [_status_cell_css(v) for v in df_cov["Status"]] +_sty_cov = ( + df_cov.style + .set_table_styles( + [ + {"selector": "th", "props": "text-align:left; background-color:#F8FAFC;"}, + {"selector": "td", "props": "text-align:left; vertical-align:top;"}, + ] + ) + .hide(axis="index") + .apply(lambda _col: _cov_status_styles, subset=["Status"]) +) +display(HTML(_sty_cov.to_html())) +``` + +### 3.2 Focus score {#sec-3-2} + +::: {.callout-note collapse="true" title="Metric explanation"} +**Summary table metrics** + +- **Tiles in focus** — primary slide-level verdict. The percentage of tissue tiles classified as in-focus, where the classification comes from a two-component Gaussian Mixture Model on focus score and Laplacian variance. The GMM is trained on tissue tiles only and classifies each tile by posterior probability. Higher in-focus % is better; samples with large out-of-focus regions lose transcripts and disrupt segmentation. Brain samples often show elevated blurriness rates because lower DAPI contrast in neural nuclei mimics the dim-tissue side of the bimodality (biology-confounded — cross-check the per-channel intensity correlation in [Section 4.4.a](#sec-4-4-a) before treating brain values as a quality fail). +- **Focus score median (tissue tiles)** — per-tile DAPI image contrast (high = sharp, in-focus tissue; low = flat or out-of-focus), summarised as the median across tissue tiles. The underlying formula is **CCFS** (Coefficient of Contrast Focus Score: variance ÷ mean of pixel intensities in a region), measuring image focus — low values indicate optical blurriness. +- **Median Laplacian variance** (informational) — the median value of **Laplacian variance** (variance of an edge-detection filter's output: high when sharp edges exist, low when the image is smooth or blurry) across tissue tiles. Reported for calibration tracking only; it is no longer a standalone pass/fail gate. Laplacian variance scales with image brightness (roughly the square of it), so an absolute floor is unreliable on dim sections; the GMM above already uses Laplacian variance as one of its two features. +- **Focus-Laplacian correlation (Spearman)** — rank correlation between per-tile CCFS and per-tile Laplacian — two independent sharpness measures. High values indicate the two metrics agree, giving a reliable focus call; low values indicate they disagree, so one metric may be responding to intensity gradients or saturation rather than focus. `N/A` when the slide is uniformly sharp (low CV on both metrics). +- **2D GMM Laplacian component separation** — Cohen's *d* between the two GMM components on a `log(1 + Laplacian variance)` scale. PASS → blurry/in-focus split is meaningful; WARN → interpret cautiously; FAIL → heavy overlap, don't use blurry (GMM) % alone. Tissues with diffuse morphology (brain neuropil, adipose) often show reduced separation even when imaging is fine. + +**Visualisations** + +- **Spatial focus-score map** — per-tile focus score laid out spatially across the Xenium region. +- **Focus – DAPI intensity dependence** — per-tile focus score vs mean DAPI brightness. +- **Focus score distribution** — histogram of focus scores across all tiles. The histogram x-axis and the Focus-vs-DAPI-intensity scatter both use a *normalised* focus score (RobustScaler) for cross-run comparability; the spatial map uses *raw* CCFS computed on raw camera counts (16-bit ADU). +::: + +::: {.callout-tip collapse="true" title="Interpretation help"} +**Spatial focus-score map** + +- Relatively uniform warm colours across tissue → consistently sharp imaging. +- Large contiguous cool regions → systematic defocus (optical tilt, stage drift, tissue fold). +- Gradients from sharp to soft → uneven tissue thickness, mounting issues, or optical field curvature. +- Edge-only dips are common and usually reflect partial-tissue tiles rather than genuine blurriness. + +**Focus - DAPI intensity dependence** + +- No correlation (cloud) → focus and intensity independent; image quality is not confounded by signal level. +- Positive correlation → dim tiles are also out of focus; understained areas may coincide with physical defocus. +- A distinct low-intensity / low-focus cluster usually marks background or partial-tissue tiles that should be excluded by the tissue gate. + +**Focus score distribution** + +- **Single tight peak** → most tiles have similar focus — uniformly sharp (or uniformly blurry); check absolute value to distinguish. +- **Long left tail** → a minority of tiles are notably softer; check the spatial heatmap to see if they cluster (localised) or scatter (noise). +- **Bimodal** → two distinct populations — genuine blurry/in-focus split; check the component-separation metric in the summary table. +::: + +#### Summary + +```{python roi-focus-blur-summary} +#| echo: false + +rows = [] + +# Inputs / cutoffs +# 2D first, for the same reason as the overall verdict above. +blur = roi_metrics.get("blur_gmm_2d") or roi_metrics.get("blur_gmm_1d") or {} +lap = roi_metrics.get("laplacian_sharpness") or {} +dapi_cut = (((roi_cutoffs.get("image_qc") or {}).get("channels") or {}).get("DAPI") or {}) +focus_cut = ((roi_cutoffs.get("image_qc") or {}).get("focus") or {}) + +# Check whether tissue mask was generated — if not, focus metrics are invalid +_tissue_qc_sec6 = roi_metrics.get("tissue_mask_qc") if isinstance(roi_metrics.get("tissue_mask_qc"), dict) else {} +_tissue_mask_ok = _tissue_qc_sec6.get("tissue_mask_generated", True) +_rois_in_tissue_sec6 = _tissue_qc_sec6.get("rois_in_tissue", 0) +# Also guard against edge case: mask "generated" but 0 tissue tiles +_tissue_mask_ok = _tissue_mask_ok and (isinstance(_rois_in_tissue_sec6, (int, float)) and _rois_in_tissue_sec6 > 0) + + +def _status_from_thresholds(value, warn=None, fail=None, higher_is_better=True): + if not isinstance(value, (int, float)) or not np.isfinite(float(value)): + return "N/A" + if not isinstance(fail, (int, float)) or not np.isfinite(float(fail)): + return "N/A" + v = float(value) + f = float(fail) + if warn is None or not isinstance(warn, (int, float)) or not np.isfinite(float(warn)): + if higher_is_better: + return "FAIL" if v < f else "PASS" + return "FAIL" if v > f else "PASS" + w = float(warn) + if higher_is_better: + if v < f: + return "FAIL" + if v < w: + return "WARN" + return "PASS" + if v > f: + return "FAIL" + if v > w: + return "WARN" + return "PASS" + + +# 1) Tiles in focus (%) — primary status metric +if not _tissue_mask_ok: + # No tissue mask → focus metrics are meaningless + pct_blur = np.nan + pct_focus = np.nan + status_focus = "N/A" + den_label = "no tissue mask" +else: + pct_blur_tissue = blur.get("pct_blurred_gmm_tissue_filtered") + pct_blur_all = blur.get("pct_blurred_gmm") + if isinstance(pct_blur_tissue, (int, float)) and np.isfinite(float(pct_blur_tissue)): + pct_blur = float(pct_blur_tissue) + den_label = "tiles within the tissue mask" + else: + pct_blur = float(pct_blur_all) if isinstance(pct_blur_all, (int, float)) and np.isfinite(float(pct_blur_all)) else np.nan + den_label = "all tiles" + pct_focus = 100.0 - pct_blur if np.isfinite(pct_blur) else np.nan +focus_warn = dapi_cut.get("focus_warn") +focus_fail = dapi_cut.get("focus_fail") +if _tissue_mask_ok: + status_focus = _status_from_thresholds( + pct_focus if np.isfinite(pct_focus) else None, + warn=(100.0 * (1.0 - float(focus_warn))) if isinstance(focus_warn, (int, float)) else None, + fail=(100.0 * (1.0 - float(focus_fail))) if isinstance(focus_fail, (int, float)) else None, + higher_is_better=True, + ) +rows.append( + { + "Metric": "Tiles in focus", + "Status": status_focus, + "Value": f"{pct_focus:.1f}%" if np.isfinite(pct_focus) else "not available", + "Cutoffs": ( + f"PASS >= {100.0*(1.0-float(focus_warn)):.1f}%; WARN {100.0*(1.0-float(focus_fail)):.1f}-{100.0*(1.0-float(focus_warn)):.1f}%; FAIL < {100.0*(1.0-float(focus_fail)):.1f}%" + if isinstance(focus_warn, (int, float)) and isinstance(focus_fail, (int, float)) + else "not configured" + ), + "Notes": f"Denominator: {den_label}", + } +) + +# 2) Focus score median (tissue tiles) +_fs = roi_metrics.get("focus_score") or {} +focus_med = _fs.get("tissue_median", _fs.get("median")) +_focus_med_warn = focus_cut.get("focus_median_warn") +_focus_med_status = "N/A" +if not _tissue_mask_ok: + focus_med = None # suppress misleading value +elif isinstance(focus_med, (int, float)) and np.isfinite(float(focus_med)): + if isinstance(_focus_med_warn, (int, float)) and float(focus_med) < float(_focus_med_warn): + _focus_med_status = "WARN" + else: + _focus_med_status = "PASS" +rows.append( + { + "Metric": "Focus score median (tissue tiles)", + "Status": _focus_med_status, + "Value": f"{float(focus_med):.3g}" if isinstance(focus_med, (int, float)) and np.isfinite(float(focus_med)) else "not available", + "Cutoffs": f"WARN < {float(_focus_med_warn):.0f}" if isinstance(_focus_med_warn, (int, float)) else "not configured", + "Notes": "Tissue-dependent: brain/small-nucleus tissues score lower. Interpret with the blurry-tile %.", + } +) + +# Focus-Laplacian Spearman correlation +lap_corr = None if not _tissue_mask_ok else lap.get("focus_lap_spearman_corr") +lap_corr_status = "N/A" if not _tissue_mask_ok else lap.get("focus_lap_corr_status") +corr_warn = dapi_cut.get("lap_focus_corr_warn") +corr_fail = dapi_cut.get("lap_focus_corr_fail") +if lap_corr_status == "uniform": + rows.append( + { + "Metric": "Focus-Laplacian correlation (Spearman)", + "Status": "N/A", + "Value": "Uniform — N/A", + "Cutoffs": "skipped (both metrics have CV < 0.1)", + "Notes": "Uniformly sharp slide — correlation is unstable and uninformative", + } + ) +else: + rows.append( + { + "Metric": "Focus-Laplacian correlation (Spearman)", + "Status": _status_from_thresholds( + float(lap_corr) if isinstance(lap_corr, (int, float)) and np.isfinite(float(lap_corr)) else None, + warn=float(corr_warn) if isinstance(corr_warn, (int, float)) else None, + fail=float(corr_fail) if isinstance(corr_fail, (int, float)) else None, + higher_is_better=True, + ), + "Value": f"{float(lap_corr):.3f}" if isinstance(lap_corr, (int, float)) and np.isfinite(float(lap_corr)) else "N/A", + "Cutoffs": ( + f"PASS >= {float(corr_warn):.2f}; WARN {float(corr_fail):.2f}-{float(corr_warn):.2f}; FAIL < {float(corr_fail):.2f}" + if isinstance(corr_warn, (int, float)) and isinstance(corr_fail, (int, float)) + else "not configured" + ), + "Notes": "Low correlation indicates the two sharpness metrics disagree", + } + ) + +# 6) 2D GMM Laplacian component separation +comp_sep = None if not _tissue_mask_ok else lap.get("component_separation") +comp_sep_warn = focus_cut.get("gmm_2d_laplacian_component_separation_warn") +comp_sep_fail = focus_cut.get("gmm_2d_laplacian_component_separation_fail") +rows.append( + { + "Metric": "2D GMM Laplacian component separation", + "Status": _status_from_thresholds( + comp_sep, + warn=comp_sep_warn if isinstance(comp_sep_warn, (int, float)) else None, + fail=comp_sep_fail if isinstance(comp_sep_fail, (int, float)) else None, + higher_is_better=True, + ), + "Value": f"{float(comp_sep):.2f}" if isinstance(comp_sep, (int, float)) and np.isfinite(float(comp_sep)) else "N/A", + "Cutoffs": ( + f"PASS >= {float(comp_sep_warn):.1f}; WARN {float(comp_sep_fail):.1f}-{float(comp_sep_warn):.1f}; FAIL < {float(comp_sep_fail):.1f}" + if isinstance(comp_sep_warn, (int, float)) and isinstance(comp_sep_fail, (int, float)) + else "not configured" + ), + "Notes": "Cohen's d between blurry/focused populations on a log(1 + Laplacian variance) scale", + } +) + +# 7) Boundary tiles excluded — diagnostic-only; surface the row only when it fires +n_boundary_excluded = 0 if not _tissue_mask_ok else lap.get("n_boundary_excluded", 0) +n_rois_used = 0 if not _tissue_mask_ok else lap.get("n_rois_used", 0) +if isinstance(n_boundary_excluded, (int, float)) and float(n_boundary_excluded) > 0: + rows.append( + { + "Metric": "Boundary tiles excluded", + "Status": "info", + "Value": f"{n_boundary_excluded} excluded, {n_rois_used} used", + "Cutoffs": "N/A", + "Notes": "Edge tiles excluded from Laplacian normalisation (reflect-padding artefact)", + } + ) + +df_focus = pd.DataFrame(rows, columns=["Metric", "Status", "Value", "Cutoffs", "Notes"]) + +# Cross-row contradiction notes — anchored to the row whose status drives the panel +_focus_extra_caveats: dict[str, str] = {} +if _tissue_mask_ok and _focus_med_status == "PASS" and status_focus in ("WARN", "FAIL"): + _focus_extra_caveats["Tiles in focus"] = ( + f"**Focus score median is PASS but blurry-tile fraction is {status_focus}** " + "— this is not contradictory. The **median** is the 50th-percentile focus score " + "across all tissue tiles: with " + f"{pct_focus:.0f}% of tiles in focus, the median falls inside the sharp population " + "and reports a healthy value. The **blurry-tile %** counts how many tiles the GMM placed " + "in the blurry component — a substantial minority can be blurry while the majority " + "(and therefore the median) remains sharp — the slide has " + "**regionally concentrated blurriness** rather than uniform degradation. Check the spatial " + "blurriness heatmap below to locate affected zones." + ) +elif _tissue_mask_ok and _focus_med_status == "WARN" and status_focus == "PASS": + _focus_extra_caveats["Focus score median"] = ( + "**Blurry-tile fraction is PASS but focus score median is WARN** — the GMM classifies " + "few tiles as blurry, yet the overall median focus score is low. This typically means " + "the slide is **uniformly soft** rather than having a distinct out-of-focus population: " + "most tiles have similar (mediocre) sharpness, so the GMM does not find two separable " + "components. Check the component separation metric — " + "low separation (d < 1) would confirm a unimodal focus distribution. Tissue type " + "matters here: brain and small-nucleus tissues inherently produce lower focus scores." + ) + +# Anchor sample-specific caveats to the Metric cell when Status is WARN/FAIL +_TILES_IN_FOCUS_TISSUE_CAVEAT = ( + "**Cross-check morphology metrics in [Section 2.1](#sec-2-1)** — Tiles in focus failures can be " + "tissue-context driven, not true imaging defects. Lumen-rich tissues (lung, intestine, " + "glandular epithelia) and cell-poor regions (white matter, adipose) naturally produce " + "localised low-focus zones because their tile-level CCFS is biased by tissue architecture " + "rather than by optical sharpness. Confirm whether the blurriness is spatially concentrated " + "(regional, often biology-driven) or diffuse (global, usually imaging-driven) before " + "treating this as a sample-level failure." +) + +def _build_status_part_list(name, status, value, cutoffs, notes): + parts = [f"**{name}** is currently **{status}**."] + val = str(value).strip() + if val and val.lower() not in ("n/a", "not available", ""): + parts.append(f"Value: **{val}**.") + cuts = str(cutoffs).strip() + if cuts and cuts.lower() not in ("n/a", "not configured", ""): + parts.append(f"Thresholds: {cuts}.") + nts = str(notes).strip() + if nts: + parts.append(nts + ".") + return parts + +for _i, _r in df_focus.iterrows(): + _s = str(_r["Status"]).strip().upper() + if _s not in ("WARN", "FAIL"): + continue + _name = str(_r["Metric"]) + _base_parts = _build_status_part_list(_name, _s, _r["Value"], _r["Cutoffs"], _r["Notes"]) + _extra = _focus_extra_caveats.get(_name) + + if _name == "Tiles in focus": + # Three-bullet caveat: 1) status/value/thresholds, 2) tissue-context cross-check, + # 3) cross-row contradiction note (when present, e.g. median-PASS but blurry-tile-FAIL). + _bullets = [" ".join(_base_parts), _TILES_IN_FOCUS_TISSUE_CAVEAT] + if _extra: + _bullets.append(_extra) + _body = "
    " + "".join( + f"
  • {b}
  • " for b in _bullets + ) + "
" + df_focus.at[_i, "Metric"] = metric_cell_with_caveat(_name, _s, _body) + else: + if _extra: + _base_parts.append(_extra) + df_focus.at[_i, "Metric"] = metric_cell_with_caveat(_name, _s, " ".join(_base_parts)) + +display_status_table(df_focus, ["Status"], narrow_cols={"Cutoffs": "16em", "Notes": "14em"}) + +# Warning when tissue mask failed — focus metrics are invalid +if not _tissue_mask_ok: + display(Markdown( + "> ⚠ **No tissue mask — focus score analysis is invalid.** " + "Tissue mask generation failed (0 tissue tiles detected), so blurry (GMM) classification, " + "Laplacian sharpness, and intensity QC could not distinguish tissue from background. " + "All focus/blurriness metrics above should be treated as **N/A**. This can occur with tissue " + "types that have very low or diffuse DAPI signal (e.g. skin, adipose, decalcified bone) " + "where the automatic tissue segmentation cannot find a foreground population. " + "Review the morphology overview and quality masks in [Section 2](#sec-2) to understand why mask " + "generation failed, and consider lowering the tile intensity floor in the thresholds " + "configuration if the tissue is genuinely present but below the default gate." + )) + +# Conditional warning when GMM component separation is poor +if isinstance(comp_sep, (int, float)) and np.isfinite(float(comp_sep)): + _cs = float(comp_sep) + _cs_fail = float(comp_sep_fail) if isinstance(comp_sep_fail, (int, float)) else 0.5 + _cs_warn = float(comp_sep_warn) if isinstance(comp_sep_warn, (int, float)) else 1.0 + if _cs < _cs_warn: + _severity = "FAIL" if _cs < _cs_fail else "WARN" + _gmm_msg = ( + f"> ⚠ **{'Low' if _severity == 'FAIL' else 'Moderate'} GMM component " + f"separation ({_severity}):** Cohen's d = {_cs:.2f} — the GMM cannot " + + ("reliably distinguish blurry from focused tiles. " + "**Do not use blurry cells (GMM) % for quality decisions on this sample.** " + if _severity == "FAIL" else + "blurry/focused populations are not well separated. " + "blurry cells (GMM) % should be interpreted with caution. ") + + "This can occur on uniformly well-focused slides (no real out-of-focus population) " + "or on tissues with diffuse morphology (brain, adipose) where Laplacian " + "variance is naturally low." + ) + display(Markdown(_gmm_msg)) + + # --- Honest fallback slide-level note (2026-06-22) --- + # GMM separation is poor, so blur cannot be assessed automatically. We do + # NOT issue a FAIL here: the only GMM-independent tile metric was the + # absolute Laplacian floor, which scales with brightness² and is least + # valid on the dim/degraded slides that reach this branch + # (qc_drift_analysis). CCFS low-nuclear-texture is the one per-cell signal + # worth surfacing; high CCFS → REVIEW, otherwise inspect manually. + # lap_var_median_raw is shown for reference only, not as a verdict. + _cm = cell_metrics or {} + _pct_ccfs = _cm.get("pct_low_nuclear_texture") + if _pct_ccfs is None and _cm.get("cells_low_nuclear_texture") and _cm.get("total_cells"): + _pct_ccfs = round(100.0 * _cm["cells_low_nuclear_texture"] / _cm["total_cells"], 2) + _ccfs_warn_thr = focus_cut.get("low_texture_cell_warn", 0.10) + _ccfs_bad = isinstance(_pct_ccfs, (int, float)) and _pct_ccfs >= _ccfs_warn_thr * 100 + + _lap_med_raw = lap.get("lap_var_median_raw") if isinstance(lap, dict) else None + _lap_val = f"{float(_lap_med_raw):.1f}" if isinstance(_lap_med_raw, (int, float)) else "N/A" + _ccfs_val = f"{_pct_ccfs:.1f}%" if isinstance(_pct_ccfs, (int, float)) else "N/A" + + if _ccfs_bad: + _verdict = "REVIEW" + _icon = "🔍" + _detail = ( + "A notable fraction of cells show low nuclear contrast (CCFS). This can reflect " + "tissue-specific nuclear morphology (e.g. large neurons) or localised defocus. " + "**Review spatial CCFS maps and focus heatmaps** to determine whether low-CCFS cells " + "cluster in specific regions before drawing a focus conclusion." + ) + else: + _verdict = "INSPECT" + _icon = "🔍" + _detail = ( + "No elevated cell-level low-texture signal, but blur cannot be assessed automatically " + "when GMM separation is this low. **Review the spatial focus heatmaps manually** rather " + "than relying on the blurry (GMM) % for this sample." + ) + + display(Markdown( + f"> {_icon} **Slide focus verdict (GMM unreliable):** **{_verdict}**\n>\n" + f"> - Median Laplacian variance (informational, not gated): {_lap_val}\n" + f"> - Cell-level CCFS nuclear texture: **{_ccfs_val}** low texture " + f"(threshold = {focus_cut.get('ccfs_low_texture_threshold', 0.02)})\n>\n" + f"> {_detail}" + )) + +if not lap: + tissue_qc = roi_metrics.get("tissue_mask_qc") or {} + if not tissue_qc.get("tissue_mask_generated", True): + display(Markdown( + "> **Note:** Laplacian sharpness metrics are not available for this sample. " + "Tissue mask generation failed (0 tissue tiles detected), so there are no " + "tissue tiles to compute Laplacian variance on. " + "Laplacian sharpness requires a valid tissue mask to distinguish in-tissue " + "focus quality from background. Review the tissue mask in section 4 and " + "consider whether the sample type or staining quality prevented mask generation." + )) + else: + display(Markdown( + "> **Note:** Laplacian sharpness metrics are not available for this sample. " + "This may indicate the analysis was run with an older pipeline version " + "that did not compute Laplacian variance." + )) +``` + +#### Spatial focus-score map across the Xenium region + +Puts the **per-tile focus score** in **space**: each square of the grid is coloured by how sharp that patch of tissue looks, so you can see **regional** patterns (e.g. edges, folds, or a soft band across the slide). + +```{python fig-focus-heatmap} +#| echo: false +display(fig_html("figures/grid_roi_focus_heatmap.png")) +``` + +#### Focus score - DAPI intensity dependence + +Each point is one tissue-overlapping tile: **horizontal axis** is **mean DAPI brightness** in that tile, **vertical axis** is **focus score**. It shows whether **dim or bright** regions systematically differ in sharpness (e.g. out of focus vs simply understained). + +```{python fig-focus-vs-intensity} +#| echo: false +display(fig_html("figures/roi_focus_vs_intensity.png")) +``` + +#### Focus score distribution + +Histogram of normalized focus scores across **tissue-overlapping tiles**, with the **GMM 2D classification** overlaid. The spread and skew indicate whether most of the slide is similarly sharp or whether a tail of poor tiles drives concern. + +```{python fig-focus-distribution-tissue} +#| echo: false +_p_focus_tissue = base / "figures" / "roi_focus_distribution_tissue.png" +if _p_focus_tissue.exists(): + display(fig_html("figures/roi_focus_distribution_tissue.png")) +else: + display(Markdown( + "*Focus distribution unavailable — tissue mask was not generated for this sample.*" + )) +``` + +### 3.3 Stain intensity {#sec-3-3} + +::: {.callout-note collapse="true" title="Metric explanation"} +**Per-channel metrics** (DAPI, Boundary, IntRNA). + +- **Status** — intensity is **advisory (WARN only, no FAIL)**: brightness does not track data quality, especially after XOA 4.0, so a dim sample is flagged for review, never failed on intensity alone. Flagged **WARN** when the below-floor % exceeds the WARN cutoff *or* the mean tile intensity itself falls below the intensity floor. +- **Mean intensity** — average raw signal in 16-bit counts across tissue tiles. +- **Intensity floor** — per-channel value below which a tile is counted as "below floor". The floor is **XOA-version-specific** (XOA 4.0 images are much dimmer than 3.x), selected from the bundle's analysis software version. +- **Percentage of tissue tiles below threshold** — fraction of tissue tiles whose mean signal falls below the per-channel intensity floor. +- **Mask-independent stain p99 (whole grid)** — the 99th-percentile per-tile intensity per channel, computed over the **whole tile grid with no tissue mask**, so it survives a tissue-mask-generation failure (when the masked metrics above read N/A). It is a **diagnostic, not a gate**: its Status mirrors the tissue-mask flag, and the values explain *why* a mask failed — a collapsed p99 across channels means the stain was too dim to detect tissue (e.g. faint skin), while a healthy p99 means the stain was fine and the mask failed on tissue geometry (a structural detection failure). It is **not a numeric floor** — an absolute p99 does not separate mask-PASS from mask-FAIL across the cohort, so it is read for cause, not used to pass/fail. + +**Threshold rationale** — defaults are tuned **per channel** to tolerate biologically expected heterogeneity (acellular stroma, adipose, necrotic areas, vessel-rich regions) where many tissue-overlapping tiles can be genuinely dim. + +```{python intensity-default-thresholds} +#| echo: false +# Render the default-threshold line dynamically from intensity_stats so it +# stays in sync with YAML cutoff changes without prose edits. Uniform cutoffs +# across DAPI / Boundary / IntRNA (2026-05-15 unification — see YAML comments). +# If cutoffs ever diverge again in future YAML edits, fall back to per-channel. +def _fmt_pct(v): + return f"{float(v):.0f}%" if isinstance(v, (int, float)) else "?" +_ch_stats = intensity_stats if isinstance(intensity_stats, dict) else {} +_d_cut = _ch_stats.get("dapi", {}) +_b_cut = _ch_stats.get("boundary", {}) +_i_cut = _ch_stats.get("intrna", {}) +_w_d, _f_d = _d_cut.get("pct_warn_threshold"), _d_cut.get("pct_fail_threshold") +_w_b, _f_b = _b_cut.get("pct_warn_threshold"), _b_cut.get("pct_fail_threshold") +_w_i, _f_i = _i_cut.get("pct_warn_threshold"), _i_cut.get("pct_fail_threshold") +# Intensity is WARN-only (no FAIL tier) since 2026-06-22: pct_fail_threshold is +# None. Render warn cutoffs only; the floor itself is XOA-version-specific. +_uniform = _w_d == _w_b == _w_i +if _uniform: + display(Markdown( + f"Intensity is **advisory (WARN only, no FAIL)**. WARN triggers when the " + f"**fraction of tissue tiles below the intensity floor** reaches " + f"`{_fmt_pct(_w_d)}` (uniform across DAPI / Boundary / IntRNA). The intensity " + f"floor itself is XOA-version-specific (XOA 4.0 images are much dimmer than 3.x)." + )) +else: + display(Markdown( + f"Intensity is **advisory (WARN only, no FAIL)**. WARN triggers on the " + f"**fraction of tissue tiles below the intensity floor**: " + f"**DAPI** `warn ≥ {_fmt_pct(_w_d)}`; " + f"**Boundary** `warn ≥ {_fmt_pct(_w_b)}`; " + f"**IntRNA** `warn ≥ {_fmt_pct(_w_i)}`. " + f"Floors are XOA-version-specific." + )) +``` + +Review against tissue context: the appropriate stringency varies significantly by tissue type and by channel, as described below. + +- **Boundary** is the most tissue-dependent channel and the most problematic in brain. The boundary stain cocktail (ATP1A1, E-Cadherin, CD45) targets epithelial and immune cell membranes. Brain parenchyma contains very few of these cell types across most regions — neurons, astrocytes, oligodendrocytes, and microglia do not express these markers in a way that produces clean membrane signal. Consequently, boundary stain is expected to be genuinely sparse and dim in brain tissue; this reflects biology, not assay failure. Flagging brain boundary tiles as poor quality based on standard thresholds is likely to produce false positives. Thresholds for this channel should be relaxed when processing brain samples, and QC failures here interpreted with caution. +- **IntRNA (18S)** signal is diffuse in brain due to the complex morphology of neural cells: cytoplasmic RNA is distributed across long axonal and dendritic processes rather than concentrated in the soma. This causes 18S signal to appear fragmented and low-contrast on a per-tile basis even in well-stained tissue. Dim tiles in this channel should not be interpreted as staining failure without corroborating evidence from the other channels. + +::: + +::: {.callout-tip collapse="true" title="Interpretation help"} +**Stain intensity histogram + heatmap** + +- A WARN on DAPI is the most actionable: weak DAPI undermines segmentation and propagates downstream. (Intensity is advisory only — it never hard-fails a sample.) +- A WARN on Boundary or IntRNA in isolation is often tissue-driven; cross-check with the tissue composition you expect. + +**Intensity assessment figure** + +- Look for channels that are globally too dim, heavily saturated, or spatially inconsistent. +- Use this figure and table to establish whether weak or uneven staining could affect downstream analysis — most notably cell segmentation. +- Spatial patchiness (one area noticeably dimmer) often points to uneven illumination or tissue thickness — distinguish from sample-wide failures. +::: + +#### Mask-independent stain p99 (mask-failure diagnostic) + +```{python stain-p99-diagnostic} +#| echo: false +# Stain p99 over the WHOLE tile grid (mask-independent, so it survives a +# mask-generation failure when the tissue-filtered summary below is N/A). +# Status mirrors the tissue-mask generation flag — a mask FAIL is the gate (the +# sample is unusable and also FAILs Stain intensity in Section 1). The p99 +# values are diagnostic only: they explain WHY a mask failed (collapsed p99 = +# dim stain, e.g. faint skin; healthy p99 = a structural mask-detection failure +# on otherwise-bright tissue). p99 is NOT a numeric floor — an absolute p99 does +# not separate mask-PASS from mask-FAIL across the cohort. +_sp = roi_metrics.get("stain_percentiles_whole_grid") or {} +_tm_p99 = roi_metrics.get("tissue_mask_qc") if isinstance(roi_metrics.get("tissue_mask_qc"), dict) else {} +_mask_status = str(_tm_p99.get("status", "")).upper() + +def _fmt_p99(ch): + v = (_sp.get(ch) or {}).get("p99") + return f"{float(v):.0f}" if isinstance(v, (int, float)) else "N/A" + +_p99_rows = [{ + "Metric": "Mask-independent stain p99 (whole grid)", + "Status": _mask_status if _mask_status in ("PASS", "WARN", "FAIL") else "N/A", + "p99 DAPI": _fmt_p99("dapi"), + "p99 Boundary": _fmt_p99("boundary"), + "p99 IntRNA": _fmt_p99("intrna"), +}] +display_status_table(pd.DataFrame(_p99_rows), status_cols=["Status"]) +if _mask_status == "FAIL": + display(Markdown( + "> ⚠ **Tissue-mask generation FAILED — sample unusable.** Intensity could " + "not be measured on a mask, so this sample also FAILs **Stain intensity** " + "in [Section 1](#sec-1). The p99 values above are mask-independent and " + "indicate *why*: a collapsed p99 across channels means the stain was too " + "dim to detect tissue (e.g. faint skin); a healthy p99 means the stain " + "was fine and the mask failed on tissue geometry (a structural detection " + "failure). The p99 is diagnostic, not a threshold." + )) +``` + +#### Summary + +```{python intensity-table} +#| echo: false + +if intensity_stats: + def _fmt_num1(v): + return ( + f"{float(v):.1f}" + if isinstance(v, (int, float)) and np.isfinite(float(v)) + else "N/A" + ) + # Display names for channels (case-sensitive user-facing labels). + _CHANNEL_DISPLAY = {"dapi": "DAPI", "boundary": "Boundary", "intrna": "IntRNA"} + ch_rows = [] + for ch in ("dapi", "boundary", "intrna"): + if ch not in intensity_stats: + continue + s = intensity_stats[ch] + crit_t = s.get("critical_threshold") + pct_bc = s.get("pct_tissue_roi_below_critical") + pct_warn_t = s.get("pct_warn_threshold") + pct_fail_t = s.get("pct_fail_threshold") + qs_raw = str(s.get("quality_status", "")).strip().lower() + ch_display = _CHANNEL_DISPLAY.get(ch, ch.upper()) + ch_label = ch_display + if qs_raw in ("warn", "fail", "warning", "critical"): + display_status = "WARN" if qs_raw in ("warn", "warning") else "FAIL" + _parts = [f"**{ch_display}** intensity is currently **{display_status}**."] + if isinstance(pct_bc, (int, float)) and np.isfinite(float(pct_bc)): + _parts.append(f"{float(pct_bc):.1f}% of tissue tiles are below the intensity floor.") + if isinstance(pct_warn_t, (int, float)): + _parts.append(f"WARN > {float(pct_warn_t):g}% (advisory; no FAIL tier).") + if ch == "dapi": + _parts.append( + "Low DAPI undermines segmentation; review the spatial intensity figure below " + "and the morphology mask in [Section 2](#sec-2) before treating it as a sample failure." + ) + elif ch == "intrna": + _parts.append( + "IntRNA is the most tissue-dependent channel. White matter, adipose, " + "or necrotic tumour areas can produce many low-IntRNA tiles in otherwise valid " + "slides — interpret against tissue composition." + ) + else: + _parts.append( + "Boundary signal can decouple from DAPI in stromal, lymphoid, multinucleated, " + "or mixed tissue/background tiles. Cross-check with morphology channel overview " + "in [Section 2.1](#sec-2-1)." + ) + ch_label = metric_cell_with_caveat(ch_display, display_status, " ".join(_parts)) + ch_rows.append( + { + "Channel": ch_label, + "Status": _intensity_display_status(str(s.get("quality_status", ""))), + "Mean intensity": _fmt_num1(s.get("mean")), + "Intensity floor": f"{int(float(crit_t))}" if isinstance(crit_t, (int, float)) and np.isfinite(float(crit_t)) else "N/A", + "Percentage of tissue tiles below threshold": ( + f"{float(pct_bc):.1f}%" + if isinstance(pct_bc, (int, float)) and np.isfinite(float(pct_bc)) + else "N/A" + ), + } + ) + + # Overall footer row — aggregates the per-channel statuses with an escalation rule + if ch_rows and "overall_quality" in intensity_stats: + overall_status = _intensity_display_status(str(intensity_stats["overall_quality"])) + overall_label = "Overall" + if overall_status in ("WARN", "FAIL"): + _o_parts = [ + f"**Overall intensity quality** is **{overall_status}**.", + "Status logic: **WARN** if any channel is WARN, otherwise PASS. " + "Intensity is advisory only — there is no FAIL tier (intensity does " + "not track quality after XOA 4.0).", + ] + overall_label = metric_cell_with_caveat("Overall", overall_status, " ".join(_o_parts)) + ch_rows.append( + { + "Channel": overall_label, + "Status": overall_status, + "Mean intensity": "—", + "Intensity floor": "—", + "Percentage of tissue tiles below threshold": "—", + } + ) + + if ch_rows: + _df_int = pd.DataFrame(ch_rows) + _sty_int = ( + _df_int.style + .set_table_styles( + [ + {"selector": "th", "props": "text-align:left; vertical-align:middle; background-color:#F8FAFC; white-space:normal;"}, + {"selector": "td", "props": "text-align:left; vertical-align:middle; white-space:nowrap; line-height:1.5;"}, + {"selector": "td details[open]", "props": "white-space:normal;"}, + {"selector": "td details[open] > div", "props": "white-space:normal;"}, + ] + ) + .hide(axis="index") + ) + if "Status" in _df_int.columns: + _q_styles = [_status_cell_css(v) for v in _df_int["Status"]] + _sty_int = _sty_int.apply(lambda _col: _q_styles, subset=["Status"]) + # Colour % tissue tiles below threshold: green/orange/red based on per-channel YAML thresholds + def _pct_below_css(col): + styles = [] + for idx, v in enumerate(col): + try: + val = float(str(v).rstrip("%")) + ch = _df_int.iloc[idx]["Channel"] + ch_s = intensity_stats.get(ch, {}) + # Intensity is WARN-only (pct_fail_threshold is None): no red. + fail_t = ch_s.get("pct_fail_threshold") + warn_t = ch_s.get("pct_warn_threshold", 15.0) + if isinstance(fail_t, (int, float)) and val > fail_t: + styles.append(_status_cell_css("CRITICAL")) + elif val > warn_t: + styles.append(_status_cell_css("WARNING")) + else: + styles.append(_status_cell_css("PASS")) + except (ValueError, TypeError): + styles.append("") + return styles + if "Percentage of tissue tiles below threshold" in _df_int.columns: + _sty_int = _sty_int.apply(_pct_below_css, subset=["Percentage of tissue tiles below threshold"]) + # Mark the Overall summary row visually distinct from per-channel rows + _sty_int = _sty_int.apply(_summary_row_styles, axis=1) + display(HTML(_sty_int.to_html())) + # Check if all channels are not_available — tissue mask failure + _int_all_na = all( + str(intensity_stats.get(ch, {}).get("quality_status", "")).strip().lower() == "not_available" + for ch in ("dapi", "boundary", "intrna") + if ch in intensity_stats + ) + if _int_all_na and ch_rows: + _tm_int = roi_metrics.get("tissue_mask_qc") if isinstance(roi_metrics.get("tissue_mask_qc"), dict) else {} + _mask_gen_int = _tm_int.get("tissue_mask_generated", True) + _rit_int = _tm_int.get("rois_in_tissue", -1) + if not _mask_gen_int or (isinstance(_rit_int, (int, float)) and _rit_int == 0): + display(Markdown( + "> ⚠ **All channels are unavailable.** " + "Tissue mask generation failed (0 tissue tiles detected), so per-channel " + "intensity thresholds could not be evaluated — there are no tissue tiles " + "to classify as dim or adequate. This is expected for tissue types with very " + "low or diffuse DAPI signal (e.g. skin, adipose, decalcified bone). " + "Cross-check with the morphology masks in [Section 2.1](#sec-2-1)." + )) + else: + display(Markdown( + "> ⚠ **All channels are unavailable.** " + "Intensity QC ran but could not assess any channel. " + "Check the pipeline logs and the intensity assessment outputs for details." + )) +else: + print("_No intensity_assessment.json — skip table._") +``` + +#### Stain intensity: histograms and spatial heatmaps + +Multipanel figure: +*top row* — per-tile intensity histograms for DAPI, Boundary, and IntRNA, with dashed lines marking each channel's intensity floor; +*bottom row* — spatial heatmaps of the same per-tile intensities laid over the slide footprint, so the reader can see *where* any dim regions concentrate. + +```{python fig-intensity} +#| echo: false +display(fig_html("figures/intensity_assessment.png", max_width="95%")) +``` + + + + +### 3.4 Signal-to-Noise Ratio {#sec-3-4} + +Signal-to-noise ratio at tile resolution, computed from image-derived (DAPI contrast) and transcript-derived (decoded spots vs negative-control probes) sources. These two views are complementary: image SNR catches optical-acquisition issues, transcript SNR catches probe-decoding issues. Low SNR compromises downstream expression and neighbourhood analyses. + +::: {.callout-note collapse="true" title="Metric explanation"} +*dB = decibel, a logarithmic unit for expressing ratios. The pipeline uses the amplitude convention `20·log₁₀(signal/noise)`, so +6 dB ≈ 2× signal, +20 dB ≈ 10× signal.* + +**Image SNR (tile quartile, dB)** — image-only contrast across the slide. All tissue tiles are sorted by their mean DAPI intensity, then split into the brightest 25% (foreground) and dimmest 25% (background). The score is `20·log₁₀(mean_top_quartile / std_bottom_quartile)`. Using quartile splits limits the influence of a few outlier tiles. Higher value = brighter tissue stands out more cleanly from dim regions. +*Best used for:* slide-wide staining homogeneity and cross-sample comparison — it's a single global aggregate and will **not** flag localised problems (use the Negative-control spatial uniformity metric and the SNR spatial heatmap for those). + +**Image SNR (Otsu split)** — same dB-like scoring as above, but the foreground/background split is done *per tile* on the tile's own pixels, not across the slide. Otsu's method picks a histogram cut-point that maximises separation between dim and bright pixels in that tile, then `20·log₁₀(mean_signal_pixels / std_background_pixels)` is computed. The slide-level value is the median across tiles. Same dB thresholds as above. +*Best used for:* per-tile contrast quality under uneven staining — adapting the split to each tile's local distribution is more tolerant of inter-tile intensity variation than the global tile-quartile metric, so it surfaces problems that are visible *within* tiles rather than just *between* them. Median-across-tiles still aggregates the slide; for rare localised artefacts use the Negative-control spatial uniformity metric below. + +**Transcript SNR (per tile)** — molecular SNR from decoded transcripts (not pixel intensity). Each decoded spot is assigned to a tile by its xy coordinates. Per tile, two quantities: `real / negative` (target-to-noise ratio) and `negative / total` (negative-probe burden). Slide-level uses the median across tiles. *Best used for:* catching molecular-side decoding problems independent of imaging quality — a slide can have high optical quality yet decode poorly (or the reverse); the image-based SNRs above do not capture this difference. + +**Negative-control spatial uniformity** — checks whether the negative-probe burden is spread evenly across the slide. Two measurements: +- *Quadrant spread* (always computed): tiles are split into four spatial quadrants, the mean negative-probe fraction is computed in each, and a spread index summarises the disagreement across the four. High spread = noise concentrated in one region. Defaults: WARN > 0.5, FAIL > 1.0. +- *Moran's I* (optional, when `esda` / `libpysal` are installed): spatial autocorrelation across all tiles. Detects small contiguous noise blobs that quadrant averaging would smooth over. +*Best used for:* localised noise concentration — folds, mounting drift, washing artefacts, debris. A high score means the *average* noise level might still pass while the noise is patchy in space; it's the complement to Image SNR (tile quartile, dB), which can't see such localised effects. + +**Note on units** — Image SNR (tile quartile, Otsu split) is reported in dB (`20·log₁₀(mean_foreground / std_background)`; same amplitude convention as the [Section 3.4](#sec-3-4) intro). Transcript SNR is a unitless ratio (decoded target count / negative-control count per tile). Negative-control spatial uniformity is a unitless spread index across slide quadrants. + +**Visualisations** + +- **SNR spatial heatmap** — two-panel slide overview at tile resolution, scoped to tiles inside the tissue mask. *Left:* per-tile fraction of negative-control probes (`negative / total`); dark = clean, bright = high noise burden. *Right:* per-tile real-to-negative ratio on a `log₁₀` scale; dark = high SNR, bright = noise approaching signal. Both panels use fixed colorbars (left 0–40%; right 1×–1000× log) so colours mean the same thing across samples. Tiles past the cap saturate at the deepest red with an `extend` triangle. Dashed and solid lines on each colorbar mark WARN (orange) and FAIL (red) cut-points: 15% / 30% on the left, 60× / 30× on the right. + + *Pseudocount note on the right panel:* the ratio is rendered as `(real+1)/(neg+1)` with an α=1 Laplace pseudocount (display-only — does not affect verdicts or JSON). Without it, tiles with `neg=0` would give `inf` and tiles with `real=0` would give `0`, both dropping out as NaN on the log axis. With it, those tiles paint at the colorbar extremes (green for "no noise detected", red for "all noise") instead of disappearing. Effect on well-populated tiles is negligible. +::: + +::: {.callout-tip collapse="true" title="Interpretation help"} +**Per-component SNR (image-derived)** + +- Lower **Image SNR (tile quartile, dB)** → weaker optical contrast between signal-rich and background-like regions. +- Lower **Image SNR (Otsu split)** → poorer separation of tissue signal from low-intensity background. +- Lower **Transcript SNR (per tile)** ratio (or higher negative fraction) → weak molecular separation in decoded transcript space. +- Higher **Negative-control spatial uniformity** unevenness → localised technical artefacts rather than uniform background. + + +**SNR spatial heatmap** + +- Red regions on the **left** panel (negative-control fraction) → high noise burden in those tiles. +- Red regions on the **right** panel (real-to-negative ratio) → low signal separation. +- Look for spatial overlap with focus / blurriness maps from [Section 3.2](#sec-3-2); correlated red zones suggest a single root cause (fold, optical issue) rather than independent failures. + +**Image-derived quality vs transcript-derived quality** + +These three panels test whether image quality predicts transcript quality at the tile level, each using a different image-quality axis. The **top and middle panels** share a y-axis (transcript SNR ratio, high = clean), so healthy tiles cluster in the upper-right quadrant — sharper optics or higher image contrast tend to coincide with stronger molecular signal. The **bottom panel** inverts the y-axis convention (negative-probe fraction, low = clean), so it is interpreted separately in the bottom-panel bullet below. For the top and middle panels, weak overall correlation has two interpretations: both axes uniformly high (expected for good data), or molecular quality varying independently of imaging (a concern); off-pattern clusters point to specific failure modes covered below. + +- *Optical sharpness vs transcript quality (top panel):* a positive correlation → out-of-focus regions also have poorer transcript detection; image focus is constraining the molecular readout. A weak correlation is decoupled imaging and transcript quality — check whether that's because both are uniformly high (ideal) or because molecular quality varies independently of imaging (probe panel / hybridisation, not acquisition). +- *Image SNR vs transcript quality (middle panel):* a positive correlation → tiles with poor local foreground/background contrast also have poor transcript SNR; image acquisition (contrast, staining, exposure) is limiting downstream decoding. A weak correlation → image SNR and transcript SNR are tracking independent quality axes. +- *Signal strength vs noise contamination (bottom panel):* the y-axis (negative-probe fraction) is bounded above by the small fraction of negative-control codewords in the panel, so most tiles cluster near zero regardless of DAPI level — a roughly flat trend is the expected null. A strong positive slope is anomalous (potential contamination scaling with cell density). A strong negative slope can arise when per-area background noise dominates dim-DAPI tiles. + +*The quadrant patterns below apply to the **top and middle panels** only (both with y = transcript SNR ratio). The bottom panel's interpretation is its own bullet above.* + +- *Upper-right cluster (high image + high transcript quality):* typical clean tissue. Most tiles fall here in a healthy sample; a sparse upper-right population indicates either poor tissue coverage or widespread low quality. +- *Lower-left cluster (low image + low transcript quality):* tiles where both optical and molecular quality are degraded. Concentration at tissue periphery is an edge-effect; concentration in interior tissue indicates a real quality problem (folds, debris, focus drift). +- *Upper-left — high focus but low transcript SNR:* tiles are sharp optically yet molecular noise is high. Points to probe panel design or hybridisation issues rather than acquisition; cross-check [Section 3.3](#sec-3-3) stain intensity for systematic channel underperformance. +- *Lower-right — low focus but high transcript SNR:* unusual — out-of-focus tiles where transcripts still decode well. This usually indicates the focus metric is responding to low nuclear texture (sparse nuclei, neural tissue with diffuse chromatin) rather than genuine defocus; cross-check [Section 4.4.a](#sec-4-4-a) per-channel intensity correlation for whether DAPI is the **strongest predictor of transcript yield** in this tissue. +- *Points coloured by blurry (GMM) classification (red = blurry, blue = in-focus):* red cluster overlapping the low-SNR region means the two QC systems agree on the same tiles being problematic. Red scattered across all SNR levels means the GMM is catching something other than what affects molecular quality (often biology, especially in neural tissue). +- *Threshold lines:* mark WARN (orange dashed) and FAIL (red dashed) from the thresholds configuration. Most tiles should sit safely above WARN. +::: + +#### Summary + +```{python snr-overall} +#| echo: false + +if not snr_summary: + print("_No SNR block found (run image QC with SNR enabled, or add snr_metrics.json)._") +else: + comp = snr_summary.get("components", {}) if isinstance(snr_summary.get("components", {}), dict) else {} + roi_keys = ("SNR_image_roi_quartile_db", "SNR_image_otsu", "SNR_roi_tx", "SNR_roi_neg_spatial") + matrix_keys = () # Plummer/SpatialQM moved to molecule QC (v5 restructure) + + rank = {"PASS": 0, "WARN": 1, "FAIL": 2} + def _display_verdict(payload): + reason = str((payload or {}).get("reason", "") or (payload or {}).get("moran_note", "")).lower() + if "modulenotfounderror" in reason or "no module named" in reason: + return "N/A" + return str((payload or {}).get("verdict", "")).upper() + def _rollup(keys): + vals = [] + for k in keys: + vu = _display_verdict(comp.get(k) or {}) if isinstance(comp.get(k), dict) else "" + if vu in rank: + vals.append(vu) + if not vals: + return "NOT_COMPUTED" + return max(vals, key=lambda x: rank[x]) + + roi_ov = _rollup(roi_keys) + matrix_ov = _rollup(matrix_keys) + # Rollup across the four tile-level SNR components — drives the single + # [Section 3.4](#sec-3-4) Overall pill in the Summary table. matrix_keys is empty + # post-v5 (Plummer/SpatialQM moved to molecule QC). + snr_ov = _rollup(roi_keys + matrix_keys) +``` + +```{python snr-components} +#| echo: false + +# Image-derived SNR component keys (tile / image / transcript / spatial uniformity). +_SNR_ROI_KEYS = ( + "SNR_image_roi_quartile_db", + "SNR_image_otsu", + "SNR_roi_tx", + "SNR_roi_neg_spatial", +) +# Matrix-derived SNR component keys (Plummer + SpatialQM) moved to molecule +# QC under the v5 restructure. Empty tuple preserved so downstream concat +# expressions (roi_keys + matrix_keys) keep working without further edits. +_SNR_MATRIX_KEYS = () + +# Cutoff strings per [Section 3.4](#sec-3-4) metric, derived from conf/roi_image_qc_thresholds.yaml. +# Used by the Detailed-metrics tabs to populate the Cutoffs column. +_snr_cut = ((roi_cutoffs.get("image_qc") or {}).get("snr") or {}) if isinstance(roi_cutoffs, dict) else {} +_img_snr_warn = float((_snr_cut.get("image_snr_db") or {}).get("warn", 15.0)) +_img_snr_fail = float((_snr_cut.get("image_snr_db") or {}).get("fail", 10.0)) +_roi_tx_t = _snr_cut.get("roi_tx") or {} +_roi_tx_rwarn = float(_roi_tx_t.get("ratio_warn", 3.0)) +_roi_tx_rfail = float(_roi_tx_t.get("ratio_fail", 1.5)) +_roi_tx_nwarn = float(_roi_tx_t.get("neg_pct_warn", 0.15)) +_roi_tx_nfail = float(_roi_tx_t.get("neg_pct_fail", 0.30)) +_neg_spat_t = _snr_cut.get("neg_spatial") or {} +_qs_warn = float(_neg_spat_t.get("quadrant_spread_warn", 0.5)) +_qs_fail = float(_neg_spat_t.get("quadrant_spread_fail", 1.0)) + +_SNR_TILE_CUTOFFS = { + "SNR_image_roi_quartile_db": { + "snr_db": f"PASS ≥ {_img_snr_warn:g} dB; WARN {_img_snr_fail:g}–{_img_snr_warn:g} dB; FAIL < {_img_snr_fail:g} dB", + }, + "SNR_image_otsu": { + "snr_db_median": f"PASS ≥ {_img_snr_warn:g} dB; WARN {_img_snr_fail:g}–{_img_snr_warn:g} dB; FAIL < {_img_snr_fail:g} dB", + }, + "SNR_roi_tx": { + "median_roi_tx_snr_ratio": f"PASS ≥ {_roi_tx_rwarn:g}; WARN {_roi_tx_rfail:g}–{_roi_tx_rwarn:g}; FAIL < {_roi_tx_rfail:g}", + "median_neg_pct": f"PASS ≤ {_roi_tx_nwarn:g}; WARN {_roi_tx_nwarn:g}–{_roi_tx_nfail:g}; FAIL > {_roi_tx_nfail:g}", + }, + "SNR_roi_neg_spatial": { + "quadrant_spread_index": f"PASS ≤ {_qs_warn:g}; WARN {_qs_warn:g}–{_qs_fail:g}; FAIL > {_qs_fail:g}", + }, +} + +# Human-readable display names for SNR component JSON IDs rendered in the [Section 3.4](#sec-3-4) Summary table. +_SNR_DISPLAY_NAMES = { + "SNR_image_roi_quartile_db": "Image SNR (tile quartile, dB)", + "SNR_image_otsu": "Image SNR (Otsu split)", + "SNR_roi_tx": "Transcript SNR (per tile)", + "SNR_roi_neg_spatial": "Negative-control spatial uniformity", +} + +# Method-code → short human-readable algorithm name. +_SNR_METHOD_NAMES = { + "per_roi_otsu": "Otsu thresholding", + "roi_df_quartiles": "Quartile-based dB ratio", + "roi_tx_target_vs_neg": "Decoded target vs negative-control", + "quadrant_spread": "Quadrant variance + Moran's I", +} + +# Method-code → one-line description of what the method computes. +_SNR_METHOD_DESCRIPTIONS = { + "per_roi_otsu": "Otsu threshold separates signal from background tiles; their contrast is converted to a dB-like SNR.", + "roi_df_quartiles": "Compares robust upper- vs lower-quartile DAPI intensity per tile; ratio converted to a dB-like SNR.", + "roi_tx_target_vs_neg": "Per tile, counts decoded target transcripts vs negative-control probes; reports the median target-to-negative ratio and median negative-probe burden across tiles.", + "quadrant_spread": "Measures dispersion / clustering of negative-control burden across image quadrants (with optional Moran's I).", +} + + +def _info_label(name: str, body_md: str) -> str: + """Wrap a label in an inline
with a neutral info ⓘ chip (no warning colour).""" + import re as _re + body_html = _re.sub(r"\*\*(.+?)\*\*", r"\1", body_md) + body_html = _re.sub(r"`([^`]+)`", r"\1", body_html) + chip = ( + "" + "ⓘ▾" + "" + ) + return ( + f"
" + f"" + f"{name}{chip}" + f"" + f"
" + f"{body_html}" + f"
" + ) + + +def display_snr_detailed_metrics(sub: dict, *, show_cutoffs: bool = True, cutoff_overrides: dict | None = None) -> None: + """Render one SNR component's payload as a Metric / Value [/ Cutoffs] table. + + Strips status, verdict, reason, and method-machinery keys from the body. + Pulls method metadata into an expandable on a synthetic top row, and + formats `thresholds_used` (when present) into the Cutoffs column on the + row whose key drives the verdict. + + Parameters + ---------- + sub : dict + Component payload from snr_summary["components"][...]. + show_cutoffs : bool, optional + If False, omit the Cutoffs column entirely. Use for [Section 3.4](#sec-3-4) + components whose thresholds live in external YAML rather than the + payload itself, so the column would always be blank. + """ + if not isinstance(sub, dict): + print("_Metric not available for this run._") + return + + _SUPPRESS = { + "status", "verdict", "quadrant_verdict", "reason", "error", + "moran_note", "moran_subsample", "moran_p_pseudo", "moran_neighbors", + "moran_i", "moran_z", "report_note", "per_roi_table_file", + } + _METHOD_KEYS = ("method", "method_primary", "intensity_col", "map_key", "formula") + + method_code = str(sub.get("method", sub.get("method_primary", "")) or "") + method_name = _SNR_METHOD_NAMES.get(method_code, method_code) if method_code else "" + method_desc = _SNR_METHOD_DESCRIPTIONS.get(method_code, "") + + method_extras = [f"**{k}**: `{sub[k]}`" for k in _METHOD_KEYS if k in sub] + + # Threshold formatting → Cutoffs column. + _thr = sub.get("thresholds_used") if isinstance(sub.get("thresholds_used"), dict) else None + cutoffs_target_key = None + cutoffs_str = "" + if _thr: + if "metric" in _thr: + cutoffs_target_key = _thr["metric"] + warn = _thr.get("warn") + fail = _thr.get("fail") + if isinstance(warn, (int, float)) and isinstance(fail, (int, float)): + cutoffs_str = f"PASS ≥ {warn:g}; FAIL < {fail:g}" + elif "pass_min" in _thr or "fail_below" in _thr: + cutoffs_target_key = "snr" + pmin = _thr.get("pass_min") + fbelow = _thr.get("fail_below") + if isinstance(pmin, (int, float)) and isinstance(fbelow, (int, float)): + cutoffs_str = f"PASS ≥ {pmin:g}; FAIL < {fbelow:g}" + + rows = [] + + # Compose the Method-info body that will live in the Metric column header + # (instead of a synthetic top row). + _header_body_parts = [] + if method_name and method_name != method_code: + _header_body_parts.append(f"**Method:** {method_name}") + elif method_code: + _header_body_parts.append(f"**Method:** `{method_code}`") + if method_desc: + _header_body_parts.append(method_desc) + if method_extras: + _header_body_parts.append("Internal keys: " + "; ".join(method_extras) + ".") + metric_header = ( + _info_label("Metric", " ".join(_header_body_parts)) + if _header_body_parts + else "Metric" + ) + + for k, v in sub.items(): + if k in _SUPPRESS or k in _METHOD_KEYS or k == "thresholds_used": + continue + if isinstance(v, dict): + continue + if isinstance(v, list): + v_str = ", ".join(f"{x:.3g}" if isinstance(x, float) else str(x) for x in v[:8]) + if len(v) > 8: + v_str += ", …" + elif isinstance(v, float): + v_str = f"{v:.4g}" + else: + v_str = str(v) + row = {"Metric": k, "Value": v_str} + if show_cutoffs: + if cutoff_overrides and k in cutoff_overrides: + row["Cutoffs"] = cutoff_overrides[k] + elif cutoffs_target_key and k == cutoffs_target_key and cutoffs_str: + row["Cutoffs"] = cutoffs_str + else: + row["Cutoffs"] = "—" + rows.append(row) + + if not rows: + print("_No metrics to display._") + return + + columns = ["Metric", "Value", "Cutoffs"] if show_cutoffs else ["Metric", "Value"] + df = pd.DataFrame(rows, columns=columns) + + # Build per-row Value-cell coloring using the component-level verdict. + # A row is "verdict-bearing" when its Metric key is referenced by the + # cutoffs configuration (i.e., appears in cutoff_overrides, or matches + # cutoffs_target_key from thresholds_used). The component verdict + # (sub["verdict"]) colours those cells; others stay uncoloured. + _verdict = str(sub.get("verdict", "")).upper() + _verdict_keys = set() + if cutoff_overrides: + _verdict_keys.update(cutoff_overrides.keys()) + if cutoffs_target_key: + _verdict_keys.add(cutoffs_target_key) + _value_cell_styles = [ + _status_cell_css(_verdict) + if (_verdict in {"PASS", "WARN", "FAIL"} and r["Metric"] in _verdict_keys) + else "" + for r in rows + ] + + sty = ( + df.style + .set_table_styles([ + {"selector": "th", "props": "text-align:left; vertical-align:top; background-color:#F8FAFC;"}, + {"selector": "td", "props": "text-align:left; vertical-align:top;"}, + ]) + .hide(axis="index") + .format_index( + lambda c: metric_header if c == "Metric" else c, + axis=1, + escape=None, + ) + .apply(lambda _col: _value_cell_styles, subset=["Value"]) + ) + display(HTML(sty.to_html())) + +if snr_summary and "components" in snr_summary: + comp = snr_summary["components"] + recs = [] + for name in _SNR_ROI_KEYS + _SNR_MATRIX_KEYS: + payload = comp.get(name) + if not isinstance(payload, dict): + continue + reason = str(payload.get("reason", "") or payload.get("moran_note", "")) + display_verdict = _display_verdict(payload) + display_name = _SNR_DISPLAY_NAMES.get(name, name) + method_code = str(payload.get("method", payload.get("method_primary", ""))) + if not method_code: + # Fallback for older JSON payloads that didn't emit a method key. + method_code = { + "SNR_roi_tx": "roi_tx_target_vs_neg", + }.get(name, "") + method_name = _SNR_METHOD_NAMES.get(method_code, method_code) + method_desc = _SNR_METHOD_DESCRIPTIONS.get(method_code, "") + if reason: + note_text = f"{method_desc} ({reason})" if method_desc else reason + else: + note_text = method_desc + metric_label = display_name + if str(display_verdict).upper() in ("WARN", "FAIL"): + _parts = [f"**{display_name}** verdict is **{display_verdict}**."] + if reason: + _parts.append(f"Reason: {reason}.") + _hints = { + "SNR_image_roi_quartile_db": ( + "Weak optical contrast between signal-rich and background-like tiles. " + "Check intensity ([Section 3.3](#sec-3-3)) and focus ([Section 3.2](#sec-3-2)) before treating it as a noise problem." + ), + "SNR_image_otsu": ( + "Otsu split between tissue signal and low-intensity background is poor. " + "Often co-occurs with low DAPI intensity — see [Section 3.3](#sec-3-3)." + ), + "SNR_roi_tx": ( + "Decoded transcripts vs negative controls are close at tile level. " + "Inspect the SNR spatial heatmap below for whether the issue is global or patchy." + ), + "SNR_roi_neg_spatial": ( + "Negative-control burden is spatially uneven — likely a localised technical " + "artefact rather than uniform background. Check tissue mask and edge zones in [Section 2](#sec-2)." + ), + } + if name in _hints: + _parts.append(_hints[name]) + metric_label = metric_cell_with_caveat(display_name, display_verdict, " ".join(_parts)) + recs.append( + { + "Metric": metric_label, + "Status": display_verdict, + "Method": method_name, + "Note": note_text, + } + ) + if recs: + # Overall footer row — worst PASS/WARN/FAIL across the four tile-level components. + overall_label = "Overall" + if str(snr_ov).upper() in ("WARN", "FAIL"): + _o_parts = [ + f"**SNR verdict** is **{snr_ov}**.", + "Aggregation: worst PASS/WARN/FAIL across **Image SNR (tile quartile, dB)**, " + "**Image SNR (Otsu split)**, **Transcript SNR (per tile)**, and " + "**Negative-control spatial uniformity**.", + "Cross-check against [Section 3.3](#sec-3-3) (intensity) and [Section 3.2](#sec-3-2) (focus); a tile-level " + "SNR failure often co-occurs with low DAPI intensity or out-of-focus zones.", + ] + overall_label = metric_cell_with_caveat("Overall", str(snr_ov), " ".join(_o_parts)) + recs.append( + { + "Metric": overall_label, + "Status": str(snr_ov), + "Method": "—", + "Note": "—", + } + ) + display_status_table(pd.DataFrame(recs), ["Status"]) + else: + print("_No SNR components in JSON (check SNR enabled and inputs)._") + + # Notes for specific components + quartile = comp.get("SNR_image_roi_quartile_db") or {} + if quartile.get("status") == "skipped" and quartile.get("reason") in ("too_few_tissue_rois",): + tissue_qc = roi_metrics.get("tissue_mask_qc") or {} + if not tissue_qc.get("tissue_mask_generated", True): + display(Markdown( + "> **Note:** Quartile image SNR was skipped because tissue mask generation " + "failed (0 tissue tiles detected). This metric requires tissue tiles " + "to separate foreground from background intensity. Review the tissue mask in " + "section 4." + )) + + neg_spatial = comp.get("SNR_roi_neg_spatial") or {} + report_note = neg_spatial.get("report_note") + if report_note: + display(Markdown(f"> **Note:** {report_note}")) +``` + +#### SNR Spatial Heatmap + +Spatial distribution of transcript-level noise across the slide. + +**Left:** fraction of negative-control probes per tile — red regions have high noise burden. +**Right:** real-to-negative transcript ratio per tile (log scale) — red regions have low signal separation. Dashed/solid lines on each colorbar mark WARN (orange dashed) and FAIL (red solid) thresholds. + +```{python snr-heatmap-fig} +#| echo: false +display(fig_html("figures/snr_heatmap.png")) +``` + +#### Image-derived quality vs transcript-derived quality (per-tile) + +Per-tile relationships between image-derived and transcript-derived quality metrics. + +**Top:** Focus score vs transcript SNR. **Middle:** Image SNR (Otsu split, per-tile dB) vs transcript SNR. **Bottom:** DAPI intensity vs negative-probe burden. + + +```{python concordance-fig} +#| echo: false +_concordance_path = base / "figures" / "cross_section_concordance.png" +if _concordance_path.exists(): + display(fig_html("figures/cross_section_concordance.png", max_width="65%")) +else: + display(Markdown("*Image-derived quality vs transcript-derived quality plot not available (requires both focus scores and transcript SNR).*")) +``` + +## 4. Cell-level metrics {#sec-4} + +Per-cell metrics derived from segmentation masks — rendered only when the bundle supplies segmentation. Verdicts are advisory (PASS / WARN only, no FAIL tier): candidates for review, not sample-level filtering decisions. Metric definitions and biological caveats live in the subsection callouts. + +```{python cell-section} +#| echo: false + +if cell_metrics is None: + print("_No cell-level image QC metrics for this run (expected for bundle-only / no cells path)._") +else: + display(Markdown("### 4.1 Cell quality scores summary {#sec-4-1}")) + display(Markdown( + "Use this table to identify cells (or whole clusters) that may need " + "verification or filtering before downstream analysis. Inspect flagged " + "cells before deciding whether to filter — some will be biologically " + "real, not QC failures." + )) + + # Cutoffs read from YAML at render time (no new keys — all reuse existing). + # Loaded BEFORE the 📖 callout so the callout can interpolate live values. + _a_focus_cut = (roi_cutoffs.get("image_qc") or {}).get("focus") or {} + _a_cw = _a_focus_cut.get("low_texture_cell_warn", 0.10) + _a_cf = _a_focus_cut.get("low_texture_cell_fail", 0.25) + _a_n_total = cell_metrics.get("total_cells", 0) + _a_ccfs_thr = cell_metrics.get("ccfs_low_texture_threshold", 0.02) + _a_ch_cfg = ((roi_cutoffs.get("image_qc") or {}).get("channels") or {}) + # Show the ACTUAL per-channel floor the cell-level flagging used. The + # cell-level code applies the XOA-version-specific floor (same as the tile + # level), emitted as intensity_quality..critical_threshold; fall back to + # the flat YAML intensity_critical only if that's unavailable. Avoids showing + # the bright-era 500/100/300 when a dim XOA-4.0 floor was actually applied. + _a_iq = ( + intensity_stats + if intensity_stats + else (roi_metrics.get("intensity_quality") or {}) + ) + def _a_floor(_iq_key, _cfg_key, _default): + _v = (_a_iq.get(_iq_key) or {}).get("critical_threshold") + if isinstance(_v, (int, float)): + return int(_v) + return (_a_ch_cfg.get(_cfg_key) or {}).get("intensity_critical", _default) + _a_ic_dapi = _a_floor("dapi", "DAPI", 500) + _a_ic_boundary = _a_floor("boundary", "boundary", 100) + _a_ic_intrna = _a_floor("intrna", "intRNA", 300) + _a_spatial_cfg = ((roi_cutoffs.get("image_qc") or {}).get("spatial_context") or {}) + _a_artif_warn = _a_spatial_cfg.get("in_artifact_warn") + _a_artif_fail = _a_spatial_cfg.get("in_artifact_fail") + # Cell-level advisory cutoffs (PASS / WARN only, no FAIL). Read here so the + # metric-explanation callout can state them; reused by the §4.1/§4.2 rows. + _a_cl_cut = (roi_cutoffs.get("image_qc") or {}).get("cell_level") or {} + _a_nuc_warn = _a_cl_cut.get("pct_low_nuclear_texture_warn", 5.0) + _a_int_warn = _a_cl_cut.get("pct_cells_below_intensity_warn", 10.0) + _a_art_warn = _a_cl_cut.get("pct_cells_in_optically_dense_regions_warn", 30.0) + _a_gmm_warn = _a_cl_cut.get("pct_blurred_gmm_2d_roi_warn", 20.0) + + # Pre-formatted threshold strings for the 📖 callout (use Unicode ≤/≥ to + # avoid `<` being mistaken for an HTML tag start by Pandoc / browsers). + _a_pct_cutoff_str = ( + f"PASS below {int(_a_cw * 100)}% / WARN ≥ {int(_a_cw * 100)}% / FAIL ≥ {int(_a_cf * 100)}%" + ) + _a_artif_cutoff_str = ( + f"PASS below {int(float(_a_artif_warn) * 100)}% / WARN ≥ {int(float(_a_artif_warn) * 100)}% / FAIL ≥ {int(float(_a_artif_fail) * 100)}%" + if isinstance(_a_artif_warn, (int, float)) + and isinstance(_a_artif_fail, (int, float)) + and float(_a_artif_warn) <= float(_a_artif_fail) + else "thresholds not configured" + ) + + display(Markdown( + f"""::: {{.callout-note collapse="true" title="Metric explanation"}} +**Percentage of cells with low nuclear texture** — flags cells where the DAPI signal is too smooth/dim across the nucleus for reliable segmentation. Computed per cell as **CCFS_DAPI** (variance ÷ mean of DAPI pixel intensities inside the nucleus mask; low = smooth/uniform). *Not a blurriness metric* — captures nuclear texture quality, which depends on nuclear morphology, staining, and cell type. Routinely flags different cells to the blurry cells (GMM) metric ([Section 4.2](#sec-4-2)). *Verdict:* **advisory — PASS / WARN only, no FAIL.** A cell is flagged when CCFS_DAPI ≤ {_a_ccfs_thr}; the row turns **WARN** at ≥ {_a_nuc_warn:.0f}% flagged cells, otherwise **PASS**. + +**Percentage of cells with low DAPI / Boundary / IntRNA intensity** — three rows, one per channel. A cell is flagged when the average channel signal (measured over the **nucleus mask** for DAPI and the **whole-cell mask** for Boundary and IntRNA) is too dim for confident downstream interpretation. +*Cutoffs:* DAPI < {_a_ic_dapi}, Boundary < {_a_ic_boundary}, IntRNA < {_a_ic_intrna}. + +**Percentage of cells in optically dense regions** — flags cells overlapping regions of unusually high pixel intensity not attributable to tissue staining (e.g. folds, debris, coverslip contamination — sometimes genuine densely-packed tissue). These regions degrade segmentation accuracy and intensity-based metrics for the cells inside them. + +**Verdict (all rows here):** advisory — **PASS or WARN only, there is no FAIL tier**. Each row reports the *percentage of cells flagged* by that metric; it turns **WARN** when that percentage reaches low nuclear texture ≥ {_a_nuc_warn:.0f}%, low DAPI/Boundary/IntRNA intensity ≥ {_a_int_warn:.0f}%, or cells in optically dense regions ≥ {_a_art_warn:.0f}%; otherwise **PASS**. (The per-cell intensity floors above decide whether an individual cell is "low intensity"; the percentages here decide the row verdict.) These are flags to investigate, not sample-level gates. **Why this can differ from §3.3 (tiles):** cell-level intensity uses the same per-XOA-version floor as the tile metric, but it flags at ≥ {_a_int_warn:.0f}% of cells (the tile metric needs ≥ 40% of tissue tiles) and is measured per nucleus / whole-cell mask rather than per tile — so a channel can WARN here while §3.3 tiles PASS. +:::""" + )) + + display(Markdown( + """::: {.callout-tip collapse="true" title="Interpretation help"} +- **High percentage of cells with low nuclear texture**: many cells with degraded nuclear contrast. If the sample is **neural**, this can be biology (large neurons → naturally lower nuclear contrast); cross-check the per-channel intensity correlation heatmap in [Section 4.4.a](#sec-4-4-a) to see whether DAPI is the strongest predictor of transcript yield for this tissue. For non-neural tissue, look at the [Section 4.5](#sec-4-5) cell-flagged maps to localise the affected region. +- **High percentage of cells with low DAPI/Boundary/IntRNA intensity**: a single-channel issue often co-occurs with the same channel's tile-level intensity issue ([Section 3.3](#sec-3-3)); cross-check there. For tissues where DAPI isn't the strongest predictor of transcript yield (brain): consult the correlation heatmap in [Section 4.4.a](#sec-4-4-a) — Boundary or IntRNA may correlate more strongly with transcripts, in which case a low-DAPI flag carries less weight. +- **High percentage of cells in optically dense regions**: usually tissue folds, debris, or coverslip contamination — cross-check the optically-dense-region map in [Section 2.3](#sec-2-3) Masks. +- **Tissue-biology context**: a high percentage of low-IntRNA cells in skin, or a high percentage of cells in optically dense regions in spleen or specific brain regions, often reflects tissue biology rather than acquisition failure. Cross-check [Section 4.4.a](#sec-4-4-a) before excluding cells flagged by a single channel. +:::""" + )) + + def _a_value_str(_pct, _n_below, _n_total_local): + if not isinstance(_pct, (int, float)): + return "—" + if isinstance(_n_below, (int, float)) and isinstance(_n_total_local, (int, float)) and _n_total_local > 0: + return f"{_pct:.2f}% ({int(_n_below):,}/{int(_n_total_local):,})" + return f"{_pct:.2f}%" + + _a_rows = [] + + # ---- [Section 4.1](#sec-4-1) Cell quality scores summary — per-cell metrics ---- + + # Tier 1 row 1: pct_low_nuclear_texture (per-cell CCFS_DAPI column ≤ threshold) + _a_pct_low = cell_metrics.get("pct_low_nuclear_texture") + _a_n_low = cell_metrics.get("cells_low_nuclear_texture") + _a_rows.append({ + "Metric": "Percentage of cells with low nuclear texture", + "Value": _a_value_str(_a_pct_low, _a_n_low, _a_n_total), + "Advisory": _advisory_pass_warn(_a_pct_low, _a_nuc_warn), + "Filtering column": "is_low_nuclear_texture", + }) + + # Tier 1 rows 2–4: per-channel intensity below intensity_critical + # The pct_cells_below_intensity_* JSON keys are emitted as % only — derive + # the absolute count locally via pct × total / 100 so the Value column + # shows "X.XX% (n/total)" matching the nuclear texture / optically-dense rows. + for _ch_label, _emit_key, _ic, _filter_col in ( + ("DAPI", "pct_cells_below_intensity_DAPI", _a_ic_dapi, "mean_intensity"), + ("Boundary", "pct_cells_below_intensity_Boundary", _a_ic_boundary, "mean_intensity_Boundary"), + ("IntRNA", "pct_cells_below_intensity_IntRNA", _a_ic_intrna, "mean_intensity_IntRNA"), + ): + _ipct = cell_metrics.get(_emit_key) + _i_n_below = ( + int(round(float(_ipct) * _a_n_total / 100.0)) + if isinstance(_ipct, (int, float)) and _a_n_total > 0 + else None + ) + _a_rows.append({ + "Metric": f"Percentage of cells with low {_ch_label} intensity", + "Value": _a_value_str(_ipct, _i_n_below, _a_n_total), + "Advisory": _advisory_pass_warn(_ipct, _a_int_warn), + "Filtering column": f"{_filter_col}", + }) + + # Tier 1 row 5: % cells in optically dense regions (derived from cells_with_artifacts) + _a_n_artif = cell_metrics.get("cells_with_artifacts") + _a_pct_artif = ( + 100.0 * int(_a_n_artif) / _a_n_total + if isinstance(_a_n_artif, (int, float)) and _a_n_total > 0 else None + ) + _a_rows.append({ + "Metric": "Percentage of cells in optically dense regions", + "Value": _a_value_str(_a_pct_artif, _a_n_artif, _a_n_total), + "Advisory": _advisory_pass_warn(_a_pct_artif, _a_art_warn), + "Filtering column": "has_artifacts", + }) + + # ---- Render [Section 4.1](#sec-4-1) table; transition to [Section 4.2](#sec-4-2) ---- + display_status_table( + pd.DataFrame(_a_rows), + status_cols=["Advisory"], + narrow_cols={"Filtering column": "16em"}, + ) + display(Markdown("### 4.2 Cell quality (inherited from tile scores summary) {#sec-4-2}")) + + display(Markdown( + """::: {.callout-note collapse="true" title="Metric explanation"} +**Blurry cells (GMM)** — inherits the tile-level blurriness classification (the 2D GMM in [Section 3.2](#sec-3-2)) for each cell via spatial overlap. Computed over **solid-tissue cells only** (tile coverage ≥ 0.2), to match the tile-level "tiles in focus" in [Section 3.2](#sec-3-2). Cells in the emptiest tiles are left out because they are force-labelled blurry for coverage reasons, not optical focus, and including them pushes this % far above the tile figure. The exact cutoff is reported as `pct_blurred_gmm_2d_roi_denominator` in `image_qc_metrics.json`. + +**Cells in low-coverage tiles** — flags cells whose assigned tile lies in low-coverage tissue (`roi_tissue_coverage` < 0.5). This uses a different cutoff from the row above, so the two rows overlap: a cell in a tile at 0.35 coverage is counted in both, and the two counts can add up to more than the total cell count. Read each on its own rather than as a split of the cells. +:::""" + )) + + display(Markdown( + """::: {.callout-tip collapse="true" title="Interpretation help"} +If there is a high percentage of blurry cells, check whether Cohen's *d* between the GMM components is a PASS in [Section 3.2](#sec-3-2). If not, the metric may not be detecting real blurriness. Also check whether there is a high percentage of cells in low-coverage tiles, which may indicate a holey tissue where cells receive a 'blurry' label for biological reasons. +:::""" + )) + + _a_rows = [] + + # ---- [Section 4.2](#sec-4-2) Tile-mapped per-cell metrics (% cells in flagged tiles) ---- + + # [Section 4.2](#sec-4-2) row 1: Blurry cells (GMM), over solid-tissue cells + # (denominator = cells_evaluated_for_blur) so it matches the tile-level metric. + _a_pct_gmm = cell_metrics.get("pct_blurred_gmm_2d_roi") + _a_n_gmm = cell_metrics.get("cells_blurred_gmm_2d_roi") + _a_blur_denom = cell_metrics.get("cells_evaluated_for_blur", _a_n_total) + if _a_pct_gmm is None and isinstance(_a_n_gmm, (int, float)) and _a_blur_denom: + _a_pct_gmm = round(100.0 * float(_a_n_gmm) / float(_a_blur_denom), 2) + _a_rows.append({ + "Metric": "Blurry cells (GMM)", + "Value": _a_value_str(_a_pct_gmm, _a_n_gmm, _a_blur_denom), + "Advisory": _advisory_pass_warn(_a_pct_gmm, _a_gmm_warn), + "Filtering column": "is_blurred_gmm_2d_roi", + }) + + # Tier 2 row 2: Cells in low-coverage tiles (informational — no Advisory) + _a_pct_lc = cell_metrics.get("pct_cells_in_low_coverage_tiles") + _a_n_lc = cell_metrics.get("cells_in_low_coverage_tiles") + if isinstance(_a_pct_lc, (int, float)) or isinstance(_a_n_lc, (int, float)): + _a_rows.append({ + "Metric": "Cells in low-coverage tiles", + "Value": _a_value_str(_a_pct_lc, _a_n_lc, _a_n_total), + "Advisory": "—", + "Filtering column": "roi_tissue_coverage", + }) + + # ---- Render [Section 4.2](#sec-4-2) table; transition to [Section 4.3](#sec-4-3) ---- + display_status_table( + pd.DataFrame(_a_rows), + status_cols=["Advisory"], + narrow_cols={"Filtering column": "16em"}, + ) + display(Markdown("### 4.3 Cell quality cluster bias summary {#sec-4-3}")) + + display(Markdown( + """::: {.callout-note collapse="true" title="Metric explanation"} +WARN when an outlier cluster is detected; PASS when none are. + +**Outlier cluster** — any cluster where blurry-cell rate or low-nuclear texture rate exceeds max(2× sample-wide rate, 10%) AND n_cells ≥ 50. + +**Worst outlier (blurriness / low nuclear texture)** — if one or more clusters with unusually high levels of blurriness or low nuclear texture have been identified, the worst one is shown. +:::""" + )) + + display(Markdown( + """::: {.callout-tip collapse="true" title="Interpretation help"} +A cluster composed largely of blurry or low-nuclear-texture cells is likely defined by that image-quality defect rather than by biology. Because these metrics are sensitive to technical factors, such outlier clusters warrant caution: a cluster with high blurriness reflects a regional optical problem and is unlikely to represent a genuine cell population. A high proportion of low-nuclear-texture cells in a cluster may also indicate a technical problem, though it can also reflect genuine cell-type properties. If a technical cause is likely, the cluster can be filtered out before downstream analysis. +:::""" + )) + + _a_rows = [] + + # ---- [Section 4.3](#sec-4-3) Per-cluster outliers ---- + + # Tier 3 row 1: cluster_outlier_focus (worst tile-blurriness cluster outlier) + # When cluster_blur falls back to is_low_nuclear_texture, the two [Section 4.3](#sec-4-3) + # rows compute identical outlier sets — surface that to the reader so + # they don't read row 1 + row 2 as independent signals. + _a_blur_method = cell_metrics.get("cluster_blur_method", "is_blurred_gmm_2d_roi") + _a_blur_outliers = cell_metrics.get("cluster_blur_outliers") + _a_blur_fallback_suffix = ( + " — fallback method (no blurry-(GMM) signal; values equivalent to the low-nuclear texture row below)" + if _a_blur_method == "is_low_nuclear_texture" else "" + ) + if isinstance(_a_blur_outliers, dict) and _a_blur_outliers: + _a_worst_blur = max(_a_blur_outliers, key=lambda k: _a_blur_outliers[k]["pct_blurred"]) + _a_worst_blur_info = _a_blur_outliers[_a_worst_blur] + _a_rows.append({ + "Metric": "Worst tile-blurriness cluster outlier", + "Value": ( + f"C{_a_worst_blur} " + f"({float(_a_worst_blur_info['pct_blurred']):.0f}% blurry cells (GMM), " + f"n={int(_a_worst_blur_info['n_cells']):,})" + + _a_blur_fallback_suffix + ), + "Advisory": "WARN", + "Filtering column": "Cluster_kmeans10", + }) + else: + _a_rows.append({ + "Metric": "Worst tile-blurriness cluster outlier", + "Value": "no outliers" + _a_blur_fallback_suffix, + "Advisory": "PASS", + "Filtering column": "—", + }) + + # Tier 3 row 2: cluster_outlier_CCFS (worst low-nuclear texture cluster outlier) + # JSON key kept as `cluster_ccfs_outliers` for backward compatibility; the + # user-facing row name now uses "low-nuclear texture" terminology consistent + # with the rest of §4. + _a_ccfs_outliers = cell_metrics.get("cluster_ccfs_outliers") + if isinstance(_a_ccfs_outliers, dict) and _a_ccfs_outliers: + _a_worst_ccfs = max(_a_ccfs_outliers, key=lambda k: _a_ccfs_outliers[k]["pct_low_texture"]) + _a_worst_ccfs_info = _a_ccfs_outliers[_a_worst_ccfs] + _a_rows.append({ + "Metric": "Worst low-nuclear texture cluster outlier", + "Value": ( + f"C{_a_worst_ccfs} " + f"({float(_a_worst_ccfs_info['pct_low_texture']):.0f}% low-texture, " + f"n={int(_a_worst_ccfs_info['n_cells']):,})" + ), + "Advisory": "WARN", + "Filtering column": "Cluster_kmeans10", + }) + elif "cluster_ccfs" in cell_metrics: + _a_rows.append({ + "Metric": "Worst low-nuclear texture cluster outlier", + "Value": "no outliers", + "Advisory": "PASS", + "Filtering column": "—", + }) + + # Render [Section 4.3](#sec-4-3) table. Advisory column = WARN when an outlier cluster + # fires the upstream rule (>2× sample-wide rate AND ≥50 cells), PASS + # otherwise. Verdict is presence/absence of outliers, not a value-vs- + # threshold comparison — so no YAML cutoff is involved here. + display_status_table( + pd.DataFrame(_a_rows), + status_cols=["Advisory"], + narrow_cols={"Filtering column": "16em"}, + ) + + # ============================================================================= + # [Section 4.4](#sec-4-4)–4.5 cell-level supporting figures — four sub-subsections + conditional drill-down (Phase 14, extended r6) + # ============================================================================= + # Replaces v3's 9.A figure displays (CCFS spatial, ccfs_thresholded, + # nuclear_texture_vs_transcripts_log) + v3's 9.B figures (cell_focus_distribution, + # gmm_focus_vs_transcripts, spatial_comparison) + v3's 9.D figures + # (nuclear_texture_proportions, nuclear_texture_density, tile_blur_proportions_roi). + # v3 9.C / 9.D / 9.E and the UMAP overlay block all dissolved — UMAP figures + # remain unsurfaced latent artefacts produced by bin/image_qc.py. + # v4 r5: spatial_comparison_nuclei_vs_roi (2-panel CCFS-vs-tile-focus) is + # now also latent — split-replaced by single-panel `tile_focus_gmm_spatial` + # (right side only; CCFS side was dropped per user feedback as concordance + # with tile-focus is known to be low). cell_focus_distribution trimmed + # from 2×2 to 1×2 — bottom-left dropped (focus density per cluster, redundant + # with nuclear_texture_density), bottom-right extracted as standalone + # `blur_prob_density_by_cluster` and moved to Per-cluster subsection. + display(Markdown("### 4.4 Technical effects on transcript count {#sec-4-4}")) + display(Markdown( + "Per-cell scatters of each quality metric against transcript count. " + )) + + display(Markdown( + """::: {.callout-tip collapse="true" title="Interpretation help"} +**Nuclear texture vs transcript count** + +The correlation is tissue-dependent. Mildly positive in most tissues, but can flip negative in tissues with large nuclei and diffuse chromatin (e.g. brain). A weak or negative correlation here is not automatically a quality problem — read against the tissue context. + +**Stain intensity vs transcript count** + +- **IntRNA** is the most consistent predictor of transcript yield — expect a strong positive correlation. +- **DAPI** is tissue-variable. Positive in most tissues, but can flip negative in brain — large pyramidal neurons have diffuse, low-density chromatin (low DAPI signal) but high transcript counts. +- **Boundary** is a moderate predictor. Can exceed DAPI in samples where membrane staining tracks cell density better than nuclear contrast. + +**Focus score vs transcript count** + +- The diagnostic signal is the **x-axis spread** within each focus band — imaging defects push blurry cells (red) to lower transcript counts at the same focus level. +- Biology-driven low-count regions (white matter, adipose) stay co-located with well-focused tissue. +- Don't read the vertical red/blue split as a finding — the GMM uses focus scores as input. +:::""" + )) + + # ----- [Section 4.4.a](#sec-4-4-a) Per-cell quality metric correlations (heatmap, first) ----- + # Headline view: 6×6 Spearman ρ matrix across the per-cell quality + # axes (nuclear texture, focus score, DAPI/Boundary/IntRNA intensities) + # vs transcript counts. Comes first so a reader sees the overall + # relationships before drilling into individual scatters below. + display(Markdown("#### 4.4.a Per-cell quality metric correlations {#sec-4-4-a}")) + display(Markdown( + "Spearman correlation matrix across nuclear texture (`CCFS_DAPI`), focus score, " + "DAPI / Boundary / IntRNA mean intensity, and transcript counts. " + )) + display(fig_html("figures/intensity_transcript_correlation.png", max_width="55%")) + + # ----- [Section 4.4.b](#sec-4-4-b) Nuclear texture vs transcript count ----- + display(Markdown("#### 4.4.b Nuclear texture vs transcript count {#sec-4-4-b}")) + display(Markdown( + "Points are coloured by local 2D cell density (viridis: bright yellow = " + "many overlapping cells, dark purple = sparse outliers)." + )) + display(fig_html("figures/nuclear_texture_vs_transcripts_log.png")) + + # ----- [Section 4.4.c](#sec-4-4-c) Stain intensity vs transcript count ----- + # Three per-channel scatters showing DAPI / Boundary / IntRNA intensity + # vs transcript count. The high-level Spearman view across all channels + # is in [Section 4.4.a](#sec-4-4-a) above. + display(Markdown("#### 4.4.c Stain intensity vs transcript count {#sec-4-4-c}")) + display(Markdown("**DAPI mean intensity vs transcript count**")) + display(fig_html("figures/dapi_intensity_vs_transcripts_log.png")) + + display(Markdown("**Boundary mean intensity vs transcript count**")) + display(fig_html("figures/boundary_intensity_vs_transcripts_log.png")) + + display(Markdown("**IntRNA mean intensity vs transcript count**")) + display(fig_html("figures/intrna_intensity_vs_transcripts_log.png")) + + # ----- [Section 4.4.d](#sec-4-4-d) Focus score vs transcript count ----- + display(Markdown("#### 4.4.d Focus score vs transcript count {#sec-4-4-d}")) + display(Markdown( + "Focus score is computed per tile; each cell inherits the focus score of " + "its tile. Points are coloured by GMM classification (blue = in focus, " + "red = blurry)." + )) + display(fig_html("figures/gmm_focus_vs_transcripts.png")) + + # ============================================================================= + # [Section 4.5](#sec-4-5) Spatial distribution of low-quality flagged cells + # ============================================================================= + display(Markdown("### 4.5 Spatial distribution of low-quality flagged cells {#sec-4-5}")) + display(Markdown( + "Whole-sample maps marking cells flagged by low nuclear texture or tile " + "blurriness. Use these to see whether quality loss is spatially clustered " + "(physical optical issue) or scattered (baseline noise)." + )) + + display(fig_html("figures/cell_flagged_maps.png")) + + # Cell-level focus score distribution figure removed 2026-05-15 (user + # decision). The tile-level focus distribution in [Section 3.2](#sec-3-2) is the canonical + # view; the cell-level version was the tile distribution re-weighted by + # per-tile cell density, which conflicted with [Section 3.2](#sec-3-2) on samples with + # heterogeneous cell density and added no signal for the advisory + # filtering use-case of §4. The cell-level GMM classification (verdict + # in [Section 4.2](#sec-4-2)'s "Blurry cells (GMM)" row) carries the actionable signal. + + # ----- Cluster bias supporting figures (companion to [Section 4.3](#sec-4-3) table) ----- + display(Markdown("### 4.6 Cluster bias supporting figures {#sec-4-6}")) + + display(Markdown( + """::: {.callout-tip collapse="true" title="Interpretation help"} +- **Nuclear texture proportions by cluster**: clusters should have similar texture proportions. Cluster-specific bias may reflect nuclear morphology differences between cell types — nuclear texture captures nuclear contrast, not optical blurriness, so cluster separation may be biology rather than quality loss. Cross-check with cluster spatial distribution before excluding cells. +- **Tile-blurriness proportions by cluster (GMM)**: clusters should have similar blurriness proportions. Cluster-specific enrichment may indicate spatially biased quality loss, likely confounded by imaging artefacts rather than real biology. +- **Per-cluster blurriness breakdown table**: Cross-check the cell-flagged maps ([Section 4.5](#sec-4-5)) for whether flagged clusters overlap blurry regions before excluding cells. + - High blurriness but low low-texture in a cluster: possible imaging issue localised to that cluster's spatial position; nuclear contrast is unaffected. + - High low-texture but low blurriness: nuclear texture loss without an imaging defect; likely reflects cell-type-specific nuclear morphology rather than quality loss. + - Both high: likely indicates a real spatial quality problem. Consider excluding the affected cluster. +:::""" + )) + + if "clusters_present" in cell_metrics: + display(Markdown(f"**Clusters present:** {cell_metrics['clusters_present']}")) + + display(Markdown("**Nuclear texture proportions by cluster**")) + display(fig_html("figures/nuclear_texture_proportions.png")) + + display(Markdown("**Tile-blurriness proportions by cluster (GMM)**")) + display(fig_html("figures/tile_blur_proportions_roi.png")) + + # ----- Per-cluster blurriness breakdown ----- + # Two independent signals are shown for this section: + # (1) a sample-wide "overall blurriness" advisory line, fired when the + # whole-sample blurry-cell rate is high (reuses the §4.2 advisory gate + # pct_blurred_gmm_2d_roi_warn, default 20%). It fires even when no + # single cluster stands out, which is the broadly-blurry case. + # (2) a per-cluster WARN for clusters flagged as outliers by image_qc.py + # (cluster_blur_outliers / cluster_ccfs_outliers). The QMD only tests + # membership; the detection rule (robust MAD z-score + 15% floor) lives + # in bin/image_qc.py so the producer and this report never drift. + _cluster_blur = cell_metrics.get("cluster_blur") + _cluster_ccfs = cell_metrics.get("cluster_ccfs") + _blur_outliers = cell_metrics.get("cluster_blur_outliers") or {} + _ccfs_outliers = cell_metrics.get("cluster_ccfs_outliers") or {} + display(Markdown("**Per-cluster blurriness breakdown**")) + + # (1) Sample-wide advisory line (PASS/WARN, no FAIL) above the detail table. + _cl_cut_46 = (roi_cutoffs.get("image_qc") or {}).get("cell_level") or {} + _gmm_warn_46 = float(_cl_cut_46.get("pct_blurred_gmm_2d_roi_warn", 20.0)) + _sample_blur = cell_metrics.get("pct_blurred_gmm_2d_roi") + if isinstance(_sample_blur, (int, float)): + _overall_caveat = ( + f"Overall blurriness is high ({_sample_blur:.1f}% of cells flagged " + f"blurry, at or above the {_gmm_warn_46:.0f}% advisory level). A high " + "sample-wide rate means image quality is a concern across the whole " + "section, so the per-cluster table below may show no single standout " + "while the sample is still affected. Review the focus heatmap " + "([Section 3.2](#sec-3-2)) and the sample-wide blurry-cells row " + "([Section 4.2](#sec-4-2))." + ) + _overall_status = "WARN" if _sample_blur >= _gmm_warn_46 else "PASS" + _overall_label = metric_cell_with_caveat( + f"{_sample_blur:.1f}%", _overall_status, _overall_caveat, + ) if _overall_status == "WARN" else f"{_sample_blur:.1f}%" + _overall_df = pd.DataFrame([{ + "Overall blurriness (sample-wide)": _overall_label, + "Status": _overall_status, + }]) + display_status_table(_overall_df, status_cols=["Status"]) + _cl_rows = [] + # Iterate the union of cluster IDs from both dicts so a CCFS-only + # trigger (where _cluster_blur may be empty) still renders rows. Each + # column falls back to "—" when its source dict lacks the cluster key. + # Sort by number of cells descending (largest clusters first); clusters + # with no cell-count data fall to the end. Secondary key on + # int(cluster_id) keeps tied rows deterministic. + _blur_keys = set(_cluster_blur.keys()) if isinstance(_cluster_blur, dict) else set() + _ccfs_keys = set(_cluster_ccfs.keys()) if isinstance(_cluster_ccfs, dict) else set() + def _cluster_sort_key(cid): + _b = _cluster_blur.get(cid) if isinstance(_cluster_blur, dict) else None + _c = _cluster_ccfs.get(cid) if isinstance(_cluster_ccfs, dict) else None + _n = (_b.get("n_cells") if isinstance(_b, dict) else None) \ + or (_c.get("n_cells") if isinstance(_c, dict) else None) + _has = isinstance(_n, (int, float)) + # Tuple: (no-data clusters sort after; then n_cells desc; then cluster id asc) + return (0 if _has else 1, -int(_n) if _has else 0, int(cid)) + _all_cluster_ids = sorted(_blur_keys | _ccfs_keys, key=_cluster_sort_key) + for cl_id in _all_cluster_ids: + _blur_info = _cluster_blur.get(cl_id) if isinstance(_cluster_blur, dict) else None + _ccfs_info = _cluster_ccfs.get(cl_id) if isinstance(_cluster_ccfs, dict) else None + _has_blur_data = isinstance(_blur_info, dict) + _has_ccfs_data = isinstance(_ccfs_info, dict) + _is_blur_out = cl_id in _blur_outliers + _is_ccfs_out = cl_id in _ccfs_outliers + is_outlier = _is_blur_out or _is_ccfs_out + _cluster_label = f"C{cl_id}" + if is_outlier: + if _is_blur_out and _is_ccfs_out: + _trigger = "high blurriness and low nuclear texture" + elif _is_blur_out: + _trigger = "high blurriness" + else: + _trigger = "low nuclear texture" + _caveat = ( + f"**Cluster outlier: {_trigger}.** This cluster's rate is far " + "above the other clusters in this sample, which points to " + "spatially concentrated quality loss that can affect downstream " + "analysis of this cell population. Check the focus heatmap " + "([Section 3.2](#sec-3-2)) to see whether the cluster sits over an " + "out-of-focus region, then decide whether to exclude or flag its " + "cells." + ) + _cluster_label = metric_cell_with_caveat( + _cluster_label, "WARN", _caveat, + ) + _n_cells = ( + _blur_info.get("n_cells") if _has_blur_data + else (_ccfs_info.get("n_cells") if _has_ccfs_data else None) + ) + _cl_rows.append({ + "Cluster": _cluster_label, + "Cells": f"{int(_n_cells):,}" if isinstance(_n_cells, (int, float)) else "—", + "Blurry Cells": f'{_blur_info["n_blurred"]:,}' if _has_blur_data and "n_blurred" in _blur_info else "—", + "Percentage of blurry cells": f'{_blur_info["pct_blurred"]:.1f}%' if _has_blur_data and "pct_blurred" in _blur_info else "—", + "Percentage of low-texture cells": f'{float(_ccfs_info["pct_low_texture"]):.1f}%' if _has_ccfs_data and "pct_low_texture" in _ccfs_info else "—", + "Median nuclear texture": f'{_blur_info["median_ccfs_dapi"]:.3f}' if _has_blur_data and "median_ccfs_dapi" in _blur_info else "—", + "Status": "WARN" if is_outlier else "—", + }) + if _cl_rows: + _df_cl = pd.DataFrame(_cl_rows) + display_status_table(_df_cl, status_cols=["Status"]) + else: + # No cluster-level data emitted in this run's JSON (cluster_blur and + # cluster_ccfs both missing). Render a clarifying note. + display(Markdown( + "*Per-cluster detail (`cluster_blur` / `cluster_ccfs`) is not " + "available in this run's JSON — cluster-level breakdown skipped.*" + )) + +``` + +## 5. Metadata {#sec-5} + + +```{python report-metadata} +#| echo: false + +from datetime import datetime + +# Set when this HTML is rendered (local time). +_report_stamp = datetime.now().astimezone().strftime("%Y-%m-%d %H:%M %Z") + +_resolved_out = base.resolve() +_sample = str(SAMPLE_NAME).strip() if SAMPLE_NAME else "" +if not _sample: + if _resolved_out.name == "image_qc": + _sample = _resolved_out.parent.name + else: + _sample = _resolved_out.name + +_bundle = str(XENIUM_BUNDLE).strip() if XENIUM_BUNDLE else "" +_bundle_display = ( + f"{_bundle}" + if _bundle + else "— (pass -P XENIUM_BUNDLE:… for local render; Nextflow passes samplesheet path)" +) + +_pub = str(SAMPLE_PUBLISHED_OUTDIR).strip() if SAMPLE_PUBLISHED_OUTDIR else "" +_pub_display = ( + f"{_pub}" + if _pub + else "— (pass -P SAMPLE_PUBLISHED_OUTDIR:… or run via Nextflow)" +) + +# Collapse versions.yml into a single one-line summary (joined key=value pairs). +if versions_text: + _v_pairs = [] + for _ln in versions_text.splitlines(): + _s = _ln.strip() + if not _s: + continue + # Skip YAML section-header lines (e.g. "CELLPOSE:") that have no inline value. + if _s.endswith(":") and ":" not in _s[:-1]: + continue + if ":" in _s: + _k, _v = _s.split(":", 1) + _v = _v.strip() + if _v: + _v_pairs.append(f"{_k.strip()}={_v}") + _versions_one_line = "; ".join(_v_pairs) if _v_pairs else versions_text.strip().replace("\n", " ") +else: + _versions_one_line = "— (versions.yml not found)" + +_seg_sw = roi_metrics.get("segmentation_software") or "—" +# XOA (onboard analysis) version, emitted by image_qc.py from the bundle's +# experiment.xenium. Strip the "xenium-" prefix for display. Blank on bundles +# processed before this field was added — reprocess to populate. +_xoa_raw = roi_metrics.get("xoa_version") +_xoa_version = ( + str(_xoa_raw).split("-", 1)[-1] + if _xoa_raw + else "— (not recorded; reprocess to populate)" +) + +_meta = pd.DataFrame( + [ + {"Field": "Report generated", "Value": _report_stamp}, + {"Field": "Xenium bundle", "Value": _bundle_display}, + {"Field": "XOA version", "Value": _xoa_version}, + {"Field": "Segmentation software", "Value": _seg_sw}, + {"Field": "Image QC output folder path", "Value": _pub_display}, + {"Field": "Software versions", "Value": _versions_one_line}, + ] +) +_tbl = _meta.to_html(index=False, header=False, classes="table report-meta", border=0, escape=False) +display( + HTML( + "" + + _tbl + ) +) +``` + + +```{python authors-footer} +#| echo: false +_pipeline_version = str(PIPELINE_VERSION).strip() if PIPELINE_VERSION else "" +_version_clause = f" v{_pipeline_version}" if _pipeline_version else "" +display(HTML( + '
' + 'Generated by ' + f'nf-xenium-processing{_version_clause} — pipeline maintained by Altos ' + 'Labs Spatial Bioinformatics. Authors: Malwina Prater, Hanneke Okkenaug ' + 'Nell Yu Nie & Christel Krueger.' + '
' +)) +``` +```{python versions} +#| echo: false + +# The nf-core quarto/notebook module requires the notebook to export the +# versions of the packages it uses to versions.csv (package,version — no +# header), which the module folds into the pipeline's versions topic. Read at +# runtime rather than hardcoded, per the pipeline's version-reporting rules. +from importlib.metadata import PackageNotFoundError, version as _pkg_version + +with open("versions.csv", "w", encoding="utf-8") as _fh: + for _pkg in ("numpy", "pandas", "ipython"): + try: + _fh.write(f"{_pkg},{_pkg_version(_pkg)}\n") + except PackageNotFoundError: + continue +``` diff --git a/bin/xenium_patch_stitch_postprocess.py b/bin/xenium_patch_stitch_postprocess.py index 7144b1ac..d600b83d 100755 --- a/bin/xenium_patch_stitch_postprocess.py +++ b/bin/xenium_patch_stitch_postprocess.py @@ -68,6 +68,11 @@ def reassign_dropped(csv_path: str, dropped_cells: set) -> None: fieldnames = reader.fieldnames rows = list(reader) + if fieldnames is None: + raise ValueError( + f"No header row in {csv_path}; cannot rewrite transcript assignments" + ) + reassigned = 0 for row in rows: if row["cell"] in dropped_cells: @@ -87,8 +92,12 @@ def parse_args() -> argparse.Namespace: parser = argparse.ArgumentParser( description="Clean stitched GeoJSON polygons and reconcile transcript CSV." ) - parser.add_argument("--geojson", required=True, help="Path to xr-cell-polygons.geojson") - parser.add_argument("--csv", required=True, help="Path to xr-transcript-metadata.csv") + parser.add_argument( + "--geojson", required=True, help="Path to xr-cell-polygons.geojson" + ) + parser.add_argument( + "--csv", required=True, help="Path to xr-transcript-metadata.csv" + ) return parser.parse_args() diff --git a/conf/base.config b/conf/base.config index 8cd5a228..d1a5e827 100644 --- a/conf/base.config +++ b/conf/base.config @@ -176,6 +176,17 @@ process { time = { 16.h * task.attempt } } + // Optionally-GPU image QC. GPU concerns only — cpus/memory/time come from + // the combinatorial labels the process also carries. `accelerator` tells the + // executor how many devices to request; it does not restrict what CUDA can + // see, so the process must additionally cap the devices it uses (image QC + // passes `--max-gpus`), otherwise a task requesting one GPU that lands on a + // multi-GPU instance detects and uses all of them. + withLabel:process_gpu_qc { + ext.use_gpu = { params.use_gpu } + accelerator = { params.use_gpu ? (params.image_qc_gpus as int) : null } + } + // ========================================================================= // Extra labels // ========================================================================= diff --git a/conf/modules.config b/conf/modules.config index 51860d55..ef41946f 100644 --- a/conf/modules.config +++ b/conf/modules.config @@ -387,4 +387,81 @@ process { mode: params.publish_dir_mode, ] } + + // ---------------------------- image / transcript QC ------------------------- + + // The QC processes read no `params.*` themselves; every knob arrives as + // `ext.*` from here. Four keys below MUST stay explicit: the module tests + // them negated (`if (!task.ext.x)`) or interpolates them unconditionally, so + // leaving one unset silently flips behaviour — unset `figures` disables all + // figures, unset `stream_tiles` writes full-resolution planes (~154 GB of + // scratch on a 5.5 gigapixel sample), unset `snr_no_moran` turns Moran's I + // on, and unset `max_gpus` renders `--max-gpus null`. + withName: '.*IMAGE_QC:ANALYSIS' { + ext.prefix = 'image_qc' + ext.tile_size = { params.tile_size } + ext.legacy_focus = { params.legacy_focus } + ext.no_snr = { params.image_qc_no_snr } + ext.snr_no_roi_tx_table = { params.image_qc_snr_no_roi_tx_table } + ext.snr_otsu_max_rois = { params.image_qc_snr_otsu_max_rois } + ext.snr_no_moran = { params.image_qc_snr_no_moran } + ext.save_dapi_maps_tiff = { params.image_qc_save_dapi_maps_tiff } + ext.stream_tiles = { params.image_qc_stream_tiles } + ext.figure_source_tables = { params.image_qc_figure_source_tables } + ext.figures = { params.image_qc_figures } + ext.max_gpus = { params.image_qc_gpus } + ext.lap_sigma = { params.image_qc_lap_sigma } + // The analysis output directory is itself named by ext.prefix, so + // publish into qc/ rather than qc/image_qc to avoid double-nesting. + publishDir = [ + path: { "${params.outdir}/${params.mode}/qc" }, + mode: params.publish_dir_mode, + ] + } + + withName: '.*TRANSCRIPT_QC:ANALYSIS' { + ext.prefix = 'transcript_qc' + ext.non_gene_prefix = { params.neg_control_prefix } + ext.num_row_groups = null + ext.pipeline_segmentation = { params.method ?: 'skip' } + ext.is_resegmented = false + publishDir = [ + path: { "${params.outdir}/${params.mode}/qc" }, + mode: params.publish_dir_mode, + ] + } + + // Report renderers. Quarto's HTML pipeline runs a second pandoc pass + // (--embed-resources) that re-reads the whole rendered document, so its + // memory scales with REPORT SIZE, not input size: a ~60 MB production image + // QC report OOM-killed that pass under QUARTONOTEBOOK's own process_low + // (12 GB) while the complete HTML was already on disk, and quarto exited 1 + // with an empty stderr. process_medium's 42 GB covers the measured need. + withName: '.*IMAGE_QC:REPORT' { + // The stock QUARTONOTEBOOK container has no pandas; render in the image + // QC container, which carries the report's Python dependencies. + container = 'quay.io/dongzehe/image_qc:1.0.0' + ext.prefix = 'image_qc' + cpus = { 6 * task.attempt } + memory = { 42.GB * task.attempt } + time = { 8.h * task.attempt } + publishDir = [ + path: { "${params.outdir}/${params.mode}/qc/image_qc" }, + mode: params.publish_dir_mode, + pattern: '*.html', + ] + } + + withName: '.*TRANSCRIPT_QC:REPORT' { + container = 'quay.io/dongzehe/transcript_qc:1.1.0' + ext.prefix = 'transcript_qc' + cpus = { 6 * task.attempt } + memory = { 42.GB * task.attempt } + time = { 8.h * task.attempt } + publishDir = [ + path: { "${params.outdir}/${params.mode}/qc/transcript_qc" }, + mode: params.publish_dir_mode, + pattern: '*.html', + ] + } } diff --git a/docs/output.md b/docs/output.md index 20442cb2..cf05b207 100644 --- a/docs/output.md +++ b/docs/output.md @@ -124,6 +124,19 @@ The pipeline is built using [Nextflow](https://www.nextflow.io/) and processes d
Output files +- `/qc/image_qc/` + - `image_qc.html` rendered image QC report (focus, signal-to-noise and morphology assessment) + - `image_qc_metrics.json`, `image_qc_metrics.csv` per-sample image QC metrics + - `image_qc_cell_metrics.json`, `image_qc_cell_metrics.csv` per-cell metrics, when the bundle contains cell data + - `roi_qc_metrics.json`, `grid_roi_focus_scores.csv` per-ROI focus and intensity metrics + - `snr_metrics.json`, `SNR_roi_tx.parquet` signal-to-noise metrics and the per-ROI transcript table + - `intensity_assessment.json`, `dense_intensity_regions_summary.csv` intensity grading + - `image_qc_status.json` analysis status marker; the report renders a QC-FAILED banner when the analysis could not complete + - `figures/` figure PDFs and PNGs embedded in the report +- `/qc/transcript_qc/` + - `transcript_qc.html` rendered transcript QC report (per-transcript, per-cell and per-field-of-view assessment) + - `transcript_qc_metrics.json` per-sample transcript QC metrics, including the noise threshold, retained genes, and per-FoV quality summary + - `figures/` figure PDFs and PNGs embedded in the report - `opt/` - `flip/` - `*.fa` the forward oriented fasta file diff --git a/main.nf b/main.nf index 15827672..5e80e86b 100644 --- a/main.nf +++ b/main.nf @@ -80,6 +80,8 @@ workflow NFCORE_SPATIALAXE { params.tiling, params.xeniumranger_only, params.spoqc, + params.roi_image_qc_thresholds_yaml, + params.transcript_qc_thresholds_yaml, ) emit: multiqc_report = SPATIALAXE.out.multiqc_report // channel: /path/to/multiqc_report.html diff --git a/modules.json b/modules.json index 6bcc4bde..aa0d7855 100644 --- a/modules.json +++ b/modules.json @@ -35,6 +35,11 @@ "installed_by": ["modules"], "patch": "modules/nf-core/opt/track/opt-track.diff" }, + "quarto/notebook": { + "branch": "master", + "git_sha": "b0c002b4eeaae7b0fa0181b098bb318269b11cf5", + "installed_by": ["modules"] + }, "stardist": { "branch": "master", "git_sha": "4e783502ab661bed13f15189401b73c93966831f", diff --git a/modules/local/image_qc/Dockerfile b/modules/local/image_qc/Dockerfile new file mode 100644 index 00000000..6a54ef30 --- /dev/null +++ b/modules/local/image_qc/Dockerfile @@ -0,0 +1,24 @@ +# Container for the image QC module (analysis + Quarto report). +# Built from environment.yml in this directory. Tagged and pushed as +# quay.io/dongzehe/image_qc:1.0.0 (to be migrated to the nf-core org for release). +# +# docker build -t quay.io//image_qc:1.0.0 modules/local/image_qc +# +# Note: behind a corporate proxy / private conda mirror you may need to provide a +# CA bundle and channel config at build time (e.g. via BuildKit --mount=type=secret); +# a standard build resolves the pinned packages directly from conda-forge/bioconda. +FROM mambaorg/micromamba:2.0.5 +COPY --chown=$MAMBA_USER:$MAMBA_USER environment.yml /tmp/environment.yml +RUN micromamba install -y -n base -f /tmp/environment.yml \ + && micromamba clean -a -f -y \ + && rm -f /tmp/environment.yml +ENV PATH="/opt/conda/bin:$PATH" \ + QUARTO_DENO=/opt/conda/bin/deno \ + QUARTO_DENO_DOM=/opt/conda/lib/deno_dom.so \ + QUARTO_PANDOC=/opt/conda/bin/pandoc \ + QUARTO_ESBUILD=/opt/conda/bin/esbuild \ + QUARTO_TYPST=/opt/conda/bin/typst \ + QUARTO_DART_SASS=/opt/conda/bin/sass \ + QUARTO_SHARE_PATH=/opt/conda/share/quarto \ + QUARTO_CONDA_PREFIX=/opt/conda \ + ES2_LIBRARY=/opt/conda/lib/libGLESv2.so.2 diff --git a/modules/local/image_qc/environment.yml b/modules/local/image_qc/environment.yml new file mode 100644 index 00000000..00ab029b --- /dev/null +++ b/modules/local/image_qc/environment.yml @@ -0,0 +1,51 @@ +--- +# yaml-language-server: $schema=https://raw.githubusercontent.com/nf-core/modules/master/modules/environment-schema.json +channels: + - conda-forge + - bioconda +dependencies: + # procps-ng provides `ps`, required by Nextflow for task-metric collection + - conda-forge::procps-ng=4.0.4 + - conda-forge::python=3.11.0 + - conda-forge::numpy=2.3.3 + - conda-forge::pandas=2.3.2 + - conda-forge::scipy=1.16.2 + - conda-forge::matplotlib=3.10.6 + - conda-forge::seaborn=0.13.2 + - conda-forge::scikit-image=0.25.2 + - conda-forge::scikit-learn=1.7.2 + - conda-forge::numba=0.62.0 + - conda-forge::tifffile=2025.9.20 + - conda-forge::zarr=2.18.7 + - conda-forge::imagecodecs=2025.8.2 + - conda-forge::pillow=12.0.0 + - conda-forge::pyyaml=6.0.3 + - conda-forge::h5py=3.14.0 + - conda-forge::pyarrow=21.0.0 + - conda-forge::click=8.3.0 + - conda-forge::tqdm=4.67.1 + - conda-forge::tbb=2023.0.0 + # Report rendering (Quarto) — this container also renders the image QC HTML report + - conda-forge::quarto=1.6.42 + - conda-forge::jupyter=1.1.1 + - conda-forge::ipython=9.5.0 + - conda-forge::ipykernel=6.30.1 + - conda-forge::papermill=2.6.0 + - conda-forge::napari-simpleitk-image-processing=0.4.9 + - conda-forge::napari-skimage-regionprops=0.10.1 + # GL ES runtime for the napari->vispy import chain pulled in by the two napari + # plugins above. Never used for actual rendering (headless run); vispy merely + # dlopens libGLESv2 at import. The container sets ES2_LIBRARY to this library + # so vispy finds it without ldconfig visibility of /opt/conda/lib. + - conda-forge::libgles=1.7.0 + # Spatial autocorrelation (Moran's I) for the ROI / SNR QC metrics + - conda-forge::libpysal=4.14.1 + - conda-forge::esda=2.9.0 + - conda-forge::pysal=26.1 + - conda-forge::pip=26.2.1 + # GPU acceleration (optional at runtime). cupy imports are guarded by try/except + # in image_qc.py, so this environment also runs correctly on CPU-only hosts. + - pip: + - cupy-cuda12x==14.0.1 + - nvidia-cuda-nvrtc-cu12==12.9.86 + - nvidia-cuda-runtime-cu12==12.9.79 diff --git a/modules/local/image_qc/main.nf b/modules/local/image_qc/main.nf new file mode 100644 index 00000000..1e27258c --- /dev/null +++ b/modules/local/image_qc/main.nf @@ -0,0 +1,164 @@ +process IMAGE_QC_ANALYSIS { + tag "${meta.id}" + // process_gpu_qc carries only the GPU request; cpus/memory/time come from + // process_xl (30 cpus / 240 GB / 24 h), which matches what a 5.5 gigapixel + // production bundle actually needed. + label 'process_gpu_qc' + label 'process_xl' + + conda "${moduleDir}/environment.yml" + // Built from environment.yml in this directory (see the module Dockerfile). + // Hosted on the author's quay.io namespace for now; to be migrated to the + // nf-core org before release. + container "quay.io/dongzehe/image_qc:1.0.0" + + input: + tuple val(meta), val(parameters), path(input_files) + path(roi_thresholds_yaml) + + output: + tuple val(meta), path(outdir), emit: outdir + tuple val("${task.process}"), val('python'), eval("python3 --version | sed 's/Python //'"), topic: versions, emit: versions_python + tuple val("${task.process}"), val('numpy'), eval("python3 -c 'import numpy; print(numpy.__version__)'"), topic: versions, emit: versions_numpy + tuple val("${task.process}"), val('scikit-image'), eval("python3 -c 'import skimage; print(skimage.__version__)'"), topic: versions, emit: versions_skimage + + when: + task.ext.when == null || task.ext.when + + script: + def prefix = task.ext.prefix ?: "${meta.id}" + outdir = prefix + + // Convert parameters to script arguments + def args = [] + + // Required parameters + args << "--xenium-bundle-dir '${input_files[0]}'" + args << "--outdir '${outdir}'" + args << "--sample-id '${meta.id}'" + + // ROI thresholds YAML (staged file) + if (roi_thresholds_yaml.name != 'NO_FILE') { + args << "--roi-thresholds-yaml '${roi_thresholds_yaml}'" + } + // Parse optional parameters from the parameters list + def param_map = [:] + if (parameters) { + parameters.collate(2).each { key, value -> + if (value != null && value != '') { + param_map[key] = value + } + } + } + + // --stain-names is deliberately not passed: the script parses it and then + // discards it (every figure title and metric key is hard-coded, and those + // keys are the contract the threshold YAML and the report read), so the + // pipeline parameter was dropped rather than shipping a knob with no effect. + + // Tile size parameter (a.k.a. ROI size internally). Per-sample + // `parameters` map can override via ROI_SIZE; otherwise the module-level + // `task.ext.tile_size` (wired from the Seqera-visible pipeline param) is + // used. Default 35 px. + if (param_map.containsKey('ROI_SIZE') && param_map['ROI_SIZE']) { + args << "--roi-size ${param_map['ROI_SIZE']}" + } else { + args << "--roi-size ${task.ext.tile_size ?: 35}" + } + + // Legacy focus score mode (CPU for-loop instead of GPU convolution) + if (task.ext.legacy_focus) { + args << "--legacy-focus" + } + + if (task.ext.no_snr) { + args << '--no-snr' + } + if (task.ext.snr_no_roi_tx_table) { + args << '--snr-no-roi-tx-table' + } + if (task.ext.snr_otsu_max_rois != null) { + args << "--snr-otsu-max-rois ${task.ext.snr_otsu_max_rois}" + } + // snr_no_moran true (default): Moran off — no flag. false: opt in to Moran. + if (!task.ext.snr_no_moran) { + args << '--snr-with-moran' + } + if (task.ext.save_dapi_maps_tiff) { + args << '--save-dapi-maps-tiff' + } + // Streaming is the script default: each tile is reduced and dropped, so no + // full-resolution pixel plane is written. The planes cost ~154 GB of scratch on + // a 5.5 gigapixel sample and their writeback to S3 is what pushed run + // 3nkeHOEV1ONlbK past the 4 h wall. Opt out only to reproduce the old path. + if (!task.ext.stream_tiles) { + args << '--no-stream-tiles' + } + // Per-figure figures_source/*.csv source-data exports are unused downstream + // (the QMD embeds only the PNGs) and the big ROI-table dumps cost ~80 s each. + // Off by default; opt in to write them. + if (task.ext.figure_source_tables) { + args << '--figure-source-tables' + } + // Figure generation master toggle. Off skips ALL figure rendering for a + // metrics-only fast run; metrics/JSON/parquet are always produced. + if (!task.ext.figures) { + args << '--no-figures' + } + // Cap the devices the script uses. `accelerator` only tells AWS Batch how many + // GPUs to request -- it does not restrict what CUDA can see, so a task asking + // for 1 GPU that lands on a 4-GPU instance would otherwise detect and use all + // four. Observed on run 3nkeHOEV1ONlbK: image_qc_gpus=1 placed on a + // g6e.12xlarge and the script reported "4 GPU(s)". + args << "--max-gpus ${task.ext.max_gpus}" + if (task.ext.lap_sigma != null) { + args << "--lap-sigma ${task.ext.lap_sigma}" + } + + if (param_map.containsKey('PIPELINE_SEGMENTATION') && param_map['PIPELINE_SEGMENTATION']) { + args << "--pipeline-segmentation '${param_map['PIPELINE_SEGMENTATION']}'" + } + + // Boolean flag — emit only when truthy (never '--is-resegmented false') + if (param_map.containsKey('IS_RESEGMENTED') && param_map['IS_RESEGMENTED'].toString() == 'true') { + args << "--is-resegmented" + } + + """ + export MKL_NUM_THREADS="$task.cpus" + export OPENBLAS_NUM_THREADS="$task.cpus" + export OMP_NUM_THREADS="$task.cpus" + export NUMBA_NUM_THREADS="$task.cpus" + + # Capture the analysis exit code without aborting (Nextflow runs with set -e). + # A very dim sample can make image_qc.py exit 1 (e.g. no tissue tiles clear the + # intensity gate); we still want a report, so on exit 1 we record a failed + # status and exit 0. The Quarto report reads image_qc_status.json and renders a + # QC-FAILED banner. Signal / OOM / preemption codes (104, 130-145) are re-raised + # so Nextflow's retry errorStrategy still fires. + rc=0 + image_qc.py \\ + ${args.join(' \\\n ')} || rc=\$? + + mkdir -p "${outdir}" + + if [ "\$rc" -eq 0 ]; then + echo '{"status": "ok", "sample_id": "${meta.id}"}' > "${outdir}/image_qc_status.json" + elif [ "\$rc" -eq 1 ]; then + echo "WARNING: image_qc.py exited 1 (Python error); writing failed status for report" >&2 + echo '{"status": "failed", "exit_code": 1, "sample_id": "${meta.id}"}' > "${outdir}/image_qc_status.json" + else + echo "image_qc.py exited \$rc (signal/OOM); propagating for errorStrategy" >&2 + exit \$rc + fi + """ + + stub: + def prefix = task.ext.prefix ?: "${meta.id}" + outdir = prefix + """ + mkdir -p "${outdir}/figures" + touch "${outdir}/image_qc_metrics.json" + touch "${outdir}/image_qc_metrics.csv" + """ +} diff --git a/modules/local/image_qc/meta.yml b/modules/local/image_qc/meta.yml new file mode 100644 index 00000000..a1681f1f --- /dev/null +++ b/modules/local/image_qc/meta.yml @@ -0,0 +1,101 @@ +name: "image_qc_analysis" +description: Image-level quality control for a Xenium output bundle - focus/blur maps, per-channel intensity metrics, signal-to-noise ratio metrics and QC figures. +keywords: + - quality control + - spatialomics + - imaging + - xenium + - signal to noise +tools: + - "image_qc": + description: "Combined image QC for Xenium bundles: GPU-accelerated tile/pixel focus maps, cell-based QC figures and mapping of tile results to cells." + homepage: "https://github.com/nf-core/spatialaxe" + documentation: "https://nf-co.re/spatialaxe" + tool_dev_url: "https://github.com/nf-core/spatialaxe" + doi: "" + licence: ["MIT"] + +input: + - - meta: + type: map + description: | + Groovy Map containing sample information + e.g. `[ id:'sample_id' ]` + - parameters: + type: list + description: | + Flat list of alternating key/value entries with per-sample overrides, + collated in pairs by the module. Recognised keys are `ROI_SIZE`, + `PIPELINE_SEGMENTATION` and `IS_RESEGMENTED`. Empty or null values are + ignored. + - input_files: + type: directory + description: | + Staged input paths for the sample. The first element is used as the + Xenium output bundle directory. + - - roi_thresholds_yaml: + type: file + description: | + YAML file of per-channel ROI QC thresholds. Pass a file literally + named `NO_FILE` to run without threshold overrides. + pattern: "*.{yml,yaml}" + +output: + outdir: + - - meta: + type: map + description: | + Groovy Map containing sample information + e.g. `[ id:'sample_id' ]` + - "${prefix}": + type: directory + description: | + Image QC output directory containing `image_qc_metrics.json`, + `image_qc_metrics.csv`, `image_qc_status.json`, per-ROI parquet + tables and a `figures/` subdirectory. + versions_python: + - - ${task.process}: + type: string + description: The process the versions were collected from + - python: + type: string + description: The tool name + - "python3 --version | sed 's/Python //'": + type: eval + description: The expression to obtain the version of the tool + versions_numpy: + - - ${task.process}: + type: string + description: The process the versions were collected from + - numpy: + type: string + description: The tool name + - "python3 -c 'import numpy; print(numpy.__version__)'": + type: eval + description: The expression to obtain the version of the tool + versions_skimage: + - - ${task.process}: + type: string + description: The process the versions were collected from + - scikit-image: + type: string + description: The tool name + - "python3 -c 'import skimage; print(skimage.__version__)'": + type: eval + description: The expression to obtain the version of the tool +topics: + versions: + - - ${task.process}: + type: string + description: The process the versions were collected from + - python: + type: string + description: The tool name + - "python3 --version | sed 's/Python //'": + type: eval + description: The expression to obtain the version of the tool + +authors: + - "@an-altosian" +maintainers: + - "@an-altosian" diff --git a/modules/local/image_qc/tests/main.nf.test b/modules/local/image_qc/tests/main.nf.test new file mode 100644 index 00000000..bb27d9aa --- /dev/null +++ b/modules/local/image_qc/tests/main.nf.test @@ -0,0 +1,50 @@ +nextflow_process { + + name "Test Process IMAGE_QC_ANALYSIS" + script "../main.nf" + process "IMAGE_QC_ANALYSIS" + config "./nextflow.config" + + tag "modules" + tag "modules_local" + tag "image_qc" + tag "qc" + + test("image qc analysis - stub") { + + options "-stub" + + when { + process { + """ + // Fixtures are built here at test time instead of being committed + // to the repository (reviewer request). The stub block never runs + // image_qc.py, so an empty bundle skeleton is sufficient. + def fixtureRoot = new File("${outputDir}/image_qc_fixtures") + def bundleDir = new File(fixtureRoot, "bundle") + bundleDir.mkdirs() + new File(bundleDir, "experiment.xenium").text = '{"analysis_sw_version": "xenium-4.0.1.0"}' + def roiYaml = new File(fixtureRoot, "roi_thresholds.yaml") + roiYaml.text = 'channels: {}' + + input[0] = [ + [ id:'test' ], + [], + [ file(bundleDir.toString(), checkIfExists: true) ] + ] + input[1] = file(roiYaml.toString(), checkIfExists: true) + """ + } + } + + then { + assertAll( + { assert process.success }, + { assert file(process.out.outdir[0][1]).isDirectory() }, + { assert file(process.out.outdir[0][1] + "/image_qc_metrics.json").exists() }, + { assert file(process.out.outdir[0][1] + "/image_qc_metrics.csv").exists() }, + { assert file(process.out.outdir[0][1] + "/figures").isDirectory() } + ) + } + } +} diff --git a/modules/local/image_qc/tests/nextflow.config b/modules/local/image_qc/tests/nextflow.config new file mode 100644 index 00000000..8983e39f --- /dev/null +++ b/modules/local/image_qc/tests/nextflow.config @@ -0,0 +1,29 @@ +process { + + resourceLimits = [ + cpus: 4, + memory: '8.GB', + time: '2.h', + ] + + // Worked example of the module's `task.ext.*` contract. The process reads no + // `params.*` (nf-core reviewer requirement), so every knob upstream exposed + // as a pipeline parameter is supplied here. Values below reproduce upstream + // nf-xenium-processing defaults; the pipeline-level wiring in + // conf/modules.config must set the same keys. + withName: 'IMAGE_QC_ANALYSIS' { + ext.tile_size = 35 + ext.legacy_focus = false + ext.no_snr = false + ext.snr_no_roi_tx_table = false + ext.snr_otsu_max_rois = null + ext.snr_no_moran = true // true = Moran's I off (upstream default) + ext.save_dapi_maps_tiff = false + ext.stream_tiles = true // true = stream tiles (upstream default) + ext.figure_source_tables = false + ext.figures = true // true = render figures (upstream default) + ext.max_gpus = 1 + ext.lap_sigma = null + } + +} diff --git a/modules/local/transcript_qc/Dockerfile b/modules/local/transcript_qc/Dockerfile new file mode 100644 index 00000000..691bf3a4 --- /dev/null +++ b/modules/local/transcript_qc/Dockerfile @@ -0,0 +1,23 @@ +# Container for the transcript QC module (analysis + Quarto report). +# Built from environment.yml in this directory. Tagged and pushed as +# quay.io/dongzehe/transcript_qc:1.0.0 (to be migrated to the nf-core org for release). +# +# docker build -t quay.io//transcript_qc:1.0.0 modules/local/transcript_qc +# +# Note: behind a corporate proxy / private conda mirror you may need to provide a +# CA bundle and channel config at build time (e.g. via BuildKit --mount=type=secret); +# a standard build resolves the pinned packages directly from conda-forge/bioconda. +FROM mambaorg/micromamba:2.0.5 +COPY --chown=$MAMBA_USER:$MAMBA_USER environment.yml /tmp/environment.yml +RUN micromamba install -y -n base -f /tmp/environment.yml \ + && micromamba clean -a -f -y \ + && rm -f /tmp/environment.yml +ENV PATH="/opt/conda/bin:$PATH" \ + QUARTO_DENO=/opt/conda/bin/deno \ + QUARTO_DENO_DOM=/opt/conda/lib/deno_dom.so \ + QUARTO_PANDOC=/opt/conda/bin/pandoc \ + QUARTO_ESBUILD=/opt/conda/bin/esbuild \ + QUARTO_TYPST=/opt/conda/bin/typst \ + QUARTO_DART_SASS=/opt/conda/bin/sass \ + QUARTO_SHARE_PATH=/opt/conda/share/quarto \ + QUARTO_CONDA_PREFIX=/opt/conda diff --git a/modules/local/transcript_qc/environment.yml b/modules/local/transcript_qc/environment.yml new file mode 100644 index 00000000..03440f6e --- /dev/null +++ b/modules/local/transcript_qc/environment.yml @@ -0,0 +1,32 @@ +--- +# yaml-language-server: $schema=https://raw.githubusercontent.com/nf-core/modules/master/modules/environment-schema.json +channels: + - conda-forge + - bioconda +dependencies: + # procps-ng provides `ps`, required by Nextflow for task-metric collection + - conda-forge::procps-ng=4.0.4 + - conda-forge::python=3.11.0 + # transcript_qc_processing.py: numpy, pandas, matplotlib, seaborn, scanpy + # (-> anndata, h5py for cell_feature_matrix.h5), pyarrow.parquet, scipy.stats + # (calculate_noise_bound) and pyyaml (parse_tool_versions). + - conda-forge::numpy=2.3.3 + - conda-forge::pandas=2.3.2 + - conda-forge::scipy=1.16.2 + - conda-forge::matplotlib=3.10.6 + - conda-forge::seaborn=0.13.2 + - conda-forge::scanpy=1.11.4 + - conda-forge::anndata=0.12.2 + - conda-forge::h5py=3.14.0 + - conda-forge::pyyaml=6.0.3 + # transcript_stream.py: pyarrow.dataset / .compute / acero streaming engine + - conda-forge::pyarrow=20.0.0 + - conda-forge::numba=0.62.0 + - conda-forge::tqdm=4.67.1 + # Report rendering (Quarto) — this container also renders the transcript QC + # HTML report from bin/transcript_qc.qmd + - conda-forge::quarto=1.6.42 + - conda-forge::jupyter=1.1.1 + - conda-forge::ipython=9.5.0 + - conda-forge::ipykernel=6.30.1 + - conda-forge::papermill=2.6.0 diff --git a/modules/local/transcript_qc/main.nf b/modules/local/transcript_qc/main.nf new file mode 100644 index 00000000..06bd0bd1 --- /dev/null +++ b/modules/local/transcript_qc/main.nf @@ -0,0 +1,110 @@ +process TRANSCRIPT_QC_PROCESSING { + tag "${meta.id}" + label 'process_high' + + conda "${moduleDir}/environment.yml" + // Built from environment.yml in this directory (see the module Dockerfile). + // Hosted on the author's quay.io namespace for now; to be migrated to the + // nf-core org before release. + container "quay.io/dongzehe/transcript_qc:1.1.0" + + input: + tuple val(meta), val(parameters), path(input_files) + + output: + tuple val(meta), path(outdir), emit: outdir + tuple val("${task.process}"), val('python'), eval("python3 --version | sed 's/Python //'"), topic: versions, emit: versions_python + tuple val("${task.process}"), val('numpy'), eval("python3 -c 'import numpy; print(numpy.__version__)'"), topic: versions, emit: versions_numpy + tuple val("${task.process}"), val('pandas'), eval("python3 -c 'import pandas; print(pandas.__version__)'"), topic: versions, emit: versions_pandas + tuple val("${task.process}"), val('pyarrow'), eval("python3 -c 'import pyarrow; print(pyarrow.__version__)'"), topic: versions, emit: versions_pyarrow + tuple val("${task.process}"), val('scanpy'), eval("python3 -c 'import scanpy; print(scanpy.__version__)'"), topic: versions, emit: versions_scanpy + tuple val("${task.process}"), val('anndata'), eval("python3 -c 'import anndata; print(anndata.__version__)'"), topic: versions, emit: versions_anndata + tuple val("${task.process}"), val('scipy'), eval("python3 -c 'import scipy; print(scipy.__version__)'"), topic: versions, emit: versions_scipy + tuple val("${task.process}"), val('matplotlib'), eval("python3 -c 'import matplotlib; print(matplotlib.__version__)'"), topic: versions, emit: versions_matplotlib + tuple val("${task.process}"), val('seaborn'), eval("python3 -c 'import seaborn; print(seaborn.__version__)'"), topic: versions, emit: versions_seaborn + + when: + task.ext.when == null || task.ext.when + + script: + def prefix = task.ext.prefix ?: "${meta.id}" + outdir = prefix + + // Convert parameters to script arguments + def args = [] + + // Required parameters + args << "--xenium-bundle-dir '${input_files[0]}'" + args << "--outdir '${outdir}'" + args << "--threads ${task.cpus}" + args << "--task-process '${task.process}'" + + // Parse optional parameters from the parameters list + def param_map = [:] + if (parameters) { + parameters.collate(2).each { key, value -> + if (value != null && value != '') { + param_map[key] = value + } + } + } + + // Add optional parameters. Per-sample `parameters` entries win; the + // module-level `task.ext.*` keys are the fallback, so the process never + // reads `params.*` directly (nf-core reviewer requirement). + def non_gene_prefix = param_map['NON_GENE_PREFIX'] ?: task.ext.non_gene_prefix + if (non_gene_prefix) { + def prefixes = non_gene_prefix.toString().split(';').collect { "'${it.trim()}'" }.join(' ') + args << "--non-gene-prefix ${prefixes}" + } + + // --stain-names is deliberately not passed: the script parses it and never + // uses it, so the pipeline parameter was dropped instead of exposing a knob + // with no effect. + + def num_row_groups = param_map['NUM_ROW_GROUPS'] ?: task.ext.num_row_groups + if (num_row_groups) { + args << "--num-row-groups ${num_row_groups}" + } + + def pipeline_segmentation = param_map['PIPELINE_SEGMENTATION'] ?: task.ext.pipeline_segmentation + if (pipeline_segmentation) { + args << "--pipeline-segmentation '${pipeline_segmentation}'" + } + + // Boolean flag — emit only when truthy (never '--is-resegmented false') + def is_resegmented = param_map.containsKey('IS_RESEGMENTED') ? param_map['IS_RESEGMENTED'] : task.ext.is_resegmented + if (is_resegmented != null && is_resegmented.toString() == 'true') { + args << "--is-resegmented" + } + + // Run-global segmentation versions.yml (staged as input_files[1] on + // post-seg runs; input_files[0] is always the bundle dir). Used to label + // the pipeline tool version. + if (input_files.size() > 1) { + args << "--seg-versions-file '${input_files[1]}'" + } + + """ + # print() is block-buffered when stdout is not a tty, so an OOM SIGKILL + # discards every buffered progress line and the task log shows only the kill. + # That is what made this module's OOM undiagnosable. + export PYTHONUNBUFFERED=1 + export MKL_NUM_THREADS="$task.cpus" + export OPENBLAS_NUM_THREADS="$task.cpus" + export OMP_NUM_THREADS="$task.cpus" + export NUMBA_NUM_THREADS="$task.cpus" + + + transcript_qc_processing.py \\ + ${args.join(' \\\n ')} + """ + + stub: + def prefix = task.ext.prefix ?: "${meta.id}" + outdir = prefix + """ + mkdir -p "${outdir}/figures" + touch "${outdir}/transcript_qc_metrics.json" + """ +} diff --git a/modules/local/transcript_qc/meta.yml b/modules/local/transcript_qc/meta.yml new file mode 100644 index 00000000..11eba6ef --- /dev/null +++ b/modules/local/transcript_qc/meta.yml @@ -0,0 +1,160 @@ +name: "transcript_qc_processing" +description: Transcript-level quality control for a Xenium output bundle - codeword category breakdown, quality-value (QV) distributions, per-FoV consistency and density, per-feature and per-cell transcript counts, transcript-to-cell assignment and QC figures. +keywords: + - quality control + - spatialomics + - transcriptomics + - xenium + - transcripts +tools: + - "transcript_qc": + description: "Streaming transcript QC for Xenium bundles: reads transcripts.parquet with a pyarrow acero pipeline, derives sample-, FoV-, feature- and cell-level QC metrics and renders the accompanying figures." + homepage: "https://github.com/nf-core/spatialaxe" + documentation: "https://nf-co.re/spatialaxe" + tool_dev_url: "https://github.com/nf-core/spatialaxe" + doi: "" + licence: ["MIT"] + +input: + - - meta: + type: map + description: | + Groovy Map containing sample information + e.g. `[ id:'sample_id' ]` + - parameters: + type: list + description: | + Flat list of alternating key/value entries with per-sample overrides, + collated in pairs by the module. Recognised keys are + `NON_GENE_PREFIX`, `NUM_ROW_GROUPS`, + `PIPELINE_SEGMENTATION` and `IS_RESEGMENTED`. Empty or null values are + ignored and the corresponding `task.ext.*` fallback is used instead. + - input_files: + type: directory + description: | + Staged input paths for the sample. The first element is the Xenium + output bundle directory. An optional second element is a run-global + segmentation `versions.yml`, used to label the segmentation software + on post-segmentation runs. + +output: + outdir: + - - meta: + type: map + description: | + Groovy Map containing sample information + e.g. `[ id:'sample_id' ]` + - "${prefix}": + type: directory + description: | + Transcript QC output directory containing + `transcript_qc_metrics.json`, `versions.yml`, + `num_transcripts_per_cell.csv`, `num_genes_per_cell.csv`, a + `figures/` subdirectory of PDF/PNG figures and a `figures_source/` + subdirectory of per-figure CSV source data. + versions_python: + - - ${task.process}: + type: string + description: The process the versions were collected from + - python: + type: string + description: The tool name + - "python3 --version | sed 's/Python //'": + type: eval + description: The expression to obtain the version of the tool + versions_numpy: + - - ${task.process}: + type: string + description: The process the versions were collected from + - numpy: + type: string + description: The tool name + - "python3 -c 'import numpy; print(numpy.__version__)'": + type: eval + description: The expression to obtain the version of the tool + versions_pandas: + - - ${task.process}: + type: string + description: The process the versions were collected from + - pandas: + type: string + description: The tool name + - "python3 -c 'import pandas; print(pandas.__version__)'": + type: eval + description: The expression to obtain the version of the tool + versions_pyarrow: + - - ${task.process}: + type: string + description: The process the versions were collected from + - pyarrow: + type: string + description: The tool name + - "python3 -c 'import pyarrow; print(pyarrow.__version__)'": + type: eval + description: The expression to obtain the version of the tool + versions_scanpy: + - - ${task.process}: + type: string + description: The process the versions were collected from + - scanpy: + type: string + description: The tool name + - "python3 -c 'import scanpy; print(scanpy.__version__)'": + type: eval + description: The expression to obtain the version of the tool + versions_anndata: + - - ${task.process}: + type: string + description: The process the versions were collected from + - anndata: + type: string + description: The tool name + - "python3 -c 'import anndata; print(anndata.__version__)'": + type: eval + description: The expression to obtain the version of the tool + versions_scipy: + - - ${task.process}: + type: string + description: The process the versions were collected from + - scipy: + type: string + description: The tool name + - "python3 -c 'import scipy; print(scipy.__version__)'": + type: eval + description: The expression to obtain the version of the tool + versions_matplotlib: + - - ${task.process}: + type: string + description: The process the versions were collected from + - matplotlib: + type: string + description: The tool name + - "python3 -c 'import matplotlib; print(matplotlib.__version__)'": + type: eval + description: The expression to obtain the version of the tool + versions_seaborn: + - - ${task.process}: + type: string + description: The process the versions were collected from + - seaborn: + type: string + description: The tool name + - "python3 -c 'import seaborn; print(seaborn.__version__)'": + type: eval + description: The expression to obtain the version of the tool +topics: + versions: + - - ${task.process}: + type: string + description: The process the versions were collected from + - python: + type: string + description: The tool name + - "python3 --version | sed 's/Python //'": + type: eval + description: The expression to obtain the version of the tool + +authors: + - "@an-altosian" +maintainers: + - "@an-altosian" diff --git a/modules/local/transcript_qc/tests/main.nf.test b/modules/local/transcript_qc/tests/main.nf.test new file mode 100644 index 00000000..b6e2ba58 --- /dev/null +++ b/modules/local/transcript_qc/tests/main.nf.test @@ -0,0 +1,87 @@ +nextflow_process { + + name "Test Process TRANSCRIPT_QC_PROCESSING" + script "../main.nf" + process "TRANSCRIPT_QC_PROCESSING" + config "./nextflow.config" + + tag "modules" + tag "modules_local" + tag "transcript_qc" + tag "qc" + + test("transcript qc processing - stub") { + + options "-stub" + + when { + process { + """ + // Fixtures are built here at test time instead of being committed + // to the repository (reviewer request). The stub block never runs + // transcript_qc_processing.py, so an empty bundle skeleton is + // sufficient. + def fixtureRoot = new File("${outputDir}/transcript_qc_fixtures") + def bundleDir = new File(fixtureRoot, "bundle") + bundleDir.mkdirs() + new File(bundleDir, "experiment.xenium").text = '{"analysis_sw_version": "xenium-4.0.1.0"}' + + input[0] = [ + [ id:'test' ], + [], + [ file(bundleDir.toString(), checkIfExists: true) ] + ] + """ + } + } + + then { + assertAll( + { assert process.success }, + { assert file(process.out.outdir[0][1]).isDirectory() }, + { assert file(process.out.outdir[0][1] + "/transcript_qc_metrics.json").exists() }, + { assert file(process.out.outdir[0][1] + "/figures").isDirectory() } + ) + } + } + + test("transcript qc processing - stub - per-sample parameters override") { + + options "-stub" + + when { + process { + """ + def fixtureRoot = new File("${outputDir}/transcript_qc_fixtures_params") + def bundleDir = new File(fixtureRoot, "bundle") + bundleDir.mkdirs() + new File(bundleDir, "experiment.xenium").text = '{"analysis_sw_version": "xenium-4.0.1.0"}' + def segVersions = new File(fixtureRoot, "seg_versions.yml") + segVersions.text = 'CELLPOSE:\\n cellpose: 3.0.6\\n' + + input[0] = [ + [ id:'test_reseg' ], + [ + 'NON_GENE_PREFIX', 'NegControlProbe;NegControlCodeword', + 'NUM_ROW_GROUPS', 4, + 'PIPELINE_SEGMENTATION', 'cellpose', + 'IS_RESEGMENTED', 'true', + ], + [ + file(bundleDir.toString(), checkIfExists: true), + file(segVersions.toString(), checkIfExists: true), + ] + ] + """ + } + } + + then { + assertAll( + { assert process.success }, + { assert file(process.out.outdir[0][1]).isDirectory() }, + { assert file(process.out.outdir[0][1] + "/transcript_qc_metrics.json").exists() } + ) + } + } +} diff --git a/modules/local/transcript_qc/tests/nextflow.config b/modules/local/transcript_qc/tests/nextflow.config new file mode 100644 index 00000000..a7256b51 --- /dev/null +++ b/modules/local/transcript_qc/tests/nextflow.config @@ -0,0 +1,21 @@ +process { + + resourceLimits = [ + cpus: 4, + memory: '8.GB', + time: '2.h', + ] + + // Worked example of the module's `task.ext.*` contract. The process reads no + // `params.*` (nf-core reviewer requirement), so every knob upstream exposed + // as a pipeline parameter is supplied here. Values below reproduce upstream + // nf-xenium-processing defaults; the pipeline-level wiring in + // conf/modules.config must set the same keys. + withName: 'TRANSCRIPT_QC_PROCESSING' { + ext.non_gene_prefix = null // null -> script default 'NegControlProbe' + ext.num_row_groups = null // null -> read every row group + ext.pipeline_segmentation = 'skip' // upstream default (params.segmentation) + ext.is_resegmented = false // true only for post-resegmentation bundles + } + +} diff --git a/modules/nf-core/quarto/notebook/environment.yml b/modules/nf-core/quarto/notebook/environment.yml new file mode 100644 index 00000000..88aba3ac --- /dev/null +++ b/modules/nf-core/quarto/notebook/environment.yml @@ -0,0 +1,14 @@ +--- +# yaml-language-server: $schema=https://raw.githubusercontent.com/nf-core/modules/master/modules/environment-schema.json +channels: + - conda-forge + - bioconda +dependencies: + # renovate: datasource=conda depName=conda-forge/quarto + - conda-forge::jupyter=1.1.1 + - conda-forge::matplotlib=3.10.3 + - conda-forge::papermill=2.6.0 + - conda-forge::python=3.12.11 + - conda-forge::quarto=1.7.31 + - conda-forge::r-base=4.4.3 + - conda-forge::r-rmarkdown=2.29 diff --git a/modules/nf-core/quarto/notebook/main.nf b/modules/nf-core/quarto/notebook/main.nf new file mode 100644 index 00000000..8a1bee78 --- /dev/null +++ b/modules/nf-core/quarto/notebook/main.nf @@ -0,0 +1,123 @@ +// NB 1: You'll likely want to override this with a container containing all +// required dependencies for your analyses. Or use wave to build the container +// for you from the environment.yml You'll at least need Quarto itself, +// Papermill and whatever language you are running your analyses on; you can see +// an example in this module's environment file. +// +// NB 2: You'll need to export the versions of the packages you are using inside +// your notebook to a `versions.csv` file (formatted as `package,version`), +// which will be added to the `versions` topic; module versions are handled +// separately by `eval()` statements. +process QUARTO_NOTEBOOK { + tag "${meta.id}" + label 'process_low' + conda "${moduleDir}/environment.yml" + container "${workflow.containerEngine in ['singularity', 'apptainer'] && !task.ext.singularity_pull_docker_container + ? 'https://community-cr-prod.seqera.io/docker/registry/v2/blobs/sha256/28/28717ccd9ce22dbfc219f3db088d5a1fc2ca1f575b5c65621218596dcdbaac95/data' + : 'community.wave.seqera.io/library/jupyter_matplotlib_papermill_quarto_r-rmarkdown:6d15193ce3dfc665'}" + + input: + tuple val(meta), path(notebook) + val(parameters) + path input_files + path extensions + + output: + tuple val(meta), path("*.html") , emit: html + tuple val(meta), path(notebook) , emit: notebook + tuple val(meta), path("params.yml") , emit: params_yaml + tuple val(meta), path("${notebook_parameters.artifact_dir}/*") , emit: artifacts , optional: true + tuple val(meta), path("_extensions") , emit: extensions , optional: true + path "versions.yml" , emit: versions , topic: versions + tuple val("${task.process}"), val('quarto'), eval('quarto -v') , emit: versions_quarto , topic: versions + tuple val("${task.process}"), val('papermill'), eval('papermill --version | cut -f1 -d" "'), emit: versions_papermill, topic: versions + + when: + task.ext.when == null || task.ext.when + + script: + def args = task.ext.args ?: '' + def prefix = task.ext.prefix ?: "${meta.id}" + // Implicit parameters can be overwritten by supplying a value with parameters + notebook_parameters = [ + meta: meta, + cpus: task.cpus, + artifact_dir: "artifacts", + ] + (parameters ?: [:]) + // Parse parameters through a YAML file, which is better than CLI because: + // - No issue with escaping + // - Allows passing nested maps instead of just single values + // - Allows running with the language-agnostic `--execute-params` + def yamlBuilder = new groovy.yaml.YamlBuilder() + yamlBuilder.call(notebook_parameters) + def yaml_content = yamlBuilder.toString().tokenize('\n').join("\n ") + """ + # Dump parameters to yaml file + cat <<- END_YAML_PARAMS > params.yml + ${yaml_content} + END_YAML_PARAMS + + # Create output directory + mkdir "${notebook_parameters.artifact_dir}" + + # Set environment variables needed for Quarto rendering + export XDG_CACHE_HOME="./.xdg_cache_home" + export XDG_DATA_HOME="./.xdg_data_home" + + # Fix Quarto for Apptainer (see https://community.seqera.io/t/confusion-over-why-a-tool-works-in-docker-but-fails-in-singularity-when-the-installation-doesnt-differ-i-e-using-wave-micromamba/1244) + ENV_QUARTO=/opt/conda/etc/conda/activate.d/quarto.sh + set +u + if [ -z "\${QUARTO_DENO}" ] && [ -f "\${ENV_QUARTO}" ]; then + source "\${ENV_QUARTO}" + fi + set -u + + # Set parallelism for BLAS/MKL etc. to avoid over-booking of resources + export MKL_NUM_THREADS="${task.cpus}" + export OPENBLAS_NUM_THREADS="${task.cpus}" + export OMP_NUM_THREADS="${task.cpus}" + export NUMBA_NUM_THREADS="${task.cpus}" + + # Render notebook + quarto render \\ + ${notebook} \\ + ${args} \\ + --execute-params params.yml \\ + --output ${prefix}.html + + # Check that notebook package versions is exported + if [ ! -f versions.csv ]; then + echo "ERROR: versions.csv not found; the notebook must write out [tool,version] pairs used within it." >&2 + exit 1 + fi + + # Write notebook package versions to YAML + cat <<- END_VERSIONS > versions.yml + "${task.process}": + \$(awk -F',' '{printf " %s: %s\\n", \$1, \$2}' versions.csv) + END_VERSIONS + """ + + stub: + def prefix = task.ext.prefix ?: "${meta.id}" + // Implicit parameters can be overwritten by supplying a value with parameters + notebook_parameters = [ + meta: meta, + cpus: task.cpus, + artifact_dir: "artifacts", + ] + (parameters ?: [:]) + """ + # Fix Quarto for Apptainer (see https://community.seqera.io/t/confusion-over-why-a-tool-works-in-docker-but-fails-in-singularity-when-the-installation-doesnt-differ-i-e-using-wave-micromamba/1244) + # Note: This is needed in the stub for `quarto -v` to work. + ENV_QUARTO=/opt/conda/etc/conda/activate.d/quarto.sh + set +u + if [ -z "\${QUARTO_DENO}" ] && [ -f "\${ENV_QUARTO}" ]; then + source "\${ENV_QUARTO}" + fi + set -u + + touch ${prefix}.html + touch params.yml + touch versions.yml + """ +} diff --git a/modules/nf-core/quarto/notebook/meta.yml b/modules/nf-core/quarto/notebook/meta.yml new file mode 100644 index 00000000..57ce16ab --- /dev/null +++ b/modules/nf-core/quarto/notebook/meta.yml @@ -0,0 +1,166 @@ +name: "quarto_notebook" +description: Render a Quarto notebook, including parametrization. +keywords: + - quarto + - notebook + - reports + - python + - r +tools: + - quarto: + description: An open-source scientific and technical publishing system. + homepage: https://quarto.org/ + documentation: https://quarto.org/docs/reference/ + tool_dev_url: https://github.com/quarto-dev/quarto-cli + licence: + - "MIT" + identifier: "" + - papermill: + description: Parameterize, execute, and analyze notebooks + homepage: https://github.com/nteract/papermill + documentation: http://papermill.readthedocs.io/en/latest/ + tool_dev_url: https://github.com/nteract/papermill + licence: + - "BSD 3-clause" + identifier: "" +input: + - - meta: + type: map + description: | + Groovy Map containing sample information + e.g. `[ id:'sample1', single_end:false ]`. + - notebook: + type: file + description: The Quarto notebook to be rendered. + pattern: "*.{qmd}" + ontologies: [] + - parameters: + type: map + description: | + Groovy map with notebook parameters which will be passed to Quarto to + generate parametrized reports. + - input_files: + type: file + description: One or multiple files serving as input data for the notebook. + pattern: "*" + ontologies: [] + - extensions: + type: file + description: | + A quarto `_extensions` directory with custom template(s) to be + available for rendering. + pattern: "*" + ontologies: [] +output: + html: + - - meta: + type: map + description: | + Groovy Map containing sample information + e.g. `[ id:'sample1', single_end:false ]`. + - "*.html": + type: file + description: HTML report generated by Quarto. + pattern: "*.html" + ontologies: [] + notebook: + - - meta: + type: map + description: | + Groovy Map containing sample information + e.g. `[ id:'sample1', single_end:false ]`. + - notebook: + type: file + description: The Quarto notebook that was rendered. Allows user to continue working on the notebook. + pattern: "*.{qmd}" + ontologies: [] + params_yaml: + - - meta: + type: map + description: | + Groovy Map containing sample information + e.g. `[ id:'sample1', single_end:false ]`. + - params.yml: + type: file + description: Parameters used during report rendering. + pattern: "*" + ontologies: [] + artifacts: + - - meta: + type: map + description: | + Groovy Map containing sample information + e.g. `[ id:'sample1', single_end:false ]`. + - ${notebook_parameters.artifact_dir}/*: + type: file + description: Artifacts generated during report rendering. + pattern: "*" + ontologies: [] + extensions: + - - meta: + type: map + description: | + Groovy Map containing sample information + e.g. `[ id:'sample1', single_end:false ]`. + - _extensions: + type: file + description: Quarto extensions used during report rendering. + pattern: "*" + ontologies: [] + versions: + - versions.yml: + type: file + description: Optional YAML file containing software versions + pattern: "versions.yml" + ontologies: + - edam: http://edamontology.org/format_3750 + versions_quarto: + - - ${task.process}: + type: string + description: The name of the process + - quarto: + type: string + description: The name of the tool + - quarto -v: + type: eval + description: The expression to obtain the version of the tool + versions_papermill: + - - ${task.process}: + type: string + description: The name of the process + - papermill: + type: string + description: The name of the tool + - papermill --version | cut -f1 -d" ": + type: eval + description: The expression to obtain the version of the tool +topics: + versions: + - versions.yml: + type: file + description: Optional YAML file containing software versions + pattern: "versions.yml" + ontologies: + - edam: http://edamontology.org/format_3750 + - - ${task.process}: + type: string + description: The name of the process + - quarto: + type: string + description: The name of the tool + - quarto -v: + type: eval + description: The expression to obtain the version of the tool + - - ${task.process}: + type: string + description: The name of the process + - papermill: + type: string + description: The name of the tool + - papermill --version | cut -f1 -d" ": + type: eval + description: The expression to obtain the version of the tool +authors: + - "@fasterius" +maintainers: + - "@fasterius" diff --git a/modules/nf-core/quarto/notebook/tests/main.nf.test b/modules/nf-core/quarto/notebook/tests/main.nf.test new file mode 100644 index 00000000..bb10b752 --- /dev/null +++ b/modules/nf-core/quarto/notebook/tests/main.nf.test @@ -0,0 +1,231 @@ +nextflow_process { + + name "Test Process QUARTO_NOTEBOOK" + script "../main.nf" + process "QUARTO_NOTEBOOK" + config "./nextflow.config" + + tag "modules" + tag "modules_nfcore" + tag "quarto" + tag "quarto/notebook" + + test("test notebook - [qmd:r]") { + + when { + process { + """ + input[0] = [ + [ id:'test' ], // meta map + file(params.modules_testdata_base_path + 'generic/notebooks/quarto/quarto_r.qmd', checkIfExists: true) // Notebook + ] + input[1] = [:] // Parameters + input[2] = [] // Input files + input[3] = [] // Extensions + """ + } + } + + then { + assertAll( + { assert process.success }, + { assert snapshot( + process.out.findAll { key, val -> key.startsWith('versions') }, + process.out.artifacts, + process.out.params_yaml + ).match() }, + { assert path(process.out.html[0][1]).readLines().any { it.contains('Hello World 1') } }, + ) + } + + } + + test("test notebook - [qmd:python]") { + + when { + process { + """ + input[0] = [ + [ id:'test' ], // meta map + file(params.modules_testdata_base_path + 'generic/notebooks/quarto/quarto_python.qmd', checkIfExists: true) // Notebook + ] + input[1] = [:] // Parameters + input[2] = [] // Input files + input[3] = [] // Extensions + """ + } + } + + then { + assertAll( + { assert process.success }, + { assert snapshot( + process.out.findAll { key, val -> key.startsWith('versions') }, + process.out.artifacts, + process.out.params_yaml + ).match() }, + { assert path(process.out.html[0][1]).readLines().any { it.contains('Hello World 1') } }, + ) + } + + } + + test("test notebook - parametrized - [qmd:r]") { + + when { + process { + """ + input[0] = [ + [ id:'test' ], // meta map + file(params.modules_testdata_base_path + 'generic/notebooks/quarto/quarto_r.qmd', checkIfExists: true) // Notebook + ] + input[1] = [input_filename: "hello.txt", n_iter: 12] // Parameters + input[2] = file(params.modules_testdata_base_path + 'generic/txt/hello.txt', checkIfExists: true) // Input files + input[3] = [] // Extensions + """ + } + } + + then { + assertAll( + { assert process.success }, + { assert snapshot( + process.out.findAll { key, val -> key.startsWith('versions') }, + process.out.artifacts, + process.out.params_yaml + ).match() }, + { assert path(process.out.html[0][1]).readLines().any { it.contains('Hello World 1') } }, + { assert path(process.out.params_yaml[0][1]).readLines().any { it.contains('meta:') } }, + ) + } + + } + + test("test notebook - parametrized - [qmd:python]") { + + when { + process { + """ + input[0] = [ + [ id:'test' ], // meta map + file(params.modules_testdata_base_path + 'generic/notebooks/quarto/quarto_python.qmd', checkIfExists: true) // Notebook + ] + input[1] = [input_filename: "hello.txt", n_iter: 12] // Parameters + input[2] = file(params.modules_testdata_base_path + 'generic/txt/hello.txt', checkIfExists: true) // Input files + input[3] = [] // Extensions + """ + } + } + + then { + assertAll( + { assert process.success }, + { assert snapshot( + process.out.findAll { key, val -> key.startsWith('versions') }, + process.out.artifacts, + process.out.params_yaml + ).match() }, + { assert path(process.out.html[0][1]).readLines().any { it.contains('Hello World 1') } }, + { assert path(process.out.params_yaml[0][1]).readLines().any { it.contains('meta:') } }, + ) + } + + } + + test("test notebook - parametrized - [rmd]") { + + when { + process { + """ + input[0] = [ + [ id:'test' ], // meta map + file(params.modules_testdata_base_path + 'generic/notebooks/rmarkdown/rmarkdown_notebook.Rmd', checkIfExists: true) // notebook + ] + input[1] = [input_filename: "hello.txt", n_iter: 12] // Parameters + input[2] = file(params.modules_testdata_base_path + 'generic/txt/hello.txt', checkIfExists: true) // Input files + input[3] = [] // Extensions + """ + } + } + + then { + assertAll( + { assert process.success }, + { assert snapshot( + process.out.findAll { key, val -> key.startsWith('versions') }, + process.out.artifacts, + process.out.params_yaml + ).match() }, + { assert path(process.out.html[0][1]).readLines().any { it.contains('Hello World 1') } }, + { assert path(process.out.params_yaml[0][1]).readLines().any { it.contains('meta:') } }, + ) + } + + } + + test("test notebook - parametrized - [ipynb]") { + + when { + params { + // The Jupyter Notebook test need to explicitly execute the code + // in the notebook instead of just rendering the notebook to + // HTML (which is the default), in order to correctly output + // `versions.csv`. + module_args = '--execute' + } + process { + """ + input[0] = [ + [ id:'test' ], // meta map + file(params.modules_testdata_base_path + 'generic/notebooks/jupyter/ipython_notebook.ipynb', checkIfExists: true) // notebook + ] + input[1] = [input_filename: "hello.txt", n_iter: 12] // Parameters + input[2] = file(params.modules_testdata_base_path + 'generic/txt/hello.txt', checkIfExists: true) // Input files + input[3] = [] // Extensions + """ + } + } + + then { + assertAll( + { assert process.success }, + { assert snapshot( + process.out.findAll { key, val -> key.startsWith('versions') }, + process.out.artifacts, + process.out.params_yaml + ).match() }, + { assert path(process.out.html[0][1]).readLines().any { it.contains('Hello World') } }, + { assert path(process.out.params_yaml[0][1]).readLines().any { it.contains('meta:') } }, + ) + } + + } + + test("test notebook - stub - [qmd:r]") { + + options "-stub" + + when { + process { + """ + input[0] = [ + [ id:'test' ], // meta map + file(params.modules_testdata_base_path + 'generic/notebooks/quarto/quarto_r.qmd', checkIfExists: true) // Notebook + ] + input[1] = [:] // Parameters + input[2] = [] // Input files + input[3] = [] // Extensions + """ + } + } + + then { + assertAll( + { assert process.success }, + { assert snapshot(process.out).match() }, + ) + } + + } + +} diff --git a/modules/nf-core/quarto/notebook/tests/main.nf.test.snap b/modules/nf-core/quarto/notebook/tests/main.nf.test.snap new file mode 100644 index 00000000..8c39ae3a --- /dev/null +++ b/modules/nf-core/quarto/notebook/tests/main.nf.test.snap @@ -0,0 +1,371 @@ +{ + "test notebook - stub - [qmd:r]": { + "content": [ + { + "0": [ + [ + { + "id": "test" + }, + "test.html:md5,d41d8cd98f00b204e9800998ecf8427e" + ] + ], + "1": [ + [ + { + "id": "test" + }, + "quarto_r.qmd:md5,a6a7392a32a4a2f5637fd8245b936031" + ] + ], + "2": [ + [ + { + "id": "test" + }, + "params.yml:md5,d41d8cd98f00b204e9800998ecf8427e" + ] + ], + "3": [ + + ], + "4": [ + + ], + "5": [ + "versions.yml:md5,d41d8cd98f00b204e9800998ecf8427e" + ], + "6": [ + [ + "QUARTO_NOTEBOOK", + "quarto", + "1.7.31" + ] + ], + "7": [ + [ + "QUARTO_NOTEBOOK", + "papermill", + "2.6.0" + ] + ], + "artifacts": [ + + ], + "extensions": [ + + ], + "html": [ + [ + { + "id": "test" + }, + "test.html:md5,d41d8cd98f00b204e9800998ecf8427e" + ] + ], + "notebook": [ + [ + { + "id": "test" + }, + "quarto_r.qmd:md5,a6a7392a32a4a2f5637fd8245b936031" + ] + ], + "params_yaml": [ + [ + { + "id": "test" + }, + "params.yml:md5,d41d8cd98f00b204e9800998ecf8427e" + ] + ], + "versions": [ + "versions.yml:md5,d41d8cd98f00b204e9800998ecf8427e" + ], + "versions_papermill": [ + [ + "QUARTO_NOTEBOOK", + "papermill", + "2.6.0" + ] + ], + "versions_quarto": [ + [ + "QUARTO_NOTEBOOK", + "quarto", + "1.7.31" + ] + ] + } + ], + "timestamp": "2026-08-18T09:19:47.986553", + "meta": { + "nf-test": "0.9.5", + "nextflow": "26.04.6" + } + }, + "test notebook - [qmd:r]": { + "content": [ + { + "versions": [ + "versions.yml:md5,ffbc211e9dcc1252c411bdd2f072b92e" + ], + "versions_papermill": [ + [ + "QUARTO_NOTEBOOK", + "papermill", + "2.6.0" + ] + ], + "versions_quarto": [ + [ + "QUARTO_NOTEBOOK", + "quarto", + "1.7.31" + ] + ] + }, + [ + [ + { + "id": "test" + }, + "artifact.txt:md5,b10a8db164e0754105b7a99be72e3fe5" + ] + ], + [ + [ + { + "id": "test" + }, + "params.yml:md5,6d2ec3cf3e2f15958ab8ab440d651f8d" + ] + ] + ], + "timestamp": "2026-08-18T15:12:25.10271", + "meta": { + "nf-test": "0.9.5", + "nextflow": "26.04.6" + } + }, + "test notebook - parametrized - [qmd:python]": { + "content": [ + { + "versions": [ + "versions.yml:md5,68fd39542815eb5537cdb4b234e2d30d" + ], + "versions_papermill": [ + [ + "QUARTO_NOTEBOOK", + "papermill", + "2.6.0" + ] + ], + "versions_quarto": [ + [ + "QUARTO_NOTEBOOK", + "quarto", + "1.7.31" + ] + ] + }, + [ + [ + { + "id": "test" + }, + "artifact.txt:md5,8ddd8be4b179a529afa5f2ffae4b9858" + ] + ], + [ + [ + { + "id": "test" + }, + "params.yml:md5,4fd353560b2c3366515af2cc4d49957d" + ] + ] + ], + "timestamp": "2026-08-18T15:13:37.473159", + "meta": { + "nf-test": "0.9.5", + "nextflow": "26.04.6" + } + }, + "test notebook - parametrized - [rmd]": { + "content": [ + { + "versions": [ + "versions.yml:md5,ffbc211e9dcc1252c411bdd2f072b92e" + ], + "versions_papermill": [ + [ + "QUARTO_NOTEBOOK", + "papermill", + "2.6.0" + ] + ], + "versions_quarto": [ + [ + "QUARTO_NOTEBOOK", + "quarto", + "1.7.31" + ] + ] + }, + [ + [ + { + "id": "test" + }, + "artifact.txt:md5,b10a8db164e0754105b7a99be72e3fe5" + ] + ], + [ + [ + { + "id": "test" + }, + "params.yml:md5,4fd353560b2c3366515af2cc4d49957d" + ] + ] + ], + "timestamp": "2026-08-18T15:13:53.297883", + "meta": { + "nf-test": "0.9.5", + "nextflow": "26.04.6" + } + }, + "test notebook - parametrized - [ipynb]": { + "content": [ + { + "versions": [ + "versions.yml:md5,68fd39542815eb5537cdb4b234e2d30d" + ], + "versions_papermill": [ + [ + "QUARTO_NOTEBOOK", + "papermill", + "2.6.0" + ] + ], + "versions_quarto": [ + [ + "QUARTO_NOTEBOOK", + "quarto", + "1.7.31" + ] + ] + }, + [ + [ + { + "id": "test" + }, + "artifact.txt:md5,8ddd8be4b179a529afa5f2ffae4b9858" + ] + ], + [ + [ + { + "id": "test" + }, + "params.yml:md5,4fd353560b2c3366515af2cc4d49957d" + ] + ] + ], + "timestamp": "2026-08-18T15:14:10.605053", + "meta": { + "nf-test": "0.9.5", + "nextflow": "26.04.6" + } + }, + "test notebook - [qmd:python]": { + "content": [ + { + "versions": [ + "versions.yml:md5,68fd39542815eb5537cdb4b234e2d30d" + ], + "versions_papermill": [ + [ + "QUARTO_NOTEBOOK", + "papermill", + "2.6.0" + ] + ], + "versions_quarto": [ + [ + "QUARTO_NOTEBOOK", + "quarto", + "1.7.31" + ] + ] + }, + [ + [ + { + "id": "test" + }, + "artifact.txt:md5,8ddd8be4b179a529afa5f2ffae4b9858" + ] + ], + [ + [ + { + "id": "test" + }, + "params.yml:md5,6d2ec3cf3e2f15958ab8ab440d651f8d" + ] + ] + ], + "timestamp": "2026-08-18T15:12:45.306028", + "meta": { + "nf-test": "0.9.5", + "nextflow": "26.04.6" + } + }, + "test notebook - parametrized - [qmd:r]": { + "content": [ + { + "versions": [ + "versions.yml:md5,ffbc211e9dcc1252c411bdd2f072b92e" + ], + "versions_papermill": [ + [ + "QUARTO_NOTEBOOK", + "papermill", + "2.6.0" + ] + ], + "versions_quarto": [ + [ + "QUARTO_NOTEBOOK", + "quarto", + "1.7.31" + ] + ] + }, + [ + [ + { + "id": "test" + }, + "artifact.txt:md5,b10a8db164e0754105b7a99be72e3fe5" + ] + ], + [ + [ + { + "id": "test" + }, + "params.yml:md5,4fd353560b2c3366515af2cc4d49957d" + ] + ] + ], + "timestamp": "2026-08-18T15:13:18.571016", + "meta": { + "nf-test": "0.9.5", + "nextflow": "26.04.6" + } + } +} \ No newline at end of file diff --git a/modules/nf-core/quarto/notebook/tests/nextflow.config b/modules/nf-core/quarto/notebook/tests/nextflow.config new file mode 100644 index 00000000..021909a3 --- /dev/null +++ b/modules/nf-core/quarto/notebook/tests/nextflow.config @@ -0,0 +1,18 @@ +// Initialise `module_args` parameter +params { + module_args = null +} + +process { + withName: QUARTO_NOTEBOOK { + + // Enable per-test parameters + ext.args = { params.module_args ?: '' } + + // Pinned: `task.cpus` is embedded in `params.yml` and included in the + // rendered output, so leaving it unset breaks snapshot reproducibility + // across machines. + cpus = 1 + + } +} diff --git a/mypy.ini b/mypy.ini new file mode 100644 index 00000000..1d3fe5b6 --- /dev/null +++ b/mypy.ini @@ -0,0 +1,17 @@ +[mypy] +# The QC scripts below are vendored from the upstream nf-xenium-processing repo +# and are kept byte-identical to it apart from documented, mechanical +# adaptations (the inlined helper block that replaces the xenium_helpers +# dependency, and the recorded molecule -> transcript rename). They are +# maintained and type-checked upstream and are re-synced wholesale, so they are +# excluded here: annotating them locally would fork them from upstream and turn +# every future re-sync into a manual merge. All other bin/ scripts are still +# type-checked by `make typecheck`. +# +# Re-sync procedure for these files is documented in the port manifests. +exclude = (?x)( + ^bin/image_qc\.py$ + | ^bin/snr_metrics\.py$ + | ^bin/transcript_qc_processing\.py$ + | ^bin/transcript_stream\.py$ + ) diff --git a/nextflow.config b/nextflow.config index 66d98351..d690be31 100644 --- a/nextflow.config +++ b/nextflow.config @@ -114,6 +114,23 @@ params { offtarget_probe_tracking = false // whether to run off-target probe tracking (provide probe_fasta, reference sequences, gene synonyms ) spoqc = false // whether to run spoqc + // image / transcript QC + image_qc_gpus = 1 // GPUs to request for image QC, and the cap passed to the script (0 = CPU only) + roi_image_qc_thresholds_yaml = null // override the bundled image QC threshold config + transcript_qc_thresholds_yaml = null // override the bundled transcript QC threshold config + tile_size = 35 // image QC ROI/tile size in pixels + neg_control_prefix = 'NegControl' // non-gene probe prefix(es) used by the transcript QC noise model + legacy_focus = false // CPU for-loop focus scores instead of the GPU convolution + image_qc_no_snr = false // skip the SNR metrics entirely + image_qc_snr_no_roi_tx_table = false // skip the per-ROI transcript table + image_qc_snr_otsu_max_rois = null // cap the ROIs used for Otsu thresholding + image_qc_snr_no_moran = true // Moran's I is off by default; false opts in + image_qc_save_dapi_maps_tiff = false // write full-resolution DAPI maps as TIFF + image_qc_stream_tiles = true // reduce and drop each tile; false keeps full-resolution planes + image_qc_figures = true // false skips all figure rendering (metrics-only fast run) + image_qc_figure_source_tables = false // write per-figure figures_source/*.csv exports + image_qc_lap_sigma = null // Laplacian-of-Gaussian sigma override + // utility modules csplit_x_bins = 2 // number of tiles along the x axis (total number of bins is product of x_bins * y_bins) csplit_y_bins = 2 // number of tiles along the y axis @@ -284,6 +301,12 @@ profiles { containerOptions = { "--shm-size ${task.memory.toGiga()}g" } queue = { params.cellpose_queue ?: params.gpu_queue ?: null } } + withLabel:process_gpu_qc { + ext.use_gpu = { params.use_gpu } + accelerator = { params.use_gpu ? (params.image_qc_gpus as int) : null } + containerOptions = { "--shm-size ${task.memory.toGiga()}g" } + queue = { params.gpu_queue ?: null } + } } } test { includeConfig 'conf/test.config' } diff --git a/nextflow_schema.json b/nextflow_schema.json index 8ba05233..e25f5c25 100644 --- a/nextflow_schema.json +++ b/nextflow_schema.json @@ -122,11 +122,6 @@ "description": "Options for the segmentation layer of the spatialaxe pipeline", "default": "", "properties": { - "run_qc": { - "type": "boolean", - "description": "Whether to run the qc layer in the pipeline.", - "default": true - }, "offtarget_probe_tracking": { "type": "boolean", "description": "Whether to run the off-target probe tracking." @@ -424,8 +419,90 @@ "description": "", "default": "", "properties": { + "run_qc": { + "type": "boolean", + "description": "Whether to run the qc layer in the pipeline.", + "default": true + }, "spoqc": { - "type": "boolean" + "type": "boolean", + "description": "Run the spoQC subworkflow." + }, + "image_qc_gpus": { + "type": "integer", + "default": 1, + "minimum": 0, + "description": "Number of GPUs to request for image QC, and the device cap passed to the script.", + "help_text": "0 runs image QC on CPU only. The value is both the `accelerator` request and the `--max-gpus` cap, because `accelerator` does not restrict what CUDA can see." + }, + "roi_image_qc_thresholds_yaml": { + "type": "string", + "format": "file-path", + "pattern": "^\\S+\\.(yml|yaml)$", + "description": "Override the bundled image QC threshold config." + }, + "transcript_qc_thresholds_yaml": { + "type": "string", + "format": "file-path", + "pattern": "^\\S+\\.(yml|yaml)$", + "description": "Override the bundled transcript QC threshold config." + }, + "tile_size": { + "type": "integer", + "default": 35, + "minimum": 1, + "description": "Image QC ROI/tile size in pixels." + }, + "neg_control_prefix": { + "type": "string", + "default": "NegControl", + "description": "Prefix identifying non-gene (negative control) probes for the transcript QC noise model." + }, + "legacy_focus": { + "type": "boolean", + "description": "Compute focus scores with the legacy CPU loop instead of the GPU convolution." + }, + "image_qc_no_snr": { + "type": "boolean", + "description": "Skip the SNR metrics entirely." + }, + "image_qc_snr_no_roi_tx_table": { + "type": "boolean", + "description": "Skip the per-ROI transcript table." + }, + "image_qc_snr_otsu_max_rois": { + "type": "integer", + "minimum": 1, + "description": "Cap the number of ROIs used for Otsu thresholding." + }, + "image_qc_snr_no_moran": { + "type": "boolean", + "default": true, + "description": "Disable Moran's I spatial autocorrelation (default). Set false to opt in." + }, + "image_qc_save_dapi_maps_tiff": { + "type": "boolean", + "description": "Write full-resolution DAPI maps as TIFF." + }, + "image_qc_stream_tiles": { + "type": "boolean", + "default": true, + "description": "Reduce and drop each tile instead of writing full-resolution planes.", + "help_text": "Disabling this writes full-resolution pixel planes, which cost ~154 GB of scratch on a 5.5 gigapixel sample." + }, + "image_qc_figures": { + "type": "boolean", + "default": true, + "description": "Render figures. Set false for a metrics-only fast run." + }, + "image_qc_figure_source_tables": { + "type": "boolean", + "description": "Write per-figure source-data CSV exports." + }, + "image_qc_lap_sigma": { + "type": "number", + "minimum": 0, + "description": "Laplacian-of-Gaussian sigma override for focus scoring." } } }, diff --git a/subworkflows/local/image_qc/main.nf b/subworkflows/local/image_qc/main.nf new file mode 100644 index 00000000..4f0f9524 --- /dev/null +++ b/subworkflows/local/image_qc/main.nf @@ -0,0 +1,49 @@ +// +// IMAGE_QC: image-based quality control for a Xenium bundle. +// +// Runs the image QC analysis (focus / SNR / morphology metrics + figures) and +// renders an HTML report from the analysis outputs with the nf-core +// QUARTO_NOTEBOOK module. Versions are reported via the `versions` topic channel +// by each module, so this subworkflow does not thread versions through emit. +// + +include { IMAGE_QC_ANALYSIS as ANALYSIS } from '../../../modules/local/image_qc/main' +include { QUARTO_NOTEBOOK as REPORT } from '../../../modules/nf-core/quarto/notebook/main' + +workflow IMAGE_QC { + take: + ch_input // channel: [ val(meta), val(parameters), path(input_files) ] + ch_thresholds // channel: path(roi_image_qc_thresholds.yaml) + notebook // path: the image QC report .qmd + outdir_label // val: publish subdirectory name, used for the report's self-reference + + main: + ANALYSIS(ch_input, ch_thresholds.first()) + + // Render with QUARTONOTEBOOK. Its four inputs are separate channels paired + // by emission order, so every per-sample channel is derived from one + // upstream channel to guarantee alignment. The analysis output directory and + // the thresholds YAML are staged as input files; the notebook's parameters + // cell receives their staged names through params.yml. + ch_report = ANALYSIS.out.outdir.combine(ch_thresholds.first()).map { meta, analysis_outdir, thresholds -> + def parameters = [ + INDIR : analysis_outdir.name, + SAMPLE_NAME : meta.id, + XENIUM_BUNDLE : meta.samplesheet_xenium_bundle ?: '', + SAMPLE_PUBLISHED_OUTDIR: outdir_label, + ROI_THRESHOLDS_YAML : thresholds.name, + PIPELINE_VERSION : workflow.manifest.version, + ] + [meta, parameters, [analysis_outdir, thresholds]] + } + REPORT( + ch_report.map { meta, _parameters, _input_files -> [meta, notebook] }, + ch_report.map { _meta, parameters, _input_files -> parameters }, + ch_report.map { _meta, _parameters, input_files -> input_files }, + [], + ) + + emit: + outdir = ANALYSIS.out.outdir // channel: [ val(meta), path(outdir) ] + report = REPORT.out.html // channel: [ val(meta), path(html) ] +} diff --git a/subworkflows/local/qc/main.nf b/subworkflows/local/qc/main.nf new file mode 100644 index 00000000..4e0d9473 --- /dev/null +++ b/subworkflows/local/qc/main.nf @@ -0,0 +1,65 @@ +// +// QC: image and transcript quality control for a Xenium bundle. +// +// Thin wrapper that runs the two QC subworkflows — image QC and transcript QC — +// on one bundle and emits their analysis directories and HTML reports. It +// deliberately holds no pre/post-segmentation logic and no MultiQC: when and on +// which bundle QC runs is decided by the caller (workflows/spatialaxe.nf). +// +// Following the pipeline convention, only the entry workflow reads `params.*`; +// everything this subworkflow needs arrives through `take:`. +// + +include { IMAGE_QC } from '../image_qc/main' +include { TRANSCRIPT_QC } from '../transcript_qc/main' + +workflow QC { + take: + ch_bundle // channel: [ val(meta), path(xenium_bundle) ] + ch_image_thresholds // channel: path(roi_image_qc_thresholds.yaml) + ch_transcript_thresholds // channel: path(transcript_qc_thresholds.yaml) + image_notebook // path: image QC report .qmd + transcript_notebook // path: transcript QC report .qmd + image_outdir_label // val: publish subdirectory for image QC + transcript_outdir_label // val: publish subdirectory for transcript QC + + main: + // Shared per-sample meta for both reports. The bundle path is carried in + // meta so the notebooks can state which bundle they describe. + ch_qc_meta = ch_bundle.map { meta, bundle -> + [ + [ + id: meta.id, + samplesheet_xenium_bundle: bundle.toString(), + xenium_bundle_source: meta.xenium_bundle_source ?: '', + cropped: meta.cropped ?: false, + ], + bundle, + ] + } + + // Per-sample values reach the analysis modules through the `parameters` + // list; module-wide settings come from `task.ext.*` in conf/modules.config, + // so the modules never read `params.*` themselves. + ch_qc_input = ch_qc_meta.map { meta, bundle -> tuple(meta, [], [bundle]) } + + IMAGE_QC( + ch_qc_input, + ch_image_thresholds, + image_notebook, + image_outdir_label, + ) + + TRANSCRIPT_QC( + ch_qc_input, + ch_transcript_thresholds, + transcript_notebook, + transcript_outdir_label, + ) + + emit: + image_qc_outdir = IMAGE_QC.out.outdir // channel: [ val(meta), path(outdir) ] + image_qc_report = IMAGE_QC.out.report // channel: [ val(meta), path(html) ] + transcript_qc_outdir = TRANSCRIPT_QC.out.outdir // channel: [ val(meta), path(outdir) ] + transcript_qc_report = TRANSCRIPT_QC.out.report // channel: [ val(meta), path(html) ] +} diff --git a/subworkflows/local/transcript_qc/main.nf b/subworkflows/local/transcript_qc/main.nf new file mode 100644 index 00000000..6e5d5555 --- /dev/null +++ b/subworkflows/local/transcript_qc/main.nf @@ -0,0 +1,47 @@ +// +// TRANSCRIPT_QC: transcript-level quality control for a Xenium bundle. +// +// Runs the transcript QC analysis (per-transcript, per-cell and per-FoV metrics +// + figures) and renders an HTML report from the analysis outputs with the +// nf-core QUARTO_NOTEBOOK module. Versions are reported via the `versions` topic +// channel by each module. +// + +include { TRANSCRIPT_QC_PROCESSING as ANALYSIS } from '../../../modules/local/transcript_qc/main' +include { QUARTO_NOTEBOOK as REPORT } from '../../../modules/nf-core/quarto/notebook/main' + +workflow TRANSCRIPT_QC { + take: + ch_input // channel: [ val(meta), val(parameters), path(input_files) ] + ch_thresholds // channel: path(transcript_qc_thresholds.yaml) + notebook // path: the transcript QC report .qmd + outdir_label // val: publish subdirectory name, used for the report's self-reference + + main: + ANALYSIS(ch_input) + + // Render with QUARTONOTEBOOK — see the note in subworkflows/local/image_qc + // on why every per-sample channel derives from one upstream channel. + ch_report = ANALYSIS.out.outdir.combine(ch_thresholds.first()).map { meta, analysis_outdir, thresholds -> + def parameters = [ + INDIR : analysis_outdir.name, + SAMPLE_NAME : meta.id, + XENIUM_BUNDLE : meta.samplesheet_xenium_bundle ?: '', + SAMPLE_PUBLISHED_OUTDIR: outdir_label, + // The transcript QC notebook reads the staged thresholds file under + // the same ROI_THRESHOLDS_YAML name the image QC notebook uses. + ROI_THRESHOLDS_YAML : thresholds.name, + ] + [meta, parameters, [analysis_outdir, thresholds]] + } + REPORT( + ch_report.map { meta, _parameters, _input_files -> [meta, notebook] }, + ch_report.map { _meta, parameters, _input_files -> parameters }, + ch_report.map { _meta, _parameters, input_files -> input_files }, + [], + ) + + emit: + outdir = ANALYSIS.out.outdir // channel: [ val(meta), path(outdir) ] + report = REPORT.out.html // channel: [ val(meta), path(html) ] +} diff --git a/tests/.nftignore b/tests/.nftignore index ace7f33c..dca04467 100644 --- a/tests/.nftignore +++ b/tests/.nftignore @@ -87,3 +87,11 @@ qc/spatialdata/write/spatialdata/test_run/raw_bundle/points/** qc/spatialdata/write/spatialdata/test_run/raw_bundle/shapes/** qc/spatialdata/write/spatialdata/test_run/raw_bundle/tables/** qc/spatialdata/write/spatialdata/test_run/raw_bundle/zarr.json + +# Image QC / transcript QC reports embed a render timestamp, and their +# matplotlib figure PDFs embed a CreationDate, so their content is not +# run-to-run reproducible; names and existence are still asserted. +**/qc/image_qc/image_qc.html +**/qc/image_qc/figures/*.pdf +**/qc/transcript_qc/transcript_qc.html +**/qc/transcript_qc/figures/*.pdf diff --git a/tests/coordinate_mode.nf.test.snap b/tests/coordinate_mode.nf.test.snap index 5b9979b4..f59ed087 100644 --- a/tests/coordinate_mode.nf.test.snap +++ b/tests/coordinate_mode.nf.test.snap @@ -325,7 +325,7 @@ "zarr.json:md5,b98263d3964a5236145ccf6c808f8919" ] ], - "timestamp": "2026-08-14T15:24:34.837557935", + "timestamp": "2026-08-27T14:59:26.2268814", "meta": { "nf-test": "0.9.5", "nextflow": "26.04.6" @@ -334,12 +334,18 @@ "-profile test stub": { "content": [ { + "ANALYSIS": { + "seaborn": "0.13.2" + }, "PROSEG2BAYSOR": { "proseg": "3.1.0" }, "PROSEG": { "proseg": "3.1.0" }, + "REPORT": { + "quarto": "1.6.42" + }, "SPATIALDATA_MERGE_RAW_REDEFINED": { "spatialdata": "0.7.2" }, @@ -387,6 +393,16 @@ "coordinate/proseg/proseg2baysor/test_run", "coordinate/proseg/proseg2baysor/test_run/cell-polygons.geojson", "coordinate/proseg/proseg2baysor/test_run/transcript-metadata.csv", + "coordinate/qc", + "coordinate/qc/image_qc", + "coordinate/qc/image_qc/figures", + "coordinate/qc/image_qc/image_qc.html", + "coordinate/qc/image_qc/image_qc_metrics.csv", + "coordinate/qc/image_qc/image_qc_metrics.json", + "coordinate/qc/transcript_qc", + "coordinate/qc/transcript_qc/figures", + "coordinate/qc/transcript_qc/transcript_qc.html", + "coordinate/qc/transcript_qc/transcript_qc_metrics.json", "coordinate/spatialdata", "coordinate/spatialdata/merge", "coordinate/spatialdata/merge/spatialdata", @@ -443,13 +459,16 @@ ".stub:md5,d41d8cd98f00b204e9800998ecf8427e", ".stub:md5,d41d8cd98f00b204e9800998ecf8427e", "multiqc_report.html:md5,d41d8cd98f00b204e9800998ecf8427e", + "image_qc_metrics.csv:md5,d41d8cd98f00b204e9800998ecf8427e", + "image_qc_metrics.json:md5,d41d8cd98f00b204e9800998ecf8427e", + "transcript_qc_metrics.json:md5,d41d8cd98f00b204e9800998ecf8427e", "fake_file.txt:md5,d41d8cd98f00b204e9800998ecf8427e", "fake_file.txt:md5,d41d8cd98f00b204e9800998ecf8427e", "fake_file.txt:md5,d41d8cd98f00b204e9800998ecf8427e", "fake_file.txt:md5,d41d8cd98f00b204e9800998ecf8427e" ] ], - "timestamp": "2026-08-14T15:28:47.60775781", + "timestamp": "2026-08-27T15:03:17.494829997", "meta": { "nf-test": "0.9.5", "nextflow": "26.04.6" @@ -458,12 +477,18 @@ "-profile test qc": { "content": [ { + "ANALYSIS": { + "seaborn": "0.13.2" + }, "PROSEG2BAYSOR": { "proseg": "3.1.0" }, "PROSEG": { "proseg": "3.1.0" }, + "REPORT": { + "quarto": "1.6.42" + }, "SPATIALDATA_MERGE_RAW_REDEFINED": { "spatialdata": "0.7.2" }, @@ -515,6 +540,91 @@ "coordinate/proseg/proseg2baysor/test_run", "coordinate/proseg/proseg2baysor/test_run/cell-polygons.geojson", "coordinate/proseg/proseg2baysor/test_run/transcript-metadata.csv", + "coordinate/qc", + "coordinate/qc/image_qc", + "coordinate/qc/image_qc/SNR_roi_tx.parquet", + "coordinate/qc/image_qc/dense_intensity_regions_summary.csv", + "coordinate/qc/image_qc/figures", + "coordinate/qc/image_qc/figures/ccfs_thresholded.png", + "coordinate/qc/image_qc/figures/cell_flagged_maps.png", + "coordinate/qc/image_qc/figures/dapi_intensity_vs_transcripts_log.png", + "coordinate/qc/image_qc/figures/distance_map_edge.pdf", + "coordinate/qc/image_qc/figures/distance_map_edge.png", + "coordinate/qc/image_qc/figures/distance_map_holes.pdf", + "coordinate/qc/image_qc/figures/distance_map_holes.png", + "coordinate/qc/image_qc/figures/distance_maps.pdf", + "coordinate/qc/image_qc/figures/distance_maps.png", + "coordinate/qc/image_qc/figures/figures_methodology_assessment", + "coordinate/qc/image_qc/figures/gmm_focus_vs_transcripts.png", + "coordinate/qc/image_qc/figures/grid_roi_focus_heatmap.png", + "coordinate/qc/image_qc/figures/imageqc_masks.pdf", + "coordinate/qc/image_qc/figures/imageqc_masks.png", + "coordinate/qc/image_qc/figures/intensity_assessment.pdf", + "coordinate/qc/image_qc/figures/intensity_assessment.png", + "coordinate/qc/image_qc/figures/nuclear_texture_vs_transcripts_log.png", + "coordinate/qc/image_qc/figures/roi_focus_vs_intensity.pdf", + "coordinate/qc/image_qc/figures/roi_focus_vs_intensity.png", + "coordinate/qc/image_qc/figures/tile_blur_proportions_roi.png", + "coordinate/qc/image_qc/figures/tile_focus_gmm_spatial.png", + "coordinate/qc/image_qc/figures_cell_centred", + "coordinate/qc/image_qc/figures_cell_centred/cell_flagged_maps.png", + "coordinate/qc/image_qc/figures_cell_centred/dapi_intensity_vs_transcripts_log.png", + "coordinate/qc/image_qc/figures_cell_centred/gmm_focus_vs_transcripts.png", + "coordinate/qc/image_qc/figures_cell_centred/tile_blur_proportions_roi.png", + "coordinate/qc/image_qc/figures_cell_centred/tile_focus_gmm_spatial.png", + "coordinate/qc/image_qc/grid_roi_focus_scores.csv", + "coordinate/qc/image_qc/image_qc.html", + "coordinate/qc/image_qc/image_qc_cell_metrics.csv", + "coordinate/qc/image_qc/image_qc_cell_metrics.json", + "coordinate/qc/image_qc/image_qc_metrics.csv", + "coordinate/qc/image_qc/image_qc_metrics.json", + "coordinate/qc/image_qc/image_qc_status.json", + "coordinate/qc/image_qc/intensity_assessment.json", + "coordinate/qc/image_qc/roi_blur_threshold.json", + "coordinate/qc/image_qc/roi_count.txt", + "coordinate/qc/image_qc/roi_qc_metrics.json", + "coordinate/qc/image_qc/roi_size.txt", + "coordinate/qc/image_qc/snr_metrics.json", + "coordinate/qc/image_qc/versions.yml", + "coordinate/qc/transcript_qc", + "coordinate/qc/transcript_qc/df_spatial_quality.csv", + "coordinate/qc/transcript_qc/figures", + "coordinate/qc/transcript_qc/figures/cell_size_distribution.pdf", + "coordinate/qc/transcript_qc/figures/cell_size_distribution.png", + "coordinate/qc/transcript_qc/figures/cell_size_distribution_log.pdf", + "coordinate/qc/transcript_qc/figures/cell_size_distribution_log.png", + "coordinate/qc/transcript_qc/figures/nucleus_to_cell_size_fraction_per_cell_distribution.pdf", + "coordinate/qc/transcript_qc/figures/nucleus_to_cell_size_fraction_per_cell_distribution.png", + "coordinate/qc/transcript_qc/figures/nucleus_transcript_fraction_per_cell_distribution.pdf", + "coordinate/qc/transcript_qc/figures/nucleus_transcript_fraction_per_cell_distribution.png", + "coordinate/qc/transcript_qc/figures/num_genes_per_cell.pdf", + "coordinate/qc/transcript_qc/figures/num_genes_per_cell.png", + "coordinate/qc/transcript_qc/figures/num_transcripts_per_cell.pdf", + "coordinate/qc/transcript_qc/figures/num_transcripts_per_cell.png", + "coordinate/qc/transcript_qc/figures/num_transcripts_per_feature.pdf", + "coordinate/qc/transcript_qc/figures/num_transcripts_per_feature.png", + "coordinate/qc/transcript_qc/figures/quality_by_fov.pdf", + "coordinate/qc/transcript_qc/figures/quality_by_fov.png", + "coordinate/qc/transcript_qc/figures/quality_distributions_comprehensive.pdf", + "coordinate/qc/transcript_qc/figures/quality_distributions_comprehensive.png", + "coordinate/qc/transcript_qc/figures/unassigned_per_gene.pdf", + "coordinate/qc/transcript_qc/figures/unassigned_per_gene.png", + "coordinate/qc/transcript_qc/figures_source", + "coordinate/qc/transcript_qc/figures_source/cell_size_distribution.csv", + "coordinate/qc/transcript_qc/figures_source/nucleus_to_cell_size_fraction_per_cell_distribution.csv", + "coordinate/qc/transcript_qc/figures_source/nucleus_transcript_fraction_per_cell_distribution.csv", + "coordinate/qc/transcript_qc/figures_source/num_genes_per_cell.csv", + "coordinate/qc/transcript_qc/figures_source/num_transcripts_per_cell.csv", + "coordinate/qc/transcript_qc/figures_source/num_transcripts_per_feature.csv", + "coordinate/qc/transcript_qc/figures_source/quality_by_fov.csv", + "coordinate/qc/transcript_qc/figures_source/quality_distributions_comprehensive.csv", + "coordinate/qc/transcript_qc/figures_source/unassigned_per_gene.csv", + "coordinate/qc/transcript_qc/num_genes_per_cell.csv", + "coordinate/qc/transcript_qc/num_transcripts_per_cell.csv", + "coordinate/qc/transcript_qc/retained_genes.csv", + "coordinate/qc/transcript_qc/transcript_qc.html", + "coordinate/qc/transcript_qc/transcript_qc_metrics.json", + "coordinate/qc/transcript_qc/versions.yml", "coordinate/spatialdata", "coordinate/spatialdata/merge", "coordinate/spatialdata/merge/spatialdata", @@ -922,9 +1032,68 @@ ], [ "multiqc_citations.txt:md5,4c806e63a283ec1b7e78cdae3a923d4f", - "multiqc_software_versions.txt:md5,6b25b317ed5e9bda0d2c9ff6fe85eafb", + "multiqc_software_versions.txt:md5,19333911e0128339d0cb1415a6c5e539", "multiqc_citations.txt:md5,4c806e63a283ec1b7e78cdae3a923d4f", - "multiqc_software_versions.txt:md5,6b25b317ed5e9bda0d2c9ff6fe85eafb", + "multiqc_software_versions.txt:md5,19333911e0128339d0cb1415a6c5e539", + "SNR_roi_tx.parquet:md5,7636c8a75be03c1aacc1662225f1862d", + "dense_intensity_regions_summary.csv:md5,cae928f6aa311af4b0e1bfc17fbd092e", + "ccfs_thresholded.png:md5,032f6b10b6878f403efbbe01af8b7110", + "cell_flagged_maps.png:md5,0ba882744ff82f34a1a6bbc83f976c8f", + "dapi_intensity_vs_transcripts_log.png:md5,d8fdf16b335d6b8b94d8a6267e0b033a", + "distance_map_edge.png:md5,1fd0072b0f726b8f2cab8488ced54ec6", + "distance_map_holes.png:md5,bec4e0e137dd6c6054f89f813628f392", + "distance_maps.png:md5,642f233c2f4069ff44b951dec56d407d", + "gmm_focus_vs_transcripts.png:md5,3a5342221bf9d12076b8cba0ebdf8260", + "grid_roi_focus_heatmap.png:md5,fe63ecb24348e4b8c51a8abce368d39e", + "imageqc_masks.png:md5,406baa0d4058345872d4911669c633f5", + "intensity_assessment.png:md5,d2edac7a810058fe776f4ff3768a09fc", + "nuclear_texture_vs_transcripts_log.png:md5,b0a8bfb561526af53b34033d43a18bab", + "roi_focus_vs_intensity.png:md5,7cd8798673ac7e95d69f7bdd33dc8f52", + "tile_blur_proportions_roi.png:md5,7f36d91b1938443d796aa692480a515c", + "tile_focus_gmm_spatial.png:md5,53a6bfe4a7c73750ca8a85438b1d2517", + "cell_flagged_maps.png:md5,0ba882744ff82f34a1a6bbc83f976c8f", + "dapi_intensity_vs_transcripts_log.png:md5,d8fdf16b335d6b8b94d8a6267e0b033a", + "gmm_focus_vs_transcripts.png:md5,3a5342221bf9d12076b8cba0ebdf8260", + "tile_blur_proportions_roi.png:md5,8d2c5b6aafcc1666b27d09075dd191c6", + "tile_focus_gmm_spatial.png:md5,53a6bfe4a7c73750ca8a85438b1d2517", + "grid_roi_focus_scores.csv:md5,b2cfd1da35fd51bd43cc3f44837a931c", + "image_qc_cell_metrics.csv:md5,f9b6d14445af2f3ed9e9fadcc0f8075a", + "image_qc_cell_metrics.json:md5,c970ee84992e3f38f24b4cd4b7449992", + "image_qc_metrics.csv:md5,f9b6d14445af2f3ed9e9fadcc0f8075a", + "image_qc_metrics.json:md5,69f2ae0bdf4603c3dd8aecf9dcb2912a", + "image_qc_status.json:md5,14f825da7a7d6f84385cdcf179ab1f98", + "intensity_assessment.json:md5,ca3a3ab1a25d10d1726707ee92b0a381", + "roi_blur_threshold.json:md5,1f3640c94ebb2087b5a650cae7f352a4", + "roi_count.txt:md5,c881617af0ae19b38dd0546548cd53ad", + "roi_qc_metrics.json:md5,be710c04ad674d86641685aed4517812", + "roi_size.txt:md5,649ee93d50739c656e94ec88a32c7ffe", + "snr_metrics.json:md5,a2037280229900d1d049998fa663980f", + "versions.yml:md5,e1d8ab29fd14c338cd8a7b67c8c44c9e", + "df_spatial_quality.csv:md5,3dcb5878080b65eb80e56c81c0e469a4", + "cell_size_distribution.png:md5,9f3f93c8a0d69d72238b6feaa5d77f5d", + "cell_size_distribution_log.png:md5,f663e879592bafe3b526b87536a695e4", + "nucleus_to_cell_size_fraction_per_cell_distribution.png:md5,214fd9b389a9b2a29a03ef322e93eb4d", + "nucleus_transcript_fraction_per_cell_distribution.png:md5,406c6ec940050080d83cf769205d0756", + "num_genes_per_cell.png:md5,c9fe04ed2c11dd1fbb3255dd258f7fea", + "num_transcripts_per_cell.png:md5,c555e864011cdf4a44e5636b4e6411dd", + "num_transcripts_per_feature.png:md5,1aeaae3be7287677b916d07286aa12b0", + "quality_by_fov.png:md5,2878179b8869887f57f70fb2d6a81790", + "quality_distributions_comprehensive.png:md5,2f0f5f5c1f791e39987df00c110eb5ce", + "unassigned_per_gene.png:md5,373320b0b7a93f78e908a9e817d9e611", + "cell_size_distribution.csv:md5,6c8aee45e7c0045918502fe4d4d441e0", + "nucleus_to_cell_size_fraction_per_cell_distribution.csv:md5,f753929ec13d1e3dfaf94576bbd7d0d5", + "nucleus_transcript_fraction_per_cell_distribution.csv:md5,9c92b67705b28fe0ae017e3bf9a2dcdb", + "num_genes_per_cell.csv:md5,1dc6c4f75a58c595163d192182f73d92", + "num_transcripts_per_cell.csv:md5,e7378613f3d9966c3d25e449df4d690a", + "num_transcripts_per_feature.csv:md5,ce592e962496b30c4c7bc0a4a4b83ccf", + "quality_by_fov.csv:md5,04b3fd532a8c12dc29e9c084a6c891f8", + "quality_distributions_comprehensive.csv:md5,3dcb5878080b65eb80e56c81c0e469a4", + "unassigned_per_gene.csv:md5,d000d944fd21c9417a1f2923d07ac37f", + "num_genes_per_cell.csv:md5,dda1841ad514e161fc4c6afad7816c99", + "num_transcripts_per_cell.csv:md5,7babff329ac62e8c2d7d95ee4b2168a4", + "retained_genes.csv:md5,9d531795ce1d22cd60c0f70ccdec0ac6", + "transcript_qc_metrics.json:md5,c031bdf67c7f36a03523cd105902840b", + "versions.yml:md5,e027a59df4703317fe257afdafa30397", "0:md5,397dc21e479c58fd2ffcbe6c3e627683", "1:md5,f1a6b9238c3095bad7867b22f293e5e6", "zarr.json:md5,893b4d8873fc997d364288c63ef0871f", @@ -1047,7 +1216,7 @@ "zarr.json:md5,b98263d3964a5236145ccf6c808f8919" ] ], - "timestamp": "2026-08-14T15:28:21.376352619", + "timestamp": "2026-08-27T15:02:49.795786082", "meta": { "nf-test": "0.9.5", "nextflow": "26.04.6" diff --git a/tests/default.nf.test.snap b/tests/default.nf.test.snap index b24374ad..ed93dac2 100644 --- a/tests/default.nf.test.snap +++ b/tests/default.nf.test.snap @@ -2,12 +2,18 @@ "-profile test stub": { "content": [ { + "ANALYSIS": { + "seaborn": "0.13.2" + }, "PROSEG2BAYSOR": { "proseg": "3.1.0" }, "PROSEG": { "proseg": "3.1.0" }, + "REPORT": { + "quarto": "1.6.42" + }, "SPATIALDATA_MERGE_RAW_REDEFINED": { "spatialdata": "0.7.2" }, @@ -55,6 +61,16 @@ "coordinate/proseg/proseg2baysor/test_run", "coordinate/proseg/proseg2baysor/test_run/cell-polygons.geojson", "coordinate/proseg/proseg2baysor/test_run/transcript-metadata.csv", + "coordinate/qc", + "coordinate/qc/image_qc", + "coordinate/qc/image_qc/figures", + "coordinate/qc/image_qc/image_qc.html", + "coordinate/qc/image_qc/image_qc_metrics.csv", + "coordinate/qc/image_qc/image_qc_metrics.json", + "coordinate/qc/transcript_qc", + "coordinate/qc/transcript_qc/figures", + "coordinate/qc/transcript_qc/transcript_qc.html", + "coordinate/qc/transcript_qc/transcript_qc_metrics.json", "coordinate/spatialdata", "coordinate/spatialdata/merge", "coordinate/spatialdata/merge/spatialdata", @@ -111,13 +127,16 @@ ".stub:md5,d41d8cd98f00b204e9800998ecf8427e", ".stub:md5,d41d8cd98f00b204e9800998ecf8427e", "multiqc_report.html:md5,d41d8cd98f00b204e9800998ecf8427e", + "image_qc_metrics.csv:md5,d41d8cd98f00b204e9800998ecf8427e", + "image_qc_metrics.json:md5,d41d8cd98f00b204e9800998ecf8427e", + "transcript_qc_metrics.json:md5,d41d8cd98f00b204e9800998ecf8427e", "fake_file.txt:md5,d41d8cd98f00b204e9800998ecf8427e", "fake_file.txt:md5,d41d8cd98f00b204e9800998ecf8427e", "fake_file.txt:md5,d41d8cd98f00b204e9800998ecf8427e", "fake_file.txt:md5,d41d8cd98f00b204e9800998ecf8427e" ] ], - "timestamp": "2026-08-14T11:35:32.776028015", + "timestamp": "2026-08-21T19:48:17.283517278", "meta": { "nf-test": "0.9.5", "nextflow": "26.04.6" diff --git a/tests/image_mode.nf.test.snap b/tests/image_mode.nf.test.snap index 0c9770d6..f1c51d36 100644 --- a/tests/image_mode.nf.test.snap +++ b/tests/image_mode.nf.test.snap @@ -2,6 +2,9 @@ "-profile test stub": { "content": [ { + "ANALYSIS": { + "seaborn": "0.13.2" + }, "BAYSOR_PREPROCESS_TRANSCRIPTS": { "python": "3.14.4" }, @@ -11,6 +14,9 @@ "CELLPOSE_CELLS": { "torch": "2.10.0+cu128" }, + "REPORT": { + "quarto": "1.6.42" + }, "RESIZE_TIF": { "tifffile": "2026.2.24" }, @@ -62,6 +68,16 @@ "image/multiqc/redefined_bundle/multiqc_plots", "image/multiqc/redefined_bundle/multiqc_plots/.stub", "image/multiqc/redefined_bundle/multiqc_report.html", + "image/qc", + "image/qc/image_qc", + "image/qc/image_qc/figures", + "image/qc/image_qc/image_qc.html", + "image/qc/image_qc/image_qc_metrics.csv", + "image/qc/image_qc/image_qc_metrics.json", + "image/qc/transcript_qc", + "image/qc/transcript_qc/figures", + "image/qc/transcript_qc/transcript_qc.html", + "image/qc/transcript_qc/transcript_qc_metrics.json", "image/spatialdata", "image/spatialdata/merge", "image/spatialdata/merge/spatialdata", @@ -126,6 +142,9 @@ ".stub:md5,d41d8cd98f00b204e9800998ecf8427e", ".stub:md5,d41d8cd98f00b204e9800998ecf8427e", "multiqc_report.html:md5,d41d8cd98f00b204e9800998ecf8427e", + "image_qc_metrics.csv:md5,d41d8cd98f00b204e9800998ecf8427e", + "image_qc_metrics.json:md5,d41d8cd98f00b204e9800998ecf8427e", + "transcript_qc_metrics.json:md5,d41d8cd98f00b204e9800998ecf8427e", "fake_file.txt:md5,d41d8cd98f00b204e9800998ecf8427e", "fake_file.txt:md5,d41d8cd98f00b204e9800998ecf8427e", "fake_file.txt:md5,d41d8cd98f00b204e9800998ecf8427e", @@ -155,7 +174,7 @@ "resized_morphology_focus_0000.ome_cp_masks.tif.tif:md5,d41d8cd98f00b204e9800998ecf8427e" ] ], - "timestamp": "2026-08-14T11:37:37.301529638", + "timestamp": "2026-08-21T19:48:51.059737492", "meta": { "nf-test": "0.9.5", "nextflow": "26.04.6" diff --git a/tests/preview_mode.nf.test.snap b/tests/preview_mode.nf.test.snap index 60f47591..7159e830 100644 --- a/tests/preview_mode.nf.test.snap +++ b/tests/preview_mode.nf.test.snap @@ -2,6 +2,9 @@ "-profile test stub": { "content": [ { + "ANALYSIS": { + "seaborn": "0.13.2" + }, "BAYSOR_CREATE_DATASET": { "baysor": "0.7.1" }, @@ -14,6 +17,9 @@ "PARQUET_TO_CSV": { "pyarrow": "24.0.0" }, + "REPORT": { + "quarto": "1.6.42" + }, "UNTAR": { "untar": 1.34 }, @@ -45,6 +51,16 @@ "preview/multiqc/redefined_bundle/multiqc_plots", "preview/multiqc/redefined_bundle/multiqc_plots/.stub", "preview/multiqc/redefined_bundle/multiqc_report.html", + "preview/qc", + "preview/qc/image_qc", + "preview/qc/image_qc/figures", + "preview/qc/image_qc/image_qc.html", + "preview/qc/image_qc/image_qc_metrics.csv", + "preview/qc/image_qc/image_qc_metrics.json", + "preview/qc/transcript_qc", + "preview/qc/transcript_qc/figures", + "preview/qc/transcript_qc/transcript_qc.html", + "preview/qc/transcript_qc/transcript_qc_metrics.json", "preview/untar", "preview/untar/test_run", "preview/untar/test_run/.end-of-run", @@ -91,6 +107,9 @@ ".stub:md5,d41d8cd98f00b204e9800998ecf8427e", ".stub:md5,d41d8cd98f00b204e9800998ecf8427e", "multiqc_report.html:md5,d41d8cd98f00b204e9800998ecf8427e", + "image_qc_metrics.csv:md5,d41d8cd98f00b204e9800998ecf8427e", + "image_qc_metrics.json:md5,d41d8cd98f00b204e9800998ecf8427e", + "transcript_qc_metrics.json:md5,d41d8cd98f00b204e9800998ecf8427e", ".end-of-run:md5,d41d8cd98f00b204e9800998ecf8427e", "analysis.tar.gz:md5,d41d8cd98f00b204e9800998ecf8427e", "analysis.zarr.zip:md5,d41d8cd98f00b204e9800998ecf8427e", @@ -121,7 +140,7 @@ "umap_mqc.tsv:md5,d41d8cd98f00b204e9800998ecf8427e" ] ], - "timestamp": "2026-08-14T11:36:14.951666276", + "timestamp": "2026-08-21T19:49:15.173411089", "meta": { "nf-test": "0.9.5", "nextflow": "26.04.6" diff --git a/tests/segfree_mode.nf.test.snap b/tests/segfree_mode.nf.test.snap index 02026e9e..c9f64ae8 100644 --- a/tests/segfree_mode.nf.test.snap +++ b/tests/segfree_mode.nf.test.snap @@ -2,12 +2,18 @@ "-profile test stub": { "content": [ { + "ANALYSIS": { + "seaborn": "0.13.2" + }, "BAYSOR_PREPROCESS_TRANSCRIPTS": { "python": "3.14.4" }, "BAYSOR_SEGFREE": { "baysor": "0.7.1" }, + "REPORT": { + "quarto": "1.6.42" + }, "UNTAR": { "untar": 1.34 }, @@ -39,6 +45,16 @@ "segfree/multiqc/redefined_bundle/multiqc_plots", "segfree/multiqc/redefined_bundle/multiqc_plots/.stub", "segfree/multiqc/redefined_bundle/multiqc_report.html", + "segfree/qc", + "segfree/qc/image_qc", + "segfree/qc/image_qc/figures", + "segfree/qc/image_qc/image_qc.html", + "segfree/qc/image_qc/image_qc_metrics.csv", + "segfree/qc/image_qc/image_qc_metrics.json", + "segfree/qc/transcript_qc", + "segfree/qc/transcript_qc/figures", + "segfree/qc/transcript_qc/transcript_qc.html", + "segfree/qc/transcript_qc/transcript_qc_metrics.json", "segfree/untar", "segfree/untar/test_run", "segfree/untar/test_run/.end-of-run", @@ -74,6 +90,9 @@ ".stub:md5,d41d8cd98f00b204e9800998ecf8427e", ".stub:md5,d41d8cd98f00b204e9800998ecf8427e", "multiqc_report.html:md5,d41d8cd98f00b204e9800998ecf8427e", + "image_qc_metrics.csv:md5,d41d8cd98f00b204e9800998ecf8427e", + "image_qc_metrics.json:md5,d41d8cd98f00b204e9800998ecf8427e", + "transcript_qc_metrics.json:md5,d41d8cd98f00b204e9800998ecf8427e", ".end-of-run:md5,d41d8cd98f00b204e9800998ecf8427e", "analysis.tar.gz:md5,d41d8cd98f00b204e9800998ecf8427e", "analysis.zarr.zip:md5,d41d8cd98f00b204e9800998ecf8427e", @@ -98,7 +117,7 @@ "transcripts.zarr.zip:md5,d41d8cd98f00b204e9800998ecf8427e" ] ], - "timestamp": "2026-08-14T11:36:49.488134053", + "timestamp": "2026-08-21T19:49:39.057225447", "meta": { "nf-test": "0.9.5", "nextflow": "26.04.6" diff --git a/tests/spoqc.nf.test.snap b/tests/spoqc.nf.test.snap index c6aea8b7..5cb637a9 100644 --- a/tests/spoqc.nf.test.snap +++ b/tests/spoqc.nf.test.snap @@ -2,6 +2,12 @@ "-profile test": { "content": [ { + "ANALYSIS": { + "seaborn": "0.13.2" + }, + "REPORT": { + "quarto": "1.6.42" + }, "SPATIALDATA_WRITE_RAW_BUNDLE": { "spatialdata": "0.7.2" }, @@ -139,6 +145,91 @@ "qc/multiqc/redefined_bundle/MultiQC-Post-Xeniumranger-import-segmentation-Run_multiqc_report_data/xenium_segmentation.txt", "qc/multiqc/redefined_bundle/MultiQC-Post-Xeniumranger-import-segmentation-Run_multiqc_report_data/xenium_transcript_quality_per_sample_table.txt", "qc/multiqc/redefined_bundle/MultiQC-Post-Xeniumranger-import-segmentation-Run_multiqc_report_data/xenium_transcripts_per_gene.txt", + "qc/qc", + "qc/qc/image_qc", + "qc/qc/image_qc/SNR_roi_tx.parquet", + "qc/qc/image_qc/dense_intensity_regions_summary.csv", + "qc/qc/image_qc/figures", + "qc/qc/image_qc/figures/ccfs_thresholded.png", + "qc/qc/image_qc/figures/cell_flagged_maps.png", + "qc/qc/image_qc/figures/dapi_intensity_vs_transcripts_log.png", + "qc/qc/image_qc/figures/distance_map_edge.pdf", + "qc/qc/image_qc/figures/distance_map_edge.png", + "qc/qc/image_qc/figures/distance_map_holes.pdf", + "qc/qc/image_qc/figures/distance_map_holes.png", + "qc/qc/image_qc/figures/distance_maps.pdf", + "qc/qc/image_qc/figures/distance_maps.png", + "qc/qc/image_qc/figures/figures_methodology_assessment", + "qc/qc/image_qc/figures/gmm_focus_vs_transcripts.png", + "qc/qc/image_qc/figures/grid_roi_focus_heatmap.png", + "qc/qc/image_qc/figures/imageqc_masks.pdf", + "qc/qc/image_qc/figures/imageqc_masks.png", + "qc/qc/image_qc/figures/intensity_assessment.pdf", + "qc/qc/image_qc/figures/intensity_assessment.png", + "qc/qc/image_qc/figures/nuclear_texture_vs_transcripts_log.png", + "qc/qc/image_qc/figures/roi_focus_vs_intensity.pdf", + "qc/qc/image_qc/figures/roi_focus_vs_intensity.png", + "qc/qc/image_qc/figures/tile_blur_proportions_roi.png", + "qc/qc/image_qc/figures/tile_focus_gmm_spatial.png", + "qc/qc/image_qc/figures_cell_centred", + "qc/qc/image_qc/figures_cell_centred/cell_flagged_maps.png", + "qc/qc/image_qc/figures_cell_centred/dapi_intensity_vs_transcripts_log.png", + "qc/qc/image_qc/figures_cell_centred/gmm_focus_vs_transcripts.png", + "qc/qc/image_qc/figures_cell_centred/tile_blur_proportions_roi.png", + "qc/qc/image_qc/figures_cell_centred/tile_focus_gmm_spatial.png", + "qc/qc/image_qc/grid_roi_focus_scores.csv", + "qc/qc/image_qc/image_qc.html", + "qc/qc/image_qc/image_qc_cell_metrics.csv", + "qc/qc/image_qc/image_qc_cell_metrics.json", + "qc/qc/image_qc/image_qc_metrics.csv", + "qc/qc/image_qc/image_qc_metrics.json", + "qc/qc/image_qc/image_qc_status.json", + "qc/qc/image_qc/intensity_assessment.json", + "qc/qc/image_qc/roi_blur_threshold.json", + "qc/qc/image_qc/roi_count.txt", + "qc/qc/image_qc/roi_qc_metrics.json", + "qc/qc/image_qc/roi_size.txt", + "qc/qc/image_qc/snr_metrics.json", + "qc/qc/image_qc/versions.yml", + "qc/qc/transcript_qc", + "qc/qc/transcript_qc/df_spatial_quality.csv", + "qc/qc/transcript_qc/figures", + "qc/qc/transcript_qc/figures/cell_size_distribution.pdf", + "qc/qc/transcript_qc/figures/cell_size_distribution.png", + "qc/qc/transcript_qc/figures/cell_size_distribution_log.pdf", + "qc/qc/transcript_qc/figures/cell_size_distribution_log.png", + "qc/qc/transcript_qc/figures/nucleus_to_cell_size_fraction_per_cell_distribution.pdf", + "qc/qc/transcript_qc/figures/nucleus_to_cell_size_fraction_per_cell_distribution.png", + "qc/qc/transcript_qc/figures/nucleus_transcript_fraction_per_cell_distribution.pdf", + "qc/qc/transcript_qc/figures/nucleus_transcript_fraction_per_cell_distribution.png", + "qc/qc/transcript_qc/figures/num_genes_per_cell.pdf", + "qc/qc/transcript_qc/figures/num_genes_per_cell.png", + "qc/qc/transcript_qc/figures/num_transcripts_per_cell.pdf", + "qc/qc/transcript_qc/figures/num_transcripts_per_cell.png", + "qc/qc/transcript_qc/figures/num_transcripts_per_feature.pdf", + "qc/qc/transcript_qc/figures/num_transcripts_per_feature.png", + "qc/qc/transcript_qc/figures/quality_by_fov.pdf", + "qc/qc/transcript_qc/figures/quality_by_fov.png", + "qc/qc/transcript_qc/figures/quality_distributions_comprehensive.pdf", + "qc/qc/transcript_qc/figures/quality_distributions_comprehensive.png", + "qc/qc/transcript_qc/figures/unassigned_per_gene.pdf", + "qc/qc/transcript_qc/figures/unassigned_per_gene.png", + "qc/qc/transcript_qc/figures_source", + "qc/qc/transcript_qc/figures_source/cell_size_distribution.csv", + "qc/qc/transcript_qc/figures_source/nucleus_to_cell_size_fraction_per_cell_distribution.csv", + "qc/qc/transcript_qc/figures_source/nucleus_transcript_fraction_per_cell_distribution.csv", + "qc/qc/transcript_qc/figures_source/num_genes_per_cell.csv", + "qc/qc/transcript_qc/figures_source/num_transcripts_per_cell.csv", + "qc/qc/transcript_qc/figures_source/num_transcripts_per_feature.csv", + "qc/qc/transcript_qc/figures_source/quality_by_fov.csv", + "qc/qc/transcript_qc/figures_source/quality_distributions_comprehensive.csv", + "qc/qc/transcript_qc/figures_source/unassigned_per_gene.csv", + "qc/qc/transcript_qc/num_genes_per_cell.csv", + "qc/qc/transcript_qc/num_transcripts_per_cell.csv", + "qc/qc/transcript_qc/retained_genes.csv", + "qc/qc/transcript_qc/transcript_qc.html", + "qc/qc/transcript_qc/transcript_qc_metrics.json", + "qc/qc/transcript_qc/versions.yml", "qc/spatialdata", "qc/spatialdata/write", "qc/spatialdata/write/spatialdata", @@ -412,9 +503,68 @@ ], [ "multiqc_citations.txt:md5,4c806e63a283ec1b7e78cdae3a923d4f", - "multiqc_software_versions.txt:md5,1774c1ba6c0b28109544da460f02aad4", + "multiqc_software_versions.txt:md5,c12e42bcb9029df6313d72d822cb0cf9", "multiqc_citations.txt:md5,4c806e63a283ec1b7e78cdae3a923d4f", - "multiqc_software_versions.txt:md5,1774c1ba6c0b28109544da460f02aad4", + "multiqc_software_versions.txt:md5,c12e42bcb9029df6313d72d822cb0cf9", + "SNR_roi_tx.parquet:md5,7636c8a75be03c1aacc1662225f1862d", + "dense_intensity_regions_summary.csv:md5,cae928f6aa311af4b0e1bfc17fbd092e", + "ccfs_thresholded.png:md5,032f6b10b6878f403efbbe01af8b7110", + "cell_flagged_maps.png:md5,0ba882744ff82f34a1a6bbc83f976c8f", + "dapi_intensity_vs_transcripts_log.png:md5,d8fdf16b335d6b8b94d8a6267e0b033a", + "distance_map_edge.png:md5,1fd0072b0f726b8f2cab8488ced54ec6", + "distance_map_holes.png:md5,bec4e0e137dd6c6054f89f813628f392", + "distance_maps.png:md5,642f233c2f4069ff44b951dec56d407d", + "gmm_focus_vs_transcripts.png:md5,3a5342221bf9d12076b8cba0ebdf8260", + "grid_roi_focus_heatmap.png:md5,fe63ecb24348e4b8c51a8abce368d39e", + "imageqc_masks.png:md5,406baa0d4058345872d4911669c633f5", + "intensity_assessment.png:md5,d2edac7a810058fe776f4ff3768a09fc", + "nuclear_texture_vs_transcripts_log.png:md5,b0a8bfb561526af53b34033d43a18bab", + "roi_focus_vs_intensity.png:md5,7cd8798673ac7e95d69f7bdd33dc8f52", + "tile_blur_proportions_roi.png:md5,7f36d91b1938443d796aa692480a515c", + "tile_focus_gmm_spatial.png:md5,53a6bfe4a7c73750ca8a85438b1d2517", + "cell_flagged_maps.png:md5,0ba882744ff82f34a1a6bbc83f976c8f", + "dapi_intensity_vs_transcripts_log.png:md5,d8fdf16b335d6b8b94d8a6267e0b033a", + "gmm_focus_vs_transcripts.png:md5,3a5342221bf9d12076b8cba0ebdf8260", + "tile_blur_proportions_roi.png:md5,8d2c5b6aafcc1666b27d09075dd191c6", + "tile_focus_gmm_spatial.png:md5,53a6bfe4a7c73750ca8a85438b1d2517", + "grid_roi_focus_scores.csv:md5,b2cfd1da35fd51bd43cc3f44837a931c", + "image_qc_cell_metrics.csv:md5,f9b6d14445af2f3ed9e9fadcc0f8075a", + "image_qc_cell_metrics.json:md5,c970ee84992e3f38f24b4cd4b7449992", + "image_qc_metrics.csv:md5,f9b6d14445af2f3ed9e9fadcc0f8075a", + "image_qc_metrics.json:md5,69f2ae0bdf4603c3dd8aecf9dcb2912a", + "image_qc_status.json:md5,14f825da7a7d6f84385cdcf179ab1f98", + "intensity_assessment.json:md5,ca3a3ab1a25d10d1726707ee92b0a381", + "roi_blur_threshold.json:md5,1f3640c94ebb2087b5a650cae7f352a4", + "roi_count.txt:md5,c881617af0ae19b38dd0546548cd53ad", + "roi_qc_metrics.json:md5,be710c04ad674d86641685aed4517812", + "roi_size.txt:md5,649ee93d50739c656e94ec88a32c7ffe", + "snr_metrics.json:md5,a2037280229900d1d049998fa663980f", + "versions.yml:md5,e1d8ab29fd14c338cd8a7b67c8c44c9e", + "df_spatial_quality.csv:md5,3dcb5878080b65eb80e56c81c0e469a4", + "cell_size_distribution.png:md5,9f3f93c8a0d69d72238b6feaa5d77f5d", + "cell_size_distribution_log.png:md5,f663e879592bafe3b526b87536a695e4", + "nucleus_to_cell_size_fraction_per_cell_distribution.png:md5,214fd9b389a9b2a29a03ef322e93eb4d", + "nucleus_transcript_fraction_per_cell_distribution.png:md5,406c6ec940050080d83cf769205d0756", + "num_genes_per_cell.png:md5,c9fe04ed2c11dd1fbb3255dd258f7fea", + "num_transcripts_per_cell.png:md5,c555e864011cdf4a44e5636b4e6411dd", + "num_transcripts_per_feature.png:md5,1aeaae3be7287677b916d07286aa12b0", + "quality_by_fov.png:md5,2878179b8869887f57f70fb2d6a81790", + "quality_distributions_comprehensive.png:md5,2f0f5f5c1f791e39987df00c110eb5ce", + "unassigned_per_gene.png:md5,373320b0b7a93f78e908a9e817d9e611", + "cell_size_distribution.csv:md5,6c8aee45e7c0045918502fe4d4d441e0", + "nucleus_to_cell_size_fraction_per_cell_distribution.csv:md5,f753929ec13d1e3dfaf94576bbd7d0d5", + "nucleus_transcript_fraction_per_cell_distribution.csv:md5,9c92b67705b28fe0ae017e3bf9a2dcdb", + "num_genes_per_cell.csv:md5,1dc6c4f75a58c595163d192182f73d92", + "num_transcripts_per_cell.csv:md5,e7378613f3d9966c3d25e449df4d690a", + "num_transcripts_per_feature.csv:md5,ce592e962496b30c4c7bc0a4a4b83ccf", + "quality_by_fov.csv:md5,04b3fd532a8c12dc29e9c084a6c891f8", + "quality_distributions_comprehensive.csv:md5,3dcb5878080b65eb80e56c81c0e469a4", + "unassigned_per_gene.csv:md5,d000d944fd21c9417a1f2923d07ac37f", + "num_genes_per_cell.csv:md5,dda1841ad514e161fc4c6afad7816c99", + "num_transcripts_per_cell.csv:md5,7babff329ac62e8c2d7d95ee4b2168a4", + "retained_genes.csv:md5,9d531795ce1d22cd60c0f70ccdec0ac6", + "transcript_qc_metrics.json:md5,c031bdf67c7f36a03523cd105902840b", + "versions.yml:md5,e027a59df4703317fe257afdafa30397", "0:md5,397dc21e479c58fd2ffcbe6c3e627683", "1:md5,f1a6b9238c3095bad7867b22f293e5e6", "zarr.json:md5,893b4d8873fc997d364288c63ef0871f", @@ -475,7 +625,7 @@ "transcripts.zarr.zip:md5,807e63a2ef8340e085cd899507f45395" ] ], - "timestamp": "2026-08-15T17:13:09.795328271", + "timestamp": "2026-08-27T15:18:03.157962084", "meta": { "nf-test": "0.9.5", "nextflow": "26.04.6" @@ -484,6 +634,12 @@ "-profile test stub": { "content": [ { + "ANALYSIS": { + "seaborn": "0.13.2" + }, + "REPORT": { + "quarto": "1.6.42" + }, "SPATIALDATA_WRITE_RAW_BUNDLE": { "spatialdata": "0.7.2" }, @@ -601,6 +757,16 @@ "qc/multiqc/redefined_bundle/multiqc_plots", "qc/multiqc/redefined_bundle/multiqc_plots/.stub", "qc/multiqc/redefined_bundle/multiqc_report.html", + "qc/qc", + "qc/qc/image_qc", + "qc/qc/image_qc/figures", + "qc/qc/image_qc/image_qc.html", + "qc/qc/image_qc/image_qc_metrics.csv", + "qc/qc/image_qc/image_qc_metrics.json", + "qc/qc/transcript_qc", + "qc/qc/transcript_qc/figures", + "qc/qc/transcript_qc/transcript_qc.html", + "qc/qc/transcript_qc/transcript_qc_metrics.json", "qc/spatialdata", "qc/spatialdata/write", "qc/spatialdata/write/spatialdata", @@ -641,6 +807,9 @@ ".stub:md5,d41d8cd98f00b204e9800998ecf8427e", ".stub:md5,d41d8cd98f00b204e9800998ecf8427e", "multiqc_report.html:md5,d41d8cd98f00b204e9800998ecf8427e", + "image_qc_metrics.csv:md5,d41d8cd98f00b204e9800998ecf8427e", + "image_qc_metrics.json:md5,d41d8cd98f00b204e9800998ecf8427e", + "transcript_qc_metrics.json:md5,d41d8cd98f00b204e9800998ecf8427e", "fake_file.txt:md5,d41d8cd98f00b204e9800998ecf8427e", ".end-of-run:md5,d41d8cd98f00b204e9800998ecf8427e", "analysis.tar.gz:md5,d41d8cd98f00b204e9800998ecf8427e", @@ -666,7 +835,7 @@ "transcripts.zarr.zip:md5,d41d8cd98f00b204e9800998ecf8427e" ] ], - "timestamp": "2026-08-15T17:20:09.102748685", + "timestamp": "2026-08-21T20:09:09.735409159", "meta": { "nf-test": "0.9.5", "nextflow": "26.04.6" diff --git a/tests/test_xenium_patch/test_stitch_transcripts.py b/tests/test_xenium_patch/test_stitch_transcripts.py index 5419c897..d0677570 100644 --- a/tests/test_xenium_patch/test_stitch_transcripts.py +++ b/tests/test_xenium_patch/test_stitch_transcripts.py @@ -8,16 +8,30 @@ import pyarrow as pa import pyarrow.csv as pa_csv import pytest -from shapely.geometry import Polygon, mapping + +# stitch_transcripts.py and this module import shapely and sopa at module scope, +# and those only exist in the stitch module's container. Skip before importing +# them: an ImportError here aborts collection for the whole pytest run, not just +# this file. These guards must stay above the shapely import below. +pytest.importorskip( + "shapely", reason="shapely is only installed in the stitch module container" +) +pytest.importorskip( + "sopa", reason="sopa is only installed in the stitch module container" +) + +from shapely.geometry import Polygon, mapping # noqa: E402 # --------------------------------------------------------------------------- -# Import the standalone script from module resources +# Import the standalone script from the pipeline's bin/ # --------------------------------------------------------------------------- -_SCRIPT = ( - Path(__file__).resolve().parents[2] - / "modules/local/xenium_patch/stitch/resources/usr/bin/stitch_transcripts.py" -) +_SCRIPT = Path(__file__).resolve().parents[2] / "bin/stitch_transcripts.py" +if not _SCRIPT.is_file(): + raise FileNotFoundError( + f"Script under test not found: {_SCRIPT}. This test imports it by path, so a " + "stale path silently takes the whole module out of service instead of failing." + ) _spec = importlib.util.spec_from_file_location("stitch_transcripts", _SCRIPT) _mod = importlib.util.module_from_spec(_spec) sys.modules["stitch_transcripts"] = _mod diff --git a/workflows/spatialaxe.nf b/workflows/spatialaxe.nf index ed72343c..53814261 100644 --- a/workflows/spatialaxe.nf +++ b/workflows/spatialaxe.nf @@ -47,6 +47,7 @@ include { SPATIALDATA_WRITE_META_MERGE } from '../subworkflo // qc layer subworkflows include { OPT_FLIP_TRACK_STAT } from '../subworkflows/local/opt_flip_track_stat/main' include { SPOQC } from '../subworkflows/local/spoqc/main' +include { QC } from '../subworkflows/local/qc/main' /* ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ @@ -97,6 +98,8 @@ workflow SPATIALAXE { tiling xeniumranger_only spoqc + roi_image_qc_thresholds_yaml + transcript_qc_thresholds_yaml main: @@ -366,9 +369,15 @@ workflow SPATIALAXE { return [meta, gene_panel_file] } } - else { - - // gene panel to use if only --relabel_genes is provided + else if (do_relabel) { + + // Gene panel from the bundle, used when only --relabel_genes is given. + // Guarded by do_relabel: the file(checkIfExists:) inside .map runs for + // every sample even when the channel is never consumed, so building this + // unconditionally fails any bundle without the optional gene_panel.json + // (and any remote tarball input). When relabelling is off, ch_gene_panel + // keeps its channel.empty() initialisation, and its only consumer is + // already inside `if (do_relabel)`. ch_gene_panel = ch_input.map { meta, bundle, _image, _annotation, _stainings -> def gene_panel_file = file( file(bundle).toUriString().replaceFirst(/\/$/, '') + "/gene_panel.json", @@ -667,6 +676,28 @@ workflow SPATIALAXE { ) } + // Image QC and transcript QC on the validated Xenium bundle. The + // threshold configs and notebooks are resolved inside this block so + // their `checkIfExists` never runs when the QC layer is skipped. + ch_image_qc_thresholds = channel.fromPath( + roi_image_qc_thresholds_yaml ?: "${projectDir}/bin/roi_image_qc_thresholds.yaml", + checkIfExists: true, + ) + ch_transcript_qc_thresholds = channel.fromPath( + transcript_qc_thresholds_yaml ?: "${projectDir}/bin/transcript_qc_thresholds.yaml", + checkIfExists: true, + ) + + QC( + ch_bundle_path, + ch_image_qc_thresholds, + ch_transcript_qc_thresholds, + file("${projectDir}/bin/xenium_image_qc_report.qmd", checkIfExists: true), + file("${projectDir}/bin/transcript_qc.qmd", checkIfExists: true), + "${outdir}/${mode}/qc/image_qc", + "${outdir}/${mode}/qc/transcript_qc", + ) + } @@ -709,13 +740,30 @@ workflow SPATIALAXE { SPATIALAXE - COLLATE & SAVE SOFTWARE VERSIONS ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ */ - // Collect versions published via topic channels (local modules) - ch_topic_versions = channel.topic('versions') + // Collect versions published via topic channels. Two shapes arrive here: + // local modules emit (process, tool, version) tuples, while some nf-core + // modules emit a `versions.yml` path (quarto/notebook builds one from the + // versions.csv its notebook exports). Handle both — destructuring a path + // would fail, and dropping it would discard real version information. + ch_topic_raw = channel.topic('versions') + + ch_topic_versions = ch_topic_raw + .filter { entry -> entry instanceof List && entry.size() == 3 } .map { process, tool, version -> "\"${process}\":\n ${tool}: ${version}" } - softwareVersionsToYAML(ch_versions.mix(ch_topic_versions)) + // softwareVersionsToYAML parses YAML *content*, so read the file in. Drop + // empty content: -stub runs produce an empty versions.yml, and parsing that + // yields null, which the downstream collectEntries would fail on. + ch_topic_version_files = ch_topic_raw + .filter { entry -> !(entry instanceof List) } + .map { versions_file -> file(versions_file).text.trim() } + .filter { content -> content } + + softwareVersionsToYAML( + ch_versions.mix(ch_topic_versions).mix(ch_topic_version_files) + ) .collectFile( storeDir: "${outdir}/pipeline_info", name: 'nf_core_' + 'spatialaxe_software_' + 'mqc_' + 'versions.yml',