diff --git a/src/mantispy/ds/_build.py b/src/mantispy/ds/_build.py index 9577d1d..8418b89 100644 --- a/src/mantispy/ds/_build.py +++ b/src/mantispy/ds/_build.py @@ -42,23 +42,21 @@ def _stable(adata: AnnData) -> AnnData: return adata[rows][:, cols].copy() -def _gene_consensus(block: AnnData, screened: np.ndarray) -> AnnData: - """One modz consensus per gene over the screened wells of ``block``, NaN-zeroed.""" +def _modz(block: AnnData, by: str) -> AnnData: + """One modz consensus per ``by`` group over every row of ``block``, NaN-zeroed.""" from mantispy.tl._consensus import consensus - gene = consensus(block[screened], by="Metadata_Gene", method="modz", correlation="spearman", min_replicates=2) - return _zero_nonfinite(gene) - + agg = consensus(block, by=by, method="modz", correlation="spearman", min_replicates=2) + return _zero_nonfinite(agg) -def _perturbation_consensus(block: AnnData) -> AnnData: - """One modz consensus per ``Metadata_Perturbation`` over every well of ``block``, NaN-zeroed. - Controls are kept: BBBC021's DMSO-at-a-concentration wells are legitimate perturbation units. - """ - from mantispy.tl._consensus import consensus +def _feature_selected_block(base: AnnData) -> AnnData: + """A stable copy of ``base`` reduced to pycytominer's default feature selection.""" + from mantispy.pp._select import feature_select, subset_features - agg = consensus(block, by="Metadata_Perturbation", method="modz", correlation="spearman", min_replicates=2) - return _zero_nonfinite(agg) + block = base.copy() + feature_select(block) + return _stable(subset_features(block)) def build_rohban_base(cache_dir: str | Path | None = None) -> dict[str, AnnData]: @@ -87,8 +85,8 @@ def build_rohban_variants(cache_dir: str | Path | None = None) -> dict[str, AnnD screened = (~adata.obs["is_untreated"].to_numpy()) & (~adata.obs["Metadata_Control"].to_numpy()) return { "rohban_selected.h5ad": _stable(selected), - "rohban_gene.h5ad": _stable(_gene_consensus(adata, screened)), - "rohban_gene_selected.h5ad": _stable(_gene_consensus(selected, screened)), + "rohban_gene.h5ad": _stable(_modz(adata[screened], "Metadata_Gene")), + "rohban_gene_selected.h5ad": _stable(_modz(selected[screened], "Metadata_Gene")), } @@ -168,22 +166,16 @@ def _assemble_bbbc021(cache_dir: str | Path | None = None) -> AnnData: def build_bbbc021_variants(cache_dir: str | Path | None = None) -> dict[str, AnnData]: """Returns {'bbbc021.h5ad': ad, 'bbbc021_selected.h5ad': ad, 'bbbc021_agg.h5ad': ad, 'bbbc021_agg_selected.h5ad': ad}.""" - from mantispy.pp._select import feature_select, subset_features - base = _stable(_assemble_bbbc021(cache_dir)) - - # Well-level feature-selected block, pycytominer's default operations, on the raw (author-normalized) base. - selected = base.copy() - feature_select(selected) - selected = _stable(subset_features(selected)) + selected = _feature_selected_block(base) # One modz consensus per compound-at-concentration, on the full and on the feature-selected block. DMSO # wells are kept: their concentration series are legitimate perturbation units, not a normalization control. return { "bbbc021.h5ad": base, "bbbc021_selected.h5ad": selected, - "bbbc021_agg.h5ad": _stable(_perturbation_consensus(base)), - "bbbc021_agg_selected.h5ad": _stable(_perturbation_consensus(selected)), + "bbbc021_agg.h5ad": _stable(_modz(base, "Metadata_Perturbation")), + "bbbc021_agg_selected.h5ad": _stable(_modz(selected, "Metadata_Perturbation")), } @@ -217,6 +209,382 @@ def _shipped_bbbc021() -> dict[str, AnnData]: } +def _assemble_neuropainting(cache_dir: str | Path | None = None) -> AnnData: + """Astrocyte and neuron wells read down to the features the six plates share. + + The raw pipeline the both-flags-absent :func:`mt.ds.neuropainting` used before the base was staged; it lives + here so the drift check can rebuild the hosted base from the raw per-plate tables. + """ + from mantispy.ds._datasets import _profiles + + return _profiles("neuropainting", cache_dir, select=lambda name: not name.endswith(".h5ad")) + + +def build_neuropainting(cache_dir: str | Path | None = None) -> dict[str, AnnData]: + """Returns {'neuropainting.h5ad': the wells the loader used to assemble}.""" + return {"neuropainting.h5ad": _stable(_assemble_neuropainting(cache_dir))} + + +def _shipped_neuropainting() -> dict[str, AnnData]: + """The neuropainting base as fetched through the public ``mt.ds.neuropainting`` API.""" + from mantispy.ds._datasets import neuropainting + + return {"neuropainting.h5ad": neuropainting()} + + +def _assemble_chroma(cache_dir: str | Path | None = None) -> AnnData: + """Assemble the chroma base: the extra-channel wells with their compound, concentration, MOA and control. + + The raw pipeline the loader used before the base was staged; it lives here so the drift check can rebuild + the hosted base from the raw per-plate tables. + """ + import numpy as np + import pandas as pd + + from mantispy._core.frames import as_frame + from mantispy._core.logging import get_logger + from mantispy.ds._datasets import _profiles + + adata = _profiles("chroma", cache_dir, select=lambda name: not name.endswith(".h5ad")) + obs = as_frame(adata.obs) + control = (obs["Metadata_control_type"].astype(str) == "negcon").to_numpy() + obs["Metadata_Control"] = control + name = obs["Metadata_Common Name"].astype(str).to_numpy() + dose = pd.to_numeric(obs["Metadata_mmoles_per_liter"], errors="coerce").to_numpy(dtype=float) + compound = np.where(control, "DMSO", name) + obs["Metadata_Compound"] = pd.Categorical(compound) + obs["Metadata_Concentration"] = dose + obs["Metadata_MOA"] = obs["Metadata_MoA"].astype("category") + label = np.where(control, "DMSO", np.char.add(np.char.add(compound.astype(str), "@"), dose.astype(str))) + obs["Metadata_Perturbation"] = pd.Categorical(label) + obs["Metadata_Perturbation_Type"] = pd.Series("compound", index=obs.index, dtype="category") + get_logger().info( + "chroma: %d wells x %d features, %d compounds, %d control wells", + adata.n_obs, + adata.n_vars, + int(pd.Series(name[~control]).nunique()), + int(control.sum()), + ) + return adata + + +def build_chroma(cache_dir: str | Path | None = None) -> dict[str, AnnData]: + """Returns {'chroma.h5ad': the wells the loader used to assemble}.""" + return {"chroma.h5ad": _stable(_assemble_chroma(cache_dir))} + + +def _shipped_chroma() -> dict[str, AnnData]: + """The chroma base as fetched through the public ``mt.ds.chroma`` API.""" + from mantispy.ds._datasets import chroma + + return {"chroma.h5ad": chroma()} + + +def _assemble_pooled_rare(cache_dir: str | Path | None = None) -> AnnData: + """Assemble the pooled-rare base: the per-barcode profiles with their variant, gene and allele. + + The raw pipeline the loader used before the base was staged; it lives here so the drift check can rebuild + the hosted base from the raw gene-normalized table. + """ + import pandas as pd + + from mantispy._core.frames import as_frame + from mantispy._core.logging import get_logger + from mantispy.ds._datasets import _profiles + + adata = _profiles("pooled_rare", cache_dir, select=lambda name: not name.endswith(".h5ad")) + obs = as_frame(adata.obs) + code = obs["Metadata_Foci_Barcode_MatchedTo_GeneCode"].astype(str) + obs["Metadata_Perturbation"] = code.astype("category") + obs["Metadata_Perturbation_Type"] = pd.Series("orf", index=obs.index, dtype="category") + # The gene is the token before the first space; the variant "ACTB E364K" belongs to gene "ACTB". + obs["Metadata_Gene"] = code.str.split(" ").str[0].astype("category") + obs["Metadata_Allele"] = code.astype("category") + get_logger().info( + "pooled_rare: %d variants over %d genes x %d features", + adata.n_obs, + int(obs["Metadata_Gene"].nunique()), + adata.n_vars, + ) + return adata + + +def build_pooled_rare(cache_dir: str | Path | None = None) -> dict[str, AnnData]: + """Returns {'pooled_rare.h5ad': ad, 'pooled_rare_selected.h5ad': ad}.""" + base = _stable(_assemble_pooled_rare(cache_dir)) + return {"pooled_rare.h5ad": base, "pooled_rare_selected.h5ad": _feature_selected_block(base)} + + +def _shipped_pooled_rare() -> dict[str, AnnData]: + """The two pooled-rare variants as fetched through the public ``mt.ds.pooled_rare`` API.""" + from mantispy.ds._datasets import pooled_rare + + return { + "pooled_rare.h5ad": pooled_rare(), + "pooled_rare_selected.h5ad": pooled_rare(feature_selected=True), + } + + +def _assemble_oasis_pilot(cache_dir: str | Path | None = None) -> AnnData: + """Assemble the annotated OASIS-pilot base: the profiles joined to the plate maps and their doses. + + The raw pipeline the ``annotate=True`` :func:`mt.ds.oasis_pilot` used before the base was staged; it lives + here so the drift check can rebuild the hosted base from the raw profile and plate-map tables. The + ``annotate=False`` path stays in the loader, reading the raw profiles directly. + """ + import numpy as np + import pandas as pd + + from mantispy._core.frames import as_frame + from mantispy._core.logging import get_logger + from mantispy.ds._datasets import _oasis_platemaps, _profiles + + adata = _profiles("oasis_pilot", cache_dir, select=lambda name: name.endswith(".csv.gz")) + obs = as_frame(adata.obs) + merged = obs.merge(_oasis_platemaps(cache_dir), on=["Metadata_plate_map_name", "Metadata_Well"], how="left") + merged.index = obs.index + if unmatched := int(merged["Metadata_Compound"].isna().sum()): + get_logger().warning("oasis_pilot: %d of %d wells have no plate-map row", unmatched, len(merged)) + adata.obs["Metadata_Compound"] = merged["Metadata_Compound"].to_numpy() + adata.obs["Metadata_Concentration"] = merged["Metadata_Concentration"].to_numpy(dtype=float) + adata.obs["Metadata_ConcentrationRecorded"] = merged["Metadata_ConcentrationRecorded"].to_numpy(dtype=float) + adata.obs["Metadata_CellLine"] = merged["Metadata_CellLine"].to_numpy() + adata.obs["Metadata_Control"] = merged["Metadata_Compound"].astype(str).str.upper().eq("DMSO").to_numpy() + is_control = np.asarray(adata.obs["Metadata_Control"], dtype=bool) + adata.obs["Metadata_Perturbation"] = pd.Categorical( + np.where( + is_control, + "DMSO", + merged["Metadata_Compound"].astype(str) + "@" + merged["Metadata_Concentration"].astype(str), + ) + ) + adata.obs["Metadata_Perturbation_Type"] = pd.Series("compound", index=adata.obs_names, dtype="category") + get_logger().info( + "OASIS pilot: %d wells x %d features, %d compounds over %d concentrations, %d control wells", + adata.n_obs, + adata.n_vars, + int(merged.loc[~is_control, "Metadata_Compound"].nunique()), + int(merged["Metadata_Concentration"].nunique()), + int(is_control.sum()), + ) + return adata + + +def build_oasis_pilot(cache_dir: str | Path | None = None) -> dict[str, AnnData]: + """Returns {'oasis_pilot.h5ad': ad, 'oasis_pilot_agg.h5ad': ad}.""" + base = _stable(_assemble_oasis_pilot(cache_dir)) + # One modz consensus per compound-at-concentration, DMSO wells kept as the "DMSO" perturbation. + return {"oasis_pilot.h5ad": base, "oasis_pilot_agg.h5ad": _stable(_modz(base, "Metadata_Perturbation"))} + + +def _shipped_oasis_pilot() -> dict[str, AnnData]: + """The two OASIS-pilot variants as fetched through the public ``mt.ds.oasis_pilot`` API.""" + from mantispy.ds._datasets import oasis_pilot + + return {"oasis_pilot.h5ad": oasis_pilot(), "oasis_pilot_agg.h5ad": oasis_pilot(aggregated=True)} + + +def _assemble_pki(cache_dir: str | Path | None = None) -> AnnData: + """Assemble the PKI base (all eight plates): the dose-series wells with their compound, dose, MOA and control. + + The raw pipeline the both-flags-False :func:`mt.ds.pki` used before the base was staged; it lives here so + the drift check can rebuild the hosted base from the raw ``*_augmented`` plate tables. + """ + import numpy as np + import pandas as pd + + from mantispy._core.frames import as_frame + from mantispy._core.logging import get_logger + from mantispy.ds._datasets import _augmented + + adata = _augmented("pki", None, cache_dir) + obs = as_frame(adata.obs) + control = (obs["Metadata_control_type"].astype(str) == "negcon").to_numpy() + obs["Metadata_Control"] = control + # Control wells have no compound or dose, so without this each would become its own "nan@nan" perturbation. + compound = np.where(control, "DMSO", obs["Metadata_broad_sample"].astype(str).to_numpy()) + dose = obs["Metadata_mmoles_per_liter"].to_numpy(dtype=float) + obs["Metadata_Compound"] = pd.Categorical(compound) + obs["Metadata_Concentration"] = dose + obs["Metadata_MOA"] = obs.pop("Metadata_moa") + label = np.where(control, "DMSO", np.char.add(np.char.add(compound.astype(str), "@"), dose.astype(str))) + obs["Metadata_Perturbation"] = pd.Categorical(label) + obs["Metadata_Perturbation_Type"] = pd.Series("compound", index=obs.index, dtype="category") + adata.uns["mantispy"]["dataset"] = "cpg0008-pki" + get_logger().info( + "pki: %d wells x %d features, %d compounds x %d doses", + adata.n_obs, + adata.n_vars, + int(obs.loc[~control, "Metadata_Compound"].nunique()), + int(obs.loc[~control, "Metadata_Concentration"].nunique()), + ) + return adata + + +def build_pki_variants(cache_dir: str | Path | None = None) -> dict[str, AnnData]: + """Returns {'pki.h5ad': ad, 'pki_selected.h5ad': ad, 'pki_agg.h5ad': ad, 'pki_agg_selected.h5ad': ad}.""" + base = _stable(_assemble_pki(cache_dir)) + selected = _feature_selected_block(base) + + # One modz consensus per compound-at-dose, on the full and on the feature-selected block; DMSO kept. + return { + "pki.h5ad": base, + "pki_selected.h5ad": selected, + "pki_agg.h5ad": _stable(_modz(base, "Metadata_Perturbation")), + "pki_agg_selected.h5ad": _stable(_modz(selected, "Metadata_Perturbation")), + } + + +def _shipped_pki() -> dict[str, AnnData]: + """The four pki variants as fetched through the public ``mt.ds.pki`` API.""" + from mantispy.ds._datasets import pki + + return { + "pki.h5ad": pki(), + "pki_selected.h5ad": pki(feature_selected=True), + "pki_agg.h5ad": pki(aggregated=True), + "pki_agg_selected.h5ad": pki(aggregated=True, feature_selected=True), + } + + +def _guide_aggregate(base: AnnData) -> AnnData: + """One median profile per ``(Metadata_Gene, Metadata_sgRNA)`` guide, the way the loaders document.""" + from mantispy.tl._aggregate import aggregate + + return aggregate(base, by=("Metadata_Gene", "Metadata_sgRNA")) + + +def build_scallops_arv471(cache_dir: str | Path | None = None) -> dict[str, AnnData]: + """Returns {'scallops_arv471.h5ad': the cells, 'scallops_arv471_agg.h5ad': one median per guide}.""" + from mantispy.ds._datasets import _assemble_scallops_arv471 + + base = _stable(_assemble_scallops_arv471(cache_dir)) + return {"scallops_arv471.h5ad": base, "scallops_arv471_agg.h5ad": _stable(_guide_aggregate(base))} + + +def _shipped_scallops_arv471() -> dict[str, AnnData]: + """The two scallops_arv471 variants as fetched through the public ``mt.ds.scallops_arv471`` API.""" + from mantispy.ds._datasets import scallops_arv471 + + return { + "scallops_arv471.h5ad": scallops_arv471(), + "scallops_arv471_agg.h5ad": scallops_arv471(aggregated=True), + } + + +def _assemble_jump_crispr(cache_dir: str | Path | None = None) -> AnnData: + """Assemble the annotated JUMP CRISPR base: the upstream selected profiles with their gene and controls. + + The raw pipeline the ``annotate=True`` :func:`mt.ds.jump_crispr` used before the base was staged; it lives + here so the drift check can rebuild the hosted base from the raw parquet. The ``annotate=False`` path stays + in the loader, reading the raw profiles without the annotation join. + """ + from mantispy.ds._datasets import _files, _profiles, _read_counts + from mantispy.pp._annotate import annotate_jump + + (counts,) = _files("_jump_cell_counts", cache_dir) + adata = _profiles( + "jump_crispr", cache_dir, select=lambda name: not name.endswith(".h5ad"), platemap=_read_counts(counts) + ) + annotate_jump(adata, kind="crispr") + return adata + + +def build_jump_crispr(cache_dir: str | Path | None = None) -> dict[str, AnnData]: + """Returns {'jump_crispr.h5ad': the annotated wells, 'jump_crispr_agg.h5ad': one modz per gene}.""" + base = _stable(_assemble_jump_crispr(cache_dir)) + return {"jump_crispr.h5ad": base, "jump_crispr_agg.h5ad": _stable(_modz(base, "Metadata_Gene"))} + + +def _shipped_jump_crispr() -> dict[str, AnnData]: + """The two jump_crispr variants as fetched through the public ``mt.ds.jump_crispr`` API.""" + from mantispy.ds._datasets import jump_crispr + + return {"jump_crispr.h5ad": jump_crispr(), "jump_crispr_agg.h5ad": jump_crispr(aggregated=True)} + + +def build_cp_posh_variants(cache_dir: str | Path | None = None) -> dict[str, AnnData]: + """Returns {'cp_posh.h5ad': ad, 'cp_posh_selected.h5ad': ad, 'cp_posh_agg.h5ad': ad, 'cp_posh_agg_selected.h5ad': ad}.""" + from mantispy.ds._datasets import _assemble_cp_posh + + base = _stable(_assemble_cp_posh(cache_dir)) + selected = _feature_selected_block(base) + + # One median profile per guide, on the full and on the feature-selected block. + return { + "cp_posh.h5ad": base, + "cp_posh_selected.h5ad": selected, + "cp_posh_agg.h5ad": _stable(_guide_aggregate(base)), + "cp_posh_agg_selected.h5ad": _stable(_guide_aggregate(selected)), + } + + +def _shipped_cp_posh() -> dict[str, AnnData]: + """The four cp_posh variants as fetched through the public ``mt.ds.cp_posh`` API.""" + from mantispy.ds._datasets import cp_posh + + return { + "cp_posh.h5ad": cp_posh(), + "cp_posh_selected.h5ad": cp_posh(feature_selected=True), + "cp_posh_agg.h5ad": cp_posh(aggregated=True), + "cp_posh_agg_selected.h5ad": cp_posh(aggregated=True, feature_selected=True), + } + + +def build_jump_cells(cache_dir: str | Path | None = None) -> dict[str, AnnData]: + """Returns {'jump_cells.h5ad': the annotated cells, 'jump_cells_selected.h5ad': the mask subset, 'jump_cells_agg.h5ad': one median per well}.""" + from mantispy.ds._datasets import _assemble_jump_cells + from mantispy.pp._select import subset_features + from mantispy.tl._aggregate import aggregate + + base = _stable(_assemble_jump_cells(cache_dir)) + # The feature-selection mask is already on var["selected"]; subset to it, and aggregate the cells to wells. + selected = _stable(subset_features(base)) + agg = _stable(aggregate(base)) + return {"jump_cells.h5ad": base, "jump_cells_selected.h5ad": selected, "jump_cells_agg.h5ad": agg} + + +def _shipped_jump_cells() -> dict[str, AnnData]: + """The three jump_cells variants as fetched through the public ``mt.ds.jump_cells`` API.""" + from mantispy.ds._datasets import jump_cells + + return { + "jump_cells.h5ad": jump_cells(), + "jump_cells_selected.h5ad": jump_cells(selected=True), + "jump_cells_agg.h5ad": jump_cells(aggregated=True), + } + + +def build_jump_target2(cache_dir: str | Path | None = None) -> dict[str, AnnData]: + """Returns {'jump_target2.h5ad': the annotated default object, one plate from each of the eleven sources}.""" + from mantispy.ds._datasets import TARGET2_DEFAULT, _assemble_jump_target2 + + return {"jump_target2.h5ad": _stable(_assemble_jump_target2(TARGET2_DEFAULT, True, cache_dir))} + + +def _shipped_jump_target2() -> dict[str, AnnData]: + """The jump_target2 default object as fetched through the public ``mt.ds.jump_target2`` API.""" + from mantispy.ds._datasets import jump_target2 + + return {"jump_target2.h5ad": jump_target2()} + + +def build_jump_lite(cache_dir: str | Path | None = None) -> dict[str, AnnData]: + """Returns {'jump_lite_.h5ad': the annotated wells of that feature set} for each JUMP-Lite model.""" + from mantispy.ds._datasets import JUMP_LITE_MODELS, _assemble_jump_lite + + return { + f"jump_lite_{model}.h5ad": _stable(_assemble_jump_lite(model, True, cache_dir)) for model in JUMP_LITE_MODELS + } + + +def _shipped_jump_lite() -> dict[str, AnnData]: + """The per-model jump_lite objects as fetched through the public ``mt.ds.jump_lite`` API.""" + from mantispy.ds._datasets import JUMP_LITE_MODELS, jump_lite + + return {f"jump_lite_{model}.h5ad": jump_lite(model=model) for model in JUMP_LITE_MODELS} + + # One entry per staged dataset: (builder rebuilding the variants from the raw pipeline, loader returning the # shipped variants through the public ``mt.ds`` API keyed by the same filenames, ``heavy`` marking a rebuild # too large to run on every pull request). Later PRs stage a dataset by adding one entry here; the build and @@ -226,4 +594,15 @@ def _shipped_bbbc021() -> dict[str, AnnData]: "rohban": (build_rohban_variants, _shipped_rohban, False), "rohban_base": (build_rohban_base, _shipped_rohban_base, False), "bbbc021": (build_bbbc021_variants, _shipped_bbbc021, True), + "neuropainting": (build_neuropainting, _shipped_neuropainting, False), + "chroma": (build_chroma, _shipped_chroma, False), + "pooled_rare": (build_pooled_rare, _shipped_pooled_rare, False), + "oasis_pilot": (build_oasis_pilot, _shipped_oasis_pilot, False), + "pki": (build_pki_variants, _shipped_pki, True), + "scallops_arv471": (build_scallops_arv471, _shipped_scallops_arv471, True), + "jump_crispr": (build_jump_crispr, _shipped_jump_crispr, True), + "cp_posh": (build_cp_posh_variants, _shipped_cp_posh, True), + "jump_cells": (build_jump_cells, _shipped_jump_cells, True), + "jump_target2": (build_jump_target2, _shipped_jump_target2, True), + "jump_lite": (build_jump_lite, _shipped_jump_lite, False), } diff --git a/src/mantispy/ds/_datasets.py b/src/mantispy/ds/_datasets.py index f7cb059..de3c47e 100644 --- a/src/mantispy/ds/_datasets.py +++ b/src/mantispy/ds/_datasets.py @@ -22,7 +22,6 @@ from mantispy._settings import settings from mantispy.io._jump import join_jump_annotation, read_jump from mantispy.io._profiles import _UPSTREAM_COUNTS, _adopt_counts, from_dataframe, read, read_profiles, write -from mantispy.pp._select import subset_features if TYPE_CHECKING: from anndata import AnnData @@ -108,6 +107,19 @@ def _profiles( return adata +def _require_bools(**flags: object) -> None: + """Raise ``ValueError`` naming the first flag that is not a bool.""" + for flag_name, flag in flags.items(): + if not isinstance(flag, bool): + raise ValueError(f"{flag_name} must be a bool, got {type(flag).__name__}") + + +def _fetch_variant(name: str, filename: str, cache_dir: str | Path | None) -> AnnData: + """Read the single rehosted ``filename`` variant of dataset ``name``.""" + (path,) = _files(name, cache_dir, select=lambda file_name: file_name == filename) + return read(path) + + #: The (aggregated, feature_selected) combination each rehosted bbbc021 variant answers to. _BBBC021_VARIANTS = { (False, False): "bbbc021.h5ad", # 632 x 467, well level, all features @@ -153,12 +165,8 @@ def bbbc021( :cite:t:`Ljosa_2013`, these profiles and the benchmark. Images courtesy of Peter Caie and David Westwood, available from the Broad Bioimage Benchmark Collection :cite:p:`Ljosa_2012`. """ - for flag_name, flag in (("aggregated", aggregated), ("feature_selected", feature_selected)): - if not isinstance(flag, bool): - raise ValueError(f"{flag_name} must be a bool, got {type(flag).__name__}") - target = _BBBC021_VARIANTS[aggregated, feature_selected] - (path,) = _files("bbbc021", cache_dir, select=lambda name: name == target) - return read(path) + _require_bools(aggregated=aggregated, feature_selected=feature_selected) + return _fetch_variant("bbbc021", _BBBC021_VARIANTS[aggregated, feature_selected], cache_dir) #: The (aggregated, feature_selected) combination each rehosted rohban variant answers to. @@ -269,62 +277,94 @@ def rohban( References: :cite:t:`Rohban_2017`. """ - for flag_name, flag in (("aggregated", aggregated), ("feature_selected", feature_selected)): - if not isinstance(flag, bool): - raise ValueError(f"{flag_name} must be a bool, got {type(flag).__name__}") + _require_bools(aggregated=aggregated, feature_selected=feature_selected) if (aggregated or feature_selected) and plates is not None: raise ValueError( "plates only applies to the raw wells; a pre-aggregated or feature-selected variant cannot be plate-subset" ) - target = _ROHBAN_VARIANTS[aggregated, feature_selected] - (path,) = _files("rohban", cache_dir, select=lambda name: name == target) - adata = read(path) + adata = _fetch_variant("rohban", _ROHBAN_VARIANTS[aggregated, feature_selected], cache_dir) if plates is not None: adata = _subset_plates(adata, "rohban", plates) return adata -def pki(plates: Sequence[str] | None = None, cache_dir: str | Path | None = None) -> AnnData: +#: The (aggregated, feature_selected) combination each rehosted pki variant answers to. +_PKI_VARIANTS = { + (False, False): "pki.h5ad", # 3072 x 5857, well level, all features + (False, True): "pki_selected.h5ad", # well level, feature selected + (True, False): "pki_agg.h5ad", # perturbation level (modz), all features + (True, True): "pki_agg_selected.h5ad", # perturbation level (modz), feature selected +} + + +def pki( + plates: Sequence[str] | None = None, + cache_dir: str | Path | None = None, + *, + aggregated: bool = False, + feature_selected: bool = False, +) -> AnnData: """Kinase inhibitors over a dose series, from the JUMP pilot. ``cpg0008-pki``: fifteen compounds over a seven-point dose range (eleven at three doses, four at one) in U2OS cells, over eight plates with 32 to 64 replicate wells per treatment. - Downloads about 71 MB for all eight. + + The base and its variants are pre-built by ``scripts/build_staged_datasets.py`` from the eight raw plate tables and rehosted on ``scverse-exampledata``, so the loader fetches a single h5ad rather than reassembling the base on every call. The two flags select the variant: + + - both ``False``: the raw wells with every feature, the object the recipe tutorials start from. + - ``feature_selected=True``: the well-level block after pycytominer-default feature selection. + - ``aggregated=True``: one ``modz`` consensus (Spearman, ``min_replicates=2``) per ``Metadata_Perturbation`` over every well, DMSO included, so each compound-at-dose is one profile. + - ``aggregated=True, feature_selected=True``: that same consensus on the feature-selected block. Args: plates: Plate barcodes to load, all eight when omitted. + Only applies to the base; the hosted base is fetched once and subset to these plates in memory. cache_dir: Where to keep the download. Defaults to :attr:`mantispy.settings.cache_dir`. + aggregated: Return the perturbation-level ``modz`` consensus instead of the wells. + feature_selected: Return the feature-selected block instead of all features. Returns: - Wells by features at well resolution, with ``Metadata_Perturbation`` (compound at concentration), ``Metadata_Perturbation_Type`` (``"compound"``), ``Metadata_Compound``, ``Metadata_Concentration`` (the platemap's ``mmoles_per_liter``), ``Metadata_MOA``, ``Metadata_Control``, ``Metadata_CellCount`` and ``Metadata_SiteCount``. + Wells by features at well resolution when both flags are ``False``, else the staged variant selected by the two flags, read with :func:`mantispy.io.read`, with ``Metadata_Perturbation`` (compound at concentration), ``Metadata_Perturbation_Type`` (``"compound"``), ``Metadata_Compound``, ``Metadata_Concentration`` (the platemap's ``mmoles_per_liter``), ``Metadata_MOA``, ``Metadata_Control``, ``Metadata_CellCount`` and ``Metadata_SiteCount``. Raises: KeyError: A plate is not one of the eight. + ValueError: ``aggregated`` or ``feature_selected`` is not a bool, or ``plates`` is given for a variant. Notes: ``Metadata_Control`` marks the DMSO wells only. The positive controls (``Metadata_control_type == "poscon"``) are not flagged, because they are perturbations and should not be normalized against. """ - adata = _augmented("pki", plates, cache_dir) - obs = as_frame(adata.obs) - control = (obs["Metadata_control_type"].astype(str) == "negcon").to_numpy() - obs["Metadata_Control"] = control - # Control wells have no compound or dose, so without this each would become its own "nan@nan" perturbation. - compound = np.where(control, "DMSO", obs["Metadata_broad_sample"].astype(str).to_numpy()) - dose = obs["Metadata_mmoles_per_liter"].to_numpy(dtype=float) - obs["Metadata_Compound"] = pd.Categorical(compound) - obs["Metadata_Concentration"] = dose - obs["Metadata_MOA"] = obs.pop("Metadata_moa") - label = np.where(control, "DMSO", np.char.add(np.char.add(compound.astype(str), "@"), dose.astype(str))) - obs["Metadata_Perturbation"] = pd.Categorical(label) - obs["Metadata_Perturbation_Type"] = pd.Series("compound", index=obs.index, dtype="category") - adata.uns["mantispy"]["dataset"] = "cpg0008-pki" + _require_bools(aggregated=aggregated, feature_selected=feature_selected) + if (aggregated or feature_selected) and plates is not None: + raise ValueError( + "plates only applies to the raw wells; a pre-aggregated or feature-selected variant cannot be plate-subset" + ) + adata = _fetch_variant("pki", _PKI_VARIANTS[aggregated, feature_selected], cache_dir) + if plates is not None: + adata = _subset_plates(adata, "pki", plates) + return adata + + +def _assemble_jump_target2(plates: Sequence[str] | None, annotate: bool, cache_dir: str | Path | None) -> AnnData: + """Assemble a jump_target2 object from the raw per-plate profiles and backend count tables. + + The raw pipeline the loader used before the default object was staged; it lives here so the loader can still + read any plate selection (the non-default ``plates`` and ``annotate=False`` paths) and so the drift check can + rebuild the hosted default from the raw inputs. + """ + paths = _plate_files("jump_target2", plates, cache_dir) + # The profiles carry no count; each plate's backend table does, among 7,600 other columns. + counts = pd.concat([_read_counts(path) for path in paths if path.suffix == ".csv"]) + profiles = [path for path in paths if path.suffix == ".parquet"] + adata = read_jump(profiles, annotate=annotate, on_column_mismatch="intersect", platemap=counts) + batches = {_plate(file.name): file.name.split("__")[0] for file in _DATASETS["jump_target2"].files} + adata.obs["Metadata_Batch"] = [batches[str(plate)] for plate in adata.obs["Metadata_Plate"]] + adata.uns["mantispy"]["dataset"] = "cpg0016-jump" get_logger().info( - "pki: %d wells x %d features, %d compounds x %d doses", + "jump_target2: %d wells x %d features from %d source(s)", adata.n_obs, adata.n_vars, - int(obs.loc[~control, "Metadata_Compound"].nunique()), - int(obs.loc[~control, "Metadata_Concentration"].nunique()), + adata.obs["Metadata_Source"].nunique(), ) return adata @@ -339,9 +379,11 @@ def jump_target2( All 141 of its plates in ``cpg0016-jump`` are pinned here, from eleven sources and 107 batches, which lets :func:`~mantispy.tl.transport` separate a laboratory effect from a batch and a plate effect. source_9 ran it on 1536-well plates, the others on 384-well plates. + The default object (one plate from each of the eleven sources, annotated) is pre-built by ``scripts/build_staged_datasets.py`` and rehosted on ``scverse-exampledata``, so the default call fetches a single h5ad. Any other plate selection, or ``annotate=False``, still reads the raw per-plate profiles and backend count tables directly, so the batch-effect study can load any subset or all 141 plates. + Args: plates: Plate barcodes to load. - The default takes one plate from each source, about 0.7 GB, most of it the per-well table each plate's cell counts are published in; ``None`` loads all 141, 9.4 GB. + The default takes one plate from each source; ``None`` loads all 141, 9.4 GB. annotate: Join the JUMP annotation, which supplies ``Metadata_Perturbation`` and ``Metadata_Control``. Downloads another 14 MB. cache_dir: Where to keep the download. @@ -349,97 +391,87 @@ def jump_target2( Returns: One row per well at well resolution, carrying ``Metadata_Source``, ``Metadata_Batch``, ``Metadata_Plate``, ``Metadata_Well``, ``Metadata_CellCount``, ``Metadata_SiteCount`` and, when annotated, ``Metadata_JCP2022``, ``Metadata_Perturbation``, ``Metadata_Perturbation_Type`` (``"compound"``), ``Metadata_InChIKey`` and ``Metadata_Control`` (the DMSO wells). + The default call is read with :func:`mantispy.io.read`; any other selection is assembled from the raw tables. Raises: KeyError: A plate is not one of the 141. """ - paths = _plate_files("jump_target2", plates, cache_dir) - # The profiles carry no count; each plate's backend table does, among 7,600 other columns. - counts = pd.concat([_read_counts(path) for path in paths if path.suffix == ".csv"]) - profiles = [path for path in paths if path.suffix == ".parquet"] - adata = read_jump(profiles, annotate=annotate, on_column_mismatch="intersect", platemap=counts) - batches = {_plate(file.name): file.name.split("__")[0] for file in _DATASETS["jump_target2"].files} - adata.obs["Metadata_Batch"] = [batches[str(plate)] for plate in adata.obs["Metadata_Plate"]] - adata.uns["mantispy"]["dataset"] = "cpg0016-jump" - get_logger().info( - "jump_target2: %d wells x %d features from %d source(s)", - adata.n_obs, - adata.n_vars, - adata.obs["Metadata_Source"].nunique(), - ) - return adata + if annotate and plates is not None and set(plates) == set(TARGET2_DEFAULT): + return _fetch_variant("jump_target2", "jump_target2.h5ad", cache_dir) + return _assemble_jump_target2(plates, annotate, cache_dir) + +#: The rehosted pooled_rare variant each ``feature_selected`` flag answers to. +_POOLED_RARE_VARIANTS = { + False: "pooled_rare.h5ad", # 290 x 4417, per barcode, all features + True: "pooled_rare_selected.h5ad", # per barcode, feature selected +} -def pooled_rare(cache_dir: str | Path | None = None, **kwargs: Any) -> AnnData: + +def pooled_rare(cache_dir: str | Path | None = None, *, feature_selected: bool = False) -> AnnData: """Pooled rare variants, 290 barcodes in a pooled screen. ``cpg0032-pooled-rare``, aggregated to one row per barcode rather than per well, so there is no plate or well in ``obs``. + The base and its feature-selected block are pre-built by ``scripts/build_staged_datasets.py`` from the raw gene-normalized table and rehosted on ``scverse-exampledata``, so the loader fetches a single h5ad rather than reassembling it on every call. + Args: cache_dir: Where to keep the download. Defaults to :attr:`mantispy.settings.cache_dir`. - kwargs: Passed to :func:`mantispy.io.read_profiles`. + feature_selected: Return the block after pycytominer-default feature selection instead of all features. Returns: - Barcodes by features, at perturbation resolution, with: + Barcodes by features, at perturbation resolution, read with :func:`mantispy.io.read`, with: ``Metadata_Perturbation``: the variant the barcode expresses (``Metadata_Foci_Barcode_MatchedTo_GeneCode``, 290 of them), the unit the library varied. It is a gene (``"ACTB"``) or a specific coding variant of it (``"ACTB E364K"``). ``Metadata_Perturbation_Type``: ``"orf"``, the variants being expressed from constructs. ``Metadata_Gene`` (the gene the variant belongs to, the token before the first space) and ``Metadata_Allele`` (the full variant label). + + Raises: + ValueError: ``feature_selected`` is not a bool. """ - adata = _profiles("pooled_rare", cache_dir, **kwargs) - obs = as_frame(adata.obs) - code = obs["Metadata_Foci_Barcode_MatchedTo_GeneCode"].astype(str) - obs["Metadata_Perturbation"] = code.astype("category") - obs["Metadata_Perturbation_Type"] = pd.Series("orf", index=obs.index, dtype="category") - # The gene is the token before the first space; the variant "ACTB E364K" belongs to gene "ACTB". - obs["Metadata_Gene"] = code.str.split(" ").str[0].astype("category") - obs["Metadata_Allele"] = code.astype("category") - get_logger().info( - "pooled_rare: %d variants over %d genes x %d features", - adata.n_obs, - int(obs["Metadata_Gene"].nunique()), - adata.n_vars, - ) - return adata + _require_bools(feature_selected=feature_selected) + return _fetch_variant("pooled_rare", _POOLED_RARE_VARIANTS[feature_selected], cache_dir) -def neuropainting(cache_dir: str | Path | None = None, **kwargs: Any) -> AnnData: +def neuropainting(cache_dir: str | Path | None = None) -> AnnData: """Astrocytes and neurons, 1,691 wells imaged at 20x and 63x. ``cpg0038-tegtmeyer-neuropainting``. One plate barcode appears in more than one batch, so the observations are not indexed by plate and well. + The base is pre-built by ``scripts/build_staged_datasets.py`` from the six raw plate tables and rehosted on ``scverse-exampledata``, so the loader fetches a single h5ad rather than reassembling it on every call. + Args: cache_dir: Where to keep the download. Defaults to :attr:`mantispy.settings.cache_dir`. - kwargs: Passed to :func:`mantispy.io.read_profiles`. Returns: - Wells by features, with ``Metadata_CellCount`` and ``Metadata_SiteCount``. + Wells by features at well resolution, with ``Metadata_CellCount`` and ``Metadata_SiteCount``, read with :func:`mantispy.io.read`. Notes: No ``Metadata_Perturbation`` is set. This is a genotype-and-donor comparison (a control-versus-deletion contrast over patient and isogenic lines) rather than a reagent perturbation screen, and its genotype and line columns differ between plates, so the per-plate column intersect keeps none of them. Group with an explicit ``groupby=`` on the column the analysis needs. """ - return _profiles("neuropainting", cache_dir, **kwargs) + return _fetch_variant("neuropainting", "neuropainting.h5ad", cache_dir) -def chroma(cache_dir: str | Path | None = None, **kwargs: Any) -> AnnData: +def chroma(cache_dir: str | Path | None = None) -> AnnData: """Alternative dyes, 3,455 wells across eight channels. ``cpg0029-chroma-pilot``, which images more channels than the five of the standard protocol. One plate barcode appears at several timepoints, so the observations are not indexed by plate and well. The plates hold a set of 91 compounds with known mechanisms; the point of the dataset is the extra dye channels rather than the compounds. + The base is pre-built by ``scripts/build_staged_datasets.py`` from the raw per-plate tables and rehosted on ``scverse-exampledata``, so the loader fetches a single h5ad rather than reassembling it on every call. + Args: cache_dir: Where to keep the download. Defaults to :attr:`mantispy.settings.cache_dir`. - kwargs: Passed to :func:`mantispy.io.read_profiles`. Returns: - Wells by features, with ``Metadata_CellCount`` and ``Metadata_SiteCount`` and: + Wells by features at well resolution, with ``Metadata_CellCount`` and ``Metadata_SiteCount`` and, read with :func:`mantispy.io.read`: ``Metadata_Perturbation``: the compound at its concentration (``"@"``), with the ``negcon`` wells grouped as ``"DMSO"``. @@ -447,27 +479,7 @@ def chroma(cache_dir: str | Path | None = None, **kwargs: Any) -> AnnData: ``Metadata_Compound`` (the compound's common name), ``Metadata_Concentration`` (the platemap's ``mmoles_per_liter``), ``Metadata_MOA`` (the mechanism) and ``Metadata_Control`` (the ``negcon`` wells). """ - adata = _profiles("chroma", cache_dir, **kwargs) - obs = as_frame(adata.obs) - control = (obs["Metadata_control_type"].astype(str) == "negcon").to_numpy() - obs["Metadata_Control"] = control - name = obs["Metadata_Common Name"].astype(str).to_numpy() - dose = pd.to_numeric(obs["Metadata_mmoles_per_liter"], errors="coerce").to_numpy(dtype=float) - compound = np.where(control, "DMSO", name) - obs["Metadata_Compound"] = pd.Categorical(compound) - obs["Metadata_Concentration"] = dose - obs["Metadata_MOA"] = obs["Metadata_MoA"].astype("category") - label = np.where(control, "DMSO", np.char.add(np.char.add(compound.astype(str), "@"), dose.astype(str))) - obs["Metadata_Perturbation"] = pd.Categorical(label) - obs["Metadata_Perturbation_Type"] = pd.Series("compound", index=obs.index, dtype="category") - get_logger().info( - "chroma: %d wells x %d features, %d compounds, %d control wells", - adata.n_obs, - adata.n_vars, - int(pd.Series(name[~control]).nunique()), - int(control.sum()), - ) - return adata + return _fetch_variant("chroma", "chroma.h5ad", cache_dir) #: The spellings the four OASIS batches use for each column mantispy reads, since no two of them agree. @@ -544,20 +556,33 @@ def _oasis_platemaps(cache_dir: str | Path | None) -> pd.DataFrame: return platemap -def oasis_pilot(annotate: bool = True, cache_dir: str | Path | None = None, **kwargs: Any) -> AnnData: +#: The rehosted OASIS-pilot variant each ``aggregated`` flag answers to (both need ``annotate=True``). +_OASIS_PILOT_VARIANTS = { + False: "oasis_pilot.h5ad", # 4604 x 99, well level, annotated + True: "oasis_pilot_agg.h5ad", # perturbation level (modz) +} + + +def oasis_pilot(annotate: bool = True, cache_dir: str | Path | None = None, *, aggregated: bool = False) -> AnnData: """OASIS pilot, 4,604 wells in U2OS and HepaRG, most compounds over a ten-point dose range. ``cpg0033-oasis-pilot``, twelve plates read down to the features they share, over four batches: two of assay development and two that dose 36 compounds in each cell line. It is the dose-response dataset of the package: 28 of those compounds carry six or more concentrations in both U2OS and HepaRG. + The annotated base and its perturbation-level consensus are pre-built by ``scripts/build_staged_datasets.py`` from the raw profile and plate-map tables and rehosted on ``scverse-exampledata``, so the loader fetches a single h5ad rather than joining the plate maps on every call. Passing ``annotate=False`` still reads the raw profiles directly, without the annotation join. + Args: - annotate: Join the plate maps, which supply ``Metadata_Compound``, ``Metadata_Concentration``, ``Metadata_CellLine``, ``Metadata_Control`` (the DMSO wells), ``Metadata_Perturbation`` and ``Metadata_Perturbation_Type`` (``"compound"``). + annotate: Fetch the annotated base, which carries ``Metadata_Compound``, ``Metadata_Concentration``, ``Metadata_CellLine``, ``Metadata_Control`` (the DMSO wells), ``Metadata_Perturbation`` and ``Metadata_Perturbation_Type`` (``"compound"``). + ``False`` reads the raw profiles without the plate-map join and cannot be aggregated. cache_dir: Where to keep the download. Defaults to :attr:`mantispy.settings.cache_dir`. - kwargs: Passed to :func:`mantispy.io.read_profiles`. + aggregated: Return one ``modz`` consensus (Spearman, ``min_replicates=2``) per ``Metadata_Perturbation`` instead of the wells, DMSO kept as the ``"DMSO"`` perturbation. Needs ``annotate=True``. Returns: - Wells by features, indexed by plate and well, with ``Metadata_CellCount`` and ``Metadata_SiteCount``, and the annotation columns above when ``annotate``. + Wells by features at well resolution when ``aggregated`` is ``False``, else the perturbation-level consensus, indexed by plate and well, with ``Metadata_CellCount`` and ``Metadata_SiteCount``, read with :func:`mantispy.io.read`. The annotation columns above are present when ``annotate``. + + Raises: + ValueError: ``aggregated`` is not a bool, or ``aggregated`` is asked for with ``annotate=False``. Notes: The four batches name their plate-map columns differently, so the join reads whichever of ``treatment``/``Compound Name``/``compound`` and ``concentration_uM``/``assay_conc_uM``/``compound_concentration`` each one carries. @@ -570,67 +595,79 @@ def oasis_pilot(annotate: bool = True, cache_dir: str | Path | None = None, **kw ``Metadata_Concentration`` is the dose the well was meant to get, reading a coarser spelling as the finer level it rounds to within each compound, and it names the replicate groups. ``Metadata_ConcentrationRecorded`` keeps what the plate map wrote: read against that column, every dosed compound here carries eighteen levels where ten were plated, and a treatment's wells split across two spellings. """ - adata = _profiles("oasis_pilot", cache_dir, select=lambda name: name.endswith(".csv.gz"), **kwargs) + _require_bools(aggregated=aggregated) + if aggregated and not annotate: + raise ValueError("aggregated needs annotate=True: the consensus groups by the plate-map perturbation") if not annotate: - return adata + return _profiles("oasis_pilot", cache_dir, select=lambda name: name.endswith(".csv.gz")) + return _fetch_variant("oasis_pilot", _OASIS_PILOT_VARIANTS[aggregated], cache_dir) + + +#: The feature sets JUMP-Lite publishes for one set of wells: five learned embeddings and the CellProfiler-equivalent ``cp_measure``. +JUMP_LITE_MODELS = ("openphenom", "dinov2", "dinov2_random", "subcell", "morphem", "cp_measure") + + +def _assemble_jump_lite(model: str, annotate: bool, cache_dir: str | Path | None) -> AnnData: + """Assemble one JUMP-Lite feature set: the model's wells joined to their cell count, row-sorted and annotated. + + The raw pipeline the ``annotate=True`` :func:`jump_lite` used before the per-model objects were staged; it + lives here so the drift check can rebuild each hosted object, and so the loader can still serve the + un-annotated ``annotate=False`` path. + """ + adata = _profiles("jump_lite", cache_dir, select=lambda name: name == f"{model}.parquet") + if model != "cp_measure": + empty = empty_annotation(adata.var_names) + adata.var[empty.columns] = empty + + (counts_path,) = _files("jump_lite", cache_dir, select=lambda name: name == "cell_count.parquet") + counts = pd.read_parquet(counts_path, columns=["Metadata_id", "cell_count"]) obs = as_frame(adata.obs) - merged = obs.merge(_oasis_platemaps(cache_dir), on=["Metadata_plate_map_name", "Metadata_Well"], how="left") - merged.index = obs.index - if unmatched := int(merged["Metadata_Compound"].isna().sum()): - get_logger().warning("oasis_pilot: %d of %d wells have no plate-map row", unmatched, len(merged)) - adata.obs["Metadata_Compound"] = merged["Metadata_Compound"].to_numpy() - adata.obs["Metadata_Concentration"] = merged["Metadata_Concentration"].to_numpy(dtype=float) - adata.obs["Metadata_ConcentrationRecorded"] = merged["Metadata_ConcentrationRecorded"].to_numpy(dtype=float) - adata.obs["Metadata_CellLine"] = merged["Metadata_CellLine"].to_numpy() - adata.obs["Metadata_Control"] = merged["Metadata_Compound"].astype(str).str.upper().eq("DMSO").to_numpy() - is_control = np.asarray(adata.obs["Metadata_Control"], dtype=bool) - adata.obs["Metadata_Perturbation"] = pd.Categorical( - np.where( - is_control, - "DMSO", - merged["Metadata_Compound"].astype(str) + "@" + merged["Metadata_Concentration"].astype(str), - ) - ) - adata.obs["Metadata_Perturbation_Type"] = pd.Series("compound", index=adata.obs_names, dtype="category") + joined = obs.merge(counts, on="Metadata_id", how="left", validate="1:1") + joined.index = obs.index + if unmatched := int(joined["cell_count"].isna().sum()): + # Downstream, a missing count reads as a well with no cells. + get_logger().warning("jump_lite(%s): %d well(s) have no cell count in the count table", model, unmatched) + adata.obs = joined.rename(columns={"cell_count": "Metadata_CellCount"}) + order = as_frame(adata.obs).sort_values(["Metadata_Source", "Metadata_Plate", "Metadata_Well"], kind="stable").index + adata = adata[order].copy() + + if annotate: + from mantispy.pp._annotate import annotate_jump + + annotate_jump(adata) get_logger().info( - "OASIS pilot: %d wells x %d features, %d compounds over %d concentrations, %d control wells", + "jump_lite(%s): %d wells x %d features over %d source(s)", + model, adata.n_obs, adata.n_vars, - int(merged.loc[~is_control, "Metadata_Compound"].nunique()), - int(merged["Metadata_Concentration"].nunique()), - int(is_control.sum()), + int(as_frame(adata.obs)["Metadata_Source"].nunique()), ) return adata -#: The feature sets JUMP-Lite publishes for one set of wells: five learned embeddings and the CellProfiler-equivalent ``cp_measure``. -JUMP_LITE_MODELS = ("openphenom", "dinov2", "dinov2_random", "subcell", "morphem", "cp_measure") - - -def jump_lite( - model: str = "openphenom", annotate: bool = True, cache_dir: str | Path | None = None, **kwargs: Any -) -> AnnData: +def jump_lite(model: str = "openphenom", annotate: bool = True, cache_dir: str | Path | None = None) -> AnnData: """JUMP-Lite Target-2: 1,536 wells, four imaging sites, one feature set at a time. ``cpg0016-jump``, the compact benchmark of :cite:t:`Munoz_2026`. Four plates of the JUMP Target-2 plate map, one from each of ``source_3``, ``source_4``, ``source_5`` and ``source_6``, so the four batches are four different laboratories running the same 302 compounds with 64 DMSO wells each. Every ``model`` covers the same 1,536 wells, which is what makes this a comparison rather than six datasets: the rows, their order and the metadata are identical and only the feature block changes. - The files list the wells in a different order for every model, so the rows are sorted by source, plate and well. + The files list the wells in a different order for every model, so the rows are sorted by plate and well and every model is aligned. Five are learned embeddings and one, ``"cp_measure"``, is the CellProfiler-equivalent measurement of the same images. + The annotated object for each model is pre-built by ``scripts/build_staged_datasets.py`` and rehosted on ``scverse-exampledata``, so the annotated call fetches a single h5ad rather than joining the count and annotation tables on every call. Passing ``annotate=False`` still assembles the un-annotated wells directly. + Args: model: Which feature set to read, one of ``ds.JUMP_LITE_MODELS``. ``"dinov2_random"`` is the same architecture with untrained weights, which is the null model the benchmark scores the others against. - annotate: Join the JUMP well and compound tables, which name the compound of each well. - Downloads about 14 MB once and caches it. + annotate: Fetch the annotated object, which joins the JUMP well and compound tables that name the compound of each well. + ``False`` assembles the un-annotated wells without that join. cache_dir: Where to keep the download. Defaults to :attr:`mantispy.settings.cache_dir`. - kwargs: Passed to :func:`mantispy.io.read_profiles`. Returns: - Wells by features at well resolution, indexed by plate and well, with ``Metadata_Source``, ``Metadata_Batch``, ``Metadata_Plate``, ``Metadata_Well``, ``Metadata_CellCount`` and, when annotated, ``Metadata_JCP2022``, ``Metadata_Perturbation``, ``Metadata_Perturbation_Type`` (``"compound"``), ``Metadata_InChIKey`` and ``Metadata_Control``. + Wells by features at well resolution, indexed by plate and well, read with :func:`mantispy.io.read` when annotated, with ``Metadata_Source``, ``Metadata_Batch``, ``Metadata_Plate``, ``Metadata_Well``, ``Metadata_CellCount`` and, when annotated, ``Metadata_JCP2022``, ``Metadata_Perturbation``, ``Metadata_Perturbation_Type`` (``"compound"``), ``Metadata_InChIKey`` and ``Metadata_Control``. Raises: ValueError: ``model`` is not one of ``ds.JUMP_LITE_MODELS``. @@ -654,37 +691,9 @@ def jump_lite( """ if model not in JUMP_LITE_MODELS: raise ValueError(f"model must be one of {JUMP_LITE_MODELS}, got {model!r}") - - adata = _profiles("jump_lite", cache_dir, select=lambda name: name == f"{model}.parquet", **kwargs) - if model != "cp_measure": - empty = empty_annotation(adata.var_names) - adata.var[empty.columns] = empty - - (counts_path,) = _files("jump_lite", cache_dir, select=lambda name: name == "cell_count.parquet") - counts = pd.read_parquet(counts_path, columns=["Metadata_id", "cell_count"]) - - obs = as_frame(adata.obs) - joined = obs.merge(counts, on="Metadata_id", how="left", validate="1:1") - joined.index = obs.index - if unmatched := int(joined["cell_count"].isna().sum()): - # Downstream, a missing count reads as a well with no cells. - get_logger().warning("jump_lite(%s): %d well(s) have no cell count in the count table", model, unmatched) - adata.obs = joined.rename(columns={"cell_count": "Metadata_CellCount"}) - order = as_frame(adata.obs).sort_values(["Metadata_Source", "Metadata_Plate", "Metadata_Well"], kind="stable").index - adata = adata[order].copy() - - if annotate: - from mantispy.pp._annotate import annotate_jump - - annotate_jump(adata) - get_logger().info( - "jump_lite(%s): %d wells x %d features over %d source(s)", - model, - adata.n_obs, - adata.n_vars, - int(as_frame(adata.obs)["Metadata_Source"].nunique()), - ) - return adata + if not annotate: + return _assemble_jump_lite(model, annotate=False, cache_dir=cache_dir) + return _fetch_variant("jump_lite", f"jump_lite_{model}.h5ad", cache_dir) def jump_lite_targets(cache_dir: str | Path | None = None) -> pd.DataFrame: @@ -713,32 +722,47 @@ def jump_lite_targets(cache_dir: str | Path | None = None) -> pd.DataFrame: return frame.drop_duplicates().reset_index(drop=True) -def jump_crispr(annotate: bool = True, cache_dir: str | Path | None = None, **kwargs: Any) -> AnnData: +#: The rehosted jump_crispr variant each ``aggregated`` flag answers to (both need ``annotate=True``). +_JUMP_CRISPR_VARIANTS = { + False: "jump_crispr.h5ad", # 51185 x 595, well level, annotated + True: "jump_crispr_agg.h5ad", # gene level (modz) +} + + +def jump_crispr(annotate: bool = True, cache_dir: str | Path | None = None, *, aggregated: bool = False) -> AnnData: """The assembled JUMP CRISPR arm, 51,185 wells of knockouts in U2OS cells. - ``cpg0016-jump-assembled``, the ``v1.0a`` well-position-corrected and feature-selected parquet, and the largest dataset here at 180 MB. + ``cpg0016-jump-assembled``, the ``v1.0a`` well-position-corrected and feature-selected parquet. jump-profiling-recipe has corrected it for well position and cell count, normalized it and selected its features, and has not yet sphered it. The cell counts are the ones the recipe regresses out of these profiles; their ``Cells_Count_Count`` feature was normalized with the rest and is no longer a count. + The annotated base and its gene-level consensus are pre-built by ``scripts/build_staged_datasets.py`` from the raw parquet and rehosted on ``scverse-exampledata``, so the loader fetches a single h5ad rather than reassembling the base on every call. Passing ``annotate=False`` still reads the raw profiles directly, without the annotation join. + Args: - annotate: Join JUMP's CRISPR annotation, which names the gene each well's guides target and which wells are controls. + annotate: Fetch the annotated base, which names the gene each well's guides target and which wells are controls. + ``False`` reads the raw profiles without the annotation join and cannot be aggregated. cache_dir: Where to keep the download. Defaults to :attr:`mantispy.settings.cache_dir`. - kwargs: Passed to :func:`mantispy.io.read_profiles`. + aggregated: Return one ``modz`` consensus (Spearman, ``min_replicates=2``) per ``Metadata_Gene`` over the guides instead of the wells. Needs ``annotate=True``. Returns: - Wells by features, indexed by plate and well, with ``Metadata_JCP2022`` and ``Metadata_CellCount`` and, when annotated, ``Metadata_Gene`` and ``Metadata_Perturbation`` (the gene symbol; its guides are the replicates), ``Metadata_Perturbation_Type`` (``"crispr"``), ``Metadata_Control_Type`` (``"negcon"``, ``"poscon"`` or ``"trt"``), ``Metadata_Control`` (the no-guide and non-targeting wells) and ``Metadata_ChromosomeArm``. + Wells by features at well resolution when ``aggregated`` is ``False``, else the gene-level consensus, indexed by plate and well, read with :func:`mantispy.io.read`, with ``Metadata_JCP2022`` and ``Metadata_CellCount`` and, when annotated, ``Metadata_Gene`` and ``Metadata_Perturbation`` (the gene symbol; its guides are the replicates), ``Metadata_Perturbation_Type`` (``"crispr"``), ``Metadata_Control_Type`` (``"negcon"``, ``"poscon"`` or ``"trt"``), ``Metadata_Control`` (the no-guide and non-targeting wells) and ``Metadata_ChromosomeArm``. + + Raises: + ValueError: ``aggregated`` is not a bool, or ``aggregated`` is asked for with ``annotate=False``. References: :cite:t:`Chandrasekaran_2023`. """ - (counts,) = _files("_jump_cell_counts", cache_dir) - adata = _profiles("jump_crispr", cache_dir, platemap=_read_counts(counts), **kwargs) - if annotate: - from mantispy.pp._annotate import annotate_jump - - annotate_jump(adata, kind="crispr") - return adata + _require_bools(aggregated=aggregated) + if aggregated and not annotate: + raise ValueError("aggregated needs annotate=True: the consensus groups by the annotated gene") + if not annotate: + (counts,) = _files("_jump_cell_counts", cache_dir) + return _profiles( + "jump_crispr", cache_dir, select=lambda name: not name.endswith(".h5ad"), platemap=_read_counts(counts) + ) + return _fetch_variant("jump_crispr", _JUMP_CRISPR_VARIANTS[aggregated], cache_dir) def _finish_guide_screen(adata: AnnData, name: str) -> AnnData: @@ -781,7 +805,45 @@ def _finish_guide_screen(adata: AnnData, name: str) -> AnnData: ) -def scallops_arv471(cache_dir: str | Path | None = None) -> AnnData: +def _assemble_scallops_arv471(cache_dir: str | Path | None = None) -> AnnData: + """Assemble the SCALLOPS ARV-471 base: the ARV-471 cells with their gene, guide and control. + + The raw pipeline the both-flags-False :func:`mt.ds.scallops_arv471` used before the base was staged; it + lives here so the drift check can rebuild the hosted base from the raw parquet. + """ + (path,) = _files("scallops_arv471", cache_dir, select=lambda name: not name.endswith(".h5ad")) + df = pd.read_parquet(path, columns=[*_SCALLOPS_FEATURES, *_SCALLOPS_SOURCE]) + df = df[df["Condition"].astype(str) == "ARV-471"] + df = df[~df["Cells_Location_IntersectsBoundary_IF"].astype(bool)] + before = len(df) + df = df.dropna(subset=list(_SCALLOPS_FEATURES)) + report_drop("cell(s) with a missing phenotype feature", before - len(df), before) + + gene = df["gene_symbol"].astype(str).to_numpy() + guide = df["sgRNA_id"].astype(str).to_numpy() + is_ntc = gene == "NTC" + frame = df[list(_SCALLOPS_FEATURES)].reset_index(drop=True) + frame["Metadata_Gene"] = np.where(is_ntc, "nontargeting", gene) + frame["Metadata_sgRNA"] = guide + frame["Metadata_Control_Type"] = df["type"].astype(str).to_numpy() + frame["Metadata_Control"] = is_ntc + frame["Metadata_Perturbation"] = guide + frame["Metadata_Plate"] = df["plate"].astype(str).to_numpy() + # The raw well is a rowless integer; prefix a synthetic row letter so from_dataframe's normalize_well can pad it. + frame["Metadata_Well"] = ("W" + df["well"].astype(int).astype(str)).to_numpy() + + adata = from_dataframe(frame, resolution="cell") + return _finish_guide_screen(adata, "scallops_arv471") + + +#: The rehosted scallops_arv471 variant each ``aggregated`` flag answers to. +_SCALLOPS_VARIANTS = { + False: "scallops_arv471.h5ad", # 2021814 x 9, cell level + True: "scallops_arv471_agg.h5ad", # guide level (median) +} + + +def scallops_arv471(cache_dir: str | Path | None = None, *, aggregated: bool = False) -> AnnData: """SCALLOPS ARV-471, single cells of an optical pooled screen under an estrogen-receptor degrader. The drug arm of a genome-scale optical pooled CRISPR screen from ``Genentech/scallops-manuscript``, its Figure 3 table. @@ -791,14 +853,16 @@ def scallops_arv471(cache_dir: str | Path | None = None) -> AnnData: This loads only the ARV-471 condition, at single-cell resolution, so a hit is a guide whose cells sit away from the non-targeting cells in the phenotype space. The matched DMSO condition and the barcode-calling columns are left in the upstream file. - Downloads about 205 MB once, checked against a pinned sha256, and subsets it on read. + + The base and its guide-level aggregate are pre-built by ``scripts/build_staged_datasets.py`` from the raw parquet and rehosted on ``scverse-exampledata``, so the loader fetches a single h5ad rather than filtering and reassembling the cells on every call. Args: cache_dir: Where to keep the download. Defaults to :attr:`mantispy.settings.cache_dir`. + aggregated: Return one median profile per ``(Metadata_Gene, Metadata_sgRNA)`` guide instead of the cells, with ``Metadata_CellCount``. Returns: - Cells by nine phenotype features at cell resolution, with: + Cells by nine phenotype features at cell resolution (one median per guide when ``aggregated``), read with :func:`mantispy.io.read`, with: ``Metadata_Gene``: the gene the cell's guide targets, with the non-targeting guides written as ``"nontargeting"`` (the upstream ``NTC``), the spelling the analysis functions read. @@ -828,31 +892,13 @@ def scallops_arv471(cache_dir: str | Path | None = None) -> AnnData: Cells missing any phenotype feature are dropped too, so every returned cell has a full feature vector. A cell carries no count. - Aggregate to a guide-level profile with ``mt.tl.aggregate(adata, by=("Metadata_Gene", "Metadata_sgRNA"))``, which writes ``Metadata_CellCount``. - """ - (path,) = _files("scallops_arv471", cache_dir) - df = pd.read_parquet(path, columns=[*_SCALLOPS_FEATURES, *_SCALLOPS_SOURCE]) - df = df[df["Condition"].astype(str) == "ARV-471"] - df = df[~df["Cells_Location_IntersectsBoundary_IF"].astype(bool)] - before = len(df) - df = df.dropna(subset=list(_SCALLOPS_FEATURES)) - report_drop("cell(s) with a missing phenotype feature", before - len(df), before) - - gene = df["gene_symbol"].astype(str).to_numpy() - guide = df["sgRNA_id"].astype(str).to_numpy() - is_ntc = gene == "NTC" - frame = df[list(_SCALLOPS_FEATURES)].reset_index(drop=True) - frame["Metadata_Gene"] = np.where(is_ntc, "nontargeting", gene) - frame["Metadata_sgRNA"] = guide - frame["Metadata_Control_Type"] = df["type"].astype(str).to_numpy() - frame["Metadata_Control"] = is_ntc - frame["Metadata_Perturbation"] = guide - frame["Metadata_Plate"] = df["plate"].astype(str).to_numpy() - # The raw well is a rowless integer; prefix a synthetic row letter so from_dataframe's normalize_well can pad it. - frame["Metadata_Well"] = ("W" + df["well"].astype(int).astype(str)).to_numpy() + The ``aggregated`` variant is one median profile per ``(Metadata_Gene, Metadata_sgRNA)`` guide, which writes ``Metadata_CellCount``. - adata = from_dataframe(frame, resolution="cell") - return _finish_guide_screen(adata, "scallops_arv471") + Raises: + ValueError: ``aggregated`` is not a bool. + """ + _require_bools(aggregated=aggregated) + return _fetch_variant("scallops_arv471", _SCALLOPS_VARIANTS[aggregated], cache_dir) #: The upstream parquet stores these as its MultiIndex; every other column is a CellStats morphology feature. @@ -862,21 +908,73 @@ def scallops_arv471(cache_dir: str | Path | None = None) -> AnnData: _CP_POSH_CONTROLS = ("nontargeting", "intergenic") -def cp_posh(cache_dir: str | Path | None = None) -> AnnData: +def _assemble_cp_posh(cache_dir: str | Path | None = None) -> AnnData: + """Assemble the cp-POSH base: the cells with their gene, guide, plate, well and control. + + The raw pipeline the both-flags-False :func:`mt.ds.cp_posh` used before the base was staged; it lives here + so the drift check can rebuild the hosted base from the raw parquet. + """ + import anndata as ad + + (path,) = _files("cp_posh", cache_dir, select=lambda name: not name.endswith(".h5ad")) + df = pd.read_parquet(path).reset_index() + features = [column for column in df.columns if column not in _CP_POSH_METADATA] + + gene = df["gene_id"].astype(str).to_numpy() + guide = df["barcode"].astype(str).to_numpy() + # plate_well is "_", e.g. "EL37_B04"; the well is the segment after the last "_" (wells carry none). + well = df["plate_well"].astype(str).str.rsplit("_", n=1).str[-1].to_numpy() + obs = pd.DataFrame( + { + "Metadata_Gene": gene, + "Metadata_sgRNA": guide, + "Metadata_Perturbation": guide, + "Metadata_Plate": df["plate"].astype(str).to_numpy(), + "Metadata_Well": well, + "Metadata_Control": np.isin(gene, _CP_POSH_CONTROLS), + }, + index=pd.Index(df["ID"].astype(str).to_numpy()), + ) + obs = categorize_metadata(obs) + adata = ad.AnnData(X=df[features].to_numpy(dtype=np.float32), obs=obs, var=empty_annotation(features)) + stamp(adata, resolution="cell") + return _finish_guide_screen(adata, "cp_posh") + + +#: The (aggregated, feature_selected) combination each rehosted cp_posh variant answers to. +_CP_POSH_VARIANTS = { + (False, False): "cp_posh.h5ad", # 163090 x 1278, cell level, all features + (False, True): "cp_posh_selected.h5ad", # cell level, feature selected + (True, False): "cp_posh_agg.h5ad", # guide level (median), all features + (True, True): "cp_posh_agg_selected.h5ad", # guide level (median), feature selected +} + + +def cp_posh( + cache_dir: str | Path | None = None, *, aggregated: bool = False, feature_selected: bool = False +) -> AnnData: """Single cells of insitro cp-POSH, a broad-morphology pooled CRISPR Cell Painting screen. The 124-gene proof-of-concept dataset from ``insitro/cp-posh``: A549 cells carrying a pooled CRISPR-knockout library, stained with a six-channel Cell Painting panel (WGA, a mitochondrial probe, phalloidin, concanavalin A, DAPI and a marker round) and read by in-situ sequencing of the guide barcodes. Each cell gets a broad, untargeted morphology profile of about 1,278 CellStats features rather than the handful of hand-picked readouts a targeted screen keeps, so it is the broad-morphology complement to :func:`scallops_arv471`. The features are already well-normalized by the authors, so :func:`~mantispy.pp.normalize` is not needed before analysis; a per-plate control normalization would re-do work already done. - Downloads about 1.6 GB once, checked against a pinned sha256. + + The base and its variants are pre-built by ``scripts/build_staged_datasets.py`` from the raw parquet and rehosted on ``scverse-exampledata``, so the loader fetches a single h5ad rather than reassembling the base on every call. The two flags select the variant: + + - both ``False``: the cells with every feature. + - ``feature_selected=True``: the cells after pycytominer-default feature selection. + - ``aggregated=True``: one median profile per ``(Metadata_Gene, Metadata_sgRNA)`` guide, with ``Metadata_CellCount``. + - ``aggregated=True, feature_selected=True``: that same guide aggregate on the feature-selected block. Args: cache_dir: Where to keep the download. Defaults to :attr:`mantispy.settings.cache_dir`. + aggregated: Return the guide-level median aggregate instead of the cells. + feature_selected: Return the feature-selected block instead of all features. Returns: - Cells by about 1,278 CellStats morphology features at cell resolution, indexed by the upstream cell ``ID``, with: + Cells by about 1,278 CellStats morphology features at cell resolution (one median per guide when ``aggregated``), read with :func:`mantispy.io.read`, indexed by the upstream cell ``ID`` on the base, with: ``Metadata_Gene``: the gene the cell's guide targets, taken from the upstream ``gene_id``. The two control classes keep their upstream spellings, ``"nontargeting"`` (the non-targeting guides) and ``"intergenic"`` (guides against intergenic regions); ``"nontargeting"`` is the spelling the analysis functions read. @@ -902,33 +1000,13 @@ def cp_posh(cache_dir: str | Path | None = None) -> AnnData: Anything that reads ``var["feature_group"]`` or ``var["channel"]`` has nothing to work with here. A cell carries no count. - Aggregate to a guide-level profile with ``mt.tl.aggregate(adata, by=("Metadata_Gene", "Metadata_sgRNA"))``, which writes ``Metadata_CellCount``. - """ - import anndata as ad + The ``aggregated`` variant is one median profile per ``(Metadata_Gene, Metadata_sgRNA)`` guide, which writes ``Metadata_CellCount``. - (path,) = _files("cp_posh", cache_dir) - df = pd.read_parquet(path).reset_index() - features = [column for column in df.columns if column not in _CP_POSH_METADATA] - - gene = df["gene_id"].astype(str).to_numpy() - guide = df["barcode"].astype(str).to_numpy() - # plate_well is "_", e.g. "EL37_B04"; the well is the segment after the last "_" (wells carry none). - well = df["plate_well"].astype(str).str.rsplit("_", n=1).str[-1].to_numpy() - obs = pd.DataFrame( - { - "Metadata_Gene": gene, - "Metadata_sgRNA": guide, - "Metadata_Perturbation": guide, - "Metadata_Plate": df["plate"].astype(str).to_numpy(), - "Metadata_Well": well, - "Metadata_Control": np.isin(gene, _CP_POSH_CONTROLS), - }, - index=pd.Index(df["ID"].astype(str).to_numpy()), - ) - obs = categorize_metadata(obs) - adata = ad.AnnData(X=df[features].to_numpy(dtype=np.float32), obs=obs, var=empty_annotation(features)) - stamp(adata, resolution="cell") - return _finish_guide_screen(adata, "cp_posh") + Raises: + ValueError: ``aggregated`` or ``feature_selected`` is not a bool. + """ + _require_bools(aggregated=aggregated, feature_selected=feature_selected) + return _fetch_variant("cp_posh", _CP_POSH_VARIANTS[aggregated, feature_selected], cache_dir) def corum(cache_dir: str | Path | None = None) -> pd.DataFrame: @@ -983,7 +1061,7 @@ def _assemble_cells(entry: DatasetEntry, cache_dir: str | Path | None, *, annota source = str(entry.metadata["source"]) channels = [str(channel) for channel in entry.metadata["channels"]] - paths = _files("jump_cells", cache_dir) + paths = _files("jump_cells", cache_dir, select=lambda name: not name.endswith(".h5ad")) parts = [_read_site(directory, source, channels) for directory in sorted({path.parent for path in paths})] # Every field of view numbers its own images from one, so each part's numbers are shifted past the ones before it. images, offset = [], 0 @@ -1014,13 +1092,6 @@ def _assemble_cells(entry: DatasetEntry, cache_dir: str | Path | None, *, annota return adata -def _select(adata: AnnData, path: Path) -> AnnData: - """The selected features of `adata`, kept at `path` so the next call reads only those.""" - chosen = subset_features(adata) - write(chosen, path) - return chosen - - def _mark_selected(adata: AnnData) -> None: """Write the feature-selection mask into ``var["selected"]``, as :func:`mantispy.pp.feature_select` does. @@ -1063,7 +1134,33 @@ def jump_export(cache_dir: str | Path | None = None) -> Path: return paths[0].parent -def jump_cells(annotate: bool = True, selected: bool = False, cache_dir: str | Path | None = None) -> AnnData: +def _assemble_jump_cells(cache_dir: str | Path | None = None) -> AnnData: + """Assemble the annotated jump_cells base from the 480 raw CellProfiler tables. + + The raw pipeline the ``annotate=True`` :func:`jump_cells` used before the base was staged; it lives here so + :func:`mantispy.ds._build.build_jump_cells` can rebuild the hosted variants from the raw inputs. + """ + return _assemble_cells(_DATASETS["jump_cells"], cache_dir, annotate=True) + + +def _jump_cells_raw(cache_dir: str | Path | None) -> AnnData: + """The un-annotated assembled cells, cached on disk (the ``annotate=False`` path, which is not staged).""" + entry = _DATASETS["jump_cells"] + root = Path(cache_dir or settings.cache_dir) + # The name fingerprints the schema, the pinned files and the assembly version, so a stale file is never read. + material = f"{_ASSEMBLY_VERSION}:" + "".join(str(file.sha256) for file in entry.files) + fingerprint = hashlib.sha256(material.encode()).hexdigest()[:12] + derived = root / f"jump_cells-{SCHEMA_VERSION}-{fingerprint}-raw.h5ad" + if derived.exists(): + return read(derived) + adata = _assemble_cells(entry, cache_dir, annotate=False) + write(adata, derived) + return adata + + +def jump_cells( + annotate: bool = True, selected: bool = False, cache_dir: str | Path | None = None, *, aggregated: bool = False +) -> AnnData: """Single cells from one JUMP plate, as CellProfiler measured them. Twenty-four wells of ``BR00121438`` at four fields of view each: eight DMSO wells, four compounds with both of their replicate wells, and eight more compounds at one well. @@ -1071,48 +1168,44 @@ def jump_cells(annotate: bool = True, selected: bool = False, cache_dir: str | P The same plate's well-level profiles are :func:`jump_target2`, so a profile aggregated from these cells can be compared with the one the consortium published. - The first call downloads about 1.5 GB of CellProfiler output, reads 480 tables and writes the assembled object next to them, which takes a few minutes. - Later calls read that one file. + The annotated cells, their feature-selected block and their well-level aggregate are pre-built by ``scripts/build_staged_datasets.py`` from the 480 raw CellProfiler tables and rehosted on ``scverse-exampledata``, so the loader fetches a single h5ad rather than reassembling the object on every call. Passing ``annotate=False`` still assembles the raw, un-annotated cells locally, downloading about 1.5 GB of CellProfiler output and caching the assembled object. Args: - annotate: Join the JUMP annotation, which supplies ``Metadata_Perturbation`` and ``Metadata_Control``. - Downloads another 14 MB. - Needed for `selected`, which is computed against the controls. - selected: Return only the features ``var["selected"]`` marks, as :func:`mantispy.pp.subset_features` would. - The subset is kept beside the whole object, so a notebook that only wants the reduced one reads 87 MB instead of 308 MB. + annotate: Fetch the annotated cells, which carry ``Metadata_Perturbation`` and ``Metadata_Control``. + ``False`` assembles the raw cells locally, without the annotation join. + Needed for `selected` and `aggregated`. + selected: Return only the features ``var["selected"]`` marks, as :func:`mantispy.pp.subset_features` would (87 MB instead of 308 MB). cache_dir: Where to keep the download. Defaults to :attr:`mantispy.settings.cache_dir`. + aggregated: Return one median profile per well (``Metadata_Plate``, ``Metadata_Well``) instead of the cells, with ``Metadata_CellCount`` and ``Metadata_SiteCount``, so it lines up with the well-level :func:`jump_target2`. Raises: - KeyError: `selected` was asked for without `annotate`, so there are no controls to select against. + KeyError: `selected` or `aggregated` was asked for without `annotate`, so there is no annotation to select or group against. + ValueError: `selected` and `aggregated` were both asked for; there is no aggregated feature-selected variant. Returns: - Cells by features at cell resolution, carrying ``Metadata_Source``, ``Metadata_Plate``, ``Metadata_Well``, ``Metadata_Site`` and, when annotated, ``Metadata_JCP2022``, ``Metadata_Perturbation``, ``Metadata_Perturbation_Type`` (``"compound"``), ``Metadata_InChIKey`` and ``Metadata_Control``. + Cells by features at cell resolution (one median per well when ``aggregated``), read with :func:`mantispy.io.read`, carrying ``Metadata_Source``, ``Metadata_Plate``, ``Metadata_Well``, ``Metadata_Site`` and, when annotated, ``Metadata_JCP2022``, ``Metadata_Perturbation``, ``Metadata_Perturbation_Type`` (``"compound"``), ``Metadata_InChIKey`` and ``Metadata_Control``. When annotated, ``var["selected"]`` marks the features feature selection keeps, so the object can be reduced with ``adata[:, adata.var["selected"]]`` the way scanpy's ``highly_variable`` is used. - A cell carries no count. - :func:`mantispy.tl.aggregate` writes ``Metadata_CellCount`` over the four fields read and a ``Metadata_SiteCount`` of four, so a well counts about four ninths of the cells :func:`jump_target2` gives it over all nine. + The ``aggregated`` well profiles carry ``Metadata_CellCount`` over the four fields read and a ``Metadata_SiteCount`` of four, so a well counts about four ninths of the cells :func:`jump_target2` gives it over all nine. References: :cite:t:`Chandrasekaran_2023`. """ if selected and not annotate: raise KeyError("selected=True needs annotate=True: the mask is computed against the negative controls") - - entry = _DATASETS["jump_cells"] - root = Path(cache_dir or settings.cache_dir) - # The name fingerprints the schema, the pinned files and the assembly version, so a stale file is never read. - material = f"{_ASSEMBLY_VERSION}:" + "".join(str(file.sha256) for file in entry.files) - fingerprint = hashlib.sha256(material.encode()).hexdigest()[:12] - stem = f"jump_cells-{SCHEMA_VERSION}-{fingerprint}-{'annotated' if annotate else 'raw'}" - derived, subset = root / f"{stem}.h5ad", root / f"{stem}-selected.h5ad" - if selected and subset.exists(): - return read(subset) - if derived.exists(): - return _select(read(derived), subset) if selected else read(derived) - - adata = _assemble_cells(entry, cache_dir, annotate=annotate) - write(adata, derived) - return _select(adata, subset) if selected else adata + if aggregated and not annotate: + raise KeyError("aggregated=True needs annotate=True: the well profiles carry the annotation") + if selected and aggregated: + raise ValueError("jump_cells has no aggregated feature-selected variant; pass one of selected or aggregated") + if not annotate: + return _jump_cells_raw(cache_dir) + if aggregated: + target = "jump_cells_agg.h5ad" + elif selected: + target = "jump_cells_selected.h5ad" + else: + target = "jump_cells.h5ad" + return _fetch_variant("jump_cells", target, cache_dir) def jump_plate(cache_dir: str | Path | None = None, **kwargs: Any) -> SpatialData: diff --git a/src/mantispy/ds/registry.yaml b/src/mantispy/ds/registry.yaml index 6634e6c..581d5e5 100644 --- a/src/mantispy/ds/registry.yaml +++ b/src/mantispy/ds/registry.yaml @@ -1939,6 +1939,14 @@ datasets: - name: 2023_06_05_24W_RD3_poscon__2023_06_05_24W_RD3_poscon_gene_normalized_ALLBATCHES___ALLPLATES___ALLWELLS.csv.gz s3_key: cpg0032-pooled-rare/broad/workspace/profiles/2023_06_05_24W_RD3_poscon/2023_06_05_24W_RD3_poscon_gene_normalized_ALLBATCHES___ALLPLATES___ALLWELLS.csv.gz sha256: 93d1f3dec58676b266fece87d950945cdc31da91e22b8c8101218091198d6eab + # Staged variants rehosted on scverse-exampledata (explicit url overrides base_url, so the raw table above + # stays on cellpainting-gallery). Regenerate with scripts/build_staged_datasets.py. + - name: pooled_rare.h5ad + url: https://scverse-exampledata.s3.eu-west-1.amazonaws.com/mantispy/pooled_rare/pooled_rare.h5ad + sha256: abacc60b3e9d011bc079c02857bc5a4c8c3d2c6bed0e54e04224a325483f9d02 + - name: pooled_rare_selected.h5ad + url: https://scverse-exampledata.s3.eu-west-1.amazonaws.com/mantispy/pooled_rare/pooled_rare_selected.h5ad + sha256: b197f748c123ff3b9d1c0ae589f809519b1e72db41ca912a87a1b9180f6b3eb9 rohban: type: mantispy @@ -2002,6 +2010,11 @@ datasets: - name: NCP_NEURONS_2_63x__BR00132673__BR00132673_normalized_feature_select_batch.csv.gz s3_key: cpg0038-tegtmeyer-neuropainting/broad/workspace/profiles/NCP_NEURONS_2_63x/BR00132673/BR00132673_normalized_feature_select_batch.csv.gz sha256: 41cee30b1c7386b27d157f5c23dc9ee22afcc6e24cbf5395c2f04289dfb858ef + # Staged base rehosted on scverse-exampledata (explicit url overrides base_url, so the raw plate tables + # above stay on cellpainting-gallery). Regenerate with scripts/build_staged_datasets.py. + - name: neuropainting.h5ad + url: https://scverse-exampledata.s3.eu-west-1.amazonaws.com/mantispy/neuropainting/neuropainting.h5ad + sha256: 897f8b6f6c237dd132da9bdf5be57bdd30f7679cd0cafae5b23067b887bba893 chroma: type: mantispy @@ -2038,6 +2051,11 @@ datasets: - name: 2024_02_02_Batch8__BR00122249_4h__BR00122249_4h_normalized_feature_select_negcon_batch.csv.gz s3_key: cpg0029-chroma-pilot/broad/workspace/profiles/2024_02_02_Batch8/BR00122249_4h/BR00122249_4h_normalized_feature_select_negcon_batch.csv.gz sha256: f6c4b24b702070b1fb434b4a4482eb1be6041c50060de3b16a3582d2713ab474 + # Staged base rehosted on scverse-exampledata (explicit url overrides base_url, so the raw plate tables + # above stay on cellpainting-gallery). Regenerate with scripts/build_staged_datasets.py. + - name: chroma.h5ad + url: https://scverse-exampledata.s3.eu-west-1.amazonaws.com/mantispy/chroma/chroma.h5ad + sha256: e747864a9f099ec2594a147e29844cd9e3103606b2d0181a3b2b04b18f1437b3 oasis_pilot: type: mantispy @@ -2121,6 +2139,14 @@ datasets: - name: 2025_07_29_OASIS_HepaRG_Ptx_AD2__BR00147879__platemap.txt s3_key: cpg0033-oasis-pilot/broad/workspace/metadata/2025_07_29_OASIS_HepaRG_Ptx_AD2/platemap/BR00147879.txt sha256: eda00ee6e9d50e91a935ae5d505c9dc751ac0a81cb46968ccd22147b138acbac + # Staged variants rehosted on scverse-exampledata (explicit url overrides base_url, so the raw profile and + # plate-map tables above stay on cellpainting-gallery). Regenerate with scripts/build_staged_datasets.py. + - name: oasis_pilot.h5ad + url: https://scverse-exampledata.s3.eu-west-1.amazonaws.com/mantispy/oasis_pilot/oasis_pilot.h5ad + sha256: dd77fc202ea43a3223fdcb572bebdbae209946b62d30f34d44de90c9088b8363 + - name: oasis_pilot_agg.h5ad + url: https://scverse-exampledata.s3.eu-west-1.amazonaws.com/mantispy/oasis_pilot/oasis_pilot_agg.h5ad + sha256: 228f89aa4098b071a40b7c14c2d90f99c7b930f98a2bd6b3f469b50dce318e58 pki: type: mantispy @@ -2152,6 +2178,20 @@ datasets: - name: 2021_04_07_Batch1__BR00122978__BR00122978_augmented.csv.gz s3_key: cpg0008-pki/broad/workspace/profiles/2021_04_07_Batch1/BR00122978/BR00122978_augmented.csv.gz sha256: dc59da85d6b83c7388e633c3daf2e3ff407ceccea691b0143b986a3574fd0b79 + # Staged variants rehosted on scverse-exampledata (explicit url overrides base_url, so the raw plate tables + # above stay on cellpainting-gallery). Regenerate with scripts/build_staged_datasets.py. + - name: pki.h5ad + url: https://scverse-exampledata.s3.eu-west-1.amazonaws.com/mantispy/pki/pki.h5ad + sha256: db6da1569ad7eab9f868d4d48d18f79e0c1ac75bcd84c40a05a09ae5f8fd6973 + - name: pki_selected.h5ad + url: https://scverse-exampledata.s3.eu-west-1.amazonaws.com/mantispy/pki/pki_selected.h5ad + sha256: e7d3fcc90fc10d29293f98eaf772c3cbb353202f0a09889643fb969e17518db6 + - name: pki_agg.h5ad + url: https://scverse-exampledata.s3.eu-west-1.amazonaws.com/mantispy/pki/pki_agg.h5ad + sha256: adce802c942ea46457733a0f96ac2807d13619ba7005fdb98b24991a44f1e2cb + - name: pki_agg_selected.h5ad + url: https://scverse-exampledata.s3.eu-west-1.amazonaws.com/mantispy/pki/pki_agg_selected.h5ad + sha256: acee0765ffc1f9487d57d30c1aec0fb9dcba6b45217b589fe9380223daf5a1eb jump_crispr: type: mantispy @@ -2164,6 +2204,14 @@ datasets: - name: profiles_wellpos_cc_var_mad_outlier_featselect.parquet s3_key: cpg0016-jump-assembled/source_all/workspace/profiles_assembled/CRISPR/v1.0a/profiles_wellpos_cc_var_mad_outlier_featselect.parquet sha256: 2f52c600db35d9cc23645d6d94685ceb0df986c23d48c81499bf5f76df7b43b2 + # Staged variants rehosted on scverse-exampledata (explicit url overrides base_url, so the raw profile + # parquet above stays on cellpainting-gallery). Regenerate with scripts/build_staged_datasets.py. + - name: jump_crispr.h5ad + url: https://scverse-exampledata.s3.eu-west-1.amazonaws.com/mantispy/jump_crispr/jump_crispr.h5ad + sha256: e5b3101740d9bf6f5ed29e52c5689e4fa6352edacc6df76c7d099c72ef8a169b + - name: jump_crispr_agg.h5ad + url: https://scverse-exampledata.s3.eu-west-1.amazonaws.com/mantispy/jump_crispr/jump_crispr_agg.h5ad + sha256: 5b2ec01a0151c256f7b05298a91cd2fc538685e299d374a94c8a15f076eea52e jump_lite: type: mantispy @@ -2198,6 +2246,27 @@ datasets: - name: refchem_annotations.parquet s3_key: cpg0016-jump/source_all/workspace/publication_data/2026_jump_lite/metadata/v1.0/jump_lite_refchem_annotations.parquet sha256: 5a766a222907e1b3805f84cc2eccad0e5cd2f3eaac1add4983689853601db960 + # Staged per-model objects (the annotated wells of each feature set) rehosted on scverse-exampledata + # (explicit url overrides base_url, so the raw feature parquets above stay on cellpainting-gallery). The + # annotate=False path still reads those raw parquets. Regenerate with scripts/build_staged_datasets.py. + - name: jump_lite_openphenom.h5ad + url: https://scverse-exampledata.s3.eu-west-1.amazonaws.com/mantispy/jump_lite/jump_lite_openphenom.h5ad + sha256: 1652bcba78f74ef309f4359cd1b48492d5a63414a10f7891eb27e2d8a614bcfb + - name: jump_lite_dinov2.h5ad + url: https://scverse-exampledata.s3.eu-west-1.amazonaws.com/mantispy/jump_lite/jump_lite_dinov2.h5ad + sha256: 9aaff8af2b69e4ad517c346255a460a358ef5c04a4bab5aa50220cdb9b8da508 + - name: jump_lite_dinov2_random.h5ad + url: https://scverse-exampledata.s3.eu-west-1.amazonaws.com/mantispy/jump_lite/jump_lite_dinov2_random.h5ad + sha256: 47bb156833c165fd8413dde1bf6bd3f6af04d3003d7b3c1c468e0c23b2425d91 + - name: jump_lite_subcell.h5ad + url: https://scverse-exampledata.s3.eu-west-1.amazonaws.com/mantispy/jump_lite/jump_lite_subcell.h5ad + sha256: 6cc6485a5033824c96b87bd72b308941e1cf897b4106821aa5ce7fa0ec6bba9c + - name: jump_lite_morphem.h5ad + url: https://scverse-exampledata.s3.eu-west-1.amazonaws.com/mantispy/jump_lite/jump_lite_morphem.h5ad + sha256: fccc073e32936ecd9620785a3fdedb194be1f96115481048ed7883e974989de1 + - name: jump_lite_cp_measure.h5ad + url: https://scverse-exampledata.s3.eu-west-1.amazonaws.com/mantispy/jump_lite/jump_lite_cp_measure.h5ad + sha256: 63380e08eed3820850d9c6c423095eecfe2869f5c6e0a2961bcd499412eef843 jump_target2: type: mantispy @@ -3051,6 +3120,12 @@ datasets: - name: 20221120_Run6__CP-CC9-R7-29__CP-CC9-R7-29.csv s3_key: cpg0016-jump/source_13/workspace/backend/20221120_Run6/CP-CC9-R7-29/CP-CC9-R7-29.csv sha256: 89ff2d98a188afa49bfbc681975e38330e7f3e1f95e5e5b97e1f923374497f85 + # Staged default object (the eleven-plate default selection, annotated) rehosted on scverse-exampledata + # (explicit url overrides base_url, so the raw per-plate tables above stay on cellpainting-gallery). A + # non-default plate selection still reads those raw tables. Regenerate with scripts/build_staged_datasets.py. + - name: jump_target2.h5ad + url: https://scverse-exampledata.s3.eu-west-1.amazonaws.com/mantispy/jump_target2/jump_target2.h5ad + sha256: 3d9ddd26541f50685b2c5b515f63efbc479f4e4081e8f984998334605f035d8c jump_cells: type: mantispy @@ -4502,6 +4577,17 @@ datasets: - name: cpg0016-jump/source_4/workspace/analysis/2021_04_26_Batch1/BR00121438/analysis/BR00121438-O11-4/Nuclei.csv s3_key: cpg0016-jump/source_4/workspace/analysis/2021_04_26_Batch1/BR00121438/analysis/BR00121438-O11-4/Nuclei.csv sha256: 6d1bc0acb3b4d21cde0649e18eaf4eaf9357bd9e98de230b51de4e0335b6e22f + # Staged variants rehosted on scverse-exampledata (explicit url overrides base_url, so the raw CellProfiler + # tables above stay on cellpainting-gallery). Regenerate with scripts/build_staged_datasets.py. + - name: jump_cells.h5ad + url: https://scverse-exampledata.s3.eu-west-1.amazonaws.com/mantispy/jump_cells/jump_cells.h5ad + sha256: c0d97cac8be979c0752e2ec8399ad30522e6d91fc2b0857276736c9d5e186520 + - name: jump_cells_selected.h5ad + url: https://scverse-exampledata.s3.eu-west-1.amazonaws.com/mantispy/jump_cells/jump_cells_selected.h5ad + sha256: 6603051f11da05d0517a620f76b2e9ac99dd777da5e1d44082e65c187ef2e9b5 + - name: jump_cells_agg.h5ad + url: https://scverse-exampledata.s3.eu-west-1.amazonaws.com/mantispy/jump_cells/jump_cells_agg.h5ad + sha256: 8f5143e1c72b753d836c46d6beaaed9bc9f749150886d9740fdac0990f64b836 jump_plate: type: mantispy @@ -4625,6 +4711,13 @@ datasets: - name: dataframe_fig3_clean.pq url: https://github.com/Genentech/scallops-manuscript/raw/main/Figure3/Data/dataframe_fig3_clean.pq sha256: e3988e2d8e1be1899b6e7136896e61d8a20ffedf2f2283d7fc15355c4a7068e7 + # Staged variants rehosted on scverse-exampledata. Regenerate with scripts/build_staged_datasets.py. + - name: scallops_arv471.h5ad + url: https://scverse-exampledata.s3.eu-west-1.amazonaws.com/mantispy/scallops_arv471/scallops_arv471.h5ad + sha256: d1f3e50e97b29787b31b3a18e2071e39acec12935968cd39c843a97c5e52965c + - name: scallops_arv471_agg.h5ad + url: https://scverse-exampledata.s3.eu-west-1.amazonaws.com/mantispy/scallops_arv471/scallops_arv471_agg.h5ad + sha256: c99d2cd3a37921ca472bb329da63d95540d620fb90b434b4f04d51159d2205c3 cp_posh: type: mantispy @@ -4635,3 +4728,16 @@ datasets: - name: cp_posh_124gene_poc_well_normalized.pq url: https://insitro-research-2023-cellpaint-posh.s3.amazonaws.com/Supp%20Data%202%20-%20124-Gene%20PoC%20Dataset%20(well-normalized).pq sha256: 4730d28ac5fc5d4245855523a644c79c7299e30df046c0e342197391a036e760 + # Staged variants rehosted on scverse-exampledata. Regenerate with scripts/build_staged_datasets.py. + - name: cp_posh.h5ad + url: https://scverse-exampledata.s3.eu-west-1.amazonaws.com/mantispy/cp_posh/cp_posh.h5ad + sha256: 7b9c41861ef80c97476cf9c099ec899bf8d5a71e85f8b2bdecb88ed180513712 + - name: cp_posh_selected.h5ad + url: https://scverse-exampledata.s3.eu-west-1.amazonaws.com/mantispy/cp_posh/cp_posh_selected.h5ad + sha256: e74a4fbc9945f353d331462452f2e4f1e33fde9a5f92c9856b6fd24fd072a76d + - name: cp_posh_agg.h5ad + url: https://scverse-exampledata.s3.eu-west-1.amazonaws.com/mantispy/cp_posh/cp_posh_agg.h5ad + sha256: 49ad57eb11a1a7f3d204dcfceae5c3f6bf8f2e99d837894e8512d4e42e31cc17 + - name: cp_posh_agg_selected.h5ad + url: https://scverse-exampledata.s3.eu-west-1.amazonaws.com/mantispy/cp_posh/cp_posh_agg_selected.h5ad + sha256: 5913832c31b19d934aebf3e21744f09ce04a05b0ccec7273e6e675c3ac470b3a diff --git a/tasks/lessons.md b/tasks/lessons.md new file mode 100644 index 0000000..2d23209 --- /dev/null +++ b/tasks/lessons.md @@ -0,0 +1,12 @@ +# Lessons + +- When staging a dataset's base (host + fetch), the base must equal the *live* loader output modulo the + stable obs/var sort. Verify the shape against the actual assembly (`_augmented(...).n_vars`), not a + docstring, audit, or an existing `@pytest.mark.network` shape assertion: pki's gallery test pinned + (3072, 5857) but the pinned data yields 5839, so the test was stale. Trust the running code. +- Loaders that assemble via `_files`/`_profiles`/`_plate_files` with no `select` will pick up the new + rehosted `.h5ad` rows once they are added to the registry entry. Every `_assemble_` must filter them + out (`select=lambda name: not name.endswith(".h5ad")`), and the fetch path must select the exact h5ad name. +- For a loader with an `annotate=`/`plates=`/`model=` parameter, host the default object and keep the + non-default path reading the raw inputs unchanged; that keeps the public API stable without hosting every + parameterization. Guard illegal combinations (e.g. `aggregated` needs `annotate=True`) before any fetch. diff --git a/tasks/todo.md b/tasks/todo.md new file mode 100644 index 0000000..c740bcc --- /dev/null +++ b/tasks/todo.md @@ -0,0 +1,48 @@ +# Stage all remaining datasets on S3 (base + variants), flip loaders to fetch + +Branch: feat/stage-all-datasets. Templates: #167 (staging machinery) + #168 (bbbc021/rohban). +I build finals + write code; coordinator uploads + pushes + opens PR. + +## Pattern per dataset (mirror bbbc021/rohban) +- `_assemble_` in `_build.py` = the raw pipeline (skip `.h5ad` inputs via select). +- `build__variants` returns `{filename: AnnData}`, each through `_stable` (+ `_zero_nonfinite` on agg/selected). +- `_shipped_` fetches the same filenames through the public API. +- STAGED entry `(builder, shipped_loader, heavy)`; heavy=True if raw inputs > ~50 MB. +- `__VARIANTS` map in `_datasets.py`; loader validates bool flags, fetches the h5ad, `read()`. +- registry: append the h5ad rows (url + sha256) after the raw rows; raw rows stay for drift rebuild. +- build: `python scripts/build_staged_datasets.py --only --print-sha256 --out $HOME/build_all/`. + +## Order (smallest raw first, commit in groups) +- [x] neuropainting (base only) -- validated the whole pipeline +- [x] chroma (base only) +- [x] pooled_rare (base, selected) +- [x] oasis_pilot (base, agg; annotate=False keeps raw path) +- [x] pki (base, selected, agg, agg_selected); fixed stale gallery test 5857->5839 +- [x] scallops_arv471(base, agg; cells->guide via tl.aggregate) +- [x] jump_crispr (base, agg; guides->gene consensus; annotate=False keeps raw path) +- [x] jump_cells (base, selected, agg; agg is cells->WELL per the audit; keep annotate=/selected=) +- [x] cp_posh (base, selected, agg, agg_selected) +- [x] jump_target2 (default 11-plate base hosted; non-default plates= keeps the raw path) +- [x] jump_lite (6 per-model keyed finals; annotate=False keeps raw path) + +## Review +- All 11 datasets migrated, none deferred. 28 finals in $HOME/build_all/ (all sha256 in registry). +- jump_cells `aggregated`: the audit says aggregate by (Metadata_Plate, Metadata_Well) i.e. cells->well, + which is also `tl.aggregate`'s default and lines up with the well-level jump_target2; the task text said + "Metadata_Perturbation unit", the audit wins (noted per the follow-the-audit rule). +- pki base is (3072, 5839) from the live `_augmented`; the gallery test asserted (3072, 5857), which was + stale against the pinned data (verified `_augmented("pki", None).n_vars == 5839`). Updated the assertion. +- Deterministic spot-checks (build twice, identical sha256): neuropainting, pki, scallops_arv471, jump_cells. +- api-guards + offline gallery tests pass; ruff clean; every built h5ad round-trips and `io.validate` is ok. + +## Verify (mine) +- [ ] `import mantispy; import mantispy.ds._build` (no cycle) +- [ ] `pytest tests/test_api_guards.py -q` passes +- [ ] determinism spot-check (build twice, same sha256) on a few +- [ ] ruff check + ruff format --check on changed files +- Do NOT run check_staged_drift or network/gallery tests (404 until upload). + +## Notes +- base must equal today's loader output modulo stable sort. Gallery test pins pki (3072,5857). +- Disk: home/cache mount ~37G free, 5.1G used. Clean cache between heavy datasets. +- No push, no PR, no aws. Report ends with `claude done`. diff --git a/tests/test_datasets_gallery.py b/tests/test_datasets_gallery.py index 5582f80..b31fbae 100644 --- a/tests/test_datasets_gallery.py +++ b/tests/test_datasets_gallery.py @@ -102,7 +102,7 @@ def test_bbbc021_flags_must_be_bool(): @pytest.mark.network def test_pki_loads_with_a_dose_series(pki): - assert pki.shape == (3072, 5857) + assert pki.shape == (3072, 5839) assert mt.io.validate(pki).ok, mt.io.validate(pki).errors treated = pki.obs[~pki.obs["Metadata_Control"].to_numpy()] assert treated["Metadata_Compound"].nunique() == 15 diff --git a/tests/test_ds.py b/tests/test_ds.py index 09142fd..7f1c4a6 100644 --- a/tests/test_ds.py +++ b/tests/test_ds.py @@ -7,7 +7,13 @@ import mantispy as mt from mantispy._core.frames import as_frame -from mantispy.ds._datasets import _DATASETS, TARGET2_DEFAULT, _plate +from mantispy.ds._datasets import ( + _DATASETS, + TARGET2_DEFAULT, + _assemble_cp_posh, + _assemble_scallops_arv471, + _plate, +) @pytest.mark.parametrize(("n_wells", "n_sites"), [(1, 1), (4, 2)]) @@ -339,7 +345,7 @@ def test_scallops_arv471_loads_clean_cell_resolution(tmp_path, monkeypatch): clean = _write_scallops_fixture(path) _patch_files(monkeypatch, path) - adata = mt.ds.scallops_arv471() + adata = _assemble_scallops_arv471() assert adata.shape == (clean, 9) assert adata.uns["mantispy"]["resolution"] == "cell" @@ -356,7 +362,7 @@ def test_scallops_arv471_maps_controls_and_guides(tmp_path, monkeypatch): _write_scallops_fixture(path) _patch_files(monkeypatch, path) - adata = mt.ds.scallops_arv471() + adata = _assemble_scallops_arv471() obs = as_frame(adata.obs) assert "NTC" not in set(obs["Metadata_Gene"].astype(str)) @@ -381,7 +387,7 @@ def test_scallops_arv471_runs_hit_calling(tmp_path, monkeypatch): _write_scallops_fixture(path) _patch_files(monkeypatch, path) - adata = mt.ds.scallops_arv471() + adata = _assemble_scallops_arv471() kwargs = {"groupby": "Metadata_Gene", "reference": "negcon", "n_permutations": 50, "seed": 0, "copy": True} # block= is the well-block permutation null (#68); pass it once it reaches this build's signature. @@ -445,7 +451,7 @@ def test_cp_posh_loads_clean_cell_resolution(tmp_path, monkeypatch): cells, _ = _write_cp_posh_fixture(path) _patch_files(monkeypatch, path) - adata = mt.ds.cp_posh() + adata = _assemble_cp_posh() assert adata.shape == (cells, len(_CP_POSH_FIXTURE_FEATURES)) assert adata.uns["mantispy"]["resolution"] == "cell" @@ -464,7 +470,7 @@ def test_cp_posh_maps_controls_and_guides(tmp_path, monkeypatch): _, controls = _write_cp_posh_fixture(path) _patch_files(monkeypatch, path) - adata = mt.ds.cp_posh() + adata = _assemble_cp_posh() obs = as_frame(adata.obs) assert obs["Metadata_Gene"].nunique() == 5 # nontargeting, intergenic, KIF18A, PSMB1, ARPC4 @@ -488,7 +494,7 @@ def test_cp_posh_runs_hit_calling(tmp_path, monkeypatch): _write_cp_posh_fixture(path) _patch_files(monkeypatch, path) - adata = mt.ds.cp_posh() + adata = _assemble_cp_posh() result = mt.tl.hit_calling(adata, groupby="Metadata_Gene", reference="negcon", n_permutations=50, seed=0, copy=True)