diff --git a/.github/workflows/crisprworks-fit.yml b/.github/workflows/crisprworks-fit.yml index 4a6b511a..84782440 100644 --- a/.github/workflows/crisprworks-fit.yml +++ b/.github/workflows/crisprworks-fit.yml @@ -45,6 +45,8 @@ jobs: python-version: ${{ matrix.python }} - name: Install experimental package run: python -m pip install ./packages/crisprworks-fit + - name: Require compiled engine + run: python -c 'from crisprworks_fit.kernels import resolved_engine; assert resolved_engine("native") == "native"' - name: Test scientific and workflow parity run: python -m unittest discover -s packages/crisprworks-fit/tests -v - name: Check installed command @@ -90,7 +92,7 @@ jobs: echo 'run=true' >> "$GITHUB_OUTPUT" else git diff --name-only "$BEFORE" HEAD > /tmp/fit-changed-paths - if grep -Eq '^packages/crisprworks-fit/(src/|benchmarks/.*\.py|pyproject\.toml|container-constraints\.txt)' /tmp/fit-changed-paths; then + if grep -Eq '^packages/crisprworks-fit/(src/|benchmarks/.*\.py|pyproject\.toml|setup\.py|container-constraints\.txt)' /tmp/fit-changed-paths; then echo 'run=true' >> "$GITHUB_OUTPUT" else echo 'run=false' >> "$GITHUB_OUTPUT" @@ -104,7 +106,7 @@ jobs: if: steps.changed.outputs.run == 'true' - name: Paired public full-library benchmark with numerical parity if: steps.changed.outputs.run == 'true' - run: python packages/crisprworks-fit/benchmarks/public_hap1.py --out-dir fit-hap1 --repeats 3 --cohort "${{ inputs.cohort || 'four-guide' }}" + run: python packages/crisprworks-fit/benchmarks/public_hap1.py --out-dir fit-hap1 --repeats 3 --kernel native --compare-numpy --cohort "${{ inputs.cohort || 'four-guide' }}" - uses: actions/upload-artifact@v7 if: always() with: diff --git a/packages/crisprworks-fit/.gitignore b/packages/crisprworks-fit/.gitignore index 8e75778f..6918fb99 100644 --- a/packages/crisprworks-fit/.gitignore +++ b/packages/crisprworks-fit/.gitignore @@ -4,3 +4,7 @@ __pycache__/ build/ dist/ .venv/ + +*.so +*.pyd +*.o diff --git a/packages/crisprworks-fit/README.md b/packages/crisprworks-fit/README.md index 46fc5b41..bed261b7 100644 --- a/packages/crisprworks-fit/README.md +++ b/packages/crisprworks-fit/README.md @@ -3,7 +3,8 @@ **Experimental acceleration of MAGeCK2 MLE for pooled CRISPR screens.** Fit runs MAGeCK2's count-table-to-gene-results workflow with faster numerical -kernels and worker reuse. It belongs to the CRISPRWorks family alongside +kernels and worker reuse. Alpha 2 adds a compiled C EM loop; the earlier +NumPy engine remains selectable for reproducible comparisons. It belongs to the CRISPRWorks family alongside CRISPRWorks Count, powered by DotMatch. This is a separate installable package; the existing DotMatch package remains independent of its dependencies. @@ -23,7 +24,11 @@ crisprworks-fit demo --out-dir fit-demo/ MAGeCK2 is an installation dependency and currently builds its bundled C++ helpers, so its installation requires a C++ compiler and `make`. Python 3.10+ -is required by Fit. The package has not been published to PyPI. +is required by Fit. Building the optional native loop requires a C compiler. +`--kernel auto` uses the native loop when available and otherwise NumPy; +`--kernel native` fails explicitly if the compiled engine is unavailable; +`--kernel numpy` selects the previous engine. The provenance manifest records +the engine actually used. The package has not been published to PyPI. The demo creates a small synthetic count table and design, then writes gene and guide summaries. It demonstrates software behavior, not biological accuracy. @@ -80,6 +85,9 @@ processes; `--blas-threads` controls BLAS threads within each worker. ## What is faster +- Execute the EM loop in compiled C, with working storage reused across + iterations. For finite inputs, process exact nonzero design entries while + preserving row summation order; nonfinite inputs use the dense calculation. - Apply IRLS weights by vector multiplication and solve the smaller ridge system directly. No dense diagonal weight matrix is constructed. - Calculate the sandwich covariance after the last iteration, and compute the @@ -108,18 +116,25 @@ checks dense-versus-compact covariance algebra, restarts, strict permutation tails including ties and NaNs, FDR, source compatibility, failure restoration, invalid designs, the public upstream count-table fixture and spawned workers. -The [genome-wide HAP1 cohort measurement](benchmarks/HAP1.md) covered 17,445 -complete four-guide gene labels: median wall time was **199.74 s reference vs -75.63 s accelerated (2.64×)** across three runs per backend. Printed gene -summaries were byte-identical, permutation p-values/FDR matched exactly, and -full-precision beta differences were below `1.25e-14`. This selects complete -four-guide labels before normalization; it is not a timing for the unfiltered -table or evidence of biological hit accuracy. - -The unfiltered 71,090-guide table also passed a separate cross-environment -full-precision comparison with identical printed gene results; see the HAP1 -record for how the CI reference and local accelerated outputs were compared. -No unfiltered-table timing claim is made. +The [native-engine HAP1 measurement](benchmarks/NATIVE.md) covered 17,445 +complete four-guide gene labels and 69,780 guides. Three paired runs per engine +measured **332.45 s MAGeCK2 reference, 131.06 s NumPy accelerator, and 38.68 s +native accelerator**: **8.59× versus reference and 3.39× versus NumPy**. +All nine printed gene summaries were byte-identical; permutation p-values and +FDR matched exactly. Maximum full-precision beta differences were `2.49e-14`. +These timings include startup and diagnostic output on one shared CI runner. + +This cohort selects complete four-guide labels before normalization and +excludes larger control bins. The unfiltered 71,090-guide, 18,056-label table +also matched the reference in a separate cross-environment full-precision +comparison. That comparison establishes numerical agreement, with no +unfiltered-table timing ratio claimed. These checks preserve the reference +results; they do not establish superior biological hit accuracy or a universal +speedup. Chronos and JACKS have not been benchmarked in this comparison. + +The [earlier NumPy measurement](benchmarks/HAP1.md) remains archived with its +original implementation commit and runner timings. Compare engines within each +paired experiment rather than combining times from different environments. Initial measurements are documented in [benchmarks/RESULTS.md](benchmarks/RESULTS.md). The initial records cover two bounded workloads at commit `e900ecdb`; they do @@ -162,8 +177,9 @@ commands and input hashes. The underlying covariance algebra is tested at CNV correction and `--debug-gene` are rejected until separately evaluated. Experimental Bayes options retain upstream's explicit rejection. The upstream active fitting path does not apply `--remove-outliers`; Fit preserves that -behavior. Fit is currently an optional Python/NumPy accelerator, with native -linear algebra provided by BLAS. It does not implement counting or change +behavior. Fit offers a compiled C EM engine and a NumPy fallback, with BLAS +used for covariance calculations. The native loop omits exact zero design +entries for finite inputs and retains a dense path for nonfinite inputs. It does not implement counting or change DotMatch's read-assignment rules. Keep counting comparisons separate from inference comparisons. A faster fit diff --git a/packages/crisprworks-fit/benchmarks/NATIVE.md b/packages/crisprworks-fit/benchmarks/NATIVE.md new file mode 100644 index 00000000..ac831005 --- /dev/null +++ b/packages/crisprworks-fit/benchmarks/NATIVE.md @@ -0,0 +1,64 @@ +# Native-engine HAP1 benchmark, 2026-09-30 + +CRISPRWorks Fit alpha 2 at implementation commit +`2938b8e6e2219e80404280facf14f323e55ee0c3` analyzed **17,445 gene labels and +69,780 guides** from the pinned public Hart Lab BAGEL HAP1 table. All three +engines used identical counts, design, options and random seed. Execution order +rotated across three repetitions per engine on the same shared CI runner. + +| Engine | Median wall time | Individual wall times | +| --- | ---: | --- | +| Unmodified MAGeCK2 0.3.0 fitting functions | 332.452 s | 336.551, 330.837, 332.452 s | +| Previous Fit NumPy engine | 131.060 s | 131.060, 129.386, 132.549 s | +| Fit native engine | 38.681 s | 38.838, 38.590, 38.681 s | + +**8.59× faster than reference; 3.39× faster than the NumPy accelerator.** +Wall time includes interpreter startup, the count-table workflow and +full-precision diagnostic output. All nine printed gene summaries were +byte-identical. All permutation p-values and FDR values, including both tails, +matched exactly. Maximum absolute differences were `2.49e-14` for beta, +`3.27e-14` for Wald z-scores and `2.01e-14` for Wald FDR. + +The design has one T0 baseline and three T18 samples with one treatment effect. +Settings: updated guide efficiencies, mean-variance modeling using 1,000 genes, +two permutation rounds, seed 42, one worker and one BLAS thread. This cohort +selects labels with exactly four guides before normalization, excluding +incomplete labels and larger control bins. It is not the unfiltered table. + +The [machine-readable record](hap1-native-four-guide.json) includes dataset +provenance, hashes, all commands and run manifests, versions, BLAS configuration +and numerical differences. The [successful CI run](https://github.com/dnncha/dotmatch/actions/runs/36724219194) +retains all nine complete result sets. Artifact ID: `11102737825`; archive +SHA-256: `cf30d6e98a92efb8e2252e66592e9c59fd6afcc3bf8e4a2c8645e9c954b097ae`. + +To reproduce with a compiled install: + +```bash +python packages/crisprworks-fit/benchmarks/public_hap1.py \ + --out-dir hap1-native --cohort four-guide --repeats 3 \ + --kernel native --compare-numpy +``` + +The native implementation uses a checked SciPy special-function C API, +partial-pivot LU without extra regularization, and compiler settings that +exclude fast-math and floating-point contraction. It omits exact zero design +entries for finite inputs while retaining a dense path for nonfinite inputs. +Tests cover multiple designs, efficiency modes, low counts, matrix adapters, +nonfinite initialization, singular solves, invalid buffers, engine fallback and +state restoration. Linux, macOS, installed wheels and the container passed CI. +Local fitting tests also passed AddressSanitizer and undefined-behavior checks. + +The [unfiltered-table parity record](hap1-native-full-parity.json) separately +covers all 71,090 guides and 18,056 labels. It compares a completed CI reference +with local native outputs using identical input hashes, settings and seed. +Printed gene summaries matched byte for byte, all permutation statistics +matched exactly, and maximum absolute differences were below `5.53e-14`. +It is a cross-environment numerical comparison; no timing ratio is claimed. +Upstream's default skip and permutation fallback for the 96-guide label remain. + +These results establish a measured performance improvement with numerical +compatibility on this dataset. They do not establish universal performance, +biological superiority or overall state of the art. Chronos, JACKS, legacy +MAGeCK releases and independent multi-screen hit-quality benchmarks remain +outside this comparison. Earlier NumPy timing records use different runners +and must not be combined with these times to calculate a speedup. diff --git a/packages/crisprworks-fit/benchmarks/hap1-native-four-guide.json b/packages/crisprworks-fit/benchmarks/hap1-native-four-guide.json new file mode 100644 index 00000000..331e73aa --- /dev/null +++ b/packages/crisprworks-fit/benchmarks/hap1-native-four-guide.json @@ -0,0 +1,1278 @@ +{ + "implementation_commit": "2938b8e6e2219e80404280facf14f323e55ee0c3", + "ci_run": 36724219194, + "artifact_id": 11102737825, + "artifact_digest": "sha256:cf30d6e98a92efb8e2252e66592e9c59fd6afcc3bf8e4a2c8645e9c954b097ae", + "provenance": { + "source_url": "https://raw.githubusercontent.com/hart-lab/bagel/53388adbb4fb0931e5c9dda135502be19e4555f0/reads_hap1.txt", + "source_commit": "53388adbb4fb0931e5c9dda135502be19e4555f0", + "source_sha256": "7638bc6237cf8a3e2302fcd969d014b041819669b37f2f87bc1f7cfbc45ca8a7", + "counts_sha256": "bf24bf255b75cde8cb01279290bdbbbbbba5773596245b8c0300b07b879c36bf", + "cohort": "four-guide", + "guides": 69780, + "gene_labels_including_controls": 17445, + "selection": "labels having exactly four guides; incomplete labels and larger control bins excluded before normalization", + "design": "one T0 baseline; three T18 replicates; single treatment effect", + "scope": "numerical parity against MAGeCK2, not validation of biological hits", + "control_caveat": "Upstream default skips fitting labels with >=40 guides; their permutation tails use the preceding guide-count group", + "source_license": "MIT; dataset retrieved separately, not redistributed in this package" + }, + "benchmark": { + "dataset": "supplied count table; provenance and validation must be recorded separately", + "genes": 17445, + "guides_per_gene": null, + "samples": 4, + "total_guides": 69780, + "permutation_rounds": 2, + "seed": 42, + "worker_processes": 1, + "blas_threads_per_process": 1, + "reference_commit": "630aea0b6fc152a81006435f21d629297275c911", + "python": "3.12.14", + "platform": "Linux-6.17.0-1022-azure-x86_64-with-glibc2.39", + "counts_sha256": "bf24bf255b75cde8cb01279290bdbbbbbba5773596245b8c0300b07b879c36bf", + "design_sha256": "bc4f06905509e0180146e20ae3d3228a59406b2422ed193e7f0e0e7cc9fd5d2e", + "runs": [ + { + "backend": "reference", + "repeat": 0, + "wall_seconds_including_startup": 336.550612808, + "workflow_seconds": 335.542093153, + "command": [ + "/opt/hostedtoolcache/Python/3.12.14/x64/bin/python", + "-m", + "crisprworks_fit", + "mle", + "-k", + "fit-hap1/counts.tsv", + "-d", + "fit-hap1/design.tsv", + "-n", + "fit-hap1/paired/reference-0", + "--backend", + "reference", + "--threads", + "1", + "--blas-threads", + "1", + "--seed", + "42", + "--permutation-round", + "2", + "--kernel", + "native", + "--write-fit-details", + "--genes-varmodeling", + "1000", + "--update-efficiency" + ], + "manifest": { + "schema_version": 1, + "status": "complete", + "tool": "CRISPRWorks Fit", + "version": "0.1.0a2", + "backend": "reference", + "kernel": "reference", + "mageck2_version": "0.3.0", + "reference_commit": "630aea0b6fc152a81006435f21d629297275c911", + "python": "3.12.14", + "numpy": "2.3.5", + "scipy": "1.17.0", + "platform": "Linux-6.17.0-1022-azure-x86_64-with-glibc2.39", + "options": { + "subcmd": "mle", + "count_table": "fit-hap1/counts.tsv", + "design_matrix": "fit-hap1/design.tsv", + "day0_label": null, + "output_prefix": "fit-hap1/paired/reference-0", + "include_samples": null, + "beta_labels": null, + "control_sgrna": null, + "control_gene": null, + "cnv_norm": null, + "cell_line": null, + "cnv_est": null, + "debug": false, + "debug_gene": null, + "norm_method": "median", + "genes_varmodeling": 1000, + "permutation_round": 2, + "no_permutation_by_group": false, + "max_sgrnapergene_permutation": 40, + "remove_outliers": false, + "threads": 1, + "adjust_method": "fdr", + "sgrna_efficiency": null, + "sgrna_eff_name_column": 0, + "sgrna_eff_score_column": 1, + "update_efficiency": true, + "bayes": false, + "PPI_prior": false, + "PPI_weighting": null, + "negative_control": null, + "backend": "reference", + "kernel": "native", + "seed": 42, + "blas_threads": 1, + "write_fit_details": true + }, + "inputs": { + "count_table": { + "path": "/home/runner/work/dotmatch/dotmatch/fit-hap1/counts.tsv", + "sha256": "bf24bf255b75cde8cb01279290bdbbbbbba5773596245b8c0300b07b879c36bf" + }, + "design_matrix": { + "path": "/home/runner/work/dotmatch/dotmatch/fit-hap1/design.tsv", + "sha256": "bc4f06905509e0180146e20ae3d3228a59406b2422ed193e7f0e0e7cc9fd5d2e" + }, + "sgrna_efficiency": null, + "control_sgrna": null, + "control_gene": null + }, + "blas": [ + { + "user_api": "blas", + "internal_api": "openblas", + "num_threads": 1, + "prefix": "libscipy_openblas", + "filepath": "/opt/hostedtoolcache/Python/3.12.14/x64/lib/python3.12/site-packages/numpy.libs/libscipy_openblas64_-fdde5778.so", + "version": "0.3.30", + "threading_layer": "pthreads", + "architecture": "Haswell" + }, + { + "user_api": "blas", + "internal_api": "openblas", + "num_threads": 1, + "prefix": "libscipy_openblas", + "filepath": "/opt/hostedtoolcache/Python/3.12.14/x64/lib/python3.12/site-packages/scipy.libs/libscipy_openblas-6cdc3b4a.so", + "version": "0.3.30", + "threading_layer": "pthreads", + "architecture": "Haswell" + } + ], + "outputs": { + "gene_summary": { + "path": "/home/runner/work/dotmatch/dotmatch/fit-hap1/paired/reference-0.gene_summary.txt", + "sha256": "dfeb15a7e1f113891a56b8cd23d575ffb124289d6c948367d1027c20ce2256ee" + }, + "sgrna_summary": { + "path": "/home/runner/work/dotmatch/dotmatch/fit-hap1/paired/reference-0.sgrna_summary.txt", + "sha256": "16f5ded25d4b6321808ea405132cbf39c6fe1185aec1bf6dced07ca9e9a2bd29" + }, + "fit_details": { + "path": "/home/runner/work/dotmatch/dotmatch/fit-hap1/paired/reference-0.fit-details.json", + "sha256": "4317c2db4819af2c6ad2fe8659af4ce2bd23cba69ad843bc8a4a091aaae7c082" + } + }, + "elapsed_seconds": 335.542093153 + } + }, + { + "backend": "accelerated", + "repeat": 0, + "wall_seconds_including_startup": 38.837820647, + "workflow_seconds": 37.80452142799999, + "command": [ + "/opt/hostedtoolcache/Python/3.12.14/x64/bin/python", + "-m", + "crisprworks_fit", + "mle", + "-k", + "fit-hap1/counts.tsv", + "-d", + "fit-hap1/design.tsv", + "-n", + "fit-hap1/paired/accelerated-0", + "--backend", + "accelerated", + "--threads", + "1", + "--blas-threads", + "1", + "--seed", + "42", + "--permutation-round", + "2", + "--kernel", + "native", + "--write-fit-details", + "--genes-varmodeling", + "1000", + "--update-efficiency" + ], + "manifest": { + "schema_version": 1, + "status": "complete", + "tool": "CRISPRWorks Fit", + "version": "0.1.0a2", + "backend": "accelerated", + "kernel": "native", + "mageck2_version": "0.3.0", + "reference_commit": "630aea0b6fc152a81006435f21d629297275c911", + "python": "3.12.14", + "numpy": "2.3.5", + "scipy": "1.17.0", + "platform": "Linux-6.17.0-1022-azure-x86_64-with-glibc2.39", + "options": { + "subcmd": "mle", + "count_table": "fit-hap1/counts.tsv", + "design_matrix": "fit-hap1/design.tsv", + "day0_label": null, + "output_prefix": "fit-hap1/paired/accelerated-0", + "include_samples": null, + "beta_labels": null, + "control_sgrna": null, + "control_gene": null, + "cnv_norm": null, + "cell_line": null, + "cnv_est": null, + "debug": false, + "debug_gene": null, + "norm_method": "median", + "genes_varmodeling": 1000, + "permutation_round": 2, + "no_permutation_by_group": false, + "max_sgrnapergene_permutation": 40, + "remove_outliers": false, + "threads": 1, + "adjust_method": "fdr", + "sgrna_efficiency": null, + "sgrna_eff_name_column": 0, + "sgrna_eff_score_column": 1, + "update_efficiency": true, + "bayes": false, + "PPI_prior": false, + "PPI_weighting": null, + "negative_control": null, + "backend": "accelerated", + "kernel": "native", + "seed": 42, + "blas_threads": 1, + "write_fit_details": true + }, + "inputs": { + "count_table": { + "path": "/home/runner/work/dotmatch/dotmatch/fit-hap1/counts.tsv", + "sha256": "bf24bf255b75cde8cb01279290bdbbbbbba5773596245b8c0300b07b879c36bf" + }, + "design_matrix": { + "path": "/home/runner/work/dotmatch/dotmatch/fit-hap1/design.tsv", + "sha256": "bc4f06905509e0180146e20ae3d3228a59406b2422ed193e7f0e0e7cc9fd5d2e" + }, + "sgrna_efficiency": null, + "control_sgrna": null, + "control_gene": null + }, + "blas": [ + { + "user_api": "blas", + "internal_api": "openblas", + "num_threads": 1, + "prefix": "libscipy_openblas", + "filepath": "/opt/hostedtoolcache/Python/3.12.14/x64/lib/python3.12/site-packages/numpy.libs/libscipy_openblas64_-fdde5778.so", + "version": "0.3.30", + "threading_layer": "pthreads", + "architecture": "Haswell" + }, + { + "user_api": "blas", + "internal_api": "openblas", + "num_threads": 1, + "prefix": "libscipy_openblas", + "filepath": "/opt/hostedtoolcache/Python/3.12.14/x64/lib/python3.12/site-packages/scipy.libs/libscipy_openblas-6cdc3b4a.so", + "version": "0.3.30", + "threading_layer": "pthreads", + "architecture": "Haswell" + } + ], + "outputs": { + "gene_summary": { + "path": "/home/runner/work/dotmatch/dotmatch/fit-hap1/paired/accelerated-0.gene_summary.txt", + "sha256": "dfeb15a7e1f113891a56b8cd23d575ffb124289d6c948367d1027c20ce2256ee" + }, + "sgrna_summary": { + "path": "/home/runner/work/dotmatch/dotmatch/fit-hap1/paired/accelerated-0.sgrna_summary.txt", + "sha256": "16f5ded25d4b6321808ea405132cbf39c6fe1185aec1bf6dced07ca9e9a2bd29" + }, + "fit_details": { + "path": "/home/runner/work/dotmatch/dotmatch/fit-hap1/paired/accelerated-0.fit-details.json", + "sha256": "3d0a1ad86a65e537f72ab6dd1cebf971743c12e2ca4024c787364f01a7107e11" + } + }, + "elapsed_seconds": 37.80452142799999 + } + }, + { + "backend": "numpy", + "repeat": 0, + "wall_seconds_including_startup": 131.05977701799998, + "workflow_seconds": 130.05577189500002, + "command": [ + "/opt/hostedtoolcache/Python/3.12.14/x64/bin/python", + "-m", + "crisprworks_fit", + "mle", + "-k", + "fit-hap1/counts.tsv", + "-d", + "fit-hap1/design.tsv", + "-n", + "fit-hap1/paired/numpy-0", + "--backend", + "accelerated", + "--threads", + "1", + "--blas-threads", + "1", + "--seed", + "42", + "--permutation-round", + "2", + "--kernel", + "numpy", + "--write-fit-details", + "--genes-varmodeling", + "1000", + "--update-efficiency" + ], + "manifest": { + "schema_version": 1, + "status": "complete", + "tool": "CRISPRWorks Fit", + "version": "0.1.0a2", + "backend": "accelerated", + "kernel": "numpy", + "mageck2_version": "0.3.0", + "reference_commit": "630aea0b6fc152a81006435f21d629297275c911", + "python": "3.12.14", + "numpy": "2.3.5", + "scipy": "1.17.0", + "platform": "Linux-6.17.0-1022-azure-x86_64-with-glibc2.39", + "options": { + "subcmd": "mle", + "count_table": "fit-hap1/counts.tsv", + "design_matrix": "fit-hap1/design.tsv", + "day0_label": null, + "output_prefix": "fit-hap1/paired/numpy-0", + "include_samples": null, + "beta_labels": null, + "control_sgrna": null, + "control_gene": null, + "cnv_norm": null, + "cell_line": null, + "cnv_est": null, + "debug": false, + "debug_gene": null, + "norm_method": "median", + "genes_varmodeling": 1000, + "permutation_round": 2, + "no_permutation_by_group": false, + "max_sgrnapergene_permutation": 40, + "remove_outliers": false, + "threads": 1, + "adjust_method": "fdr", + "sgrna_efficiency": null, + "sgrna_eff_name_column": 0, + "sgrna_eff_score_column": 1, + "update_efficiency": true, + "bayes": false, + "PPI_prior": false, + "PPI_weighting": null, + "negative_control": null, + "backend": "accelerated", + "kernel": "numpy", + "seed": 42, + "blas_threads": 1, + "write_fit_details": true + }, + "inputs": { + "count_table": { + "path": "/home/runner/work/dotmatch/dotmatch/fit-hap1/counts.tsv", + "sha256": "bf24bf255b75cde8cb01279290bdbbbbbba5773596245b8c0300b07b879c36bf" + }, + "design_matrix": { + "path": "/home/runner/work/dotmatch/dotmatch/fit-hap1/design.tsv", + "sha256": "bc4f06905509e0180146e20ae3d3228a59406b2422ed193e7f0e0e7cc9fd5d2e" + }, + "sgrna_efficiency": null, + "control_sgrna": null, + "control_gene": null + }, + "blas": [ + { + "user_api": "blas", + "internal_api": "openblas", + "num_threads": 1, + "prefix": "libscipy_openblas", + "filepath": "/opt/hostedtoolcache/Python/3.12.14/x64/lib/python3.12/site-packages/numpy.libs/libscipy_openblas64_-fdde5778.so", + "version": "0.3.30", + "threading_layer": "pthreads", + "architecture": "Haswell" + }, + { + "user_api": "blas", + "internal_api": "openblas", + "num_threads": 1, + "prefix": "libscipy_openblas", + "filepath": "/opt/hostedtoolcache/Python/3.12.14/x64/lib/python3.12/site-packages/scipy.libs/libscipy_openblas-6cdc3b4a.so", + "version": "0.3.30", + "threading_layer": "pthreads", + "architecture": "Haswell" + } + ], + "outputs": { + "gene_summary": { + "path": "/home/runner/work/dotmatch/dotmatch/fit-hap1/paired/numpy-0.gene_summary.txt", + "sha256": "dfeb15a7e1f113891a56b8cd23d575ffb124289d6c948367d1027c20ce2256ee" + }, + "sgrna_summary": { + "path": "/home/runner/work/dotmatch/dotmatch/fit-hap1/paired/numpy-0.sgrna_summary.txt", + "sha256": "16f5ded25d4b6321808ea405132cbf39c6fe1185aec1bf6dced07ca9e9a2bd29" + }, + "fit_details": { + "path": "/home/runner/work/dotmatch/dotmatch/fit-hap1/paired/numpy-0.fit-details.json", + "sha256": "b5693aa3831b79707e6e5ec2ad5409cc1ef7bcca1cf5b8160ffef496137d3c05" + } + }, + "elapsed_seconds": 130.05577189500002 + } + }, + { + "backend": "accelerated", + "repeat": 1, + "wall_seconds_including_startup": 38.589507603000015, + "workflow_seconds": 37.62997877099997, + "command": [ + "/opt/hostedtoolcache/Python/3.12.14/x64/bin/python", + "-m", + "crisprworks_fit", + "mle", + "-k", + "fit-hap1/counts.tsv", + "-d", + "fit-hap1/design.tsv", + "-n", + "fit-hap1/paired/accelerated-1", + "--backend", + "accelerated", + "--threads", + "1", + "--blas-threads", + "1", + "--seed", + "42", + "--permutation-round", + "2", + "--kernel", + "native", + "--write-fit-details", + "--genes-varmodeling", + "1000", + "--update-efficiency" + ], + "manifest": { + "schema_version": 1, + "status": "complete", + "tool": "CRISPRWorks Fit", + "version": "0.1.0a2", + "backend": "accelerated", + "kernel": "native", + "mageck2_version": "0.3.0", + "reference_commit": "630aea0b6fc152a81006435f21d629297275c911", + "python": "3.12.14", + "numpy": "2.3.5", + "scipy": "1.17.0", + "platform": "Linux-6.17.0-1022-azure-x86_64-with-glibc2.39", + "options": { + "subcmd": "mle", + "count_table": "fit-hap1/counts.tsv", + "design_matrix": "fit-hap1/design.tsv", + "day0_label": null, + "output_prefix": "fit-hap1/paired/accelerated-1", + "include_samples": null, + "beta_labels": null, + "control_sgrna": null, + "control_gene": null, + "cnv_norm": null, + "cell_line": null, + "cnv_est": null, + "debug": false, + "debug_gene": null, + "norm_method": "median", + "genes_varmodeling": 1000, + "permutation_round": 2, + "no_permutation_by_group": false, + "max_sgrnapergene_permutation": 40, + "remove_outliers": false, + "threads": 1, + "adjust_method": "fdr", + "sgrna_efficiency": null, + "sgrna_eff_name_column": 0, + "sgrna_eff_score_column": 1, + "update_efficiency": true, + "bayes": false, + "PPI_prior": false, + "PPI_weighting": null, + "negative_control": null, + "backend": "accelerated", + "kernel": "native", + "seed": 42, + "blas_threads": 1, + "write_fit_details": true + }, + "inputs": { + "count_table": { + "path": "/home/runner/work/dotmatch/dotmatch/fit-hap1/counts.tsv", + "sha256": "bf24bf255b75cde8cb01279290bdbbbbbba5773596245b8c0300b07b879c36bf" + }, + "design_matrix": { + "path": "/home/runner/work/dotmatch/dotmatch/fit-hap1/design.tsv", + "sha256": "bc4f06905509e0180146e20ae3d3228a59406b2422ed193e7f0e0e7cc9fd5d2e" + }, + "sgrna_efficiency": null, + "control_sgrna": null, + "control_gene": null + }, + "blas": [ + { + "user_api": "blas", + "internal_api": "openblas", + "num_threads": 1, + "prefix": "libscipy_openblas", + "filepath": "/opt/hostedtoolcache/Python/3.12.14/x64/lib/python3.12/site-packages/numpy.libs/libscipy_openblas64_-fdde5778.so", + "version": "0.3.30", + "threading_layer": "pthreads", + "architecture": "Haswell" + }, + { + "user_api": "blas", + "internal_api": "openblas", + "num_threads": 1, + "prefix": "libscipy_openblas", + "filepath": "/opt/hostedtoolcache/Python/3.12.14/x64/lib/python3.12/site-packages/scipy.libs/libscipy_openblas-6cdc3b4a.so", + "version": "0.3.30", + "threading_layer": "pthreads", + "architecture": "Haswell" + } + ], + "outputs": { + "gene_summary": { + "path": "/home/runner/work/dotmatch/dotmatch/fit-hap1/paired/accelerated-1.gene_summary.txt", + "sha256": "dfeb15a7e1f113891a56b8cd23d575ffb124289d6c948367d1027c20ce2256ee" + }, + "sgrna_summary": { + "path": "/home/runner/work/dotmatch/dotmatch/fit-hap1/paired/accelerated-1.sgrna_summary.txt", + "sha256": "16f5ded25d4b6321808ea405132cbf39c6fe1185aec1bf6dced07ca9e9a2bd29" + }, + "fit_details": { + "path": "/home/runner/work/dotmatch/dotmatch/fit-hap1/paired/accelerated-1.fit-details.json", + "sha256": "3d0a1ad86a65e537f72ab6dd1cebf971743c12e2ca4024c787364f01a7107e11" + } + }, + "elapsed_seconds": 37.62997877099997 + } + }, + { + "backend": "numpy", + "repeat": 1, + "wall_seconds_including_startup": 129.38647600599995, + "workflow_seconds": 128.41233042300007, + "command": [ + "/opt/hostedtoolcache/Python/3.12.14/x64/bin/python", + "-m", + "crisprworks_fit", + "mle", + "-k", + "fit-hap1/counts.tsv", + "-d", + "fit-hap1/design.tsv", + "-n", + "fit-hap1/paired/numpy-1", + "--backend", + "accelerated", + "--threads", + "1", + "--blas-threads", + "1", + "--seed", + "42", + "--permutation-round", + "2", + "--kernel", + "numpy", + "--write-fit-details", + "--genes-varmodeling", + "1000", + "--update-efficiency" + ], + "manifest": { + "schema_version": 1, + "status": "complete", + "tool": "CRISPRWorks Fit", + "version": "0.1.0a2", + "backend": "accelerated", + "kernel": "numpy", + "mageck2_version": "0.3.0", + "reference_commit": "630aea0b6fc152a81006435f21d629297275c911", + "python": "3.12.14", + "numpy": "2.3.5", + "scipy": "1.17.0", + "platform": "Linux-6.17.0-1022-azure-x86_64-with-glibc2.39", + "options": { + "subcmd": "mle", + "count_table": "fit-hap1/counts.tsv", + "design_matrix": "fit-hap1/design.tsv", + "day0_label": null, + "output_prefix": "fit-hap1/paired/numpy-1", + "include_samples": null, + "beta_labels": null, + "control_sgrna": null, + "control_gene": null, + "cnv_norm": null, + "cell_line": null, + "cnv_est": null, + "debug": false, + "debug_gene": null, + "norm_method": "median", + "genes_varmodeling": 1000, + "permutation_round": 2, + "no_permutation_by_group": false, + "max_sgrnapergene_permutation": 40, + "remove_outliers": false, + "threads": 1, + "adjust_method": "fdr", + "sgrna_efficiency": null, + "sgrna_eff_name_column": 0, + "sgrna_eff_score_column": 1, + "update_efficiency": true, + "bayes": false, + "PPI_prior": false, + "PPI_weighting": null, + "negative_control": null, + "backend": "accelerated", + "kernel": "numpy", + "seed": 42, + "blas_threads": 1, + "write_fit_details": true + }, + "inputs": { + "count_table": { + "path": "/home/runner/work/dotmatch/dotmatch/fit-hap1/counts.tsv", + "sha256": "bf24bf255b75cde8cb01279290bdbbbbbba5773596245b8c0300b07b879c36bf" + }, + "design_matrix": { + "path": "/home/runner/work/dotmatch/dotmatch/fit-hap1/design.tsv", + "sha256": "bc4f06905509e0180146e20ae3d3228a59406b2422ed193e7f0e0e7cc9fd5d2e" + }, + "sgrna_efficiency": null, + "control_sgrna": null, + "control_gene": null + }, + "blas": [ + { + "user_api": "blas", + "internal_api": "openblas", + "num_threads": 1, + "prefix": "libscipy_openblas", + "filepath": "/opt/hostedtoolcache/Python/3.12.14/x64/lib/python3.12/site-packages/numpy.libs/libscipy_openblas64_-fdde5778.so", + "version": "0.3.30", + "threading_layer": "pthreads", + "architecture": "Haswell" + }, + { + "user_api": "blas", + "internal_api": "openblas", + "num_threads": 1, + "prefix": "libscipy_openblas", + "filepath": "/opt/hostedtoolcache/Python/3.12.14/x64/lib/python3.12/site-packages/scipy.libs/libscipy_openblas-6cdc3b4a.so", + "version": "0.3.30", + "threading_layer": "pthreads", + "architecture": "Haswell" + } + ], + "outputs": { + "gene_summary": { + "path": "/home/runner/work/dotmatch/dotmatch/fit-hap1/paired/numpy-1.gene_summary.txt", + "sha256": "dfeb15a7e1f113891a56b8cd23d575ffb124289d6c948367d1027c20ce2256ee" + }, + "sgrna_summary": { + "path": "/home/runner/work/dotmatch/dotmatch/fit-hap1/paired/numpy-1.sgrna_summary.txt", + "sha256": "16f5ded25d4b6321808ea405132cbf39c6fe1185aec1bf6dced07ca9e9a2bd29" + }, + "fit_details": { + "path": "/home/runner/work/dotmatch/dotmatch/fit-hap1/paired/numpy-1.fit-details.json", + "sha256": "b5693aa3831b79707e6e5ec2ad5409cc1ef7bcca1cf5b8160ffef496137d3c05" + } + }, + "elapsed_seconds": 128.41233042300007 + } + }, + { + "backend": "reference", + "repeat": 1, + "wall_seconds_including_startup": 330.83726006200004, + "workflow_seconds": 329.8490432809999, + "command": [ + "/opt/hostedtoolcache/Python/3.12.14/x64/bin/python", + "-m", + "crisprworks_fit", + "mle", + "-k", + "fit-hap1/counts.tsv", + "-d", + "fit-hap1/design.tsv", + "-n", + "fit-hap1/paired/reference-1", + "--backend", + "reference", + "--threads", + "1", + "--blas-threads", + "1", + "--seed", + "42", + "--permutation-round", + "2", + "--kernel", + "native", + "--write-fit-details", + "--genes-varmodeling", + "1000", + "--update-efficiency" + ], + "manifest": { + "schema_version": 1, + "status": "complete", + "tool": "CRISPRWorks Fit", + "version": "0.1.0a2", + "backend": "reference", + "kernel": "reference", + "mageck2_version": "0.3.0", + "reference_commit": "630aea0b6fc152a81006435f21d629297275c911", + "python": "3.12.14", + "numpy": "2.3.5", + "scipy": "1.17.0", + "platform": "Linux-6.17.0-1022-azure-x86_64-with-glibc2.39", + "options": { + "subcmd": "mle", + "count_table": "fit-hap1/counts.tsv", + "design_matrix": "fit-hap1/design.tsv", + "day0_label": null, + "output_prefix": "fit-hap1/paired/reference-1", + "include_samples": null, + "beta_labels": null, + "control_sgrna": null, + "control_gene": null, + "cnv_norm": null, + "cell_line": null, + "cnv_est": null, + "debug": false, + "debug_gene": null, + "norm_method": "median", + "genes_varmodeling": 1000, + "permutation_round": 2, + "no_permutation_by_group": false, + "max_sgrnapergene_permutation": 40, + "remove_outliers": false, + "threads": 1, + "adjust_method": "fdr", + "sgrna_efficiency": null, + "sgrna_eff_name_column": 0, + "sgrna_eff_score_column": 1, + "update_efficiency": true, + "bayes": false, + "PPI_prior": false, + "PPI_weighting": null, + "negative_control": null, + "backend": "reference", + "kernel": "native", + "seed": 42, + "blas_threads": 1, + "write_fit_details": true + }, + "inputs": { + "count_table": { + "path": "/home/runner/work/dotmatch/dotmatch/fit-hap1/counts.tsv", + "sha256": "bf24bf255b75cde8cb01279290bdbbbbbba5773596245b8c0300b07b879c36bf" + }, + "design_matrix": { + "path": "/home/runner/work/dotmatch/dotmatch/fit-hap1/design.tsv", + "sha256": "bc4f06905509e0180146e20ae3d3228a59406b2422ed193e7f0e0e7cc9fd5d2e" + }, + "sgrna_efficiency": null, + "control_sgrna": null, + "control_gene": null + }, + "blas": [ + { + "user_api": "blas", + "internal_api": "openblas", + "num_threads": 1, + "prefix": "libscipy_openblas", + "filepath": "/opt/hostedtoolcache/Python/3.12.14/x64/lib/python3.12/site-packages/numpy.libs/libscipy_openblas64_-fdde5778.so", + "version": "0.3.30", + "threading_layer": "pthreads", + "architecture": "Haswell" + }, + { + "user_api": "blas", + "internal_api": "openblas", + "num_threads": 1, + "prefix": "libscipy_openblas", + "filepath": "/opt/hostedtoolcache/Python/3.12.14/x64/lib/python3.12/site-packages/scipy.libs/libscipy_openblas-6cdc3b4a.so", + "version": "0.3.30", + "threading_layer": "pthreads", + "architecture": "Haswell" + } + ], + "outputs": { + "gene_summary": { + "path": "/home/runner/work/dotmatch/dotmatch/fit-hap1/paired/reference-1.gene_summary.txt", + "sha256": "dfeb15a7e1f113891a56b8cd23d575ffb124289d6c948367d1027c20ce2256ee" + }, + "sgrna_summary": { + "path": "/home/runner/work/dotmatch/dotmatch/fit-hap1/paired/reference-1.sgrna_summary.txt", + "sha256": "16f5ded25d4b6321808ea405132cbf39c6fe1185aec1bf6dced07ca9e9a2bd29" + }, + "fit_details": { + "path": "/home/runner/work/dotmatch/dotmatch/fit-hap1/paired/reference-1.fit-details.json", + "sha256": "4317c2db4819af2c6ad2fe8659af4ce2bd23cba69ad843bc8a4a091aaae7c082" + } + }, + "elapsed_seconds": 329.8490432809999 + } + }, + { + "backend": "numpy", + "repeat": 2, + "wall_seconds_including_startup": 132.548807613, + "workflow_seconds": 131.574272264, + "command": [ + "/opt/hostedtoolcache/Python/3.12.14/x64/bin/python", + "-m", + "crisprworks_fit", + "mle", + "-k", + "fit-hap1/counts.tsv", + "-d", + "fit-hap1/design.tsv", + "-n", + "fit-hap1/paired/numpy-2", + "--backend", + "accelerated", + "--threads", + "1", + "--blas-threads", + "1", + "--seed", + "42", + "--permutation-round", + "2", + "--kernel", + "numpy", + "--write-fit-details", + "--genes-varmodeling", + "1000", + "--update-efficiency" + ], + "manifest": { + "schema_version": 1, + "status": "complete", + "tool": "CRISPRWorks Fit", + "version": "0.1.0a2", + "backend": "accelerated", + "kernel": "numpy", + "mageck2_version": "0.3.0", + "reference_commit": "630aea0b6fc152a81006435f21d629297275c911", + "python": "3.12.14", + "numpy": "2.3.5", + "scipy": "1.17.0", + "platform": "Linux-6.17.0-1022-azure-x86_64-with-glibc2.39", + "options": { + "subcmd": "mle", + "count_table": "fit-hap1/counts.tsv", + "design_matrix": "fit-hap1/design.tsv", + "day0_label": null, + "output_prefix": "fit-hap1/paired/numpy-2", + "include_samples": null, + "beta_labels": null, + "control_sgrna": null, + "control_gene": null, + "cnv_norm": null, + "cell_line": null, + "cnv_est": null, + "debug": false, + "debug_gene": null, + "norm_method": "median", + "genes_varmodeling": 1000, + "permutation_round": 2, + "no_permutation_by_group": false, + "max_sgrnapergene_permutation": 40, + "remove_outliers": false, + "threads": 1, + "adjust_method": "fdr", + "sgrna_efficiency": null, + "sgrna_eff_name_column": 0, + "sgrna_eff_score_column": 1, + "update_efficiency": true, + "bayes": false, + "PPI_prior": false, + "PPI_weighting": null, + "negative_control": null, + "backend": "accelerated", + "kernel": "numpy", + "seed": 42, + "blas_threads": 1, + "write_fit_details": true + }, + "inputs": { + "count_table": { + "path": "/home/runner/work/dotmatch/dotmatch/fit-hap1/counts.tsv", + "sha256": "bf24bf255b75cde8cb01279290bdbbbbbba5773596245b8c0300b07b879c36bf" + }, + "design_matrix": { + "path": "/home/runner/work/dotmatch/dotmatch/fit-hap1/design.tsv", + "sha256": "bc4f06905509e0180146e20ae3d3228a59406b2422ed193e7f0e0e7cc9fd5d2e" + }, + "sgrna_efficiency": null, + "control_sgrna": null, + "control_gene": null + }, + "blas": [ + { + "user_api": "blas", + "internal_api": "openblas", + "num_threads": 1, + "prefix": "libscipy_openblas", + "filepath": "/opt/hostedtoolcache/Python/3.12.14/x64/lib/python3.12/site-packages/numpy.libs/libscipy_openblas64_-fdde5778.so", + "version": "0.3.30", + "threading_layer": "pthreads", + "architecture": "Haswell" + }, + { + "user_api": "blas", + "internal_api": "openblas", + "num_threads": 1, + "prefix": "libscipy_openblas", + "filepath": "/opt/hostedtoolcache/Python/3.12.14/x64/lib/python3.12/site-packages/scipy.libs/libscipy_openblas-6cdc3b4a.so", + "version": "0.3.30", + "threading_layer": "pthreads", + "architecture": "Haswell" + } + ], + "outputs": { + "gene_summary": { + "path": "/home/runner/work/dotmatch/dotmatch/fit-hap1/paired/numpy-2.gene_summary.txt", + "sha256": "dfeb15a7e1f113891a56b8cd23d575ffb124289d6c948367d1027c20ce2256ee" + }, + "sgrna_summary": { + "path": "/home/runner/work/dotmatch/dotmatch/fit-hap1/paired/numpy-2.sgrna_summary.txt", + "sha256": "16f5ded25d4b6321808ea405132cbf39c6fe1185aec1bf6dced07ca9e9a2bd29" + }, + "fit_details": { + "path": "/home/runner/work/dotmatch/dotmatch/fit-hap1/paired/numpy-2.fit-details.json", + "sha256": "b5693aa3831b79707e6e5ec2ad5409cc1ef7bcca1cf5b8160ffef496137d3c05" + } + }, + "elapsed_seconds": 131.574272264 + } + }, + { + "backend": "reference", + "repeat": 2, + "wall_seconds_including_startup": 332.45245166199993, + "workflow_seconds": 331.4501148710001, + "command": [ + "/opt/hostedtoolcache/Python/3.12.14/x64/bin/python", + "-m", + "crisprworks_fit", + "mle", + "-k", + "fit-hap1/counts.tsv", + "-d", + "fit-hap1/design.tsv", + "-n", + "fit-hap1/paired/reference-2", + "--backend", + "reference", + "--threads", + "1", + "--blas-threads", + "1", + "--seed", + "42", + "--permutation-round", + "2", + "--kernel", + "native", + "--write-fit-details", + "--genes-varmodeling", + "1000", + "--update-efficiency" + ], + "manifest": { + "schema_version": 1, + "status": "complete", + "tool": "CRISPRWorks Fit", + "version": "0.1.0a2", + "backend": "reference", + "kernel": "reference", + "mageck2_version": "0.3.0", + "reference_commit": "630aea0b6fc152a81006435f21d629297275c911", + "python": "3.12.14", + "numpy": "2.3.5", + "scipy": "1.17.0", + "platform": "Linux-6.17.0-1022-azure-x86_64-with-glibc2.39", + "options": { + "subcmd": "mle", + "count_table": "fit-hap1/counts.tsv", + "design_matrix": "fit-hap1/design.tsv", + "day0_label": null, + "output_prefix": "fit-hap1/paired/reference-2", + "include_samples": null, + "beta_labels": null, + "control_sgrna": null, + "control_gene": null, + "cnv_norm": null, + "cell_line": null, + "cnv_est": null, + "debug": false, + "debug_gene": null, + "norm_method": "median", + "genes_varmodeling": 1000, + "permutation_round": 2, + "no_permutation_by_group": false, + "max_sgrnapergene_permutation": 40, + "remove_outliers": false, + "threads": 1, + "adjust_method": "fdr", + "sgrna_efficiency": null, + "sgrna_eff_name_column": 0, + "sgrna_eff_score_column": 1, + "update_efficiency": true, + "bayes": false, + "PPI_prior": false, + "PPI_weighting": null, + "negative_control": null, + "backend": "reference", + "kernel": "native", + "seed": 42, + "blas_threads": 1, + "write_fit_details": true + }, + "inputs": { + "count_table": { + "path": "/home/runner/work/dotmatch/dotmatch/fit-hap1/counts.tsv", + "sha256": "bf24bf255b75cde8cb01279290bdbbbbbba5773596245b8c0300b07b879c36bf" + }, + "design_matrix": { + "path": "/home/runner/work/dotmatch/dotmatch/fit-hap1/design.tsv", + "sha256": "bc4f06905509e0180146e20ae3d3228a59406b2422ed193e7f0e0e7cc9fd5d2e" + }, + "sgrna_efficiency": null, + "control_sgrna": null, + "control_gene": null + }, + "blas": [ + { + "user_api": "blas", + "internal_api": "openblas", + "num_threads": 1, + "prefix": "libscipy_openblas", + "filepath": "/opt/hostedtoolcache/Python/3.12.14/x64/lib/python3.12/site-packages/numpy.libs/libscipy_openblas64_-fdde5778.so", + "version": "0.3.30", + "threading_layer": "pthreads", + "architecture": "Haswell" + }, + { + "user_api": "blas", + "internal_api": "openblas", + "num_threads": 1, + "prefix": "libscipy_openblas", + "filepath": "/opt/hostedtoolcache/Python/3.12.14/x64/lib/python3.12/site-packages/scipy.libs/libscipy_openblas-6cdc3b4a.so", + "version": "0.3.30", + "threading_layer": "pthreads", + "architecture": "Haswell" + } + ], + "outputs": { + "gene_summary": { + "path": "/home/runner/work/dotmatch/dotmatch/fit-hap1/paired/reference-2.gene_summary.txt", + "sha256": "dfeb15a7e1f113891a56b8cd23d575ffb124289d6c948367d1027c20ce2256ee" + }, + "sgrna_summary": { + "path": "/home/runner/work/dotmatch/dotmatch/fit-hap1/paired/reference-2.sgrna_summary.txt", + "sha256": "16f5ded25d4b6321808ea405132cbf39c6fe1185aec1bf6dced07ca9e9a2bd29" + }, + "fit_details": { + "path": "/home/runner/work/dotmatch/dotmatch/fit-hap1/paired/reference-2.fit-details.json", + "sha256": "4317c2db4819af2c6ad2fe8659af4ce2bd23cba69ad843bc8a4a091aaae7c082" + } + }, + "elapsed_seconds": 331.4501148710001 + } + }, + { + "backend": "accelerated", + "repeat": 2, + "wall_seconds_including_startup": 38.68067711100002, + "workflow_seconds": 37.72639602000004, + "command": [ + "/opt/hostedtoolcache/Python/3.12.14/x64/bin/python", + "-m", + "crisprworks_fit", + "mle", + "-k", + "fit-hap1/counts.tsv", + "-d", + "fit-hap1/design.tsv", + "-n", + "fit-hap1/paired/accelerated-2", + "--backend", + "accelerated", + "--threads", + "1", + "--blas-threads", + "1", + "--seed", + "42", + "--permutation-round", + "2", + "--kernel", + "native", + "--write-fit-details", + "--genes-varmodeling", + "1000", + "--update-efficiency" + ], + "manifest": { + "schema_version": 1, + "status": "complete", + "tool": "CRISPRWorks Fit", + "version": "0.1.0a2", + "backend": "accelerated", + "kernel": "native", + "mageck2_version": "0.3.0", + "reference_commit": "630aea0b6fc152a81006435f21d629297275c911", + "python": "3.12.14", + "numpy": "2.3.5", + "scipy": "1.17.0", + "platform": "Linux-6.17.0-1022-azure-x86_64-with-glibc2.39", + "options": { + "subcmd": "mle", + "count_table": "fit-hap1/counts.tsv", + "design_matrix": "fit-hap1/design.tsv", + "day0_label": null, + "output_prefix": "fit-hap1/paired/accelerated-2", + "include_samples": null, + "beta_labels": null, + "control_sgrna": null, + "control_gene": null, + "cnv_norm": null, + "cell_line": null, + "cnv_est": null, + "debug": false, + "debug_gene": null, + "norm_method": "median", + "genes_varmodeling": 1000, + "permutation_round": 2, + "no_permutation_by_group": false, + "max_sgrnapergene_permutation": 40, + "remove_outliers": false, + "threads": 1, + "adjust_method": "fdr", + "sgrna_efficiency": null, + "sgrna_eff_name_column": 0, + "sgrna_eff_score_column": 1, + "update_efficiency": true, + "bayes": false, + "PPI_prior": false, + "PPI_weighting": null, + "negative_control": null, + "backend": "accelerated", + "kernel": "native", + "seed": 42, + "blas_threads": 1, + "write_fit_details": true + }, + "inputs": { + "count_table": { + "path": "/home/runner/work/dotmatch/dotmatch/fit-hap1/counts.tsv", + "sha256": "bf24bf255b75cde8cb01279290bdbbbbbba5773596245b8c0300b07b879c36bf" + }, + "design_matrix": { + "path": "/home/runner/work/dotmatch/dotmatch/fit-hap1/design.tsv", + "sha256": "bc4f06905509e0180146e20ae3d3228a59406b2422ed193e7f0e0e7cc9fd5d2e" + }, + "sgrna_efficiency": null, + "control_sgrna": null, + "control_gene": null + }, + "blas": [ + { + "user_api": "blas", + "internal_api": "openblas", + "num_threads": 1, + "prefix": "libscipy_openblas", + "filepath": "/opt/hostedtoolcache/Python/3.12.14/x64/lib/python3.12/site-packages/numpy.libs/libscipy_openblas64_-fdde5778.so", + "version": "0.3.30", + "threading_layer": "pthreads", + "architecture": "Haswell" + }, + { + "user_api": "blas", + "internal_api": "openblas", + "num_threads": 1, + "prefix": "libscipy_openblas", + "filepath": "/opt/hostedtoolcache/Python/3.12.14/x64/lib/python3.12/site-packages/scipy.libs/libscipy_openblas-6cdc3b4a.so", + "version": "0.3.30", + "threading_layer": "pthreads", + "architecture": "Haswell" + } + ], + "outputs": { + "gene_summary": { + "path": "/home/runner/work/dotmatch/dotmatch/fit-hap1/paired/accelerated-2.gene_summary.txt", + "sha256": "dfeb15a7e1f113891a56b8cd23d575ffb124289d6c948367d1027c20ce2256ee" + }, + "sgrna_summary": { + "path": "/home/runner/work/dotmatch/dotmatch/fit-hap1/paired/accelerated-2.sgrna_summary.txt", + "sha256": "16f5ded25d4b6321808ea405132cbf39c6fe1185aec1bf6dced07ca9e9a2bd29" + }, + "fit_details": { + "path": "/home/runner/work/dotmatch/dotmatch/fit-hap1/paired/accelerated-2.fit-details.json", + "sha256": "3d0a1ad86a65e537f72ab6dd1cebf971743c12e2ca4024c787364f01a7107e11" + } + }, + "elapsed_seconds": 37.72639602000004 + } + } + ], + "identical_gene_summary_bytes": true, + "full_precision_max_absolute_errors": { + "beta_estimate": 2.4868995751603507e-14, + "beta_zscore": 3.26405569239796e-14, + "w_estimate": 1.4432899320127035e-15, + "beta_pval": 1.9761969838327786e-14, + "beta_pval_fdr": 2.0095036745715333e-14, + "beta_permute_pval": 0, + "beta_permute_pval_fdr": 0, + "beta_permute_pval_neg": 0, + "beta_permute_pval_pos": 0, + "beta_permute_pval_neg_fdr": 0, + "beta_permute_pval_pos_fdr": 0 + }, + "full_precision_tolerances": { + "rtol": 1e-07, + "atol": 1e-08 + }, + "median_wall_seconds": { + "reference": 332.45245166199993, + "accelerated": 38.68067711100002, + "numpy": 131.05977701799998 + }, + "median_speedup": 8.594794002906868, + "speedup_over_numpy": 3.3882492967200193 + } +} diff --git a/packages/crisprworks-fit/benchmarks/hap1-native-full-parity.json b/packages/crisprworks-fit/benchmarks/hap1-native-full-parity.json new file mode 100644 index 00000000..d980aad0 --- /dev/null +++ b/packages/crisprworks-fit/benchmarks/hap1-native-full-parity.json @@ -0,0 +1,122 @@ +{ + "scope": "Cross-environment full-table numerical parity only; timing ratios are not valid", + "implementation_commit": "2938b8e6e2219e80404280facf14f323e55ee0c3", + "reference_ci_run": 36712090800, + "reference_artifact": 11095429626, + "genes": 18056, + "identical_gene_summary_bytes": true, + "full_precision_max_absolute_errors": { + "beta_estimate": 9.131584377541913e-15, + "beta_zscore": 5.5289106626332796e-14, + "w_estimate": 2.220446049250313e-15, + "beta_pval": 3.5416114485542494e-14, + "beta_pval_fdr": 3.7636560534792807e-14, + "beta_permute_pval": 0.0, + "beta_permute_pval_fdr": 0.0, + "beta_permute_pval_neg": 0.0, + "beta_permute_pval_pos": 0.0, + "beta_permute_pval_neg_fdr": 0.0, + "beta_permute_pval_pos_fdr": 0.0 + }, + "native_manifest": { + "schema_version": 1, + "status": "complete", + "tool": "CRISPRWorks Fit", + "version": "0.1.0a2", + "backend": "accelerated", + "kernel": "native", + "mageck2_version": "0.3.0", + "reference_commit": "630aea0b6fc152a81006435f21d629297275c911", + "python": "3.12.14", + "numpy": "2.3.5", + "scipy": "1.17.0", + "platform": "Linux-6.18.44-x86_64-with-glibc2.39", + "options": { + "subcmd": "mle", + "count_table": "fit-hap1-inputs/source.tsv", + "design_matrix": "fit-hap1-inputs/design.tsv", + "day0_label": null, + "output_prefix": "fit-sparse-full/screen", + "include_samples": null, + "beta_labels": null, + "control_sgrna": null, + "control_gene": null, + "cnv_norm": null, + "cell_line": null, + "cnv_est": null, + "debug": false, + "debug_gene": null, + "norm_method": "median", + "genes_varmodeling": 1000, + "permutation_round": 2, + "no_permutation_by_group": false, + "max_sgrnapergene_permutation": 40, + "remove_outliers": false, + "threads": 1, + "adjust_method": "fdr", + "sgrna_efficiency": null, + "sgrna_eff_name_column": 0, + "sgrna_eff_score_column": 1, + "update_efficiency": true, + "bayes": false, + "PPI_prior": false, + "PPI_weighting": null, + "negative_control": null, + "backend": "accelerated", + "kernel": "native", + "seed": 42, + "blas_threads": 1, + "write_fit_details": true + }, + "inputs": { + "count_table": { + "path": "/workspace/scratch/cba091351f29/fit-hap1-inputs/source.tsv", + "sha256": "7638bc6237cf8a3e2302fcd969d014b041819669b37f2f87bc1f7cfbc45ca8a7" + }, + "design_matrix": { + "path": "/workspace/scratch/cba091351f29/fit-hap1-inputs/design.tsv", + "sha256": "bc4f06905509e0180146e20ae3d3228a59406b2422ed193e7f0e0e7cc9fd5d2e" + }, + "sgrna_efficiency": null, + "control_sgrna": null, + "control_gene": null + }, + "blas": [ + { + "user_api": "blas", + "internal_api": "openblas", + "num_threads": 1, + "prefix": "libscipy_openblas", + "filepath": "/opt/codex/runtimes/codex-primary-runtime/dependencies/python/lib/python3.12/site-packages/numpy.libs/libscipy_openblas64_-fdde5778.so", + "version": "0.3.30", + "threading_layer": "pthreads", + "architecture": "SkylakeX" + }, + { + "user_api": "blas", + "internal_api": "openblas", + "num_threads": 1, + "prefix": "libscipy_openblas", + "filepath": "/opt/codex/runtimes/codex-primary-runtime/dependencies/python/lib/python3.12/site-packages/scipy.libs/libscipy_openblas-6cdc3b4a.so", + "version": "0.3.30", + "threading_layer": "pthreads", + "architecture": "SkylakeX" + } + ], + "outputs": { + "gene_summary": { + "path": "/workspace/scratch/cba091351f29/fit-sparse-full/screen.gene_summary.txt", + "sha256": "0304f6cd4afaf1b43b2cbc82c2166948200786ed99885b7a05a910e701875792" + }, + "sgrna_summary": { + "path": "/workspace/scratch/cba091351f29/fit-sparse-full/screen.sgrna_summary.txt", + "sha256": "8785435d5a14be7c82354009d5a87289222fc44d475d4fd380627e4b100990d2" + }, + "fit_details": { + "path": "/workspace/scratch/cba091351f29/fit-sparse-full/screen.fit-details.json", + "sha256": "d897f4c7cd2af93a85e67fb8791f6ea6a9eeff855ffcca15dbda451b2e6adbf5" + } + }, + "elapsed_seconds": 151.5760336050007 + } +} diff --git a/packages/crisprworks-fit/benchmarks/public_hap1.py b/packages/crisprworks-fit/benchmarks/public_hap1.py index 4dde78a4..32a2730e 100644 --- a/packages/crisprworks-fit/benchmarks/public_hap1.py +++ b/packages/crisprworks-fit/benchmarks/public_hap1.py @@ -20,6 +20,8 @@ def main(): parser = argparse.ArgumentParser() parser.add_argument("--out-dir", type=Path, required=True) parser.add_argument("--repeats", type=int, default=3) + parser.add_argument("--compare-numpy", action="store_true") + parser.add_argument("--kernel", choices=("auto", "native", "numpy"), default="auto") parser.add_argument("--cohort", choices=("full", "four-guide"), default="full", help="Full table or complete four-guide gene labels; selection is recorded") parser.add_argument("--local-counts", type=Path, @@ -59,6 +61,8 @@ def main(): "--out-dir", str(args.out_dir / "paired"), "--count-table", str(counts), "--design-matrix", str(design), "--update-efficiency", "--genes-varmodeling", "1000", "--repeats", str(args.repeats), + "--kernel", args.kernel, + *(["--compare-numpy"] if args.compare_numpy else []), ], check=True) print((args.out_dir / "paired" / "benchmark.json").read_text(), flush=True) diff --git a/packages/crisprworks-fit/benchmarks/run.py b/packages/crisprworks-fit/benchmarks/run.py index dfe51fc2..94d072db 100644 --- a/packages/crisprworks-fit/benchmarks/run.py +++ b/packages/crisprworks-fit/benchmarks/run.py @@ -24,6 +24,8 @@ def main(): parser.add_argument("--design-matrix", type=Path) parser.add_argument("--update-efficiency", action="store_true") parser.add_argument("--genes-varmodeling", type=int, default=0) + parser.add_argument("--compare-numpy", action="store_true", help="Also benchmark the previous NumPy accelerator") + parser.add_argument("--kernel", choices=("auto", "native", "numpy"), default="auto") parser.add_argument("--timeout", type=int, default=3600, help="Maximum seconds per backend run (default: 3600)") args = parser.parse_args() @@ -60,16 +62,17 @@ def main(): } summaries = [] precisions = [] + backends = ("reference", "accelerated", "numpy") if args.compare_numpy else ("reference", "accelerated") for repeat in range(args.repeats): # Alternate order to reduce a systematic warm-cache advantage. - order = ("reference", "accelerated") if repeat % 2 == 0 else ("accelerated", "reference") + order = backends[repeat % len(backends):] + backends[:repeat % len(backends)] for backend in order: prefix = args.out_dir / f"{backend}-{repeat}" command = [ sys.executable, "-m", "crisprworks_fit", "mle", "-k", str(counts), - "-d", str(design), "-n", str(prefix), "--backend", backend, + "-d", str(design), "-n", str(prefix), "--backend", "accelerated" if backend == "numpy" else backend, "--threads", "1", "--blas-threads", "1", "--seed", "42", "--permutation-round", "2", - "--write-fit-details", "--genes-varmodeling", str(args.genes_varmodeling), + "--kernel", "numpy" if backend == "numpy" else args.kernel, "--write-fit-details", "--genes-varmodeling", str(args.genes_varmodeling), ] if args.update_efficiency: command.append("--update-efficiency") @@ -110,10 +113,12 @@ def main(): medians = { backend: statistics.median(run["wall_seconds_including_startup"] for run in report["runs"] if run["backend"] == backend) - for backend in ("reference", "accelerated") + for backend in backends } report["median_wall_seconds"] = medians report["median_speedup"] = medians["reference"] / medians["accelerated"] + if args.compare_numpy: + report["speedup_over_numpy"] = medians["numpy"] / medians["accelerated"] output = args.out_dir / "benchmark.json" output.write_text(json.dumps(report, indent=2) + "\n") print(json.dumps({key: report[key] for key in ( diff --git a/packages/crisprworks-fit/pyproject.toml b/packages/crisprworks-fit/pyproject.toml index 8b5a36d2..eecc8a5b 100644 --- a/packages/crisprworks-fit/pyproject.toml +++ b/packages/crisprworks-fit/pyproject.toml @@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta" [project] name = "crisprworks-fit" -version = "0.1.0a1" +version = "0.1.0a2" description = "Experimental acceleration of MAGeCK2 MLE for pooled CRISPR screens" readme = "README.md" requires-python = ">=3.10" diff --git a/packages/crisprworks-fit/setup.py b/packages/crisprworks-fit/setup.py new file mode 100644 index 00000000..8586d011 --- /dev/null +++ b/packages/crisprworks-fit/setup.py @@ -0,0 +1,6 @@ +from setuptools import Extension, setup + +setup(ext_modules=[Extension( + "crisprworks_fit._native", ["src/crisprworks_fit/_native.c"], + extra_compile_args=["-O3", "-fno-fast-math", "-ffp-contract=off"], optional=True, +)]) diff --git a/packages/crisprworks-fit/src/crisprworks_fit/__init__.py b/packages/crisprworks-fit/src/crisprworks_fit/__init__.py index 72e13b93..a01aeb74 100644 --- a/packages/crisprworks-fit/src/crisprworks_fit/__init__.py +++ b/packages/crisprworks-fit/src/crisprworks_fit/__init__.py @@ -1,3 +1,3 @@ """CRISPRWorks Fit: experimental acceleration for MAGeCK2 MLE.""" -__version__ = "0.1.0a1" +__version__ = "0.1.0a2" diff --git a/packages/crisprworks-fit/src/crisprworks_fit/_native.c b/packages/crisprworks-fit/src/crisprworks_fit/_native.c new file mode 100644 index 00000000..6ee3e2f1 --- /dev/null +++ b/packages/crisprworks-fit/src/crisprworks_fit/_native.c @@ -0,0 +1,255 @@ +/* MAGeCK-compatible EM loop. BSD-3-Clause; see LICENSE. + * No fast-math, NumPy ABI, or linked BLAS dependency. SciPy special-function + * capsules are checked before use. Input buffers stay alive while the GIL is + * released; all mutable working storage belongs to this call. + */ +#define PY_SSIZE_T_CLEAN +#include +#include +#include +#include +#include + +typedef double (*special1)(double, int); +typedef double (*special2)(double, double, int); +static special1 loggamma_fn; +static special2 xlog1py_fn; + +static int initialize_special(void) { + PyObject *module = PyImport_ImportModule("scipy.special.cython_special"); + if (!module) return -1; + PyObject *api = PyObject_GetAttrString(module, "__pyx_capi__"); + Py_DECREF(module); + if (!api) return -1; + PyObject *gamma = PyDict_GetItemString(api, "gammaln"); + PyObject *xlog = PyDict_GetItemString(api, "__pyx_fuse_1xlog1py"); + if (!gamma || !xlog) { + Py_DECREF(api); + PyErr_SetString(PyExc_ImportError, "Unsupported SciPy special-function C API"); + return -1; + } + loggamma_fn = (special1)PyCapsule_GetPointer(gamma, "double (double, int __pyx_skip_dispatch)"); + xlog1py_fn = (special2)PyCapsule_GetPointer(xlog, "double (double, double, int __pyx_skip_dispatch)"); + Py_DECREF(api); + if (!loggamma_fn || !xlog1py_fn) { + PyErr_Clear(); + PyErr_SetString(PyExc_ImportError, "Unsupported SciPy special-function capsule signature"); + return -1; + } + return 0; +} + +/* Partial-pivot LU for the small ridge system. No extra regularization or + * pseudoinverse fallback: a singular system is reported to the caller. */ +static int solve(double *a, double *b, int p) { + for (int k=0; k fabs(a[pivot*p+k])) pivot=i; + if (a[pivot*p+k] == 0.0) return -1; + if (pivot != k) { + for (int j=0; j=0; i--) { + double value=b[i]; + for (int j=i+1; j0 && p>0 && p<=1) || isnan(k)) return 0; + if (!(k>=0 && k<=INFINITY && k==floor(k))) return -INFINITY; + double value=loggamma_fn(r+k,1)-factorial-loggamma_fn(r,1) + + r*log(p)+xlog1py_fn(k,-p,1); + return isnan(value) ? 0 : value; +} + +static PyObject *loop(PyObject *self, PyObject *args) { + PyObject *objects[6]; Py_buffer buffers[6]={{0}}; + int n,c,other,estimate,update; + double ridge; + PyObject *result=NULL; + double *storage=NULL; + Py_ssize_t *row_offsets=NULL; + int *columns=NULL; + if (!PyArg_ParseTuple(args,"OOOOOOiiidpp",&objects[0],&objects[1],&objects[2], + &objects[3],&objects[4],&objects[5],&n,&c,&other,&ridge,&estimate,&update)) return NULL; + if (n<1 || c<1 || other<1 || n>10000 || c>10000 || other>10000) { + PyErr_SetString(PyExc_ValueError,"Invalid native model dimensions"); return NULL; + } + Py_ssize_t p=(Py_ssize_t)n+c, m=(Py_ssize_t)n*(2*other+1); + if (p>10000 || m>1000000 || m*p>100000000) { + PyErr_SetString(PyExc_ValueError,"Native model exceeds bounded working dimensions"); return NULL; + } + Py_ssize_t lengths[6]={m*p,m,m,p,n,m}; + for (int i=0;i<6;i++) { + if (PyObject_GetBuffer(objects[i],&buffers[i],PyBUF_FORMAT|PyBUF_C_CONTIGUOUS)<0) goto cleanup; + if (!buffers[i].format || strcmp(buffers[i].format,"d") || buffers[i].itemsize!=sizeof(double) + || buffers[i].len!=lengths[i]*(Py_ssize_t)sizeof(double) + || (uintptr_t)buffers[i].buf % _Alignof(double)) { + PyErr_SetString(PyExc_ValueError,"Expected contiguous native float64 buffers of matching size"); goto cleanup; + } + } + /* Seven m-vectors, two p-vectors, one n-vector, and three p*p matrices. */ + Py_ssize_t length=7*m+2*p+n+3*p*p; + storage=PyMem_Calloc((size_t)length,sizeof(double)); + if (!storage) {PyErr_NoMemory(); goto cleanup;} + const double *x=buffers[0].buf,*counts=buffers[1].buf,*sizes=buffers[2].buf,*alpha=buffers[5].buf; + double *beta=storage, *eff=beta+p, *mu=eff+n, *res=mu+m, *weights=res+m; + double *response=weights+m,*rounded=response+m,*factorial=rounded+m,*logmean=factorial+m; + double *proposal=logmean+m,*gram=proposal+p,*regularized=gram+p*p,*factor=regularized+p*p; + memcpy(beta,buffers[3].buf,p*sizeof(double)); memcpy(eff,buffers[4].buf,n*sizeof(double)); + /* Preserve row summation order, but omit exact zero design entries for + * finite inputs. Invalid values use the dense path so 0*NaN conventions + * remain unchanged. No statistical approximation or sparsity threshold. */ + Py_ssize_t nonzero=0; + int finite_design=1; + for (Py_ssize_t i=0;i=n) coefficient=i100) difference=100; + if (isnan(difference) || difference=n) { + double coefficient=imaximum) maximum=w; + } + double floor_weight=missing_weight ? NAN : maximum/100; + for (Py_ssize_t i=0;ilargest) largest=fabs(proposal[j]); + if (j>=n) {double delta=proposal[j]-beta[j];difference+=delta*delta;beta_size+=proposal[j]*proposal[j];} + } + if (isnan(sum) || largest>100) {status=2;break;} + memcpy(beta,proposal,p*sizeof(double)); iteration++; + if (fabs(beta_size)<1e-9) beta_size=1; + if (difference/beta_size<1e-9) break; + if (iteration>1000) {status=1;break;} + } + Py_END_ALLOW_THREADS + if (singular) {PyErr_SetString(PyExc_ArithmeticError,"Singular ridge system");goto cleanup;} + result=PyTuple_New(8); + if (!result) goto cleanup; + PyObject *status_object=PyLong_FromLong(status); + if (!status_object) {Py_CLEAR(result);goto cleanup;} + PyTuple_SET_ITEM(result,0,status_object); + double *outputs[7]={eff,beta,mu,res,weights,gram,regularized}; + Py_ssize_t sizes_out[7]={n,p,m,m,m,p*p,p*p}; + for (int i=0;i<7;i++) { + PyObject *bytes=PyBytes_FromStringAndSize((char*)outputs[i],sizes_out[i]*sizeof(double)); + if (!bytes) {Py_CLEAR(result);goto cleanup;} + PyTuple_SET_ITEM(result,i+1,bytes); + } +cleanup: + PyMem_Free(columns); + PyMem_Free(row_offsets); + PyMem_Free(storage); + for (int i=0;i<6;i++) if (buffers[i].obj) PyBuffer_Release(&buffers[i]); + return result; +} + +static PyMethodDef methods[]={{"loop",loop,METH_VARARGS,"Execute the validated native EM loop."},{NULL,NULL,0,NULL}}; +static struct PyModuleDef module={PyModuleDef_HEAD_INIT,"_native",NULL,-1,methods}; +PyMODINIT_FUNC PyInit__native(void) { + if (initialize_special()<0) return NULL; + return PyModule_Create(&module); +} diff --git a/packages/crisprworks-fit/src/crisprworks_fit/backend.py b/packages/crisprworks-fit/src/crisprworks_fit/backend.py index 30c15b55..390a0a0a 100644 --- a/packages/crisprworks-fit/src/crisprworks_fit/backend.py +++ b/packages/crisprworks-fit/src/crisprworks_fit/backend.py @@ -39,8 +39,9 @@ def verify_upstream(): raise RuntimeError(f"Untested MAGeCK2 implementation of {name}; use the pinned release") -def _worker_init(blas_threads): +def _worker_init(blas_threads, kernel): verify_upstream() + kernels._engine = kernels.resolved_engine(kernel) from mageck2 import mleem from mageck2.mledesignmat import DesignMatCache DesignMatCache.cache = {} @@ -63,10 +64,11 @@ def _fit_one(task): class Runner: """Reuse a worker pool across fitting stages and permutation rounds.""" - def __init__(self, blas_threads=1): + def __init__(self, blas_threads=1, kernel="auto"): self.executor = None self.workers = None self.blas_threads = blas_threads + self.kernel = kernels.resolved_engine(kernel) def close(self): if self.executor is not None: @@ -100,7 +102,7 @@ def __call__(self, allgenedict, args, nproc=1, argsdict=None): self.executor = ProcessPoolExecutor( max_workers=nproc, mp_context=multiprocessing.get_context("spawn"), initializer=_worker_init, - initargs=(self.blas_threads,), + initargs=(self.blas_threads, self.kernel), ) # map yields in input order, preserving gene order for permutations. chunk = max(1, len(tasks) // (nproc * 4)) @@ -109,7 +111,7 @@ def __call__(self, allgenedict, args, nproc=1, argsdict=None): @contextmanager -def accelerated(blas_threads=1): +def accelerated(blas_threads=1, kernel="auto"): """Enable kernels for one workflow, restoring upstream state on exit. Patches are process-global: concurrent calls to unwrapped MAGeCK2 in the @@ -124,7 +126,9 @@ def accelerated(blas_threads=1): from mageck2.mledesignmat import DesignMatCache original = (mleem.em_whileloop, mlemultiprocessing.runem_multiproc, mlemultiprocessing.assign_p_value_from_permuted_beta, DesignMatCache.cache) - runner = Runner(blas_threads) + runner = Runner(blas_threads, kernel) + previous_engine = kernels._engine + kernels._engine = runner.kernel _active = True DesignMatCache.cache = {} mleem.em_whileloop = kernels.em_whileloop @@ -138,4 +142,5 @@ def accelerated(blas_threads=1): finally: (mleem.em_whileloop, mlemultiprocessing.runem_multiproc, mlemultiprocessing.assign_p_value_from_permuted_beta, DesignMatCache.cache) = original + kernels._engine = previous_engine _active = False diff --git a/packages/crisprworks-fit/src/crisprworks_fit/cli.py b/packages/crisprworks-fit/src/crisprworks_fit/cli.py index 521dd17d..edb7f78c 100644 --- a/packages/crisprworks-fit/src/crisprworks_fit/cli.py +++ b/packages/crisprworks-fit/src/crisprworks_fit/cli.py @@ -36,6 +36,7 @@ def parser(): arg_mle(subcommands) mle = subcommands.choices["mle"] mle.add_argument("--backend", choices=("accelerated", "reference"), default="accelerated") + mle.add_argument("--kernel", choices=("auto", "native", "numpy"), default="auto") mle.add_argument("--seed", type=seed_value, default=0, help="Permutation RNG seed; default 0") mle.add_argument("--blas-threads", type=positive_int, default=1, help="BLAS threads per process; default 1") @@ -44,6 +45,7 @@ def parser(): demo = subcommands.add_parser("demo", help="Run a reproducible synthetic screen") demo.add_argument("--out-dir", type=Path, required=True) demo.add_argument("--backend", choices=("accelerated", "reference"), default="accelerated") + demo.add_argument("--kernel", choices=("auto", "native", "numpy"), default="auto") demo.add_argument("--threads", type=positive_int, default=1) return result @@ -101,6 +103,8 @@ def run(args): from .backend import accelerated, verify_upstream, REFERENCE_COMMIT verify_upstream() + from .kernels import resolved_engine + kernel = resolved_engine(args.kernel) if args.backend == "accelerated" else "reference" if args.threads < 1 or args.permutation_round < 1: raise ValueError("--threads and --permutation-round must be at least 1") if args.max_sgrnapergene_permutation < 2: @@ -114,7 +118,7 @@ def run(args): manifest_path = Path(str(prefix) + ".crisprworks.json") manifest = { "schema_version": 1, "status": "running", "tool": "CRISPRWorks Fit", - "version": __version__, "backend": args.backend, + "version": __version__, "backend": args.backend, "kernel": kernel, "mageck2_version": mageck2.__version__, "reference_commit": REFERENCE_COMMIT, "python": platform.python_version(), "numpy": np.__version__, "scipy": scipy.__version__, "platform": platform.platform(), "options": vars(args).copy(), @@ -129,7 +133,7 @@ def run(args): try: np.random.seed(args.seed) DesignMatCache.cache = {} - context = accelerated(args.blas_threads) if args.backend == "accelerated" else nullcontext() + context = accelerated(args.blas_threads, args.kernel) if args.backend == "accelerated" else nullcontext() with threadpool_limits(limits=args.blas_threads), context: manifest["blas"] = threadpool_info() result = mageckmle_main(parsedargs=args) @@ -173,7 +177,7 @@ def main(argv=None): counts, design = synthetic_screen(args.out_dir) args = cli.parse_args([ "mle", "-k", str(counts), "-d", str(design), "-n", str(args.out_dir / "screen"), - "--backend", args.backend, "--threads", str(args.threads), + "--backend", args.backend, "--kernel", args.kernel, "--threads", str(args.threads), ]) try: run(args) diff --git a/packages/crisprworks-fit/src/crisprworks_fit/kernels.py b/packages/crisprworks-fit/src/crisprworks_fit/kernels.py index e065c251..deff8989 100644 --- a/packages/crisprworks-fit/src/crisprworks_fit/kernels.py +++ b/packages/crisprworks-fit/src/crisprworks_fit/kernels.py @@ -7,6 +7,54 @@ import numpy as np from scipy.special import gammaln, xlog1py +try: + from . import _native +except ImportError: + _native = None + +_engine = "auto" + + +def resolved_engine(requested="auto"): + if requested not in ("auto", "native", "numpy"): + raise ValueError("Kernel must be auto, native or numpy") + if requested == "native" and _native is None: + raise RuntimeError("Native Fit extension unavailable; build the package or select --kernel numpy") + return "native" if requested != "numpy" and _native is not None else "numpy" + + +def em_whileloop(sk, beta_init_mat_0, size_vec, wfrac_list_0, + alpha_dispersion, alpha_val, estimateeff, updateeff, + removeoutliers, debug): + if debug or resolved_engine(_engine) == "numpy": + return em_whileloop_numpy(sk, beta_init_mat_0, size_vec, wfrac_list_0, + alpha_dispersion, alpha_val, estimateeff, updateeff, + removeoutliers, debug) + from mageck2.mledesignmat import DesignMatCache + n = sk.nb_count.shape[1] + other = sk.design_mat.shape[0] - 1 + conditions = sk.design_mat.shape[1] - 1 + design = np.require(np.asarray(DesignMatCache.get_record(n)[2]), dtype=np.float64, requirements=["C", "A"]) + m, p = design.shape + def buffer(values): + return np.require(np.asarray(values).reshape(-1), dtype=np.float64, requirements=["C", "A"]) + dispersion = np.require(np.broadcast_to(np.asarray(alpha_dispersion), (m,)), dtype=np.float64, requirements=["C", "A"]) + try: + result = _native.loop(design, buffer(sk.sgrna_kvalue), buffer(size_vec), + buffer(beta_init_mat_0), buffer(wfrac_list_0), dispersion, + n, conditions, other, alpha_val, estimateeff, updateeff) + except ArithmeticError as error: + raise np.linalg.LinAlgError(str(error)) from error + efficiency, beta, mean, residual, weights, gram, regularized = ( + np.frombuffer(value, dtype=np.float64) for value in result[1:] + ) + mean = mean.reshape(m, 1) + residual = residual.reshape(m, 1) + covariance = uncertainty(design, weights, gram.reshape(p, p), + regularized.reshape(p, p), residual / mean, n) + return (result[0], efficiency.copy(), np.matrix(beta.reshape(p, 1)), + np.matrix(covariance), np.matrix(mean), np.matrix(residual)) + def negative_binomial_loglikelihood(counts, mean, dispersion): """Evaluate the reference NB log PMF without scipy.stats argument parsing. @@ -63,7 +111,7 @@ def uncertainty(design, weights, gram, regularized, residual_ratio, n_guides): return scale * covariance -def em_whileloop(sk, beta_init_mat_0, size_vec, wfrac_list_0, +def em_whileloop_numpy(sk, beta_init_mat_0, size_vec, wfrac_list_0, alpha_dispersion, alpha_val, estimateeff, updateeff, removeoutliers, debug): """Replacement for the pinned MAGeCK2 EM loop, retaining its conventions. diff --git a/packages/crisprworks-fit/tests/test_native.py b/packages/crisprworks-fit/tests/test_native.py new file mode 100644 index 00000000..717599c7 --- /dev/null +++ b/packages/crisprworks-fit/tests/test_native.py @@ -0,0 +1,83 @@ +"""Compiled engine contracts and direct comparison with the NumPy accelerator.""" +import copy +import unittest +import warnings +from unittest.mock import patch +import numpy as np +from numpy.testing import assert_allclose +from mageck2 import mleem +from crisprworks_fit import kernels +from crisprworks_fit.backend import accelerated +from test_fit import gene_case, VarianceModel + + +class NativeTests(unittest.TestCase): + def test_selection_and_state_restoration(self): + previous = kernels._engine + with accelerated(kernel="numpy"): + self.assertEqual(kernels._engine, "numpy") + self.assertEqual(kernels._engine, previous) + with patch.object(kernels, "_native", None): + self.assertEqual(kernels.resolved_engine("auto"), "numpy") + with self.assertRaisesRegex(RuntimeError, "unavailable"): + kernels.resolved_engine("native") + with self.assertRaises(ValueError): + kernels.resolved_engine("other") + + @unittest.skipIf(kernels._native is None, "Native extension not installed") + def test_native_matches_numpy_for_matrix_designs_and_low_counts(self): + for guides in (1, 4, 12, 24): + for seed in (2, 17): + original = gene_case(seed, guides) + original.design_mat = np.matrix(original.design_mat) + original.nb_count += 1 + original.nb_count[-1, 0] = 1 + original.nb_count[-2, 0] = 2 + settings = dict(debug=False, logem=False, estimateeff=True, + updateeff=True, meanvarmodel=VarianceModel()) + expected, actual = copy.deepcopy(original), copy.deepcopy(original) + with accelerated(kernel="numpy"): + mleem.iteratenbem(expected, **settings) + with accelerated(kernel="native"): + mleem.iteratenbem(actual, **settings) + for field in ("beta_estimate", "beta_zscore", "beta_pval", "w_estimate", + "mu_estimate", "sgrna_residule"): + assert_allclose(getattr(actual, field), getattr(expected, field), + rtol=1e-7, atol=1e-8, err_msg=field) + + @unittest.skipIf(kernels._native is None, "Native extension not installed") + def test_invalid_buffers_are_rejected_before_computation(self): + arrays = [np.ones((3, 2)), np.ones(3), np.ones(3), np.ones(2), np.ones(1), np.ones(3)] + for index in range(6): + bad = list(arrays) + bad[index] = np.ones(1, dtype=np.float32) + with self.assertRaises(ValueError): + kernels._native.loop(*bad, 1, 1, 1, .01, True, True) + unaligned = np.ndarray((3,), dtype=np.float64, buffer=bytearray(25), offset=1) + arrays[1] = unaligned + with self.assertRaises(ValueError): + kernels._native.loop(*arrays, 1, 1, 1, .01, True, True) + with self.assertRaises(ValueError): + kernels._native.loop(*arrays, -1, 1, 1, .01, True, True) + + @unittest.skipIf(kernels._native is None, "Native extension not installed") + def test_nonfinite_initialization_matches_numpy_failure_conventions(self): + original = gene_case(guides=4) + original.nb_count[0, 0] = 0 + expected, actual = copy.deepcopy(original), copy.deepcopy(original) + with warnings.catch_warnings(): + warnings.simplefilter("ignore", RuntimeWarning) + for kernel, gene in (("numpy", expected), ("native", actual)): + with accelerated(kernel=kernel): + mleem.iteratenbem(gene, debug=False, logem=False, estimateeff=True, updateeff=True) + for field in ("beta_estimate", "beta_zscore", "beta_pval", "w_estimate", + "mu_estimate", "sgrna_residule"): + assert_allclose(getattr(actual, field), getattr(expected, field), + rtol=1e-7, atol=1e-8, equal_nan=True) + + @unittest.skipIf(kernels._native is None, "Native extension not installed") + def test_native_singular_solve_rejects_without_extra_regularization(self): + arrays = [np.ones((3, 2)), np.full(3, 10.), np.ones(3), + np.ones(2), np.ones(1), np.full(3, .1)] + with self.assertRaisesRegex(ArithmeticError, "Singular"): + kernels._native.loop(*arrays, 1, 1, 1, 0., False, False)