From 1e9b21847bc0f0638b6759c3cf0038c876c22815 Mon Sep 17 00:00:00 2001 From: jeebjean Date: Fri, 10 Jul 2026 10:28:56 -0500 Subject: [PATCH 01/29] Add Logistic PCA (LPCA) analysis Adds an analysis.lpca config block (include, k, m, cv, transpose) with a matching LpcaAnalysis schema, an lpca_analysis Snakemake rule restricted to algorithms with multiple parameter combinations, and an LPCA analysis module that builds the binary edge-by-run matrix via summarize_networks and runs the logisticPCA container through run_container_and_log. m is fixed by default; cross-validation and matrix transposition are opt-in. --- Snakefile | 21 ++++++++++ config/config.yaml | 16 +++++++ spras/analysis/lpca.py | 94 ++++++++++++++++++++++++++++++++++++++++++ spras/config/config.py | 5 +++ spras/config/schema.py | 10 +++++ 5 files changed, 146 insertions(+) create mode 100644 spras/analysis/lpca.py diff --git a/Snakefile b/Snakefile index 5ad7aa185..82f208e4f 100644 --- a/Snakefile +++ b/Snakefile @@ -91,6 +91,9 @@ def make_final_input(wildcards): final_input.extend(expand('{out_dir}{sep}{dataset}-ml{sep}ensemble-pathway.txt',out_dir=out_dir,sep=SEP,dataset=dataset_labels,algorithm_params=algorithms_with_params)) final_input.extend(expand('{out_dir}{sep}{dataset}-ml{sep}jaccard-matrix.txt',out_dir=out_dir,sep=SEP,dataset=dataset_labels,algorithm_params=algorithms_with_params)) final_input.extend(expand('{out_dir}{sep}{dataset}-ml{sep}jaccard-heatmap.png',out_dir=out_dir,sep=SEP,dataset=dataset_labels,algorithm_params=algorithms_with_params)) + + if _config.config.analysis_include_lpca: + final_input.extend(expand('{out_dir}{sep}{dataset}-ml{sep}{algorithm}-lpca-scores.csv',out_dir=out_dir, sep=SEP, dataset=dataset_labels, algorithm=algorithms_mult_param_combos)) if _config.config.analysis_include_ml_aggregate_algo: final_input.extend(expand('{out_dir}{sep}{dataset}-ml{sep}{algorithm}-pca.png',out_dir=out_dir,sep=SEP,dataset=dataset_labels,algorithm=algorithms_mult_param_combos)) @@ -356,6 +359,7 @@ rule ml_analysis: ml.hac_horizontal(summary_df, output.hac_image_horizontal, output.hac_clusters_horizontal, **hac_params) ml.pca(summary_df, output.pca_image, output.pca_variance, output.pca_coordinates, **pca_params) + # Calculated Jaccard similarity between output pathways for each dataset rule jaccard_similarity: input: @@ -403,6 +407,23 @@ rule ml_analysis_aggregate_algo: ml.hac_horizontal(summary_df, output.hac_image_horizontal, output.hac_clusters_horizontal, **hac_params) ml.pca(summary_df, output.pca_image, output.pca_variance, output.pca_coordinates, **pca_params) +rule lpca_analysis: + input: + pathways = collect_pathways_per_algo + output: + lpca_scores = SEP.join([out_dir, '{dataset}-ml', '{algorithm}-lpca-scores.csv']) + run: + from spras.analysis import lpca + lpca.run_lpca( + input.pathways, + output.lpca_scores, + k=_config.config.lpca_params.k, + m=_config.config.lpca_params.m, + cv=_config.config.lpca_params.cv, + transpose=_config.config.lpca_params.transpose, + container_settings=container_settings + ) + # Ensemble the output pathways for each dataset per algorithm rule ensemble_per_algo: input: diff --git a/config/config.yaml b/config/config.yaml index ef51de739..55c8d3afa 100644 --- a/config/config.yaml +++ b/config/config.yaml @@ -272,3 +272,19 @@ analysis: # adds evaluation per algorithm per dataset-goldstandard pair # evaluation per algorithm will not run unless ml include and ml aggregate_per_algorithm are set to true aggregate_per_algorithm: true + # Logistic PCA (LPCA) of the binary edge-by-run matrix, one embedding per algorithm + # only runs for algorithms with multiple parameter combinations chosen + lpca: + # run the LPCA analysis per algorithm + # LPCA needs enough observations to be meaningful + include: false + # number of principal components to compute + k: 2 + # fixed value of the logisticPCA tuning parameter m, used when cv is false + m: 6 + # if true, choose m by cross-validation; if false, use the fixed m above + cv: false + # matrix orientation given to LPCA: + # false = edges x runs (observations are the edges) + # true = runs x edges (observations are the runs, mirroring the classic ml pca analysis) + transpose: false diff --git a/spras/analysis/lpca.py b/spras/analysis/lpca.py new file mode 100644 index 000000000..295fad1ae --- /dev/null +++ b/spras/analysis/lpca.py @@ -0,0 +1,94 @@ +# Logistic PCA analysis for SPRAS +# Runs LPCA on the binary edge x algorithm matrix produced by summarize_networks. +# Configured through the analysis.lpca block (k, m, cv, transpose). + +from os import PathLike +from pathlib import Path +from typing import Iterable, Union + +import pandas as pd + +from spras.analysis.ml import summarize_networks +from spras.config.container_schema import ProcessedContainerSettings +from spras.containers import prepare_volume, run_container_and_log + +# Published as docker.io/reedcompbio/lpca:v1. Only the suffix is given here; +# the registry prefix is resolved from the container settings. +LPCA_CONTAINER_SUFFIX = 'lpca:v1' +LPCA_WORK_DIR = '/app' + + +def run_lpca( + file_paths: Iterable[Union[str, PathLike]], + output_scores: str, + k: int = 2, + m: float = 6, + cv: bool = False, + transpose: bool = False, + container_settings=None +) -> None: + """ + Runs Logistic PCA on the binary edge x algorithm matrix built from SPRAS + algorithm output files. + + @param file_paths: pathway.txt file paths from SPRAS algorithm outputs + @param output_scores: path to write the LPCA PC scores CSV + @param k: number of principal components (default 2) + @param m: fixed logisticPCA tuning parameter, used when cv is False + @param cv: if True, determine m by cross-validation; if False, use the + fixed m directly (default False) + @param transpose: if True, run LPCA on the transposed (runs x edges) matrix + instead of the default (edges x runs) + @param container_settings: configure the container runtime (Docker or Singularity) + """ + if not container_settings: + container_settings = ProcessedContainerSettings() + + # Step 1: build the binary edge x algorithm matrix + print('LPCA: Building binary edge x algorithm matrix...') + matrix = summarize_networks(file_paths) + + # Optionally transpose so the runs become the observations rather than the edges + if transpose: + matrix = matrix.T + print(f'LPCA: Matrix shape: {matrix.shape}') + + # Step 2: write the matrix next to the outputs, namespaced by algorithm + output_dir = Path(output_scores).parent + output_dir.mkdir(parents=True, exist_ok=True) + algo_name = Path(output_scores).name.replace('-lpca-scores.csv', '') + matrix_path = str(output_dir / f'{algo_name}-lpca_binary_matrix.csv') + matrix.to_csv(matrix_path) + print(f'LPCA: Binary matrix saved to {matrix_path}') + + # Step 3: mount the matrix and the scores output + volumes = [] + bind_path, mapped_matrix = prepare_volume(matrix_path, LPCA_WORK_DIR, container_settings) + volumes.append(bind_path) + bind_path, mapped_scores = prepare_volume(output_scores, LPCA_WORK_DIR, container_settings) + volumes.append(bind_path) + + # Step 4: choose m, optionally via cross-validation + if cv: + cv_output_path = str(output_dir / f'{algo_name}-lpca_cv_result.csv') + bind_path, mapped_cv_output = prepare_volume(cv_output_path, LPCA_WORK_DIR, container_settings) + volumes.append(bind_path) + + print(f'LPCA: Running cross-validation with k={k}...') + command_cv = ['Rscript', '/app/run_cv.R', mapped_matrix, mapped_cv_output, str(k)] + run_container_and_log('LPCA-CV', LPCA_CONTAINER_SUFFIX, command_cv, volumes, + LPCA_WORK_DIR, None, container_settings) + + m_used = pd.read_csv(cv_output_path)['best_m'][0] + print(f'LPCA: Best m found by CV: {m_used}') + else: + m_used = m + print(f'LPCA: Using fixed m={m_used}') + + # Step 5: run LPCA with the chosen k and m + print(f'LPCA: Running LPCA with k={k}, m={m_used}...') + command_lpca = ['Rscript', '/app/run_lpca.R', mapped_matrix, mapped_scores, str(k), str(m_used)] + run_container_and_log('LPCA', LPCA_CONTAINER_SUFFIX, command_lpca, volumes, + LPCA_WORK_DIR, None, container_settings) + + print(f'LPCA: Done! Scores saved to {output_scores}') diff --git a/spras/config/config.py b/spras/config/config.py index ebf10faad..e99d965c9 100644 --- a/spras/config/config.py +++ b/spras/config/config.py @@ -86,6 +86,8 @@ def __init__(self, raw_config: dict[str, Any]): self.evaluation_params = self.analysis_params.evaluation # A dict with the ML settings self.ml_params = self.analysis_params.ml + # A dict with the LPCA settings + self.lpca_params = self.analysis_params.lpca # A Boolean specifying whether to run ML analysis for individual algorithms self.analysis_include_ml_aggregate_algo = None # A dict with the PCA settings @@ -96,6 +98,8 @@ def __init__(self, raw_config: dict[str, Any]): self.analysis_include_summary = None # A Boolean specifying whether to run the Cytoscape analysis self.analysis_include_cytoscape = None + # A Boolean specifying whether to run the LPCA analysis + self.analysis_include_lpca = None # A Boolean specifying whether to run the ML analysis self.analysis_include_ml = None # A Boolean specifying whether to run the Evaluation analysis @@ -254,6 +258,7 @@ def process_analysis(self, raw_config: RawConfig): self.analysis_include_summary = raw_config.analysis.summary.include self.analysis_include_cytoscape = raw_config.analysis.cytoscape.include self.analysis_include_ml = raw_config.analysis.ml.include + self.analysis_include_lpca = raw_config.analysis.lpca.include self.analysis_include_evaluation = raw_config.analysis.evaluation.include # Only run ML aggregate per algorithm if analysis include ML is set to True diff --git a/spras/config/schema.py b/spras/config/schema.py index 1a965c75c..2a85fe993 100644 --- a/spras/config/schema.py +++ b/spras/config/schema.py @@ -67,10 +67,20 @@ class EvaluationAnalysis(BaseModel): model_config = ConfigDict(extra='forbid') +class LpcaAnalysis(BaseModel): + include: bool + k: int = 2 + m: float = 6 + cv: bool = False + transpose: bool = False + + model_config = ConfigDict(extra='forbid') + class Analysis(BaseModel): summary: SummaryAnalysis = SummaryAnalysis(include=False) cytoscape: CytoscapeAnalysis = CytoscapeAnalysis(include=False) ml: MlAnalysis = MlAnalysis(include=False) + lpca: LpcaAnalysis = LpcaAnalysis(include=False) evaluation: EvaluationAnalysis = EvaluationAnalysis(include=False) model_config = ConfigDict(extra='forbid') From bb338a81efa621a66ff22e06fad7009d01f0e078 Mon Sep 17 00:00:00 2001 From: jeebjean Date: Fri, 10 Jul 2026 11:23:10 -0500 Subject: [PATCH 02/29] Add LPCA Docker wrapper Adds docker-wrappers/lpca with a pinned rocker/r-base Dockerfile that installs logisticPCA and its ggplot2 dependencies, the run_lpca.R and run_cv.R scripts under /app, and a README documenting the config options, script contracts, and how to build and publish reedcompbio/lpca:v1. --- docker-wrappers/lpca/Dockerfile | 31 ++++++++++++++++ docker-wrappers/lpca/README.md | 65 +++++++++++++++++++++++++++++++++ docker-wrappers/lpca/run_cv.R | 36 ++++++++++++++++++ docker-wrappers/lpca/run_lpca.R | 30 +++++++++++++++ 4 files changed, 162 insertions(+) create mode 100644 docker-wrappers/lpca/Dockerfile create mode 100644 docker-wrappers/lpca/README.md create mode 100644 docker-wrappers/lpca/run_cv.R create mode 100644 docker-wrappers/lpca/run_lpca.R diff --git a/docker-wrappers/lpca/Dockerfile b/docker-wrappers/lpca/Dockerfile new file mode 100644 index 000000000..69691fabd --- /dev/null +++ b/docker-wrappers/lpca/Dockerfile @@ -0,0 +1,31 @@ +# Logistic PCA (logisticPCA) wrapper for SPRAS. +# Invoked by spras/analysis/lpca.py, which supplies the full command +# (Rscript /app/run_lpca.R ... or /app/run_cv.R ...), so no ENTRYPOINT is set. + +# Pinned R version for reproducibility. Bump deliberately, not to :latest. +FROM rocker/r-base:4.4.2 + +LABEL org.opencontainers.image.source="https://github.com/Reed-CompBio/spras" +LABEL org.opencontainers.image.description="Logistic PCA (logisticPCA) wrapper for SPRAS" + +# System libraries needed to compile ggplot2 (a hard Import of logisticPCA) +# and its dependency stack from source on Debian. +RUN apt-get update && apt-get install -y --no-install-recommends \ + libcurl4-openssl-dev \ + libssl-dev \ + libxml2-dev \ + libfontconfig1-dev \ + libfreetype6-dev \ + libpng-dev \ + libtiff5-dev \ + libjpeg-dev \ + && rm -rf /var/lib/apt/lists/* + +# logisticPCA is still on CRAN (last published 2016) and pulls in ggplot2. +RUN Rscript -e "install.packages('logisticPCA', repos='https://cran.r-project.org')" \ + && Rscript -e "library(logisticPCA)" + +COPY run_lpca.R /app/run_lpca.R +COPY run_cv.R /app/run_cv.R + +WORKDIR /app \ No newline at end of file diff --git a/docker-wrappers/lpca/README.md b/docker-wrappers/lpca/README.md new file mode 100644 index 000000000..37dabc71e --- /dev/null +++ b/docker-wrappers/lpca/README.md @@ -0,0 +1,65 @@ +# LPCA (Logistic PCA) wrapper + +This wrapper runs [logisticPCA](https://github.com/andland/logisticPCA) +(Landgraf & Lee, 2020) as a SPRAS analysis step. It reduces the binary +edge-by-run matrix built from a set of pathway reconstruction outputs to a small +number of components and reports the proportion of deviance explained. + +The analysis is driven by the `analysis.lpca` config block and the +`lpca_analysis` Snakemake rule, and is implemented in `spras/analysis/lpca.py`. + +## Configuration + + analysis: + lpca: + include: false # run the LPCA analysis per algorithm + k: 2 # number of principal components + m: 6 # fixed logisticPCA tuning parameter, used when cv is false + cv: false # true: choose m by cross-validation; false: use the fixed m + transpose: false # false: edges x runs; true: runs x edges (mirrors the ml pca analysis) + +LPCA only runs for algorithms with multiple parameter combinations, so that the +binary matrix has more than one column. It also needs a reasonable number of +observations to be meaningful; very small inputs (such as the bundled example +datasets) produce degenerate results, which is why it is disabled by default. + +## Scripts + +The image contains two R scripts under `/app`: + +- `run_lpca.R `: runs logisticPCA with a fixed `m` and + writes the scores CSV plus a sibling `_deviance.txt`. +- `run_cv.R `: cross-validates `m` over 1..20 and writes a + CSV with a `best_m` column (plus a `_curve.csv` with the full CV curve). Only + used when `cv: true`. + +Both read a CSV whose first column holds row labels and whose remaining columns +are binary (0/1) features, and coerce missing values to 0. + +## Dependency note + +`logisticPCA` declares `ggplot2` as a hard `Imports` dependency, so building the +image compiles the ggplot2 stack. The Dockerfile installs the required Debian +system libraries for that. To build from a source CRAN mirror the `repos` +argument already points at `https://cran.r-project.org`. + +## Building and publishing the image + +For the SPRAS default registry to resolve the image, it must be published as +`docker.io/reedcompbio/lpca:v1`, which requires access to the `reedcompbio` +Docker Hub organization: + + docker build -t reedcompbio/lpca:v1 docker-wrappers/lpca/ + docker push reedcompbio/lpca:v1 + +## How SPRAS resolves the image + +SPRAS builds the image reference as `//`. +The default is `docker.io/reedcompbio`, combined with the tag `lpca:v1`, so the +resolved image is `docker.io/reedcompbio/lpca:v1`. The owner is never hardcoded; +it comes from config. + +To test against a local image before it is hosted under `reedcompbio`, tag the +built image with that name locally (Docker uses a local image without pulling): + + docker tag reedcompbio/lpca:v1 \ No newline at end of file diff --git a/docker-wrappers/lpca/run_cv.R b/docker-wrappers/lpca/run_cv.R new file mode 100644 index 000000000..7301d3419 --- /dev/null +++ b/docker-wrappers/lpca/run_cv.R @@ -0,0 +1,36 @@ +# run_cv.R +# Finds the optimal m for a given k using cross-validation + +args = commandArgs(trailingOnly = TRUE) +input_file = args[1] +output_file = args[2] +k = as.integer(args[3]) + +set.seed(42) + +# Load data +library(logisticPCA) +data = read.csv(input_file, row.names = NULL) +data = data[, -1] +data_matrix = as.matrix(data) +data_matrix[is.na(data_matrix)] = 0 + +# Cross-validation over m, fixed k +cv_result = cv.lpca(data_matrix, ks = k, ms = 1:20) +best_m = which.min(cv_result) + +cat("Cross-validation done for k =", k, "\n") +cat("Best m:", best_m, "\n") + +# Save best m +write.csv(data.frame(k = k, best_m = best_m), output_file, row.names = FALSE) + +# Save full CV curve (all m values and their reconstruction error) +cv_curve_file = sub("\\.csv$", "_curve.csv", output_file) +cv_df = data.frame( + m = 1:20, + reconstruction_error = as.numeric(cv_result), + is_best = (1:20) == best_m +) +write.csv(cv_df, cv_curve_file, row.names = FALSE) +cat("CV curve saved to", cv_curve_file, "\n") \ No newline at end of file diff --git a/docker-wrappers/lpca/run_lpca.R b/docker-wrappers/lpca/run_lpca.R new file mode 100644 index 000000000..a0cf095a1 --- /dev/null +++ b/docker-wrappers/lpca/run_lpca.R @@ -0,0 +1,30 @@ +# Read command line arguments +args = commandArgs(trailingOnly = TRUE) +input_file = args[1] +output_file = args[2] +k = as.integer(args[3]) +m = as.numeric(args[4]) + +# Load data +library(logisticPCA) +data = read.csv(input_file, row.names = NULL) +party = data[, 1] +data = data[, -1] +data_matrix = as.matrix(data) +data_matrix[is.na(data_matrix)] = 0 + +# Run LPCA +model = logisticPCA(data_matrix, k = k, m = m) + +# Save scores +scores = model$PCs +rownames(scores) = party +write.csv(scores, output_file, row.names = TRUE) + +# Save deviance explained +deviance_file = sub("\\.csv$", "_deviance.txt", output_file) +writeLines(as.character(model$prop_deviance_expl), deviance_file) + +cat("LPCA done! Scores saved to", output_file, "\n") +cat("Score dimensions:", nrow(scores), "x", ncol(scores), "\n") +cat("Proportion of deviance explained:", model$prop_deviance_expl, "\n") \ No newline at end of file From efd10058eeb723b6ec2be8bfb45b3f98b5e41ce5 Mon Sep 17 00:00:00 2001 From: jeebjean Date: Tue, 21 Jul 2026 12:40:30 -0500 Subject: [PATCH 03/29] Add unit tests for LPCA analysis --- test/analysis/test_lpca.py | 83 ++++++++++++++++++++++++++++++++++++++ 1 file changed, 83 insertions(+) create mode 100644 test/analysis/test_lpca.py diff --git a/test/analysis/test_lpca.py b/test/analysis/test_lpca.py new file mode 100644 index 000000000..9b861333f --- /dev/null +++ b/test/analysis/test_lpca.py @@ -0,0 +1,83 @@ +from pathlib import Path + +import pandas as pd + +import spras.config.config as config +from spras.analysis.lpca import run_lpca + +config.init_from_file("config/config.yaml") + +TEST_DIR = Path('test/analysis/') +OUT_DIR = TEST_DIR / 'output' + +# Reuse pathway files from the evaluate test directory +INPUT_FILES = [ + 'test/evaluate/input/data-test-params-123/pathway.txt', + 'test/evaluate/input/data-test-params-456/pathway.txt', + 'test/evaluate/input/data-test-params-789/pathway.txt', +] + +class TestLpca: + """ + Run Logistic PCA (LPCA) analysis tests + """ + @classmethod + def setup_class(cls): + OUT_DIR.mkdir(parents=True, exist_ok=True) + + def test_lpca_output_exists(self): + """Test that LPCA produces an output scores file""" + out_path = OUT_DIR / 'lpca-scores.csv' + out_path.unlink(missing_ok=True) + + run_lpca( + file_paths=INPUT_FILES, + output_scores=str(out_path), + k=2, + m=4, + cv=False, + transpose=False + ) + + assert out_path.exists(), "LPCA scores file was not created" + + def test_lpca_output_shape(self): + """Test that LPCA scores have the correct shape (edges x k)""" + out_path = OUT_DIR / 'lpca-scores-shape.csv' + out_path.unlink(missing_ok=True) + + run_lpca( + file_paths=INPUT_FILES, + output_scores=str(out_path), + k=2, + m=4, + cv=False, + transpose=False + ) + + scores = pd.read_csv(out_path, index_col=0) + # k=2 so should have 2 columns + assert scores.shape[1] == 2, f"Expected 2 PC columns, got {scores.shape[1]}" + # Should have at least 1 row (edge) + assert scores.shape[0] > 0, "Scores file is empty" + + def test_lpca_transposed_shape(self): + """Test that transposed LPCA scores have the correct shape (runs x k)""" + out_path = OUT_DIR / 'lpca-scores-transposed.csv' + out_path.unlink(missing_ok=True) + + run_lpca( + file_paths=INPUT_FILES, + output_scores=str(out_path), + k=2, + m=4, + cv=False, + transpose=True + ) + + scores = pd.read_csv(out_path, index_col=0) + # k=2 so should have 2 columns + assert scores.shape[1] == 2, f"Expected 2 PC columns, got {scores.shape[1]}" + # Should have 3 rows (one per pathway run) + assert scores.shape[0] == len(INPUT_FILES), \ + f"Expected {len(INPUT_FILES)} rows, got {scores.shape[0]}" From 8f891195783869279441348dbfa76dfef41aa64c Mon Sep 17 00:00:00 2001 From: jeebjean Date: Tue, 21 Jul 2026 14:33:48 -0500 Subject: [PATCH 04/29] Add LPCA visualization: generate lpca.png and lpca-coordinates.txt --- Snakefile | 11 ++++++++++- spras/analysis/lpca.py | 44 ++++++++++++++++++++++++++++++++++++++++++ 2 files changed, 54 insertions(+), 1 deletion(-) diff --git a/Snakefile b/Snakefile index 82f208e4f..38614bcb5 100644 --- a/Snakefile +++ b/Snakefile @@ -94,6 +94,8 @@ def make_final_input(wildcards): if _config.config.analysis_include_lpca: final_input.extend(expand('{out_dir}{sep}{dataset}-ml{sep}{algorithm}-lpca-scores.csv',out_dir=out_dir, sep=SEP, dataset=dataset_labels, algorithm=algorithms_mult_param_combos)) + final_input.extend(expand('{out_dir}{sep}{dataset}-ml{sep}{algorithm}-lpca.png',out_dir=out_dir, sep=SEP,dataset=dataset_labels,algorithm=algorithms_mult_param_combos)) + final_input.extend(expand('{out_dir}{sep}{dataset}-ml{sep}{algorithm}-lpca-coordinates.txt',out_dir=out_dir, sep=SEP, dataset=dataset_labels,algorithm=algorithms_mult_param_combos)) if _config.config.analysis_include_ml_aggregate_algo: final_input.extend(expand('{out_dir}{sep}{dataset}-ml{sep}{algorithm}-pca.png',out_dir=out_dir,sep=SEP,dataset=dataset_labels,algorithm=algorithms_mult_param_combos)) @@ -411,7 +413,9 @@ rule lpca_analysis: input: pathways = collect_pathways_per_algo output: - lpca_scores = SEP.join([out_dir, '{dataset}-ml', '{algorithm}-lpca-scores.csv']) + lpca_scores = SEP.join([out_dir, '{dataset}-ml', '{algorithm}-lpca-scores.csv']), + lpca_png = SEP.join([out_dir, '{dataset}-ml', '{algorithm}-lpca.png']), + lpca_coord = SEP.join([out_dir, '{dataset}-ml', '{algorithm}-lpca-coordinates.txt']) run: from spras.analysis import lpca lpca.run_lpca( @@ -423,6 +427,11 @@ rule lpca_analysis: transpose=_config.config.lpca_params.transpose, container_settings=container_settings ) + lpca.plot_lpca( + output.lpca_scores, + output.lpca_png, + output.lpca_coord + ) # Ensemble the output pathways for each dataset per algorithm rule ensemble_per_algo: diff --git a/spras/analysis/lpca.py b/spras/analysis/lpca.py index 295fad1ae..ddab3ca4e 100644 --- a/spras/analysis/lpca.py +++ b/spras/analysis/lpca.py @@ -6,7 +6,9 @@ from pathlib import Path from typing import Iterable, Union +import matplotlib.pyplot as plt import pandas as pd +import seaborn as sns from spras.analysis.ml import summarize_networks from spras.config.container_schema import ProcessedContainerSettings @@ -92,3 +94,45 @@ def run_lpca( LPCA_WORK_DIR, None, container_settings) print(f'LPCA: Done! Scores saved to {output_scores}') + +def plot_lpca(scores_file: str, output_png: str, output_coord: str, labels: bool = True) -> None: + """ + Creates a scatterplot of the first two LPCA principal components. + @param scores_file: path to the LPCA scores CSV file + @param output_png: path to save the scatterplot PNG + @param output_coord: path to save the PC coordinates + @param labels: if True, adds algorithm labels to the plot + """ + scores = pd.read_csv(scores_file, index_col=0) + + if scores.empty: + print('LPCA: Scores file is empty, skipping plot.') + return + + # Extract algorithm names from the index + column_names = [idx.split('-')[-3] if '-' in idx else idx for idx in scores.index] + + X = scores.values + fig, ax = plt.subplots(figsize=(10, 8)) + + sns.scatterplot(x=X[:, 0], y=X[:, 1], hue=column_names, s=70, ax=ax) + + if labels: + for i, label in enumerate(scores.index): + ax.annotate(label, (X[i, 0], X[i, 1]), fontsize=6, alpha=0.7) + + ax.set_xlabel('PC1') + ax.set_ylabel('PC2') + ax.set_title('Logistic PCA') + + plt.tight_layout() + + # Save PNG + Path(output_png).parent.mkdir(parents=True, exist_ok=True) + plt.savefig(output_png, dpi=200) + plt.close() + + # Save coordinates + coord_df = pd.DataFrame(X, columns=['PC1', 'PC2'], index=scores.index) + coord_df.to_csv(output_coord) + print(f'LPCA: Plot saved to {output_png}') From 3d9a38fff9dc1d2c26cf753a397bd01aade95302 Mon Sep 17 00:00:00 2001 From: jeebjean Date: Tue, 21 Jul 2026 15:11:11 -0500 Subject: [PATCH 05/29] Increase Docker timeout to 600s for large LPCA matrices --- config/config.yaml | 26 +++++++++++++------------- spras/containers.py | 2 +- 2 files changed, 14 insertions(+), 14 deletions(-) diff --git a/config/config.yaml b/config/config.yaml index 55c8d3afa..395b97c0d 100644 --- a/config/config.yaml +++ b/config/config.yaml @@ -103,13 +103,13 @@ algorithms: include: true runs: run1: - b: [5, 6] - w: np.linspace(0,5,2) + b: [2, 4, 6, 8, 10] + w: np.linspace(0,5,5) d: 10 dummy_mode: "file" # Or "terminals", "all", "others" - name: "omicsintegrator2" - include: true + include: false runs: run1: b: 4 @@ -119,7 +119,7 @@ algorithms: g: 3 - name: "meo" - include: true + include: false runs: run1: max_path_length: 3 @@ -127,46 +127,46 @@ algorithms: rand_restarts: 10 - name: "mincostflow" - include: true + include: false runs: run1: flow: 1 capacity: 1 - name: "allpairs" - include: true + include: false - name: "domino" - include: true + include: false runs: run1: slice_threshold: 0.3 module_threshold: 0.05 - name: "strwr" - include: true + include: false runs: run1: alpha: [0.85] threshold: [100, 200] - name: "rwr" - include: true + include: false runs: run1: alpha: [0.85] threshold: [100, 200] - name: "bowtiebuilder" - include: true + include: false - name: "responsenet" - include: true + include: false runs: run1: gamma: [10] - name: "diamond" - include: true + include: false runs: run1: n: 1 @@ -277,7 +277,7 @@ analysis: lpca: # run the LPCA analysis per algorithm # LPCA needs enough observations to be meaningful - include: false + include: true # number of principal components to compute k: 2 # fixed value of the logisticPCA tuning parameter m, used when cv is false diff --git a/spras/containers.py b/spras/containers.py index c30697f3f..15573dc8d 100644 --- a/spras/containers.py +++ b/spras/containers.py @@ -359,7 +359,7 @@ def run_container_docker(container: str, command: List[str], volumes: List[Tuple # Initialize a Docker client using environment variables try: - client = docker.from_env() + client = docker.from_env(timeout=600) except Exception as err: err.add_note("An error occurred when fetching the docker daemon: is docker installed and is dockerd running?") raise err From e4b46797afbb3dc5a4ec3278fbda7c97615852cb Mon Sep 17 00:00:00 2001 From: jeebjean Date: Tue, 21 Jul 2026 15:35:16 -0500 Subject: [PATCH 06/29] Restore config.yaml to main defaults, keep only lpca block --- config/config.yaml | 54 +++++++++++----------------------------------- 1 file changed, 13 insertions(+), 41 deletions(-) diff --git a/config/config.yaml b/config/config.yaml index 395b97c0d..6cfdb4f40 100644 --- a/config/config.yaml +++ b/config/config.yaml @@ -50,30 +50,6 @@ containers: # requirements = versionGE(split(Target.CondorVersion)[1], "24.8.0") && (isenforcingdiskusage =!= true) enable_profiling: false - # Override the default container image for specific algorithms. - # Keys are algorithm names (as they appear in the algorithms list below). - # Values are interpreted based on the container framework: - # - # Image reference (e.g., "pathlinker:v3"): - # Prepends the registry prefix. Works with both Docker and Apptainer. - # - # Full image reference with registry (e.g., "ghcr.io/myorg/pathlinker:v3"): - # Used as-is (prefix NOT prepended). Works with both Docker and Apptainer. - # - # Local .sif file path (e.g., "images/pathlinker_v2.sif"): - # Apptainer/Singularity only. Skips pulling from registry and uses the - # pre-built .sif directly. When running via HTCondor with shared-fs-usage: none (set - # via the spras_profile config when running SPRAS against HTCondor), .sif paths listed - # here are automatically included in htcondor_transfer_input_files. - # Ignored with a warning if the framework is Docker. - # - # Example (one of each type): - # images: - # omicsintegrator1: "images/omics-integrator-1_v2.sif" # local .sif (Apptainer only) - # pathlinker: "pathlinker:v1234" # image name only (base_url/owner prepended) - # omicsintegrator2: "some-other-owner/oi2:latest" # owner/image (base_url prepended) - # mincostflow: "ghcr.io/reed-compbio/mincostflow:v2" # full registry reference (used as-is) - # This list of algorithms should be generated by a script which checks the filesystem for installs. # It shouldn't be changed by mere mortals. (alternatively, we could add a path to executable for each algorithm # in the list to reduce the number of assumptions of the program at the cost of making the config a little more involved) @@ -103,13 +79,13 @@ algorithms: include: true runs: run1: - b: [2, 4, 6, 8, 10] - w: np.linspace(0,5,5) + b: [5, 6] + w: np.linspace(0,5,2) d: 10 dummy_mode: "file" # Or "terminals", "all", "others" - name: "omicsintegrator2" - include: false + include: true runs: run1: b: 4 @@ -119,7 +95,7 @@ algorithms: g: 3 - name: "meo" - include: false + include: true runs: run1: max_path_length: 3 @@ -127,46 +103,46 @@ algorithms: rand_restarts: 10 - name: "mincostflow" - include: false + include: true runs: run1: flow: 1 capacity: 1 - name: "allpairs" - include: false + include: true - name: "domino" - include: false + include: true runs: run1: slice_threshold: 0.3 module_threshold: 0.05 - name: "strwr" - include: false + include: true runs: run1: alpha: [0.85] threshold: [100, 200] - name: "rwr" - include: false + include: true runs: run1: alpha: [0.85] threshold: [100, 200] - name: "bowtiebuilder" - include: false + include: true - name: "responsenet" - include: false + include: true runs: run1: gamma: [10] - name: "diamond" - include: false + include: true runs: run1: n: 1 @@ -272,12 +248,8 @@ analysis: # adds evaluation per algorithm per dataset-goldstandard pair # evaluation per algorithm will not run unless ml include and ml aggregate_per_algorithm are set to true aggregate_per_algorithm: true - # Logistic PCA (LPCA) of the binary edge-by-run matrix, one embedding per algorithm - # only runs for algorithms with multiple parameter combinations chosen lpca: - # run the LPCA analysis per algorithm - # LPCA needs enough observations to be meaningful - include: true + include: false # number of principal components to compute k: 2 # fixed value of the logisticPCA tuning parameter m, used when cv is false From 41b64b7187c4a68f0ab0f51533d3a4ff0e9a7c26 Mon Sep 17 00:00:00 2001 From: jeebjean Date: Tue, 21 Jul 2026 15:53:01 -0500 Subject: [PATCH 07/29] Skip LPCA tests if Docker image not available --- test/analysis/test_lpca.py | 26 +++++++++++++++++++++----- 1 file changed, 21 insertions(+), 5 deletions(-) diff --git a/test/analysis/test_lpca.py b/test/analysis/test_lpca.py index 9b861333f..cf10c87d1 100644 --- a/test/analysis/test_lpca.py +++ b/test/analysis/test_lpca.py @@ -1,6 +1,8 @@ from pathlib import Path +import docker import pandas as pd +import pytest import spras.config.config as config from spras.analysis.lpca import run_lpca @@ -10,13 +12,28 @@ TEST_DIR = Path('test/analysis/') OUT_DIR = TEST_DIR / 'output' -# Reuse pathway files from the evaluate test directory INPUT_FILES = [ 'test/evaluate/input/data-test-params-123/pathway.txt', 'test/evaluate/input/data-test-params-456/pathway.txt', 'test/evaluate/input/data-test-params-789/pathway.txt', ] +def lpca_image_available(): + """Check if the LPCA Docker image is available locally or on Docker Hub""" + try: + client = docker.from_env() + client.images.get('reedcompbio/lpca:v1') + return True + except docker.errors.ImageNotFound: + return False + except Exception: + return False + +skip_if_no_lpca_image = pytest.mark.skipif( + not lpca_image_available(), + reason='reedcompbio/lpca:v1 Docker image not available' +) + class TestLpca: """ Run Logistic PCA (LPCA) analysis tests @@ -25,6 +42,7 @@ class TestLpca: def setup_class(cls): OUT_DIR.mkdir(parents=True, exist_ok=True) + @skip_if_no_lpca_image def test_lpca_output_exists(self): """Test that LPCA produces an output scores file""" out_path = OUT_DIR / 'lpca-scores.csv' @@ -41,6 +59,7 @@ def test_lpca_output_exists(self): assert out_path.exists(), "LPCA scores file was not created" + @skip_if_no_lpca_image def test_lpca_output_shape(self): """Test that LPCA scores have the correct shape (edges x k)""" out_path = OUT_DIR / 'lpca-scores-shape.csv' @@ -56,11 +75,10 @@ def test_lpca_output_shape(self): ) scores = pd.read_csv(out_path, index_col=0) - # k=2 so should have 2 columns assert scores.shape[1] == 2, f"Expected 2 PC columns, got {scores.shape[1]}" - # Should have at least 1 row (edge) assert scores.shape[0] > 0, "Scores file is empty" + @skip_if_no_lpca_image def test_lpca_transposed_shape(self): """Test that transposed LPCA scores have the correct shape (runs x k)""" out_path = OUT_DIR / 'lpca-scores-transposed.csv' @@ -76,8 +94,6 @@ def test_lpca_transposed_shape(self): ) scores = pd.read_csv(out_path, index_col=0) - # k=2 so should have 2 columns assert scores.shape[1] == 2, f"Expected 2 PC columns, got {scores.shape[1]}" - # Should have 3 rows (one per pathway run) assert scores.shape[0] == len(INPUT_FILES), \ f"Expected {len(INPUT_FILES)} rows, got {scores.shape[0]}" From c49362f78a85c4569eaab500cfe14da81f7de546 Mon Sep 17 00:00:00 2001 From: jeebjean Date: Thu, 23 Jul 2026 13:55:31 -0500 Subject: [PATCH 08/29] Add rARPACK for memory-efficient LPCA with partial_decomp=TRUE --- docker-wrappers/lpca/Dockerfile | 4 ++-- docker-wrappers/lpca/run_lpca.R | 2 +- 2 files changed, 3 insertions(+), 3 deletions(-) diff --git a/docker-wrappers/lpca/Dockerfile b/docker-wrappers/lpca/Dockerfile index 69691fabd..66b27893a 100644 --- a/docker-wrappers/lpca/Dockerfile +++ b/docker-wrappers/lpca/Dockerfile @@ -22,8 +22,8 @@ RUN apt-get update && apt-get install -y --no-install-recommends \ && rm -rf /var/lib/apt/lists/* # logisticPCA is still on CRAN (last published 2016) and pulls in ggplot2. -RUN Rscript -e "install.packages('logisticPCA', repos='https://cran.r-project.org')" \ - && Rscript -e "library(logisticPCA)" +RUN Rscript -e "install.packages(c('logisticPCA', 'rARPACK'), repos='https://cran.r-project.org')" \ + && Rscript -e "library(logisticPCA); library(rARPACK)" COPY run_lpca.R /app/run_lpca.R COPY run_cv.R /app/run_cv.R diff --git a/docker-wrappers/lpca/run_lpca.R b/docker-wrappers/lpca/run_lpca.R index a0cf095a1..5947d8e4c 100644 --- a/docker-wrappers/lpca/run_lpca.R +++ b/docker-wrappers/lpca/run_lpca.R @@ -14,7 +14,7 @@ data_matrix = as.matrix(data) data_matrix[is.na(data_matrix)] = 0 # Run LPCA -model = logisticPCA(data_matrix, k = k, m = m) +model = logisticPCA(data_matrix, k = k, m = m, partial_decomp = TRUE) # Save scores scores = model$PCs From 7536d8ccd97c75d86cd878397ed6f7a1a383da8d Mon Sep 17 00:00:00 2001 From: jeebjean Date: Mon, 27 Jul 2026 09:58:25 -0500 Subject: [PATCH 09/29] Address PR review: palette, CV check, KDE note, config docs, rename to LPCA, Snakefile cleanup --- Snakefile | 1 - config/config.yaml | 2 ++ docker-wrappers/{lpca => LPCA}/Dockerfile | 0 docker-wrappers/{lpca => LPCA}/README.md | 2 +- docker-wrappers/{lpca => LPCA}/run_cv.R | 0 docker-wrappers/{lpca => LPCA}/run_lpca.R | 0 spras/analysis/lpca.py | 14 +++++++++++--- spras/analysis/ml.py | 3 ++- 8 files changed, 16 insertions(+), 6 deletions(-) rename docker-wrappers/{lpca => LPCA}/Dockerfile (100%) rename docker-wrappers/{lpca => LPCA}/README.md (96%) rename docker-wrappers/{lpca => LPCA}/run_cv.R (100%) rename docker-wrappers/{lpca => LPCA}/run_lpca.R (100%) diff --git a/Snakefile b/Snakefile index 38614bcb5..a842407c0 100644 --- a/Snakefile +++ b/Snakefile @@ -408,7 +408,6 @@ rule ml_analysis_aggregate_algo: ml.hac_vertical(summary_df, output.hac_image_vertical, output.hac_clusters_vertical, **hac_params) ml.hac_horizontal(summary_df, output.hac_image_horizontal, output.hac_clusters_horizontal, **hac_params) ml.pca(summary_df, output.pca_image, output.pca_variance, output.pca_coordinates, **pca_params) - rule lpca_analysis: input: pathways = collect_pathways_per_algo diff --git a/config/config.yaml b/config/config.yaml index 6cfdb4f40..4e4802661 100644 --- a/config/config.yaml +++ b/config/config.yaml @@ -249,6 +249,8 @@ analysis: # evaluation per algorithm will not run unless ml include and ml aggregate_per_algorithm are set to true aggregate_per_algorithm: true lpca: + # if true, runs LPCA in addition to the existing PCA (both analyses will run in parallel). + # Running both helps compare the two methods on the same data. include: false # number of principal components to compute k: 2 diff --git a/docker-wrappers/lpca/Dockerfile b/docker-wrappers/LPCA/Dockerfile similarity index 100% rename from docker-wrappers/lpca/Dockerfile rename to docker-wrappers/LPCA/Dockerfile diff --git a/docker-wrappers/lpca/README.md b/docker-wrappers/LPCA/README.md similarity index 96% rename from docker-wrappers/lpca/README.md rename to docker-wrappers/LPCA/README.md index 37dabc71e..2db543e1f 100644 --- a/docker-wrappers/lpca/README.md +++ b/docker-wrappers/LPCA/README.md @@ -1,7 +1,7 @@ # LPCA (Logistic PCA) wrapper This wrapper runs [logisticPCA](https://github.com/andland/logisticPCA) -(Landgraf & Lee, 2020) as a SPRAS analysis step. It reduces the binary +([Landgraf & Lee, 2020](https://doi.org/10.1016/j.jmva.2020.104668)) as a SPRAS analysis step. It reduces the binary edge-by-run matrix built from a set of pathway reconstruction outputs to a small number of components and reports the proportion of deviance explained. diff --git a/docker-wrappers/lpca/run_cv.R b/docker-wrappers/LPCA/run_cv.R similarity index 100% rename from docker-wrappers/lpca/run_cv.R rename to docker-wrappers/LPCA/run_cv.R diff --git a/docker-wrappers/lpca/run_lpca.R b/docker-wrappers/LPCA/run_lpca.R similarity index 100% rename from docker-wrappers/lpca/run_lpca.R rename to docker-wrappers/LPCA/run_lpca.R diff --git a/spras/analysis/lpca.py b/spras/analysis/lpca.py index ddab3ca4e..94c3a31c4 100644 --- a/spras/analysis/lpca.py +++ b/spras/analysis/lpca.py @@ -10,7 +10,7 @@ import pandas as pd import seaborn as sns -from spras.analysis.ml import summarize_networks +from spras.analysis.ml import create_palette, summarize_networks from spras.config.container_schema import ProcessedContainerSettings from spras.containers import prepare_volume, run_container_and_log @@ -42,6 +42,9 @@ def run_lpca( @param transpose: if True, run LPCA on the transposed (runs x edges) matrix instead of the default (edges x runs) @param container_settings: configure the container runtime (Docker or Singularity) + + Note: KDE-based parameter selection (used by PCA) always uses PCA scores, even when LPCA + is also enabled. KDE integration with LPCA may be added in a future update. """ if not container_settings: container_settings = ProcessedContainerSettings() @@ -61,7 +64,6 @@ def run_lpca( algo_name = Path(output_scores).name.replace('-lpca-scores.csv', '') matrix_path = str(output_dir / f'{algo_name}-lpca_binary_matrix.csv') matrix.to_csv(matrix_path) - print(f'LPCA: Binary matrix saved to {matrix_path}') # Step 3: mount the matrix and the scores output volumes = [] @@ -81,6 +83,11 @@ def run_lpca( run_container_and_log('LPCA-CV', LPCA_CONTAINER_SUFFIX, command_cv, volumes, LPCA_WORK_DIR, None, container_settings) + if not Path(cv_output_path).exists(): + raise FileNotFoundError( + f'LPCA: Cross-validation output not found at {cv_output_path}. ' + 'Check the LPCA Docker container logs for errors.' + ) m_used = pd.read_csv(cv_output_path)['best_m'][0] print(f'LPCA: Best m found by CV: {m_used}') else: @@ -115,7 +122,8 @@ def plot_lpca(scores_file: str, output_png: str, output_coord: str, labels: bool X = scores.values fig, ax = plt.subplots(figsize=(10, 8)) - sns.scatterplot(x=X[:, 0], y=X[:, 1], hue=column_names, s=70, ax=ax) + label_color_map = create_palette(column_names) + sns.scatterplot(x=X[:, 0], y=X[:, 1], hue=column_names, palette=label_color_map, s=70, ax=ax) if labels: for i, label in enumerate(scores.index): diff --git a/spras/analysis/ml.py b/spras/analysis/ml.py index 55abae5a8..459852770 100644 --- a/spras/analysis/ml.py +++ b/spras/analysis/ml.py @@ -156,7 +156,8 @@ def pca(dataframe: pd.DataFrame, output_png: str | PathLike, output_var: str | P # center binary data by subtracting the column-wise mean # allows PCA to focus on edge inclusion patterns across runs rather than raw output volume. - # TODO: replace PCA https://github.com/Reed-CompBio/spras/issues/271 + # TODO: consider replacing PCA with LPCA for binary data https://github.com/Reed-CompBio/spras/issues/271 + # LPCA is now available as an alternative analysis (analysis.lpca in config) scaler = StandardScaler(with_std=False) scaler.fit(X) # compute mean inclusion rate per edge X_scaled = scaler.transform(X) From 63d941d3ebda47ca64e0e5d2ef305fba45f8f2f8 Mon Sep 17 00:00:00 2001 From: jeebjean Date: Mon, 27 Jul 2026 10:37:04 -0500 Subject: [PATCH 10/29] Remove transpose option, fix m doc, update README, rename party variable --- Snakefile | 1 - config/config.yaml | 8 +++----- docker-wrappers/LPCA/README.md | 16 ++++------------ docker-wrappers/LPCA/run_lpca.R | 4 ++-- spras/analysis/lpca.py | 9 ++------- spras/config/schema.py | 1 - 6 files changed, 11 insertions(+), 28 deletions(-) diff --git a/Snakefile b/Snakefile index a842407c0..6476caee7 100644 --- a/Snakefile +++ b/Snakefile @@ -423,7 +423,6 @@ rule lpca_analysis: k=_config.config.lpca_params.k, m=_config.config.lpca_params.m, cv=_config.config.lpca_params.cv, - transpose=_config.config.lpca_params.transpose, container_settings=container_settings ) lpca.plot_lpca( diff --git a/config/config.yaml b/config/config.yaml index 4e4802661..f7d98d648 100644 --- a/config/config.yaml +++ b/config/config.yaml @@ -254,11 +254,9 @@ analysis: include: false # number of principal components to compute k: 2 - # fixed value of the logisticPCA tuning parameter m, used when cv is false + # fixed value of the logisticPCA tuning parameter m, used when cv is false. + # default of 6 was selected based on cross-validation experiments across multiple + # SPRAS algorithms (see https://github.com/Jeebjean/lpca-spras for details) m: 6 # if true, choose m by cross-validation; if false, use the fixed m above cv: false - # matrix orientation given to LPCA: - # false = edges x runs (observations are the edges) - # true = runs x edges (observations are the runs, mirroring the classic ml pca analysis) - transpose: false diff --git a/docker-wrappers/LPCA/README.md b/docker-wrappers/LPCA/README.md index 2db543e1f..6a3a17e85 100644 --- a/docker-wrappers/LPCA/README.md +++ b/docker-wrappers/LPCA/README.md @@ -1,5 +1,9 @@ # LPCA (Logistic PCA) wrapper +Docker image: https://hub.docker.com/r/reedcompbio/lpca + +This wrapper runs [logisticPCA](https://github.com/andland/logisticPCA) + This wrapper runs [logisticPCA](https://github.com/andland/logisticPCA) ([Landgraf & Lee, 2020](https://doi.org/10.1016/j.jmva.2020.104668)) as a SPRAS analysis step. It reduces the binary edge-by-run matrix built from a set of pathway reconstruction outputs to a small @@ -16,7 +20,6 @@ The analysis is driven by the `analysis.lpca` config block and the k: 2 # number of principal components m: 6 # fixed logisticPCA tuning parameter, used when cv is false cv: false # true: choose m by cross-validation; false: use the fixed m - transpose: false # false: edges x runs; true: runs x edges (mirrors the ml pca analysis) LPCA only runs for algorithms with multiple parameter combinations, so that the binary matrix has more than one column. It also needs a reasonable number of @@ -52,14 +55,3 @@ Docker Hub organization: docker build -t reedcompbio/lpca:v1 docker-wrappers/lpca/ docker push reedcompbio/lpca:v1 -## How SPRAS resolves the image - -SPRAS builds the image reference as `//`. -The default is `docker.io/reedcompbio`, combined with the tag `lpca:v1`, so the -resolved image is `docker.io/reedcompbio/lpca:v1`. The owner is never hardcoded; -it comes from config. - -To test against a local image before it is hosted under `reedcompbio`, tag the -built image with that name locally (Docker uses a local image without pulling): - - docker tag reedcompbio/lpca:v1 \ No newline at end of file diff --git a/docker-wrappers/LPCA/run_lpca.R b/docker-wrappers/LPCA/run_lpca.R index 5947d8e4c..1a3049e56 100644 --- a/docker-wrappers/LPCA/run_lpca.R +++ b/docker-wrappers/LPCA/run_lpca.R @@ -8,7 +8,7 @@ m = as.numeric(args[4]) # Load data library(logisticPCA) data = read.csv(input_file, row.names = NULL) -party = data[, 1] +row_labels = data[, 1] data = data[, -1] data_matrix = as.matrix(data) data_matrix[is.na(data_matrix)] = 0 @@ -18,7 +18,7 @@ model = logisticPCA(data_matrix, k = k, m = m, partial_decomp = TRUE) # Save scores scores = model$PCs -rownames(scores) = party +rownames(scores) = row_labels write.csv(scores, output_file, row.names = TRUE) # Save deviance explained diff --git a/spras/analysis/lpca.py b/spras/analysis/lpca.py index 94c3a31c4..4dce4c5eb 100644 --- a/spras/analysis/lpca.py +++ b/spras/analysis/lpca.py @@ -26,7 +26,6 @@ def run_lpca( k: int = 2, m: float = 6, cv: bool = False, - transpose: bool = False, container_settings=None ) -> None: """ @@ -39,8 +38,6 @@ def run_lpca( @param m: fixed logisticPCA tuning parameter, used when cv is False @param cv: if True, determine m by cross-validation; if False, use the fixed m directly (default False) - @param transpose: if True, run LPCA on the transposed (runs x edges) matrix - instead of the default (edges x runs) @param container_settings: configure the container runtime (Docker or Singularity) Note: KDE-based parameter selection (used by PCA) always uses PCA scores, even when LPCA @@ -52,10 +49,8 @@ def run_lpca( # Step 1: build the binary edge x algorithm matrix print('LPCA: Building binary edge x algorithm matrix...') matrix = summarize_networks(file_paths) - - # Optionally transpose so the runs become the observations rather than the edges - if transpose: - matrix = matrix.T + # Transpose so runs are the observations, mirroring the classic PCA analysis + matrix = matrix.T print(f'LPCA: Matrix shape: {matrix.shape}') # Step 2: write the matrix next to the outputs, namespaced by algorithm diff --git a/spras/config/schema.py b/spras/config/schema.py index 2a85fe993..f2e0ba353 100644 --- a/spras/config/schema.py +++ b/spras/config/schema.py @@ -72,7 +72,6 @@ class LpcaAnalysis(BaseModel): k: int = 2 m: float = 6 cv: bool = False - transpose: bool = False model_config = ConfigDict(extra='forbid') From a44bb50f345b9eb424144d40305de1f3ad7c0752 Mon Sep 17 00:00:00 2001 From: jeebjean Date: Mon, 27 Jul 2026 10:51:57 -0500 Subject: [PATCH 11/29] Restore deleted comments in config.yaml --- config/config.yaml | 24 ++++++++++++++++++++++++ 1 file changed, 24 insertions(+) diff --git a/config/config.yaml b/config/config.yaml index f7d98d648..9f6b00fad 100644 --- a/config/config.yaml +++ b/config/config.yaml @@ -50,6 +50,30 @@ containers: # requirements = versionGE(split(Target.CondorVersion)[1], "24.8.0") && (isenforcingdiskusage =!= true) enable_profiling: false + # Override the default container image for specific algorithms. + # Keys are algorithm names (as they appear in the algorithms list below). + # Values are interpreted based on the container framework: + # + # Image reference (e.g., "pathlinker:v3"): + # Prepends the registry prefix. Works with both Docker and Apptainer. + # + # Full image reference with registry (e.g., "ghcr.io/myorg/pathlinker:v3"): + # Used as-is (prefix NOT prepended). Works with both Docker and Apptainer. + # + # Local .sif file path (e.g., "images/pathlinker_v2.sif"): + # Apptainer/Singularity only. Skips pulling from registry and uses the + # pre-built .sif directly. When running via HTCondor with shared-fs-usage: none (set + # via the spras_profile config when running SPRAS against HTCondor), .sif paths listed + # here are automatically included in htcondor_transfer_input_files. + # Ignored with a warning if the framework is Docker. + # + # Example (one of each type): + # images: + # omicsintegrator1: "images/omics-integrator-1_v2.sif" # local .sif (Apptainer only) + # pathlinker: "pathlinker:v1234" # image name only (base_url/owner prepended) + # omicsintegrator2: "some-other-owner/oi2:latest" # owner/image (base_url prepended) + # mincostflow: "ghcr.io/reed-compbio/mincostflow:v2" # full registry reference (used as-is) + # This list of algorithms should be generated by a script which checks the filesystem for installs. # It shouldn't be changed by mere mortals. (alternatively, we could add a path to executable for each algorithm # in the list to reduce the number of assumptions of the program at the cost of making the config a little more involved) From 58ceac3dc19534d33d1bc4ad6a09e3ae1a78abb7 Mon Sep 17 00:00:00 2001 From: jeebjean Date: Mon, 27 Jul 2026 11:04:00 -0500 Subject: [PATCH 12/29] Pass summarize_networks output to run_lpca, matching PCA convention --- Snakefile | 3 ++- spras/analysis/lpca.py | 11 ++++------- 2 files changed, 6 insertions(+), 8 deletions(-) diff --git a/Snakefile b/Snakefile index 6476caee7..8d00d3cdb 100644 --- a/Snakefile +++ b/Snakefile @@ -417,8 +417,9 @@ rule lpca_analysis: lpca_coord = SEP.join([out_dir, '{dataset}-ml', '{algorithm}-lpca-coordinates.txt']) run: from spras.analysis import lpca + summary_df = ml.summarize_networks(input.pathways) lpca.run_lpca( - input.pathways, + summary_df, output.lpca_scores, k=_config.config.lpca_params.k, m=_config.config.lpca_params.m, diff --git a/spras/analysis/lpca.py b/spras/analysis/lpca.py index 4dce4c5eb..ce4b3b0b5 100644 --- a/spras/analysis/lpca.py +++ b/spras/analysis/lpca.py @@ -2,15 +2,13 @@ # Runs LPCA on the binary edge x algorithm matrix produced by summarize_networks. # Configured through the analysis.lpca block (k, m, cv, transpose). -from os import PathLike from pathlib import Path -from typing import Iterable, Union import matplotlib.pyplot as plt import pandas as pd import seaborn as sns -from spras.analysis.ml import create_palette, summarize_networks +from spras.analysis.ml import create_palette from spras.config.container_schema import ProcessedContainerSettings from spras.containers import prepare_volume, run_container_and_log @@ -21,7 +19,7 @@ def run_lpca( - file_paths: Iterable[Union[str, PathLike]], + dataframe: pd.DataFrame, output_scores: str, k: int = 2, m: float = 6, @@ -32,7 +30,7 @@ def run_lpca( Runs Logistic PCA on the binary edge x algorithm matrix built from SPRAS algorithm output files. - @param file_paths: pathway.txt file paths from SPRAS algorithm outputs + @param dataframe: binary dataframe of edge comparison between algorithms from summarize_networks @param output_scores: path to write the LPCA PC scores CSV @param k: number of principal components (default 2) @param m: fixed logisticPCA tuning parameter, used when cv is False @@ -47,8 +45,7 @@ def run_lpca( container_settings = ProcessedContainerSettings() # Step 1: build the binary edge x algorithm matrix - print('LPCA: Building binary edge x algorithm matrix...') - matrix = summarize_networks(file_paths) + matrix = dataframe # Transpose so runs are the observations, mirroring the classic PCA analysis matrix = matrix.T print(f'LPCA: Matrix shape: {matrix.shape}') From 33f0435480e114fbdb132c2837ee51eae27a3246 Mon Sep 17 00:00:00 2001 From: jeebjean Date: Mon, 27 Jul 2026 11:15:22 -0500 Subject: [PATCH 13/29] Move binary matrix to Snakefile output, pass path to run_lpca --- Snakefile | 7 +++++-- spras/analysis/lpca.py | 9 ++++++--- 2 files changed, 11 insertions(+), 5 deletions(-) diff --git a/Snakefile b/Snakefile index 8d00d3cdb..a52cfff45 100644 --- a/Snakefile +++ b/Snakefile @@ -96,6 +96,7 @@ def make_final_input(wildcards): final_input.extend(expand('{out_dir}{sep}{dataset}-ml{sep}{algorithm}-lpca-scores.csv',out_dir=out_dir, sep=SEP, dataset=dataset_labels, algorithm=algorithms_mult_param_combos)) final_input.extend(expand('{out_dir}{sep}{dataset}-ml{sep}{algorithm}-lpca.png',out_dir=out_dir, sep=SEP,dataset=dataset_labels,algorithm=algorithms_mult_param_combos)) final_input.extend(expand('{out_dir}{sep}{dataset}-ml{sep}{algorithm}-lpca-coordinates.txt',out_dir=out_dir, sep=SEP, dataset=dataset_labels,algorithm=algorithms_mult_param_combos)) + final_input.extend(expand('{out_dir}{sep}{dataset}-ml{sep}{algorithm}-lpca-binary-matrix.csv',out_dir=out_dir, sep=SEP, dataset=dataset_labels,algorithm=algorithms_mult_param_combos)) if _config.config.analysis_include_ml_aggregate_algo: final_input.extend(expand('{out_dir}{sep}{dataset}-ml{sep}{algorithm}-pca.png',out_dir=out_dir,sep=SEP,dataset=dataset_labels,algorithm=algorithms_mult_param_combos)) @@ -408,19 +409,21 @@ rule ml_analysis_aggregate_algo: ml.hac_vertical(summary_df, output.hac_image_vertical, output.hac_clusters_vertical, **hac_params) ml.hac_horizontal(summary_df, output.hac_image_horizontal, output.hac_clusters_horizontal, **hac_params) ml.pca(summary_df, output.pca_image, output.pca_variance, output.pca_coordinates, **pca_params) -rule lpca_analysis: +rrule lpca_analysis: input: pathways = collect_pathways_per_algo output: lpca_scores = SEP.join([out_dir, '{dataset}-ml', '{algorithm}-lpca-scores.csv']), lpca_png = SEP.join([out_dir, '{dataset}-ml', '{algorithm}-lpca.png']), - lpca_coord = SEP.join([out_dir, '{dataset}-ml', '{algorithm}-lpca-coordinates.txt']) + lpca_coord = SEP.join([out_dir, '{dataset}-ml', '{algorithm}-lpca-coordinates.txt']), + lpca_matrix = SEP.join([out_dir, '{dataset}-ml', '{algorithm}-lpca-binary-matrix.csv']) run: from spras.analysis import lpca summary_df = ml.summarize_networks(input.pathways) lpca.run_lpca( summary_df, output.lpca_scores, + output.lpca_matrix, k=_config.config.lpca_params.k, m=_config.config.lpca_params.m, cv=_config.config.lpca_params.cv, diff --git a/spras/analysis/lpca.py b/spras/analysis/lpca.py index ce4b3b0b5..68728388c 100644 --- a/spras/analysis/lpca.py +++ b/spras/analysis/lpca.py @@ -21,6 +21,7 @@ def run_lpca( dataframe: pd.DataFrame, output_scores: str, + output_matrix: str, k: int = 2, m: float = 6, cv: bool = False, @@ -31,6 +32,7 @@ def run_lpca( algorithm output files. @param dataframe: binary dataframe of edge comparison between algorithms from summarize_networks + @param output_matrix: path to write the binary matrix CSV (used as Docker volume mount) @param output_scores: path to write the LPCA PC scores CSV @param k: number of principal components (default 2) @param m: fixed logisticPCA tuning parameter, used when cv is False @@ -50,11 +52,11 @@ def run_lpca( matrix = matrix.T print(f'LPCA: Matrix shape: {matrix.shape}') - # Step 2: write the matrix next to the outputs, namespaced by algorithm + # Step 2: write the binary matrix for Docker volume mount output_dir = Path(output_scores).parent output_dir.mkdir(parents=True, exist_ok=True) - algo_name = Path(output_scores).name.replace('-lpca-scores.csv', '') - matrix_path = str(output_dir / f'{algo_name}-lpca_binary_matrix.csv') + Path(output_matrix).parent.mkdir(parents=True, exist_ok=True) + matrix_path = output_matrix matrix.to_csv(matrix_path) # Step 3: mount the matrix and the scores output @@ -66,6 +68,7 @@ def run_lpca( # Step 4: choose m, optionally via cross-validation if cv: + algo_name = Path(output_scores).name.replace('-lpca-scores.csv', '') cv_output_path = str(output_dir / f'{algo_name}-lpca_cv_result.csv') bind_path, mapped_cv_output = prepare_volume(cv_output_path, LPCA_WORK_DIR, container_settings) volumes.append(bind_path) From 90b509470d7275f3103b1c78897c73f45ff9afb4 Mon Sep 17 00:00:00 2001 From: jeebjean Date: Mon, 27 Jul 2026 11:23:45 -0500 Subject: [PATCH 14/29] Update tests: remove skipif, use summarize_networks, remove transpose --- test/analysis/test_lpca.py | 55 +++++++------------------------------- 1 file changed, 10 insertions(+), 45 deletions(-) diff --git a/test/analysis/test_lpca.py b/test/analysis/test_lpca.py index cf10c87d1..ad76f4100 100644 --- a/test/analysis/test_lpca.py +++ b/test/analysis/test_lpca.py @@ -1,11 +1,10 @@ from pathlib import Path -import docker import pandas as pd -import pytest import spras.config.config as config from spras.analysis.lpca import run_lpca +from spras.analysis.ml import summarize_networks config.init_from_file("config/config.yaml") @@ -18,22 +17,6 @@ 'test/evaluate/input/data-test-params-789/pathway.txt', ] -def lpca_image_available(): - """Check if the LPCA Docker image is available locally or on Docker Hub""" - try: - client = docker.from_env() - client.images.get('reedcompbio/lpca:v1') - return True - except docker.errors.ImageNotFound: - return False - except Exception: - return False - -skip_if_no_lpca_image = pytest.mark.skipif( - not lpca_image_available(), - reason='reedcompbio/lpca:v1 Docker image not available' -) - class TestLpca: """ Run Logistic PCA (LPCA) analysis tests @@ -42,58 +25,40 @@ class TestLpca: def setup_class(cls): OUT_DIR.mkdir(parents=True, exist_ok=True) - @skip_if_no_lpca_image def test_lpca_output_exists(self): """Test that LPCA produces an output scores file""" out_path = OUT_DIR / 'lpca-scores.csv' + matrix_path = OUT_DIR / 'lpca-binary-matrix.csv' out_path.unlink(missing_ok=True) + summary_df = summarize_networks(INPUT_FILES) run_lpca( - file_paths=INPUT_FILES, + dataframe=summary_df, output_scores=str(out_path), + output_matrix=str(matrix_path), k=2, m=4, cv=False, - transpose=False ) assert out_path.exists(), "LPCA scores file was not created" - @skip_if_no_lpca_image def test_lpca_output_shape(self): - """Test that LPCA scores have the correct shape (edges x k)""" + """Test that LPCA scores have the correct shape (runs x k)""" out_path = OUT_DIR / 'lpca-scores-shape.csv' + matrix_path = OUT_DIR / 'lpca-binary-matrix-shape.csv' out_path.unlink(missing_ok=True) + summary_df = summarize_networks(INPUT_FILES) run_lpca( - file_paths=INPUT_FILES, + dataframe=summary_df, output_scores=str(out_path), + output_matrix=str(matrix_path), k=2, m=4, cv=False, - transpose=False ) scores = pd.read_csv(out_path, index_col=0) assert scores.shape[1] == 2, f"Expected 2 PC columns, got {scores.shape[1]}" assert scores.shape[0] > 0, "Scores file is empty" - - @skip_if_no_lpca_image - def test_lpca_transposed_shape(self): - """Test that transposed LPCA scores have the correct shape (runs x k)""" - out_path = OUT_DIR / 'lpca-scores-transposed.csv' - out_path.unlink(missing_ok=True) - - run_lpca( - file_paths=INPUT_FILES, - output_scores=str(out_path), - k=2, - m=4, - cv=False, - transpose=True - ) - - scores = pd.read_csv(out_path, index_col=0) - assert scores.shape[1] == 2, f"Expected 2 PC columns, got {scores.shape[1]}" - assert scores.shape[0] == len(INPUT_FILES), \ - f"Expected {len(INPUT_FILES)} rows, got {scores.shape[0]}" From 20ff9018603cd99994320fa99df9335ef906f0e7 Mon Sep 17 00:00:00 2001 From: jeebjean Date: Mon, 27 Jul 2026 12:02:40 -0500 Subject: [PATCH 15/29] Add lpca_analysis_all rule for cross-algorithm LPCA --- Snakefile | 17 +++++++++++------ 1 file changed, 11 insertions(+), 6 deletions(-) diff --git a/Snakefile b/Snakefile index a52cfff45..c71625db6 100644 --- a/Snakefile +++ b/Snakefile @@ -97,6 +97,10 @@ def make_final_input(wildcards): final_input.extend(expand('{out_dir}{sep}{dataset}-ml{sep}{algorithm}-lpca.png',out_dir=out_dir, sep=SEP,dataset=dataset_labels,algorithm=algorithms_mult_param_combos)) final_input.extend(expand('{out_dir}{sep}{dataset}-ml{sep}{algorithm}-lpca-coordinates.txt',out_dir=out_dir, sep=SEP, dataset=dataset_labels,algorithm=algorithms_mult_param_combos)) final_input.extend(expand('{out_dir}{sep}{dataset}-ml{sep}{algorithm}-lpca-binary-matrix.csv',out_dir=out_dir, sep=SEP, dataset=dataset_labels,algorithm=algorithms_mult_param_combos)) + final_input.extend(expand('{out_dir}{sep}{dataset}-ml{sep}lpca.png',out_dir=out_dir, sep=SEP, dataset=dataset_labels)) + final_input.extend(expand('{out_dir}{sep}{dataset}-ml{sep}lpca-scores.csv',out_dir=out_dir, sep=SEP, dataset=dataset_labels)) + final_input.extend(expand('{out_dir}{sep}{dataset}-ml{sep}lpca-coordinates.txt',out_dir=out_dir, sep=SEP, dataset=dataset_labels)) + final_input.extend(expand('{out_dir}{sep}{dataset}-ml{sep}lpca-binary-matrix.csv',out_dir=out_dir, sep=SEP, dataset=dataset_labels)) if _config.config.analysis_include_ml_aggregate_algo: final_input.extend(expand('{out_dir}{sep}{dataset}-ml{sep}{algorithm}-pca.png',out_dir=out_dir,sep=SEP,dataset=dataset_labels,algorithm=algorithms_mult_param_combos)) @@ -409,14 +413,15 @@ rule ml_analysis_aggregate_algo: ml.hac_vertical(summary_df, output.hac_image_vertical, output.hac_clusters_vertical, **hac_params) ml.hac_horizontal(summary_df, output.hac_image_horizontal, output.hac_clusters_horizontal, **hac_params) ml.pca(summary_df, output.pca_image, output.pca_variance, output.pca_coordinates, **pca_params) -rrule lpca_analysis: + +rule lpca_analysis_all: input: - pathways = collect_pathways_per_algo + pathways = expand('{out_dir}{sep}{{dataset}}-{algorithm_params}{sep}pathway.txt', out_dir=out_dir, sep=SEP, algorithm_params=algorithms_with_params) output: - lpca_scores = SEP.join([out_dir, '{dataset}-ml', '{algorithm}-lpca-scores.csv']), - lpca_png = SEP.join([out_dir, '{dataset}-ml', '{algorithm}-lpca.png']), - lpca_coord = SEP.join([out_dir, '{dataset}-ml', '{algorithm}-lpca-coordinates.txt']), - lpca_matrix = SEP.join([out_dir, '{dataset}-ml', '{algorithm}-lpca-binary-matrix.csv']) + lpca_scores = SEP.join([out_dir, '{dataset}-ml', 'lpca-scores.csv']), + lpca_png = SEP.join([out_dir, '{dataset}-ml', 'lpca.png']), + lpca_coord = SEP.join([out_dir, '{dataset}-ml', 'lpca-coordinates.txt']), + lpca_matrix = SEP.join([out_dir, '{dataset}-ml', 'lpca-binary-matrix.csv']) run: from spras.analysis import lpca summary_df = ml.summarize_networks(input.pathways) From a16ba198570542889a901f720059b44cc4077ec6 Mon Sep 17 00:00:00 2001 From: jeebjean Date: Mon, 27 Jul 2026 13:19:19 -0500 Subject: [PATCH 16/29] Add dedicated LPCA test inputs and stronger tests --- test/analysis/input/lpca/pathway-params-1.txt | 6 ++++ test/analysis/input/lpca/pathway-params-2.txt | 6 ++++ test/analysis/input/lpca/pathway-params-3.txt | 6 ++++ test/analysis/input/lpca/pathway-params-4.txt | 6 ++++ test/analysis/test_lpca.py | 35 +++++++++++++++---- 5 files changed, 53 insertions(+), 6 deletions(-) create mode 100644 test/analysis/input/lpca/pathway-params-1.txt create mode 100644 test/analysis/input/lpca/pathway-params-2.txt create mode 100644 test/analysis/input/lpca/pathway-params-3.txt create mode 100644 test/analysis/input/lpca/pathway-params-4.txt diff --git a/test/analysis/input/lpca/pathway-params-1.txt b/test/analysis/input/lpca/pathway-params-1.txt new file mode 100644 index 000000000..e8e75157a --- /dev/null +++ b/test/analysis/input/lpca/pathway-params-1.txt @@ -0,0 +1,6 @@ +Node1 Node2 Rank Direction +A B 1 U +B C 1 U +C D 1 U +D E 1 U +E F 1 U diff --git a/test/analysis/input/lpca/pathway-params-2.txt b/test/analysis/input/lpca/pathway-params-2.txt new file mode 100644 index 000000000..c53dc06f0 --- /dev/null +++ b/test/analysis/input/lpca/pathway-params-2.txt @@ -0,0 +1,6 @@ +Node1 Node2 Rank Direction +A B 1 U +B C 1 U +C G 1 U +G H 1 U +H I 1 U diff --git a/test/analysis/input/lpca/pathway-params-3.txt b/test/analysis/input/lpca/pathway-params-3.txt new file mode 100644 index 000000000..349ffdfab --- /dev/null +++ b/test/analysis/input/lpca/pathway-params-3.txt @@ -0,0 +1,6 @@ +Node1 Node2 Rank Direction +A B 1 U +D E 1 U +E F 1 U +F J 1 U +J K 1 U diff --git a/test/analysis/input/lpca/pathway-params-4.txt b/test/analysis/input/lpca/pathway-params-4.txt new file mode 100644 index 000000000..2ae2296a0 --- /dev/null +++ b/test/analysis/input/lpca/pathway-params-4.txt @@ -0,0 +1,6 @@ +Node1 Node2 Rank Direction +B C 1 U +C D 1 U +G H 1 U +H I 1 U +I L 1 U diff --git a/test/analysis/test_lpca.py b/test/analysis/test_lpca.py index ad76f4100..8c3d004d0 100644 --- a/test/analysis/test_lpca.py +++ b/test/analysis/test_lpca.py @@ -3,7 +3,7 @@ import pandas as pd import spras.config.config as config -from spras.analysis.lpca import run_lpca +from spras.analysis.lpca import plot_lpca, run_lpca from spras.analysis.ml import summarize_networks config.init_from_file("config/config.yaml") @@ -12,9 +12,10 @@ OUT_DIR = TEST_DIR / 'output' INPUT_FILES = [ - 'test/evaluate/input/data-test-params-123/pathway.txt', - 'test/evaluate/input/data-test-params-456/pathway.txt', - 'test/evaluate/input/data-test-params-789/pathway.txt', + 'test/analysis/input/lpca/pathway-params-1.txt', + 'test/analysis/input/lpca/pathway-params-2.txt', + 'test/analysis/input/lpca/pathway-params-3.txt', + 'test/analysis/input/lpca/pathway-params-4.txt', ] class TestLpca: @@ -44,7 +45,7 @@ def test_lpca_output_exists(self): assert out_path.exists(), "LPCA scores file was not created" def test_lpca_output_shape(self): - """Test that LPCA scores have the correct shape (runs x k)""" + """Test that LPCA scores have shape (runs x k)""" out_path = OUT_DIR / 'lpca-scores-shape.csv' matrix_path = OUT_DIR / 'lpca-binary-matrix-shape.csv' out_path.unlink(missing_ok=True) @@ -61,4 +62,26 @@ def test_lpca_output_shape(self): scores = pd.read_csv(out_path, index_col=0) assert scores.shape[1] == 2, f"Expected 2 PC columns, got {scores.shape[1]}" - assert scores.shape[0] > 0, "Scores file is empty" + assert scores.shape[0] == len(INPUT_FILES), \ + f"Expected {len(INPUT_FILES)} rows, got {scores.shape[0]}" + + def test_lpca_plot_output(self): + """Test that LPCA plot and coordinates files are created""" + scores_path = OUT_DIR / 'lpca-scores-plot.csv' + matrix_path = OUT_DIR / 'lpca-binary-matrix-plot.csv' + png_path = OUT_DIR / 'lpca-plot.png' + coord_path = OUT_DIR / 'lpca-coordinates.txt' + + summary_df = summarize_networks(INPUT_FILES) + run_lpca( + dataframe=summary_df, + output_scores=str(scores_path), + output_matrix=str(matrix_path), + k=2, + m=4, + cv=False, + ) + plot_lpca(str(scores_path), str(png_path), str(coord_path)) + + assert png_path.exists(), "LPCA plot PNG was not created" + assert coord_path.exists(), "LPCA coordinates file was not created" From b063bbd7db5c37dc361c396076734b5c5313c250 Mon Sep 17 00:00:00 2001 From: jeebjean Date: Mon, 27 Jul 2026 13:27:13 -0500 Subject: [PATCH 17/29] Add LPCA to Docker image build CI workflow --- .github/workflows/build-containers.yml | 5 +++++ 1 file changed, 5 insertions(+) diff --git a/.github/workflows/build-containers.yml b/.github/workflows/build-containers.yml index 811b8e034..8bebdabdd 100644 --- a/.github/workflows/build-containers.yml +++ b/.github/workflows/build-containers.yml @@ -73,3 +73,8 @@ jobs: # of detected changes. always_build: true context: ./ + build-and-remove-lpca: + uses: "./.github/workflows/build-and-remove-template.yml" + with: + path: docker-wrappers/LPCA + container: reedcompbio/lpca From b1905c04c3553ae8594d2041ce2bc0913104887a Mon Sep 17 00:00:00 2001 From: jeebjean Date: Mon, 27 Jul 2026 13:40:23 -0500 Subject: [PATCH 18/29] Fix test input files: use tab-separated format --- test/analysis/input/lpca/pathway-params-1.txt | 12 ++++++------ test/analysis/input/lpca/pathway-params-2.txt | 12 ++++++------ test/analysis/input/lpca/pathway-params-3.txt | 12 ++++++------ test/analysis/input/lpca/pathway-params-4.txt | 12 ++++++------ 4 files changed, 24 insertions(+), 24 deletions(-) diff --git a/test/analysis/input/lpca/pathway-params-1.txt b/test/analysis/input/lpca/pathway-params-1.txt index e8e75157a..affde9164 100644 --- a/test/analysis/input/lpca/pathway-params-1.txt +++ b/test/analysis/input/lpca/pathway-params-1.txt @@ -1,6 +1,6 @@ -Node1 Node2 Rank Direction -A B 1 U -B C 1 U -C D 1 U -D E 1 U -E F 1 U +Node1 Node2 Rank Direction +A B 1 U +B C 1 U +C D 1 U +D E 1 U +E F 1 U diff --git a/test/analysis/input/lpca/pathway-params-2.txt b/test/analysis/input/lpca/pathway-params-2.txt index c53dc06f0..b1c72f8b4 100644 --- a/test/analysis/input/lpca/pathway-params-2.txt +++ b/test/analysis/input/lpca/pathway-params-2.txt @@ -1,6 +1,6 @@ -Node1 Node2 Rank Direction -A B 1 U -B C 1 U -C G 1 U -G H 1 U -H I 1 U +Node1 Node2 Rank Direction +A B 1 U +B C 1 U +C G 1 U +G H 1 U +H I 1 U diff --git a/test/analysis/input/lpca/pathway-params-3.txt b/test/analysis/input/lpca/pathway-params-3.txt index 349ffdfab..8211a5a56 100644 --- a/test/analysis/input/lpca/pathway-params-3.txt +++ b/test/analysis/input/lpca/pathway-params-3.txt @@ -1,6 +1,6 @@ -Node1 Node2 Rank Direction -A B 1 U -D E 1 U -E F 1 U -F J 1 U -J K 1 U +Node1 Node2 Rank Direction +A B 1 U +D E 1 U +E F 1 U +F J 1 U +J K 1 U diff --git a/test/analysis/input/lpca/pathway-params-4.txt b/test/analysis/input/lpca/pathway-params-4.txt index 2ae2296a0..67f1cf554 100644 --- a/test/analysis/input/lpca/pathway-params-4.txt +++ b/test/analysis/input/lpca/pathway-params-4.txt @@ -1,6 +1,6 @@ -Node1 Node2 Rank Direction -B C 1 U -C D 1 U -G H 1 U -H I 1 U -I L 1 U +Node1 Node2 Rank Direction +B C 1 U +C D 1 U +G H 1 U +H I 1 U +I L 1 U From ec14dc8bfee3307c51eb4d61558cc77e91ed69c4 Mon Sep 17 00:00:00 2001 From: jeebjean Date: Thu, 30 Jul 2026 14:17:26 -0500 Subject: [PATCH 19/29] Add lpca aggregate_per_algorithm option and per-algorithm rule; enable lpca in CI; document partial_decomp --- Snakefile | 36 ++++++++++++++++++++++++++++++---- config/config.yaml | 4 +++- docker-wrappers/LPCA/README.md | 14 ++++++++++++- spras/config/config.py | 6 ++++++ spras/config/schema.py | 1 + 5 files changed, 55 insertions(+), 6 deletions(-) diff --git a/Snakefile b/Snakefile index c71625db6..ff80ba7ae 100644 --- a/Snakefile +++ b/Snakefile @@ -93,15 +93,17 @@ def make_final_input(wildcards): final_input.extend(expand('{out_dir}{sep}{dataset}-ml{sep}jaccard-heatmap.png',out_dir=out_dir,sep=SEP,dataset=dataset_labels,algorithm_params=algorithms_with_params)) if _config.config.analysis_include_lpca: - final_input.extend(expand('{out_dir}{sep}{dataset}-ml{sep}{algorithm}-lpca-scores.csv',out_dir=out_dir, sep=SEP, dataset=dataset_labels, algorithm=algorithms_mult_param_combos)) - final_input.extend(expand('{out_dir}{sep}{dataset}-ml{sep}{algorithm}-lpca.png',out_dir=out_dir, sep=SEP,dataset=dataset_labels,algorithm=algorithms_mult_param_combos)) - final_input.extend(expand('{out_dir}{sep}{dataset}-ml{sep}{algorithm}-lpca-coordinates.txt',out_dir=out_dir, sep=SEP, dataset=dataset_labels,algorithm=algorithms_mult_param_combos)) - final_input.extend(expand('{out_dir}{sep}{dataset}-ml{sep}{algorithm}-lpca-binary-matrix.csv',out_dir=out_dir, sep=SEP, dataset=dataset_labels,algorithm=algorithms_mult_param_combos)) final_input.extend(expand('{out_dir}{sep}{dataset}-ml{sep}lpca.png',out_dir=out_dir, sep=SEP, dataset=dataset_labels)) final_input.extend(expand('{out_dir}{sep}{dataset}-ml{sep}lpca-scores.csv',out_dir=out_dir, sep=SEP, dataset=dataset_labels)) final_input.extend(expand('{out_dir}{sep}{dataset}-ml{sep}lpca-coordinates.txt',out_dir=out_dir, sep=SEP, dataset=dataset_labels)) final_input.extend(expand('{out_dir}{sep}{dataset}-ml{sep}lpca-binary-matrix.csv',out_dir=out_dir, sep=SEP, dataset=dataset_labels)) + if _config.config.analysis_include_lpca_aggregate_algo: + final_input.extend(expand('{out_dir}{sep}{dataset}-ml{sep}{algorithm}-lpca-scores.csv',out_dir=out_dir, sep=SEP, dataset=dataset_labels, algorithm=algorithms_mult_param_combos)) + final_input.extend(expand('{out_dir}{sep}{dataset}-ml{sep}{algorithm}-lpca.png',out_dir=out_dir, sep=SEP,dataset=dataset_labels,algorithm=algorithms_mult_param_combos)) + final_input.extend(expand('{out_dir}{sep}{dataset}-ml{sep}{algorithm}-lpca-coordinates.txt',out_dir=out_dir, sep=SEP, dataset=dataset_labels,algorithm=algorithms_mult_param_combos)) + final_input.extend(expand('{out_dir}{sep}{dataset}-ml{sep}{algorithm}-lpca-binary-matrix.csv',out_dir=out_dir, sep=SEP, dataset=dataset_labels,algorithm=algorithms_mult_param_combos)) + if _config.config.analysis_include_ml_aggregate_algo: final_input.extend(expand('{out_dir}{sep}{dataset}-ml{sep}{algorithm}-pca.png',out_dir=out_dir,sep=SEP,dataset=dataset_labels,algorithm=algorithms_mult_param_combos)) final_input.extend(expand('{out_dir}{sep}{dataset}-ml{sep}{algorithm}-pca-variance.txt',out_dir=out_dir,sep=SEP,dataset=dataset_labels,algorithm=algorithms_mult_param_combos)) @@ -440,6 +442,32 @@ rule lpca_analysis_all: output.lpca_coord ) +rule lpca_analysis_aggregate_algo: + input: + pathways = collect_pathways_per_algo + output: + lpca_scores = SEP.join([out_dir, '{dataset}-ml', '{algorithm}-lpca-scores.csv']), + lpca_png = SEP.join([out_dir, '{dataset}-ml', '{algorithm}-lpca.png']), + lpca_coord = SEP.join([out_dir, '{dataset}-ml', '{algorithm}-lpca-coordinates.txt']), + lpca_matrix = SEP.join([out_dir, '{dataset}-ml', '{algorithm}-lpca-binary-matrix.csv']) + run: + from spras.analysis import lpca + summary_df = ml.summarize_networks(input.pathways) + lpca.run_lpca( + summary_df, + output.lpca_scores, + output.lpca_matrix, + k=_config.config.lpca_params.k, + m=_config.config.lpca_params.m, + cv=_config.config.lpca_params.cv, + container_settings=container_settings + ) + lpca.plot_lpca( + output.lpca_scores, + output.lpca_png, + output.lpca_coord + ) + # Ensemble the output pathways for each dataset per algorithm rule ensemble_per_algo: input: diff --git a/config/config.yaml b/config/config.yaml index 9f6b00fad..33ebc9720 100644 --- a/config/config.yaml +++ b/config/config.yaml @@ -275,7 +275,9 @@ analysis: lpca: # if true, runs LPCA in addition to the existing PCA (both analyses will run in parallel). # Running both helps compare the two methods on the same data. - include: false + include: true + # adds separate LPCA analysis for each algorithm that has multiple parameter combinations + aggregate_per_algorithm: false # number of principal components to compute k: 2 # fixed value of the logisticPCA tuning parameter m, used when cv is false. diff --git a/docker-wrappers/LPCA/README.md b/docker-wrappers/LPCA/README.md index 6a3a17e85..820a8482c 100644 --- a/docker-wrappers/LPCA/README.md +++ b/docker-wrappers/LPCA/README.md @@ -4,7 +4,6 @@ Docker image: https://hub.docker.com/r/reedcompbio/lpca This wrapper runs [logisticPCA](https://github.com/andland/logisticPCA) -This wrapper runs [logisticPCA](https://github.com/andland/logisticPCA) ([Landgraf & Lee, 2020](https://doi.org/10.1016/j.jmva.2020.104668)) as a SPRAS analysis step. It reduces the binary edge-by-run matrix built from a set of pathway reconstruction outputs to a small number of components and reports the proportion of deviance explained. @@ -26,6 +25,19 @@ binary matrix has more than one column. It also needs a reasonable number of observations to be meaningful; very small inputs (such as the bundled example datasets) produce degenerate results, which is why it is disabled by default. + ### `partial_decomp` + + The LPCA wrapper always runs `logisticSVD` with `partial_decomp = TRUE`, which + uses a truncated (rARPACK-based) decomposition instead of a full one. This is + hardcoded rather than exposed as a parameter: + + - On small datasets it has no practical effect on the result. + - On large datasets it is required to avoid out-of-memory (OOMKilled) errors + that occur with the full decomposition. + + Because it is beneficial on large inputs and harmless on small ones, it is + enabled unconditionally and is not a user-facing configuration option. + ## Scripts The image contains two R scripts under `/app`: diff --git a/spras/config/config.py b/spras/config/config.py index e99d965c9..f29bc3bef 100644 --- a/spras/config/config.py +++ b/spras/config/config.py @@ -267,6 +267,12 @@ def process_analysis(self, raw_config: RawConfig): else: self.analysis_include_ml_aggregate_algo = False + # Only run LPCA aggregate per algorithm if analysis include LPCA is set to True + if self.analysis_include_lpca and raw_config.analysis.lpca.aggregate_per_algorithm: + self.analysis_include_lpca_aggregate_algo = raw_config.analysis.lpca.aggregate_per_algorithm + else: + self.analysis_include_lpca_aggregate_algo = False + # Raises an error if Evaluation is enabled but no gold standard data is provided if self.gold_standards == {} and self.analysis_include_evaluation: raise ValueError("Evaluation analysis cannot run as gold standard data not provided. " diff --git a/spras/config/schema.py b/spras/config/schema.py index f2e0ba353..6bda7efdd 100644 --- a/spras/config/schema.py +++ b/spras/config/schema.py @@ -69,6 +69,7 @@ class EvaluationAnalysis(BaseModel): class LpcaAnalysis(BaseModel): include: bool + aggregate_per_algorithm: bool = False k: int = 2 m: float = 6 cv: bool = False From 1704c475551acec676ec3ca9868805f9d916fe5c Mon Sep 17 00:00:00 2001 From: jeebjean Date: Thu, 30 Jul 2026 14:31:37 -0500 Subject: [PATCH 20/29] Set lpca include back to false after confirming CI run --- config/config.yaml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/config/config.yaml b/config/config.yaml index 33ebc9720..10bec536b 100644 --- a/config/config.yaml +++ b/config/config.yaml @@ -275,7 +275,7 @@ analysis: lpca: # if true, runs LPCA in addition to the existing PCA (both analyses will run in parallel). # Running both helps compare the two methods on the same data. - include: true + include: false # adds separate LPCA analysis for each algorithm that has multiple parameter combinations aggregate_per_algorithm: false # number of principal components to compute From 5b64162a4b93691f02615696e62f7542ddda753d Mon Sep 17 00:00:00 2001 From: jeebjean Date: Thu, 30 Jul 2026 14:43:50 -0500 Subject: [PATCH 21/29] Add stronger LPCA test comparing scores to known outputs (sign-robust) --- test/analysis/test_lpca.py | 38 ++++++++++++++++++++++++++++++++++++++ 1 file changed, 38 insertions(+) diff --git a/test/analysis/test_lpca.py b/test/analysis/test_lpca.py index 8c3d004d0..75f84f451 100644 --- a/test/analysis/test_lpca.py +++ b/test/analysis/test_lpca.py @@ -85,3 +85,41 @@ def test_lpca_plot_output(self): assert png_path.exists(), "LPCA plot PNG was not created" assert coord_path.exists(), "LPCA coordinates file was not created" + + def test_lpca_known_output(self): + """Test that LPCA produces the expected scores for known inputs. + + Compared in absolute value because LPCA components, like PCA, are only + defined up to a sign (an axis can flip without changing the result). + """ + out_path = OUT_DIR / 'lpca-scores-known.csv' + matrix_path = OUT_DIR / 'lpca-binary-matrix-known.csv' + out_path.unlink(missing_ok=True) + + summary_df = summarize_networks(INPUT_FILES) + run_lpca( + dataframe=summary_df, + output_scores=str(out_path), + output_matrix=str(matrix_path), + k=2, + m=4, + cv=False, + ) + + scores = pd.read_csv(out_path, index_col=0) + + expected = pd.DataFrame({ + 'V1': [4.96756989310632, -7.42875819587666, 12.7488840407735, -11.2979872320273], + 'V2': [-7.80927209388169, 7.21294742051829, 1.711517188498, -5.51380214217161], + }) + + assert scores.shape == expected.shape, \ + f"Expected shape {expected.shape}, got {scores.shape}" + + # Compare in absolute value to stay robust to sign flips (sign non-identifiability) + pd.testing.assert_frame_equal( + scores.reset_index(drop=True).abs(), + expected.abs(), + check_dtype=False, + atol=1e-4, + ) From 2ced8afa9f6271c56e2ff3d10e3e5b2e46df204d Mon Sep 17 00:00:00 2001 From: Anthony Gitter Date: Tue, 29 Sep 2026 12:11:54 -0500 Subject: [PATCH 22/29] Add robustness to LPCA R wrapper and test code --- docker-wrappers/LPCA/Dockerfile | 23 +- docker-wrappers/LPCA/README.md | 103 ++++--- docker-wrappers/LPCA/run_cv.R | 36 --- docker-wrappers/LPCA/run_lpca.R | 113 ++++++-- docker-wrappers/LPCA/test_container.py | 375 +++++++++++++++++++++++++ 5 files changed, 543 insertions(+), 107 deletions(-) delete mode 100644 docker-wrappers/LPCA/run_cv.R create mode 100644 docker-wrappers/LPCA/test_container.py diff --git a/docker-wrappers/LPCA/Dockerfile b/docker-wrappers/LPCA/Dockerfile index 66b27893a..241ba6dd5 100644 --- a/docker-wrappers/LPCA/Dockerfile +++ b/docker-wrappers/LPCA/Dockerfile @@ -1,6 +1,6 @@ # Logistic PCA (logisticPCA) wrapper for SPRAS. -# Invoked by spras/analysis/lpca.py, which supplies the full command -# (Rscript /app/run_lpca.R ... or /app/run_cv.R ...), so no ENTRYPOINT is set. +# The caller supplies Rscript /app/run_lpca.R and its arguments. +# Cross-validation is not supported, and no ENTRYPOINT is set. # Pinned R version for reproducibility. Bump deliberately, not to :latest. FROM rocker/r-base:4.4.2 @@ -21,11 +21,20 @@ RUN apt-get update && apt-get install -y --no-install-recommends \ libjpeg-dev \ && rm -rf /var/lib/apt/lists/* -# logisticPCA is still on CRAN (last published 2016) and pulls in ggplot2. -RUN Rscript -e "install.packages(c('logisticPCA', 'rARPACK'), repos='https://cran.r-project.org')" \ - && Rscript -e "library(logisticPCA); library(rARPACK)" +# Install tested versions of required R packages including rARPACK's RSpectra backend. +# remotes finds these versions in CRAN or its archive. +# Install only required dependencies and do not upgrade previously installed ones. +# Transitive and system dependencies are not fully pinned by this Dockerfile. +RUN Rscript -e "options(repos = c(CRAN = 'https://cran.r-project.org')); \ + install.packages('remotes'); \ + versions <- c(RSpectra = '0.16-2', rARPACK = '0.11-0', logisticPCA = '0.2'); \ + for (pkg in names(versions)) { \ + remotes::install_version(pkg, version = versions[[pkg]], \ + dependencies = NA, upgrade = 'never'); \ + stopifnot(packageVersion(pkg) == package_version(versions[[pkg]])); \ + }; \ + library(logisticPCA); library(rARPACK); library(RSpectra)" COPY run_lpca.R /app/run_lpca.R -COPY run_cv.R /app/run_cv.R -WORKDIR /app \ No newline at end of file +WORKDIR /app diff --git a/docker-wrappers/LPCA/README.md b/docker-wrappers/LPCA/README.md index 820a8482c..808cf6a17 100644 --- a/docker-wrappers/LPCA/README.md +++ b/docker-wrappers/LPCA/README.md @@ -2,61 +2,71 @@ Docker image: https://hub.docker.com/r/reedcompbio/lpca -This wrapper runs [logisticPCA](https://github.com/andland/logisticPCA) +This wrapper uses [logisticPCA](https://github.com/andland/logisticPCA) +([Landgraf & Lee, 2020](https://doi.org/10.1016/j.jmva.2020.104668)) for an +exploratory, two-dimensional comparison of reconstructed networks. It calls +`logisticPCA`, not `logisticSVD`. It supports only two components and a fixed, +finite, strictly positive `m`. Cross-validation for automatic selection of `m` +and modifying `k` are not supported. -([Landgraf & Lee, 2020](https://doi.org/10.1016/j.jmva.2020.104668)) as a SPRAS analysis step. It reduces the binary -edge-by-run matrix built from a set of pathway reconstruction outputs to a small -number of components and reports the proportion of deviance explained. +## Container interface -The analysis is driven by the `analysis.lpca` config block and the -`lpca_analysis` Snakemake rule, and is implemented in `spras/analysis/lpca.py`. +The image contains one R script, invoked explicitly by the caller: -## Configuration +```text +Rscript /app/run_lpca.R +``` - analysis: - lpca: - include: false # run the LPCA analysis per algorithm - k: 2 # number of principal components - m: 6 # fixed logisticPCA tuning parameter, used when cv is false - cv: false # true: choose m by cross-validation; false: use the fixed m +### Input -LPCA only runs for algorithms with multiple parameter combinations, so that the -binary matrix has more than one column. It also needs a reasonable number of -observations to be meaningful; very small inputs (such as the bundled example -datasets) produce degenerate results, which is why it is disabled by default. +The CSV has a header row, a first column of nonempty, unique run identifiers, +and one column per binary edge feature. Rows are runs, not edges. The caller +must transpose the edge-by-run matrix returned by `summarize_networks` before +writing it. - ### `partial_decomp` +This wrapper deliberately requires at least three runs, three edge features, +and three distinct binary network profiles. All feature values must be finite +numeric zeros or ones. Missing, nonnumeric, and nonbinary entries are error. +Duplicate profiles and constant features are +retained when the input otherwise satisfies these requirements. - The LPCA wrapper always runs `logisticSVD` with `partial_decomp = TRUE`, which - uses a truncated (rARPACK-based) decomposition instead of a full one. This is - hardcoded rather than exposed as a parameter: +`k` must equal 2. `m` must be finite and strictly positive. - - On small datasets it has no practical effect on the result. - - On large datasets it is required to avoid out-of-memory (OOMKilled) errors - that occur with the full decomposition. +### Outputs and numerical checks - Because it is beneficial on large inputs and harmless on small ones, it is - enabled unconditionally and is not a user-facing configuration option. +The raw scores CSV contains `datapoint_labels`, `PC1`, and `PC2`, with one row +per input run and no synthetic centroid. It is an intermediate for Python's +PCA-compatible plotting and coordinate formatting, not a second public +coordinate table. -## Scripts +The deviance summary is a small text file: -The image contains two R scripts under `/app`: +```text +components: 2 +m: +percent_deviance_explained: <100 times the package's whole-model statistic> +``` -- `run_lpca.R `: runs logisticPCA with a fixed `m` and - writes the scores CSV plus a sibling `_deviance.txt`. -- `run_cv.R `: cross-validates `m` over 1..20 and writes a - CSV with a `best_m` column (plus a `_curve.csv` with the full CV curve). Only - used when `cv: true`. +## SPRAS configuration -Both read a CSV whose first column holds row labels and whose remaining columns -are binary (0/1) features, and coerce missing values to 0. +The `analysis.lpca` config settings are: -## Dependency note +```yaml +analysis: + lpca: + include: false + aggregate_per_algorithm: false + k: 2 + m: 6 +``` -`logisticPCA` declares `ggplot2` as a hard `Imports` dependency, so building the -image compiles the ggplot2 stack. The Dockerfile installs the required Debian -system libraries for that. To build from a source CRAN mirror the `repos` -argument already points at `https://cran.r-project.org`. +## Decomposition + +`partial_decomp=TRUE` is always set. It uses +`rARPACK` (backed by `RSpectra`) for partial decompositions where supported. The +package can fall back to full eigendecomposition. It still constructs dense +edge-by-edge matrices, so memory use can grow quadratically with the number of +edge features. ## Building and publishing the image @@ -67,3 +77,16 @@ Docker Hub organization: docker build -t reedcompbio/lpca:v1 docker-wrappers/lpca/ docker push reedcompbio/lpca:v1 +## Testing +Run the test Python script to test the Docker image +```commandline +python docker-wrappers/LPCA/test_container.py +``` +Expected output will end with +```commandline +=== RESULT: 28 passed; 0 failed === +Container exit status: 0 +``` + +## AI +GPT 6 Astra was used to refactor these files and write the test code. diff --git a/docker-wrappers/LPCA/run_cv.R b/docker-wrappers/LPCA/run_cv.R deleted file mode 100644 index 7301d3419..000000000 --- a/docker-wrappers/LPCA/run_cv.R +++ /dev/null @@ -1,36 +0,0 @@ -# run_cv.R -# Finds the optimal m for a given k using cross-validation - -args = commandArgs(trailingOnly = TRUE) -input_file = args[1] -output_file = args[2] -k = as.integer(args[3]) - -set.seed(42) - -# Load data -library(logisticPCA) -data = read.csv(input_file, row.names = NULL) -data = data[, -1] -data_matrix = as.matrix(data) -data_matrix[is.na(data_matrix)] = 0 - -# Cross-validation over m, fixed k -cv_result = cv.lpca(data_matrix, ks = k, ms = 1:20) -best_m = which.min(cv_result) - -cat("Cross-validation done for k =", k, "\n") -cat("Best m:", best_m, "\n") - -# Save best m -write.csv(data.frame(k = k, best_m = best_m), output_file, row.names = FALSE) - -# Save full CV curve (all m values and their reconstruction error) -cv_curve_file = sub("\\.csv$", "_curve.csv", output_file) -cv_df = data.frame( - m = 1:20, - reconstruction_error = as.numeric(cv_result), - is_best = (1:20) == best_m -) -write.csv(cv_df, cv_curve_file, row.names = FALSE) -cat("CV curve saved to", cv_curve_file, "\n") \ No newline at end of file diff --git a/docker-wrappers/LPCA/run_lpca.R b/docker-wrappers/LPCA/run_lpca.R index 1a3049e56..ce62c98f5 100644 --- a/docker-wrappers/LPCA/run_lpca.R +++ b/docker-wrappers/LPCA/run_lpca.R @@ -1,30 +1,95 @@ -# Read command line arguments -args = commandArgs(trailingOnly = TRUE) -input_file = args[1] -output_file = args[2] -k = as.integer(args[3]) -m = as.numeric(args[4]) +# Fixed-m, two-component logistic PCA for SPRAS. -# Load data -library(logisticPCA) -data = read.csv(input_file, row.names = NULL) -row_labels = data[, 1] -data = data[, -1] -data_matrix = as.matrix(data) -data_matrix[is.na(data_matrix)] = 0 +# Read command-line arguments. +args <- commandArgs(trailingOnly = TRUE) +if (length(args) != 5L) { + stop("Usage: run_lpca.R ") +} +input_file <- args[1] +output_file <- args[2] +k <- as.numeric(args[3]) +m <- as.numeric(args[4]) +deviance_file <- args[5] +if (!is.finite(k) || k != 2) { + stop("k must be exactly 2.") +} +if (!is.finite(m) || m <= 0) { + stop("m must be finite and positive; m=0 requests estimation.") +} -# Run LPCA -model = logisticPCA(data_matrix, k = k, m = m, partial_decomp = TRUE) +# Read labels as text and convert only the binary feature columns to numbers. +data <- read.csv(input_file, row.names = NULL, check.names = FALSE, + colClasses = "character", na.strings = character(), + fill = FALSE, blank.lines.skip = FALSE) +# Require at least 3 runs and 3 edge features, plus the run-label column. +if (nrow(data) < 3L || ncol(data) < 4L) { + stop("LPCA requires at least 3 runs and 3 binary edge features.") +} +row_labels <- data[[1]] +if (anyNA(row_labels) || any(!nzchar(trimws(row_labels))) || + anyDuplicated(row_labels)) { + stop("Run labels must be nonempty and unique.") +} +data_matrix <- as.matrix(data[, -1, drop = FALSE]) +storage.mode(data_matrix) <- "double" +if (any(!is.finite(data_matrix)) || any(!data_matrix %in% c(0, 1))) { + stop("Edge features must be complete, finite numeric zeros or ones.") +} +if (nrow(unique(data_matrix)) < 3L) { + stop("LPCA requires at least 3 distinct binary network profiles.") +} -# Save scores -scores = model$PCs -rownames(scores) = row_labels -write.csv(scores, output_file, row.names = TRUE) +# Fit LPCA. The package can roll back and truncate the loss trace on failure, +# so retain the specific deviance-increase warning check. +max_iters <- 1000L +conv_criteria <- 1e-5 +set.seed(42) +model <- withCallingHandlers( + logisticPCA::logisticPCA( + data_matrix, k = k, m = m, main_effects = TRUE, + partial_decomp = TRUE, random_start = FALSE, + max_iters = max_iters, conv_criteria = conv_criteria + ), + warning = function(w) { + if (grepl("Algorithm stopped because deviance increased", + conditionMessage(w), fixed = TRUE)) { + stop("LPCA deviance increased during optimization.") + } + } +) -# Save deviance explained -deviance_file = sub("\\.csv$", "_deviance.txt", output_file) -writeLines(as.character(model$prop_deviance_expl), deviance_file) +# Check the results and convergence before writing outputs. +scores <- model$PCs +percent_deviance <- 100 * model$prop_deviance_expl +if (!identical(dim(scores), c(nrow(data_matrix), 2L)) || + any(!is.finite(scores)) || !is.finite(percent_deviance)) { + stop("LPCA returned invalid scores or deviance explained.") +} +loss <- model$loss_trace +if (length(loss) < 2L || any(!is.finite(loss))) { + stop("LPCA returned an invalid loss trace.") +} +changes <- diff(loss) +if (any(changes > 1e-10)) { + stop("LPCA deviance increased during optimization.") +} +if (abs(tail(changes, 1L)) >= conv_criteria) { + stop(sprintf("LPCA did not converge within %d iterations.", max_iters)) +} + +# Python adds the centroid and formats the public coordinate table. +score_table <- data.frame(datapoint_labels = row_labels, + PC1 = scores[, 1], PC2 = scores[, 2]) +summary <- c( + sprintf("components: %d", k), + sprintf("m: %.17g", m), + sprintf("percent_deviance_explained: %.17g", percent_deviance) +) +write.csv(score_table, output_file, row.names = FALSE) +writeLines(summary, deviance_file) cat("LPCA done! Scores saved to", output_file, "\n") -cat("Score dimensions:", nrow(scores), "x", ncol(scores), "\n") -cat("Proportion of deviance explained:", model$prop_deviance_expl, "\n") \ No newline at end of file +cat("Score dimensions:", nrow(scores), "x", k, "\n") +cat("Percent deviance explained:", percent_deviance, "\n") +cat("Iterations:", model$iters, "\n") +cat("logisticPCA version:", as.character(packageVersion("logisticPCA")), "\n") diff --git a/docker-wrappers/LPCA/test_container.py b/docker-wrappers/LPCA/test_container.py new file mode 100644 index 000000000..dabf402a2 --- /dev/null +++ b/docker-wrappers/LPCA/test_container.py @@ -0,0 +1,375 @@ +#!/usr/bin/env python3 +"""Run standalone regression tests for the SPRAS LPCA container. + +Save this file beside run_lpca.R in docker-wrappers/LPCA/. From the repository +root, build the image and run the tests with Python >= 3.9 and Docker: + docker build -t reedcompbio/lpca:v1 docker-wrappers/LPCA + python docker-wrappers/LPCA/test_container.py + +The wrapper is located relative to this file, not the working directory. +Results go to the terminal; --log PATH also saves them to a new UTF-8 file. +Use --image TAG to select a different already-built local image. Tests never +pull an image or require a registry login. + +The runner checks the local/container wrapper match (ignoring CRLF vs LF), +package versions, numerical regressions, and input/output behavior. All +fixtures are created inside one disposable Linux container, with no network +or host mounts. Local R, pytest, and an installed SPRAS package are not needed. +Each fit runs the real five-argument CLI in a separate Rscript process; the +production wrapper needs no testing hooks. Scope: fixed positive m, exactly +two components, and whole-model deviance, without CV or per-axis variance. + +The process exits with zero only after a completed, passing test run. This +suite does not test SPRAS volume mapping, plotting, or Snakemake integration. +""" + +import argparse +import hashlib +import json +import os +import shutil +import subprocess +import sys +import uuid +from contextlib import nullcontext +from pathlib import Path + +R_TESTS = r''' +options(warn = 1) +cat("=== Environment and wrapper identity ===\n") +print(sessionInfo()) +versions <- c(logisticPCA = "0.2", rARPACK = "0.11-0", RSpectra = "0.16-2") +for (pkg in names(versions)) { + actual <- packageVersion(pkg) + cat(pkg, as.character(actual), "\n") + if (actual != package_version(versions[[pkg]])) { + stop(paste("Unexpected package version:", pkg)) + } +} + +# Compare exact source bytes except for Windows CRLF versus Unix LF newlines. +source_file <- "/app/run_lpca.R" +source_bytes <- readBin(source_file, "raw", n = file.info(source_file)$size) +source_text <- gsub("\r\n", "\n", rawToChar(source_bytes), fixed = TRUE) +normalized <- tempfile() +writeBin(charToRaw(source_text), normalized) +actual_md5 <- unname(tools::md5sum(normalized)) +unlink(normalized) +cat("Container wrapper MD5 (LF normalized):", actual_md5, "\n") +cat("=== Wrapper source being tested ===\n", source_text, "\n", sep = "") +if (!identical(actual_md5, expected_md5)) { + stop("The image does not contain the adjacent run_lpca.R. Rebuild the selected image.") +} +cat("Local/container wrapper match: PASS\n") + +work <- tempfile(pattern = "LPCA numerical checks ") +dir.create(work) +passed <- 0L +failed <- 0L +need <- function(condition, message) { + if (!isTRUE(condition)) stop(message, call. = FALSE) +} +check <- function(name, action) { + cat("\n===", name, "===\n") + tryCatch({ + action() + passed <<- passed + 1L + cat("PASS:", name, "\n") + }, error = function(e) { + failed <<- failed + 1L + cat("FAIL:", name, "|", conditionMessage(e), "\n") + }) +} +near <- function(actual, expected, tolerance, label) { + need(length(actual) == length(expected), paste(label, "length mismatch")) + delta <- max(abs(as.numeric(actual) - as.numeric(expected))) + cat(label, "maximum absolute difference:", format(delta, digits = 10), + "(tolerance", tolerance, ")\n") + need(is.finite(delta) && delta < tolerance, paste(label, "differs from reference")) +} + +# Regression reference provenance: +# The binary patterns below correspond to pathway-params-1.txt through +# pathway-params-4.txt in test/analysis/input/lpca/. The reference scores come +# from test/analysis/test_lpca.py at SPRAS commit +# 5b64162a4b93691f02615696e62f7542ddda753d. +# The deviance references were recorded on the same patterns using R 4.4.2, +# logisticPCA 0.2, rARPACK 0.11-0, and RSpectra 0.16-2 on x86_64 Linux: +# k=2, m=4: proportion 0.9192537 (recorded console precision) +# k=2, m=6: proportion 0.976137532256003 (recorded output-file precision) +# Both fits used main effects and partial_decomp=TRUE. These are observed +# regression results, not an independent proof of numerical correctness. +# Distances allow a global orthogonal change of axes. Absolute tolerances are +# 1e-3 for reference distances and deviance percentages and 1e-6 for repeated +# fits. Investigate changes before updating references or loosening tolerances, +# especially when updating the numerical package versions above. +x <- rbind( + c(1,1,1,1,1,0,0,0,0,0,0), + c(1,1,0,0,0,1,1,1,0,0,0), + c(1,0,0,1,1,0,0,0,1,1,0), + c(0,1,1,0,0,0,1,1,0,0,1) +) +labels <- c("001", "NA", "run,three", "run four") +fixture <- function(mat = x, ids = paste0("run", seq_len(nrow(mat))), + k = "2", m = "4", blank_header = FALSE) { + directory <- tempfile(pattern = "case ", tmpdir = work) + dir.create(directory) + input <- file.path(directory, "matrix input.csv") + scores <- file.path(directory, "raw scores.csv") + deviance <- file.path(directory, "separate fit summary.txt") + write.csv(data.frame(datapoint_labels = ids, mat), input, + row.names = FALSE, na = "") + if (blank_header) { + lines <- readLines(input) + lines[1] <- sub('^"datapoint_labels"', '', lines[1]) + writeLines(lines, input) + } + list(input = input, scores = scores, deviance = deviance, + args = c(input, scores, k, m, deviance), ids = ids, mat = mat) +} +invoke <- function(f, expected_error = NULL, arguments = f$args) { + before <- tools::md5sum(f$input) + # The subprocess runs inside Linux; shQuote protects file paths with spaces. + output <- suppressWarnings(system2( + file.path(R.home("bin"), "Rscript"), + c("--vanilla", shQuote(c(source_file, arguments))), + stdout = TRUE, stderr = TRUE, timeout = 90 + )) + status <- attr(output, "status") + if (is.null(status)) status <- 0L + cat(paste(output, collapse = "\n"), "\n") + cat("Wrapper exit status:", status, "\n") + need(identical(before, tools::md5sum(f$input)), "The input CSV was modified") + need(status != 124L, "The wrapper timed out; this is not an expected rejection") + if (is.null(expected_error)) { + need(status == 0L, "The valid-input fit failed") + return(invisible(NULL)) + } + need(status != 0L, "Invalid input unexpectedly succeeded") + need(any(grepl(expected_error, output, fixed = TRUE)), + paste("Missing expected error text:", expected_error)) + need(!file.exists(f$scores) && !file.exists(f$deviance), + "A rejected fit left a scores or deviance output") +} +read_results <- function(f) { + need(file.exists(f$scores) && file.exists(f$deviance), "A required output is missing") + table <- read.csv(f$scores, row.names = NULL, colClasses = "character", + na.strings = character(), check.names = FALSE) + need(identical(names(table), c("datapoint_labels", "PC1", "PC2")), + "Unexpected scores CSV columns") + need(nrow(table) == nrow(f$mat), "Wrong score-row count (no centroid belongs here)") + need(identical(table$datapoint_labels, f$ids), "Run labels/order were not preserved") + scores <- as.matrix(table[, c("PC1", "PC2"), drop = FALSE]) + storage.mode(scores) <- "double" + need(all(is.finite(scores)), "Nonfinite score value") + lines <- readLines(f$deviance) + keys <- c("components", "m", "percent_deviance_explained") + need(identical(sub(":.*$", "", lines), keys), "Unexpected fit-summary fields") + values <- as.numeric(sub("^[^:]+: *", "", lines)) + need(all(is.finite(values)) && values[1] == 2 && + values[2] == as.numeric(f$args[4]), "Invalid summary values or wrong m/k") + print(table, row.names = FALSE) + cat(paste(lines, collapse = "\n"), "\n") + list(scores = scores, percent = values[3]) +} +fit <- function(f) { + invoke(f) + read_results(f) +} + +baseline <- NULL +check("Reference fixture: m=4, geometry, deviance, labels and output schemas", function() { + f <- fixture(ids = labels, blank_header = TRUE) + result <- fit(f) + # Reference score provenance is documented above the binary fixture. + # Comparing distances permits one global orthogonal change of plotted axes, + # but does not permit arbitrary sign changes on individual observations. + expected <- cbind( + c(4.96756989310632, -7.42875819587666, 12.7488840407735, -11.2979872320273), + c(-7.80927209388169, 7.21294742051829, 1.711517188498, -5.51380214217161) + ) + near(dist(result$scores), dist(expected), 1e-3, "Reference distances") + # Convert the recorded m=4 proportion to a percentage. + near(result$percent, 91.92537, 1e-3, "Reference deviance percentage") + baseline <<- result +}) +check("Repeatability in a fresh R process; named CSV label header", function() { + need(!is.null(baseline), "Reference fixture did not pass") + result <- fit(fixture(ids = labels)) + near(dist(result$scores), dist(baseline$scores), 1e-6, "Repeated distances") + near(result$percent, baseline$percent, 1e-6, "Repeated deviance percentage") +}) +check("Second known configuration: fixed m=6", function() { + result <- fit(fixture(m = "6")) + # Convert the recorded m=6 proportion to a percentage. + near(result$percent, 97.6137532256003, 1e-3, "m=6 reference deviance percentage") +}) +check("Boundary-sized valid input: three runs and three features", function() { + fit(fixture(diag(3))) +}) +check("Constant columns with otherwise informative data", function() { + fit(fixture(cbind(x, 1, 0))) +}) +check("Duplicate profiles retained when at least three distinct profiles exist", function() { + result <- fit(fixture(rbind(x, x[1, , drop = FALSE]))) + near(result$scores[1, ], result$scores[5, ], 1e-6, "Identical-profile scores") +}) +check("Finite positive fractional m", function() { + fit(fixture(m = "4.5")) +}) + +check("Reject two runs before reaching the decomposition backend", function() { + invoke(fixture(x[1:2, , drop = FALSE]), "at least 3 runs and 3 binary edge features") +}) +for (count in 1:2) { + check(paste("Reject", count, "feature column(s)"), function() { + invoke(fixture(x[, seq_len(count), drop = FALSE]), + "at least 3 runs and 3 binary edge features") + }) +} +check("Reject identical profiles instead of exporting infinite deviance", function() { + invoke(fixture(matrix(1, nrow = 4, ncol = 5)), "at least 3 distinct") +}) +check("Reject only two distinct profiles despite four runs", function() { + invoke(fixture(x[c(1, 2, 1, 2), , drop = FALSE]), "at least 3 distinct") +}) +check("Reject duplicate run identifiers", function() { + invoke(fixture(ids = rep("same", 4)), "Run labels must be nonempty and unique") +}) +check("Reject blank run identifiers", function() { + invoke(fixture(ids = c(" ", "b", "c", "d")), "Run labels must be nonempty and unique") +}) +for (value in c("1", "3", "2.9")) { + check(paste("Reject k =", value), function() { + invoke(fixture(k = value), "k must be exactly 2") + }) +} +for (value in c("0", "-1", "Inf", "bad")) { + check(paste("Reject m =", value), function() { + invoke(fixture(m = value), "m must be finite and positive") + }) +} +for (value in c("", "NA", "Inf", "bad", "2", "0.5")) { + check(paste("Reject invalid feature value:", dQuote(value)), function() { + mat <- x + mat[1, 1] <- value + invoke(fixture(mat), "Edge features must be complete, finite numeric zeros or ones") + }) +} +check("Reject old four-argument interface without producing outputs", function() { + f <- fixture() + invoke(f, "Usage:", f$args[1:4]) +}) + +cat("\n=== RESULT:", passed, "passed;", failed, "failed ===\n") +unlink(work, recursive = TRUE) +quit(save = "no", status = if (failed == 0L) 0L else 1L) +''' + + +def main() -> int: + parser = argparse.ArgumentParser( + description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter + ) + parser.add_argument("--image", default="reedcompbio/lpca:v1", + help="Local image to test (default: reedcompbio/lpca:v1)") + parser.add_argument("--log", type=Path, + help="Also save output to a new UTF-8 file; never overwrite") + parser.add_argument("--timeout", type=int, default=600, + help="Total container timeout in seconds (default: 600)") + args = parser.parse_args() + if args.timeout <= 0: + parser.error("--timeout must be positive") + source = Path(__file__).resolve().with_name("run_lpca.R") + if not source.is_file(): + parser.error("Place test_container.py beside run_lpca.R in docker-wrappers/LPCA") + docker = shutil.which("docker") + if docker is None: + parser.error("Docker CLI was not found on PATH") + + env = os.environ.copy() + env.update({"MSYS_NO_PATHCONV": "1", "MSYS2_ARG_CONV_EXCL": "*"}) + normalized_source = source.read_bytes().replace(b"\r\n", b"\n") + expected_md5 = hashlib.md5(normalized_source, usedforsecurity=False).hexdigest() + name = "spras-lpca-tests-" + uuid.uuid4().hex[:12] + log_path = args.log.expanduser() if args.log is not None else None + # Logging is optional; an existing file is never overwritten. + log_context = (log_path.open("x", encoding="utf-8", newline="\n") + if log_path is not None else nullcontext(None)) + with log_context as log: + def emit(text: str) -> None: + if not text.endswith("\n"): + text += "\n" + print(text, end="", flush=True) + if log is not None: + log.write(text) + log.flush() + + if log_path is not None: + emit(f"Results log: {log_path}") + emit(f"Local wrapper: {source}") + emit(f"Local wrapper MD5 (LF normalized): {expected_md5}") + try: + inspected = subprocess.run( + [docker, "image", "inspect", args.image], env=env, + stdout=subprocess.PIPE, stderr=subprocess.PIPE, + text=True, encoding="utf-8", errors="replace", timeout=30, + ) + if inspected.returncode: + emit(inspected.stderr) + emit("Image inspection failed. Build the selected image first; no image is pulled.") + return 1 + image = json.loads(inspected.stdout)[0] + emit(f"Image tag: {args.image}\nImage ID: {image['Id']}") + emit(f"Image platform: {image.get('Os')} / {image.get('Architecture')}") + # Run the inspected ID, not the mutable tag, to identify exactly what was tested. + command = [docker, "run", "--name", name, "--rm", "--pull=never", + "--network=none", "-i", "--entrypoint", "Rscript", + image["Id"], "--vanilla", "/dev/stdin"] + payload = f'expected_md5 <- "{expected_md5}"\n' + R_TESTS + emit("Running container CLI tests. Their output is buffered until completion.") + try: + result = subprocess.run( + command, input=payload, env=env, stdout=subprocess.PIPE, + stderr=subprocess.STDOUT, text=True, encoding="utf-8", + errors="replace", timeout=args.timeout, + ) + finally: + # On timeout/interrupt, remove only the uniquely named test container. + # After an ordinary --rm exit, this is a harmless no-op. + try: + subprocess.run([docker, "rm", "--force", name], env=env, + stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, + timeout=30, check=False) + except (OSError, subprocess.SubprocessError) as cleanup_error: + emit(f"WARNING: Could not confirm cleanup of {name}: {cleanup_error}") + emit(result.stdout) + emit(f"Container exit status: {result.returncode}") + complete = "=== RESULT:" in result.stdout + if not complete: + emit("The test suite did not reach its summary; this run is not passing.") + return 0 if result.returncode == 0 and complete else 1 + except subprocess.TimeoutExpired as exc: + partial = exc.stdout or b"" + emit(partial.decode("utf-8", "replace") if isinstance(partial, bytes) else partial) + emit(f"ERROR: Command timed out after {exc.timeout} seconds; the run is incomplete.") + return 1 + except KeyboardInterrupt: + emit("Interrupted; the run is incomplete.") + return 130 + except (OSError, ValueError, KeyError, IndexError) as exc: + emit(f"ERROR: {exc}") + return 1 + + +if __name__ == "__main__": + try: + sys.exit(main()) + except FileExistsError as exc: + print(f"ERROR: Log already exists: {exc.filename}\n" + "Choose a different --log path, or omit --log for terminal output only.", + file=sys.stderr) + sys.exit(1) + except OSError as exc: + print(f"ERROR: {exc}", file=sys.stderr) + sys.exit(1) From e1c0be27dd7a4398569feb4eef303c726613f4a0 Mon Sep 17 00:00:00 2001 From: Anthony Gitter Date: Thu, 1 Oct 2026 13:18:16 -0500 Subject: [PATCH 23/29] Refactor LPCA wrapper and tests --- docker-wrappers/LPCA/README.md | 2 +- docker-wrappers/LPCA/run_lpca.R | 2 +- spras/analysis/lpca.py | 197 +++++++++++---------------- spras/analysis/ml.py | 2 - test/analysis/test_lpca.py | 234 ++++++++++++++++---------------- 5 files changed, 195 insertions(+), 242 deletions(-) diff --git a/docker-wrappers/LPCA/README.md b/docker-wrappers/LPCA/README.md index 808cf6a17..c21d5b1ac 100644 --- a/docker-wrappers/LPCA/README.md +++ b/docker-wrappers/LPCA/README.md @@ -74,7 +74,7 @@ For the SPRAS default registry to resolve the image, it must be published as `docker.io/reedcompbio/lpca:v1`, which requires access to the `reedcompbio` Docker Hub organization: - docker build -t reedcompbio/lpca:v1 docker-wrappers/lpca/ + docker build -t reedcompbio/lpca:v1 docker-wrappers/LPCA/ docker push reedcompbio/lpca:v1 ## Testing diff --git a/docker-wrappers/LPCA/run_lpca.R b/docker-wrappers/LPCA/run_lpca.R index ce62c98f5..42e6c0b72 100644 --- a/docker-wrappers/LPCA/run_lpca.R +++ b/docker-wrappers/LPCA/run_lpca.R @@ -77,7 +77,7 @@ if (abs(tail(changes, 1L)) >= conv_criteria) { stop(sprintf("LPCA did not converge within %d iterations.", max_iters)) } -# Python adds the centroid and formats the public coordinate table. +# Python formats the public coordinate table. score_table <- data.frame(datapoint_labels = row_labels, PC1 = scores[, 1], PC2 = scores[, 2]) summary <- c( diff --git a/spras/analysis/lpca.py b/spras/analysis/lpca.py index 68728388c..78160b00a 100644 --- a/spras/analysis/lpca.py +++ b/spras/analysis/lpca.py @@ -1,141 +1,100 @@ -# Logistic PCA analysis for SPRAS -# Runs LPCA on the binary edge x algorithm matrix produced by summarize_networks. -# Configured through the analysis.lpca block (k, m, cv, transpose). +"""Fixed-m, two-component LPCA of pathway graphs from one or more algorithms.""" +from os import PathLike from pathlib import Path +from tempfile import TemporaryDirectory import matplotlib.pyplot as plt import pandas as pd import seaborn as sns +from adjustText import adjust_text -from spras.analysis.ml import create_palette +from spras.analysis.ml import DPI, create_palette from spras.config.container_schema import ProcessedContainerSettings from spras.containers import prepare_volume, run_container_and_log +from spras.util import make_required_dirs -# Published as docker.io/reedcompbio/lpca:v1. Only the suffix is given here; -# the registry prefix is resolved from the container settings. +# The registry prefix is resolved from the container settings. LPCA_CONTAINER_SUFFIX = 'lpca:v1' LPCA_WORK_DIR = '/app' def run_lpca( dataframe: pd.DataFrame, - output_scores: str, - output_matrix: str, + output_png: str | PathLike, + output_deviance: str | PathLike, + output_coord: str | PathLike, + output_matrix: str | PathLike, k: int = 2, m: float = 6, - cv: bool = False, - container_settings=None + labels: bool = True, + container_settings: ProcessedContainerSettings | None = None, ) -> None: - """ - Runs Logistic PCA on the binary edge x algorithm matrix built from SPRAS - algorithm output files. - - @param dataframe: binary dataframe of edge comparison between algorithms from summarize_networks - @param output_matrix: path to write the binary matrix CSV (used as Docker volume mount) - @param output_scores: path to write the LPCA PC scores CSV - @param k: number of principal components (default 2) - @param m: fixed logisticPCA tuning parameter, used when cv is False - @param cv: if True, determine m by cross-validation; if False, use the - fixed m directly (default False) - @param container_settings: configure the container runtime (Docker or Singularity) + """Fit and plot the graphs in the edges-by-runs summarize_networks dataframe. - Note: KDE-based parameter selection (used by PCA) always uses PCA scores, even when LPCA - is also enabled. KDE integration with LPCA may be added in a future update. + The caller selects runs from one algorithm or across algorithms. Containerized R validates + the inputs and numerical fit. Python writes the input matrix, calls R, and + produces a plot and tab-separated coordinates with one row per input graph. + The deviance summary is written by R; its raw scores are temporary. """ - if not container_settings: + if container_settings is None: container_settings = ProcessedContainerSettings() - - # Step 1: build the binary edge x algorithm matrix - matrix = dataframe - # Transpose so runs are the observations, mirroring the classic PCA analysis - matrix = matrix.T - print(f'LPCA: Matrix shape: {matrix.shape}') - - # Step 2: write the binary matrix for Docker volume mount - output_dir = Path(output_scores).parent - output_dir.mkdir(parents=True, exist_ok=True) - Path(output_matrix).parent.mkdir(parents=True, exist_ok=True) - matrix_path = output_matrix - matrix.to_csv(matrix_path) - - # Step 3: mount the matrix and the scores output - volumes = [] - bind_path, mapped_matrix = prepare_volume(matrix_path, LPCA_WORK_DIR, container_settings) - volumes.append(bind_path) - bind_path, mapped_scores = prepare_volume(output_scores, LPCA_WORK_DIR, container_settings) - volumes.append(bind_path) - - # Step 4: choose m, optionally via cross-validation - if cv: - algo_name = Path(output_scores).name.replace('-lpca-scores.csv', '') - cv_output_path = str(output_dir / f'{algo_name}-lpca_cv_result.csv') - bind_path, mapped_cv_output = prepare_volume(cv_output_path, LPCA_WORK_DIR, container_settings) - volumes.append(bind_path) - - print(f'LPCA: Running cross-validation with k={k}...') - command_cv = ['Rscript', '/app/run_cv.R', mapped_matrix, mapped_cv_output, str(k)] - run_container_and_log('LPCA-CV', LPCA_CONTAINER_SUFFIX, command_cv, volumes, - LPCA_WORK_DIR, None, container_settings) - - if not Path(cv_output_path).exists(): - raise FileNotFoundError( - f'LPCA: Cross-validation output not found at {cv_output_path}. ' - 'Check the LPCA Docker container logs for errors.' - ) - m_used = pd.read_csv(cv_output_path)['best_m'][0] - print(f'LPCA: Best m found by CV: {m_used}') - else: - m_used = m - print(f'LPCA: Using fixed m={m_used}') - - # Step 5: run LPCA with the chosen k and m - print(f'LPCA: Running LPCA with k={k}, m={m_used}...') - command_lpca = ['Rscript', '/app/run_lpca.R', mapped_matrix, mapped_scores, str(k), str(m_used)] - run_container_and_log('LPCA', LPCA_CONTAINER_SUFFIX, command_lpca, volumes, - LPCA_WORK_DIR, None, container_settings) - - print(f'LPCA: Done! Scores saved to {output_scores}') - -def plot_lpca(scores_file: str, output_png: str, output_coord: str, labels: bool = True) -> None: - """ - Creates a scatterplot of the first two LPCA principal components. - @param scores_file: path to the LPCA scores CSV file - @param output_png: path to save the scatterplot PNG - @param output_coord: path to save the PC coordinates - @param labels: if True, adds algorithm labels to the plot - """ - scores = pd.read_csv(scores_file, index_col=0) - - if scores.empty: - print('LPCA: Scores file is empty, skipping plot.') - return - - # Extract algorithm names from the index - column_names = [idx.split('-')[-3] if '-' in idx else idx for idx in scores.index] - - X = scores.values - fig, ax = plt.subplots(figsize=(10, 8)) - - label_color_map = create_palette(column_names) - sns.scatterplot(x=X[:, 0], y=X[:, 1], hue=column_names, palette=label_color_map, s=70, ax=ax) - - if labels: - for i, label in enumerate(scores.index): - ax.annotate(label, (X[i, 0], X[i, 1]), fontsize=6, alpha=0.7) - - ax.set_xlabel('PC1') - ax.set_ylabel('PC2') - ax.set_title('Logistic PCA') - - plt.tight_layout() - - # Save PNG - Path(output_png).parent.mkdir(parents=True, exist_ok=True) - plt.savefig(output_png, dpi=200) - plt.close() - - # Save coordinates - coord_df = pd.DataFrame(X, columns=['PC1', 'PC2'], index=scores.index) - coord_df.to_csv(output_coord) - print(f'LPCA: Plot saved to {output_png}') + for path in (output_png, output_deviance, output_coord, output_matrix): + make_required_dirs(path) + output_dir = Path(output_png).parent + matrix = dataframe.T + matrix.to_csv(output_matrix, index_label='datapoint_labels') + print(f'LPCA: Matrix shape: {matrix.shape}; k={k}, m={m}') + + with TemporaryDirectory(prefix='.lpca-', dir=output_dir) as temp_dir: + scores_file = Path(temp_dir) / 'scores.csv' + matrix_volume, mapped_matrix = prepare_volume(output_matrix, LPCA_WORK_DIR, container_settings) + scores_volume, mapped_scores = prepare_volume(scores_file, LPCA_WORK_DIR, container_settings) + deviance_volume, mapped_deviance = prepare_volume(output_deviance, LPCA_WORK_DIR, container_settings) + volumes = [matrix_volume, scores_volume, deviance_volume] + command = ['Rscript', '/app/run_lpca.R', mapped_matrix, mapped_scores, + str(k), str(m), mapped_deviance] + run_container_and_log('LPCA', LPCA_CONTAINER_SUFFIX, command, volumes, + LPCA_WORK_DIR, output_dir, container_settings) + + # Preserve identifiers and require one pair of scores per input graph. + scores = pd.read_csv(scores_file, index_col='datapoint_labels', + dtype={'datapoint_labels': str}, keep_default_na=False) + if (list(scores.columns) != ['PC1', 'PC2'] or + scores.index.tolist() != matrix.index.tolist()): + raise ValueError('LPCA scores must contain PC1/PC2 and the input run labels in order.') + + with open(output_deviance) as summary_file: + summary = dict(line.strip().split(': ', 1) for line in summary_file) + percent_deviance = float(summary['percent_deviance_explained']) + plot_lpca(scores, output_png, output_coord, percent_deviance, labels=labels) + + +def plot_lpca( + scores: pd.DataFrame, + output_png: str | PathLike, + output_coord: str | PathLike, + percent_deviance: float, + labels: bool = True, +) -> None: + """Plot run scores and save their coordinates; output directories must exist.""" + scores.rename_axis('datapoint_labels').round(8).to_csv(output_coord, sep='\t') + column_names = [name.split('-')[-3] if name.count('-') >= 2 else name + for name in scores.index] + points = scores.to_numpy() + fig, ax = plt.subplots(figsize=(10, 7)) + try: + sns.scatterplot(x=points[:, 0], y=points[:, 1], hue=column_names, + palette=create_palette(column_names), s=70, ax=ax) + ax.set_xlabel('PC1') + ax.set_ylabel('PC2') + ax.set_title(f'Logistic PCA ({percent_deviance:.1f}% deviance explained)') + fig.tight_layout() + if labels: + texts = [ax.text(x, y, name, size=10) + for (x, y), name in zip(points, scores.index, strict=True)] + adjust_text(texts, ax=ax, force_points=(5.0, 5.0), + arrowprops=dict(arrowstyle='->', color='black')) + fig.savefig(output_png, dpi=DPI) + finally: + plt.close(fig) diff --git a/spras/analysis/ml.py b/spras/analysis/ml.py index 459852770..7e17b1e00 100644 --- a/spras/analysis/ml.py +++ b/spras/analysis/ml.py @@ -156,8 +156,6 @@ def pca(dataframe: pd.DataFrame, output_png: str | PathLike, output_var: str | P # center binary data by subtracting the column-wise mean # allows PCA to focus on edge inclusion patterns across runs rather than raw output volume. - # TODO: consider replacing PCA with LPCA for binary data https://github.com/Reed-CompBio/spras/issues/271 - # LPCA is now available as an alternative analysis (analysis.lpca in config) scaler = StandardScaler(with_std=False) scaler.fit(X) # compute mean inclusion rate per edge X_scaled = scaler.transform(X) diff --git a/test/analysis/test_lpca.py b/test/analysis/test_lpca.py index 75f84f451..42ff48285 100644 --- a/test/analysis/test_lpca.py +++ b/test/analysis/test_lpca.py @@ -1,125 +1,121 @@ +"""Test SPRAS integration; numerical edge cases belong in docker-wrappers/LPCA/test_container.py.""" + +import shutil from pathlib import Path +from unittest.mock import Mock +import matplotlib.pyplot as plt +import numpy as np import pandas as pd +import pytest +from scipy.spatial.distance import pdist -import spras.config.config as config -from spras.analysis.lpca import plot_lpca, run_lpca +from spras.analysis import lpca from spras.analysis.ml import summarize_networks -config.init_from_file("config/config.yaml") - -TEST_DIR = Path('test/analysis/') -OUT_DIR = TEST_DIR / 'output' - -INPUT_FILES = [ - 'test/analysis/input/lpca/pathway-params-1.txt', - 'test/analysis/input/lpca/pathway-params-2.txt', - 'test/analysis/input/lpca/pathway-params-3.txt', - 'test/analysis/input/lpca/pathway-params-4.txt', -] - -class TestLpca: - """ - Run Logistic PCA (LPCA) analysis tests - """ - @classmethod - def setup_class(cls): - OUT_DIR.mkdir(parents=True, exist_ok=True) - - def test_lpca_output_exists(self): - """Test that LPCA produces an output scores file""" - out_path = OUT_DIR / 'lpca-scores.csv' - matrix_path = OUT_DIR / 'lpca-binary-matrix.csv' - out_path.unlink(missing_ok=True) - - summary_df = summarize_networks(INPUT_FILES) - run_lpca( - dataframe=summary_df, - output_scores=str(out_path), - output_matrix=str(matrix_path), - k=2, - m=4, - cv=False, - ) - - assert out_path.exists(), "LPCA scores file was not created" - - def test_lpca_output_shape(self): - """Test that LPCA scores have shape (runs x k)""" - out_path = OUT_DIR / 'lpca-scores-shape.csv' - matrix_path = OUT_DIR / 'lpca-binary-matrix-shape.csv' - out_path.unlink(missing_ok=True) - - summary_df = summarize_networks(INPUT_FILES) - run_lpca( - dataframe=summary_df, - output_scores=str(out_path), - output_matrix=str(matrix_path), - k=2, - m=4, - cv=False, - ) - - scores = pd.read_csv(out_path, index_col=0) - assert scores.shape[1] == 2, f"Expected 2 PC columns, got {scores.shape[1]}" - assert scores.shape[0] == len(INPUT_FILES), \ - f"Expected {len(INPUT_FILES)} rows, got {scores.shape[0]}" - - def test_lpca_plot_output(self): - """Test that LPCA plot and coordinates files are created""" - scores_path = OUT_DIR / 'lpca-scores-plot.csv' - matrix_path = OUT_DIR / 'lpca-binary-matrix-plot.csv' - png_path = OUT_DIR / 'lpca-plot.png' - coord_path = OUT_DIR / 'lpca-coordinates.txt' - - summary_df = summarize_networks(INPUT_FILES) - run_lpca( - dataframe=summary_df, - output_scores=str(scores_path), - output_matrix=str(matrix_path), - k=2, - m=4, - cv=False, - ) - plot_lpca(str(scores_path), str(png_path), str(coord_path)) - - assert png_path.exists(), "LPCA plot PNG was not created" - assert coord_path.exists(), "LPCA coordinates file was not created" - - def test_lpca_known_output(self): - """Test that LPCA produces the expected scores for known inputs. - - Compared in absolute value because LPCA components, like PCA, are only - defined up to a sign (an axis can flip without changing the result). - """ - out_path = OUT_DIR / 'lpca-scores-known.csv' - matrix_path = OUT_DIR / 'lpca-binary-matrix-known.csv' - out_path.unlink(missing_ok=True) - - summary_df = summarize_networks(INPUT_FILES) - run_lpca( - dataframe=summary_df, - output_scores=str(out_path), - output_matrix=str(matrix_path), - k=2, - m=4, - cv=False, - ) - - scores = pd.read_csv(out_path, index_col=0) - - expected = pd.DataFrame({ - 'V1': [4.96756989310632, -7.42875819587666, 12.7488840407735, -11.2979872320273], - 'V2': [-7.80927209388169, 7.21294742051829, 1.711517188498, -5.51380214217161], - }) - - assert scores.shape == expected.shape, \ - f"Expected shape {expected.shape}, got {scores.shape}" - - # Compare in absolute value to stay robust to sign flips (sign non-identifiability) - pd.testing.assert_frame_equal( - scores.reset_index(drop=True).abs(), - expected.abs(), - check_dtype=False, - atol=1e-4, - ) +INPUT_DIR = Path(__file__).parent / 'input' / 'lpca' +# Existing four-graph fixture, logisticPCA 0.2, k=2, m=4. Distances allow a +# common change of axis orientation, not independent sign changes per graph. +REFERENCE_SCORES = np.array([ + [4.96756989310632, -7.80927209388169], + [-7.42875819587666, 7.21294742051829], + [12.7488840407735, 1.711517188498], + [-11.2979872320273, -5.51380214217161], +]) +REFERENCE_DEVIANCE = 91.92536834497237 + + +@pytest.fixture +def outputs(tmp_path): + return { + 'output_png': tmp_path / 'lpca.png', + 'output_deviance': tmp_path / 'lpca-deviance.txt', + 'output_coord': tmp_path / 'lpca-coordinates.txt', + 'output_matrix': tmp_path / 'lpca-binary-matrix.csv', + } + + +@pytest.mark.parametrize('per_algorithm', [False, True], ids=['all_algorithms', 'per_algorithm']) +def test_lpca_container(tmp_path, outputs, per_algorithm): + algorithms = ['allpairs'] * 4 if per_algorithm else ['allpairs', 'allpairs', 'meo', 'meo'] + paths = [] + for i, algorithm in enumerate(algorithms, start=1): + # summarize_networks derives run identifiers from parent directories. + path = tmp_path / 'input graphs' / f'dataset-{algorithm}-params-{i:07d}' / 'pathway.txt' + path.parent.mkdir(parents=True) + shutil.copyfile(INPUT_DIR / f'pathway-params-{i}.txt', path) + paths.append(path) + networks = summarize_networks(paths) + lpca.run_lpca(networks, **outputs, m=4, labels=not per_algorithm) + + assert all(path.is_file() for path in outputs.values()) + assert outputs['output_png'].read_bytes().startswith(b'\x89PNG\r\n\x1a\n') + matrix = pd.read_csv(outputs['output_matrix'], index_col='datapoint_labels') + pd.testing.assert_frame_equal(matrix, networks.T.rename_axis('datapoint_labels')) + coordinates = pd.read_csv(outputs['output_coord'], sep='\t') + assert list(coordinates.columns) == ['datapoint_labels', 'PC1', 'PC2'] + assert coordinates['datapoint_labels'].tolist() == list(networks.columns) + points = coordinates[['PC1', 'PC2']].to_numpy() + assert points.shape == (4, 2) + assert np.isfinite(points).all() + np.testing.assert_allclose(pdist(points), pdist(REFERENCE_SCORES), atol=1e-3, rtol=0) + summary = dict(line.split(': ', 1) + for line in outputs['output_deviance'].read_text().splitlines()) + assert float(summary['components']) == 2 + assert float(summary['m']) == 4 + assert float(summary['percent_deviance_explained']) == pytest.approx( + REFERENCE_DEVIANCE, abs=1e-3, rel=0) + assert not list(tmp_path.glob('.lpca-*')) + + +@pytest.mark.parametrize('labels', [True, False]) +def test_plot_lpca(outputs, monkeypatch, labels): + names = ['001', 'NA', 'dataset-meo-params-CCCCCCC', 'run-four'] + scores = pd.DataFrame(REFERENCE_SCORES, columns=['PC1', 'PC2'], index=names) + # Mock records calls without moving labels, so we can inspect what our + # plotting code passes to the label-placement function. + adjust = Mock() + # Replace the name used by lpca only for this test. The monkeypatch + # fixture restores the original function afterward, even on failure. + monkeypatch.setattr(lpca, 'adjust_text', adjust) + lpca.plot_lpca(scores, outputs['output_png'], outputs['output_coord'], + REFERENCE_DEVIANCE, labels=labels) + + coordinates = pd.read_csv(outputs['output_coord'], sep='\t', + dtype={'datapoint_labels': str}, keep_default_na=False) + assert list(coordinates.columns) == ['datapoint_labels', 'PC1', 'PC2'] + assert coordinates['datapoint_labels'].tolist() == names + np.testing.assert_allclose(coordinates[['PC1', 'PC2']], REFERENCE_SCORES, + atol=1e-8, rtol=0) + assert outputs['output_png'].read_bytes().startswith(b'\x89PNG\r\n\x1a\n') + if labels: + adjust.assert_called_once() + # call_args stores the last call; kwargs contains named arguments. + ax = adjust.call_args.kwargs['ax'] + assert len(ax.collections) == 1 # Only graph coordinates; no extra plotted points. + np.testing.assert_allclose(ax.collections[0].get_offsets(), REFERENCE_SCORES) + assert ax.get_xlabel() == 'PC1' + assert ax.get_ylabel() == 'PC2' + assert ax.get_title() == 'Logistic PCA (91.9% deviance explained)' + # args[0] is the first positional argument: the list of text labels. + assert [text.get_text() for text in adjust.call_args.args[0]] == names + assert not plt.fignum_exists(ax.figure.number) + else: + adjust.assert_not_called() + + +def test_container_failure_propagates(tmp_path, outputs, monkeypatch): + # side_effect raises this error when the mock is called, simulating a + # failed container without starting Docker. The call is still recorded. + failure = Mock(side_effect=RuntimeError('container failed')) + # Replace lpca's imported function; pytest restores it after this test. + monkeypatch.setattr(lpca, 'run_container_and_log', failure) + networks = pd.DataFrame(np.eye(3), columns=['a', 'b', 'c']) + with pytest.raises(RuntimeError, match='container failed'): + lpca.run_lpca(networks, **outputs) + failure.assert_called_once() + # The sixth positional argument is the container helper's output directory. + assert failure.call_args.args[5] == tmp_path + assert not outputs['output_png'].exists() + assert not outputs['output_coord'].exists() + assert not list(tmp_path.glob('.lpca-*')) From e2c91d2cc7557568e9f24d6bfffc88883c2bc850 Mon Sep 17 00:00:00 2001 From: Anthony Gitter Date: Thu, 1 Oct 2026 16:50:15 -0500 Subject: [PATCH 24/29] Add tests, refactor config and Snakefile --- Snakefile | 64 +++++++------- config/config.yaml | 61 +++++++------- config/egfr.yaml | 6 ++ spras/config/schema.py | 10 +-- test/test_config.py | 28 +++++++ test/test_lpca_workflow.py | 167 +++++++++++++++++++++++++++++++++++++ 6 files changed, 269 insertions(+), 67 deletions(-) create mode 100644 test/test_lpca_workflow.py diff --git a/Snakefile b/Snakefile index ff80ba7ae..a2d232a22 100644 --- a/Snakefile +++ b/Snakefile @@ -4,7 +4,7 @@ import shutil import yaml from spras.dataset import Dataset from spras.evaluation import Evaluation -from spras.analysis import ml, summary, cytoscape +from spras.analysis import ml, summary, cytoscape, lpca from spras.config.revision import detach_spras_revision import spras.config.config as _config @@ -24,6 +24,7 @@ _config.init_global(config) out_dir = _config.config.out_dir algorithm_params = _config.config.algorithm_params pca_params = _config.config.pca_params +lpca_params = _config.config.lpca_params hac_params = _config.config.hac_params container_settings = _config.config.container_settings include_aggregate_algo_eval = _config.config.analysis_include_evaluation_aggregate_algo @@ -43,6 +44,8 @@ def algo_has_mult_param_combos(algo): return len(algorithm_params.get(algo, {})) > 1 algorithms_mult_param_combos = [algo for algo in algorithms if algo_has_mult_param_combos(algo)] +# LPCA requires at least three runs; keep the PCA eligibility threshold unchanged. +algorithms_lpca = [algo for algo in algorithms if len(algorithm_params[algo]) >= 3] # Get the parameter dictionary for the specified # algorithm and parameter combination hash @@ -94,15 +97,15 @@ def make_final_input(wildcards): if _config.config.analysis_include_lpca: final_input.extend(expand('{out_dir}{sep}{dataset}-ml{sep}lpca.png',out_dir=out_dir, sep=SEP, dataset=dataset_labels)) - final_input.extend(expand('{out_dir}{sep}{dataset}-ml{sep}lpca-scores.csv',out_dir=out_dir, sep=SEP, dataset=dataset_labels)) + final_input.extend(expand('{out_dir}{sep}{dataset}-ml{sep}lpca-deviance.txt',out_dir=out_dir, sep=SEP, dataset=dataset_labels)) final_input.extend(expand('{out_dir}{sep}{dataset}-ml{sep}lpca-coordinates.txt',out_dir=out_dir, sep=SEP, dataset=dataset_labels)) final_input.extend(expand('{out_dir}{sep}{dataset}-ml{sep}lpca-binary-matrix.csv',out_dir=out_dir, sep=SEP, dataset=dataset_labels)) if _config.config.analysis_include_lpca_aggregate_algo: - final_input.extend(expand('{out_dir}{sep}{dataset}-ml{sep}{algorithm}-lpca-scores.csv',out_dir=out_dir, sep=SEP, dataset=dataset_labels, algorithm=algorithms_mult_param_combos)) - final_input.extend(expand('{out_dir}{sep}{dataset}-ml{sep}{algorithm}-lpca.png',out_dir=out_dir, sep=SEP,dataset=dataset_labels,algorithm=algorithms_mult_param_combos)) - final_input.extend(expand('{out_dir}{sep}{dataset}-ml{sep}{algorithm}-lpca-coordinates.txt',out_dir=out_dir, sep=SEP, dataset=dataset_labels,algorithm=algorithms_mult_param_combos)) - final_input.extend(expand('{out_dir}{sep}{dataset}-ml{sep}{algorithm}-lpca-binary-matrix.csv',out_dir=out_dir, sep=SEP, dataset=dataset_labels,algorithm=algorithms_mult_param_combos)) + final_input.extend(expand('{out_dir}{sep}{dataset}-ml{sep}{algorithm}-lpca-deviance.txt',out_dir=out_dir, sep=SEP, dataset=dataset_labels, algorithm=algorithms_lpca)) + final_input.extend(expand('{out_dir}{sep}{dataset}-ml{sep}{algorithm}-lpca.png',out_dir=out_dir, sep=SEP,dataset=dataset_labels,algorithm=algorithms_lpca)) + final_input.extend(expand('{out_dir}{sep}{dataset}-ml{sep}{algorithm}-lpca-coordinates.txt',out_dir=out_dir, sep=SEP, dataset=dataset_labels,algorithm=algorithms_lpca)) + final_input.extend(expand('{out_dir}{sep}{dataset}-ml{sep}{algorithm}-lpca-binary-matrix.csv',out_dir=out_dir, sep=SEP, dataset=dataset_labels,algorithm=algorithms_lpca)) if _config.config.analysis_include_ml_aggregate_algo: final_input.extend(expand('{out_dir}{sep}{dataset}-ml{sep}{algorithm}-pca.png',out_dir=out_dir,sep=SEP,dataset=dataset_labels,algorithm=algorithms_mult_param_combos)) @@ -416,57 +419,58 @@ rule ml_analysis_aggregate_algo: ml.hac_horizontal(summary_df, output.hac_image_horizontal, output.hac_clusters_horizontal, **hac_params) ml.pca(summary_df, output.pca_image, output.pca_variance, output.pca_coordinates, **pca_params) +# Track fit and plot settings so changes schedule a new fit. rule lpca_analysis_all: input: pathways = expand('{out_dir}{sep}{{dataset}}-{algorithm_params}{sep}pathway.txt', out_dir=out_dir, sep=SEP, algorithm_params=algorithms_with_params) output: - lpca_scores = SEP.join([out_dir, '{dataset}-ml', 'lpca-scores.csv']), lpca_png = SEP.join([out_dir, '{dataset}-ml', 'lpca.png']), + lpca_deviance = SEP.join([out_dir, '{dataset}-ml', 'lpca-deviance.txt']), lpca_coord = SEP.join([out_dir, '{dataset}-ml', 'lpca-coordinates.txt']), lpca_matrix = SEP.join([out_dir, '{dataset}-ml', 'lpca-binary-matrix.csv']) + params: + k = lpca_params.k, + m = lpca_params.m, + labels = lpca_params.labels, run: - from spras.analysis import lpca summary_df = ml.summarize_networks(input.pathways) lpca.run_lpca( summary_df, - output.lpca_scores, - output.lpca_matrix, - k=_config.config.lpca_params.k, - m=_config.config.lpca_params.m, - cv=_config.config.lpca_params.cv, + output_png=output.lpca_png, + output_deviance=output.lpca_deviance, + output_coord=output.lpca_coord, + output_matrix=output.lpca_matrix, + k=params.k, + m=params.m, + labels=params.labels, container_settings=container_settings ) - lpca.plot_lpca( - output.lpca_scores, - output.lpca_png, - output.lpca_coord - ) rule lpca_analysis_aggregate_algo: input: pathways = collect_pathways_per_algo output: - lpca_scores = SEP.join([out_dir, '{dataset}-ml', '{algorithm}-lpca-scores.csv']), lpca_png = SEP.join([out_dir, '{dataset}-ml', '{algorithm}-lpca.png']), + lpca_deviance = SEP.join([out_dir, '{dataset}-ml', '{algorithm}-lpca-deviance.txt']), lpca_coord = SEP.join([out_dir, '{dataset}-ml', '{algorithm}-lpca-coordinates.txt']), lpca_matrix = SEP.join([out_dir, '{dataset}-ml', '{algorithm}-lpca-binary-matrix.csv']) + params: + k = lpca_params.k, + m = lpca_params.m, + labels = lpca_params.labels, run: - from spras.analysis import lpca summary_df = ml.summarize_networks(input.pathways) lpca.run_lpca( summary_df, - output.lpca_scores, - output.lpca_matrix, - k=_config.config.lpca_params.k, - m=_config.config.lpca_params.m, - cv=_config.config.lpca_params.cv, + output_png=output.lpca_png, + output_deviance=output.lpca_deviance, + output_coord=output.lpca_coord, + output_matrix=output.lpca_matrix, + k=params.k, + m=params.m, + labels=params.labels, container_settings=container_settings ) - lpca.plot_lpca( - output.lpca_scores, - output.lpca_png, - output.lpca_coord - ) # Ensemble the output pathways for each dataset per algorithm rule ensemble_per_algo: diff --git a/config/config.yaml b/config/config.yaml index 10bec536b..b634e7c46 100644 --- a/config/config.yaml +++ b/config/config.yaml @@ -49,30 +49,29 @@ containers: # # requirements = versionGE(split(Target.CondorVersion)[1], "24.8.0") && (isenforcingdiskusage =!= true) enable_profiling: false - # Override the default container image for specific algorithms. - # Keys are algorithm names (as they appear in the algorithms list below). - # Values are interpreted based on the container framework: - # - # Image reference (e.g., "pathlinker:v3"): - # Prepends the registry prefix. Works with both Docker and Apptainer. - # - # Full image reference with registry (e.g., "ghcr.io/myorg/pathlinker:v3"): - # Used as-is (prefix NOT prepended). Works with both Docker and Apptainer. - # - # Local .sif file path (e.g., "images/pathlinker_v2.sif"): - # Apptainer/Singularity only. Skips pulling from registry and uses the - # pre-built .sif directly. When running via HTCondor with shared-fs-usage: none (set - # via the spras_profile config when running SPRAS against HTCondor), .sif paths listed - # here are automatically included in htcondor_transfer_input_files. - # Ignored with a warning if the framework is Docker. - # - # Example (one of each type): - # images: - # omicsintegrator1: "images/omics-integrator-1_v2.sif" # local .sif (Apptainer only) - # pathlinker: "pathlinker:v1234" # image name only (base_url/owner prepended) - # omicsintegrator2: "some-other-owner/oi2:latest" # owner/image (base_url prepended) - # mincostflow: "ghcr.io/reed-compbio/mincostflow:v2" # full registry reference (used as-is) +# Keys are algorithm names (as they appear in the algorithms list below). +# Values are interpreted based on the container framework: +# +# Image reference (e.g., "pathlinker:v3"): +# Prepends the registry prefix. Works with both Docker and Apptainer. +# +# Full image reference with registry (e.g., "ghcr.io/myorg/pathlinker:v3"): +# Used as-is (prefix NOT prepended). Works with both Docker and Apptainer. +# +# Local .sif file path (e.g., "images/pathlinker_v2.sif"): +# Apptainer/Singularity only. Skips pulling from registry and uses the +# pre-built .sif directly. When running via HTCondor with shared-fs-usage: none (set +# via the spras_profile config when running SPRAS against HTCondor), .sif paths listed +# here are automatically included in htcondor_transfer_input_files. +# Ignored with a warning if the framework is Docker. +# +# Example (one of each type): +# images: +# omicsintegrator1: "images/omics-integrator-1_v2.sif" # local .sif (Apptainer only) +# pathlinker: "pathlinker:v1234" # image name only (base_url/owner prepended) +# omicsintegrator2: "some-other-owner/oi2:latest" # owner/image (base_url prepended) +# mincostflow: "ghcr.io/reed-compbio/mincostflow:v2" # full registry reference (used as-is) # This list of algorithms should be generated by a script which checks the filesystem for installs. # It shouldn't be changed by mere mortals. (alternatively, we could add a path to executable for each algorithm @@ -273,16 +272,14 @@ analysis: # evaluation per algorithm will not run unless ml include and ml aggregate_per_algorithm are set to true aggregate_per_algorithm: true lpca: - # if true, runs LPCA in addition to the existing PCA (both analyses will run in parallel). - # Running both helps compare the two methods on the same data. + # Compare pathway graphs across all algorithms, independently of ml.include include: false - # adds separate LPCA analysis for each algorithm that has multiple parameter combinations + # Also fit each algorithm with at least three configured parameter combinations. + # Requires lpca.include: true; algorithms with fewer combinations are omitted. aggregate_per_algorithm: false - # number of principal components to compute + # Only a two-component fit is supported k: 2 - # fixed value of the logisticPCA tuning parameter m, used when cv is false. - # default of 6 was selected based on cross-validation experiments across multiple - # SPRAS algorithms (see https://github.com/Jeebjean/lpca-spras for details) + # Fixed, finite, strictly positive tuning parameter; no automatic selection m: 6 - # if true, choose m by cross-validation; if false, use the fixed m above - cv: false + # Show run identifiers on the LPCA plot + labels: true diff --git a/config/egfr.yaml b/config/egfr.yaml index b93c593c4..3b48d7a9c 100644 --- a/config/egfr.yaml +++ b/config/egfr.yaml @@ -134,6 +134,12 @@ analysis: labels: true kde: true remove_empty_pathways: true + lpca: + include: false + aggregate_per_algorithm: false + k: 2 + m: 6 + labels: true evaluation: include: true aggregate_per_algorithm: true diff --git a/spras/config/schema.py b/spras/config/schema.py index 6bda7efdd..36e26db10 100644 --- a/spras/config/schema.py +++ b/spras/config/schema.py @@ -10,9 +10,9 @@ - `CaseInsensitiveEnum` (see ./util.py) """ -from typing import Annotated +from typing import Annotated, Literal -from pydantic import AfterValidator, BaseModel, ConfigDict +from pydantic import AfterValidator, BaseModel, ConfigDict, Field from spras.config.algorithms import AlgorithmUnion from spras.config.container_schema import ContainerSettings @@ -70,9 +70,9 @@ class EvaluationAnalysis(BaseModel): class LpcaAnalysis(BaseModel): include: bool aggregate_per_algorithm: bool = False - k: int = 2 - m: float = 6 - cv: bool = False + k: Literal[2] = 2 # only support k=2 currently + m: float = Field(default=6, gt=0, allow_inf_nan=False) + labels: bool = True model_config = ConfigDict(extra='forbid') diff --git a/test/test_config.py b/test/test_config.py index a7231a93e..bc428a1a5 100644 --- a/test/test_config.py +++ b/test/test_config.py @@ -460,3 +460,31 @@ def test_eval_summary_coupling(self, eval_include, summary_include, expected_eva assert config.config.analysis_include_evaluation == expected_eval assert config.config.analysis_include_summary == expected_summary + + @pytest.mark.parametrize("include, aggregate", [ + (False, False), (False, True), (True, False), (True, True) + ]) + def test_lpca_options(self, include, aggregate): + test_config = get_test_config() + test_config["analysis"]["lpca"] = { + "include": include, "aggregate_per_algorithm": aggregate, + "m": 4.5, "labels": False, + } + parsed = config.Config(test_config) + assert parsed.analysis_include_lpca == include + assert parsed.analysis_include_lpca_aggregate_algo == (include and aggregate) + assert not parsed.analysis_include_ml + assert parsed.lpca_params.k == 2 + assert parsed.lpca_params.m == 4.5 + assert not parsed.lpca_params.labels + + @pytest.mark.parametrize("options", [ + {"k": 1}, {"k": 3}, {"k": 2.5}, + {"m": 0}, {"m": -1}, {"m": float("inf")}, {"m": float("nan")}, + {"cv": False}, {"cv": True}, {"kde": True}, + ]) + def test_lpca_invalid_options(self, options): + test_config = get_test_config() + test_config["analysis"]["lpca"] = {"include": True, **options} + with pytest.raises(ValueError): + config.Config(test_config) diff --git a/test/test_lpca_workflow.py b/test/test_lpca_workflow.py new file mode 100644 index 000000000..987de1548 --- /dev/null +++ b/test/test_lpca_workflow.py @@ -0,0 +1,167 @@ +"""Test LPCA scheduling and reruns using existing graph fixtures, not reconstruction.""" + +import csv +import shutil +import subprocess +import sys +from pathlib import Path + +import pandas as pd +import pytest +import yaml + +from spras.config.config import Config + +REPO = Path(__file__).resolve().parents[1] +LPCA_FILES = ('lpca.png', 'lpca-deviance.txt', 'lpca-coordinates.txt', 'lpca-binary-matrix.csv') + + +@pytest.fixture +def workflow_config(tmp_path): + raw = { + 'containers': {'registry': {}}, + 'immutable_files': False, + 'datasets': [{'label': 'toy', 'data_dir': '.', + 'node_files': [], 'edge_files': [], 'other_files': []}], + 'algorithms': [ + {'name': 'pathlinker', 'include': True, 'runs': {'test': {'k': [1, 2, 3, 4]}}}, + {'name': 'meo', 'include': True, 'runs': {'test': {'max_path_length': [1, 2]}}}, + ], + 'reconstruction_settings': {'locations': {'reconstruction_dir': 'output'}}, + 'analysis': { + # Disable PCA/HAC explicitly; LPCA does not depend on them. + 'ml': {'include': False}, + 'lpca': {'include': True, 'aggregate_per_algorithm': True, + 'm': 4, 'labels': False}, + }, + } + parsed = Config(raw) + logs_dir = tmp_path / 'output' / 'logs' + logs_dir.mkdir(parents=True) + # Seed completed reconstruction inputs. Only LPCA and the final-target rule + # are allowed below, so no reconstruction container or input dataset is needed. + for algorithm, combinations in parsed.algorithm_params.items(): + for i, params_hash in enumerate(combinations, start=1): + run = f'{algorithm}-params-{params_hash}' + pathway_file = tmp_path / 'output' / f'toy-{run}' / 'pathway.txt' + fixture_file = REPO / 'test/analysis/input/lpca' / f'pathway-params-{i}.txt' + pathway_file.parent.mkdir() + shutil.copyfile(fixture_file, pathway_file) + # The final target requires these logs, but LPCA does not read them. + parameter_log = logs_dir / f'parameters-{run}.yaml' + parameter_log.write_text('{}\n', encoding='utf-8') + dataset_log = logs_dir / 'datasets-toy.yaml' + dataset_log.write_text('{}\n', encoding='utf-8') + return raw + + +def run_workflow(tmp_path, raw, *options): + config_file = tmp_path / 'config.yaml' + config_file.write_text(yaml.safe_dump(raw), encoding='utf-8') + # Run the actual Snakefile in an isolated directory, including its metadata. + snakefile = REPO / 'Snakefile' + result = subprocess.run( + [sys.executable, '-m', 'snakemake', 'all', '--snakefile', str(snakefile), + '--configfile', str(config_file), '--cores', '1', '--nocolor', *options, + '--allowed-rules', 'all', 'lpca_analysis_all', 'lpca_analysis_aggregate_algo'], + cwd=tmp_path, capture_output=True, text=True, timeout=300, + ) + print(result.stdout) + print(result.stderr) + assert result.returncode == 0, result.stdout + result.stderr + return result.stdout + + +def lpca_summary(tmp_path, raw): + # --summary builds the real DAG without executing it and reports each output's + # status and update plan. It can test scheduling without a running Docker daemon. + text = run_workflow(tmp_path, raw, '--summary') + summary_rows = csv.DictReader(text.splitlines(), delimiter='\t') + lpca_rows = {} + for row in summary_rows: + if row['rule'].startswith('lpca_analysis_'): + output_file = row['output_file'].replace('\\', '/') + lpca_rows[output_file] = row + return lpca_rows + + +def expected_outputs(aggregate): + prefixes = [''] # The combined analysis has no algorithm prefix. + if aggregate: + prefixes.append('pathlinker-') + + outputs = set() + for prefix in prefixes: + for name in LPCA_FILES: + output_file = f'output/toy-ml/{prefix}{name}' + outputs.add(output_file) + return outputs + + +@pytest.mark.parametrize('include, aggregate', [ + (False, False), (False, True), (True, False), (True, True), +]) +def test_lpca_targets(tmp_path, workflow_config, include, aggregate): + workflow_config['analysis']['lpca'].update( + include=include, aggregate_per_algorithm=aggregate) + rows = lpca_summary(tmp_path, workflow_config) + assert set(rows) == (expected_outputs(aggregate) if include else set()) + # The two-run algorithm contributes to the combined fit but is not eligible + # for its own fit. The fixture explicitly sets analysis.ml.include=False. + assert not any('meo-lpca' in name for name in rows) + + +def test_lpca_workflow_reruns(tmp_path, workflow_config): + """Requires Docker and the LPCA image; never forces a rerun.""" + # Reuse the same directory so Snakemake retains the first run's metadata. + # Create the outputs at m=4, then change only m to verify automatic reruns. + for m in (4, 6): + workflow_config['analysis']['lpca']['m'] = m + before = lpca_summary(tmp_path, workflow_config) + assert set(before) == expected_outputs(True) + assert all(row['plan'] == 'update pending' for row in before.values()) + if m == 6: + # These outputs already exist, so the reason must be changed params. + assert all(row['status'] == 'params changed' for row in before.values()) + + # Execute normally, then check that another invocation needs no work. + run_workflow(tmp_path, workflow_config) + after = lpca_summary(tmp_path, workflow_config) + assert set(after) == expected_outputs(True) + assert all(row['plan'] == 'no update' for row in after.values()) + + # Combined: four PathLinker and two MEO graphs. Separate: PathLinker only. + output_dir = tmp_path / 'output' / 'toy-ml' + for prefix, count in [('', 6), ('pathlinker-', 4)]: + coordinates_file = output_dir / f'{prefix}lpca-coordinates.txt' + matrix_file = output_dir / f'{prefix}lpca-binary-matrix.csv' + deviance_file = output_dir / f'{prefix}lpca-deviance.txt' + coordinates = pd.read_csv(coordinates_file, sep='\t') + matrix = pd.read_csv(matrix_file) + assert len(coordinates) == count + assert coordinates['datapoint_labels'].tolist() == matrix['datapoint_labels'].tolist() + if prefix: + assert coordinates['datapoint_labels'].str.contains('-pathlinker-', regex=False).all() + + # Verify the actual fit used the new m, not just that it was scheduled. + summary = {} + for line in deviance_file.read_text().splitlines(): + key, value = line.split(': ', 1) + summary[key] = value + assert float(summary['m']) == m + + # Changing labels should schedule an update; no extra fit is needed to check this. + workflow_config['analysis']['lpca']['labels'] = True + rows = lpca_summary(tmp_path, workflow_config) + assert set(rows) == expected_outputs(True) + assert all(row['status'] == 'params changed' for row in rows.values()) + # Restore the executed setting so the next check isolates a missing output. + workflow_config['analysis']['lpca']['labels'] = False + + # A missing fit summary must schedule its producer even when the plot exists. + missing_output = 'output/toy-ml/lpca-deviance.txt' + missing_file = tmp_path / missing_output + missing_file.unlink() + rows = lpca_summary(tmp_path, workflow_config) + assert rows[missing_output]['status'] == 'missing' + assert rows[missing_output]['plan'] == 'update pending' From 2d7d91c9023bf082b240de4d0b53a77d9fc15f5a Mon Sep 17 00:00:00 2001 From: Anthony Gitter Date: Fri, 2 Oct 2026 09:25:53 -0500 Subject: [PATCH 25/29] Formatting and doc cleanup --- Snakefile | 3 +-- docker-wrappers/LPCA/README.md | 17 +++++++++++------ docker-wrappers/LPCA/test_container.py | 7 +++---- docs/fordevs/spras.analysis.rst | 9 +++++++++ spras/analysis/ml.py | 3 ++- 5 files changed, 26 insertions(+), 13 deletions(-) diff --git a/Snakefile b/Snakefile index a2d232a22..95758db5d 100644 --- a/Snakefile +++ b/Snakefile @@ -94,7 +94,7 @@ def make_final_input(wildcards): final_input.extend(expand('{out_dir}{sep}{dataset}-ml{sep}ensemble-pathway.txt',out_dir=out_dir,sep=SEP,dataset=dataset_labels,algorithm_params=algorithms_with_params)) final_input.extend(expand('{out_dir}{sep}{dataset}-ml{sep}jaccard-matrix.txt',out_dir=out_dir,sep=SEP,dataset=dataset_labels,algorithm_params=algorithms_with_params)) final_input.extend(expand('{out_dir}{sep}{dataset}-ml{sep}jaccard-heatmap.png',out_dir=out_dir,sep=SEP,dataset=dataset_labels,algorithm_params=algorithms_with_params)) - + if _config.config.analysis_include_lpca: final_input.extend(expand('{out_dir}{sep}{dataset}-ml{sep}lpca.png',out_dir=out_dir, sep=SEP, dataset=dataset_labels)) final_input.extend(expand('{out_dir}{sep}{dataset}-ml{sep}lpca-deviance.txt',out_dir=out_dir, sep=SEP, dataset=dataset_labels)) @@ -371,7 +371,6 @@ rule ml_analysis: ml.hac_horizontal(summary_df, output.hac_image_horizontal, output.hac_clusters_horizontal, **hac_params) ml.pca(summary_df, output.pca_image, output.pca_variance, output.pca_coordinates, **pca_params) - # Calculated Jaccard similarity between output pathways for each dataset rule jaccard_similarity: input: diff --git a/docker-wrappers/LPCA/README.md b/docker-wrappers/LPCA/README.md index c21d5b1ac..9f838cc2e 100644 --- a/docker-wrappers/LPCA/README.md +++ b/docker-wrappers/LPCA/README.md @@ -26,12 +26,16 @@ writing it. This wrapper deliberately requires at least three runs, three edge features, and three distinct binary network profiles. All feature values must be finite -numeric zeros or ones. Missing, nonnumeric, and nonbinary entries are error. +numeric zeros or ones. Missing, nonnumeric, and nonbinary entries are errors. Duplicate profiles and constant features are retained when the input otherwise satisfies these requirements. `k` must equal 2. `m` must be finite and strictly positive. +The default `m=6` is based on the exploratory +[LPCA-SPRAS experiments](https://github.com/Jeebjean/lpca-spras). +No single value of `m` was universally best. + ### Outputs and numerical checks The raw scores CSV contains `datapoint_labels`, `PC1`, and `PC2`, with one row @@ -58,6 +62,7 @@ analysis: aggregate_per_algorithm: false k: 2 m: 6 + labels: true ``` ## Decomposition @@ -70,12 +75,12 @@ edge features. ## Building and publishing the image -For the SPRAS default registry to resolve the image, it must be published as -`docker.io/reedcompbio/lpca:v1`, which requires access to the `reedcompbio` -Docker Hub organization: +Replace `` below with the tag from `LPCA_CONTAINER_SUFFIX` in +`spras/analysis/lpca.py`. Publishing to the default registry requires access +to the `reedcompbio` Docker Hub organization: - docker build -t reedcompbio/lpca:v1 docker-wrappers/LPCA/ - docker push reedcompbio/lpca:v1 + docker build -t "reedcompbio/lpca:" docker-wrappers/LPCA/ + docker push "reedcompbio/lpca:" ## Testing Run the test Python script to test the Docker image diff --git a/docker-wrappers/LPCA/test_container.py b/docker-wrappers/LPCA/test_container.py index dabf402a2..d7f6ffd87 100644 --- a/docker-wrappers/LPCA/test_container.py +++ b/docker-wrappers/LPCA/test_container.py @@ -1,9 +1,8 @@ #!/usr/bin/env python3 """Run standalone regression tests for the SPRAS LPCA container. -Save this file beside run_lpca.R in docker-wrappers/LPCA/. From the repository -root, build the image and run the tests with Python >= 3.9 and Docker: - docker build -t reedcompbio/lpca:v1 docker-wrappers/LPCA +Build the image as described in README.md, then run the tests from the +repository root with Python >= 3.9 and Docker: python docker-wrappers/LPCA/test_container.py The wrapper is located relative to this file, not the working directory. @@ -272,7 +271,7 @@ def main() -> int: description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter ) parser.add_argument("--image", default="reedcompbio/lpca:v1", - help="Local image to test (default: reedcompbio/lpca:v1)") + help="Local image to test (default: %(default)s)") parser.add_argument("--log", type=Path, help="Also save output to a new UTF-8 file; never overwrite") parser.add_argument("--timeout", type=int, default=600, diff --git a/docs/fordevs/spras.analysis.rst b/docs/fordevs/spras.analysis.rst index fae5b75de..970538ad3 100644 --- a/docs/fordevs/spras.analysis.rst +++ b/docs/fordevs/spras.analysis.rst @@ -15,6 +15,15 @@ :undoc-members: :show-inheritance: +**************************** + spras.analysis.lpca module +**************************** + +.. automodule:: spras.analysis.lpca + :members: + :undoc-members: + :show-inheritance: + ************************** spras.analysis.ml module ************************** diff --git a/spras/analysis/ml.py b/spras/analysis/ml.py index 7e17b1e00..2b847e87f 100644 --- a/spras/analysis/ml.py +++ b/spras/analysis/ml.py @@ -154,7 +154,8 @@ def pca(dataframe: pd.DataFrame, output_png: str | PathLike, output_var: str | P if not isinstance(labels, bool): raise ValueError(f"labels={labels} must be True or False") - # center binary data by subtracting the column-wise mean + # For the logistic PCA alternative, see spras.analysis.lpca.run_lpca. + # Center binary data by subtracting the column-wise mean # allows PCA to focus on edge inclusion patterns across runs rather than raw output volume. scaler = StandardScaler(with_std=False) scaler.fit(X) # compute mean inclusion rate per edge From 894643d4b00e14198e87f1739e29df35d404d39e Mon Sep 17 00:00:00 2001 From: Anthony Gitter Date: Fri, 2 Oct 2026 11:36:11 -0500 Subject: [PATCH 26/29] Config whitespace --- config/config.yaml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/config/config.yaml b/config/config.yaml index b634e7c46..f1b399f42 100644 --- a/config/config.yaml +++ b/config/config.yaml @@ -49,7 +49,7 @@ containers: # # requirements = versionGE(split(Target.CondorVersion)[1], "24.8.0") && (isenforcingdiskusage =!= true) enable_profiling: false - # Override the default container image for specific algorithms. +# Override the default container image for specific algorithms. # Keys are algorithm names (as they appear in the algorithms list below). # Values are interpreted based on the container framework: # From f2a24f43be09bb716053427283548923eb5e3909 Mon Sep 17 00:00:00 2001 From: Anthony Gitter Date: Fri, 2 Oct 2026 15:58:25 -0500 Subject: [PATCH 27/29] Update comments --- Snakefile | 1 + config/config.yaml | 9 +++++---- docker-wrappers/LPCA/README.md | 2 +- 3 files changed, 7 insertions(+), 5 deletions(-) diff --git a/Snakefile b/Snakefile index 95758db5d..660b2e5ef 100644 --- a/Snakefile +++ b/Snakefile @@ -101,6 +101,7 @@ def make_final_input(wildcards): final_input.extend(expand('{out_dir}{sep}{dataset}-ml{sep}lpca-coordinates.txt',out_dir=out_dir, sep=SEP, dataset=dataset_labels)) final_input.extend(expand('{out_dir}{sep}{dataset}-ml{sep}lpca-binary-matrix.csv',out_dir=out_dir, sep=SEP, dataset=dataset_labels)) + # Only run LPCA per algorithm for the algorithms collected in algorithms_lpca if _config.config.analysis_include_lpca_aggregate_algo: final_input.extend(expand('{out_dir}{sep}{dataset}-ml{sep}{algorithm}-lpca-deviance.txt',out_dir=out_dir, sep=SEP, dataset=dataset_labels, algorithm=algorithms_lpca)) final_input.extend(expand('{out_dir}{sep}{dataset}-ml{sep}{algorithm}-lpca.png',out_dir=out_dir, sep=SEP,dataset=dataset_labels,algorithm=algorithms_lpca)) diff --git a/config/config.yaml b/config/config.yaml index f1b399f42..50047e72b 100644 --- a/config/config.yaml +++ b/config/config.yaml @@ -49,6 +49,7 @@ containers: # # requirements = versionGE(split(Target.CondorVersion)[1], "24.8.0") && (isenforcingdiskusage =!= true) enable_profiling: false + # Override the default container image for specific algorithms. # Keys are algorithm names (as they appear in the algorithms list below). # Values are interpreted based on the container framework: @@ -273,11 +274,11 @@ analysis: aggregate_per_algorithm: true lpca: # Compare pathway graphs across all algorithms, independently of ml.include - include: false - # Also fit each algorithm with at least three configured parameter combinations. + include: true + # Also fit each algorithm that has at least three configured parameter combinations. # Requires lpca.include: true; algorithms with fewer combinations are omitted. - aggregate_per_algorithm: false - # Only a two-component fit is supported + aggregate_per_algorithm: true + # Only a two-component fit is supported. Keep k=2. k: 2 # Fixed, finite, strictly positive tuning parameter; no automatic selection m: 6 diff --git a/docker-wrappers/LPCA/README.md b/docker-wrappers/LPCA/README.md index 9f838cc2e..90a495da9 100644 --- a/docker-wrappers/LPCA/README.md +++ b/docker-wrappers/LPCA/README.md @@ -5,7 +5,7 @@ Docker image: https://hub.docker.com/r/reedcompbio/lpca This wrapper uses [logisticPCA](https://github.com/andland/logisticPCA) ([Landgraf & Lee, 2020](https://doi.org/10.1016/j.jmva.2020.104668)) for an exploratory, two-dimensional comparison of reconstructed networks. It calls -`logisticPCA`, not `logisticSVD`. It supports only two components and a fixed, +`logisticPCA`, not `logisticSVD`. It supports only `k = 2` components and a fixed, finite, strictly positive `m`. Cross-validation for automatic selection of `m` and modifying `k` are not supported. From 415c7996e7a9cddd3f6716b75c6a83517a9b1ae1 Mon Sep 17 00:00:00 2001 From: Anthony Gitter Date: Sun, 4 Oct 2026 08:49:02 -0500 Subject: [PATCH 28/29] Memory discussions --- config/config.yaml | 3 +++ docker-wrappers/LPCA/README.md | 4 +++- 2 files changed, 6 insertions(+), 1 deletion(-) diff --git a/config/config.yaml b/config/config.yaml index 50047e72b..46c514ccf 100644 --- a/config/config.yaml +++ b/config/config.yaml @@ -274,6 +274,9 @@ analysis: aggregate_per_algorithm: true lpca: # Compare pathway graphs across all algorithms, independently of ml.include + # LPCA allocates dense edge-by-edge matrices, requiring quadratic memory. + # Partial decomposition does not remove them. + # LPCA fits may fail due to out-of-memory errors. include: true # Also fit each algorithm that has at least three configured parameter combinations. # Requires lpca.include: true; algorithms with fewer combinations are omitted. diff --git a/docker-wrappers/LPCA/README.md b/docker-wrappers/LPCA/README.md index 90a495da9..d302deacc 100644 --- a/docker-wrappers/LPCA/README.md +++ b/docker-wrappers/LPCA/README.md @@ -71,7 +71,9 @@ analysis: `rARPACK` (backed by `RSpectra`) for partial decompositions where supported. The package can fall back to full eigendecomposition. It still constructs dense edge-by-edge matrices, so memory use can grow quadratically with the number of -edge features. +edge features. For example, the `egfr.yaml` configuration (19 runs, 18,851 edge features, `k=2`, `m=6`) +requires nearly 20GB memory. Completing this run may require modifying the memory +provided to the running container. ## Building and publishing the image From 7100ada6679ac5aa4f1333807d017400f27ffa92 Mon Sep 17 00:00:00 2001 From: Anthony Gitter Date: Sun, 4 Oct 2026 11:29:48 -0500 Subject: [PATCH 29/29] Disable LPCA by default --- config/config.yaml | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/config/config.yaml b/config/config.yaml index 46c514ccf..b2cee93f9 100644 --- a/config/config.yaml +++ b/config/config.yaml @@ -277,10 +277,10 @@ analysis: # LPCA allocates dense edge-by-edge matrices, requiring quadratic memory. # Partial decomposition does not remove them. # LPCA fits may fail due to out-of-memory errors. - include: true + include: false # Also fit each algorithm that has at least three configured parameter combinations. # Requires lpca.include: true; algorithms with fewer combinations are omitted. - aggregate_per_algorithm: true + aggregate_per_algorithm: false # Only a two-component fit is supported. Keep k=2. k: 2 # Fixed, finite, strictly positive tuning parameter; no automatic selection