diff --git a/.github/workflows/build-containers.yml b/.github/workflows/build-containers.yml index 811b8e034..8bebdabdd 100644 --- a/.github/workflows/build-containers.yml +++ b/.github/workflows/build-containers.yml @@ -73,3 +73,8 @@ jobs: # of detected changes. always_build: true context: ./ + build-and-remove-lpca: + uses: "./.github/workflows/build-and-remove-template.yml" + with: + path: docker-wrappers/LPCA + container: reedcompbio/lpca diff --git a/Snakefile b/Snakefile index 5ad7aa185..c71625db6 100644 --- a/Snakefile +++ b/Snakefile @@ -91,6 +91,16 @@ def make_final_input(wildcards): final_input.extend(expand('{out_dir}{sep}{dataset}-ml{sep}ensemble-pathway.txt',out_dir=out_dir,sep=SEP,dataset=dataset_labels,algorithm_params=algorithms_with_params)) final_input.extend(expand('{out_dir}{sep}{dataset}-ml{sep}jaccard-matrix.txt',out_dir=out_dir,sep=SEP,dataset=dataset_labels,algorithm_params=algorithms_with_params)) final_input.extend(expand('{out_dir}{sep}{dataset}-ml{sep}jaccard-heatmap.png',out_dir=out_dir,sep=SEP,dataset=dataset_labels,algorithm_params=algorithms_with_params)) + + if _config.config.analysis_include_lpca: + final_input.extend(expand('{out_dir}{sep}{dataset}-ml{sep}{algorithm}-lpca-scores.csv',out_dir=out_dir, sep=SEP, dataset=dataset_labels, algorithm=algorithms_mult_param_combos)) + final_input.extend(expand('{out_dir}{sep}{dataset}-ml{sep}{algorithm}-lpca.png',out_dir=out_dir, sep=SEP,dataset=dataset_labels,algorithm=algorithms_mult_param_combos)) + final_input.extend(expand('{out_dir}{sep}{dataset}-ml{sep}{algorithm}-lpca-coordinates.txt',out_dir=out_dir, sep=SEP, dataset=dataset_labels,algorithm=algorithms_mult_param_combos)) + final_input.extend(expand('{out_dir}{sep}{dataset}-ml{sep}{algorithm}-lpca-binary-matrix.csv',out_dir=out_dir, sep=SEP, dataset=dataset_labels,algorithm=algorithms_mult_param_combos)) + final_input.extend(expand('{out_dir}{sep}{dataset}-ml{sep}lpca.png',out_dir=out_dir, sep=SEP, dataset=dataset_labels)) + final_input.extend(expand('{out_dir}{sep}{dataset}-ml{sep}lpca-scores.csv',out_dir=out_dir, sep=SEP, dataset=dataset_labels)) + final_input.extend(expand('{out_dir}{sep}{dataset}-ml{sep}lpca-coordinates.txt',out_dir=out_dir, sep=SEP, dataset=dataset_labels)) + final_input.extend(expand('{out_dir}{sep}{dataset}-ml{sep}lpca-binary-matrix.csv',out_dir=out_dir, sep=SEP, dataset=dataset_labels)) if _config.config.analysis_include_ml_aggregate_algo: final_input.extend(expand('{out_dir}{sep}{dataset}-ml{sep}{algorithm}-pca.png',out_dir=out_dir,sep=SEP,dataset=dataset_labels,algorithm=algorithms_mult_param_combos)) @@ -356,6 +366,7 @@ rule ml_analysis: ml.hac_horizontal(summary_df, output.hac_image_horizontal, output.hac_clusters_horizontal, **hac_params) ml.pca(summary_df, output.pca_image, output.pca_variance, output.pca_coordinates, **pca_params) + # Calculated Jaccard similarity between output pathways for each dataset rule jaccard_similarity: input: @@ -403,6 +414,32 @@ rule ml_analysis_aggregate_algo: ml.hac_horizontal(summary_df, output.hac_image_horizontal, output.hac_clusters_horizontal, **hac_params) ml.pca(summary_df, output.pca_image, output.pca_variance, output.pca_coordinates, **pca_params) +rule lpca_analysis_all: + input: + pathways = expand('{out_dir}{sep}{{dataset}}-{algorithm_params}{sep}pathway.txt', out_dir=out_dir, sep=SEP, algorithm_params=algorithms_with_params) + output: + lpca_scores = SEP.join([out_dir, '{dataset}-ml', 'lpca-scores.csv']), + lpca_png = SEP.join([out_dir, '{dataset}-ml', 'lpca.png']), + lpca_coord = SEP.join([out_dir, '{dataset}-ml', 'lpca-coordinates.txt']), + lpca_matrix = SEP.join([out_dir, '{dataset}-ml', 'lpca-binary-matrix.csv']) + run: + from spras.analysis import lpca + summary_df = ml.summarize_networks(input.pathways) + lpca.run_lpca( + summary_df, + output.lpca_scores, + output.lpca_matrix, + k=_config.config.lpca_params.k, + m=_config.config.lpca_params.m, + cv=_config.config.lpca_params.cv, + container_settings=container_settings + ) + lpca.plot_lpca( + output.lpca_scores, + output.lpca_png, + output.lpca_coord + ) + # Ensemble the output pathways for each dataset per algorithm rule ensemble_per_algo: input: diff --git a/config/config.yaml b/config/config.yaml index ef51de739..9f6b00fad 100644 --- a/config/config.yaml +++ b/config/config.yaml @@ -272,3 +272,15 @@ analysis: # adds evaluation per algorithm per dataset-goldstandard pair # evaluation per algorithm will not run unless ml include and ml aggregate_per_algorithm are set to true aggregate_per_algorithm: true + lpca: + # if true, runs LPCA in addition to the existing PCA (both analyses will run in parallel). + # Running both helps compare the two methods on the same data. + include: false + # number of principal components to compute + k: 2 + # fixed value of the logisticPCA tuning parameter m, used when cv is false. + # default of 6 was selected based on cross-validation experiments across multiple + # SPRAS algorithms (see https://github.com/Jeebjean/lpca-spras for details) + m: 6 + # if true, choose m by cross-validation; if false, use the fixed m above + cv: false diff --git a/docker-wrappers/LPCA/Dockerfile b/docker-wrappers/LPCA/Dockerfile new file mode 100644 index 000000000..66b27893a --- /dev/null +++ b/docker-wrappers/LPCA/Dockerfile @@ -0,0 +1,31 @@ +# Logistic PCA (logisticPCA) wrapper for SPRAS. +# Invoked by spras/analysis/lpca.py, which supplies the full command +# (Rscript /app/run_lpca.R ... or /app/run_cv.R ...), so no ENTRYPOINT is set. + +# Pinned R version for reproducibility. Bump deliberately, not to :latest. +FROM rocker/r-base:4.4.2 + +LABEL org.opencontainers.image.source="https://github.com/Reed-CompBio/spras" +LABEL org.opencontainers.image.description="Logistic PCA (logisticPCA) wrapper for SPRAS" + +# System libraries needed to compile ggplot2 (a hard Import of logisticPCA) +# and its dependency stack from source on Debian. +RUN apt-get update && apt-get install -y --no-install-recommends \ + libcurl4-openssl-dev \ + libssl-dev \ + libxml2-dev \ + libfontconfig1-dev \ + libfreetype6-dev \ + libpng-dev \ + libtiff5-dev \ + libjpeg-dev \ + && rm -rf /var/lib/apt/lists/* + +# logisticPCA is still on CRAN (last published 2016) and pulls in ggplot2. +RUN Rscript -e "install.packages(c('logisticPCA', 'rARPACK'), repos='https://cran.r-project.org')" \ + && Rscript -e "library(logisticPCA); library(rARPACK)" + +COPY run_lpca.R /app/run_lpca.R +COPY run_cv.R /app/run_cv.R + +WORKDIR /app \ No newline at end of file diff --git a/docker-wrappers/LPCA/README.md b/docker-wrappers/LPCA/README.md new file mode 100644 index 000000000..6a3a17e85 --- /dev/null +++ b/docker-wrappers/LPCA/README.md @@ -0,0 +1,57 @@ +# LPCA (Logistic PCA) wrapper + +Docker image: https://hub.docker.com/r/reedcompbio/lpca + +This wrapper runs [logisticPCA](https://github.com/andland/logisticPCA) + +This wrapper runs [logisticPCA](https://github.com/andland/logisticPCA) +([Landgraf & Lee, 2020](https://doi.org/10.1016/j.jmva.2020.104668)) as a SPRAS analysis step. It reduces the binary +edge-by-run matrix built from a set of pathway reconstruction outputs to a small +number of components and reports the proportion of deviance explained. + +The analysis is driven by the `analysis.lpca` config block and the +`lpca_analysis` Snakemake rule, and is implemented in `spras/analysis/lpca.py`. + +## Configuration + + analysis: + lpca: + include: false # run the LPCA analysis per algorithm + k: 2 # number of principal components + m: 6 # fixed logisticPCA tuning parameter, used when cv is false + cv: false # true: choose m by cross-validation; false: use the fixed m + +LPCA only runs for algorithms with multiple parameter combinations, so that the +binary matrix has more than one column. It also needs a reasonable number of +observations to be meaningful; very small inputs (such as the bundled example +datasets) produce degenerate results, which is why it is disabled by default. + +## Scripts + +The image contains two R scripts under `/app`: + +- `run_lpca.R `: runs logisticPCA with a fixed `m` and + writes the scores CSV plus a sibling `_deviance.txt`. +- `run_cv.R `: cross-validates `m` over 1..20 and writes a + CSV with a `best_m` column (plus a `_curve.csv` with the full CV curve). Only + used when `cv: true`. + +Both read a CSV whose first column holds row labels and whose remaining columns +are binary (0/1) features, and coerce missing values to 0. + +## Dependency note + +`logisticPCA` declares `ggplot2` as a hard `Imports` dependency, so building the +image compiles the ggplot2 stack. The Dockerfile installs the required Debian +system libraries for that. To build from a source CRAN mirror the `repos` +argument already points at `https://cran.r-project.org`. + +## Building and publishing the image + +For the SPRAS default registry to resolve the image, it must be published as +`docker.io/reedcompbio/lpca:v1`, which requires access to the `reedcompbio` +Docker Hub organization: + + docker build -t reedcompbio/lpca:v1 docker-wrappers/lpca/ + docker push reedcompbio/lpca:v1 + diff --git a/docker-wrappers/LPCA/run_cv.R b/docker-wrappers/LPCA/run_cv.R new file mode 100644 index 000000000..7301d3419 --- /dev/null +++ b/docker-wrappers/LPCA/run_cv.R @@ -0,0 +1,36 @@ +# run_cv.R +# Finds the optimal m for a given k using cross-validation + +args = commandArgs(trailingOnly = TRUE) +input_file = args[1] +output_file = args[2] +k = as.integer(args[3]) + +set.seed(42) + +# Load data +library(logisticPCA) +data = read.csv(input_file, row.names = NULL) +data = data[, -1] +data_matrix = as.matrix(data) +data_matrix[is.na(data_matrix)] = 0 + +# Cross-validation over m, fixed k +cv_result = cv.lpca(data_matrix, ks = k, ms = 1:20) +best_m = which.min(cv_result) + +cat("Cross-validation done for k =", k, "\n") +cat("Best m:", best_m, "\n") + +# Save best m +write.csv(data.frame(k = k, best_m = best_m), output_file, row.names = FALSE) + +# Save full CV curve (all m values and their reconstruction error) +cv_curve_file = sub("\\.csv$", "_curve.csv", output_file) +cv_df = data.frame( + m = 1:20, + reconstruction_error = as.numeric(cv_result), + is_best = (1:20) == best_m +) +write.csv(cv_df, cv_curve_file, row.names = FALSE) +cat("CV curve saved to", cv_curve_file, "\n") \ No newline at end of file diff --git a/docker-wrappers/LPCA/run_lpca.R b/docker-wrappers/LPCA/run_lpca.R new file mode 100644 index 000000000..1a3049e56 --- /dev/null +++ b/docker-wrappers/LPCA/run_lpca.R @@ -0,0 +1,30 @@ +# Read command line arguments +args = commandArgs(trailingOnly = TRUE) +input_file = args[1] +output_file = args[2] +k = as.integer(args[3]) +m = as.numeric(args[4]) + +# Load data +library(logisticPCA) +data = read.csv(input_file, row.names = NULL) +row_labels = data[, 1] +data = data[, -1] +data_matrix = as.matrix(data) +data_matrix[is.na(data_matrix)] = 0 + +# Run LPCA +model = logisticPCA(data_matrix, k = k, m = m, partial_decomp = TRUE) + +# Save scores +scores = model$PCs +rownames(scores) = row_labels +write.csv(scores, output_file, row.names = TRUE) + +# Save deviance explained +deviance_file = sub("\\.csv$", "_deviance.txt", output_file) +writeLines(as.character(model$prop_deviance_expl), deviance_file) + +cat("LPCA done! Scores saved to", output_file, "\n") +cat("Score dimensions:", nrow(scores), "x", ncol(scores), "\n") +cat("Proportion of deviance explained:", model$prop_deviance_expl, "\n") \ No newline at end of file diff --git a/spras/analysis/lpca.py b/spras/analysis/lpca.py new file mode 100644 index 000000000..68728388c --- /dev/null +++ b/spras/analysis/lpca.py @@ -0,0 +1,141 @@ +# Logistic PCA analysis for SPRAS +# Runs LPCA on the binary edge x algorithm matrix produced by summarize_networks. +# Configured through the analysis.lpca block (k, m, cv, transpose). + +from pathlib import Path + +import matplotlib.pyplot as plt +import pandas as pd +import seaborn as sns + +from spras.analysis.ml import create_palette +from spras.config.container_schema import ProcessedContainerSettings +from spras.containers import prepare_volume, run_container_and_log + +# Published as docker.io/reedcompbio/lpca:v1. Only the suffix is given here; +# the registry prefix is resolved from the container settings. +LPCA_CONTAINER_SUFFIX = 'lpca:v1' +LPCA_WORK_DIR = '/app' + + +def run_lpca( + dataframe: pd.DataFrame, + output_scores: str, + output_matrix: str, + k: int = 2, + m: float = 6, + cv: bool = False, + container_settings=None +) -> None: + """ + Runs Logistic PCA on the binary edge x algorithm matrix built from SPRAS + algorithm output files. + + @param dataframe: binary dataframe of edge comparison between algorithms from summarize_networks + @param output_matrix: path to write the binary matrix CSV (used as Docker volume mount) + @param output_scores: path to write the LPCA PC scores CSV + @param k: number of principal components (default 2) + @param m: fixed logisticPCA tuning parameter, used when cv is False + @param cv: if True, determine m by cross-validation; if False, use the + fixed m directly (default False) + @param container_settings: configure the container runtime (Docker or Singularity) + + Note: KDE-based parameter selection (used by PCA) always uses PCA scores, even when LPCA + is also enabled. KDE integration with LPCA may be added in a future update. + """ + if not container_settings: + container_settings = ProcessedContainerSettings() + + # Step 1: build the binary edge x algorithm matrix + matrix = dataframe + # Transpose so runs are the observations, mirroring the classic PCA analysis + matrix = matrix.T + print(f'LPCA: Matrix shape: {matrix.shape}') + + # Step 2: write the binary matrix for Docker volume mount + output_dir = Path(output_scores).parent + output_dir.mkdir(parents=True, exist_ok=True) + Path(output_matrix).parent.mkdir(parents=True, exist_ok=True) + matrix_path = output_matrix + matrix.to_csv(matrix_path) + + # Step 3: mount the matrix and the scores output + volumes = [] + bind_path, mapped_matrix = prepare_volume(matrix_path, LPCA_WORK_DIR, container_settings) + volumes.append(bind_path) + bind_path, mapped_scores = prepare_volume(output_scores, LPCA_WORK_DIR, container_settings) + volumes.append(bind_path) + + # Step 4: choose m, optionally via cross-validation + if cv: + algo_name = Path(output_scores).name.replace('-lpca-scores.csv', '') + cv_output_path = str(output_dir / f'{algo_name}-lpca_cv_result.csv') + bind_path, mapped_cv_output = prepare_volume(cv_output_path, LPCA_WORK_DIR, container_settings) + volumes.append(bind_path) + + print(f'LPCA: Running cross-validation with k={k}...') + command_cv = ['Rscript', '/app/run_cv.R', mapped_matrix, mapped_cv_output, str(k)] + run_container_and_log('LPCA-CV', LPCA_CONTAINER_SUFFIX, command_cv, volumes, + LPCA_WORK_DIR, None, container_settings) + + if not Path(cv_output_path).exists(): + raise FileNotFoundError( + f'LPCA: Cross-validation output not found at {cv_output_path}. ' + 'Check the LPCA Docker container logs for errors.' + ) + m_used = pd.read_csv(cv_output_path)['best_m'][0] + print(f'LPCA: Best m found by CV: {m_used}') + else: + m_used = m + print(f'LPCA: Using fixed m={m_used}') + + # Step 5: run LPCA with the chosen k and m + print(f'LPCA: Running LPCA with k={k}, m={m_used}...') + command_lpca = ['Rscript', '/app/run_lpca.R', mapped_matrix, mapped_scores, str(k), str(m_used)] + run_container_and_log('LPCA', LPCA_CONTAINER_SUFFIX, command_lpca, volumes, + LPCA_WORK_DIR, None, container_settings) + + print(f'LPCA: Done! Scores saved to {output_scores}') + +def plot_lpca(scores_file: str, output_png: str, output_coord: str, labels: bool = True) -> None: + """ + Creates a scatterplot of the first two LPCA principal components. + @param scores_file: path to the LPCA scores CSV file + @param output_png: path to save the scatterplot PNG + @param output_coord: path to save the PC coordinates + @param labels: if True, adds algorithm labels to the plot + """ + scores = pd.read_csv(scores_file, index_col=0) + + if scores.empty: + print('LPCA: Scores file is empty, skipping plot.') + return + + # Extract algorithm names from the index + column_names = [idx.split('-')[-3] if '-' in idx else idx for idx in scores.index] + + X = scores.values + fig, ax = plt.subplots(figsize=(10, 8)) + + label_color_map = create_palette(column_names) + sns.scatterplot(x=X[:, 0], y=X[:, 1], hue=column_names, palette=label_color_map, s=70, ax=ax) + + if labels: + for i, label in enumerate(scores.index): + ax.annotate(label, (X[i, 0], X[i, 1]), fontsize=6, alpha=0.7) + + ax.set_xlabel('PC1') + ax.set_ylabel('PC2') + ax.set_title('Logistic PCA') + + plt.tight_layout() + + # Save PNG + Path(output_png).parent.mkdir(parents=True, exist_ok=True) + plt.savefig(output_png, dpi=200) + plt.close() + + # Save coordinates + coord_df = pd.DataFrame(X, columns=['PC1', 'PC2'], index=scores.index) + coord_df.to_csv(output_coord) + print(f'LPCA: Plot saved to {output_png}') diff --git a/spras/analysis/ml.py b/spras/analysis/ml.py index 55abae5a8..459852770 100644 --- a/spras/analysis/ml.py +++ b/spras/analysis/ml.py @@ -156,7 +156,8 @@ def pca(dataframe: pd.DataFrame, output_png: str | PathLike, output_var: str | P # center binary data by subtracting the column-wise mean # allows PCA to focus on edge inclusion patterns across runs rather than raw output volume. - # TODO: replace PCA https://github.com/Reed-CompBio/spras/issues/271 + # TODO: consider replacing PCA with LPCA for binary data https://github.com/Reed-CompBio/spras/issues/271 + # LPCA is now available as an alternative analysis (analysis.lpca in config) scaler = StandardScaler(with_std=False) scaler.fit(X) # compute mean inclusion rate per edge X_scaled = scaler.transform(X) diff --git a/spras/config/config.py b/spras/config/config.py index ebf10faad..e99d965c9 100644 --- a/spras/config/config.py +++ b/spras/config/config.py @@ -86,6 +86,8 @@ def __init__(self, raw_config: dict[str, Any]): self.evaluation_params = self.analysis_params.evaluation # A dict with the ML settings self.ml_params = self.analysis_params.ml + # A dict with the LPCA settings + self.lpca_params = self.analysis_params.lpca # A Boolean specifying whether to run ML analysis for individual algorithms self.analysis_include_ml_aggregate_algo = None # A dict with the PCA settings @@ -96,6 +98,8 @@ def __init__(self, raw_config: dict[str, Any]): self.analysis_include_summary = None # A Boolean specifying whether to run the Cytoscape analysis self.analysis_include_cytoscape = None + # A Boolean specifying whether to run the LPCA analysis + self.analysis_include_lpca = None # A Boolean specifying whether to run the ML analysis self.analysis_include_ml = None # A Boolean specifying whether to run the Evaluation analysis @@ -254,6 +258,7 @@ def process_analysis(self, raw_config: RawConfig): self.analysis_include_summary = raw_config.analysis.summary.include self.analysis_include_cytoscape = raw_config.analysis.cytoscape.include self.analysis_include_ml = raw_config.analysis.ml.include + self.analysis_include_lpca = raw_config.analysis.lpca.include self.analysis_include_evaluation = raw_config.analysis.evaluation.include # Only run ML aggregate per algorithm if analysis include ML is set to True diff --git a/spras/config/schema.py b/spras/config/schema.py index 1a965c75c..f2e0ba353 100644 --- a/spras/config/schema.py +++ b/spras/config/schema.py @@ -67,10 +67,19 @@ class EvaluationAnalysis(BaseModel): model_config = ConfigDict(extra='forbid') +class LpcaAnalysis(BaseModel): + include: bool + k: int = 2 + m: float = 6 + cv: bool = False + + model_config = ConfigDict(extra='forbid') + class Analysis(BaseModel): summary: SummaryAnalysis = SummaryAnalysis(include=False) cytoscape: CytoscapeAnalysis = CytoscapeAnalysis(include=False) ml: MlAnalysis = MlAnalysis(include=False) + lpca: LpcaAnalysis = LpcaAnalysis(include=False) evaluation: EvaluationAnalysis = EvaluationAnalysis(include=False) model_config = ConfigDict(extra='forbid') diff --git a/spras/containers.py b/spras/containers.py index c30697f3f..15573dc8d 100644 --- a/spras/containers.py +++ b/spras/containers.py @@ -359,7 +359,7 @@ def run_container_docker(container: str, command: List[str], volumes: List[Tuple # Initialize a Docker client using environment variables try: - client = docker.from_env() + client = docker.from_env(timeout=600) except Exception as err: err.add_note("An error occurred when fetching the docker daemon: is docker installed and is dockerd running?") raise err diff --git a/test/analysis/input/lpca/pathway-params-1.txt b/test/analysis/input/lpca/pathway-params-1.txt new file mode 100644 index 000000000..affde9164 --- /dev/null +++ b/test/analysis/input/lpca/pathway-params-1.txt @@ -0,0 +1,6 @@ +Node1 Node2 Rank Direction +A B 1 U +B C 1 U +C D 1 U +D E 1 U +E F 1 U diff --git a/test/analysis/input/lpca/pathway-params-2.txt b/test/analysis/input/lpca/pathway-params-2.txt new file mode 100644 index 000000000..b1c72f8b4 --- /dev/null +++ b/test/analysis/input/lpca/pathway-params-2.txt @@ -0,0 +1,6 @@ +Node1 Node2 Rank Direction +A B 1 U +B C 1 U +C G 1 U +G H 1 U +H I 1 U diff --git a/test/analysis/input/lpca/pathway-params-3.txt b/test/analysis/input/lpca/pathway-params-3.txt new file mode 100644 index 000000000..8211a5a56 --- /dev/null +++ b/test/analysis/input/lpca/pathway-params-3.txt @@ -0,0 +1,6 @@ +Node1 Node2 Rank Direction +A B 1 U +D E 1 U +E F 1 U +F J 1 U +J K 1 U diff --git a/test/analysis/input/lpca/pathway-params-4.txt b/test/analysis/input/lpca/pathway-params-4.txt new file mode 100644 index 000000000..67f1cf554 --- /dev/null +++ b/test/analysis/input/lpca/pathway-params-4.txt @@ -0,0 +1,6 @@ +Node1 Node2 Rank Direction +B C 1 U +C D 1 U +G H 1 U +H I 1 U +I L 1 U diff --git a/test/analysis/test_lpca.py b/test/analysis/test_lpca.py new file mode 100644 index 000000000..8c3d004d0 --- /dev/null +++ b/test/analysis/test_lpca.py @@ -0,0 +1,87 @@ +from pathlib import Path + +import pandas as pd + +import spras.config.config as config +from spras.analysis.lpca import plot_lpca, run_lpca +from spras.analysis.ml import summarize_networks + +config.init_from_file("config/config.yaml") + +TEST_DIR = Path('test/analysis/') +OUT_DIR = TEST_DIR / 'output' + +INPUT_FILES = [ + 'test/analysis/input/lpca/pathway-params-1.txt', + 'test/analysis/input/lpca/pathway-params-2.txt', + 'test/analysis/input/lpca/pathway-params-3.txt', + 'test/analysis/input/lpca/pathway-params-4.txt', +] + +class TestLpca: + """ + Run Logistic PCA (LPCA) analysis tests + """ + @classmethod + def setup_class(cls): + OUT_DIR.mkdir(parents=True, exist_ok=True) + + def test_lpca_output_exists(self): + """Test that LPCA produces an output scores file""" + out_path = OUT_DIR / 'lpca-scores.csv' + matrix_path = OUT_DIR / 'lpca-binary-matrix.csv' + out_path.unlink(missing_ok=True) + + summary_df = summarize_networks(INPUT_FILES) + run_lpca( + dataframe=summary_df, + output_scores=str(out_path), + output_matrix=str(matrix_path), + k=2, + m=4, + cv=False, + ) + + assert out_path.exists(), "LPCA scores file was not created" + + def test_lpca_output_shape(self): + """Test that LPCA scores have shape (runs x k)""" + out_path = OUT_DIR / 'lpca-scores-shape.csv' + matrix_path = OUT_DIR / 'lpca-binary-matrix-shape.csv' + out_path.unlink(missing_ok=True) + + summary_df = summarize_networks(INPUT_FILES) + run_lpca( + dataframe=summary_df, + output_scores=str(out_path), + output_matrix=str(matrix_path), + k=2, + m=4, + cv=False, + ) + + scores = pd.read_csv(out_path, index_col=0) + assert scores.shape[1] == 2, f"Expected 2 PC columns, got {scores.shape[1]}" + assert scores.shape[0] == len(INPUT_FILES), \ + f"Expected {len(INPUT_FILES)} rows, got {scores.shape[0]}" + + def test_lpca_plot_output(self): + """Test that LPCA plot and coordinates files are created""" + scores_path = OUT_DIR / 'lpca-scores-plot.csv' + matrix_path = OUT_DIR / 'lpca-binary-matrix-plot.csv' + png_path = OUT_DIR / 'lpca-plot.png' + coord_path = OUT_DIR / 'lpca-coordinates.txt' + + summary_df = summarize_networks(INPUT_FILES) + run_lpca( + dataframe=summary_df, + output_scores=str(scores_path), + output_matrix=str(matrix_path), + k=2, + m=4, + cv=False, + ) + plot_lpca(str(scores_path), str(png_path), str(coord_path)) + + assert png_path.exists(), "LPCA plot PNG was not created" + assert coord_path.exists(), "LPCA coordinates file was not created"