From b35b2c2a12d8d96884f4246c882c9cd1563473cb Mon Sep 17 00:00:00 2001 From: Martin Garrido Date: Fri, 24 Jul 2026 13:13:43 +0200 Subject: [PATCH 1/4] Add CI workflow and fix dependency conflict - Remove pytorch-lightning>=1.9,<2.0 (conflicts with lightning>=2.0; the lightning package already ships pytorch_lightning as a compat shim) - Widen lightning bound to <3.0 so pip resolves against current releases - Add [test] extra with pytest - Add .github/workflows/ci.yml: installs CPU-only torch first to avoid the CUDA wheel, then runs the full test suite on Python 3.10 and 3.11 - Add tests/create_mini_mzml.py: generates a 41 KB synthetic DIA mzML (5 cycles x 4 spectra) to exercise the augmentation pipeline in CI - Add tests/create_dummy_ckpt.py: saves a random-weight production-arch checkpoint (~190 MB) cached across runs via actions/cache - Add tests/test_install.py: import, mzML parsing, augment_spectra, tokenizer smoke tests (no GPU or checkpoint required) - Add tests/test_e2e.py: full cascadia sequence pipeline test using the cached dummy checkpoint (skipped if CASCADIA_DUMMY_CKPT is unset) --- .github/workflows/ci.yml | 68 +++++++++++++++++ pyproject.toml | 6 +- tests/__init__.py | 0 tests/create_dummy_ckpt.py | 59 ++++++++++++++ tests/create_mini_mzml.py | 153 +++++++++++++++++++++++++++++++++++++ tests/test_e2e.py | 44 +++++++++++ tests/test_install.py | 101 ++++++++++++++++++++++++ 7 files changed, 429 insertions(+), 2 deletions(-) create mode 100644 .github/workflows/ci.yml create mode 100644 tests/__init__.py create mode 100644 tests/create_dummy_ckpt.py create mode 100644 tests/create_mini_mzml.py create mode 100644 tests/test_e2e.py create mode 100644 tests/test_install.py diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml new file mode 100644 index 0000000..343c0cb --- /dev/null +++ b/.github/workflows/ci.yml @@ -0,0 +1,68 @@ +name: CI + +on: + push: + pull_request: + +permissions: + contents: read + +jobs: + test: + name: Install & smoke-test (Python ${{ matrix.python-version }}) + runs-on: ubuntu-latest + + strategy: + fail-fast: false + matrix: + python-version: ["3.10", "3.11"] + + env: + # Stable cache key – bump the suffix if the architecture changes. + CKPT_CACHE_KEY: cascadia-dummy-ckpt-arch-v1 + CKPT_PATH: ~/.cache/cascadia/dummy.ckpt + + steps: + - uses: actions/checkout@v4 + + - name: Set up Python ${{ matrix.python-version }} + uses: actions/setup-python@v5 + with: + python-version: ${{ matrix.python-version }} + cache: pip + + # Install PyTorch CPU-only first so pip does not pull in the multi-GB + # CUDA wheel when it later resolves cascadia's torch dependency. + - name: Install PyTorch (CPU-only) + run: | + pip install --upgrade pip + pip install torch --index-url https://download.pytorch.org/whl/cpu + + - name: Install cascadia and its dependencies + run: pip install ".[test]" + + # Quick import smoke-test before running the full suite. + - name: Verify import + run: python -c "import cascadia; print('cascadia imported OK')" + + # Restore a previously generated dummy checkpoint. The key is stable + # (tied only to the model architecture version) so it is shared across + # branches and only regenerated when the architecture changes. + - name: Restore dummy checkpoint cache + id: ckpt-cache + uses: actions/cache@v4 + with: + path: ~/.cache/cascadia + key: ${{ env.CKPT_CACHE_KEY }}-py${{ matrix.python-version }} + + - name: Generate dummy checkpoint (cache miss only) + if: steps.ckpt-cache.outputs.cache-hit != 'true' + run: python tests/create_dummy_ckpt.py "${{ env.CKPT_PATH }}" + + # Run the full test suite. The augmentation tests exercise the mzML + # parsing and ASF generation path; the e2e test drives the complete + # ``cascadia sequence`` pipeline using the dummy checkpoint. + - name: Run tests + env: + CASCADIA_DUMMY_CKPT: ${{ env.CKPT_PATH }} + run: pytest tests/ -v --tb=short diff --git a/pyproject.toml b/pyproject.toml index a19f8e1..fd96e89 100755 --- a/pyproject.toml +++ b/pyproject.toml @@ -14,8 +14,7 @@ classifiers = [ requires-python = ">=3.8" dependencies = [ "setuptools<70.0.0", - "lightning>=2.0,<2.1", - "pytorch-lightning>=1.9,<2.0", + "lightning>=2.0,<3.0", "pyteomics>=4.6", "torch>=2.2.0", "numpy<2.0", @@ -34,6 +33,9 @@ dependencies = [ "tensorboard", ] +[project.optional-dependencies] +test = ["pytest>=7.0"] + [project.scripts] cascadia = "cascadia.cascadia:main" diff --git a/tests/__init__.py b/tests/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/tests/create_dummy_ckpt.py b/tests/create_dummy_ckpt.py new file mode 100644 index 0000000..4fd8ed0 --- /dev/null +++ b/tests/create_dummy_ckpt.py @@ -0,0 +1,59 @@ +"""Create a random-weight Cascadia checkpoint for CI end-to-end testing. + +The checkpoint uses the *production* architecture (d_model=512, n_layers=9) +so that ``cascadia sequence`` can load it without modification. Weights are +randomly initialised with a fixed seed so the file is reproducible. + +Usage (called by the CI workflow): + python tests/create_dummy_ckpt.py +""" +import sys +from pathlib import Path + +import torch +import pytorch_lightning as pl + + +def create_dummy_ckpt(output_path: str | Path) -> Path: + output_path = Path(output_path) + output_path.parent.mkdir(parents=True, exist_ok=True) + + torch.manual_seed(0) + + from cascadia.depthcharge.tokenizers import PeptideTokenizer + from cascadia.model import AugmentedSpec2Pep + + tokenizer = PeptideTokenizer.from_massivekb( + reverse=False, replace_isoleucine_with_leucine=True + ) + + model = AugmentedSpec2Pep( + d_model=512, + n_layers=9, + n_head=8, + dim_feedforward=1024, + dropout=0, + rt_width=2, + tokenizer=tokenizer, + max_charge=10, + ) + + trainer = pl.Trainer( + max_epochs=1, + accelerator="cpu", + devices=1, + enable_progress_bar=False, + enable_model_summary=False, + logger=False, + ) + trainer.strategy.connect(model) + trainer.save_checkpoint(str(output_path)) + + size_mb = output_path.stat().st_size / 1e6 + print(f"Saved dummy checkpoint to {output_path} ({size_mb:.1f} MB)") + return output_path + + +if __name__ == "__main__": + out = Path(sys.argv[1]) if len(sys.argv) > 1 else Path("dummy.ckpt") + create_dummy_ckpt(out) diff --git a/tests/create_mini_mzml.py b/tests/create_mini_mzml.py new file mode 100644 index 0000000..7c2fec8 --- /dev/null +++ b/tests/create_mini_mzml.py @@ -0,0 +1,153 @@ +"""Generate a minimal synthetic DIA mzML file for CI testing. + +The file contains 5 DIA cycles (5 MS1 + 15 MS2 = 20 spectra total) with +realistic isolation windows and small random peak arrays. +""" +import base64 +import struct +from pathlib import Path + +import numpy as np + +RNG = np.random.default_rng(42) + +# DIA isolation windows (center, half-width) in m/z +WINDOWS = [(412.5, 12.5), (437.5, 12.5), (462.5, 12.5)] +N_CYCLES = 5 +CYCLE_INTERVAL_MIN = 0.5 # 30 s per cycle +N_MS1_PEAKS = 60 +N_MS2_PEAKS = 40 + + +def _encode_float32(arr: np.ndarray) -> str: + data = struct.pack(f"{len(arr)}f", *arr.astype(np.float64)) + return base64.b64encode(data).decode("ascii") + + +def _binary_array(arr: np.ndarray, accession: str, name: str) -> str: + encoded = _encode_float32(arr) + return ( + f' \n' + f' \n' + f' \n' + f' \n' + f" {encoded}\n" + f" " + ) + + +def _ms1_spectrum(idx: int, rt_min: float) -> str: + mz = np.sort(RNG.uniform(300, 1200, N_MS1_PEAKS)).astype(np.float32) + intensity = RNG.exponential(1e6, N_MS1_PEAKS).astype(np.float32) + mz_block = _binary_array(mz, "MS:1000514", "m/z array") + int_block = _binary_array(intensity, "MS:1000515", "intensity array") + return f"""\ + + + + + + + + +{mz_block} +{int_block} + + """ + + +def _ms2_spectrum( + idx: int, rt_min: float, center: float, lower: float, upper: float +) -> str: + mz = np.sort(RNG.uniform(100, 1500, N_MS2_PEAKS)).astype(np.float32) + intensity = RNG.exponential(1e5, N_MS2_PEAKS).astype(np.float32) + mz_block = _binary_array(mz, "MS:1000514", "m/z array") + int_block = _binary_array(intensity, "MS:1000515", "intensity array") + return f"""\ + + + + + + + + + + + + + + + + + +{mz_block} +{int_block} + + """ + + +def create_mini_mzml(output_path: str | Path) -> Path: + """Write a minimal DIA mzML to *output_path* and return the path.""" + output_path = Path(output_path) + spectra = [] + idx = 0 + for cycle in range(N_CYCLES): + rt_ms1 = (cycle + 1) * CYCLE_INTERVAL_MIN + spectra.append(_ms1_spectrum(idx, rt_ms1)) + idx += 1 + for win_idx, (center, hw) in enumerate(WINDOWS): + rt_ms2 = rt_ms1 + (win_idx + 1) * 0.005 + spectra.append(_ms2_spectrum(idx, rt_ms2, center, hw, hw)) + idx += 1 + + n_total = len(spectra) + spectrum_block = "\n".join(spectra) + + xml = f"""\ + + + + + + + + + + + + + + + + + + + + + + + + + +{spectrum_block} + + + +""" + output_path.write_text(xml, encoding="utf-8") + return output_path + + +if __name__ == "__main__": + import sys + + out = Path(sys.argv[1]) if len(sys.argv) > 1 else Path("mini_demo.mzML") + p = create_mini_mzml(out) + print(f"Wrote {p} ({p.stat().st_size} bytes)") diff --git a/tests/test_e2e.py b/tests/test_e2e.py new file mode 100644 index 0000000..955798e --- /dev/null +++ b/tests/test_e2e.py @@ -0,0 +1,44 @@ +"""End-to-end test: run ``cascadia sequence`` with a dummy checkpoint. + +Requires the dummy checkpoint to exist at the path given by the +CASCADIA_DUMMY_CKPT environment variable (set by the CI workflow). +Skipped automatically when the variable is absent (e.g. local dev without +the checkpoint). +""" +import os +import sys +from pathlib import Path + +import pytest + +TESTS_DIR = Path(__file__).parent +CKPT_PATH = os.environ.get("CASCADIA_DUMMY_CKPT", "") + + +@pytest.mark.skipif(not CKPT_PATH, reason="CASCADIA_DUMMY_CKPT not set") +def test_sequence_command(tmp_path): + """Full ``cascadia sequence`` pipeline on the mini mzML with a dummy checkpoint.""" + sys.path.insert(0, str(TESTS_DIR)) + from create_mini_mzml import create_mini_mzml + + mzml_file = create_mini_mzml(tmp_path / "mini_demo.mzML") + out_prefix = str(tmp_path / "ci_results") + + # Drive the CLI entry-point directly so we test the real code path. + sys.argv = [ + "cascadia", + "sequence", + str(mzml_file), + CKPT_PATH, + "--out", out_prefix, + "--batch_size", "4", + "--max_charge", "2", + ] + + from cascadia.cascadia import sequence + sequence() + + ssl_file = Path(out_prefix + ".ssl") + assert ssl_file.exists(), f"Expected output file {ssl_file} was not created" + content = ssl_file.read_text() + assert len(content) > 0, "Output SSL file is empty" diff --git a/tests/test_install.py b/tests/test_install.py new file mode 100644 index 0000000..0b73718 --- /dev/null +++ b/tests/test_install.py @@ -0,0 +1,101 @@ +"""Smoke tests that run on every push. + +These tests: +1. Verify the package imports correctly. +2. Parse a tiny synthetic mzML file through the augmentation pipeline, + which exercises pyteomics, numpy, and the core DIA data-prep logic — + all without requiring a GPU or a model checkpoint. +""" +import importlib +import os +import sys +import tempfile +from pathlib import Path + +import pytest + +# --------------------------------------------------------------------------- +# Helpers +# --------------------------------------------------------------------------- + +TESTS_DIR = Path(__file__).parent + + +def _make_mini_mzml(tmp_path: Path) -> Path: + """Generate the mini mzML fixture inside *tmp_path*.""" + sys.path.insert(0, str(TESTS_DIR)) + from create_mini_mzml import create_mini_mzml + + return create_mini_mzml(tmp_path / "mini_demo.mzML") + + +# --------------------------------------------------------------------------- +# Tests +# --------------------------------------------------------------------------- + + +def test_package_imports(): + """Top-level import must succeed.""" + import cascadia # noqa: F401 + + assert cascadia.__version__ is not None or True # version may be None in editable install + + +def test_core_submodules_import(): + """All critical internal modules must be importable.""" + modules = [ + "cascadia.model", + "cascadia.augment", + "cascadia.depthcharge.tokenizers", + "cascadia.depthcharge.data.spectrum_datasets", + "cascadia.depthcharge.data.preprocessing", + ] + for mod in modules: + importlib.import_module(mod) + + +def test_mzml_parsing(tmp_path): + """pyteomics must be able to read the synthetic mzML without errors.""" + from pyteomics import mzml + + mzml_file = _make_mini_mzml(tmp_path) + spectra = list(mzml.read(str(mzml_file))) + ms1 = [s for s in spectra if s["ms level"] == 1] + ms2 = [s for s in spectra if s["ms level"] == 2] + assert len(ms1) == 5, f"Expected 5 MS1 scans, got {len(ms1)}" + assert len(ms2) == 15, f"Expected 15 MS2 scans, got {len(ms2)}" + # Verify array shapes are non-empty + for s in spectra: + assert len(s["m/z array"]) > 0 + assert len(s["intensity array"]) > 0 + + +def test_augment_spectra(tmp_path): + """augment_spectra must produce a non-empty .asf file from the mini mzML.""" + from cascadia.augment import augment_spectra + + mzml_file = _make_mini_mzml(tmp_path) + temp_dir = tmp_path / "augment_out" + temp_dir.mkdir() + + asf_file, window_size, cycle_time = augment_spectra( + str(mzml_file), + str(temp_dir), + max_charge=2, + ) + + assert Path(asf_file).exists(), "ASF output file was not created" + content = Path(asf_file).read_text() + assert "BEGIN IONS" in content, "ASF file contains no spectra" + assert cycle_time is not None and cycle_time > 0, "cycle_time must be positive" + assert window_size > 0, "isolation window_size must be positive" + + +def test_tokenizer_loads(): + """The MassIVE-KB tokenizer must load without error.""" + from cascadia.depthcharge.tokenizers import PeptideTokenizer + + tok = PeptideTokenizer.from_massivekb( + reverse=False, replace_isoleucine_with_leucine=True + ) + assert len(tok) > 0 From db02cccf5aefa98a94de7c5133d1cacf3d714107 Mon Sep 17 00:00:00 2001 From: Martin Garrido Date: Fri, 24 Jul 2026 13:30:20 +0200 Subject: [PATCH 2/4] Fix CI checkpoint generation and path expansion Two bugs from the first CI run: 1. trainer.strategy.connect() + trainer.save_checkpoint() requires Lightning to have completed internal fit setup, which does not happen without an actual training call. Replaced with a direct torch.save() of the Lightning checkpoint dict (load_from_checkpoint only needs state_dict when all hyperparams are supplied as kwargs). 2. ~ in YAML env vars is not shell-expanded, so "${{ env.CKPT_PATH }}" passed the literal string ~/... to Python, causing mkdir to create a directory named ~. Fixed by resolving the path via $HOME in a dedicated workflow step that writes to GITHUB_ENV, and added Path.expanduser() in both Python scripts as a defensive measure. --- .github/workflows/ci.yml | 10 +++++++--- tests/create_dummy_ckpt.py | 26 +++++++++++++++----------- tests/test_e2e.py | 2 +- 3 files changed, 23 insertions(+), 15 deletions(-) diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 343c0cb..89c2f00 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -20,7 +20,6 @@ jobs: env: # Stable cache key – bump the suffix if the architecture changes. CKPT_CACHE_KEY: cascadia-dummy-ckpt-arch-v1 - CKPT_PATH: ~/.cache/cascadia/dummy.ckpt steps: - uses: actions/checkout@v4 @@ -31,6 +30,11 @@ jobs: python-version: ${{ matrix.python-version }} cache: pip + # Resolve the checkpoint path with the real home dir so Python never + # receives a literal ~ that Path.mkdir() would treat as a directory name. + - name: Set checkpoint path + run: echo "CKPT_PATH=$HOME/.cache/cascadia/dummy.ckpt" >> "$GITHUB_ENV" + # Install PyTorch CPU-only first so pip does not pull in the multi-GB # CUDA wheel when it later resolves cascadia's torch dependency. - name: Install PyTorch (CPU-only) @@ -52,12 +56,12 @@ jobs: id: ckpt-cache uses: actions/cache@v4 with: - path: ~/.cache/cascadia + path: ${{ env.CKPT_PATH }} key: ${{ env.CKPT_CACHE_KEY }}-py${{ matrix.python-version }} - name: Generate dummy checkpoint (cache miss only) if: steps.ckpt-cache.outputs.cache-hit != 'true' - run: python tests/create_dummy_ckpt.py "${{ env.CKPT_PATH }}" + run: python tests/create_dummy_ckpt.py "$CKPT_PATH" # Run the full test suite. The augmentation tests exercise the mzML # parsing and ASF generation path; the e2e test drives the complete diff --git a/tests/create_dummy_ckpt.py b/tests/create_dummy_ckpt.py index 4fd8ed0..6cc7d64 100644 --- a/tests/create_dummy_ckpt.py +++ b/tests/create_dummy_ckpt.py @@ -4,6 +4,12 @@ so that ``cascadia sequence`` can load it without modification. Weights are randomly initialised with a fixed seed so the file is reproducible. +We write the Lightning checkpoint format directly with ``torch.save`` rather +than going through ``Trainer.save_checkpoint``, which requires the trainer to +have completed the internal fit-setup chain (unavailable without a training +run). ``LightningModule.load_from_checkpoint`` only requires ``state_dict`` +in the checkpoint dict when all hyperparameters are supplied as kwargs. + Usage (called by the CI workflow): python tests/create_dummy_ckpt.py """ @@ -15,7 +21,7 @@ def create_dummy_ckpt(output_path: str | Path) -> Path: - output_path = Path(output_path) + output_path = Path(output_path).expanduser() output_path.parent.mkdir(parents=True, exist_ok=True) torch.manual_seed(0) @@ -38,16 +44,14 @@ def create_dummy_ckpt(output_path: str | Path) -> Path: max_charge=10, ) - trainer = pl.Trainer( - max_epochs=1, - accelerator="cpu", - devices=1, - enable_progress_bar=False, - enable_model_summary=False, - logger=False, - ) - trainer.strategy.connect(model) - trainer.save_checkpoint(str(output_path)) + checkpoint = { + "epoch": 0, + "global_step": 0, + "pytorch-lightning_version": pl.__version__, + "state_dict": model.state_dict(), + "hyper_parameters": {}, + } + torch.save(checkpoint, str(output_path)) size_mb = output_path.stat().st_size / 1e6 print(f"Saved dummy checkpoint to {output_path} ({size_mb:.1f} MB)") diff --git a/tests/test_e2e.py b/tests/test_e2e.py index 955798e..21587a1 100644 --- a/tests/test_e2e.py +++ b/tests/test_e2e.py @@ -12,7 +12,7 @@ import pytest TESTS_DIR = Path(__file__).parent -CKPT_PATH = os.environ.get("CASCADIA_DUMMY_CKPT", "") +CKPT_PATH = str(Path(os.environ["CASCADIA_DUMMY_CKPT"]).expanduser()) if os.environ.get("CASCADIA_DUMMY_CKPT") else "" @pytest.mark.skipif(not CKPT_PATH, reason="CASCADIA_DUMMY_CKPT not set") From 5586bc207839519642c83926451692702f6b84a3 Mon Sep 17 00:00:00 2001 From: Martin Garrido Date: Fri, 24 Jul 2026 14:18:54 +0200 Subject: [PATCH 3/4] Add missing psims dependency for pyteomics mzML parsing pyteomics>=4.7 requires psims for PSI format (mzML/mzXML) parsing. Was present locally as a transitive dep but missing from the declared dependencies, causing CI to fail on a clean install. --- pyproject.toml | 1 + 1 file changed, 1 insertion(+) diff --git a/pyproject.toml b/pyproject.toml index fd96e89..636b309 100755 --- a/pyproject.toml +++ b/pyproject.toml @@ -16,6 +16,7 @@ dependencies = [ "setuptools<70.0.0", "lightning>=2.0,<3.0", "pyteomics>=4.6", + "psims", "torch>=2.2.0", "numpy<2.0", "numba>=0.48.0", From 0ec7f862c371ad1ee11ae9fc2680d0ba11354171 Mon Sep 17 00:00:00 2001 From: Martin Garrido Date: Fri, 24 Jul 2026 14:28:58 +0200 Subject: [PATCH 4/4] Fix missing write_results import in cascadia.py MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit write_results is defined in cascadia/utils.py but was never imported — only .depthcharge.utils was pulled in via wildcard import. The function call on line 87 raised NameError at runtime. Exposed by CI running the full sequence() pipeline end-to-end. --- cascadia/cascadia.py | 1 + 1 file changed, 1 insertion(+) diff --git a/cascadia/cascadia.py b/cascadia/cascadia.py index aae6554..505cd76 100755 --- a/cascadia/cascadia.py +++ b/cascadia/cascadia.py @@ -9,6 +9,7 @@ import argparse from lightning.pytorch import loggers as pl_loggers from .depthcharge.utils import * +from .utils import write_results from .model import AugmentedSpec2Pep from .augment import * from datetime import datetime