Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
54 changes: 54 additions & 0 deletions .github/scripts/wheel_smoke_test.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,54 @@
"""Smoke-test an installed snputils wheel without using the source checkout."""

from importlib import import_module
from pathlib import Path
import platform

import snputils


project_root = Path(__file__).resolve().parents[2]
installed_package = Path(snputils.__file__).resolve()
try:
installed_package.relative_to(project_root)
except ValueError:
pass
else:
raise RuntimeError(f"snputils was imported from the source checkout: {installed_package}")

if snputils.__version__ == "unknown":
raise RuntimeError("The installed wheel does not expose package version metadata.")

for module_name in (
"snputils.snp.io.read._bcf",
"snputils.snp.io.write._bcf",
"snputils.snp.io._bgen",
):
import_module(module_name)

for public_name in (
"SNPObject",
"VCFReader",
"BCFReader",
"BGENReader",
"PGENReader",
"VCFWriter",
"BCFWriter",
"BGENWriter",
"PGENWriter",
):
getattr(snputils, public_name)

snpobj = snputils.build_synthetic_snp_dataset(n_samples=4, n_snps=8, seed=1)
if snpobj.n_samples != 4 or snpobj.n_snps != 8:
raise RuntimeError("The installed wheel failed the synthetic SNPObject smoke test.")

from snputils.tools.cli import main

if main(["version"]) != 0:
raise RuntimeError("The installed wheel's CLI smoke test failed.")

print(
f"Wheel smoke test passed: snputils {snputils.__version__}, "
f"{platform.system()} {platform.machine()}"
)
89 changes: 70 additions & 19 deletions .github/workflows/ci-cd.yml
Original file line number Diff line number Diff line change
Expand Up @@ -3,10 +3,11 @@
# 1. For Pull Requests:
# - Runs all tests
# - Verifies documentation builds
# - Builds and smoke-tests wheels on every supported platform
# 2. For Pre-releases:
# - Runs all tests
# - Builds documentation
# - Publishes package to PyPI
# - Publishes the tested wheels and source distribution to PyPI
# - Updates the release to a full release (removes pre-release status) and marks latest
# - Deploys documentation to GitHub Pages (at the end; non-blocking)

Expand Down Expand Up @@ -104,7 +105,21 @@ jobs:

# Build documentation and verify it builds correctly
- name: Build documentation
run: sphinx-build -b html docs docs/_build/html
run: sphinx-build -W --keep-going -b html docs docs/_build/html

- name: Verify documentation discovery files
run: |
test -s docs/_build/html/robots.txt
test -s docs/_build/html/sitemap.xml
grep -Fxq 'Sitemap: https://docs.snputils.org/sitemap.xml' docs/_build/html/robots.txt
grep -Fq '<loc>https://docs.snputils.org/' docs/_build/html/sitemap.xml
if grep -Eq '<loc>[^<]*/(search|genindex|py-modindex)\.html</loc>|<loc>[^<]*/_modules/' docs/_build/html/sitemap.xml; then
echo 'Sitemap contains an excluded utility page.' >&2
exit 1
fi

- name: Check documentation links
run: sphinx-build -b linkcheck docs docs/_build/linkcheck

# Upload Pages artifact if this is a release
- name: Upload Pages artifact
Expand All @@ -113,12 +128,51 @@ jobs:
with:
path: docs/_build/html

build-wheels:
needs: test
if: github.event_name == 'pull_request' || (github.event_name == 'release' && github.event.release.prerelease == true)
strategy:
fail-fast: false
matrix:
include:
- runner: ubuntu-latest
arch: x86_64
artifact: wheels-linux-x86_64
- runner: ubuntu-24.04-arm
arch: aarch64
artifact: wheels-linux-aarch64
- runner: macos-14
arch: arm64
artifact: wheels-macos-arm64
- runner: macos-15-intel
arch: x86_64
artifact: wheels-macos-x86_64
runs-on: ${{ matrix.runner }}
steps:
- name: Checkout repository
uses: actions/checkout@v6
with:
fetch-depth: 0

- name: Build and smoke-test wheels
uses: pypa/cibuildwheel@v4.1.0
with:
output-dir: wheelhouse
env:
CIBW_ARCHS: ${{ matrix.arch }}

- name: Upload wheels
uses: actions/upload-artifact@v4
with:
name: ${{ matrix.artifact }}
path: wheelhouse/*.whl
if-no-files-found: error

publish:
needs: [test, build-docs]
if: needs.build-docs.result == 'success' && github.event_name == 'release' && github.event.release.prerelease == true
needs: [test, build-docs, build-wheels]
if: needs.build-docs.result == 'success' && needs.build-wheels.result == 'success' && github.event_name == 'release' && github.event.release.prerelease == true
runs-on: ubuntu-latest
steps:
# Build and publish to PyPI
- name: Checkout repository
uses: actions/checkout@v6
with:
Expand All @@ -136,23 +190,20 @@ jobs:
- name: Install build dependencies
run: |
python -m pip install --upgrade pip
python -m pip install build

- name: Build sdist
run: |
rm -rf dist wheelhouse
python -m build --sdist
python -m pip install build twine

- name: Build wheels
uses: pypa/cibuildwheel@v4.1.0
- name: Download tested wheels
uses: actions/download-artifact@v4
with:
output-dir: wheelhouse
env:
CIBW_BUILD: "cp39-* cp310-* cp311-* cp312-* cp313-* cp314-*"
CIBW_SKIP: "*-musllinux_*"
pattern: wheels-*
path: dist
merge-multiple: true

- name: Build sdist
run: python -m build --sdist --outdir dist

- name: Move wheels into dist
run: cp wheelhouse/*.whl dist/
- name: Verify distributions
run: python -m twine check dist/*

- name: Publish package to PyPI
uses: pypa/gh-action-pypi-publish@release/v1
Expand Down
3 changes: 2 additions & 1 deletion .gitignore
Original file line number Diff line number Diff line change
Expand Up @@ -8,6 +8,7 @@ __pycache__/
*$py.class
*.lock
*.txt
!docs/_extra/robots.txt

# C extensions
*.so
Expand Down Expand Up @@ -161,4 +162,4 @@ snputils/tools/__test__/test_admixture_mapping_internal_large.py
*.md
!README.md
!docs/**/*.md
.benchmarks/
.benchmarks/
25 changes: 24 additions & 1 deletion README.md
Original file line number Diff line number Diff line change
Expand Up @@ -27,6 +27,29 @@ Developed in collaboration between Stanford University's Department of Biomedica

## Quickstart

Start with a fully self-contained synthetic dataset—no external files or downloads required:

```python
import snputils as su

snp = su.build_synthetic_snp_dataset(
n_samples=30,
n_snps=100,
seed=42,
)

af = snp.allele_freq()
pcs = su.PCA(n_components=2).fit_transform(snp)

print(snp.n_samples, snp.n_snps)
print(af[:5])
print(pcs.shape)
```

### Working with your own files

For file-backed workflows, `read_snp` detects the genotype format from its extension and the other readers load ancestry, phenotype, and IBD data:

```python
import snputils as su

Expand Down Expand Up @@ -182,7 +205,7 @@ The Python API remains the full surface for low-level readers/writers, object ma
## Documentation and Examples

- **Documentation**: [docs.snputils.org](https://docs.snputils.org)
- **Quickstart**: [Quickstart guide](https://docs.snputils.org/en/latest/quickstart.html)
- **Quickstart**: [Quickstart guide](https://docs.snputils.org/quickstart.html)
- **Tutorials**: PCA, mdPCA, maasMDS, SNP objects, allele frequency, local ancestry visualization, admixture mapping, and GRG workflows
- **API Reference**: Readers, writers, data objects, processing classes, statistics, datasets, and visualization helpers
- **Issues and feature requests**: [GitHub Issues](https://github.com/AI-sandbox/snputils/issues)
Expand Down
13 changes: 12 additions & 1 deletion benchmark/README.md
Original file line number Diff line number Diff line change
Expand Up @@ -8,7 +8,16 @@ On the chromosome 22 of the 1000 Genomes Project dataset:

**Legend: `genotype_mode`**:
- 🟦 `"dosage"`: ALT dosage (`0|0 -> 0`, `0|1 -> 1`, `1|0 -> 1`, `1|1 -> 2`)
- 🟧 `"phased"`: allele codes (`0|0 -> [0, 0]`, `0|1 -> [0, 1]`, `1|0 -> [1, 0]`, `1|1 -> [1, 1]`)
- 🟧 `"phased"`: allele codes for PGEN, VCF, and BCF (`0|1 -> [0, 1]`); the BGEN panel instead uses `"probabilities"` to materialize the complete `float32` genotype-probability (`GP`) tensor

The BGEN GP results are:

| Reader | Time (mean ± SD) | Peak memory (mean ± SD) |
| --- | ---: | ---: |
| **snputils** | 13.31 ± 1.11 s | 28.5082 ± 0.0002 GiB |
| bgen | 16.49 ± 0.23 s | 28.3963 ± 0.0002 GiB |
| pysnptools | 51.49 ± 1.33 s | 28.5433 ± 0.0003 GiB |
| sgkit | 58.05 ± 1.94 s | 56.8980 ± 0.2273 GiB |

## Methodology

Expand All @@ -17,6 +26,8 @@ The reader benchmark measures wall-clock read time and peak memory on chromosome
Each reader is run in an Slurm allocation with 8 AMD EPYC 9684X CPU cores and 200 GB of RAM and a Python 3.12 environment.
Python 3.12 is used because some libraries are not yet compatible with Python 3.13 or 3.14.

For BGEN, the dosage condition materializes expected alternate-allele counts from biallelic probabilities, while the orange condition materializes the complete `float32` genotype-probability tensor. The input contains 993,881 variants and 2,548 samples. This permits a like-for-like comparison across BGEN readers, including libraries that do not preserve phase. Every completed BGEN reader is checked against the corresponding output from the independent `bgen` package outside the timed and memory-profiled region. The correctness comparison is performed in variant chunks so that it does not create a second full-size comparison mask. Hail is omitted because its first GP materialization exhausted the fixed 200 GB allocation before returning the tensor.

After running the benchmark, the plotting utility `benchmark/plot_time_memory.py` writes both `benchmark/readers_benchmark.png` and `benchmark/readers_benchmark.pdf` from existing benchmark JSON files.

## Contributing
Expand Down
14 changes: 10 additions & 4 deletions benchmark/conftest.py
Original file line number Diff line number Diff line change
Expand Up @@ -27,8 +27,11 @@ def pytest_addoption(parser):
"--genotype-mode",
action="store",
default="dosage",
choices=("dosage", "phased"),
help="Whether readers should return genotype dosages or phased allele calls."
choices=("dosage", "probabilities", "phased"),
help=(
"Whether readers should return genotype dosages, genotype probabilities, "
"or phased allele calls. Probability output is currently benchmarked for BGEN."
),
)


Expand All @@ -55,5 +58,8 @@ def reader_name(request):

@pytest.fixture
def genotype_mode(request):
"""Fixture selecting dosage or phased genotype output."""
return request.config.getoption("--genotype-mode")
"""Fixture selecting the materialized genotype representation."""
mode = request.config.getoption("--genotype-mode")
if mode == "probabilities" and request.node.path.name != "read_bgen.py":
pytest.skip('genotype_mode="probabilities" is only supported by the BGEN benchmark.')
return mode
17 changes: 14 additions & 3 deletions benchmark/plot_time_memory.py
Original file line number Diff line number Diff line change
Expand Up @@ -129,6 +129,7 @@ def _draw_grouped_bars(
y_cap: float | None = None,
axis_top: float | None = None,
show_phased: bool = True,
secondary_label: str = 'genotype_mode="phased"',
) -> None:
x = np.arange(len(names))
width = 0.36 if show_phased else 0.42
Expand Down Expand Up @@ -170,7 +171,7 @@ def _draw_grouped_bars(

series = (
(-width / 2, dosage_values, 'genotype_mode="dosage"', colors["dosage"]),
(width / 2, phased_values, 'genotype_mode="phased"', colors["phased"]),
(width / 2, phased_values, secondary_label, colors["phased"]),
) if show_phased else (
(0.0, dosage_values, 'genotype_mode="dosage"', colors["dosage"]),
)
Expand Down Expand Up @@ -278,19 +279,29 @@ def plot_time_memory(
time_dosage,
time_phased,
"time",
f"{title} time",
f"{title} time" + (" (orange: GP)" if fmt == "bgen" else ""),
axis_top=600 if fmt == "vcf" else None,
show_phased=fmt != "bed",
secondary_label=(
'genotype_mode="probabilities"'
if fmt == "bgen"
else 'genotype_mode="phased"'
),
)
_draw_grouped_bars(
axs[row, 1],
memory_names,
memory_dosage,
memory_phased,
"memory",
f"{title} peak memory",
f"{title} peak memory" + (" (orange: GP)" if fmt == "bgen" else ""),
y_cap=memory_y_cap,
show_phased=fmt != "bed",
secondary_label=(
'genotype_mode="probabilities"'
if fmt == "bgen"
else 'genotype_mode="phased"'
),
)
axs[row, 0].set_ylabel("Time (seconds)")
axs[row, 1].set_ylabel("Peak memory (GiB)")
Expand Down
Loading
Loading