diff --git a/.github/workflows/python_package.yml b/.github/workflows/python_package.yml index 34d45d3..8249833 100644 --- a/.github/workflows/python_package.yml +++ b/.github/workflows/python_package.yml @@ -17,7 +17,7 @@ jobs: uses: mamba-org/setup-micromamba@v2 with: environment-file: environment.yml - environment-name: sprite + environment-name: wisp create-args: >- python=${{ matrix.python-version }} init-shell: bash diff --git a/.gitignore b/.gitignore index fbe0a42..671decc 100644 --- a/.gitignore +++ b/.gitignore @@ -1,7 +1,7 @@ .DS_Store .git.broken -src/sprite_mask/__pycache__/ +src/wisp_mask/__pycache__/ __pycache__/ *.py[cod] *.py.tmp.* diff --git a/.readthedocs.yaml b/.readthedocs.yaml index 9e5d069..93a4e08 100644 --- a/.readthedocs.yaml +++ b/.readthedocs.yaml @@ -1,4 +1,4 @@ -# Read the Docs configuration file for sprite. +# Read the Docs configuration file for wisp. # See https://docs.readthedocs.io/en/stable/config-file/v2.html version: 2 diff --git a/README.md b/README.md index be051b5..2169662 100644 --- a/README.md +++ b/README.md @@ -1,31 +1,31 @@ -`sprite` +`wisp` ==================== -`sprite` builds population count masks from BAM/CRAM alignments or all-sites VCFs. It is a companion tool for [pixy](https://github.com/ksamuk/pixy/), though it works equally well on its own. +`wisp` builds population count masks from BAM/CRAM alignments or all-sites VCFs. It is a companion tool for [pixy](https://github.com/ksamuk/pixy/), though it works equally well on its own. Population count masks let you correctly compute the denominators of π, dxy, Watterson's θ, and Tajima's D when working from a variants-only VCF — callable sites are counted per population rather than collapsed into a single cohort-wide pass/fail. -Sprite is inspired by [mop](https://github.com/RILAB/mop), [clam](https://github.com/cademirch/clam), and makes heavy use of [mosdepth](https://github.com/brentp/mosdepth). +Wisp is inspired by [mop](https://github.com/RILAB/mop), [clam](https://github.com/cademirch/clam), and makes heavy use of [mosdepth](https://github.com/brentp/mosdepth). -> **Note:** `sprite` is pre-release (alpha) and pending validation. It will be distributed on Bioconda as `sprite-mask` once stable. +> **Note:** `wisp` is pre-release (alpha) and pending validation. It will be distributed on Bioconda as `wisp-mask` once stable. ## Installation ```bash -mamba create -n sprite python=3.11 pip git -c conda-forge -mamba activate sprite +mamba create -n wisp python=3.11 pip git -c conda-forge +mamba activate wisp mamba install -c conda-forge samtools bcftools htslib mosdepth -python -m pip install "git+https://github.com/samuk-lab/sprite.git" +python -m pip install "git+https://github.com/samuk-lab/wisp.git" ``` ## Usage ### From BAM/CRAM alignments -`sprite` runs [mosdepth](https://github.com/brentp/mosdepth) on each sample and collapses per-sample pass intervals into a population count mask: +`wisp` runs [mosdepth](https://github.com/brentp/mosdepth) on each sample and collapses per-sample pass intervals into a population count mask: ```bash -sprite from-alignments \ +wisp from-alignments \ --samples samples.tsv \ --variants-vcf variants.vcf.gz \ --out results \ @@ -43,7 +43,7 @@ sample_2 popB /path/sample_2.cram If BAM/CRAM read groups include sample names, they must match the corresponding `sample_id`. -When `--variants-vcf` is provided, `sprite` uses the variants-only VCF to fill in omitted +When `--variants-vcf` is provided, `wisp` uses the variants-only VCF to fill in omitted alignment thresholds: `--min-dp`/`--max-dp` from per-sample `FORMAT/DP` and `--min-mapq` from `INFO/MQ` when those fields are available. Manually supplied threshold flags take precedence. The same VCF is also scanned for indels, symbolic structural variants, breakends, @@ -54,7 +54,7 @@ population. ### From an all-sites VCF ```bash -sprite from-vcf \ +wisp from-vcf \ --all-sites-vcf all_sites.vcf.gz \ --popfile populations.tsv \ --min-dp 10 \ @@ -73,19 +73,19 @@ sample_2 popB A few things to know about VCF mode: - Every sample in the population file must appear in the VCF. VCF samples absent from the population file are ignored (with a warning). -- Records may carry any `FILTER` value — the input is assumed to have been filtered as desired before running `sprite`. +- Records may carry any `FILTER` value — the input is assumed to have been filtered as desired before running `wisp`. - A sample passes a site when `FORMAT/DP >= --min-dp`, if supplied `FORMAT/DP <= --max-dp`, and, when `FORMAT/GT` is present, the genotype is not missing. - At duplicate `CHROM:POS` records, a sample passes a site if any duplicate passes the depth thresholds. Duplicates must be contiguous, as in a coordinate-sorted VCF. - By default, all record types are used (SNPs, indels, symbolic alleles, invariant sites). Pass `--snps-only` to exclude indel sites while retaining invariant sites. ## Output -`sprite` writes two files to `--out`: +`wisp` writes two files to `--out`: | File | Description | |---|---| -| `sprite.bed.gz` | bgzip-compressed, tabix-indexed population count mask | -| `sprite.bed.gz.tbi` | tabix index | +| `wisp.bed.gz` | bgzip-compressed, tabix-indexed population count mask | +| `wisp.bed.gz.tbi` | tabix index | Use `--output-prefix` to change the filename stem; `.bed.gz` is always appended. diff --git a/docs/about.rst b/docs/about.rst index a977552..1d2d1ad 100644 --- a/docs/about.rst +++ b/docs/about.rst @@ -1,9 +1,9 @@ About ***** -``sprite`` creates population count masks for population genomic workflows. +``wisp`` creates population count masks for population genomic workflows. Where a conventional depth mask gives a single cohort-wide pass/fail per -site, ``sprite`` reports how many samples in each population clear the depth +site, ``wisp`` reports how many samples in each population clear the depth threshold. Why population count masks? @@ -29,13 +29,13 @@ Input modes BAM/CRAM mode ------------- -In alignment mode, ``sprite`` runs ``mosdepth`` once per sample using a +In alignment mode, ``wisp`` runs ``mosdepth`` once per sample using a two-bin quantization: below threshold and at-or-above threshold. It extracts the passing intervals, optionally clips them to a mask BED, intersects all sample pass BEDs with ``bedtools multiinter``, and assembles them into a population count mask. -An optional variants-only VCF can modify this alignment workflow. ``sprite`` +An optional variants-only VCF can modify this alignment workflow. ``wisp`` can estimate omitted depth and mapping-quality thresholds from the VCF, and it subtracts indel, structural-variant, breakend, and multi-nucleotide polymorphism spans from every sample pass BED. Because the final BED is @@ -45,7 +45,7 @@ zero passing samples in every population. All-sites VCF mode ------------------ -In VCF mode, ``sprite`` reads ``FORMAT/DP`` values directly from an all-sites +In VCF mode, ``wisp`` reads ``FORMAT/DP`` values directly from an all-sites VCF. A sample passes a base when its DP value is greater than or equal to ``--min-dp`` and, if ``--max-dp`` is supplied, less than or equal to ``--max-dp``. When ``FORMAT/GT`` is present, the genotype must also be @@ -57,6 +57,6 @@ coordinate-sorted VCF. Sparse output ============= -``sprite`` omits intervals where all population counts are zero. Consumers +``wisp`` omits intervals where all population counts are zero. Consumers should treat missing intervals as zero passing samples per population, not as unknown or skipped. diff --git a/docs/api.rst b/docs/api.rst index 1bebd06..efbb721 100644 --- a/docs/api.rst +++ b/docs/api.rst @@ -1,29 +1,29 @@ API Reference ************* -The public interface is the ``sprite`` command line tool, but these modules are +The public interface is the ``wisp`` command line tool, but these modules are useful when reading or extending the implementation. Configuration ============= -.. automodule:: sprite_mask.config +.. automodule:: wisp_mask.config :members: Workflow ======== -.. automodule:: sprite_mask.workflow +.. automodule:: wisp_mask.workflow :members: run_workflow, workflow_output_paths Samples ======= -.. automodule:: sprite_mask.samples +.. automodule:: wisp_mask.samples :members: Output summaries ================ -.. automodule:: sprite_mask.summaries +.. automodule:: wisp_mask.summaries :members: summarize_population_count_bed diff --git a/docs/arguments.rst b/docs/arguments.rst index 81b35de..3c70652 100644 --- a/docs/arguments.rst +++ b/docs/arguments.rst @@ -1,12 +1,12 @@ Arguments ********* -All arguments are listed below. ``sprite --help`` shows the same information. +All arguments are listed below. ``wisp --help`` shows the same information. Commands ======== -``sprite`` has two subcommands: +``wisp`` has two subcommands: **from-alignments** Build a population count mask from BAM/CRAM files via ``mosdepth``. @@ -26,7 +26,7 @@ Core arguments is supplied. **--out PATH** - Output directory for the final ``sprite.bed.gz`` and tabix index. + Output directory for the final ``wisp.bed.gz`` and tabix index. Input-specific arguments ======================== @@ -66,8 +66,8 @@ Shared optional arguments intervals are emitted. **--output-prefix TEXT** - Output filename stem within ``--out``. Defaults to ``sprite``, - producing ``sprite.bed.gz`` and ``sprite.bed.gz.tbi``. ``.bed.gz`` + Output filename stem within ``--out``. Defaults to ``wisp``, + producing ``wisp.bed.gz`` and ``wisp.bed.gz.tbi``. ``.bed.gz`` is always appended. **--keep-work** @@ -75,11 +75,11 @@ Shared optional arguments files are removed after a successful run. **--force** - Overwrite existing final outputs. Without this flag, ``sprite`` refuses - to replace ``sprite.bed.gz`` or ``sprite.bed.gz.tbi``. + Overwrite existing final outputs. Without this flag, ``wisp`` refuses + to replace ``wisp.bed.gz`` or ``wisp.bed.gz.tbi``. **--version** - Print the installed ``sprite`` version and exit. + Print the installed ``wisp`` version and exit. **--help** Print the full help message and exit. @@ -125,7 +125,7 @@ BAM/CRAM mode: .. code-block:: console - sprite from-alignments \ + wisp from-alignments \ --samples tests/test_data/1000g_5sample_chr20_smoke/samples.tsv \ --min-dp 10 \ --variants-vcf validation/cohort.variants.vcf.gz \ @@ -140,7 +140,7 @@ All-sites VCF mode: .. code-block:: console - sprite from-vcf \ + wisp from-vcf \ --all-sites-vcf validation/cohort.all_sites.vcf.gz \ --popfile validation/sample_populations.tsv \ --min-dp 10 \ diff --git a/docs/changelog.rst b/docs/changelog.rst index 4be901f..633ea08 100644 --- a/docs/changelog.rst +++ b/docs/changelog.rst @@ -13,5 +13,5 @@ Highlights: depth/MAPQ thresholds and exclude indel, structural-variant, breakend, and multi-nucleotide polymorphism spans. * Build the same output from prefiltered all-sites VCF ``FORMAT/DP`` values. -* Write bgzipped and tabix-indexed ``sprite.bed.gz`` output. +* Write bgzipped and tabix-indexed ``wisp.bed.gz`` output. * Include JSON metadata and population column headers in the output BED. diff --git a/docs/conf.py b/docs/conf.py index 7eba0e7..40ff466 100644 --- a/docs/conf.py +++ b/docs/conf.py @@ -14,9 +14,9 @@ sys.path.insert(0, str(ROOT / "src")) -project = "sprite" -copyright = "2026, sprite contributors" -author = "sprite contributors" +project = "wisp" +copyright = "2026, wisp contributors" +author = "wisp contributors" with (ROOT / "pyproject.toml").open("rb") as handle: release = tomllib.load(handle)["project"]["version"] @@ -42,27 +42,27 @@ html_static_path = [] html_title = f"{project} {release}" -htmlhelp_basename = "spritedoc" +htmlhelp_basename = "wispdoc" latex_documents = [ ( master_doc, - "sprite.tex", - "sprite Documentation", + "wisp.tex", + "wisp Documentation", author, "manual", ), ] -man_pages = [(master_doc, "sprite", "sprite Documentation", [author], 1)] +man_pages = [(master_doc, "wisp", "wisp Documentation", [author], 1)] texinfo_documents = [ ( master_doc, - "sprite", - "sprite Documentation", + "wisp", + "wisp Documentation", author, - "sprite", + "wisp", "Build sparse depth-threshold mask BEDs from cohort alignment data or all-sites VCFs.", "Miscellaneous", ), diff --git a/docs/development.rst b/docs/development.rst index 06b800d..b990518 100644 --- a/docs/development.rst +++ b/docs/development.rst @@ -7,7 +7,7 @@ Set up a development environment .. code-block:: console mamba env create -f environment.yml - conda activate sprite + conda activate wisp python -m pip install -e ".[dev,docs]" Run checks @@ -38,7 +38,7 @@ Project layout .. code-block:: text - src/sprite_mask/ package source + src/wisp_mask/ package source tests/ unit and workflow tests tests/test_data/ small 1000 Genomes fixtures and download scripts docs/ Sphinx documentation @@ -46,7 +46,7 @@ Project layout Implementation overview ======================= -The public CLI is defined in ``sprite_mask.cli``. Parsed arguments are converted +The public CLI is defined in ``wisp_mask.cli``. Parsed arguments are converted to ``RunConfig`` and passed to ``run_workflow``. The workflow validates the input -mode, runs either the BAM/CRAM or VCF builder, writes ``sprite.bed.gz``, +mode, runs either the BAM/CRAM or VCF builder, writes ``wisp.bed.gz``, and creates the tabix index. diff --git a/docs/examples.rst b/docs/examples.rst index f372271..274c30f 100644 --- a/docs/examples.rst +++ b/docs/examples.rst @@ -8,7 +8,7 @@ This run uses the small fixture bundled with the test suite: .. code-block:: console - sprite from-alignments \ + wisp from-alignments \ --samples tests/test_data/1000g_5sample_chr20_smoke/samples.tsv \ --min-dp 10 \ --mask tests/test_data/1000g_5sample_chr20_smoke/targets.bed \ @@ -21,8 +21,8 @@ Expected outputs: .. code-block:: text - results/chr20_smoke/sprite.bed.gz - results/chr20_smoke/sprite.bed.gz.tbi + results/chr20_smoke/wisp.bed.gz + results/chr20_smoke/wisp.bed.gz.tbi Keep intermediate files ======================= @@ -32,7 +32,7 @@ Add ``--keep-work`` to inspect the mosdepth outputs, sample pass BEDs, and .. code-block:: console - sprite from-alignments \ + wisp from-alignments \ --samples tests/test_data/1000g_5sample_chr20_smoke/samples.tsv \ --min-dp 10 \ --mask tests/test_data/1000g_5sample_chr20_smoke/targets.bed \ @@ -44,12 +44,12 @@ BAM/CRAM with a variants-only VCF ================================= Use ``--variants-vcf`` when you have a variants-only VCF from the same callset -and want ``sprite`` to estimate omitted alignment thresholds and mask +and want ``wisp`` to estimate omitted alignment thresholds and mask non-SNP variant spans: .. code-block:: console - sprite from-alignments \ + wisp from-alignments \ --samples tests/test_data/1000g_5sample_chr20_smoke/samples.tsv \ --variants-vcf validation/cohort.variants.vcf.gz \ --mask tests/test_data/1000g_5sample_chr20_smoke/targets.bed \ @@ -62,7 +62,7 @@ Manual threshold flags override VCF-derived estimates: .. code-block:: console - sprite from-alignments \ + wisp from-alignments \ --samples samples.tsv \ --variants-vcf cohort.variants.vcf.gz \ --min-dp 8 \ @@ -77,7 +77,7 @@ When you have a prefiltered all-sites VCF with per-sample DP values: .. code-block:: console - sprite from-vcf \ + wisp from-vcf \ --all-sites-vcf validation/1000g_20sample_highcov_4chrom_subset/1000g_20samples_highcov_4chroms.all_sites.bam_call.trim_alt.vcf.gz \ --popfile validation/1000g_20sample_highcov_4chrom_subset/sample_populations.tsv \ --min-dp 10 \ @@ -93,7 +93,7 @@ The final BED is tabix-indexed, so you can slice it directly: .. code-block:: console - tabix results/chr20_smoke/sprite.bed.gz chr20:10000000-10010000 + tabix results/chr20_smoke/wisp.bed.gz chr20:10000000-10010000 Note that BED coordinates are 0-based and half-open, while tabix region strings are 1-based inclusive. @@ -101,11 +101,11 @@ strings are 1-based inclusive. Overwrite existing outputs ========================== -``sprite`` refuses to overwrite final outputs unless ``--force`` is passed: +``wisp`` refuses to overwrite final outputs unless ``--force`` is passed: .. code-block:: console - sprite from-alignments \ + wisp from-alignments \ --samples tests/test_data/1000g_5sample_chr20_smoke/samples.tsv \ --min-dp 10 \ --mask tests/test_data/1000g_5sample_chr20_smoke/targets.bed \ diff --git a/docs/images/sprite_logo.png b/docs/images/sprite_logo.png deleted file mode 100644 index 879982c..0000000 Binary files a/docs/images/sprite_logo.png and /dev/null differ diff --git a/docs/images/wisp_logo.png b/docs/images/wisp_logo.png new file mode 100644 index 0000000..db5a95e Binary files /dev/null and b/docs/images/wisp_logo.png differ diff --git a/docs/index.rst b/docs/index.rst index bf7186d..05ee909 100644 --- a/docs/index.rst +++ b/docs/index.rst @@ -1,17 +1,17 @@ -.. sprite documentation master file. +.. wisp documentation master file. .. raw:: html -

sprite 0.1.0

+

wisp 0.1.0

-.. image:: images/sprite_logo.png +.. image:: images/wisp_logo.png :width: 200 :align: center -What is sprite? +What is wisp? =============== -``sprite`` is a command line tool for building population count masks: sparse, +``wisp`` is a command line tool for building population count masks: sparse, population-level summaries of how many individuals have sufficient depth and mapping quality to call genotypes at sites across the genome. These masks can be produced from BAM/CRAM alignments or an all-sites VCF. @@ -48,11 +48,11 @@ Population count masks let you correctly compute the denominators of π, d\ VCF — callable sites are counted per population rather than collapsed into a single cohort-wide pass/fail. -``sprite`` is designed for use with `pixy `_, +``wisp`` is designed for use with `pixy `_, but works equally well on its own. Source code is on -`GitHub `_. +`GitHub `_. -The tool produces the same ``sprite.bed.gz`` output from two input modes: +The tool produces the same ``wisp.bed.gz`` output from two input modes: * BAM/CRAM alignments, using ``mosdepth`` to quantize each sample and ``bedtools multiinter`` to combine samples. In this mode, an optional diff --git a/docs/inputs.rst b/docs/inputs.rst index 6c767a9..a7c9bd3 100644 --- a/docs/inputs.rst +++ b/docs/inputs.rst @@ -24,7 +24,7 @@ Accepted header aliases: Without a recognized header, columns are read in order: ``sample_id``, ``population``, ``alignment``. -Alignment paths may be absolute or relative. For relative paths, ``sprite`` +Alignment paths may be absolute or relative. For relative paths, ``wisp`` first checks the current value, then the path relative to the sample table's parent directory. @@ -59,7 +59,7 @@ All-sites VCF columns. Every sample ID in ``--popfile`` must appear in the VCF. VCF samples absent from ``--popfile`` produce a warning and are ignored. -For each record, ``sprite`` reads the ``DP`` field from the sample's +For each record, ``wisp`` reads the ``DP`` field from the sample's ``FORMAT`` value. Missing DP values do not pass. Non-integer DP values are rejected. A sample passes when DP is at least ``--min-dp``, if ``--max-dp`` is supplied no greater than ``--max-dp``, and, when ``GT`` is present in @@ -77,15 +77,15 @@ coordinate system as the alignments. When sample columns are present, every sample ID in ``--samples`` must appear in the VCF; extra VCF samples are ignored for threshold estimation. -``sprite`` uses this VCF in two ways. +``wisp`` uses this VCF in two ways. Threshold estimation -------------------- -If ``--min-dp`` or ``--max-dp`` is omitted, ``sprite`` estimates it from +If ``--min-dp`` or ``--max-dp`` is omitted, ``wisp`` estimates it from positive per-sample ``FORMAT/DP`` values at variant records. ``--min-dp`` uses the smallest observed positive DP and ``--max-dp`` uses the largest -observed positive DP. If ``--min-mapq`` is omitted, ``sprite`` estimates it +observed positive DP. If ``--min-mapq`` is omitted, ``wisp`` estimates it from the smallest ``INFO/MQ`` value, rounded down to an integer. Any threshold supplied manually on the command line takes precedence over diff --git a/docs/installation.rst b/docs/installation.rst index 0b1df32..e143cfc 100644 --- a/docs/installation.rst +++ b/docs/installation.rst @@ -4,7 +4,7 @@ Installation Requirements ============ -``sprite`` requires Python 3.11 or 3.12. Runtime command line tools depend +``wisp`` requires Python 3.11 or 3.12. Runtime command line tools depend on the input mode: * BAM/CRAM mode requires ``samtools``, ``mosdepth``, ``bedtools``, ``bgzip``, and ``tabix``. @@ -21,26 +21,26 @@ From the repository root, create and activate the development environment: .. code-block:: console mamba env create -f environment.yml - conda activate sprite + conda activate wisp python -m pip install -e ".[dev]" If the environment already exists, update it instead: .. code-block:: console - mamba env update -n sprite -f environment.yml - conda activate sprite + mamba env update -n wisp -f environment.yml + conda activate wisp python -m pip install -e ".[dev]" Verify the CLI ============== -After installation, verify the ``sprite`` command: +After installation, verify the ``wisp`` command: .. code-block:: console - sprite --help - sprite --version + wisp --help + wisp --version Build these docs locally ======================== diff --git a/docs/output.rst b/docs/output.rst index 95d04d5..d5f1cd5 100644 --- a/docs/output.rst +++ b/docs/output.rst @@ -8,8 +8,8 @@ Each successful run writes two files to ``--out``: .. code-block:: text - sprite.bed.gz - sprite.bed.gz.tbi + wisp.bed.gz + wisp.bed.gz.tbi The population count mask is bgzip-compressed and indexed with ``tabix -p bed``. Use ``--output-prefix`` to choose a different filename @@ -22,7 +22,7 @@ The mask starts with two comment-prefixed header lines: .. code-block:: text - #sprite_mask_metadata {"columns":["chrom","start","end","GBR","YRI"],...} + #wisp_mask_metadata {"columns":["chrom","start","end","GBR","YRI"],...} #chrom start end GBR YRI Data rows contain: @@ -46,9 +46,9 @@ all counts are zero are omitted. Metadata ======== -The ``#sprite_mask_metadata`` line is JSON. It includes: +The ``#wisp_mask_metadata`` line is JSON. It includes: -* ``sprite_mask_version`` +* ``wisp_mask_version`` * ``format`` * ``columns`` * ``coordinate_system`` diff --git a/environment.yml b/environment.yml index a278ffd..d514ee1 100644 --- a/environment.yml +++ b/environment.yml @@ -1,4 +1,4 @@ -name: sprite +name: wisp channels: - conda-forge - bioconda diff --git a/pyproject.toml b/pyproject.toml index 0495b2c..6f07dfa 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -3,14 +3,14 @@ requires = ["setuptools>=68", "wheel"] build-backend = "setuptools.build_meta" [project] -name = "sprite-mask" +name = "wisp-mask" version = "0.1.0" description = "Build sparse depth-threshold mask BEDs from BAM/CRAM cohorts with mosdepth and bedtools." readme = "README.md" requires-python = ">=3.10,<3.15" license = "MIT" authors = [ - { name = "sprite-mask contributors" }, + { name = "wisp-mask contributors" }, ] dependencies = [] @@ -29,8 +29,8 @@ docs = [ ] [project.scripts] -sprite = "sprite_mask.cli:main" -sprite-mask = "sprite_mask.cli:main" +wisp = "wisp_mask.cli:main" +wisp-mask = "wisp_mask.cli:main" [tool.setuptools.packages.find] where = ["src"] diff --git a/src/sprite_mask/__init__.py b/src/wisp_mask/__init__.py similarity index 100% rename from src/sprite_mask/__init__.py rename to src/wisp_mask/__init__.py diff --git a/src/sprite_mask/bedio.py b/src/wisp_mask/bedio.py similarity index 100% rename from src/sprite_mask/bedio.py rename to src/wisp_mask/bedio.py diff --git a/src/sprite_mask/bedtools.py b/src/wisp_mask/bedtools.py similarity index 96% rename from src/sprite_mask/bedtools.py rename to src/wisp_mask/bedtools.py index 3190816..d5c8715 100644 --- a/src/sprite_mask/bedtools.py +++ b/src/wisp_mask/bedtools.py @@ -4,8 +4,8 @@ from collections.abc import Sequence from pathlib import Path -from sprite_mask.bedio import iter_bed3 -from sprite_mask.commands import run_pipeline +from wisp_mask.bedio import iter_bed3 +from wisp_mask.commands import run_pipeline def sort_and_merge_bed(in_bed: Path, out_bed: Path) -> Path: diff --git a/src/sprite_mask/cli.py b/src/wisp_mask/cli.py similarity index 89% rename from src/sprite_mask/cli.py rename to src/wisp_mask/cli.py index bcb3a33..09fd492 100644 --- a/src/sprite_mask/cli.py +++ b/src/wisp_mask/cli.py @@ -8,22 +8,21 @@ from pathlib import Path from typing import Any -from sprite_mask import __version__ -from sprite_mask.config import AlignmentRunConfig, VcfRunConfig -from sprite_mask.workflow import run_workflow - -HELP_BANNER = """\ - █▄ - ▄ ▀▀▄██▄ - ▄██▀█ ████▄ ████▄██ ██ ▄█▀█▄ - ▀███▄ ██ ██ ██ ██ ██ ██▄█▀ -█▄▄██▀▄████▀▄█▀ ▄██▄██▄▀█▄▄▄ - ██ - ▀ -""" - - -class SpriteArgumentParser(argparse.ArgumentParser): +from wisp_mask import __version__ +from wisp_mask.config import AlignmentRunConfig, VcfRunConfig +from wisp_mask.workflow import run_workflow + +HELP_BANNER = ( + " _ \n" + "__ _(_)___ _ __\n" + "\\ \\ /\\ / / / __| '_ \\\n" + " \\ V V /| \\__ \\ |_) |\n" + " \\_/\\_/ |_|___/ .__/\n" + " |_| " +) + + +class WispArgumentParser(argparse.ArgumentParser): def __init__(self, *args: Any, show_banner: bool = False, **kwargs: Any) -> None: super().__init__(*args, **kwargs) self._show_banner = show_banner @@ -32,7 +31,7 @@ def format_help(self) -> str: help_text = super().format_help() if not self._show_banner: return help_text - return f"{HELP_BANNER}\nsprite {__version__}\n\n{help_text}" + return f"{HELP_BANNER}\nwisp {__version__}\n\n{help_text}" def main(argv: Sequence[str] | None = None) -> int: @@ -53,7 +52,7 @@ def main(argv: Sequence[str] | None = None) -> int: except Exception as error: if getattr(args, "debug", False): raise - print(f"sprite: error: {error}", file=sys.stderr) + print(f"wisp: error: {error}", file=sys.stderr) return 1 @@ -102,8 +101,8 @@ def _cmd_from_vcf(args: argparse.Namespace) -> int: def build_parser() -> argparse.ArgumentParser: - parser = SpriteArgumentParser( - prog="sprite", + parser = WispArgumentParser( + prog="wisp", description="build population count masks", show_banner=True, ) @@ -130,7 +129,7 @@ def _add_common_run_args(p: argparse.ArgumentParser, *, min_dp_required: bool = p.add_argument("--out", required=True, help="output directory") p.add_argument( "--output-prefix", - default="sprite", + default="wisp", help="output file prefix within --out; .bed.gz is appended", ) p.add_argument("--work", help="working directory; defaults to /work") @@ -235,7 +234,7 @@ def _setup_logging(*, verbose: bool, quiet: bool) -> None: level = logging.INFO logging.basicConfig( level=level, - format="%(asctime)s [sprite] %(message)s", + format="%(asctime)s [wisp] %(message)s", datefmt="%Y-%m-%d %H:%M:%S", stream=sys.stderr, force=True, @@ -243,7 +242,7 @@ def _setup_logging(*, verbose: bool, quiet: bool) -> None: def _print_subprocess_error(error: subprocess.CalledProcessError) -> None: - print(f"sprite: command failed with exit code {error.returncode}", file=sys.stderr) + print(f"wisp: command failed with exit code {error.returncode}", file=sys.stderr) print(" ".join(str(part) for part in error.cmd), file=sys.stderr) if error.stderr: print(str(error.stderr).rstrip(), file=sys.stderr) diff --git a/src/sprite_mask/collapse.py b/src/wisp_mask/collapse.py similarity index 95% rename from src/sprite_mask/collapse.py rename to src/wisp_mask/collapse.py index c15bea7..567a65f 100644 --- a/src/sprite_mask/collapse.py +++ b/src/wisp_mask/collapse.py @@ -6,9 +6,9 @@ from pathlib import Path from typing import Any, TextIO -from sprite_mask import __version__ -from sprite_mask.models import Sample -from sprite_mask.samples import populations_in_order, sample_population_map +from wisp_mask import __version__ +from wisp_mask.models import Sample +from wisp_mask.samples import populations_in_order, sample_population_map def collapse_population_counts( @@ -92,14 +92,14 @@ def write_quantized_bed_header( metadata: dict[str, Any], ) -> None: full_metadata = { - "sprite_mask_version": __version__, + "wisp_mask_version": __version__, "columns": columns, "coordinate_system": "BED 0-based half-open", "zero_count_intervals_omitted": True, **metadata, } handle.write( - "#sprite_mask_metadata\t" + "#wisp_mask_metadata\t" + json.dumps(full_metadata, sort_keys=True, separators=(",", ":")) + "\n" ) diff --git a/src/sprite_mask/commands.py b/src/wisp_mask/commands.py similarity index 100% rename from src/sprite_mask/commands.py rename to src/wisp_mask/commands.py diff --git a/src/sprite_mask/config.py b/src/wisp_mask/config.py similarity index 94% rename from src/sprite_mask/config.py rename to src/wisp_mask/config.py index 194f389..99d25bb 100644 --- a/src/sprite_mask/config.py +++ b/src/wisp_mask/config.py @@ -23,7 +23,7 @@ class AlignmentRunConfig: keep_work: bool = False force: bool = False dry_run: bool = False - output_prefix: str = "sprite" + output_prefix: str = "wisp" @property def resolved_work_dir(self) -> Path: @@ -42,7 +42,7 @@ class VcfRunConfig: keep_work: bool = False force: bool = False dry_run: bool = False - output_prefix: str = "sprite" + output_prefix: str = "wisp" snps_only: bool = False @property diff --git a/src/sprite_mask/models.py b/src/wisp_mask/models.py similarity index 100% rename from src/sprite_mask/models.py rename to src/wisp_mask/models.py diff --git a/src/sprite_mask/mosdepth.py b/src/wisp_mask/mosdepth.py similarity index 96% rename from src/sprite_mask/mosdepth.py rename to src/wisp_mask/mosdepth.py index 28eb316..06f537a 100644 --- a/src/sprite_mask/mosdepth.py +++ b/src/wisp_mask/mosdepth.py @@ -4,8 +4,8 @@ import subprocess from pathlib import Path -from sprite_mask.config import AlignmentRunConfig -from sprite_mask.models import MosdepthOutputs, Sample +from wisp_mask.config import AlignmentRunConfig +from wisp_mask.models import MosdepthOutputs, Sample def run_mosdepth(sample: Sample, config: AlignmentRunConfig) -> MosdepthOutputs: diff --git a/src/sprite_mask/samples.py b/src/wisp_mask/samples.py similarity index 99% rename from src/sprite_mask/samples.py rename to src/wisp_mask/samples.py index 997a81b..ffa621b 100644 --- a/src/sprite_mask/samples.py +++ b/src/wisp_mask/samples.py @@ -3,7 +3,7 @@ import csv from pathlib import Path -from sprite_mask.models import Sample +from wisp_mask.models import Sample ALIGNMENT_ALIASES = {"alignment", "bam_or_cram", "bam", "cram"} SAMPLE_ID_ALIASES = {"sample_id", "sample", "id"} diff --git a/src/sprite_mask/summaries.py b/src/wisp_mask/summaries.py similarity index 100% rename from src/sprite_mask/summaries.py rename to src/wisp_mask/summaries.py diff --git a/src/sprite_mask/validation.py b/src/wisp_mask/validation.py similarity index 99% rename from src/sprite_mask/validation.py rename to src/wisp_mask/validation.py index f428d12..031dda8 100644 --- a/src/sprite_mask/validation.py +++ b/src/wisp_mask/validation.py @@ -5,7 +5,7 @@ from collections.abc import Iterable from pathlib import Path -from sprite_mask.models import Sample +from wisp_mask.models import Sample def validate_threshold( diff --git a/src/sprite_mask/vcf.py b/src/wisp_mask/vcf.py similarity index 99% rename from src/sprite_mask/vcf.py rename to src/wisp_mask/vcf.py index 61394f4..61dcee1 100644 --- a/src/sprite_mask/vcf.py +++ b/src/wisp_mask/vcf.py @@ -9,10 +9,10 @@ from pathlib import Path from typing import TextIO -from sprite_mask.bedio import iter_bed3 -from sprite_mask.collapse import write_quantized_bed_header -from sprite_mask.models import Sample -from sprite_mask.samples import populations_in_order +from wisp_mask.bedio import iter_bed3 +from wisp_mask.collapse import write_quantized_bed_header +from wisp_mask.models import Sample +from wisp_mask.samples import populations_in_order logger = logging.getLogger(__name__) diff --git a/src/sprite_mask/workflow.py b/src/wisp_mask/workflow.py similarity index 97% rename from src/sprite_mask/workflow.py rename to src/wisp_mask/workflow.py index eba9ca8..6555ee7 100644 --- a/src/sprite_mask/workflow.py +++ b/src/wisp_mask/workflow.py @@ -8,20 +8,20 @@ from dataclasses import replace from pathlib import Path -from sprite_mask.bedio import extract_merged_pass_intervals, normalize_targets_bed -from sprite_mask.bedtools import ( +from wisp_mask.bedio import extract_merged_pass_intervals, normalize_targets_bed +from wisp_mask.bedtools import ( intersect_sort_merge, run_multiinter, sort_and_merge_bed, subtract_sort_merge, write_single_input_multiinter, ) -from sprite_mask.collapse import collapse_population_counts -from sprite_mask.config import AlignmentRunConfig, RunConfig, VcfRunConfig -from sprite_mask.models import MosdepthOutputs, Sample, WorkflowOutputs -from sprite_mask.mosdepth import run_mosdepth -from sprite_mask.samples import read_popfile, read_samples -from sprite_mask.validation import ( +from wisp_mask.collapse import collapse_population_counts +from wisp_mask.config import AlignmentRunConfig, RunConfig, VcfRunConfig +from wisp_mask.models import MosdepthOutputs, Sample, WorkflowOutputs +from wisp_mask.mosdepth import run_mosdepth +from wisp_mask.samples import read_popfile, read_samples +from wisp_mask.validation import ( ensure_parent_dirs, refuse_existing_outputs, require_executables, @@ -32,7 +32,7 @@ validate_variants_vcf_input, validate_vcf_inputs, ) -from sprite_mask.vcf import ( +from wisp_mask.vcf import ( build_population_counts_from_all_sites_vcf, estimate_alignment_thresholds_from_variants_vcf, validate_vcf_sample_names, @@ -301,7 +301,7 @@ def _build_from_alignments( def workflow_output_paths( out_dir: Path, threshold: int, - output_prefix: str = "sprite", + output_prefix: str = "wisp", ) -> WorkflowOutputs: if output_prefix == "": raise ValueError("--output-prefix cannot be empty") diff --git a/tests/test_bedio.py b/tests/test_bedio.py index 6869b53..964c5d1 100644 --- a/tests/test_bedio.py +++ b/tests/test_bedio.py @@ -5,7 +5,7 @@ import pytest -from sprite_mask.bedio import ( +from wisp_mask.bedio import ( count_bed_sites, extract_merged_pass_intervals, extract_pass_intervals, diff --git a/tests/test_bedtools.py b/tests/test_bedtools.py index 9c552cb..b647627 100644 --- a/tests/test_bedtools.py +++ b/tests/test_bedtools.py @@ -7,7 +7,7 @@ import pytest -from sprite_mask.bedtools import ( +from wisp_mask.bedtools import ( build_multiinter_command, intersect_sort_merge, run_multiinter, @@ -27,7 +27,7 @@ def fake_run_pipeline(commands: Sequence[Sequence[str]], out_path: Path) -> None calls.append(([list(command) for command in commands], out_path)) out_path.write_text("merged\n") - monkeypatch.setattr("sprite_mask.bedtools.run_pipeline", fake_run_pipeline) + monkeypatch.setattr("wisp_mask.bedtools.run_pipeline", fake_run_pipeline) in_bed = tmp_path / "in.bed" out_bed = tmp_path / "out.bed" @@ -54,7 +54,7 @@ def test_intersect_sort_merge_builds_bedtools_pipeline( def fake_run_pipeline(commands: Sequence[Sequence[str]], out_path: Path) -> None: calls.append(([list(command) for command in commands], out_path)) - monkeypatch.setattr("sprite_mask.bedtools.run_pipeline", fake_run_pipeline) + monkeypatch.setattr("wisp_mask.bedtools.run_pipeline", fake_run_pipeline) a_bed = tmp_path / "a.bed" b_bed = tmp_path / "b.bed" @@ -82,7 +82,7 @@ def test_subtract_sort_merge_builds_bedtools_pipeline( def fake_run_pipeline(commands: Sequence[Sequence[str]], out_path: Path) -> None: calls.append(([list(command) for command in commands], out_path)) - monkeypatch.setattr("sprite_mask.bedtools.run_pipeline", fake_run_pipeline) + monkeypatch.setattr("wisp_mask.bedtools.run_pipeline", fake_run_pipeline) a_bed = tmp_path / "a.bed" b_bed = tmp_path / "b.bed" @@ -121,7 +121,7 @@ def fake_run( stdout.write("chrom\tstart\tend\tnum\tlist\ts1\n") return subprocess.CompletedProcess(command, 0) - monkeypatch.setattr("sprite_mask.bedtools.subprocess.run", fake_run) + monkeypatch.setattr("wisp_mask.bedtools.subprocess.run", fake_run) out_tsv = tmp_path / "nested" / "multiinter.tsv" diff --git a/tests/test_cli.py b/tests/test_cli.py index 1a3c37e..a04d453 100644 --- a/tests/test_cli.py +++ b/tests/test_cli.py @@ -10,17 +10,17 @@ import pytest -from sprite_mask import __version__ -from sprite_mask.cli import HELP_BANNER, build_parser, main -from sprite_mask.models import WorkflowOutputs +from wisp_mask import __version__ +from wisp_mask.cli import HELP_BANNER, build_parser, main +from wisp_mask.models import WorkflowOutputs -SPRITE_PROGRESS_RE = re.compile(r"^\d{4}-\d{2}-\d{2} \d{2}:\d{2}:\d{2} \[sprite\] Analysis ") +WISP_PROGRESS_RE = re.compile(r"^\d{4}-\d{2}-\d{2} \d{2}:\d{2}:\d{2} \[wisp\] Analysis ") def test_root_help_starts_with_banner_and_version() -> None: help_text = build_parser().format_help() - assert help_text.startswith(f"{HELP_BANNER}\nsprite {__version__}\n\nusage:") + assert help_text.startswith(f"{HELP_BANNER}\nwisp {__version__}\n\nusage:") def test_from_alignments_help_describes_fast_mode_replacement() -> None: @@ -37,10 +37,10 @@ def test_from_alignments_help_describes_fast_mode_replacement() -> None: assert "--strict-depth" not in help_text -def assert_sprite_progress(log_output: str, message: str) -> None: +def assert_wisp_progress(log_output: str, message: str) -> None: matching_lines = [line for line in log_output.splitlines() if message in line] assert matching_lines - assert SPRITE_PROGRESS_RE.match(matching_lines[0]) + assert WISP_PROGRESS_RE.match(matching_lines[0]) def test_main_all_sites_vcf_writes_indexed_population_bed( @@ -85,12 +85,12 @@ def test_main_all_sites_vcf_writes_indexed_population_bed( assert status == 0 captured = capsys.readouterr() assert captured.out == "" - assert_sprite_progress(captured.err, "Analysis start: validating VCF workflow inputs") - assert_sprite_progress(captured.err, "Analysis VCF: building population counts") - assert_sprite_progress(captured.err, "Analysis complete: wrote") - assert "sprite.bed.gz" in captured.err + assert_wisp_progress(captured.err, "Analysis start: validating VCF workflow inputs") + assert_wisp_progress(captured.err, "Analysis VCF: building population counts") + assert_wisp_progress(captured.err, "Analysis complete: wrote") + assert "wisp.bed.gz" in captured.err - population_bed = out_dir / "sprite.bed.gz" + population_bed = out_dir / "wisp.bed.gz" population_index = Path(f"{population_bed}.tbi") assert population_bed.exists() assert population_index.exists() @@ -129,11 +129,11 @@ def fake_run_workflow(config: object) -> WorkflowOutputs: nonlocal seen_config seen_config = config return WorkflowOutputs( - population_count_bed_gz=tmp_path / "out" / "sprite.bed.gz", - population_count_bed_index=tmp_path / "out" / "sprite.bed.gz.tbi", + population_count_bed_gz=tmp_path / "out" / "wisp.bed.gz", + population_count_bed_index=tmp_path / "out" / "wisp.bed.gz.tbi", ) - monkeypatch.setattr("sprite_mask.cli.run_workflow", fake_run_workflow) + monkeypatch.setattr("wisp_mask.cli.run_workflow", fake_run_workflow) status = main( [ @@ -172,7 +172,7 @@ def fake_run_workflow(config: object) -> WorkflowOutputs: assert seen_config is not None assert seen_config.samples_path == tmp_path / "samples.tsv" assert seen_config.min_dp == 30 - assert seen_config.output_prefix == "sprite" + assert seen_config.output_prefix == "wisp" assert seen_config.threads == 4 assert seen_config.jobs == 2 assert seen_config.mask_bed == tmp_path / "targets.bed" @@ -200,11 +200,11 @@ def fake_run_workflow(config: object) -> WorkflowOutputs: nonlocal seen_config seen_config = config return WorkflowOutputs( - population_count_bed_gz=tmp_path / "out" / "sprite.bed.gz", - population_count_bed_index=tmp_path / "out" / "sprite.bed.gz.tbi", + population_count_bed_gz=tmp_path / "out" / "wisp.bed.gz", + population_count_bed_index=tmp_path / "out" / "wisp.bed.gz.tbi", ) - monkeypatch.setattr("sprite_mask.cli.run_workflow", fake_run_workflow) + monkeypatch.setattr("wisp_mask.cli.run_workflow", fake_run_workflow) status = main( [ @@ -238,7 +238,7 @@ def fake_run_workflow(config: object) -> WorkflowOutputs: population_count_bed_index=tmp_path / "out" / "custom.bed.gz.tbi", ) - monkeypatch.setattr("sprite_mask.cli.run_workflow", fake_run_workflow) + monkeypatch.setattr("wisp_mask.cli.run_workflow", fake_run_workflow) status = main( [ @@ -271,11 +271,11 @@ def fake_run_workflow(config: object) -> WorkflowOutputs: nonlocal seen_config seen_config = config return WorkflowOutputs( - population_count_bed_gz=tmp_path / "out" / "sprite.bed.gz", - population_count_bed_index=tmp_path / "out" / "sprite.bed.gz.tbi", + population_count_bed_gz=tmp_path / "out" / "wisp.bed.gz", + population_count_bed_index=tmp_path / "out" / "wisp.bed.gz.tbi", ) - monkeypatch.setattr("sprite_mask.cli.run_workflow", fake_run_workflow) + monkeypatch.setattr("wisp_mask.cli.run_workflow", fake_run_workflow) status = main( [ @@ -308,7 +308,7 @@ def test_main_reports_subprocess_errors( def fake_run_workflow(_config: object) -> WorkflowOutputs: raise subprocess.CalledProcessError(9, ["tool", "arg"], stderr="bad things\n") - monkeypatch.setattr("sprite_mask.cli.run_workflow", fake_run_workflow) + monkeypatch.setattr("wisp_mask.cli.run_workflow", fake_run_workflow) status = main( [ @@ -326,7 +326,7 @@ def fake_run_workflow(_config: object) -> WorkflowOutputs: captured = capsys.readouterr() assert captured.out == "" assert captured.err == ( - "sprite: command failed with exit code 9\n" + "wisp: command failed with exit code 9\n" "tool arg\n" "bad things\n" ) @@ -340,7 +340,7 @@ def test_main_reports_regular_exceptions( def fake_run_workflow(_config: object) -> WorkflowOutputs: raise ValueError("bad config") - monkeypatch.setattr("sprite_mask.cli.run_workflow", fake_run_workflow) + monkeypatch.setattr("wisp_mask.cli.run_workflow", fake_run_workflow) status = main( [ @@ -357,7 +357,7 @@ def fake_run_workflow(_config: object) -> WorkflowOutputs: assert status == 1 captured = capsys.readouterr() assert captured.out == "" - assert captured.err == "sprite: error: bad config\n" + assert captured.err == "wisp: error: bad config\n" def test_main_no_subcommand_returns_error(capsys: pytest.CaptureFixture[str]) -> None: @@ -366,7 +366,7 @@ def test_main_no_subcommand_returns_error(capsys: pytest.CaptureFixture[str]) -> assert status == 1 captured = capsys.readouterr() assert captured.out == "" - assert captured.err.startswith(f"{HELP_BANNER}\nsprite {__version__}\n\nusage:") + assert captured.err.startswith(f"{HELP_BANNER}\nwisp {__version__}\n\nusage:") def test_main_dry_run_skips_execution_and_returns_zero( @@ -378,11 +378,11 @@ def test_main_dry_run_skips_execution_and_returns_zero( def fake_run_workflow(config: object) -> WorkflowOutputs: calls.append(config) return WorkflowOutputs( - population_count_bed_gz=tmp_path / "out" / "sprite.bed.gz", - population_count_bed_index=tmp_path / "out" / "sprite.bed.gz.tbi", + population_count_bed_gz=tmp_path / "out" / "wisp.bed.gz", + population_count_bed_index=tmp_path / "out" / "wisp.bed.gz.tbi", ) - monkeypatch.setattr("sprite_mask.cli.run_workflow", fake_run_workflow) + monkeypatch.setattr("wisp_mask.cli.run_workflow", fake_run_workflow) status = main( [ @@ -409,7 +409,7 @@ def test_main_debug_reraises_exception( def fake_run_workflow(_config: object) -> WorkflowOutputs: raise ValueError("internal error") - monkeypatch.setattr("sprite_mask.cli.run_workflow", fake_run_workflow) + monkeypatch.setattr("wisp_mask.cli.run_workflow", fake_run_workflow) with pytest.raises(ValueError, match="internal error"): main( diff --git a/tests/test_collapse.py b/tests/test_collapse.py index cd52338..812300b 100644 --- a/tests/test_collapse.py +++ b/tests/test_collapse.py @@ -5,8 +5,8 @@ import pytest -from sprite_mask.collapse import collapse_population_counts -from sprite_mask.models import Sample +from wisp_mask.collapse import collapse_population_counts +from wisp_mask.models import Sample def test_collapse_population_counts_merges_equal_population_vectors(tmp_path: Path) -> None: @@ -132,7 +132,7 @@ def test_collapse_population_counts_rejects_non_integer_indicators(tmp_path: Pat def _read_header_metadata(path: Path) -> dict[str, object]: first_line = path.read_text().splitlines()[0] prefix, encoded = first_line.split("\t", maxsplit=1) - assert prefix == "#sprite_mask_metadata" + assert prefix == "#wisp_mask_metadata" metadata = json.loads(encoded) assert metadata["columns"][0:3] == ["chrom", "start", "end"] assert metadata["coordinate_system"] == "BED 0-based half-open" diff --git a/tests/test_commands.py b/tests/test_commands.py index a837459..eab579a 100644 --- a/tests/test_commands.py +++ b/tests/test_commands.py @@ -6,7 +6,7 @@ import pytest -from sprite_mask.commands import run_pipeline +from wisp_mask.commands import run_pipeline def test_run_pipeline_requires_at_least_two_commands(tmp_path: Path) -> None: diff --git a/tests/test_data/1000g_10sample_highcov_subset/README.md b/tests/test_data/1000g_10sample_highcov_subset/README.md index 7dedec3..1281878 100644 --- a/tests/test_data/1000g_10sample_highcov_subset/README.md +++ b/tests/test_data/1000g_10sample_highcov_subset/README.md @@ -21,7 +21,7 @@ approximately 30x * 0.67. Set DOWNSAMPLE_FRAC=1 to keep full depth. ## Files -- `samples.tsv`: sprite-mask sample metadata +- `samples.tsv`: wisp-mask sample metadata - `sample_populations_and_sources.tsv`: sample/population/source metadata - `samples.list`: sample list used for VCF subsetting - `targets.bed`: BED interval for the selected region diff --git a/tests/test_data/1000g_20sample_highcov_4chrom_subset/1000g_20samples_highcov.vcf.gz b/tests/test_data/1000g_20sample_highcov_4chrom_subset/1000g_20samples_highcov.vcf.gz deleted file mode 120000 index 2a3ded4..0000000 --- a/tests/test_data/1000g_20sample_highcov_4chrom_subset/1000g_20samples_highcov.vcf.gz +++ /dev/null @@ -1 +0,0 @@ -1000g_20samples_highcov_4chroms.vcf.gz \ No newline at end of file diff --git a/tests/test_data/1000g_20sample_highcov_4chrom_subset/1000g_20samples_highcov.vcf.gz.tbi b/tests/test_data/1000g_20sample_highcov_4chrom_subset/1000g_20samples_highcov.vcf.gz.tbi deleted file mode 120000 index 6579c04..0000000 --- a/tests/test_data/1000g_20sample_highcov_4chrom_subset/1000g_20samples_highcov.vcf.gz.tbi +++ /dev/null @@ -1 +0,0 @@ -1000g_20samples_highcov_4chroms.vcf.gz.tbi \ No newline at end of file diff --git a/tests/test_data/1000g_20sample_highcov_4chrom_subset/README.md b/tests/test_data/1000g_20sample_highcov_4chrom_subset/README.md index f573d3e..059e6e6 100644 --- a/tests/test_data/1000g_20sample_highcov_4chrom_subset/README.md +++ b/tests/test_data/1000g_20sample_highcov_4chrom_subset/README.md @@ -28,7 +28,7 @@ approximately 30x * 0.67. Set DOWNSAMPLE_FRAC=1 to keep full depth. ## Files -- `samples.tsv`: sprite-mask sample metadata +- `samples.tsv`: wisp-mask sample metadata - `sample_populations.tsv`: sample/population metadata - `sample_populations_and_sources.tsv`: sample/population/source CRAM metadata - `samples.list`: sample list used for VCF subsetting diff --git a/tests/test_data/scripts/README.md b/tests/test_data/scripts/README.md index eb5161f..2167d3d 100644 --- a/tests/test_data/scripts/README.md +++ b/tests/test_data/scripts/README.md @@ -24,5 +24,5 @@ below the warning threshold. The scripts read remote CRAM files. htslib may need to cache reference slices for those CRAMs, and those slices can be larger than GitHub's per-file limit. By default, the cache is written outside the fixture directory at -`${XDG_CACHE_HOME:-$HOME/.cache}/sprite-test-data/ref_cache`. Override +`${XDG_CACHE_HOME:-$HOME/.cache}/wisp-test-data/ref_cache`. Override `REF_CACHE_DIR` if you want a different local cache location. diff --git a/tests/test_data/scripts/download_1000g_10sample_chr20_highcov_fixture.sh b/tests/test_data/scripts/download_1000g_10sample_chr20_highcov_fixture.sh index 37e75bd..33e73ff 100644 --- a/tests/test_data/scripts/download_1000g_10sample_chr20_highcov_fixture.sh +++ b/tests/test_data/scripts/download_1000g_10sample_chr20_highcov_fixture.sh @@ -5,7 +5,7 @@ SCRIPT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" && pwd)" REPO_ROOT="$(cd -- "${SCRIPT_DIR}/../../.." && pwd)" cd "${REPO_ROOT}" -SCRIPT_VERSION="2026-05-28-sprite-test-data-env-v5-github-size-guard" +SCRIPT_VERSION="2026-05-28-wisp-test-data-env-v5-github-size-guard" # download_1000g_10sample_chr20_highcov_fixture.sh # @@ -21,7 +21,7 @@ SCRIPT_VERSION="2026-05-28-sprite-test-data-env-v5-github-size-guard" # Dataset: # - 10 regional BAMs: 5 GBR + 5 YRI # - one indexed multi-sample VCF containing all 10 samples -# - samples.tsv metadata for sprite-mask +# - samples.tsv metadata for wisp-mask # - targets.bed for the selected region # # Source: @@ -42,7 +42,7 @@ SCRIPT_VERSION="2026-05-28-sprite-test-data-env-v5-github-size-guard" # bash tests/test_data/scripts/download_1000g_10sample_chr20_highcov_fixture.sh # # Optional: -# ENV_NAME=sprite-test-data bash tests/test_data/scripts/download_1000g_10sample_chr20_highcov_fixture.sh +# ENV_NAME=wisp-test-data bash tests/test_data/scripts/download_1000g_10sample_chr20_highcov_fixture.sh # OUTDIR=tests/test_data/1000g_10sample_highcov_subset bash tests/test_data/scripts/download_1000g_10sample_chr20_highcov_fixture.sh # REGION=chr20:10000000-10100000 bash tests/test_data/scripts/download_1000g_10sample_chr20_highcov_fixture.sh # THREADS=4 bash tests/test_data/scripts/download_1000g_10sample_chr20_highcov_fixture.sh @@ -60,7 +60,7 @@ SCRIPT_VERSION="2026-05-28-sprite-test-data-env-v5-github-size-guard" # - htslib's CRAM reference cache is stored outside the fixture directory by # default so large reference slices are not accidentally committed. -ENV_NAME="${ENV_NAME:-sprite-test-data}" +ENV_NAME="${ENV_NAME:-wisp-test-data}" OUTDIR="${OUTDIR:-tests/test_data/1000g_10sample_highcov_subset}" REGION="${REGION:-chr20:10000000-10100000}" THREADS="${THREADS:-2}" @@ -72,9 +72,9 @@ FORCE="${FORCE:-0}" MAX_GITHUB_FILE_BYTES="${MAX_GITHUB_FILE_BYTES:-52428800}" if [ -n "${XDG_CACHE_HOME:-}" ]; then - DEFAULT_REF_CACHE_DIR="${XDG_CACHE_HOME}/sprite-test-data/ref_cache" + DEFAULT_REF_CACHE_DIR="${XDG_CACHE_HOME}/wisp-test-data/ref_cache" else - DEFAULT_REF_CACHE_DIR="${HOME}/.cache/sprite-test-data/ref_cache" + DEFAULT_REF_CACHE_DIR="${HOME}/.cache/wisp-test-data/ref_cache" fi REF_CACHE_DIR="${REF_CACHE_DIR:-${DEFAULT_REF_CACHE_DIR}}" @@ -440,7 +440,7 @@ approximately 30x * ${DOWNSAMPLE_FRAC}. Set DOWNSAMPLE_FRAC=1 to keep full depth ## Files -- \`samples.tsv\`: sprite sample metadata +- \`samples.tsv\`: wisp sample metadata - \`sample_populations_and_sources.tsv\`: sample/population/source metadata - \`samples.list\`: sample list used for VCF subsetting - \`targets.bed\`: BED interval for the selected region diff --git a/tests/test_data/scripts/download_1000g_20sample_4chrom_highcov_fixture.sh b/tests/test_data/scripts/download_1000g_20sample_4chrom_highcov_fixture.sh index b156e6f..0cfa5f8 100644 --- a/tests/test_data/scripts/download_1000g_20sample_4chrom_highcov_fixture.sh +++ b/tests/test_data/scripts/download_1000g_20sample_4chrom_highcov_fixture.sh @@ -5,7 +5,7 @@ SCRIPT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" && pwd)" REPO_ROOT="$(cd -- "${SCRIPT_DIR}/../../.." && pwd)" cd "${REPO_ROOT}" -SCRIPT_VERSION="2026-05-28-sprite-test-data-20samples-4chroms-v6-github-size-guard" +SCRIPT_VERSION="2026-05-28-wisp-test-data-20samples-4chroms-v6-github-size-guard" # download_1000g_20sample_4chrom_highcov_fixture.sh # @@ -17,7 +17,7 @@ SCRIPT_VERSION="2026-05-28-sprite-test-data-20samples-4chroms-v6-github-size-gua # - 20 regional BAMs: 10 GBR + 10 YRI # - one indexed multi-sample VCF containing all 20 samples # - four 50 kb regions from four different chromosomes -# - samples.tsv metadata for sprite-mask +# - samples.tsv metadata for wisp-mask # - targets.bed for the selected regions # # Important contig-name behavior: @@ -34,7 +34,7 @@ SCRIPT_VERSION="2026-05-28-sprite-test-data-20samples-4chroms-v6-github-size-gua # bash tests/test_data/scripts/download_1000g_20sample_4chrom_highcov_fixture.sh # # Optional: -# ENV_NAME=sprite-test-data bash tests/test_data/scripts/download_1000g_20sample_4chrom_highcov_fixture.sh +# ENV_NAME=wisp-test-data bash tests/test_data/scripts/download_1000g_20sample_4chrom_highcov_fixture.sh # OUTDIR=tests/test_data/1000g_20sample_highcov_4chrom_subset bash tests/test_data/scripts/download_1000g_20sample_4chrom_highcov_fixture.sh # THREADS=4 bash tests/test_data/scripts/download_1000g_20sample_4chrom_highcov_fixture.sh # DOWNSAMPLE_FRAC=1 bash tests/test_data/scripts/download_1000g_20sample_4chrom_highcov_fixture.sh @@ -48,7 +48,7 @@ SCRIPT_VERSION="2026-05-28-sprite-test-data-20samples-4chroms-v6-github-size-gua # - htslib's CRAM reference cache is stored outside the fixture directory by # default so large reference slices are not accidentally committed. -ENV_NAME="${ENV_NAME:-sprite-test-data}" +ENV_NAME="${ENV_NAME:-wisp-test-data}" OUTDIR="${OUTDIR:-tests/test_data/1000g_20sample_highcov_4chrom_subset}" THREADS="${THREADS:-2}" FORCE="${FORCE:-0}" @@ -59,9 +59,9 @@ FORCE="${FORCE:-0}" MAX_GITHUB_FILE_BYTES="${MAX_GITHUB_FILE_BYTES:-52428800}" if [ -n "${XDG_CACHE_HOME:-}" ]; then - DEFAULT_REF_CACHE_DIR="${XDG_CACHE_HOME}/sprite-test-data/ref_cache" + DEFAULT_REF_CACHE_DIR="${XDG_CACHE_HOME}/wisp-test-data/ref_cache" else - DEFAULT_REF_CACHE_DIR="${HOME}/.cache/sprite-test-data/ref_cache" + DEFAULT_REF_CACHE_DIR="${HOME}/.cache/wisp-test-data/ref_cache" fi REF_CACHE_DIR="${REF_CACHE_DIR:-${DEFAULT_REF_CACHE_DIR}}" @@ -657,7 +657,7 @@ approximately 30x * ${DOWNSAMPLE_FRAC}. Set DOWNSAMPLE_FRAC=1 to keep full depth ## Files -- \`samples.tsv\`: sprite sample metadata +- \`samples.tsv\`: wisp sample metadata - \`sample_populations.tsv\`: sample/population metadata - \`sample_populations_and_sources.tsv\`: sample/population/source CRAM metadata - \`samples.list\`: sample list used for VCF subsetting diff --git a/tests/test_mosdepth.py b/tests/test_mosdepth.py index 512f45f..75601ea 100644 --- a/tests/test_mosdepth.py +++ b/tests/test_mosdepth.py @@ -6,9 +6,9 @@ import pytest -from sprite_mask.config import AlignmentRunConfig -from sprite_mask.models import Sample -from sprite_mask.mosdepth import ( +from wisp_mask.config import AlignmentRunConfig +from wisp_mask.models import Sample +from wisp_mask.mosdepth import ( build_mosdepth_command, mosdepth_outputs_for_prefix, run_mosdepth, @@ -51,7 +51,7 @@ def fake_run( outputs.global_dist.write_text("dist") return subprocess.CompletedProcess(command, 0) - monkeypatch.setattr("sprite_mask.mosdepth.subprocess.run", fake_run) + monkeypatch.setattr("wisp_mask.mosdepth.subprocess.run", fake_run) outputs = run_mosdepth(sample, config) @@ -79,7 +79,7 @@ def test_run_mosdepth_rejects_missing_expected_outputs( def fake_run(*_args: object, **_kwargs: object) -> subprocess.CompletedProcess[str]: return subprocess.CompletedProcess(["mosdepth"], 0) - monkeypatch.setattr("sprite_mask.mosdepth.subprocess.run", fake_run) + monkeypatch.setattr("wisp_mask.mosdepth.subprocess.run", fake_run) with pytest.raises(FileNotFoundError, match="quantized.bed.gz"): run_mosdepth(sample, config) diff --git a/tests/test_samples.py b/tests/test_samples.py index b7eefdc..458cf50 100644 --- a/tests/test_samples.py +++ b/tests/test_samples.py @@ -4,8 +4,8 @@ import pytest -from sprite_mask.models import Sample -from sprite_mask.samples import ( +from wisp_mask.models import Sample +from wisp_mask.samples import ( read_popfile, read_samples, validate_sample_populations, diff --git a/tests/test_summaries.py b/tests/test_summaries.py index 6b57fba..bb88903 100644 --- a/tests/test_summaries.py +++ b/tests/test_summaries.py @@ -5,13 +5,13 @@ import pytest -from sprite_mask.summaries import summarize_population_count_bed +from wisp_mask.summaries import summarize_population_count_bed def test_summarize_population_count_bed(tmp_path: Path) -> None: bed = tmp_path / "population_counts.bed" bed.write_text( - "#sprite_mask_metadata\t{}\n" + "#wisp_mask_metadata\t{}\n" "#chrom\tstart\tend\tpopA\tpopB\n" "chr1\t0\t10\t1\t0\n" "chr1\t10\t25\t2\t1\n" diff --git a/tests/test_validation.py b/tests/test_validation.py index c68877e..9d0533b 100644 --- a/tests/test_validation.py +++ b/tests/test_validation.py @@ -5,8 +5,8 @@ import pytest -from sprite_mask.models import Sample -from sprite_mask.validation import ( +from wisp_mask.models import Sample +from wisp_mask.validation import ( ensure_parent_dirs, refuse_existing_outputs, require_executables, @@ -87,7 +87,7 @@ def fake_run( stderr="", ) - monkeypatch.setattr("sprite_mask.validation.subprocess.run", fake_run) + monkeypatch.setattr("wisp_mask.validation.subprocess.run", fake_run) validate_alignment_sample_headers([Sample("s1", "popA", alignment)]) @@ -113,7 +113,7 @@ def fake_run( stderr="", ) - monkeypatch.setattr("sprite_mask.validation.subprocess.run", fake_run) + monkeypatch.setattr("wisp_mask.validation.subprocess.run", fake_run) with pytest.raises(ValueError, match="do not match the sample_id: s2"): validate_alignment_sample_headers([Sample("s1", "popA", alignment)]) @@ -135,7 +135,7 @@ def fake_run( ) -> object: return subprocess.CompletedProcess(command, 0, stdout="@HD\tVN:1.6\n", stderr="") - monkeypatch.setattr("sprite_mask.validation.subprocess.run", fake_run) + monkeypatch.setattr("wisp_mask.validation.subprocess.run", fake_run) validate_alignment_sample_headers([Sample("s1", "popA", alignment)]) @@ -156,7 +156,7 @@ def fake_run( ) -> object: return subprocess.CompletedProcess(command, 1, stdout="", stderr="not a BAM\n") - monkeypatch.setattr("sprite_mask.validation.subprocess.run", fake_run) + monkeypatch.setattr("wisp_mask.validation.subprocess.run", fake_run) with pytest.raises(RuntimeError, match="could not read alignment header[\\s\\S]*not a BAM"): validate_alignment_sample_headers([Sample("s1", "popA", alignment)]) @@ -168,7 +168,7 @@ def fake_which(name: str) -> str | None: return "/usr/bin/present" return None - monkeypatch.setattr("sprite_mask.validation.shutil.which", fake_which) + monkeypatch.setattr("wisp_mask.validation.shutil.which", fake_which) with pytest.raises(RuntimeError, match="missing1, missing2"): require_executables(["present", "missing1", "missing2"]) @@ -185,8 +185,8 @@ def test_ensure_parent_dirs_creates_all_parent_directories(tmp_path: Path) -> No def test_refuse_existing_outputs_rejects_existing_without_force(tmp_path: Path) -> None: - existing = tmp_path / "sprite.bed.gz" - missing = tmp_path / "sprite.bed.gz.tbi" + existing = tmp_path / "wisp.bed.gz" + missing = tmp_path / "wisp.bed.gz.tbi" existing.write_text("old") with pytest.raises(FileExistsError, match=str(existing)): @@ -194,7 +194,7 @@ def test_refuse_existing_outputs_rejects_existing_without_force(tmp_path: Path) def test_refuse_existing_outputs_allows_existing_with_force(tmp_path: Path) -> None: - existing = tmp_path / "sprite.bed.gz" + existing = tmp_path / "wisp.bed.gz" existing.write_text("old") refuse_existing_outputs([existing], force=True) diff --git a/tests/test_vcf.py b/tests/test_vcf.py index 24257bc..08710f7 100644 --- a/tests/test_vcf.py +++ b/tests/test_vcf.py @@ -6,8 +6,8 @@ import pytest -from sprite_mask.models import Sample -from sprite_mask.vcf import ( +from wisp_mask.models import Sample +from wisp_mask.vcf import ( build_population_counts_from_all_sites_vcf, estimate_alignment_thresholds_from_variants_vcf, validate_vcf_sample_names, diff --git a/tests/test_workflow_commands.py b/tests/test_workflow_commands.py index 932a5d7..54ee877 100644 --- a/tests/test_workflow_commands.py +++ b/tests/test_workflow_commands.py @@ -3,12 +3,12 @@ import argparse from pathlib import Path -from sprite_mask.bedtools import build_multiinter_command -from sprite_mask.cli import build_parser -from sprite_mask.config import AlignmentRunConfig, VcfRunConfig -from sprite_mask.models import Sample -from sprite_mask.mosdepth import build_mosdepth_command -from sprite_mask.workflow import _required_tools, workflow_output_paths +from wisp_mask.bedtools import build_multiinter_command +from wisp_mask.cli import build_parser +from wisp_mask.config import AlignmentRunConfig, VcfRunConfig +from wisp_mask.models import Sample +from wisp_mask.mosdepth import build_mosdepth_command +from wisp_mask.workflow import _required_tools, workflow_output_paths def test_build_mosdepth_command_default_omits_fast_mode(tmp_path: Path) -> None: @@ -98,10 +98,10 @@ def test_workflow_output_paths(tmp_path: Path) -> None: outputs = workflow_output_paths(tmp_path / "results", 30) assert outputs.population_count_bed_gz == ( - tmp_path / "results" / "sprite.bed.gz" + tmp_path / "results" / "wisp.bed.gz" ) assert outputs.population_count_bed_index == ( - tmp_path / "results" / "sprite.bed.gz.tbi" + tmp_path / "results" / "wisp.bed.gz.tbi" ) diff --git a/tests/test_workflow_e2e.py b/tests/test_workflow_e2e.py index 69f1811..a96dc54 100644 --- a/tests/test_workflow_e2e.py +++ b/tests/test_workflow_e2e.py @@ -233,7 +233,7 @@ def target_sites(self) -> int: def test_cli_workflow_multichromosome_fixture_outputs_expected_counts(tmp_path: Path) -> None: fixture_case = INTEGRATION_FIXTURE require_fixture_and_tools(fixture_case) - out_dir, work_dir = run_sprite_mask(tmp_path, fixture_case, keep_work=True) + out_dir, work_dir = run_wisp_mask(tmp_path, fixture_case, keep_work=True) assert_population_count_output_matches_fixture(out_dir, fixture_case) assert_expected_work_files_exist(work_dir, fixture_case) @@ -242,7 +242,7 @@ def test_cli_workflow_multichromosome_fixture_outputs_expected_counts(tmp_path: def test_cli_workflow_fast_mode_preserves_previous_fixture_counts(tmp_path: Path) -> None: fixture_case = FAST_MODE_INTEGRATION_FIXTURE require_fixture_and_tools(fixture_case) - out_dir, work_dir = run_sprite_mask( + out_dir, work_dir = run_wisp_mask( tmp_path, fixture_case, keep_work=True, @@ -257,7 +257,7 @@ def assert_population_count_output_matches_fixture( out_dir: Path, fixture_case: FixtureCase, ) -> None: - population_count_bed_gz = out_dir / "sprite.bed.gz" + population_count_bed_gz = out_dir / "wisp.bed.gz" population_count_bed_index = Path(f"{population_count_bed_gz}.tbi") assert sorted(path.name for path in out_dir.iterdir()) == [ @@ -296,9 +296,9 @@ def assert_population_count_output_matches_fixture( def test_cli_workflow_smoke_fixture_writes_only_indexed_population_bed(tmp_path: Path) -> None: fixture_case = SMOKE_FIXTURE require_fixture_and_tools(fixture_case) - out_dir, work_dir = run_sprite_mask(tmp_path, fixture_case, keep_work=False, threads=1) + out_dir, work_dir = run_wisp_mask(tmp_path, fixture_case, keep_work=False, threads=1) - population_count_bed_gz = out_dir / "sprite.bed.gz" + population_count_bed_gz = out_dir / "wisp.bed.gz" output_names = sorted(path.name for path in out_dir.iterdir()) assert output_names == [ population_count_bed_gz.name, @@ -325,7 +325,7 @@ def require_fixture_and_tools(fixture_case: FixtureCase) -> None: pytest.skip(f"external workflow tool(s) unavailable: {', '.join(missing)}") -def run_sprite_mask( +def run_wisp_mask( tmp_path: Path, fixture_case: FixtureCase, *, @@ -339,7 +339,7 @@ def run_sprite_mask( command = [ sys.executable, "-m", - "sprite_mask.cli", + "wisp_mask.cli", "from-alignments", "--samples", str(fixture_case.samples_tsv), @@ -369,8 +369,8 @@ def run_sprite_mask( text=True, ) assert completed.stdout == "" - assert "[sprite] Analysis complete: wrote" in completed.stderr - assert "sprite.bed.gz" in completed.stderr + assert "[wisp] Analysis complete: wrote" in completed.stderr + assert "wisp.bed.gz" in completed.stderr return out_dir, work_dir @@ -396,7 +396,7 @@ def read_count_bed_header(path: Path) -> dict[str, object]: metadata: dict[str, object] | None = None for line in handle: fields = line.rstrip("\n").split("\t") - if fields[0] == "#sprite_mask_metadata": + if fields[0] == "#wisp_mask_metadata": metadata = json.loads(fields[1]) continue if fields[0] == "#chrom": diff --git a/tests/test_workflow_unit.py b/tests/test_workflow_unit.py index b69e635..1b8431b 100644 --- a/tests/test_workflow_unit.py +++ b/tests/test_workflow_unit.py @@ -8,10 +8,10 @@ import pytest -from sprite_mask.config import AlignmentRunConfig, VcfRunConfig -from sprite_mask.models import MosdepthOutputs, Sample -from sprite_mask.validation import validate_vcf_inputs -from sprite_mask.workflow import ( +from wisp_mask.config import AlignmentRunConfig, VcfRunConfig +from wisp_mask.models import MosdepthOutputs, Sample +from wisp_mask.validation import validate_vcf_inputs +from wisp_mask.workflow import ( _build_from_alignments, _build_from_all_sites_vcf, _cleanup_work_files, @@ -108,14 +108,14 @@ def test_run_workflow_alignment_mode_dispatches_and_cleans_work_files( ) calls: list[tuple[list[Sample], AlignmentRunConfig]] = [] - monkeypatch.setattr("sprite_mask.workflow.read_samples", lambda _path: [sample]) - monkeypatch.setattr("sprite_mask.workflow.require_executables", lambda _names: None) + monkeypatch.setattr("wisp_mask.workflow.read_samples", lambda _path: [sample]) + monkeypatch.setattr("wisp_mask.workflow.require_executables", lambda _names: None) monkeypatch.setattr( - "sprite_mask.workflow.validate_alignment_sample_headers", + "wisp_mask.workflow.validate_alignment_sample_headers", lambda _samples: None, ) monkeypatch.setattr( - "sprite_mask.workflow._build_from_all_sites_vcf", + "wisp_mask.workflow._build_from_all_sites_vcf", lambda *_args: pytest.fail("VCF workflow should not be called"), ) @@ -133,13 +133,13 @@ def fake_build_from_alignments( generated_work_files.append(generated) monkeypatch.setattr( - "sprite_mask.workflow._build_from_alignments", + "wisp_mask.workflow._build_from_alignments", fake_build_from_alignments, ) outputs = run_workflow(config) - assert outputs.population_count_bed_gz == tmp_path / "out" / "sprite.bed.gz" + assert outputs.population_count_bed_gz == tmp_path / "out" / "wisp.bed.gz" assert calls == [([sample], config)] assert not config.resolved_work_dir.exists() @@ -207,8 +207,8 @@ def fake_sort(in_bed: Path, out_bed: Path) -> Path: out_bed.write_text(in_bed.read_text()) return out_bed - monkeypatch.setattr("sprite_mask.workflow.normalize_targets_bed", fake_normalize) - monkeypatch.setattr("sprite_mask.workflow.sort_and_merge_bed", fake_sort) + monkeypatch.setattr("wisp_mask.workflow.normalize_targets_bed", fake_normalize) + monkeypatch.setattr("wisp_mask.workflow.sort_and_merge_bed", fake_sort) config = AlignmentRunConfig( samples_path=tmp_path / "samples.tsv", @@ -273,7 +273,7 @@ def fake_sort(in_bed: Path, out_bed: Path) -> Path: out_bed.write_text(in_bed.read_text()) return out_bed - monkeypatch.setattr("sprite_mask.workflow.sort_and_merge_bed", fake_sort) + monkeypatch.setattr("wisp_mask.workflow.sort_and_merge_bed", fake_sort) generated: list[Path] = [] exclusions = _prepare_variant_exclusions(config, generated) @@ -343,8 +343,8 @@ def fake_extract(quantized_bed_gz: Path, out_bed: Path) -> Path: out_bed.write_text("chr1\t0\t10\n") return out_bed - monkeypatch.setattr("sprite_mask.workflow.run_mosdepth", fake_run_mosdepth) - monkeypatch.setattr("sprite_mask.workflow.extract_merged_pass_intervals", fake_extract) + monkeypatch.setattr("wisp_mask.workflow.run_mosdepth", fake_run_mosdepth) + monkeypatch.setattr("wisp_mask.workflow.extract_merged_pass_intervals", fake_extract) pass_bed, mosdepth_outputs, generated = _make_sample_pass_bed( Sample("s1", "popA", tmp_path / "s1.bam"), @@ -381,11 +381,11 @@ def test_make_sample_pass_bed_with_targets_clips_merged_pass_bed( calls: list[tuple[Path, Path, Path]] = [] monkeypatch.setattr( - "sprite_mask.workflow.run_mosdepth", + "wisp_mask.workflow.run_mosdepth", lambda _sample, _config: outputs, ) monkeypatch.setattr( - "sprite_mask.workflow.extract_merged_pass_intervals", + "wisp_mask.workflow.extract_merged_pass_intervals", lambda _quantized, out_bed: out_bed.write_text("chr1\t0\t20\n") or out_bed, ) @@ -394,7 +394,7 @@ def fake_intersect(merged_pass_bed: Path, targets_bed: Path, out_bed: Path) -> P out_bed.write_text("chr1\t5\t10\n") return out_bed - monkeypatch.setattr("sprite_mask.workflow.intersect_sort_merge", fake_intersect) + monkeypatch.setattr("wisp_mask.workflow.intersect_sort_merge", fake_intersect) pass_bed, returned_outputs, generated = _make_sample_pass_bed( Sample("s1", "popA", tmp_path / "s1.bam"), @@ -427,9 +427,9 @@ def test_make_sample_pass_bed_removes_variant_exclusion_regions( variant_exclusions.write_text("chr1\t12\t14\n") calls: list[tuple[Path, Path, Path]] = [] - monkeypatch.setattr("sprite_mask.workflow.run_mosdepth", lambda _sample, _config: outputs) + monkeypatch.setattr("wisp_mask.workflow.run_mosdepth", lambda _sample, _config: outputs) monkeypatch.setattr( - "sprite_mask.workflow.extract_merged_pass_intervals", + "wisp_mask.workflow.extract_merged_pass_intervals", lambda _quantized, out_bed: out_bed.write_text("chr1\t10\t20\n") or out_bed, ) @@ -438,7 +438,7 @@ def fake_subtract(pass_bed: Path, excluded_bed: Path, out_bed: Path) -> Path: out_bed.write_text("chr1\t10\t12\nchr1\t14\t20\n") return out_bed - monkeypatch.setattr("sprite_mask.workflow.subtract_sort_merge", fake_subtract) + monkeypatch.setattr("wisp_mask.workflow.subtract_sort_merge", fake_subtract) pass_bed, returned_outputs, generated = _make_sample_pass_bed( Sample("s1", "popA", tmp_path / "s1.bam"), @@ -482,7 +482,7 @@ def fake_make_sample_pass_bed( sample_log = tmp_path / f"{sample.sample_id}.log" return pass_bed, None, [sample_log] - monkeypatch.setattr("sprite_mask.workflow._make_sample_pass_bed", fake_make_sample_pass_bed) + monkeypatch.setattr("wisp_mask.workflow._make_sample_pass_bed", fake_make_sample_pass_bed) generated: list[Path] = [] pass_beds = _make_sample_pass_beds(samples, config, None, None, generated) @@ -517,7 +517,7 @@ def fake_make_sample_pass_bed( visited.append(sample.sample_id) return tmp_path / f"{sample.sample_id}.pass.bed", None, [] - monkeypatch.setattr("sprite_mask.workflow._make_sample_pass_bed", fake_make_sample_pass_bed) + monkeypatch.setattr("wisp_mask.workflow._make_sample_pass_bed", fake_make_sample_pass_bed) pass_beds = _make_sample_pass_beds(samples, config, None, None, []) @@ -541,11 +541,11 @@ def test_build_from_alignments_uses_single_input_multiinter_for_one_sample( calls: list[str] = [] monkeypatch.setattr( - "sprite_mask.workflow._prepare_targets", + "wisp_mask.workflow._prepare_targets", lambda _config, _generated: None, ) monkeypatch.setattr( - "sprite_mask.workflow._make_sample_pass_beds", + "wisp_mask.workflow._make_sample_pass_beds", lambda _samples, _config, _target_bed, _variant_exclusion_bed, _generated: [pass_bed], ) @@ -573,15 +573,15 @@ def fake_collapse( def fake_sort(in_bed: Path, out_bed_gz: Path) -> None: calls.append("sort") assert in_bed.name == "cohort.d10.population_count_quantized.bed" - assert out_bed_gz == tmp_path / "out" / "sprite.bed.gz" + assert out_bed_gz == tmp_path / "out" / "wisp.bed.gz" - monkeypatch.setattr("sprite_mask.workflow.write_single_input_multiinter", fake_single) + monkeypatch.setattr("wisp_mask.workflow.write_single_input_multiinter", fake_single) monkeypatch.setattr( - "sprite_mask.workflow.run_multiinter", + "wisp_mask.workflow.run_multiinter", lambda *_args: pytest.fail("run_multiinter should not be called"), ) - monkeypatch.setattr("sprite_mask.workflow.collapse_population_counts", fake_collapse) - monkeypatch.setattr("sprite_mask.workflow._sort_bgzip_tabix_bed", fake_sort) + monkeypatch.setattr("wisp_mask.workflow.collapse_population_counts", fake_collapse) + monkeypatch.setattr("wisp_mask.workflow._sort_bgzip_tabix_bed", fake_sort) generated: list[Path] = [] _build_from_alignments([sample], config, generated) @@ -612,11 +612,11 @@ def test_build_from_alignments_uses_bedtools_multiinter_for_multiple_samples( calls: list[tuple[list[Path], list[str], Path]] = [] monkeypatch.setattr( - "sprite_mask.workflow._prepare_targets", + "wisp_mask.workflow._prepare_targets", lambda _config, _generated: None, ) monkeypatch.setattr( - "sprite_mask.workflow._make_sample_pass_beds", + "wisp_mask.workflow._make_sample_pass_beds", lambda _samples, _config, _target_bed, _variant_exclusion_bed, _generated: pass_beds, ) @@ -625,15 +625,15 @@ def fake_multi(pass_beds_arg: Sequence[Path], names: Sequence[str], out_tsv: Pat out_tsv.write_text("chrom\tstart\tend\tnum\tlist\ts1\ts2\n") return out_tsv - monkeypatch.setattr("sprite_mask.workflow.run_multiinter", fake_multi) + monkeypatch.setattr("wisp_mask.workflow.run_multiinter", fake_multi) monkeypatch.setattr( - "sprite_mask.workflow.collapse_population_counts", + "wisp_mask.workflow.collapse_population_counts", lambda _samples, _multiinter, output_bed, *, metadata: output_bed.write_text( "#chrom\tstart\tend\tpopA\tpopB\n" ) or output_bed, ) - monkeypatch.setattr("sprite_mask.workflow._sort_bgzip_tabix_bed", lambda *_args: None) + monkeypatch.setattr("wisp_mask.workflow._sort_bgzip_tabix_bed", lambda *_args: None) _build_from_alignments(samples, config, []) @@ -696,10 +696,10 @@ def fake_sort(in_bed: Path, out_bed_gz: Path) -> None: calls["sort"] = (in_bed, out_bed_gz) monkeypatch.setattr( - "sprite_mask.workflow.build_population_counts_from_all_sites_vcf", + "wisp_mask.workflow.build_population_counts_from_all_sites_vcf", fake_build, ) - monkeypatch.setattr("sprite_mask.workflow._sort_bgzip_tabix_bed", fake_sort) + monkeypatch.setattr("wisp_mask.workflow._sort_bgzip_tabix_bed", fake_sort) generated: list[Path] = [] _build_from_all_sites_vcf(samples, config, generated) @@ -718,7 +718,7 @@ def fake_sort(in_bed: Path, out_bed_gz: Path) -> None: "mask_bed": str(targets), "snps_only": False, } - assert calls["sort"] == (output_bed, tmp_path / "out" / "sprite.bed.gz") + assert calls["sort"] == (output_bed, tmp_path / "out" / "wisp.bed.gz") def test_sort_bgzip_tabix_bed_preserves_headers_sorts_body_and_removes_temps( @@ -727,14 +727,14 @@ def test_sort_bgzip_tabix_bed_preserves_headers_sorts_body_and_removes_temps( ) -> None: in_bed = tmp_path / "population_count.bed" in_bed.write_text( - "#sprite_mask_metadata\t{}\n" + "#wisp_mask_metadata\t{}\n" "chrom\tstart\tend\tpopA\n" "chr2\t5\t6\t1\n" "chr1\t2\t3\t1\n" "\n" "chr1\t0\t1\t1\n" ) - out_bed_gz = tmp_path / "out" / "sprite.bed.gz" + out_bed_gz = tmp_path / "out" / "wisp.bed.gz" def fake_run( command: list[str], @@ -766,22 +766,22 @@ def fake_run( return subprocess.CompletedProcess(command, 0) raise AssertionError(f"unexpected command: {command}") - monkeypatch.setattr("sprite_mask.workflow.subprocess.run", fake_run) + monkeypatch.setattr("wisp_mask.workflow.subprocess.run", fake_run) _sort_bgzip_tabix_bed(in_bed, out_bed_gz) with gzip.open(out_bed_gz, "rt") as handle: assert handle.read() == ( - "#sprite_mask_metadata\t{}\n" + "#wisp_mask_metadata\t{}\n" "#chrom\tstart\tend\tpopA\n" "chr1\t0\t1\t1\n" "chr1\t2\t3\t1\n" "chr2\t5\t6\t1\n" ) assert Path(f"{out_bed_gz}.tbi").read_text() == "index" - assert not (tmp_path / "out" / "sprite.bed").exists() - assert not (tmp_path / "out" / "sprite.bed.body").exists() - assert not (tmp_path / "out" / "sprite.bed.sorted_body").exists() + assert not (tmp_path / "out" / "wisp.bed").exists() + assert not (tmp_path / "out" / "wisp.bed.body").exists() + assert not (tmp_path / "out" / "wisp.bed.sorted_body").exists() def test_cleanup_work_files_removes_known_files_and_empty_work_dir(tmp_path: Path) -> None: @@ -816,7 +816,7 @@ def test_full_all_sites_run_workflow_rejects_existing_outputs( tmp_path: Path, monkeypatch: pytest.MonkeyPatch, ) -> None: - population_bed = tmp_path / "out" / "sprite.bed.gz" + population_bed = tmp_path / "out" / "wisp.bed.gz" population_bed.parent.mkdir() population_bed.write_text("old") vcf = tmp_path / "all_sites.vcf" @@ -827,7 +827,7 @@ def test_full_all_sites_run_workflow_rejects_existing_outputs( popfile = tmp_path / "popfile.tsv" popfile.write_text("sample_id\tpopulation\ns1\tpopA\n") - monkeypatch.setattr("sprite_mask.workflow.require_executables", lambda _names: None) + monkeypatch.setattr("wisp_mask.workflow.require_executables", lambda _names: None) with pytest.raises(FileExistsError, match="pass --force"): run_workflow(