From 32110d9fea9d1d16a53c10c935a02cbbbc2071a6 Mon Sep 17 00:00:00 2001 From: Yann Vanrobaeys Date: Wed, 23 Sep 2026 13:01:09 -0400 Subject: [PATCH 01/11] Vendor modkit/dmr module ahead of upstream merge The module doesn't exist in nf-core/modules yet; this vendors it from a PR open against nf-core/modules (nf-core/modules#13021, with its own companion test-data PR nf-core/test-datasets#2289 for the bedmethyl fixture). modules.json records the source as YannVRB/modules.git@add-modkit-dmr rather than the canonical nf-core/modules.git, matching what `nf-core modules install --git-remote` produces for an unmerged module -- to be re-pointed at the canonical repo/branch/sha (via `nf-core modules update`) once that PR merges. Thin wrapper around `modkit dmr pair`: compares methylation between a pair of bgzip+tabix bedMethyl inputs over a set of regions and reports differentially methylated regions. Co-Authored-By: Claude Sonnet 5 Claude-Session: https://claude.ai/code/session_01Wit78XR11V9AMAgY7gGrMT --- modules.json | 11 ++ modules/nf-core/modkit/dmr/environment.yml | 7 + modules/nf-core/modkit/dmr/main.nf | 47 +++++ modules/nf-core/modkit/dmr/meta.yml | 122 ++++++++++++ modules/nf-core/modkit/dmr/tests/main.nf.test | 126 +++++++++++++ .../modkit/dmr/tests/main.nf.test.snap | 173 ++++++++++++++++++ .../nf-core/modkit/dmr/tests/nextflow.config | 5 + 7 files changed, 491 insertions(+) create mode 100644 modules/nf-core/modkit/dmr/environment.yml create mode 100644 modules/nf-core/modkit/dmr/main.nf create mode 100644 modules/nf-core/modkit/dmr/meta.yml create mode 100644 modules/nf-core/modkit/dmr/tests/main.nf.test create mode 100644 modules/nf-core/modkit/dmr/tests/main.nf.test.snap create mode 100644 modules/nf-core/modkit/dmr/tests/nextflow.config diff --git a/modules.json b/modules.json index ab8746ac..4fb1a9e1 100644 --- a/modules.json +++ b/modules.json @@ -249,6 +249,17 @@ } } } + }, + "https://github.com/YannVRB/modules.git": { + "modules": { + "nf-core": { + "modkit/dmr": { + "branch": "add-modkit-dmr", + "git_sha": "9d93dfaa3dbe224588d2f7c10dce082528854dec", + "installed_by": ["modules"] + } + } + } } } } diff --git a/modules/nf-core/modkit/dmr/environment.yml b/modules/nf-core/modkit/dmr/environment.yml new file mode 100644 index 00000000..62b97863 --- /dev/null +++ b/modules/nf-core/modkit/dmr/environment.yml @@ -0,0 +1,7 @@ +--- +# yaml-language-server: $schema=https://raw.githubusercontent.com/nf-core/modules/master/modules/environment-schema.json +channels: + - conda-forge + - bioconda +dependencies: + - ont-modkit=0.6.1 diff --git a/modules/nf-core/modkit/dmr/main.nf b/modules/nf-core/modkit/dmr/main.nf new file mode 100644 index 00000000..2ee99685 --- /dev/null +++ b/modules/nf-core/modkit/dmr/main.nf @@ -0,0 +1,47 @@ +process MODKIT_DMR { + tag "${meta.id}" + label 'process_medium' + + conda "${moduleDir}/environment.yml" + container "${workflow.containerEngine in ['singularity', 'apptainer'] && !task.ext.singularity_pull_docker_container + ? 'https://depot.galaxyproject.org/singularity/ont-modkit:0.6.1--hcdda2d0_0' + : 'quay.io/biocontainers/ont-modkit:0.6.1--hcdda2d0_0'}" + + input: + tuple val(meta), path(bedmethyl_a), path(bedmethyl_a_tbi) + tuple val(meta2), path(bedmethyl_b), path(bedmethyl_b_tbi) + tuple val(meta3), path(regions_bed) + tuple val(meta4), path(fasta) + + output: + tuple val(meta), path("*.bed"), emit: bed + tuple val(meta), path("*.log"), emit: log + tuple val("${task.process}"), val('modkit'), eval("modkit --version | sed 's/modkit //'"), emit: versions_modkit, topic: versions + + when: + task.ext.when == null || task.ext.when + + script: + def args = task.ext.args ?: '' + def prefix = task.ext.prefix ?: "${meta.id}" + def regions = regions_bed ? "-r ${regions_bed}" : '' + """ + modkit \\ + dmr pair \\ + -a ${bedmethyl_a} \\ + -b ${bedmethyl_b} \\ + ${regions} \\ + --ref ${fasta} \\ + -o ${prefix}.bed \\ + --log-filepath ${prefix}.log \\ + -t ${task.cpus} \\ + ${args} + """ + + stub: + def prefix = task.ext.prefix ?: "${meta.id}" + """ + touch ${prefix}.bed + touch ${prefix}.log + """ +} diff --git a/modules/nf-core/modkit/dmr/meta.yml b/modules/nf-core/modkit/dmr/meta.yml new file mode 100644 index 00000000..a20cfda7 --- /dev/null +++ b/modules/nf-core/modkit/dmr/meta.yml @@ -0,0 +1,122 @@ +name: modkit_dmr +description: Compare regions between a pair of bedMethyl samples (e.g. two haplotypes, + or tumor and normal) and report differentially methylated regions +keywords: + - methylation + - ont + - long-read + - differential methylation +tools: + - "modkit": + description: A bioinformatics tool for working with modified bases in Oxford + Nanopore sequencing data + homepage: https://github.com/nanoporetech/modkit + documentation: https://github.com/nanoporetech/modkit + tool_dev_url: https://github.com/nanoporetech/modkit + licence: ["Oxford Nanopore Technologies PLC. Public License Version 1.0"] + identifier: "" +input: + - - meta: + type: map + description: | + Groovy Map containing sample information for the first (control) input + e.g. `[ id:'test' ]` + - bedmethyl_a: + type: file + description: Bgzipped bedMethyl file for the first (control) sample, as + produced by modkit pileup with --bgzf + pattern: "*.bed.gz" + ontologies: [] + - bedmethyl_a_tbi: + type: file + description: Tabix index for bedmethyl_a. Must have the same basename with + a .tbi extension next to bedmethyl_a + pattern: "*.bed.gz.tbi" + ontologies: [] + - - meta2: + type: map + description: | + Groovy Map containing sample information for the second (experimental) + input e.g. `[ id:'test' ]` + - bedmethyl_b: + type: file + description: Bgzipped bedMethyl file for the second (experimental) sample, + as produced by modkit pileup with --bgzf + pattern: "*.bed.gz" + ontologies: [] + - bedmethyl_b_tbi: + type: file + description: Tabix index for bedmethyl_b. Must have the same basename with + a .tbi extension next to bedmethyl_b + pattern: "*.bed.gz.tbi" + ontologies: [] + - - meta3: + type: map + description: | + Groovy Map containing regions information + e.g. `[ id:'regions' ]` + - regions_bed: + type: file + description: Optional BED file of regions over which to compare methylation + levels. When omitted, methylation levels are compared at each site instead + pattern: "*.bed" + ontologies: [] + - - meta4: + type: map + description: | + Groovy Map containing reference information + e.g. `[ id:'hg38' ]` + - fasta: + type: file + description: Reference sequence in FASTA format, used during pileup/alignment + pattern: "*.{fa,fasta}" + ontologies: [] +output: + bed: + - - meta: + type: map + description: | + Groovy Map containing sample information + e.g. `[ id:'test' ]` + - "*.bed": + type: file + description: BED file with the score column indicating the magnitude of + the difference in methylation between the two samples + pattern: "*.bed" + ontologies: [] + log: + - - meta: + type: map + description: | + Groovy Map containing sample information + e.g. `[ id:'test' ]` + - "*.log": + type: file + description: File for debug logs to be written to + pattern: "*.log" + ontologies: [] + versions_modkit: + - - ${task.process}: + type: string + description: The name of the process + - modkit: + type: string + description: The name of the tool + - modkit --version | sed 's/modkit //': + type: eval + description: The expression to obtain the version of the tool +topics: + versions: + - - ${task.process}: + type: string + description: The name of the process + - modkit: + type: string + description: The name of the tool + - modkit --version | sed 's/modkit //': + type: eval + description: The expression to obtain the version of the tool +authors: + - "@YannVRB" +maintainers: + - "@YannVRB" diff --git a/modules/nf-core/modkit/dmr/tests/main.nf.test b/modules/nf-core/modkit/dmr/tests/main.nf.test new file mode 100644 index 00000000..e6d77652 --- /dev/null +++ b/modules/nf-core/modkit/dmr/tests/main.nf.test @@ -0,0 +1,126 @@ +nextflow_process { + + name "Test Process MODKIT_DMR" + script "../main.nf" + tag "modules" + tag "modules_nfcore" + tag "modkit" + tag "modkit/dmr" + process "MODKIT_DMR" + config "./nextflow.config" + + test("[bedmethyl_a, tbi], [bedmethyl_b, tbi], regions_bed, fasta") { + + when { + params { + module_args = '--base C' + } + process { + """ + input[0] = [ + [ id: 'test' ], + file(params.modules_testdata_base_path + 'genomics/homo_sapiens/nanopore/bedmethyl/test_hp1.bed.gz', checkIfExists: true), + file(params.modules_testdata_base_path + 'genomics/homo_sapiens/nanopore/bedmethyl/test_hp1.bed.gz.tbi', checkIfExists: true) + ] + input[1] = [ + [ id: 'test' ], + file(params.modules_testdata_base_path + 'genomics/homo_sapiens/nanopore/bedmethyl/test_hp2.bed.gz', checkIfExists: true), + file(params.modules_testdata_base_path + 'genomics/homo_sapiens/nanopore/bedmethyl/test_hp2.bed.gz.tbi', checkIfExists: true) + ] + input[2] = Channel.of('chr22\t0\t40001') + .collectFile(name: 'chr22.bed', newLine: true) + .map { file -> [ [ id:'chr22' ], file ] } + input[3] = [ + [ id: 'test_ref' ], + file(params.modules_testdata_base_path + 'genomics/homo_sapiens/genome/genome.fasta', checkIfExists: true) + ] + """ + } + } + + then { + assertAll ( + { assert process.success }, + { assert snapshot(process.out).match() } + ) + } + + } + + test("[bedmethyl_a, tbi], [bedmethyl_b, tbi], [], fasta") { + + when { + params { + module_args = '--base C' + } + process { + """ + input[0] = [ + [ id: 'test' ], + file(params.modules_testdata_base_path + 'genomics/homo_sapiens/nanopore/bedmethyl/test_hp1.bed.gz', checkIfExists: true), + file(params.modules_testdata_base_path + 'genomics/homo_sapiens/nanopore/bedmethyl/test_hp1.bed.gz.tbi', checkIfExists: true) + ] + input[1] = [ + [ id: 'test' ], + file(params.modules_testdata_base_path + 'genomics/homo_sapiens/nanopore/bedmethyl/test_hp2.bed.gz', checkIfExists: true), + file(params.modules_testdata_base_path + 'genomics/homo_sapiens/nanopore/bedmethyl/test_hp2.bed.gz.tbi', checkIfExists: true) + ] + input[2] = [ [ id:'regions' ], [] ] + input[3] = [ + [ id: 'test_ref' ], + file(params.modules_testdata_base_path + 'genomics/homo_sapiens/genome/genome.fasta', checkIfExists: true) + ] + """ + } + } + + then { + assertAll ( + { assert process.success }, + { assert snapshot(process.out).match() } + ) + } + + } + + test("[bedmethyl_a, tbi], [bedmethyl_b, tbi], regions_bed, fasta - stub") { + + options "-stub" + + when { + params { + module_args = '--base C' + } + process { + """ + input[0] = [ + [ id: 'test' ], + file(params.modules_testdata_base_path + 'genomics/homo_sapiens/nanopore/bedmethyl/test_hp1.bed.gz', checkIfExists: true), + file(params.modules_testdata_base_path + 'genomics/homo_sapiens/nanopore/bedmethyl/test_hp1.bed.gz.tbi', checkIfExists: true) + ] + input[1] = [ + [ id: 'test' ], + file(params.modules_testdata_base_path + 'genomics/homo_sapiens/nanopore/bedmethyl/test_hp2.bed.gz', checkIfExists: true), + file(params.modules_testdata_base_path + 'genomics/homo_sapiens/nanopore/bedmethyl/test_hp2.bed.gz.tbi', checkIfExists: true) + ] + input[2] = Channel.of('chr22\t0\t40001') + .collectFile(name: 'chr22.bed', newLine: true) + .map { file -> [ [ id:'chr22' ], file ] } + input[3] = [ + [ id: 'test_ref' ], + file(params.modules_testdata_base_path + 'genomics/homo_sapiens/genome/genome.fasta', checkIfExists: true) + ] + """ + } + } + + then { + assertAll ( + { assert process.success }, + { assert snapshot(process.out).match() } + ) + } + + } + +} diff --git a/modules/nf-core/modkit/dmr/tests/main.nf.test.snap b/modules/nf-core/modkit/dmr/tests/main.nf.test.snap new file mode 100644 index 00000000..7b82e676 --- /dev/null +++ b/modules/nf-core/modkit/dmr/tests/main.nf.test.snap @@ -0,0 +1,173 @@ +{ + "[bedmethyl_a, tbi], [bedmethyl_b, tbi], regions_bed, fasta": { + "content": [ + { + "0": [ + [ + { + "id": "test" + }, + "test.bed:md5,c57c0b35ce975dbafa19f3b4b5604dec" + ] + ], + "1": [ + [ + { + "id": "test" + }, + "test.log:md5,ee178bddc200e65da14440f5710f762e" + ] + ], + "2": [ + [ + "MODKIT_DMR", + "modkit", + "0.6.1" + ] + ], + "bed": [ + [ + { + "id": "test" + }, + "test.bed:md5,c57c0b35ce975dbafa19f3b4b5604dec" + ] + ], + "log": [ + [ + { + "id": "test" + }, + "test.log:md5,ee178bddc200e65da14440f5710f762e" + ] + ], + "versions_modkit": [ + [ + "MODKIT_DMR", + "modkit", + "0.6.1" + ] + ] + } + ], + "meta": { + "nf-test": "0.9.0", + "nextflow": "25.10.2" + }, + "timestamp": "2026-09-23T11:41:19.857952367" + }, + "[bedmethyl_a, tbi], [bedmethyl_b, tbi], [], fasta": { + "content": [ + { + "0": [ + [ + { + "id": "test" + }, + "test.bed:md5,e1ed04436fd4aee1c05fa4a612846417" + ] + ], + "1": [ + [ + { + "id": "test" + }, + "test.log:md5,8861572cf65176e7d07256479c169269" + ] + ], + "2": [ + [ + "MODKIT_DMR", + "modkit", + "0.6.1" + ] + ], + "bed": [ + [ + { + "id": "test" + }, + "test.bed:md5,e1ed04436fd4aee1c05fa4a612846417" + ] + ], + "log": [ + [ + { + "id": "test" + }, + "test.log:md5,8861572cf65176e7d07256479c169269" + ] + ], + "versions_modkit": [ + [ + "MODKIT_DMR", + "modkit", + "0.6.1" + ] + ] + } + ], + "meta": { + "nf-test": "0.9.0", + "nextflow": "25.10.2" + }, + "timestamp": "2026-09-23T11:41:25.583014363" + }, + "[bedmethyl_a, tbi], [bedmethyl_b, tbi], regions_bed, fasta - stub": { + "content": [ + { + "0": [ + [ + { + "id": "test" + }, + "test.bed:md5,d41d8cd98f00b204e9800998ecf8427e" + ] + ], + "1": [ + [ + { + "id": "test" + }, + "test.log:md5,d41d8cd98f00b204e9800998ecf8427e" + ] + ], + "2": [ + [ + "MODKIT_DMR", + "modkit", + "0.6.1" + ] + ], + "bed": [ + [ + { + "id": "test" + }, + "test.bed:md5,d41d8cd98f00b204e9800998ecf8427e" + ] + ], + "log": [ + [ + { + "id": "test" + }, + "test.log:md5,d41d8cd98f00b204e9800998ecf8427e" + ] + ], + "versions_modkit": [ + [ + "MODKIT_DMR", + "modkit", + "0.6.1" + ] + ] + } + ], + "meta": { + "nf-test": "0.9.0", + "nextflow": "25.10.2" + }, + "timestamp": "2026-09-23T11:41:31.474420326" + } +} \ No newline at end of file diff --git a/modules/nf-core/modkit/dmr/tests/nextflow.config b/modules/nf-core/modkit/dmr/tests/nextflow.config new file mode 100644 index 00000000..3ed1ba5e --- /dev/null +++ b/modules/nf-core/modkit/dmr/tests/nextflow.config @@ -0,0 +1,5 @@ +process { + withName: 'MODKIT_DMR' { + ext.args = params.module_args + } +} From 3a7b8f0498413caab8d240592eeab8d8fb047e02 Mon Sep 17 00:00:00 2001 From: Yann Vanrobaeys Date: Wed, 23 Sep 2026 13:11:19 -0400 Subject: [PATCH 02/11] Add DMR params and reference config --skip_dmr (default false, only takes effect when --modkit_phased is also set). dmr_cpg_islands_bed/dmr_gencode_gene_bed per genome in igenomes.config, both genome-wide, pre-cleaned (bin column stripped from UCSC's cpgIslandExt dump; gene-level BED derived from GENCODE/CAT-Liftoff GTFs) and chr-named consistently -- see the add-dmr-reference-fixtures branch on IntGenomicsLab/test-datasets (pending PR) for full provenance. Temporarily points at the YannVRB fork branch, to be re-pointed at IntGenomicsLab/test-datasets/main/... once that PR merges (marked with TODOs). Co-Authored-By: Claude Sonnet 5 Claude-Session: https://claude.ai/code/session_01Wit78XR11V9AMAgY7gGrMT --- conf/igenomes.config | 8 ++++++++ nextflow.config | 3 +++ nextflow_schema.json | 4 ++++ workflows/lrsomatic.nf | 2 ++ 4 files changed, 17 insertions(+) diff --git a/conf/igenomes.config b/conf/igenomes.config index 71ff9f59..d35e9966 100644 --- a/conf/igenomes.config +++ b/conf/igenomes.config @@ -24,6 +24,10 @@ params.genomes = [ vep_species : "homo_sapiens", savana_contigs : "https://raw.githubusercontent.com/cortes-ciriano-lab/savana/main/example/contigs.chr.hg38.txt", savana_g1000_vcf : "1000g_hg38", + // TODO: re-point at IntGenomicsLab/test-datasets/main/... once its add-dmr-reference-fixtures + // branch merges (currently on the YannVRB fork) + dmr_cpg_islands_bed : "https://raw.githubusercontent.com/YannVRB/test-datasets/add-dmr-reference-fixtures/references/dmr/GRCh38.cpg_islands.bed.gz", + dmr_gencode_gene_bed : "https://raw.githubusercontent.com/YannVRB/test-datasets/add-dmr-reference-fixtures/references/dmr/GRCh38.gencode_gene.bed.gz", vep_alphamissense : "https://storage.googleapis.com/dm_alphamissense/AlphaMissense_hg38.tsv.gz", vep_alphamissense_tbi : "https://g-608c0c.273595.03c0.data.globus.org/VEP_plugins/AlphaMissense_hg38.tsv.gz.tbi", // A dated release rather than the rolling vcf_GRCh38/clinvar.vcf.gz, whose VCF and @@ -54,6 +58,10 @@ params.genomes = [ vep_species : "homo_sapiens_gca009914755v4", savana_contigs : "https://raw.githubusercontent.com/IntGenomicsLab/test-datasets/main/references/savana/contigs.chr.chm13.txt", savana_g1000_vcf : "1000g_t2t", + // TODO: re-point at IntGenomicsLab/test-datasets/main/... once its add-dmr-reference-fixtures + // branch merges (currently on the YannVRB fork) + dmr_cpg_islands_bed : "https://raw.githubusercontent.com/YannVRB/test-datasets/add-dmr-reference-fixtures/references/dmr/CHM13.cpg_islands.bed.gz", + dmr_gencode_gene_bed : "https://raw.githubusercontent.com/YannVRB/test-datasets/add-dmr-reference-fixtures/references/dmr/CHM13.gencode_gene.bed.gz", vep_alphamissense_aa : "https://g-608c0c.273595.03c0.data.globus.org/VEP_plugins/alphamissense_protein_v2023_uniprot-2026_03.tsv.gz", vep_alphamissense_aa_tbi : "https://g-608c0c.273595.03c0.data.globus.org/VEP_plugins/alphamissense_protein_v2023_uniprot-2026_03.tsv.gz.tbi", // Pinned to a release rather than current_variation/, which moves at every Ensembl release diff --git a/nextflow.config b/nextflow.config index 696d5161..a4db7971 100644 --- a/nextflow.config +++ b/nextflow.config @@ -29,6 +29,9 @@ params { modkit_args = '--cpg --modified-bases 5mC' modkit_phased = false + // DMR options -- only takes effect when modkit_phased is also true + skip_dmr = false + // PON Options clairsto_pon_vcfs = null clairsto_pon_flags = null diff --git a/nextflow_schema.json b/nextflow_schema.json index 713e2baa..cdce36e3 100644 --- a/nextflow_schema.json +++ b/nextflow_schema.json @@ -547,6 +547,10 @@ "type": "boolean", "description": "Skip SAVANA (SV + copy-number calling)" }, + "skip_dmr": { + "type": "boolean", + "description": "Skip differential methylation region (DMR) calling between haplotypes. Only takes effect when --modkit_phased is also set; has no effect otherwise." + }, "skip_m6a": { "type": "boolean", "description": "Skip m6a calling by Fibertools" diff --git a/workflows/lrsomatic.nf b/workflows/lrsomatic.nf index 37bd1c4c..97de7479 100644 --- a/workflows/lrsomatic.nf +++ b/workflows/lrsomatic.nf @@ -111,6 +111,8 @@ workflow LRSOMATIC { params.bed_file = getGenomeAttribute('bed_file') params.savana_contigs = getGenomeAttribute('savana_contigs') params.savana_g1000_vcf = getGenomeAttribute('savana_g1000_vcf') + params.dmr_cpg_islands_bed = getGenomeAttribute('dmr_cpg_islands_bed') + params.dmr_gencode_gene_bed = getGenomeAttribute('dmr_gencode_gene_bed') params.vep_genome = getGenomeAttribute('vep_genome') params.vep_species = getGenomeAttribute('vep_species') params.sigprofiler_genome = getGenomeAttribute('sigprofiler_genome') From fdc73a6bae1b9c6b0af117748e9df3245b6ec100 Mon Sep 17 00:00:00 2001 From: Yann Vanrobaeys Date: Wed, 23 Sep 2026 18:24:27 -0400 Subject: [PATCH 03/11] Add haplotype DMR subworkflow, local modules, and vendored nf-core dependencies Wires the vendored modkit/dmr module into a new subworkflows/local/dmr.nf: MODKIT_PILEUP's phased hp1/hp2 bedMethyl -> HTSLIB_BGZIPTABIX -> CpG-island dual-haplotype-coverage restriction (new local DMR_HAPLOTYPE_REGIONS module, porting NCP's dmr_calling.sh pattern) -> MODKIT_DMR -> nearest-gene annotation via bedtools closest (new local DMR_NEAREST_GENE module). Runs automatically whenever --modkit_phased is set, gated by a new --skip_dmr. Adds GRCh38/CHM13 dmr_cpg_islands_bed/dmr_gencode_gene_bed igenomes.config entries (currently pointing at the IntGenomicsLab/test-datasets fork branch pending that PR's merge) and vendors three canonical nf-core/modules (htslib/bgziptabix, bedtools/intersect, bedtools/closest) plus the not-yet-merged modkit/dmr module. Fixes found via local testing (not committed as a pipeline-level nf-test yet -- a properly-sourced, public methylation-tagged fixture is still needed; none of lrsomatic's existing CI BAMs carry any MM/ML tags): - HTSLIB_BGZIPTABIX output filename collision between a sample's hp1/hp2 outputs -- ext.prefix now includes meta.haplotype. - DMR_NEAREST_GENE's `sort -k1,1V` isn't supported by BusyBox sort (bundled in the bedtools:2.31.1 biocontainer, which ships a coreutils-minimal sort) -- switched to plain `-k1,1 -k2,2n`. Co-Authored-By: Claude Sonnet 5 Claude-Session: https://claude.ai/code/session_01Wit78XR11V9AMAgY7gGrMT --- conf/modules.config | 34 ++ modules.json | 15 + .../dmr/haplotype_regions/environment.yml | 7 + modules/local/dmr/haplotype_regions/main.nf | 54 ++ modules/local/dmr/haplotype_regions/meta.yml | 76 +++ .../local/dmr/nearest_gene/environment.yml | 7 + modules/local/dmr/nearest_gene/main.nf | 45 ++ modules/local/dmr/nearest_gene/meta.yml | 51 ++ .../nf-core/bedtools/closest/environment.yml | 7 + modules/nf-core/bedtools/closest/main.nf | 53 ++ modules/nf-core/bedtools/closest/meta.yml | 74 +++ .../bedtools/closest/tests/main.nf.test | 157 ++++++ .../bedtools/closest/tests/main.nf.test.snap | 248 +++++++++ .../bedtools/intersect/environment.yml | 7 + modules/nf-core/bedtools/intersect/main.nf | 49 ++ modules/nf-core/bedtools/intersect/meta.yml | 83 +++ .../bedtools/intersect/tests/main.nf.test | 90 +++ .../intersect/tests/main.nf.test.snap | 125 +++++ .../bedtools/intersect/tests/nextflow.config | 5 + .../nf-core/htslib/bgziptabix/environment.yml | 8 + modules/nf-core/htslib/bgziptabix/main.nf | 88 +++ modules/nf-core/htslib/bgziptabix/meta.yml | 125 +++++ .../htslib/bgziptabix/tests/main.nf.test | 435 +++++++++++++++ .../htslib/bgziptabix/tests/main.nf.test.snap | 527 ++++++++++++++++++ subworkflows/local/dmr.nf | 127 +++++ workflows/lrsomatic.nf | 18 + 26 files changed, 2515 insertions(+) create mode 100644 modules/local/dmr/haplotype_regions/environment.yml create mode 100644 modules/local/dmr/haplotype_regions/main.nf create mode 100644 modules/local/dmr/haplotype_regions/meta.yml create mode 100644 modules/local/dmr/nearest_gene/environment.yml create mode 100644 modules/local/dmr/nearest_gene/main.nf create mode 100644 modules/local/dmr/nearest_gene/meta.yml create mode 100644 modules/nf-core/bedtools/closest/environment.yml create mode 100644 modules/nf-core/bedtools/closest/main.nf create mode 100644 modules/nf-core/bedtools/closest/meta.yml create mode 100644 modules/nf-core/bedtools/closest/tests/main.nf.test create mode 100644 modules/nf-core/bedtools/closest/tests/main.nf.test.snap create mode 100644 modules/nf-core/bedtools/intersect/environment.yml create mode 100644 modules/nf-core/bedtools/intersect/main.nf create mode 100644 modules/nf-core/bedtools/intersect/meta.yml create mode 100644 modules/nf-core/bedtools/intersect/tests/main.nf.test create mode 100644 modules/nf-core/bedtools/intersect/tests/main.nf.test.snap create mode 100644 modules/nf-core/bedtools/intersect/tests/nextflow.config create mode 100644 modules/nf-core/htslib/bgziptabix/environment.yml create mode 100644 modules/nf-core/htslib/bgziptabix/main.nf create mode 100644 modules/nf-core/htslib/bgziptabix/meta.yml create mode 100644 modules/nf-core/htslib/bgziptabix/tests/main.nf.test create mode 100644 modules/nf-core/htslib/bgziptabix/tests/main.nf.test.snap create mode 100644 subworkflows/local/dmr.nf diff --git a/conf/modules.config b/conf/modules.config index 8b20e4fd..7aafefb0 100644 --- a/conf/modules.config +++ b/conf/modules.config @@ -345,6 +345,40 @@ process { ] } + // + // SUBWORKFLOW: DMR -- intermediate steps (tabix-indexing, CpG-island restriction, the raw + // modkit dmr pair output) are not published; DMR_NEAREST_GENE's annotated bed is the final + // deliverable. + // + withName: '.*:DMR:HTSLIB_BGZIPTABIX' { + // meta.haplotype ('hp1'/'hp2') is stripped from meta again right after this step (both + // haplotypes are rejoined under the plain sample meta), so it has to be in the filename + // instead -- otherwise hp1 and hp2 both stage as .bed.gz into DMR_HAPLOTYPE_REGIONS. + ext.prefix = { "${meta.id}_${meta.haplotype}" } + ext.args2 = '-p bed' + publishDir = [ + enabled: false + ] + } + withName: '.*:DMR:DMR_HAPLOTYPE_REGIONS' { + publishDir = [ + enabled: false + ] + } + withName: '.*:DMR:MODKIT_DMR' { + ext.args = '--base C' + publishDir = [ + enabled: false + ] + } + withName: '.*:DMR:DMR_NEAREST_GENE' { + publishDir = [ + path: { "${params.outdir}/${meta.id}/methylation/${meta.type}/dmr" }, + mode: params.publish_dir_mode, + saveAs: { filename -> filename.equals('versions.yml') ? null : filename } + ] + } + withName: '.*:FIBERTOOLSRS_PREDICTM6A' { ext.args = { [ diff --git a/modules.json b/modules.json index 4fb1a9e1..d83f529f 100644 --- a/modules.json +++ b/modules.json @@ -55,6 +55,16 @@ "git_sha": "feef37435aea56816adf4b3bde1fc76aac327a8d", "installed_by": ["modules"] }, + "bedtools/closest": { + "branch": "master", + "git_sha": "cbe6025183336010a9b716e463f897b0c0f70f8a", + "installed_by": ["modules"] + }, + "bedtools/intersect": { + "branch": "master", + "git_sha": "cbe6025183336010a9b716e463f897b0c0f70f8a", + "installed_by": ["modules"] + }, "deepvariant/callvariants": { "branch": "master", "git_sha": "f2b138ee1d91f67d31c187317d7e83e429bf0309", @@ -84,6 +94,11 @@ "installed_by": ["modules"], "patch": "modules/nf-core/ensemblvep/vep/ensemblvep-vep.diff" }, + "htslib/bgziptabix": { + "branch": "master", + "git_sha": "cbe6025183336010a9b716e463f897b0c0f70f8a", + "installed_by": ["modules"] + }, "longphase/haplotag": { "branch": "master", "git_sha": "b8d30a43f33aee3148b0e9e9f00587984a4ac195", diff --git a/modules/local/dmr/haplotype_regions/environment.yml b/modules/local/dmr/haplotype_regions/environment.yml new file mode 100644 index 00000000..45c307b0 --- /dev/null +++ b/modules/local/dmr/haplotype_regions/environment.yml @@ -0,0 +1,7 @@ +--- +# yaml-language-server: $schema=https://raw.githubusercontent.com/nf-core/modules/master/modules/environment-schema.json +channels: + - conda-forge + - bioconda +dependencies: + - bioconda::bedtools=2.31.1 diff --git a/modules/local/dmr/haplotype_regions/main.nf b/modules/local/dmr/haplotype_regions/main.nf new file mode 100644 index 00000000..cc08f172 --- /dev/null +++ b/modules/local/dmr/haplotype_regions/main.nf @@ -0,0 +1,54 @@ +process DMR_HAPLOTYPE_REGIONS { + tag "$meta.id" + label 'process_single' + + conda "${moduleDir}/environment.yml" + container "${workflow.containerEngine in ['singularity', 'apptainer'] && !task.ext.singularity_pull_docker_container + ? 'https://depot.galaxyproject.org/singularity/bedtools:2.31.1--hf5e1c6e_0' + : 'quay.io/biocontainers/bedtools:2.31.1--hf5e1c6e_0'}" + + input: + // hp1/hp2 bedMethyl are not read for methylation values here, only for which CpG positions + // they cover -- restricting cpg_islands_bed to islands covered in BOTH haplotypes is what + // keeps modkit dmr pair from comparing real data on one haplotype against the other + // haplotype's absence of data at the same island. + tuple val(meta), path(hp1_bedmethyl), path(hp1_tbi), path(hp2_bedmethyl), path(hp2_tbi) + path cpg_islands_bed + tuple val(meta2), path(fai) + + output: + tuple val(meta), path("*.regions.bed"), emit: regions_bed + path "versions.yml" , emit: versions + + when: + task.ext.when == null || task.ext.when + + script: + def prefix = task.ext.prefix ?: "${meta.id}" + """ + cut -f1,2 ${fai} > genome.txt + + zcat ${hp1_bedmethyl} | awk 'BEGIN{OFS="\\t"}{print \$1,\$2,\$3}' | sort -k1,1 -k2,2n -u > hp1.cpg.bed + zcat ${hp2_bedmethyl} | awk 'BEGIN{OFS="\\t"}{print \$1,\$2,\$3}' | sort -k1,1 -k2,2n -u > hp2.cpg.bed + + bedtools intersect -u -a ${cpg_islands_bed} -b hp1.cpg.bed | sort -k1,1 -k2,2n > islands.hp1.bed + bedtools intersect -u -a ${cpg_islands_bed} -b hp2.cpg.bed | sort -k1,1 -k2,2n > islands.hp2.bed + bedtools intersect -u -a islands.hp1.bed -b islands.hp2.bed | bedtools sort -g genome.txt -i - > ${prefix}.regions.bed + + cat <<-END_VERSIONS > versions.yml + "${task.process}": + bedtools: \$(bedtools --version | sed 's/bedtools v//g') + END_VERSIONS + """ + + stub: + def prefix = task.ext.prefix ?: "${meta.id}" + """ + touch ${prefix}.regions.bed + + cat <<-END_VERSIONS > versions.yml + "${task.process}": + bedtools: 2.31.1 + END_VERSIONS + """ +} diff --git a/modules/local/dmr/haplotype_regions/meta.yml b/modules/local/dmr/haplotype_regions/meta.yml new file mode 100644 index 00000000..c3f80db6 --- /dev/null +++ b/modules/local/dmr/haplotype_regions/meta.yml @@ -0,0 +1,76 @@ +# yaml-language-server: $schema=https://raw.githubusercontent.com/nf-core/modules/master/modules/meta-schema.json +name: "dmr_haplotype_regions" +description: | + Restrict a CpG-islands BED to islands with real read coverage in both of a sample's two + haplotype bedMethyl files, for use as modkit dmr pair's --regions-bed. Without this, an island + covered in one haplotype but not the other would be compared against absence of data rather than + a real methylation difference. +keywords: + - methylation + - dmr + - haplotype + - cpg islands + - bedtools +tools: + - "bedtools": + description: "A powerful toolset for genome arithmetic" + homepage: "https://bedtools.readthedocs.io/" + documentation: "https://bedtools.readthedocs.io/" + licence: ["GPL-2.0-only"] + +input: + - - meta: + type: map + description: | + Groovy Map containing sample information, e.g. `[ id:'sample1' ]` + - hp1_bedmethyl: + type: file + description: Bgzip-compressed bedMethyl for haplotype 1 + pattern: "*.bed.gz" + - hp1_tbi: + type: file + description: Tabix index for hp1_bedmethyl + pattern: "*.bed.gz.tbi" + - hp2_bedmethyl: + type: file + description: Bgzip-compressed bedMethyl for haplotype 2 + pattern: "*.bed.gz" + - hp2_tbi: + type: file + description: Tabix index for hp2_bedmethyl + pattern: "*.bed.gz.tbi" + - - cpg_islands_bed: + type: file + description: Genome-wide CpG islands BED to restrict + pattern: "*.bed" + - - meta2: + type: map + description: | + Groovy Map containing reference information, e.g. `[ id:'GRCh38' ]` + - fai: + type: file + description: FASTA index for the reference genome (used only for its chrom/length columns) + pattern: "*.fai" + +output: + regions_bed: + - meta: + type: map + description: | + Groovy Map containing sample information, e.g. `[ id:'sample1' ]` + - "*.regions.bed": + type: file + description: | + cpg_islands_bed restricted to islands covered in both haplotypes, sorted by + genome/coordinate order, ready to pass to modkit dmr pair's --regions-bed + pattern: "*.regions.bed" + versions: + - "versions.yml": + type: file + description: File containing software versions + pattern: "versions.yml" + +authors: + - "@YannVRB" +maintainers: + - "@YannVRB" diff --git a/modules/local/dmr/nearest_gene/environment.yml b/modules/local/dmr/nearest_gene/environment.yml new file mode 100644 index 00000000..45c307b0 --- /dev/null +++ b/modules/local/dmr/nearest_gene/environment.yml @@ -0,0 +1,7 @@ +--- +# yaml-language-server: $schema=https://raw.githubusercontent.com/nf-core/modules/master/modules/environment-schema.json +channels: + - conda-forge + - bioconda +dependencies: + - bioconda::bedtools=2.31.1 diff --git a/modules/local/dmr/nearest_gene/main.nf b/modules/local/dmr/nearest_gene/main.nf new file mode 100644 index 00000000..8840a02d --- /dev/null +++ b/modules/local/dmr/nearest_gene/main.nf @@ -0,0 +1,45 @@ +process DMR_NEAREST_GENE { + tag "$meta.id" + label 'process_single' + + conda "${moduleDir}/environment.yml" + container "${workflow.containerEngine in ['singularity', 'apptainer'] && !task.ext.singularity_pull_docker_container + ? 'https://depot.galaxyproject.org/singularity/bedtools:2.31.1--hf5e1c6e_0' + : 'quay.io/biocontainers/bedtools:2.31.1--hf5e1c6e_0'}" + + input: + tuple val(meta), path(dmr_bed) + path gencode_gene_bed + + output: + tuple val(meta), path("*.nearestGene.bed"), emit: nearest_gene_bed + path "versions.yml" , emit: versions + + when: + task.ext.when == null || task.ext.when + + script: + def args = task.ext.args ?: '-D a -t first' + def prefix = task.ext.prefix ?: "${meta.id}" + """ + sort -k1,1 -k2,2n ${dmr_bed} > dmr.sorted.bed + + bedtools closest -a dmr.sorted.bed -b ${gencode_gene_bed} ${args} > ${prefix}.nearestGene.bed + + cat <<-END_VERSIONS > versions.yml + "${task.process}": + bedtools: \$(bedtools --version | sed 's/bedtools v//g') + END_VERSIONS + """ + + stub: + def prefix = task.ext.prefix ?: "${meta.id}" + """ + touch ${prefix}.nearestGene.bed + + cat <<-END_VERSIONS > versions.yml + "${task.process}": + bedtools: 2.31.1 + END_VERSIONS + """ +} diff --git a/modules/local/dmr/nearest_gene/meta.yml b/modules/local/dmr/nearest_gene/meta.yml new file mode 100644 index 00000000..39853526 --- /dev/null +++ b/modules/local/dmr/nearest_gene/meta.yml @@ -0,0 +1,51 @@ +# yaml-language-server: $schema=https://raw.githubusercontent.com/nf-core/modules/master/modules/meta-schema.json +name: "dmr_nearest_gene" +description: | + Sort a DMR bed (modkit dmr pair's output) by position and annotate each region with its nearest + gene from a GENCODE-derived gene BED, using bedtools closest. +keywords: + - methylation + - dmr + - gene annotation + - bedtools +tools: + - "bedtools": + description: "A powerful toolset for genome arithmetic" + homepage: "https://bedtools.readthedocs.io/" + documentation: "https://bedtools.readthedocs.io/" + licence: ["GPL-2.0-only"] + +input: + - - meta: + type: map + description: | + Groovy Map containing sample information, e.g. `[ id:'sample1' ]` + - dmr_bed: + type: file + description: modkit dmr pair's output BED + pattern: "*.bed" + - - gencode_gene_bed: + type: file + description: Genome-wide, position-sorted gene-level BED (chrom, start, end, gene_name, score, strand) + pattern: "*.bed" + +output: + nearest_gene_bed: + - meta: + type: map + description: | + Groovy Map containing sample information, e.g. `[ id:'sample1' ]` + - "*.nearestGene.bed": + type: file + description: dmr_bed with each region's nearest gene and signed distance appended + pattern: "*.nearestGene.bed" + versions: + - "versions.yml": + type: file + description: File containing software versions + pattern: "versions.yml" + +authors: + - "@YannVRB" +maintainers: + - "@YannVRB" diff --git a/modules/nf-core/bedtools/closest/environment.yml b/modules/nf-core/bedtools/closest/environment.yml new file mode 100644 index 00000000..45c307b0 --- /dev/null +++ b/modules/nf-core/bedtools/closest/environment.yml @@ -0,0 +1,7 @@ +--- +# yaml-language-server: $schema=https://raw.githubusercontent.com/nf-core/modules/master/modules/environment-schema.json +channels: + - conda-forge + - bioconda +dependencies: + - bioconda::bedtools=2.31.1 diff --git a/modules/nf-core/bedtools/closest/main.nf b/modules/nf-core/bedtools/closest/main.nf new file mode 100644 index 00000000..5f055bdc --- /dev/null +++ b/modules/nf-core/bedtools/closest/main.nf @@ -0,0 +1,53 @@ +process BEDTOOLS_CLOSEST { + tag "${meta.id}" + label 'process_single' + + conda "${moduleDir}/environment.yml" + container "${workflow.containerEngine in ['singularity', 'apptainer'] && !task.ext.singularity_pull_docker_container + ? 'https://depot.galaxyproject.org/singularity/bedtools:2.31.1--hf5e1c6e_0' + : 'quay.io/biocontainers/bedtools:2.31.1--hf5e1c6e_0'}" + + input: + tuple val(meta), path(input_1), path(input_2) + path fasta_fai + + output: + tuple val(meta), path("*.${extension}"), emit: output + tuple val("${task.process}"), val('bedtools'), eval("bedtools --version | sed -e 's/bedtools v//g'"), topic: versions, emit: versions_bedtools + + when: + task.ext.when == null || task.ext.when + + script: + def args = task.ext.args ?: '' + def prefix = task.ext.prefix ?: "${meta.id}" + + extension = input_1.extension == "gz" + ? (input_1 =~ /^.*\.(.*)\.gz$/)[0][1] + : input_1.extension + + def reference = fasta_fai ? "-g ${fasta_fai}" : "" + + if (input_1 == "${prefix}.${extension}" || input_2 == "${prefix}.${extension}") { + error("One of the input files is called the same as the output file. Please specify another prefix.") + } + + """ + bedtools closest \\ + ${args} \\ + -a ${input_1} \\ + -b ${input_2} \\ + ${reference} \\ + > ${prefix}.${extension} + """ + + stub: + def prefix = task.ext.prefix ?: "${meta.id}" + + extension = input_1.extension == "gz" + ? (input_1 =~ /^.*\.(.*)\.gz$/)[0][1] + : input_1.extension + """ + touch ${prefix}.${extension} + """ +} diff --git a/modules/nf-core/bedtools/closest/meta.yml b/modules/nf-core/bedtools/closest/meta.yml new file mode 100644 index 00000000..5e30eeb0 --- /dev/null +++ b/modules/nf-core/bedtools/closest/meta.yml @@ -0,0 +1,74 @@ +name: "bedtools_closest" +description: For each feature in A, finds the closest feature (upstream or downstream) + in B. +keywords: + - bedtools + - closest + - bed + - vcf + - gff +tools: + - bedtools: + description: | + A set of tools for genomic analysis tasks, specifically enabling genome arithmetic (merge, count, complement) on various file types. + documentation: https://bedtools.readthedocs.io/en/latest/content/tools/closest.html + licence: ["MIT"] + identifier: biotools:bedtools +input: + - - meta: + type: map + description: | + Groovy Map containing sample information + e.g. [ id:'test', single_end:false ] + - input_1: + type: file + description: The file to find the closest features of + pattern: "*.{bed,vcf,gff}(.gz)?" + ontologies: [] + - input_2: + type: list + description: The input file(s) to find the closest features from + pattern: "*.{bed,vcf,gff}(.gz)?" + - fasta_fai: + type: file + description: The index of the FASTA reference. Needed when the argument `--sorted` + is used + pattern: "*.fai" + ontologies: [] +output: + output: + - - meta: + type: map + description: | + Groovy Map containing sample information + e.g. [ id:'test', single_end:false ] + - "*.${extension}": + type: file + description: The resulting BED file containing the closest features + pattern: "*.{bed,vcf,gff}" + ontologies: [] + versions_bedtools: + - - ${task.process}: + type: string + description: The name of the process + - bedtools: + type: string + description: The name of the tool + - "bedtools --version | sed -e 's/bedtools v//g'": + type: eval + description: The expression to obtain the version of the tool +topics: + versions: + - - ${task.process}: + type: string + description: The name of the process + - bedtools: + type: string + description: The name of the tool + - "bedtools --version | sed -e 's/bedtools v//g'": + type: eval + description: The expression to obtain the version of the tool +authors: + - "@nvnieuwk" +maintainers: + - "@nvnieuwk" diff --git a/modules/nf-core/bedtools/closest/tests/main.nf.test b/modules/nf-core/bedtools/closest/tests/main.nf.test new file mode 100644 index 00000000..349903d9 --- /dev/null +++ b/modules/nf-core/bedtools/closest/tests/main.nf.test @@ -0,0 +1,157 @@ +nextflow_process { + name "Test Process BEDTOOLS_CLOSEST" + script "../main.nf" + process "BEDTOOLS_CLOSEST" + + tag "modules" + tag "modules_nfcore" + tag "bedtools" + tag "bedtools/closest" + + test("homo_sapiens") { + when { + process { + """ + input[0] = [ + [ id:'test' ], // meta map + file(params.modules_testdata_base_path + 'genomics/homo_sapiens/genome/genome.bed', checkIfExists: true), + [ + file(params.modules_testdata_base_path + 'genomics/homo_sapiens/genome/genome.multi_intervals.bed', checkIfExists: true), + file(params.modules_testdata_base_path + 'genomics/homo_sapiens/genome/chr21/sequence/multi_intervals.bed', checkIfExists: true) + ] + ] + input[1] = [] + """ + } + } + + then { + assertAll( + { assert process.success }, + { assert snapshot(process.out).match() } + ) + } + } + + test("homo_sapiens - fai") { + when { + process { + """ + input[0] = [ + [ id:'test' ], // meta map + file(params.modules_testdata_base_path + 'genomics/homo_sapiens/genome/genome.bed', checkIfExists: true), + file(params.modules_testdata_base_path + 'genomics/homo_sapiens/genome/genome.multi_intervals.bed', checkIfExists: true) + ] + input[1] = file(params.modules_testdata_base_path + 'genomics/homo_sapiens/genome/genome.fasta.fai', checkIfExists: true) + """ + } + } + + then { + assertAll( + { assert process.success }, + { assert snapshot(process.out).match() } + ) + } + } + + test("homo_sapiens - vcf") { + when { + process { + """ + input[0] = [ + [ id:'test' ], // meta map + file(params.modules_testdata_base_path + 'genomics/homo_sapiens/genome/genome.bed', checkIfExists: true), + [ + file(params.modules_testdata_base_path + 'genomics/homo_sapiens/genome/vcf/dbsnp_146.hg38.vcf.gz', checkIfExists: true), + file(params.modules_testdata_base_path + 'genomics/homo_sapiens/genome/vcf/gnomAD.r2.1.1.vcf.gz', checkIfExists: true) + ] + ] + input[1] = [] + """ + } + } + + then { + assertAll( + { assert process.success }, + { assert snapshot(process.out).match() } + ) + } + } + + test("homo_sapiens - stub") { + options "-stub" + when { + process { + """ + input[0] = [ + [ id:'test' ], // meta map + file(params.modules_testdata_base_path + 'genomics/homo_sapiens/genome/genome.bed', checkIfExists: true), + [ + file(params.modules_testdata_base_path + 'genomics/homo_sapiens/genome/genome.multi_intervals.bed', checkIfExists: true), + file(params.modules_testdata_base_path + 'genomics/homo_sapiens/genome/chr21/sequence/multi_intervals.bed', checkIfExists: true) + ] + ] + input[1] = [] + """ + } + } + + then { + assertAll( + { assert process.success }, + { assert snapshot(process.out).match() } + ) + } + } + + test("homo_sapiens - fai - stub") { + options "-stub" + when { + process { + """ + input[0] = [ + [ id:'test' ], // meta map + file(params.modules_testdata_base_path + 'genomics/homo_sapiens/genome/genome.bed', checkIfExists: true), + file(params.modules_testdata_base_path + 'genomics/homo_sapiens/genome/genome.multi_intervals.bed', checkIfExists: true) + ] + input[1] = file(params.modules_testdata_base_path + 'genomics/homo_sapiens/genome/genome.fasta.fai', checkIfExists: true) + """ + } + } + + then { + assertAll( + { assert process.success }, + { assert snapshot(process.out).match() } + ) + } + } + + test("homo_sapiens - vcf - stub") { + options "-stub" + when { + process { + """ + input[0] = [ + [ id:'test' ], // meta map + file(params.modules_testdata_base_path + 'genomics/homo_sapiens/genome/genome.bed', checkIfExists: true), + [ + file(params.modules_testdata_base_path + 'genomics/homo_sapiens/genome/vcf/dbsnp_146.hg38.vcf.gz', checkIfExists: true), + file(params.modules_testdata_base_path + 'genomics/homo_sapiens/genome/vcf/gnomAD.r2.1.1.vcf.gz', checkIfExists: true) + ] + ] + input[1] = [] + """ + } + } + + then { + assertAll( + { assert process.success }, + { assert snapshot(process.out).match() } + ) + } + } +} diff --git a/modules/nf-core/bedtools/closest/tests/main.nf.test.snap b/modules/nf-core/bedtools/closest/tests/main.nf.test.snap new file mode 100644 index 00000000..99895426 --- /dev/null +++ b/modules/nf-core/bedtools/closest/tests/main.nf.test.snap @@ -0,0 +1,248 @@ +{ + "homo_sapiens - vcf": { + "content": [ + { + "0": [ + [ + { + "id": "test" + }, + "test.bed:md5,85a5cee50322ab15f384df7826164d64" + ] + ], + "1": [ + [ + "BEDTOOLS_CLOSEST", + "bedtools", + "2.31.1" + ] + ], + "output": [ + [ + { + "id": "test" + }, + "test.bed:md5,85a5cee50322ab15f384df7826164d64" + ] + ], + "versions_bedtools": [ + [ + "BEDTOOLS_CLOSEST", + "bedtools", + "2.31.1" + ] + ] + } + ], + "meta": { + "nf-test": "0.9.3", + "nextflow": "25.10.2" + }, + "timestamp": "2026-01-21T11:10:06.402036653" + }, + "homo_sapiens - fai - stub": { + "content": [ + { + "0": [ + [ + { + "id": "test" + }, + "test.bed:md5,d41d8cd98f00b204e9800998ecf8427e" + ] + ], + "1": [ + [ + "BEDTOOLS_CLOSEST", + "bedtools", + "2.31.1" + ] + ], + "output": [ + [ + { + "id": "test" + }, + "test.bed:md5,d41d8cd98f00b204e9800998ecf8427e" + ] + ], + "versions_bedtools": [ + [ + "BEDTOOLS_CLOSEST", + "bedtools", + "2.31.1" + ] + ] + } + ], + "meta": { + "nf-test": "0.9.3", + "nextflow": "25.10.2" + }, + "timestamp": "2026-01-21T11:10:22.176626879" + }, + "homo_sapiens - stub": { + "content": [ + { + "0": [ + [ + { + "id": "test" + }, + "test.bed:md5,d41d8cd98f00b204e9800998ecf8427e" + ] + ], + "1": [ + [ + "BEDTOOLS_CLOSEST", + "bedtools", + "2.31.1" + ] + ], + "output": [ + [ + { + "id": "test" + }, + "test.bed:md5,d41d8cd98f00b204e9800998ecf8427e" + ] + ], + "versions_bedtools": [ + [ + "BEDTOOLS_CLOSEST", + "bedtools", + "2.31.1" + ] + ] + } + ], + "meta": { + "nf-test": "0.9.3", + "nextflow": "25.10.2" + }, + "timestamp": "2026-01-21T11:10:14.477862854" + }, + "homo_sapiens - fai": { + "content": [ + { + "0": [ + [ + { + "id": "test" + }, + "test.bed:md5,347f192c141a34dfc4bb7c13fffd44bb" + ] + ], + "1": [ + [ + "BEDTOOLS_CLOSEST", + "bedtools", + "2.31.1" + ] + ], + "output": [ + [ + { + "id": "test" + }, + "test.bed:md5,347f192c141a34dfc4bb7c13fffd44bb" + ] + ], + "versions_bedtools": [ + [ + "BEDTOOLS_CLOSEST", + "bedtools", + "2.31.1" + ] + ] + } + ], + "meta": { + "nf-test": "0.9.3", + "nextflow": "25.10.2" + }, + "timestamp": "2026-01-21T11:09:58.376835909" + }, + "homo_sapiens - vcf - stub": { + "content": [ + { + "0": [ + [ + { + "id": "test" + }, + "test.bed:md5,d41d8cd98f00b204e9800998ecf8427e" + ] + ], + "1": [ + [ + "BEDTOOLS_CLOSEST", + "bedtools", + "2.31.1" + ] + ], + "output": [ + [ + { + "id": "test" + }, + "test.bed:md5,d41d8cd98f00b204e9800998ecf8427e" + ] + ], + "versions_bedtools": [ + [ + "BEDTOOLS_CLOSEST", + "bedtools", + "2.31.1" + ] + ] + } + ], + "meta": { + "nf-test": "0.9.3", + "nextflow": "25.10.2" + }, + "timestamp": "2026-01-21T11:10:30.128998192" + }, + "homo_sapiens": { + "content": [ + { + "0": [ + [ + { + "id": "test" + }, + "test.bed:md5,de47864c0ae53891d09252eb63cd7269" + ] + ], + "1": [ + [ + "BEDTOOLS_CLOSEST", + "bedtools", + "2.31.1" + ] + ], + "output": [ + [ + { + "id": "test" + }, + "test.bed:md5,de47864c0ae53891d09252eb63cd7269" + ] + ], + "versions_bedtools": [ + [ + "BEDTOOLS_CLOSEST", + "bedtools", + "2.31.1" + ] + ] + } + ], + "meta": { + "nf-test": "0.9.3", + "nextflow": "25.10.2" + }, + "timestamp": "2026-01-21T11:09:50.231505255" + } +} \ No newline at end of file diff --git a/modules/nf-core/bedtools/intersect/environment.yml b/modules/nf-core/bedtools/intersect/environment.yml new file mode 100644 index 00000000..45c307b0 --- /dev/null +++ b/modules/nf-core/bedtools/intersect/environment.yml @@ -0,0 +1,7 @@ +--- +# yaml-language-server: $schema=https://raw.githubusercontent.com/nf-core/modules/master/modules/environment-schema.json +channels: + - conda-forge + - bioconda +dependencies: + - bioconda::bedtools=2.31.1 diff --git a/modules/nf-core/bedtools/intersect/main.nf b/modules/nf-core/bedtools/intersect/main.nf new file mode 100644 index 00000000..fbcbf530 --- /dev/null +++ b/modules/nf-core/bedtools/intersect/main.nf @@ -0,0 +1,49 @@ +process BEDTOOLS_INTERSECT { + tag "${meta.id}" + label 'process_single' + + conda "${moduleDir}/environment.yml" + container "${workflow.containerEngine in ['singularity', 'apptainer'] && !task.ext.singularity_pull_docker_container + ? 'https://depot.galaxyproject.org/singularity/bedtools:2.31.1--hf5e1c6e_0' + : 'quay.io/biocontainers/bedtools:2.31.1--hf5e1c6e_0'}" + + input: + tuple val(meta), path(intervals1), path(intervals2) + tuple val(meta2), path(chrom_sizes) + + output: + tuple val(meta), path("*.${extension}"), emit: intersect + tuple val("${task.process}"), val('bedtools'), eval("bedtools --version | sed -e 's/bedtools v//g'"), topic: versions, emit: versions_bedtools + + when: + task.ext.when == null || task.ext.when + + script: + def args = task.ext.args ?: '' + def prefix = task.ext.prefix ?: "${meta.id}" + //Extension of the output file. It is set by the user via "ext.suffix" in the config. Corresponds to the file format which depends on arguments (e. g., ".bed", ".bam", ".txt", etc.). + extension = task.ext.suffix ?: "${intervals1.extension}" + def sizes = chrom_sizes ? "-g ${chrom_sizes}" : '' + if ("${intervals1}" == "${prefix}.${extension}" || "${intervals2}" == "${prefix}.${extension}") { + error("Input and output names are the same, use \"task.ext.prefix\" to disambiguate!") + } + """ + bedtools \\ + intersect \\ + -a ${intervals1} \\ + -b ${intervals2} \\ + ${args} \\ + ${sizes} \\ + > ${prefix}.${extension} + """ + + stub: + def prefix = task.ext.prefix ?: "${meta.id}" + extension = task.ext.suffix ?: "bed" + if ("${intervals1}" == "${prefix}.${extension}" || "${intervals2}" == "${prefix}.${extension}") { + error("Input and output names are the same, use \"task.ext.prefix\" to disambiguate!") + } + """ + touch ${prefix}.${extension} + """ +} diff --git a/modules/nf-core/bedtools/intersect/meta.yml b/modules/nf-core/bedtools/intersect/meta.yml new file mode 100644 index 00000000..0500efb9 --- /dev/null +++ b/modules/nf-core/bedtools/intersect/meta.yml @@ -0,0 +1,83 @@ +name: bedtools_intersect +description: Allows one to screen for overlaps between two sets of genomic features. +keywords: + - bed + - intersect + - overlap +tools: + - bedtools: + description: | + A set of tools for genomic analysis tasks, specifically enabling genome arithmetic (merge, count, complement) on various file types. + documentation: https://bedtools.readthedocs.io/en/latest/content/tools/intersect.html + licence: ["MIT"] + identifier: biotools:bedtools +input: + - - meta: + type: map + description: | + Groovy Map containing sample information + e.g. [ id:'test', single_end:false ] + - intervals1: + type: file + description: BAM/BED/GFF/VCF + pattern: "*.{bam|bed|gff|vcf}" + ontologies: [] + - intervals2: + type: file + description: BAM/BED/GFF/VCF + pattern: "*.{bam|bed|gff|vcf}" + ontologies: [] + - - meta2: + type: map + description: | + Groovy Map containing reference chromosome sizes + e.g. [ id:'test' ] + - chrom_sizes: + type: file + description: Chromosome sizes file + pattern: "*{.sizes,.txt}" + ontologies: [] +output: + intersect: + - - meta: + type: map + description: | + Groovy Map containing sample information + e.g. [ id:'test', single_end:false ] + - "*.${extension}": + type: file + description: File containing the description of overlaps found between the + two features + pattern: "*.${extension}" + ontologies: [] + versions_bedtools: + - - ${task.process}: + type: string + description: The name of the process + - bedtools: + type: string + description: The name of the tool + - "bedtools --version | sed -e 's/bedtools v//g'": + type: eval + description: The expression to obtain the version of the tool +topics: + versions: + - - ${task.process}: + type: string + description: The name of the process + - bedtools: + type: string + description: The name of the tool + - "bedtools --version | sed -e 's/bedtools v//g'": + type: eval + description: The expression to obtain the version of the tool +authors: + - "@edmundmiller" + - "@sruthipsuresh" + - "@drpatelh" + - "@sidorov-si" +maintainers: + - "@edmundmiller" + - "@sruthipsuresh" + - "@drpatelh" + - "@sidorov-si" diff --git a/modules/nf-core/bedtools/intersect/tests/main.nf.test b/modules/nf-core/bedtools/intersect/tests/main.nf.test new file mode 100644 index 00000000..cd770946 --- /dev/null +++ b/modules/nf-core/bedtools/intersect/tests/main.nf.test @@ -0,0 +1,90 @@ +nextflow_process { + + name "Test Process BEDTOOLS_INTERSECT" + script "../main.nf" + process "BEDTOOLS_INTERSECT" + config "./nextflow.config" + + tag "modules" + tag "modules_nfcore" + tag "bedtools" + tag "bedtools/intersect" + + test("sarscov2 - bed - bed") { + + when { + process { + """ + input[0] = [ + [ id:'test' ], + file(params.modules_testdata_base_path + 'genomics/sarscov2/genome/bed/test.bed', checkIfExists: true), + file(params.modules_testdata_base_path + 'genomics/sarscov2/genome/bed/test2.bed', checkIfExists: true) + ] + + input[1] = [[:], []] + """ + } + } + + then { + assertAll( + { assert process.success }, + { assert snapshot(process.out).match() } + ) + } + + } + + test("sarscov2 - bam - bam") { + + when { + process { + """ + input[0] = [ + [ id:'test' ], + file(params.modules_testdata_base_path + 'genomics/sarscov2/illumina/bam/test.paired_end.bam', checkIfExists: true), + file(params.modules_testdata_base_path + 'genomics/sarscov2/genome/bed/baits.bed', checkIfExists: true) + ] + + input[1] = [[:], []] + """ + } + } + + then { + assertAll( + { assert process.success }, + { assert snapshot(process.out).match() } + ) + } + + } + + test("sarscov2 - bed - stub") { + + options "-stub" + + when { + process { + """ + input[0] = [ + [ id:'test' ], + file(params.modules_testdata_base_path + 'genomics/sarscov2/genome/bed/test.bed', checkIfExists: true), + file(params.modules_testdata_base_path + 'genomics/sarscov2/genome/bed/test2.bed', checkIfExists: true) + ] + + input[1] = [[:], []] + """ + } + } + + then { + assertAll( + { assert process.success }, + { assert snapshot(process.out).match() } + ) + } + + } + +} diff --git a/modules/nf-core/bedtools/intersect/tests/main.nf.test.snap b/modules/nf-core/bedtools/intersect/tests/main.nf.test.snap new file mode 100644 index 00000000..30da8be1 --- /dev/null +++ b/modules/nf-core/bedtools/intersect/tests/main.nf.test.snap @@ -0,0 +1,125 @@ +{ + "sarscov2 - bam - bam": { + "content": [ + { + "0": [ + [ + { + "id": "test" + }, + "test_out.bam:md5,738324efe2b1e442ceb6539a630c3fe6" + ] + ], + "1": [ + [ + "BEDTOOLS_INTERSECT", + "bedtools", + "2.31.1" + ] + ], + "intersect": [ + [ + { + "id": "test" + }, + "test_out.bam:md5,738324efe2b1e442ceb6539a630c3fe6" + ] + ], + "versions_bedtools": [ + [ + "BEDTOOLS_INTERSECT", + "bedtools", + "2.31.1" + ] + ] + } + ], + "meta": { + "nf-test": "0.9.3", + "nextflow": "25.10.2" + }, + "timestamp": "2026-01-21T11:31:18.811473651" + }, + "sarscov2 - bed - bed": { + "content": [ + { + "0": [ + [ + { + "id": "test" + }, + "test_out.bed:md5,afcbf01c2f2013aad71dbe8e34f2c15c" + ] + ], + "1": [ + [ + "BEDTOOLS_INTERSECT", + "bedtools", + "2.31.1" + ] + ], + "intersect": [ + [ + { + "id": "test" + }, + "test_out.bed:md5,afcbf01c2f2013aad71dbe8e34f2c15c" + ] + ], + "versions_bedtools": [ + [ + "BEDTOOLS_INTERSECT", + "bedtools", + "2.31.1" + ] + ] + } + ], + "meta": { + "nf-test": "0.9.3", + "nextflow": "25.10.2" + }, + "timestamp": "2026-01-21T11:31:11.122621263" + }, + "sarscov2 - bed - stub": { + "content": [ + { + "0": [ + [ + { + "id": "test" + }, + "test_out.bed:md5,d41d8cd98f00b204e9800998ecf8427e" + ] + ], + "1": [ + [ + "BEDTOOLS_INTERSECT", + "bedtools", + "2.31.1" + ] + ], + "intersect": [ + [ + { + "id": "test" + }, + "test_out.bed:md5,d41d8cd98f00b204e9800998ecf8427e" + ] + ], + "versions_bedtools": [ + [ + "BEDTOOLS_INTERSECT", + "bedtools", + "2.31.1" + ] + ] + } + ], + "meta": { + "nf-test": "0.9.3", + "nextflow": "25.10.2" + }, + "timestamp": "2026-01-21T11:31:27.957038806" + } +} \ No newline at end of file diff --git a/modules/nf-core/bedtools/intersect/tests/nextflow.config b/modules/nf-core/bedtools/intersect/tests/nextflow.config new file mode 100644 index 00000000..f1f9e693 --- /dev/null +++ b/modules/nf-core/bedtools/intersect/tests/nextflow.config @@ -0,0 +1,5 @@ +process { + withName: BEDTOOLS_INTERSECT { + ext.prefix = { "${meta.id}_out" } + } +} diff --git a/modules/nf-core/htslib/bgziptabix/environment.yml b/modules/nf-core/htslib/bgziptabix/environment.yml new file mode 100644 index 00000000..ec62c057 --- /dev/null +++ b/modules/nf-core/htslib/bgziptabix/environment.yml @@ -0,0 +1,8 @@ +--- +# yaml-language-server: $schema=https://raw.githubusercontent.com/nf-core/modules/master/modules/environment-schema.json +channels: + - conda-forge + - bioconda +dependencies: + - bioconda::htslib=1.24 + - conda-forge::xz=5.8.3 diff --git a/modules/nf-core/htslib/bgziptabix/main.nf b/modules/nf-core/htslib/bgziptabix/main.nf new file mode 100644 index 00000000..573c05fc --- /dev/null +++ b/modules/nf-core/htslib/bgziptabix/main.nf @@ -0,0 +1,88 @@ +process HTSLIB_BGZIPTABIX { + tag "${meta.id}" + label 'process_low' + + conda "${moduleDir}/environment.yml" + container "${workflow.containerEngine in ['singularity', 'apptainer'] && !task.ext.singularity_pull_docker_container + ? 'https://community-cr-prod.seqera.io/docker/registry/v2/blobs/sha256/86/863ca0dbbba30c8367fa4fbd3fa3a84393532fb7b300a5c5c2e70f0dfc475bbf/data' + : 'community.wave.seqera.io/library/htslib_xz:32f2772a564b3cd2'}" + + input: + tuple val(meta), path(infile), path(infile_tbi), path(regions) + val action + val make_index + val out_ext + + output: + tuple val(meta), path("${outfile}"), emit: output + tuple val(meta), path("${outfile}.{tbi,csi}"), emit: index, optional: true + // all htslib tools have the same version, we use bgzip + tuple val("${task.process}"), val('htslib'), eval("bgzip --version | sed '1! d; s/bgzip (htslib) //'"), topic: versions, emit: versions_htslib + tuple val("${task.process}"), val('xz'), eval("xz --version | sed '1! d; s/xz (XZ Utils) //'"), topic: versions, emit: versions_xz + + when: + task.ext.when == null || task.ext.when + + script: + def allowed_actions = ["compress", "decompress"] + if (action !in allowed_actions) { + error("htslib/bgziptabix: Invalid action: ${action}. Allowed actions are: ${allowed_actions.join(', ')}") + } + + if (action == "decompress" && make_index) { + log.warn("htslib/bgziptabix: Cannot create index when decompressing. Ignoring make_index option.") + } + + def args = task.ext.args ?: '' + def args2 = task.ext.args2 ?: '' + prefix = task.ext.prefix ?: "${meta.id}" + outfile = action == "compress" ? (out_ext ? "${prefix}.${out_ext}.gz" : "${prefix}.gz") : (out_ext ? "${prefix}.${out_ext}" : "${prefix}") + + def compress_cmd = action == "compress" ? "bgzip -c ${args} -@ ${task.cpus}" : "cat" + def bgzip_cmd = action == "compress" ? "[ '\$(basename ${infile})' != '\$(basename ${outfile})' ] && ln -s ${infile} ${outfile}" : "bgzip -c -d ${args} -@ ${task.cpus} ${infile} > ${outfile}" + + def regions_arg = regions ? "-R ${regions}" : "" + def tabix_cmd = (make_index && !infile_tbi) ? "tabix -@ ${task.cpus} ${regions_arg} ${args2} -f ${outfile}" : "" + def link_tabix_cmd = make_index && infile_tbi ? "ln -s ${infile_tbi} ${outfile}.${infile_tbi.extension}" : "" + def uncompressed_cmd = action == "compress" ? "${compress_cmd} ${infile} > ${outfile}" : (infile.getName() == outfile ? "" : "ln -s ${infile} ${outfile}") + """ + ${link_tabix_cmd} + + FILE_TYPE=\$(htsfile ${infile}) + + case "\$FILE_TYPE" in + *BGZF-compressed*) + ${bgzip_cmd} ;; + *gzip-compressed*) + [ "\$(basename ${infile})" == "\$(basename ${outfile})" ] && echo "Input and output names cannot be the same" && exit 1 + bgzip -d -c -@ ${task.cpus} ${infile} | ${compress_cmd} > ${outfile} ;; + *bzip2-compressed*) + bzcat ${infile} | ${compress_cmd} > ${outfile} ;; + *XZ-compressed*) + xzcat ${infile} | ${compress_cmd} > ${outfile} ;; + *) + ${uncompressed_cmd} ;; + esac + + ${tabix_cmd} + """ + + stub: + def args = task.ext.args ?: '' + def args2 = task.ext.args2 ?: '' + prefix = task.ext.prefix ?: "${meta.id}" + outfile = action == "compress" ? (out_ext ? "${prefix}.${out_ext}.gz" : "${prefix}.gz") : (out_ext ? "${prefix}.${out_ext}" : "${prefix}") + + def touch_cmd = action == "compress" ? "echo | bgzip -c" : "echo" + def index_fmt = args2.contains('-C') ? 'csi' : 'tbi' + def tabix_cmd = make_index ? "touch ${outfile}.${index_fmt}" : "" + def link_tabix_cmd = make_index && infile_tbi ? "ln -s ${infile_tbi} ${outfile}.${infile_tbi.extension}" : "" + """ + echo ${args} + + ${touch_cmd} > ${outfile} + + ${tabix_cmd} + ${link_tabix_cmd} + """ +} diff --git a/modules/nf-core/htslib/bgziptabix/meta.yml b/modules/nf-core/htslib/bgziptabix/meta.yml new file mode 100644 index 00000000..4cdefd0e --- /dev/null +++ b/modules/nf-core/htslib/bgziptabix/meta.yml @@ -0,0 +1,125 @@ +name: "htslib_bgziptabix" +description: "Multi-purpose module to compress, decompress and index files using bgzip + and tabix." +keywords: + - compress + - decompress + - index + - bgzip + - tabix + - gzip + - bzip + - xz +tools: + - "htslib": + description: "C library for high-throughput sequencing data formats." + homepage: "http://www.htslib.org/" + documentation: "http://www.htslib.org/doc/" + tool_dev_url: "https://github.com/samtools/htslib" + doi: "10.1093/gigascience/giab007" + licence: + - "MIT" + identifier: biotools:htslib +input: + - - meta: + type: map + description: | + Groovy Map containing sample information + e.g. [ id:'sample1' ] + - infile: + type: file + description: Input file to compress or decompress + pattern: "*" + ontologies: [] + - infile_tbi: + type: file + description: Optional tabix index for the input file. + pattern: "*.{tbi,csi}" + ontologies: + - edam: http://edamontology.org/format_3616 # tabix + - regions: + type: file + description: Optional file of regions to extract (BED or chr:start-end format). + Only used when creating an index for the output file. + pattern: "*.{bed,txt,tsv}" + ontologies: + - edam: http://edamontology.org/format_3475 # TSV + - edam: http://edamontology.org/format_3003 # BED + - action: + type: string + description: Action to perform, either `compress` or `decompress` + - make_index: + type: boolean + description: Whether to create a tabix index for the output file; only used + if `action` is `compress` + - out_ext: + type: string + description: Output file extension without `.gz` suffix (for example `vcf`) +output: + output: + - - meta: + type: map + description: | + Groovy Map containing sample information + e.g. [ id:'sample1' ] + - ${outfile}: + type: file + description: Compressed or decompressed output file + pattern: "*" + ontologies: [] + index: + - - meta: + type: map + description: | + Groovy Map containing sample information + e.g. [ id:'sample1' ] + - ${outfile}.{tbi,csi}: + type: file + description: Tabix index file for the compressed output file + pattern: "*.{tbi,csi}" + ontologies: + - edam: http://edamontology.org/format_3616 # tabix + versions_htslib: + - - ${task.process}: + type: string + description: The name of the process + - htslib: + type: string + description: The name of the tool + - bgzip --version | sed '1! d; s/bgzip (htslib) //': + type: eval + description: The expression to obtain the version of the tool + versions_xz: + - - ${task.process}: + type: string + description: The name of the process + - xz: + type: string + description: The name of the tool + - xz --version | sed '1! d; s/xz (XZ Utils) //': + type: eval + description: The expression to obtain the version of the tool +topics: + versions: + - - ${task.process}: + type: string + description: The name of the process + - htslib: + type: string + description: The name of the tool + - bgzip --version | sed '1! d; s/bgzip (htslib) //': + type: eval + description: The expression to obtain the version of the tool + - - ${task.process}: + type: string + description: The name of the process + - xz: + type: string + description: The name of the tool + - xz --version | sed '1! d; s/xz (XZ Utils) //': + type: eval + description: The expression to obtain the version of the tool +authors: + - "@itrujnara" +maintainers: + - "@itrujnara" diff --git a/modules/nf-core/htslib/bgziptabix/tests/main.nf.test b/modules/nf-core/htslib/bgziptabix/tests/main.nf.test new file mode 100644 index 00000000..a7346506 --- /dev/null +++ b/modules/nf-core/htslib/bgziptabix/tests/main.nf.test @@ -0,0 +1,435 @@ +nextflow_process { + + name "Test Process HTSLIB_BGZIPTABIX" + script "../main.nf" + process "HTSLIB_BGZIPTABIX" + + tag "modules" + tag "modules_nfcore" + tag "htslib" + tag "htslib/bgziptabix" + + test("sarscov2 - vcf - decompress") { + + when { + process { + """ + input[0] = [ + [ id:'test' ], + file(params.modules_testdata_base_path + 'genomics/sarscov2/illumina/vcf/test.vcf', checkIfExists: true), + [], + [] + ] + input[1] = 'decompress' // action + input[2] = false // make_index + input[3] = 'vcf' // out_ext + """ + } + } + + then { + assert process.success + assertAll( + { assert snapshot(sanitizeOutput(process.out)).match() }, + { assert process.out.output.get(0).get(1).endsWith('.vcf') }, + { assert process.out.index.size() == 0 } + ) + } + + } + + test("sarscov2 - vcf - compress - index") { + + when { + process { + """ + input[0] = [ + [ id:'test' ], + file(params.modules_testdata_base_path + 'genomics/sarscov2/illumina/vcf/test.vcf', checkIfExists: true), + [], + [] + ] + input[1] = 'compress' // action + input[2] = true // make_index + input[3] = 'vcf' // out_ext + """ + } + } + + then { + assert process.success + assertAll( + { assert snapshot(sanitizeOutput(process.out)).match() }, + { assert process.out.output.get(0).get(1).endsWith('.vcf.gz') }, + { assert process.out.index.get(0).get(1).endsWith('.vcf.gz.tbi') } + ) + } + + } + + test("sarscov2 - vcf + regions - compress - index") { + when { + process { + """ + input[0] = [ + [ id:'example' ], + file(params.modules_testdata_base_path + 'genomics/sarscov2/illumina/vcf/test.vcf.gz', checkIfExists: true), + file(params.modules_testdata_base_path + 'genomics/sarscov2/illumina/vcf/test.vcf.gz.tbi', checkIfExists: true), + file('https://raw.githubusercontent.com/luisas/test-datasets/refs/heads/add-bedgraph-subset-illumina/data/genomics/sarscov2/illumina/bed/test.bed', checkIfExists: true) + ] + input[1] = 'compress' // action + input[2] = true // make_index + input[3] = 'vcf' // out_ext + """ + } + } + + then { + assertAll ( + { assert process.success }, + { assert snapshot( + sanitizeOutput(process.out), + path(process.out.output[0][1]).vcf.getVariantsMD5(), + ).match() } + ) + } + } + + test("sarscov2 - bgzip - decompress") { + + when { + process { + """ + input[0] = [ + [ id:'test' ], + file(params.modules_testdata_base_path + 'genomics/sarscov2/illumina/vcf/test.vcf.gz', checkIfExists: true), + [], + [] + ] + input[1] = 'decompress' // action + input[2] = false // make_index + input[3] = 'vcf' // out_ext + """ + } + } + + then { + assert process.success + assertAll( + { assert snapshot(sanitizeOutput(process.out)).match() }, + { assert process.out.output.get(0).get(1).endsWith('.vcf') }, + { assert process.out.index.size() == 0 } + ) + } + } + + test("sarscov2 - bgzip - compress - no index") { + + when { + process { + """ + input[0] = [ + [ id:'test' ], + file(params.modules_testdata_base_path + 'genomics/sarscov2/illumina/vcf/test.vcf.gz', checkIfExists: true), + [], + [] + ] + input[1] = 'compress' // action + input[2] = false // make_index + input[3] = 'vcf' // out_ext + """ + } + } + + then { + assert process.success + assertAll( + { assert snapshot(sanitizeOutput(process.out)).match() }, + { assert process.out.output.get(0).get(1).endsWith('.vcf.gz') }, + { assert process.out.index.size() == 0 } + ) + } + + } + + test("sarscov2 - gzip - decompress") { + + when { + process { + """ + input[0] = [ + [ id:'test' ], + file(params.modules_testdata_base_path + 'genomics/sarscov2/illumina/fastq/test_1.fastq.gz', checkIfExists: true), + [], + [] + ] + input[1] = 'decompress' // action + input[2] = false // make_index + input[3] = 'fastq' // out_ext + """ + } + } + + then { + assert process.success + assertAll( + { assert snapshot(sanitizeOutput(process.out)).match() }, + { assert process.out.output.get(0).get(1).endsWith('.fastq') }, + { assert process.out.index.size() == 0 } + ) + } + } + + test("sarscov2 - gzip - (re)compress - no index") { + + when { + process { + """ + input[0] = [ + [ id:'test' ], + file(params.modules_testdata_base_path + 'genomics/sarscov2/illumina/fastq/test_1.fastq.gz', checkIfExists: true), + [], + [] + ] + input[1] = 'compress' // action + input[2] = false // make_index + input[3] = 'fastq' // out_ext + """ + } + } + + then { + assert process.success + assertAll( + { assert snapshot(sanitizeOutput(process.out)).match() }, + { assert process.out.output.get(0).get(1).endsWith('.fastq.gz') }, + { assert process.out.index.size() == 0 } + ) + } + } + + test("sarscov2 - gzip - name clash") { + + when { + process { + """ + input[0] = [ + [ id:'test_1' ], + file(params.modules_testdata_base_path + 'genomics/sarscov2/illumina/fastq/test_1.fastq.gz', checkIfExists: true), + [], + [] + ] + input[1] = 'compress' // action + input[2] = false // make_index + input[3] = 'fastq' // out_ext + """ + } + } + + then { + assert process.failed + assertAll( + { assert process.errorReport.contains("Input and output names cannot be the same") } + ) + } + } + + test("metagenome - bz2 - decompress") { + + when { + process { + """ + input[0] = [ + [ id:'test' ], + file(params.modules_testdata_base_path + 'genomics/prokaryotes/metagenome/rgi/card-data.tar.bz2', checkIfExists: true), + [], + [] + ] + input[1] = 'decompress' // action + input[2] = false // make_index + input[3] = 'tar' // out_ext + """ + } + } + + then { + assert process.success + assertAll( + { assert snapshot(sanitizeOutput(process.out)).match() }, + { assert process.out.output.get(0).get(1).endsWith('.tar') }, + { assert process.out.index.size() == 0 } + ) + } + } + + test("metagenome - bz2 - (re)compress - no index") { + + when { + process { + """ + input[0] = [ + [ id:'test' ], + file(params.modules_testdata_base_path + 'genomics/prokaryotes/metagenome/rgi/card-data.tar.bz2', checkIfExists: true), + [], + [] + ] + input[1] = 'compress' // action + input[2] = false // make_index + input[3] = 'tar' // out_ext + """ + } + } + + then { + assert process.success + assertAll( + { assert snapshot(sanitizeOutput(process.out)).match() }, + { assert process.out.output.get(0).get(1).endsWith('.tar.gz') }, + { assert process.out.index.size() == 0 } + ) + } + } + + test("metagenome - xz - decompress") { + + when { + process { + """ + input[0] = [ + [ id:'test' ], + file(params.modules_testdata_base_path + 'genomics/prokaryotes/metagenome/taxonomy/misc/taxa_sqlite.xz', checkIfExists: true), + [], + [] + ] + input[1] = 'decompress' // action + input[2] = false // make_index + input[3] = '' // out_ext + """ + } + } + + then { + assert process.success + assertAll( + { assert snapshot( + process.out, + process.out.findAll { key, val -> key.startsWith('versions') } + ).match() }, + { assert process.out.output.get(0).get(1).endsWith('test') }, + { assert process.out.index.size() == 0 } + ) + } + } + + test("metagenome - xz - (re)compress - no index") { + + when { + process { + """ + input[0] = [ + [ id:'test' ], + file(params.modules_testdata_base_path + 'genomics/prokaryotes/metagenome/taxonomy/misc/taxa_sqlite.xz', checkIfExists: true), + [], + [] + ] + input[1] = 'compress' // action + input[2] = false // make_index + input[3] = '' // out_ext + """ + } + } + + then { + assert process.success + assertAll( + { assert snapshot(sanitizeOutput(process.out)).match() }, + { assert process.out.output.get(0).get(1).endsWith('.gz') }, + { assert process.out.index.size() == 0 } + ) + } + } + + test("sarscov2 - vcf - compress - index - stub") { + + options "-stub" + + when { + process { + """ + input[0] = [ + [ id:'test' ], + file(params.modules_testdata_base_path + 'genomics/sarscov2/illumina/vcf/test.vcf', checkIfExists: true), + [], + [] + ] + input[1] = 'compress' // action + input[2] = true // make_index + input[3] = 'vcf' // out_ext + """ + } + } + + then { + assert process.success + assertAll( + { assert snapshot(sanitizeOutput(process.out)).match() } + ) + } + + } + + test("sarscov2 - vcf - decompress - stub") { + + options "-stub" + + when { + process { + """ + input[0] = [ + [ id:'test' ], + file(params.modules_testdata_base_path + 'genomics/sarscov2/illumina/vcf/test.vcf.gz', checkIfExists: true), + [], + [] + ] + input[1] = 'decompress' // action + input[2] = false // make_index + input[3] = 'vcf' // out_ext + """ + } + } + + then { + assert process.success + assertAll( + { assert snapshot(sanitizeOutput(process.out)).match() } + ) + } + + } + + test("illegal action") { + + when { + process { + """ + input[0] = [ + [ id:'test' ], + file(params.modules_testdata_base_path + 'genomics/sarscov2/illumina/vcf/test.vcf', checkIfExists: true), + [], + [] + ] + input[1] = 'invalid_action' // action + input[2] = true // make_index + input[3] = 'vcf' // out_ext + """ + } + } + + then { + assert process.failed + assert process.errorReport.contains("Invalid action: invalid_action. Allowed actions are: compress, decompress") + } + + } + +} diff --git a/modules/nf-core/htslib/bgziptabix/tests/main.nf.test.snap b/modules/nf-core/htslib/bgziptabix/tests/main.nf.test.snap new file mode 100644 index 00000000..4d8b3d25 --- /dev/null +++ b/modules/nf-core/htslib/bgziptabix/tests/main.nf.test.snap @@ -0,0 +1,527 @@ +{ + "sarscov2 - gzip - (re)compress - no index": { + "content": [ + { + "index": [ + + ], + "output": [ + [ + { + "id": "test" + }, + "test.fastq.gz:md5,4161df271f9bfcd25d5845a1e220dbec" + ] + ], + "versions_htslib": [ + [ + "HTSLIB_BGZIPTABIX", + "htslib", + "1.24" + ] + ], + "versions_xz": [ + [ + "HTSLIB_BGZIPTABIX", + "xz", + "5.8.3" + ] + ] + } + ], + "timestamp": "2026-07-10T08:07:32.981182", + "meta": { + "nf-test": "0.9.5", + "nextflow": "26.04.6" + } + }, + "metagenome - xz - (re)compress - no index": { + "content": [ + { + "index": [ + + ], + "output": [ + [ + { + "id": "test" + }, + "test.gz:md5,b8d852a2b1ee52ed64d83046dcdb9de2" + ] + ], + "versions_htslib": [ + [ + "HTSLIB_BGZIPTABIX", + "htslib", + "1.24" + ] + ], + "versions_xz": [ + [ + "HTSLIB_BGZIPTABIX", + "xz", + "5.8.3" + ] + ] + } + ], + "timestamp": "2026-07-10T08:08:57.748138", + "meta": { + "nf-test": "0.9.5", + "nextflow": "26.04.6" + } + }, + "metagenome - bz2 - decompress": { + "content": [ + { + "index": [ + + ], + "output": [ + [ + { + "id": "test" + }, + "test.tar:md5,39e9e71fd16cfd09ceca12cd46e6abce" + ] + ], + "versions_htslib": [ + [ + "HTSLIB_BGZIPTABIX", + "htslib", + "1.24" + ] + ], + "versions_xz": [ + [ + "HTSLIB_BGZIPTABIX", + "xz", + "5.8.3" + ] + ] + } + ], + "timestamp": "2026-07-10T08:07:44.215383", + "meta": { + "nf-test": "0.9.5", + "nextflow": "26.04.6" + } + }, + "sarscov2 - vcf - decompress - stub": { + "content": [ + { + "index": [ + + ], + "output": [ + [ + { + "id": "test" + }, + "test.vcf:md5,68b329da9893e34099c7d8ad5cb9c940" + ] + ], + "versions_htslib": [ + [ + "HTSLIB_BGZIPTABIX", + "htslib", + "1.24" + ] + ], + "versions_xz": [ + [ + "HTSLIB_BGZIPTABIX", + "xz", + "5.8.3" + ] + ] + } + ], + "timestamp": "2026-07-10T08:10:11.941036", + "meta": { + "nf-test": "0.9.5", + "nextflow": "26.04.6" + } + }, + "sarscov2 - gzip - decompress": { + "content": [ + { + "index": [ + + ], + "output": [ + [ + { + "id": "test" + }, + "test.fastq:md5,4161df271f9bfcd25d5845a1e220dbec" + ] + ], + "versions_htslib": [ + [ + "HTSLIB_BGZIPTABIX", + "htslib", + "1.24" + ] + ], + "versions_xz": [ + [ + "HTSLIB_BGZIPTABIX", + "xz", + "5.8.3" + ] + ] + } + ], + "timestamp": "2026-07-10T08:07:28.301585", + "meta": { + "nf-test": "0.9.5", + "nextflow": "26.04.6" + } + }, + "sarscov2 - vcf - compress - index - stub": { + "content": [ + { + "index": [ + [ + { + "id": "test" + }, + "test.vcf.gz.tbi:md5,d41d8cd98f00b204e9800998ecf8427e" + ] + ], + "output": [ + [ + { + "id": "test" + }, + "test.vcf.gz:md5,68b329da9893e34099c7d8ad5cb9c940" + ] + ], + "versions_htslib": [ + [ + "HTSLIB_BGZIPTABIX", + "htslib", + "1.24" + ] + ], + "versions_xz": [ + [ + "HTSLIB_BGZIPTABIX", + "xz", + "5.8.3" + ] + ] + } + ], + "timestamp": "2026-07-10T08:09:24.961486", + "meta": { + "nf-test": "0.9.5", + "nextflow": "26.04.6" + } + }, + "sarscov2 - vcf - decompress": { + "content": [ + { + "index": [ + + ], + "output": [ + [ + { + "id": "test" + }, + "test.vcf:md5,8e722884ffb75155212a3fc053918766" + ] + ], + "versions_htslib": [ + [ + "HTSLIB_BGZIPTABIX", + "htslib", + "1.24" + ] + ], + "versions_xz": [ + [ + "HTSLIB_BGZIPTABIX", + "xz", + "5.8.3" + ] + ] + } + ], + "timestamp": "2026-07-10T08:07:05.41219", + "meta": { + "nf-test": "0.9.5", + "nextflow": "26.04.6" + } + }, + "metagenome - bz2 - (re)compress - no index": { + "content": [ + { + "index": [ + + ], + "output": [ + [ + { + "id": "test" + }, + "test.tar.gz:md5,39e9e71fd16cfd09ceca12cd46e6abce" + ] + ], + "versions_htslib": [ + [ + "HTSLIB_BGZIPTABIX", + "htslib", + "1.24" + ] + ], + "versions_xz": [ + [ + "HTSLIB_BGZIPTABIX", + "xz", + "5.8.3" + ] + ] + } + ], + "timestamp": "2026-07-10T08:07:51.772557", + "meta": { + "nf-test": "0.9.5", + "nextflow": "26.04.6" + } + }, + "sarscov2 - vcf - compress - index": { + "content": [ + { + "index": [ + [ + { + "id": "test" + }, + "test.vcf.gz.tbi:md5,7f005943c935f2b55ba3f9d4802aa09f" + ] + ], + "output": [ + [ + { + "id": "test" + }, + "test.vcf.gz:md5,8e722884ffb75155212a3fc053918766" + ] + ], + "versions_htslib": [ + [ + "HTSLIB_BGZIPTABIX", + "htslib", + "1.24" + ] + ], + "versions_xz": [ + [ + "HTSLIB_BGZIPTABIX", + "xz", + "5.8.3" + ] + ] + } + ], + "timestamp": "2026-07-10T08:07:09.68169", + "meta": { + "nf-test": "0.9.5", + "nextflow": "26.04.6" + } + }, + "metagenome - xz - decompress": { + "content": [ + { + "0": [ + [ + { + "id": "test" + }, + "test:md5,b8d852a2b1ee52ed64d83046dcdb9de2" + ] + ], + "1": [ + + ], + "2": [ + [ + "HTSLIB_BGZIPTABIX", + "htslib", + "1.24" + ] + ], + "3": [ + [ + "HTSLIB_BGZIPTABIX", + "xz", + "5.8.3" + ] + ], + "index": [ + + ], + "output": [ + [ + { + "id": "test" + }, + "test:md5,b8d852a2b1ee52ed64d83046dcdb9de2" + ] + ], + "versions_htslib": [ + [ + "HTSLIB_BGZIPTABIX", + "htslib", + "1.24" + ] + ], + "versions_xz": [ + [ + "HTSLIB_BGZIPTABIX", + "xz", + "5.8.3" + ] + ] + }, + { + "versions_htslib": [ + [ + "HTSLIB_BGZIPTABIX", + "htslib", + "1.24" + ] + ], + "versions_xz": [ + [ + "HTSLIB_BGZIPTABIX", + "xz", + "5.8.3" + ] + ] + } + ], + "timestamp": "2026-07-10T08:08:19.920765", + "meta": { + "nf-test": "0.9.5", + "nextflow": "26.04.6" + } + }, + "sarscov2 - bgzip - compress - no index": { + "content": [ + { + "index": [ + + ], + "output": [ + [ + { + "id": "test" + }, + "test.vcf.gz:md5,8e722884ffb75155212a3fc053918766" + ] + ], + "versions_htslib": [ + [ + "HTSLIB_BGZIPTABIX", + "htslib", + "1.24" + ] + ], + "versions_xz": [ + [ + "HTSLIB_BGZIPTABIX", + "xz", + "5.8.3" + ] + ] + } + ], + "timestamp": "2026-07-10T08:07:23.405306", + "meta": { + "nf-test": "0.9.5", + "nextflow": "26.04.6" + } + }, + "sarscov2 - vcf + regions - compress - index": { + "content": [ + { + "index": [ + [ + { + "id": "example" + }, + "example.vcf.gz.tbi:md5,d22e5b84e4fcd18792179f72e6da702e" + ] + ], + "output": [ + [ + { + "id": "example" + }, + "example.vcf.gz:md5,8e722884ffb75155212a3fc053918766" + ] + ], + "versions_htslib": [ + [ + "HTSLIB_BGZIPTABIX", + "htslib", + "1.24" + ] + ], + "versions_xz": [ + [ + "HTSLIB_BGZIPTABIX", + "xz", + "5.8.3" + ] + ] + }, + "bc7bf3ee9e8430e064c539eb81e59bf9" + ], + "timestamp": "2026-07-10T08:07:15.187648", + "meta": { + "nf-test": "0.9.5", + "nextflow": "26.04.6" + } + }, + "sarscov2 - bgzip - decompress": { + "content": [ + { + "index": [ + + ], + "output": [ + [ + { + "id": "test" + }, + "test.vcf:md5,8e722884ffb75155212a3fc053918766" + ] + ], + "versions_htslib": [ + [ + "HTSLIB_BGZIPTABIX", + "htslib", + "1.24" + ] + ], + "versions_xz": [ + [ + "HTSLIB_BGZIPTABIX", + "xz", + "5.8.3" + ] + ] + } + ], + "timestamp": "2026-07-10T08:07:19.313945", + "meta": { + "nf-test": "0.9.5", + "nextflow": "26.04.6" + } + } +} \ No newline at end of file diff --git a/subworkflows/local/dmr.nf b/subworkflows/local/dmr.nf new file mode 100644 index 00000000..5b0a08e6 --- /dev/null +++ b/subworkflows/local/dmr.nf @@ -0,0 +1,127 @@ +// IMPORT MODULES +include { HTSLIB_BGZIPTABIX } from '../../modules/nf-core/htslib/bgziptabix/main' +include { MODKIT_DMR } from '../../modules/nf-core/modkit/dmr/main' +include { DMR_HAPLOTYPE_REGIONS } from '../../modules/local/dmr/haplotype_regions/main' +include { DMR_NEAREST_GENE } from '../../modules/local/dmr/nearest_gene/main' + +workflow DMR { + + take: + modkit_bedgz // [meta, bed.gz(es)] -- MODKIT_PILEUP.out.bedgz; only meaningful when modkit_phased is true + fasta // [[:], fasta] + fai // [[:], fai] + cpg_islands_bed // path + gencode_gene_bed // path + + main: + ch_versions = channel.empty() + + // Pick the hp1/hp2 pair out of MODKIT_PILEUP's glob-collected output list (which also + // includes a _combined file when --modkit_phased is set). An unphased run's single + // unsuffixed file matches neither pattern, so it's dropped by the filter below -- + // defence in depth alongside the modkit_phased gate on the caller's side. + modkit_bedgz + .map { meta, files -> + def flist = files instanceof List ? files : [files] + def hp1 = flist.find { it.name.endsWith('_hp1.bed.gz') } + def hp2 = flist.find { it.name.endsWith('_hp2.bed.gz') } + return [meta, hp1, hp2] + } + .filter { _meta, hp1, hp2 -> hp1 && hp2 } + .set { haplotype_bedmethyl } + // haplotype_bedmethyl: [meta, hp1_bedgz, hp2_bedgz] + + // + // MODULE: HTSLIB_BGZIPTABIX (label: process_low) + // Tabix-index each haplotype's bedMethyl -- modkit dmr pair requires a .tbi alongside each + // bgzip input. The bedMethyl is already bgzip-compressed (modkit pileup --bgzf), so this + // only adds the index. hp1/hp2 are tagged onto meta and mixed into one call, then split + // back apart below. + // + haplotype_bedmethyl + .flatMap { meta, hp1, hp2 -> + return [ + [meta + [haplotype: 'hp1'], hp1, [], []], + [meta + [haplotype: 'hp2'], hp2, [], []] + ] + } + .set { bgziptabix_input } + // bgziptabix_input: [meta+haplotype, bedmethyl, [], []] + + HTSLIB_BGZIPTABIX ( + bgziptabix_input, + 'compress', + true, + 'bed' + ) + + HTSLIB_BGZIPTABIX.out.output + .join(HTSLIB_BGZIPTABIX.out.index) + .map { meta, bedgz, tbi -> + def haplotype = meta.haplotype + def sample_meta = meta.findAll { it.key != 'haplotype' } + return [sample_meta, haplotype, bedgz, tbi] + } + .branch { _meta, haplotype, _bedgz, _tbi -> + hp1: haplotype == 'hp1' + hp2: haplotype == 'hp2' + } + .set { indexed_branched } + + indexed_branched.hp1 + .map { meta, _haplotype, bedgz, tbi -> [meta, bedgz, tbi] } + .set { hp1_indexed } + indexed_branched.hp2 + .map { meta, _haplotype, bedgz, tbi -> [meta, bedgz, tbi] } + .set { hp2_indexed } + + hp1_indexed + .join(hp2_indexed) + .set { dmr_haplotype_input } + // dmr_haplotype_input: [meta, hp1_bedgz, hp1_tbi, hp2_bedgz, hp2_tbi] + + // + // MODULE: DMR_HAPLOTYPE_REGIONS (label: process_single) + // Restrict cpg_islands_bed to islands covered in both haplotypes. + // + DMR_HAPLOTYPE_REGIONS ( + dmr_haplotype_input, + cpg_islands_bed, + fai.first() + ) + ch_versions = ch_versions.mix(DMR_HAPLOTYPE_REGIONS.out.versions) + + // + // MODULE: MODKIT_DMR (label: process_medium) + // Compare methylation between the two haplotypes over the restricted regions. + // + dmr_haplotype_input + .join(DMR_HAPLOTYPE_REGIONS.out.regions_bed) + .multiMap { meta, hp1_bedgz, hp1_tbi, hp2_bedgz, hp2_tbi, regions_bed -> + hp1: [meta, hp1_bedgz, hp1_tbi] + hp2: [meta, hp2_bedgz, hp2_tbi] + regions: [meta, regions_bed] + } + .set { modkit_dmr_input } + + MODKIT_DMR ( + modkit_dmr_input.hp1, + modkit_dmr_input.hp2, + modkit_dmr_input.regions, + fasta.first() + ) + + // + // MODULE: DMR_NEAREST_GENE (label: process_single) + // Annotate each DMR with its nearest gene. + // + DMR_NEAREST_GENE ( + MODKIT_DMR.out.bed, + gencode_gene_bed + ) + ch_versions = ch_versions.mix(DMR_NEAREST_GENE.out.versions) + + emit: + nearest_gene_bed = DMR_NEAREST_GENE.out.nearest_gene_bed // [meta, bed] -- DMRs annotated with nearest gene + versions = ch_versions +} diff --git a/workflows/lrsomatic.nf b/workflows/lrsomatic.nf index 97de7479..b0ffc81a 100644 --- a/workflows/lrsomatic.nf +++ b/workflows/lrsomatic.nf @@ -60,6 +60,7 @@ include { PAIRED_SMALLVAR_GERMLINE } from '../subworkflows/local/paired/p include { PHASING_HAPLOTYPING } from '../subworkflows/local/phasing_haplotyping' include { TUMORONLY_SAVANA } from '../subworkflows/local/tumor_only/tumoronly_savana' include { PAIRED_SAVANA } from '../subworkflows/local/paired/paired_savana' +include { DMR } from '../subworkflows/local/dmr' @@ -778,6 +779,23 @@ workflow LRSOMATIC { : ch_index_minimap // ch_modkit_input: [meta, bam, bai] -- BAM to pile up; meta.type selects the publish directory MODKIT_PILEUP(ch_modkit_input, ch_fasta, ch_fai, [[:],[]]) + + // + // SUBWORKFLOW: DMR (label: process_medium) + // Differential methylation between a sample's two haplotypes. Only possible when + // --modkit_phased produced hp1/hp2 bedMethyl in the first place; --skip_dmr additionally + // turns it off on top of that. + // + if (params.modkit_phased && !params.skip_dmr) { + DMR ( + MODKIT_PILEUP.out.bedgz, + ch_fasta, + ch_fai, + file(params.dmr_cpg_islands_bed, checkIfExists: true), + file(params.dmr_gencode_gene_bed, checkIfExists: true) + ) + ch_versions = ch_versions.mix(DMR.out.versions) + } } // Prepare phased VCFs for VEP: add empty 'extra' list required by ENSEMBLVEP_VEP From ddcca2cc3f941a8fa90ca4484ba8183d0d1d597a Mon Sep 17 00:00:00 2001 From: Yann Vanrobaeys Date: Wed, 23 Sep 2026 19:10:31 -0400 Subject: [PATCH 04/11] Point igenomes.config TODOs at the now-open reference-fixtures PR Cosmetic only -- IntGenomicsLab/test-datasets#4 (add-dmr-reference-fixtures) is open but not yet merged, so the URLs themselves are unchanged. Co-Authored-By: Claude Sonnet 5 Claude-Session: https://claude.ai/code/session_01Wit78XR11V9AMAgY7gGrMT --- conf/igenomes.config | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/conf/igenomes.config b/conf/igenomes.config index d35e9966..47b6a1f5 100644 --- a/conf/igenomes.config +++ b/conf/igenomes.config @@ -24,8 +24,8 @@ params.genomes = [ vep_species : "homo_sapiens", savana_contigs : "https://raw.githubusercontent.com/cortes-ciriano-lab/savana/main/example/contigs.chr.hg38.txt", savana_g1000_vcf : "1000g_hg38", - // TODO: re-point at IntGenomicsLab/test-datasets/main/... once its add-dmr-reference-fixtures - // branch merges (currently on the YannVRB fork) + // TODO: re-point at IntGenomicsLab/test-datasets/main/... once + // IntGenomicsLab/test-datasets#4 merges (currently on the YannVRB fork) dmr_cpg_islands_bed : "https://raw.githubusercontent.com/YannVRB/test-datasets/add-dmr-reference-fixtures/references/dmr/GRCh38.cpg_islands.bed.gz", dmr_gencode_gene_bed : "https://raw.githubusercontent.com/YannVRB/test-datasets/add-dmr-reference-fixtures/references/dmr/GRCh38.gencode_gene.bed.gz", vep_alphamissense : "https://storage.googleapis.com/dm_alphamissense/AlphaMissense_hg38.tsv.gz", @@ -58,8 +58,8 @@ params.genomes = [ vep_species : "homo_sapiens_gca009914755v4", savana_contigs : "https://raw.githubusercontent.com/IntGenomicsLab/test-datasets/main/references/savana/contigs.chr.chm13.txt", savana_g1000_vcf : "1000g_t2t", - // TODO: re-point at IntGenomicsLab/test-datasets/main/... once its add-dmr-reference-fixtures - // branch merges (currently on the YannVRB fork) + // TODO: re-point at IntGenomicsLab/test-datasets/main/... once + // IntGenomicsLab/test-datasets#4 merges (currently on the YannVRB fork) dmr_cpg_islands_bed : "https://raw.githubusercontent.com/YannVRB/test-datasets/add-dmr-reference-fixtures/references/dmr/CHM13.cpg_islands.bed.gz", dmr_gencode_gene_bed : "https://raw.githubusercontent.com/YannVRB/test-datasets/add-dmr-reference-fixtures/references/dmr/CHM13.gencode_gene.bed.gz", vep_alphamissense_aa : "https://g-608c0c.273595.03c0.data.globus.org/VEP_plugins/alphamissense_protein_v2023_uniprot-2026_03.tsv.gz", From a39317d3512f974d61cb48214f340e0b3733c8b7 Mon Sep 17 00:00:00 2001 From: Yann Vanrobaeys Date: Thu, 24 Sep 2026 08:32:07 -0400 Subject: [PATCH 05/11] Re-point DMR reference URLs at the now-merged IntGenomicsLab/test-datasets IntGenomicsLab/test-datasets#4 merged -- dmr_cpg_islands_bed/dmr_gencode_gene_bed for both GRCh38 and CHM13 now point at IntGenomicsLab/test-datasets/main instead of the YannVRB fork branch. Verified all four URLs resolve (HTTP 200) before committing. Co-Authored-By: Claude Sonnet 5 Claude-Session: https://claude.ai/code/session_01Wit78XR11V9AMAgY7gGrMT --- conf/igenomes.config | 12 ++++-------- 1 file changed, 4 insertions(+), 8 deletions(-) diff --git a/conf/igenomes.config b/conf/igenomes.config index 47b6a1f5..db1f1391 100644 --- a/conf/igenomes.config +++ b/conf/igenomes.config @@ -24,10 +24,8 @@ params.genomes = [ vep_species : "homo_sapiens", savana_contigs : "https://raw.githubusercontent.com/cortes-ciriano-lab/savana/main/example/contigs.chr.hg38.txt", savana_g1000_vcf : "1000g_hg38", - // TODO: re-point at IntGenomicsLab/test-datasets/main/... once - // IntGenomicsLab/test-datasets#4 merges (currently on the YannVRB fork) - dmr_cpg_islands_bed : "https://raw.githubusercontent.com/YannVRB/test-datasets/add-dmr-reference-fixtures/references/dmr/GRCh38.cpg_islands.bed.gz", - dmr_gencode_gene_bed : "https://raw.githubusercontent.com/YannVRB/test-datasets/add-dmr-reference-fixtures/references/dmr/GRCh38.gencode_gene.bed.gz", + dmr_cpg_islands_bed : "https://raw.githubusercontent.com/IntGenomicsLab/test-datasets/main/references/dmr/GRCh38.cpg_islands.bed.gz", + dmr_gencode_gene_bed : "https://raw.githubusercontent.com/IntGenomicsLab/test-datasets/main/references/dmr/GRCh38.gencode_gene.bed.gz", vep_alphamissense : "https://storage.googleapis.com/dm_alphamissense/AlphaMissense_hg38.tsv.gz", vep_alphamissense_tbi : "https://g-608c0c.273595.03c0.data.globus.org/VEP_plugins/AlphaMissense_hg38.tsv.gz.tbi", // A dated release rather than the rolling vcf_GRCh38/clinvar.vcf.gz, whose VCF and @@ -58,10 +56,8 @@ params.genomes = [ vep_species : "homo_sapiens_gca009914755v4", savana_contigs : "https://raw.githubusercontent.com/IntGenomicsLab/test-datasets/main/references/savana/contigs.chr.chm13.txt", savana_g1000_vcf : "1000g_t2t", - // TODO: re-point at IntGenomicsLab/test-datasets/main/... once - // IntGenomicsLab/test-datasets#4 merges (currently on the YannVRB fork) - dmr_cpg_islands_bed : "https://raw.githubusercontent.com/YannVRB/test-datasets/add-dmr-reference-fixtures/references/dmr/CHM13.cpg_islands.bed.gz", - dmr_gencode_gene_bed : "https://raw.githubusercontent.com/YannVRB/test-datasets/add-dmr-reference-fixtures/references/dmr/CHM13.gencode_gene.bed.gz", + dmr_cpg_islands_bed : "https://raw.githubusercontent.com/IntGenomicsLab/test-datasets/main/references/dmr/CHM13.cpg_islands.bed.gz", + dmr_gencode_gene_bed : "https://raw.githubusercontent.com/IntGenomicsLab/test-datasets/main/references/dmr/CHM13.gencode_gene.bed.gz", vep_alphamissense_aa : "https://g-608c0c.273595.03c0.data.globus.org/VEP_plugins/alphamissense_protein_v2023_uniprot-2026_03.tsv.gz", vep_alphamissense_aa_tbi : "https://g-608c0c.273595.03c0.data.globus.org/VEP_plugins/alphamissense_protein_v2023_uniprot-2026_03.tsv.gz.tbi", // Pinned to a release rather than current_variation/, which moves at every Ensembl release From dd729e5a42b9c73b11fd1b3fe6a2a2bf7b1a5720 Mon Sep 17 00:00:00 2001 From: Yann Vanrobaeys Date: Mon, 28 Sep 2026 11:43:56 -0400 Subject: [PATCH 06/11] Re-vendor modkit/dmr from current upstream head; drop unused bedtools modules Per ljwharbers's review of PR #204: Re-vendored modkit/dmr from nf-core/modules#13021's current head (d886465c) -- the previously-vendored test still pointed at test_hp1.bed.gz/test_hp2.bed.gz, which 404s now that the upstream PR moved to generating its bedmethyl fixture in-test instead. This is why CI never actually exercised the new code. git_sha in modules.json updated to match. bedtools/closest and bedtools/intersect were vendored but never included anywhere -- the local DMR modules shell out to the bedtools CLI directly instead of delegating to these processes. Removed both, plus their modules.json entries. Co-Authored-By: Claude Sonnet 5 Claude-Session: https://claude.ai/code/session_01Wit78XR11V9AMAgY7gGrMT --- modules.json | 12 +- .../nf-core/bedtools/closest/environment.yml | 7 - modules/nf-core/bedtools/closest/main.nf | 53 ---- modules/nf-core/bedtools/closest/meta.yml | 74 ------ .../bedtools/closest/tests/main.nf.test | 157 ----------- .../bedtools/closest/tests/main.nf.test.snap | 248 ------------------ .../bedtools/intersect/environment.yml | 7 - modules/nf-core/bedtools/intersect/main.nf | 49 ---- modules/nf-core/bedtools/intersect/meta.yml | 83 ------ .../bedtools/intersect/tests/main.nf.test | 90 ------- .../intersect/tests/main.nf.test.snap | 125 --------- .../bedtools/intersect/tests/nextflow.config | 5 - modules/nf-core/modkit/dmr/meta.yml | 23 +- modules/nf-core/modkit/dmr/tests/main.nf.test | 113 +++++--- .../modkit/dmr/tests/main.nf.test.snap | 14 +- .../nf-core/modkit/dmr/tests/nextflow.config | 9 + 16 files changed, 116 insertions(+), 953 deletions(-) delete mode 100644 modules/nf-core/bedtools/closest/environment.yml delete mode 100644 modules/nf-core/bedtools/closest/main.nf delete mode 100644 modules/nf-core/bedtools/closest/meta.yml delete mode 100644 modules/nf-core/bedtools/closest/tests/main.nf.test delete mode 100644 modules/nf-core/bedtools/closest/tests/main.nf.test.snap delete mode 100644 modules/nf-core/bedtools/intersect/environment.yml delete mode 100644 modules/nf-core/bedtools/intersect/main.nf delete mode 100644 modules/nf-core/bedtools/intersect/meta.yml delete mode 100644 modules/nf-core/bedtools/intersect/tests/main.nf.test delete mode 100644 modules/nf-core/bedtools/intersect/tests/main.nf.test.snap delete mode 100644 modules/nf-core/bedtools/intersect/tests/nextflow.config diff --git a/modules.json b/modules.json index d83f529f..a48fbf74 100644 --- a/modules.json +++ b/modules.json @@ -55,16 +55,6 @@ "git_sha": "feef37435aea56816adf4b3bde1fc76aac327a8d", "installed_by": ["modules"] }, - "bedtools/closest": { - "branch": "master", - "git_sha": "cbe6025183336010a9b716e463f897b0c0f70f8a", - "installed_by": ["modules"] - }, - "bedtools/intersect": { - "branch": "master", - "git_sha": "cbe6025183336010a9b716e463f897b0c0f70f8a", - "installed_by": ["modules"] - }, "deepvariant/callvariants": { "branch": "master", "git_sha": "f2b138ee1d91f67d31c187317d7e83e429bf0309", @@ -270,7 +260,7 @@ "nf-core": { "modkit/dmr": { "branch": "add-modkit-dmr", - "git_sha": "9d93dfaa3dbe224588d2f7c10dce082528854dec", + "git_sha": "d886465c00aa9f724c92fa89b30d84dd64c61ea6", "installed_by": ["modules"] } } diff --git a/modules/nf-core/bedtools/closest/environment.yml b/modules/nf-core/bedtools/closest/environment.yml deleted file mode 100644 index 45c307b0..00000000 --- a/modules/nf-core/bedtools/closest/environment.yml +++ /dev/null @@ -1,7 +0,0 @@ ---- -# yaml-language-server: $schema=https://raw.githubusercontent.com/nf-core/modules/master/modules/environment-schema.json -channels: - - conda-forge - - bioconda -dependencies: - - bioconda::bedtools=2.31.1 diff --git a/modules/nf-core/bedtools/closest/main.nf b/modules/nf-core/bedtools/closest/main.nf deleted file mode 100644 index 5f055bdc..00000000 --- a/modules/nf-core/bedtools/closest/main.nf +++ /dev/null @@ -1,53 +0,0 @@ -process BEDTOOLS_CLOSEST { - tag "${meta.id}" - label 'process_single' - - conda "${moduleDir}/environment.yml" - container "${workflow.containerEngine in ['singularity', 'apptainer'] && !task.ext.singularity_pull_docker_container - ? 'https://depot.galaxyproject.org/singularity/bedtools:2.31.1--hf5e1c6e_0' - : 'quay.io/biocontainers/bedtools:2.31.1--hf5e1c6e_0'}" - - input: - tuple val(meta), path(input_1), path(input_2) - path fasta_fai - - output: - tuple val(meta), path("*.${extension}"), emit: output - tuple val("${task.process}"), val('bedtools'), eval("bedtools --version | sed -e 's/bedtools v//g'"), topic: versions, emit: versions_bedtools - - when: - task.ext.when == null || task.ext.when - - script: - def args = task.ext.args ?: '' - def prefix = task.ext.prefix ?: "${meta.id}" - - extension = input_1.extension == "gz" - ? (input_1 =~ /^.*\.(.*)\.gz$/)[0][1] - : input_1.extension - - def reference = fasta_fai ? "-g ${fasta_fai}" : "" - - if (input_1 == "${prefix}.${extension}" || input_2 == "${prefix}.${extension}") { - error("One of the input files is called the same as the output file. Please specify another prefix.") - } - - """ - bedtools closest \\ - ${args} \\ - -a ${input_1} \\ - -b ${input_2} \\ - ${reference} \\ - > ${prefix}.${extension} - """ - - stub: - def prefix = task.ext.prefix ?: "${meta.id}" - - extension = input_1.extension == "gz" - ? (input_1 =~ /^.*\.(.*)\.gz$/)[0][1] - : input_1.extension - """ - touch ${prefix}.${extension} - """ -} diff --git a/modules/nf-core/bedtools/closest/meta.yml b/modules/nf-core/bedtools/closest/meta.yml deleted file mode 100644 index 5e30eeb0..00000000 --- a/modules/nf-core/bedtools/closest/meta.yml +++ /dev/null @@ -1,74 +0,0 @@ -name: "bedtools_closest" -description: For each feature in A, finds the closest feature (upstream or downstream) - in B. -keywords: - - bedtools - - closest - - bed - - vcf - - gff -tools: - - bedtools: - description: | - A set of tools for genomic analysis tasks, specifically enabling genome arithmetic (merge, count, complement) on various file types. - documentation: https://bedtools.readthedocs.io/en/latest/content/tools/closest.html - licence: ["MIT"] - identifier: biotools:bedtools -input: - - - meta: - type: map - description: | - Groovy Map containing sample information - e.g. [ id:'test', single_end:false ] - - input_1: - type: file - description: The file to find the closest features of - pattern: "*.{bed,vcf,gff}(.gz)?" - ontologies: [] - - input_2: - type: list - description: The input file(s) to find the closest features from - pattern: "*.{bed,vcf,gff}(.gz)?" - - fasta_fai: - type: file - description: The index of the FASTA reference. Needed when the argument `--sorted` - is used - pattern: "*.fai" - ontologies: [] -output: - output: - - - meta: - type: map - description: | - Groovy Map containing sample information - e.g. [ id:'test', single_end:false ] - - "*.${extension}": - type: file - description: The resulting BED file containing the closest features - pattern: "*.{bed,vcf,gff}" - ontologies: [] - versions_bedtools: - - - ${task.process}: - type: string - description: The name of the process - - bedtools: - type: string - description: The name of the tool - - "bedtools --version | sed -e 's/bedtools v//g'": - type: eval - description: The expression to obtain the version of the tool -topics: - versions: - - - ${task.process}: - type: string - description: The name of the process - - bedtools: - type: string - description: The name of the tool - - "bedtools --version | sed -e 's/bedtools v//g'": - type: eval - description: The expression to obtain the version of the tool -authors: - - "@nvnieuwk" -maintainers: - - "@nvnieuwk" diff --git a/modules/nf-core/bedtools/closest/tests/main.nf.test b/modules/nf-core/bedtools/closest/tests/main.nf.test deleted file mode 100644 index 349903d9..00000000 --- a/modules/nf-core/bedtools/closest/tests/main.nf.test +++ /dev/null @@ -1,157 +0,0 @@ -nextflow_process { - name "Test Process BEDTOOLS_CLOSEST" - script "../main.nf" - process "BEDTOOLS_CLOSEST" - - tag "modules" - tag "modules_nfcore" - tag "bedtools" - tag "bedtools/closest" - - test("homo_sapiens") { - when { - process { - """ - input[0] = [ - [ id:'test' ], // meta map - file(params.modules_testdata_base_path + 'genomics/homo_sapiens/genome/genome.bed', checkIfExists: true), - [ - file(params.modules_testdata_base_path + 'genomics/homo_sapiens/genome/genome.multi_intervals.bed', checkIfExists: true), - file(params.modules_testdata_base_path + 'genomics/homo_sapiens/genome/chr21/sequence/multi_intervals.bed', checkIfExists: true) - ] - ] - input[1] = [] - """ - } - } - - then { - assertAll( - { assert process.success }, - { assert snapshot(process.out).match() } - ) - } - } - - test("homo_sapiens - fai") { - when { - process { - """ - input[0] = [ - [ id:'test' ], // meta map - file(params.modules_testdata_base_path + 'genomics/homo_sapiens/genome/genome.bed', checkIfExists: true), - file(params.modules_testdata_base_path + 'genomics/homo_sapiens/genome/genome.multi_intervals.bed', checkIfExists: true) - ] - input[1] = file(params.modules_testdata_base_path + 'genomics/homo_sapiens/genome/genome.fasta.fai', checkIfExists: true) - """ - } - } - - then { - assertAll( - { assert process.success }, - { assert snapshot(process.out).match() } - ) - } - } - - test("homo_sapiens - vcf") { - when { - process { - """ - input[0] = [ - [ id:'test' ], // meta map - file(params.modules_testdata_base_path + 'genomics/homo_sapiens/genome/genome.bed', checkIfExists: true), - [ - file(params.modules_testdata_base_path + 'genomics/homo_sapiens/genome/vcf/dbsnp_146.hg38.vcf.gz', checkIfExists: true), - file(params.modules_testdata_base_path + 'genomics/homo_sapiens/genome/vcf/gnomAD.r2.1.1.vcf.gz', checkIfExists: true) - ] - ] - input[1] = [] - """ - } - } - - then { - assertAll( - { assert process.success }, - { assert snapshot(process.out).match() } - ) - } - } - - test("homo_sapiens - stub") { - options "-stub" - when { - process { - """ - input[0] = [ - [ id:'test' ], // meta map - file(params.modules_testdata_base_path + 'genomics/homo_sapiens/genome/genome.bed', checkIfExists: true), - [ - file(params.modules_testdata_base_path + 'genomics/homo_sapiens/genome/genome.multi_intervals.bed', checkIfExists: true), - file(params.modules_testdata_base_path + 'genomics/homo_sapiens/genome/chr21/sequence/multi_intervals.bed', checkIfExists: true) - ] - ] - input[1] = [] - """ - } - } - - then { - assertAll( - { assert process.success }, - { assert snapshot(process.out).match() } - ) - } - } - - test("homo_sapiens - fai - stub") { - options "-stub" - when { - process { - """ - input[0] = [ - [ id:'test' ], // meta map - file(params.modules_testdata_base_path + 'genomics/homo_sapiens/genome/genome.bed', checkIfExists: true), - file(params.modules_testdata_base_path + 'genomics/homo_sapiens/genome/genome.multi_intervals.bed', checkIfExists: true) - ] - input[1] = file(params.modules_testdata_base_path + 'genomics/homo_sapiens/genome/genome.fasta.fai', checkIfExists: true) - """ - } - } - - then { - assertAll( - { assert process.success }, - { assert snapshot(process.out).match() } - ) - } - } - - test("homo_sapiens - vcf - stub") { - options "-stub" - when { - process { - """ - input[0] = [ - [ id:'test' ], // meta map - file(params.modules_testdata_base_path + 'genomics/homo_sapiens/genome/genome.bed', checkIfExists: true), - [ - file(params.modules_testdata_base_path + 'genomics/homo_sapiens/genome/vcf/dbsnp_146.hg38.vcf.gz', checkIfExists: true), - file(params.modules_testdata_base_path + 'genomics/homo_sapiens/genome/vcf/gnomAD.r2.1.1.vcf.gz', checkIfExists: true) - ] - ] - input[1] = [] - """ - } - } - - then { - assertAll( - { assert process.success }, - { assert snapshot(process.out).match() } - ) - } - } -} diff --git a/modules/nf-core/bedtools/closest/tests/main.nf.test.snap b/modules/nf-core/bedtools/closest/tests/main.nf.test.snap deleted file mode 100644 index 99895426..00000000 --- a/modules/nf-core/bedtools/closest/tests/main.nf.test.snap +++ /dev/null @@ -1,248 +0,0 @@ -{ - "homo_sapiens - vcf": { - "content": [ - { - "0": [ - [ - { - "id": "test" - }, - "test.bed:md5,85a5cee50322ab15f384df7826164d64" - ] - ], - "1": [ - [ - "BEDTOOLS_CLOSEST", - "bedtools", - "2.31.1" - ] - ], - "output": [ - [ - { - "id": "test" - }, - "test.bed:md5,85a5cee50322ab15f384df7826164d64" - ] - ], - "versions_bedtools": [ - [ - "BEDTOOLS_CLOSEST", - "bedtools", - "2.31.1" - ] - ] - } - ], - "meta": { - "nf-test": "0.9.3", - "nextflow": "25.10.2" - }, - "timestamp": "2026-01-21T11:10:06.402036653" - }, - "homo_sapiens - fai - stub": { - "content": [ - { - "0": [ - [ - { - "id": "test" - }, - "test.bed:md5,d41d8cd98f00b204e9800998ecf8427e" - ] - ], - "1": [ - [ - "BEDTOOLS_CLOSEST", - "bedtools", - "2.31.1" - ] - ], - "output": [ - [ - { - "id": "test" - }, - "test.bed:md5,d41d8cd98f00b204e9800998ecf8427e" - ] - ], - "versions_bedtools": [ - [ - "BEDTOOLS_CLOSEST", - "bedtools", - "2.31.1" - ] - ] - } - ], - "meta": { - "nf-test": "0.9.3", - "nextflow": "25.10.2" - }, - "timestamp": "2026-01-21T11:10:22.176626879" - }, - "homo_sapiens - stub": { - "content": [ - { - "0": [ - [ - { - "id": "test" - }, - "test.bed:md5,d41d8cd98f00b204e9800998ecf8427e" - ] - ], - "1": [ - [ - "BEDTOOLS_CLOSEST", - "bedtools", - "2.31.1" - ] - ], - "output": [ - [ - { - "id": "test" - }, - "test.bed:md5,d41d8cd98f00b204e9800998ecf8427e" - ] - ], - "versions_bedtools": [ - [ - "BEDTOOLS_CLOSEST", - "bedtools", - "2.31.1" - ] - ] - } - ], - "meta": { - "nf-test": "0.9.3", - "nextflow": "25.10.2" - }, - "timestamp": "2026-01-21T11:10:14.477862854" - }, - "homo_sapiens - fai": { - "content": [ - { - "0": [ - [ - { - "id": "test" - }, - "test.bed:md5,347f192c141a34dfc4bb7c13fffd44bb" - ] - ], - "1": [ - [ - "BEDTOOLS_CLOSEST", - "bedtools", - "2.31.1" - ] - ], - "output": [ - [ - { - "id": "test" - }, - "test.bed:md5,347f192c141a34dfc4bb7c13fffd44bb" - ] - ], - "versions_bedtools": [ - [ - "BEDTOOLS_CLOSEST", - "bedtools", - "2.31.1" - ] - ] - } - ], - "meta": { - "nf-test": "0.9.3", - "nextflow": "25.10.2" - }, - "timestamp": "2026-01-21T11:09:58.376835909" - }, - "homo_sapiens - vcf - stub": { - "content": [ - { - "0": [ - [ - { - "id": "test" - }, - "test.bed:md5,d41d8cd98f00b204e9800998ecf8427e" - ] - ], - "1": [ - [ - "BEDTOOLS_CLOSEST", - "bedtools", - "2.31.1" - ] - ], - "output": [ - [ - { - "id": "test" - }, - "test.bed:md5,d41d8cd98f00b204e9800998ecf8427e" - ] - ], - "versions_bedtools": [ - [ - "BEDTOOLS_CLOSEST", - "bedtools", - "2.31.1" - ] - ] - } - ], - "meta": { - "nf-test": "0.9.3", - "nextflow": "25.10.2" - }, - "timestamp": "2026-01-21T11:10:30.128998192" - }, - "homo_sapiens": { - "content": [ - { - "0": [ - [ - { - "id": "test" - }, - "test.bed:md5,de47864c0ae53891d09252eb63cd7269" - ] - ], - "1": [ - [ - "BEDTOOLS_CLOSEST", - "bedtools", - "2.31.1" - ] - ], - "output": [ - [ - { - "id": "test" - }, - "test.bed:md5,de47864c0ae53891d09252eb63cd7269" - ] - ], - "versions_bedtools": [ - [ - "BEDTOOLS_CLOSEST", - "bedtools", - "2.31.1" - ] - ] - } - ], - "meta": { - "nf-test": "0.9.3", - "nextflow": "25.10.2" - }, - "timestamp": "2026-01-21T11:09:50.231505255" - } -} \ No newline at end of file diff --git a/modules/nf-core/bedtools/intersect/environment.yml b/modules/nf-core/bedtools/intersect/environment.yml deleted file mode 100644 index 45c307b0..00000000 --- a/modules/nf-core/bedtools/intersect/environment.yml +++ /dev/null @@ -1,7 +0,0 @@ ---- -# yaml-language-server: $schema=https://raw.githubusercontent.com/nf-core/modules/master/modules/environment-schema.json -channels: - - conda-forge - - bioconda -dependencies: - - bioconda::bedtools=2.31.1 diff --git a/modules/nf-core/bedtools/intersect/main.nf b/modules/nf-core/bedtools/intersect/main.nf deleted file mode 100644 index fbcbf530..00000000 --- a/modules/nf-core/bedtools/intersect/main.nf +++ /dev/null @@ -1,49 +0,0 @@ -process BEDTOOLS_INTERSECT { - tag "${meta.id}" - label 'process_single' - - conda "${moduleDir}/environment.yml" - container "${workflow.containerEngine in ['singularity', 'apptainer'] && !task.ext.singularity_pull_docker_container - ? 'https://depot.galaxyproject.org/singularity/bedtools:2.31.1--hf5e1c6e_0' - : 'quay.io/biocontainers/bedtools:2.31.1--hf5e1c6e_0'}" - - input: - tuple val(meta), path(intervals1), path(intervals2) - tuple val(meta2), path(chrom_sizes) - - output: - tuple val(meta), path("*.${extension}"), emit: intersect - tuple val("${task.process}"), val('bedtools'), eval("bedtools --version | sed -e 's/bedtools v//g'"), topic: versions, emit: versions_bedtools - - when: - task.ext.when == null || task.ext.when - - script: - def args = task.ext.args ?: '' - def prefix = task.ext.prefix ?: "${meta.id}" - //Extension of the output file. It is set by the user via "ext.suffix" in the config. Corresponds to the file format which depends on arguments (e. g., ".bed", ".bam", ".txt", etc.). - extension = task.ext.suffix ?: "${intervals1.extension}" - def sizes = chrom_sizes ? "-g ${chrom_sizes}" : '' - if ("${intervals1}" == "${prefix}.${extension}" || "${intervals2}" == "${prefix}.${extension}") { - error("Input and output names are the same, use \"task.ext.prefix\" to disambiguate!") - } - """ - bedtools \\ - intersect \\ - -a ${intervals1} \\ - -b ${intervals2} \\ - ${args} \\ - ${sizes} \\ - > ${prefix}.${extension} - """ - - stub: - def prefix = task.ext.prefix ?: "${meta.id}" - extension = task.ext.suffix ?: "bed" - if ("${intervals1}" == "${prefix}.${extension}" || "${intervals2}" == "${prefix}.${extension}") { - error("Input and output names are the same, use \"task.ext.prefix\" to disambiguate!") - } - """ - touch ${prefix}.${extension} - """ -} diff --git a/modules/nf-core/bedtools/intersect/meta.yml b/modules/nf-core/bedtools/intersect/meta.yml deleted file mode 100644 index 0500efb9..00000000 --- a/modules/nf-core/bedtools/intersect/meta.yml +++ /dev/null @@ -1,83 +0,0 @@ -name: bedtools_intersect -description: Allows one to screen for overlaps between two sets of genomic features. -keywords: - - bed - - intersect - - overlap -tools: - - bedtools: - description: | - A set of tools for genomic analysis tasks, specifically enabling genome arithmetic (merge, count, complement) on various file types. - documentation: https://bedtools.readthedocs.io/en/latest/content/tools/intersect.html - licence: ["MIT"] - identifier: biotools:bedtools -input: - - - meta: - type: map - description: | - Groovy Map containing sample information - e.g. [ id:'test', single_end:false ] - - intervals1: - type: file - description: BAM/BED/GFF/VCF - pattern: "*.{bam|bed|gff|vcf}" - ontologies: [] - - intervals2: - type: file - description: BAM/BED/GFF/VCF - pattern: "*.{bam|bed|gff|vcf}" - ontologies: [] - - - meta2: - type: map - description: | - Groovy Map containing reference chromosome sizes - e.g. [ id:'test' ] - - chrom_sizes: - type: file - description: Chromosome sizes file - pattern: "*{.sizes,.txt}" - ontologies: [] -output: - intersect: - - - meta: - type: map - description: | - Groovy Map containing sample information - e.g. [ id:'test', single_end:false ] - - "*.${extension}": - type: file - description: File containing the description of overlaps found between the - two features - pattern: "*.${extension}" - ontologies: [] - versions_bedtools: - - - ${task.process}: - type: string - description: The name of the process - - bedtools: - type: string - description: The name of the tool - - "bedtools --version | sed -e 's/bedtools v//g'": - type: eval - description: The expression to obtain the version of the tool -topics: - versions: - - - ${task.process}: - type: string - description: The name of the process - - bedtools: - type: string - description: The name of the tool - - "bedtools --version | sed -e 's/bedtools v//g'": - type: eval - description: The expression to obtain the version of the tool -authors: - - "@edmundmiller" - - "@sruthipsuresh" - - "@drpatelh" - - "@sidorov-si" -maintainers: - - "@edmundmiller" - - "@sruthipsuresh" - - "@drpatelh" - - "@sidorov-si" diff --git a/modules/nf-core/bedtools/intersect/tests/main.nf.test b/modules/nf-core/bedtools/intersect/tests/main.nf.test deleted file mode 100644 index cd770946..00000000 --- a/modules/nf-core/bedtools/intersect/tests/main.nf.test +++ /dev/null @@ -1,90 +0,0 @@ -nextflow_process { - - name "Test Process BEDTOOLS_INTERSECT" - script "../main.nf" - process "BEDTOOLS_INTERSECT" - config "./nextflow.config" - - tag "modules" - tag "modules_nfcore" - tag "bedtools" - tag "bedtools/intersect" - - test("sarscov2 - bed - bed") { - - when { - process { - """ - input[0] = [ - [ id:'test' ], - file(params.modules_testdata_base_path + 'genomics/sarscov2/genome/bed/test.bed', checkIfExists: true), - file(params.modules_testdata_base_path + 'genomics/sarscov2/genome/bed/test2.bed', checkIfExists: true) - ] - - input[1] = [[:], []] - """ - } - } - - then { - assertAll( - { assert process.success }, - { assert snapshot(process.out).match() } - ) - } - - } - - test("sarscov2 - bam - bam") { - - when { - process { - """ - input[0] = [ - [ id:'test' ], - file(params.modules_testdata_base_path + 'genomics/sarscov2/illumina/bam/test.paired_end.bam', checkIfExists: true), - file(params.modules_testdata_base_path + 'genomics/sarscov2/genome/bed/baits.bed', checkIfExists: true) - ] - - input[1] = [[:], []] - """ - } - } - - then { - assertAll( - { assert process.success }, - { assert snapshot(process.out).match() } - ) - } - - } - - test("sarscov2 - bed - stub") { - - options "-stub" - - when { - process { - """ - input[0] = [ - [ id:'test' ], - file(params.modules_testdata_base_path + 'genomics/sarscov2/genome/bed/test.bed', checkIfExists: true), - file(params.modules_testdata_base_path + 'genomics/sarscov2/genome/bed/test2.bed', checkIfExists: true) - ] - - input[1] = [[:], []] - """ - } - } - - then { - assertAll( - { assert process.success }, - { assert snapshot(process.out).match() } - ) - } - - } - -} diff --git a/modules/nf-core/bedtools/intersect/tests/main.nf.test.snap b/modules/nf-core/bedtools/intersect/tests/main.nf.test.snap deleted file mode 100644 index 30da8be1..00000000 --- a/modules/nf-core/bedtools/intersect/tests/main.nf.test.snap +++ /dev/null @@ -1,125 +0,0 @@ -{ - "sarscov2 - bam - bam": { - "content": [ - { - "0": [ - [ - { - "id": "test" - }, - "test_out.bam:md5,738324efe2b1e442ceb6539a630c3fe6" - ] - ], - "1": [ - [ - "BEDTOOLS_INTERSECT", - "bedtools", - "2.31.1" - ] - ], - "intersect": [ - [ - { - "id": "test" - }, - "test_out.bam:md5,738324efe2b1e442ceb6539a630c3fe6" - ] - ], - "versions_bedtools": [ - [ - "BEDTOOLS_INTERSECT", - "bedtools", - "2.31.1" - ] - ] - } - ], - "meta": { - "nf-test": "0.9.3", - "nextflow": "25.10.2" - }, - "timestamp": "2026-01-21T11:31:18.811473651" - }, - "sarscov2 - bed - bed": { - "content": [ - { - "0": [ - [ - { - "id": "test" - }, - "test_out.bed:md5,afcbf01c2f2013aad71dbe8e34f2c15c" - ] - ], - "1": [ - [ - "BEDTOOLS_INTERSECT", - "bedtools", - "2.31.1" - ] - ], - "intersect": [ - [ - { - "id": "test" - }, - "test_out.bed:md5,afcbf01c2f2013aad71dbe8e34f2c15c" - ] - ], - "versions_bedtools": [ - [ - "BEDTOOLS_INTERSECT", - "bedtools", - "2.31.1" - ] - ] - } - ], - "meta": { - "nf-test": "0.9.3", - "nextflow": "25.10.2" - }, - "timestamp": "2026-01-21T11:31:11.122621263" - }, - "sarscov2 - bed - stub": { - "content": [ - { - "0": [ - [ - { - "id": "test" - }, - "test_out.bed:md5,d41d8cd98f00b204e9800998ecf8427e" - ] - ], - "1": [ - [ - "BEDTOOLS_INTERSECT", - "bedtools", - "2.31.1" - ] - ], - "intersect": [ - [ - { - "id": "test" - }, - "test_out.bed:md5,d41d8cd98f00b204e9800998ecf8427e" - ] - ], - "versions_bedtools": [ - [ - "BEDTOOLS_INTERSECT", - "bedtools", - "2.31.1" - ] - ] - } - ], - "meta": { - "nf-test": "0.9.3", - "nextflow": "25.10.2" - }, - "timestamp": "2026-01-21T11:31:27.957038806" - } -} \ No newline at end of file diff --git a/modules/nf-core/bedtools/intersect/tests/nextflow.config b/modules/nf-core/bedtools/intersect/tests/nextflow.config deleted file mode 100644 index f1f9e693..00000000 --- a/modules/nf-core/bedtools/intersect/tests/nextflow.config +++ /dev/null @@ -1,5 +0,0 @@ -process { - withName: BEDTOOLS_INTERSECT { - ext.prefix = { "${meta.id}_out" } - } -} diff --git a/modules/nf-core/modkit/dmr/meta.yml b/modules/nf-core/modkit/dmr/meta.yml index a20cfda7..b59ff450 100644 --- a/modules/nf-core/modkit/dmr/meta.yml +++ b/modules/nf-core/modkit/dmr/meta.yml @@ -26,13 +26,16 @@ input: description: Bgzipped bedMethyl file for the first (control) sample, as produced by modkit pileup with --bgzf pattern: "*.bed.gz" - ontologies: [] + ontologies: + - edam: "http://edamontology.org/format_3003" # BED + - edam: "http://edamontology.org/format_3989" # GZIP format - bedmethyl_a_tbi: type: file description: Tabix index for bedmethyl_a. Must have the same basename with a .tbi extension next to bedmethyl_a pattern: "*.bed.gz.tbi" - ontologies: [] + ontologies: + - edam: "http://edamontology.org/format_3700" # Tabix index format - - meta2: type: map description: | @@ -43,13 +46,16 @@ input: description: Bgzipped bedMethyl file for the second (experimental) sample, as produced by modkit pileup with --bgzf pattern: "*.bed.gz" - ontologies: [] + ontologies: + - edam: "http://edamontology.org/format_3003" # BED + - edam: "http://edamontology.org/format_3989" # GZIP format - bedmethyl_b_tbi: type: file description: Tabix index for bedmethyl_b. Must have the same basename with a .tbi extension next to bedmethyl_b pattern: "*.bed.gz.tbi" - ontologies: [] + ontologies: + - edam: "http://edamontology.org/format_3700" # Tabix index format - - meta3: type: map description: | @@ -60,7 +66,8 @@ input: description: Optional BED file of regions over which to compare methylation levels. When omitted, methylation levels are compared at each site instead pattern: "*.bed" - ontologies: [] + ontologies: + - edam: "http://edamontology.org/format_3003" # BED - - meta4: type: map description: | @@ -70,7 +77,8 @@ input: type: file description: Reference sequence in FASTA format, used during pileup/alignment pattern: "*.{fa,fasta}" - ontologies: [] + ontologies: + - edam: "http://edamontology.org/format_1929" # FASTA output: bed: - - meta: @@ -83,7 +91,8 @@ output: description: BED file with the score column indicating the magnitude of the difference in methylation between the two samples pattern: "*.bed" - ontologies: [] + ontologies: + - edam: "http://edamontology.org/format_3003" # BED log: - - meta: type: map diff --git a/modules/nf-core/modkit/dmr/tests/main.nf.test b/modules/nf-core/modkit/dmr/tests/main.nf.test index e6d77652..d1525874 100644 --- a/modules/nf-core/modkit/dmr/tests/main.nf.test +++ b/modules/nf-core/modkit/dmr/tests/main.nf.test @@ -6,9 +6,68 @@ nextflow_process { tag "modules_nfcore" tag "modkit" tag "modkit/dmr" + tag "modkit/pileup" + tag "htslib/bgziptabix" process "MODKIT_DMR" config "./nextflow.config" + setup { + run("MODKIT_PILEUP") { + script "../../pileup/main.nf" + process { + """ + input[0] = [ + [ id: 'test' ], + file(params.modules_testdata_base_path + 'genomics/homo_sapiens/nanopore/bam/test.sorted.phased.bam', checkIfExists: true), + file(params.modules_testdata_base_path + 'genomics/homo_sapiens/nanopore/bam/test.sorted.phased.bam.bai', checkIfExists: true) + ] + input[1] = [ + [ id: 'test_ref' ], + file(params.modules_testdata_base_path + 'genomics/homo_sapiens/genome/genome.fasta', checkIfExists: true), + file(params.modules_testdata_base_path + 'genomics/homo_sapiens/genome/genome.fasta.fai', checkIfExists: true) + ] + input[2] = [[],[]] + """ + } + } + + run("HTSLIB_BGZIPTABIX") { + script "../../../htslib/bgziptabix/main.nf" + process { + """ + // MODKIT_PILEUP --phased emits one [meta, [combined, hp1, hp2]] tuple per sample + // ("bedgz_files" is a single Path rather than a List when only one file matches + // the module's output glob -- true for -stub, since modkit pileup's stub script + // never splits by haplotype -- so normalize via [x].flatten() first). Split into + // one [meta, bedgz] tuple per haplotype so each can be tabix-indexed (modkit + // pileup --bgzf already bgzips, but never indexes) and fed to MODKIT_DMR as its + // own bedmethyl_a/bedmethyl_b input, exactly like real --modkit_phased usage. + input[0] = MODKIT_PILEUP.out.bedgz + .flatMap { meta, bedgz_files -> + def files = [ bedgz_files ].flatten() + def phased = files.findAll { f -> f.name =~ /_hp[12]\\.bed\\.gz\$/ } + if (phased) { + phased.collect { f -> + def haplotype = (f.name =~ /_(hp[12])\\.bed\\.gz\$/)[0][1] + [ meta + [ id: "\${meta.id}_\${haplotype}" ], f, [], [] ] + } + } else { + // -stub: modkit pileup's stub always emits one generic, unsplit file + // (it never reads --phased in stub mode) -- reuse it for both + // haplotypes, since MODKIT_DMR's own stub never reads its inputs. + files.collectMany { f -> + ['hp1', 'hp2'].collect { haplotype -> [ meta + [ id: "\${meta.id}_\${haplotype}" ], f, [], [] ] } + } + } + } + input[1] = "compress" + input[2] = true + input[3] = [] + """ + } + } + } + test("[bedmethyl_a, tbi], [bedmethyl_b, tbi], regions_bed, fasta") { when { @@ -17,16 +76,14 @@ nextflow_process { } process { """ - input[0] = [ - [ id: 'test' ], - file(params.modules_testdata_base_path + 'genomics/homo_sapiens/nanopore/bedmethyl/test_hp1.bed.gz', checkIfExists: true), - file(params.modules_testdata_base_path + 'genomics/homo_sapiens/nanopore/bedmethyl/test_hp1.bed.gz.tbi', checkIfExists: true) - ] - input[1] = [ - [ id: 'test' ], - file(params.modules_testdata_base_path + 'genomics/homo_sapiens/nanopore/bedmethyl/test_hp2.bed.gz', checkIfExists: true), - file(params.modules_testdata_base_path + 'genomics/homo_sapiens/nanopore/bedmethyl/test_hp2.bed.gz.tbi', checkIfExists: true) - ] + input[0] = HTSLIB_BGZIPTABIX.out.output + .join(HTSLIB_BGZIPTABIX.out.index) + .filter { meta, bedgz, tbi -> meta.id.endsWith('_hp1') } + .map { meta, bedgz, tbi -> [ [ id: 'test' ], bedgz, tbi ] } + input[1] = HTSLIB_BGZIPTABIX.out.output + .join(HTSLIB_BGZIPTABIX.out.index) + .filter { meta, bedgz, tbi -> meta.id.endsWith('_hp2') } + .map { meta, bedgz, tbi -> [ [ id: 'test' ], bedgz, tbi ] } input[2] = Channel.of('chr22\t0\t40001') .collectFile(name: 'chr22.bed', newLine: true) .map { file -> [ [ id:'chr22' ], file ] } @@ -55,16 +112,14 @@ nextflow_process { } process { """ - input[0] = [ - [ id: 'test' ], - file(params.modules_testdata_base_path + 'genomics/homo_sapiens/nanopore/bedmethyl/test_hp1.bed.gz', checkIfExists: true), - file(params.modules_testdata_base_path + 'genomics/homo_sapiens/nanopore/bedmethyl/test_hp1.bed.gz.tbi', checkIfExists: true) - ] - input[1] = [ - [ id: 'test' ], - file(params.modules_testdata_base_path + 'genomics/homo_sapiens/nanopore/bedmethyl/test_hp2.bed.gz', checkIfExists: true), - file(params.modules_testdata_base_path + 'genomics/homo_sapiens/nanopore/bedmethyl/test_hp2.bed.gz.tbi', checkIfExists: true) - ] + input[0] = HTSLIB_BGZIPTABIX.out.output + .join(HTSLIB_BGZIPTABIX.out.index) + .filter { meta, bedgz, tbi -> meta.id.endsWith('_hp1') } + .map { meta, bedgz, tbi -> [ [ id: 'test' ], bedgz, tbi ] } + input[1] = HTSLIB_BGZIPTABIX.out.output + .join(HTSLIB_BGZIPTABIX.out.index) + .filter { meta, bedgz, tbi -> meta.id.endsWith('_hp2') } + .map { meta, bedgz, tbi -> [ [ id: 'test' ], bedgz, tbi ] } input[2] = [ [ id:'regions' ], [] ] input[3] = [ [ id: 'test_ref' ], @@ -93,16 +148,14 @@ nextflow_process { } process { """ - input[0] = [ - [ id: 'test' ], - file(params.modules_testdata_base_path + 'genomics/homo_sapiens/nanopore/bedmethyl/test_hp1.bed.gz', checkIfExists: true), - file(params.modules_testdata_base_path + 'genomics/homo_sapiens/nanopore/bedmethyl/test_hp1.bed.gz.tbi', checkIfExists: true) - ] - input[1] = [ - [ id: 'test' ], - file(params.modules_testdata_base_path + 'genomics/homo_sapiens/nanopore/bedmethyl/test_hp2.bed.gz', checkIfExists: true), - file(params.modules_testdata_base_path + 'genomics/homo_sapiens/nanopore/bedmethyl/test_hp2.bed.gz.tbi', checkIfExists: true) - ] + input[0] = HTSLIB_BGZIPTABIX.out.output + .join(HTSLIB_BGZIPTABIX.out.index) + .filter { meta, bedgz, tbi -> meta.id.endsWith('_hp1') } + .map { meta, bedgz, tbi -> [ [ id: 'test' ], bedgz, tbi ] } + input[1] = HTSLIB_BGZIPTABIX.out.output + .join(HTSLIB_BGZIPTABIX.out.index) + .filter { meta, bedgz, tbi -> meta.id.endsWith('_hp2') } + .map { meta, bedgz, tbi -> [ [ id: 'test' ], bedgz, tbi ] } input[2] = Channel.of('chr22\t0\t40001') .collectFile(name: 'chr22.bed', newLine: true) .map { file -> [ [ id:'chr22' ], file ] } diff --git a/modules/nf-core/modkit/dmr/tests/main.nf.test.snap b/modules/nf-core/modkit/dmr/tests/main.nf.test.snap index 7b82e676..b8ff0798 100644 --- a/modules/nf-core/modkit/dmr/tests/main.nf.test.snap +++ b/modules/nf-core/modkit/dmr/tests/main.nf.test.snap @@ -15,7 +15,7 @@ { "id": "test" }, - "test.log:md5,ee178bddc200e65da14440f5710f762e" + "test.log:md5,c30c1b33203d139e5648d847cac4ab05" ] ], "2": [ @@ -38,7 +38,7 @@ { "id": "test" }, - "test.log:md5,ee178bddc200e65da14440f5710f762e" + "test.log:md5,c30c1b33203d139e5648d847cac4ab05" ] ], "versions_modkit": [ @@ -54,7 +54,7 @@ "nf-test": "0.9.0", "nextflow": "25.10.2" }, - "timestamp": "2026-09-23T11:41:19.857952367" + "timestamp": "2026-09-24T09:10:47.411811907" }, "[bedmethyl_a, tbi], [bedmethyl_b, tbi], [], fasta": { "content": [ @@ -72,7 +72,7 @@ { "id": "test" }, - "test.log:md5,8861572cf65176e7d07256479c169269" + "test.log:md5,256a0a5feab178024236ffc0b3a18c38" ] ], "2": [ @@ -95,7 +95,7 @@ { "id": "test" }, - "test.log:md5,8861572cf65176e7d07256479c169269" + "test.log:md5,256a0a5feab178024236ffc0b3a18c38" ] ], "versions_modkit": [ @@ -111,7 +111,7 @@ "nf-test": "0.9.0", "nextflow": "25.10.2" }, - "timestamp": "2026-09-23T11:41:25.583014363" + "timestamp": "2026-09-24T09:10:55.676443925" }, "[bedmethyl_a, tbi], [bedmethyl_b, tbi], regions_bed, fasta - stub": { "content": [ @@ -168,6 +168,6 @@ "nf-test": "0.9.0", "nextflow": "25.10.2" }, - "timestamp": "2026-09-23T11:41:31.474420326" + "timestamp": "2026-09-24T09:11:03.403287933" } } \ No newline at end of file diff --git a/modules/nf-core/modkit/dmr/tests/nextflow.config b/modules/nf-core/modkit/dmr/tests/nextflow.config index 3ed1ba5e..a6b69568 100644 --- a/modules/nf-core/modkit/dmr/tests/nextflow.config +++ b/modules/nf-core/modkit/dmr/tests/nextflow.config @@ -1,4 +1,13 @@ process { + withName: 'MODKIT_PILEUP' { + // --modified-bases 5mC 5hmC (matching modkit/pileup's own phased test): if this BAM's + // basecalling calls both, requesting only one leaves the other counted as "other" in + // modkit's bedMethyl output, which MODKIT_DMR's internal consistency check rejects. + ext.args = '--phased --modified-bases 5mC 5hmC' + } + withName: 'HTSLIB_BGZIPTABIX' { + ext.args2 = '-p bed' + } withName: 'MODKIT_DMR' { ext.args = params.module_args } From 0818fe318fe6ec683e305dea5e858b625ee86cd5 Mon Sep 17 00:00:00 2001 From: Yann Vanrobaeys Date: Mon, 28 Sep 2026 11:44:10 -0400 Subject: [PATCH 07/11] Declare dmr_cpg_islands_bed/dmr_gencode_gene_bed; skip DMR gracefully when unset Per ljwharbers's review: with --modkit_phased, any genome other than GRCh38/CHM13 hit a hard failure, because these two params only ever got a value via getGenomeAttribute() -- there was no schema entry, no nextflow.config default, and (the real bug) no way for an explicit --dmr_cpg_islands_bed/--dmr_gencode_gene_bed to survive: getGenomeAttribute() unconditionally overwrote whatever the user passed. - nextflow.config: both default to null. - nextflow_schema.json: both declared under reference_genome_options as optional file-path params. - workflows/lrsomatic.nf: params.dmr_cpg_islands_bed ?: getGenomeAttribute(...) instead of a bare overwrite, matching the 'explicit wins' pattern already used for VEP plugin resources in this same file. The DMR invocation block now checks both are present before running DMR; when they're not (no built-in default and no override), it logs a warning and skips DMR instead of crashing the run. Co-Authored-By: Claude Sonnet 5 Claude-Session: https://claude.ai/code/session_01Wit78XR11V9AMAgY7gGrMT --- nextflow.config | 7 ++++++- nextflow_schema.json | 14 ++++++++++++++ workflows/lrsomatic.nf | 42 ++++++++++++++++++++++++++++++------------ 3 files changed, 50 insertions(+), 13 deletions(-) diff --git a/nextflow.config b/nextflow.config index a4db7971..66dc85aa 100644 --- a/nextflow.config +++ b/nextflow.config @@ -29,8 +29,13 @@ params { modkit_args = '--cpg --modified-bases 5mC' modkit_phased = false - // DMR options -- only takes effect when modkit_phased is also true + // DMR options -- only take effect when modkit_phased is also true. dmr_cpg_islands_bed/ + // dmr_gencode_gene_bed default to null here; --genome GRCh38/CHM13 fill them in from + // igenomes.config unless explicitly overridden, and any other genome must supply both + // explicitly or DMR is skipped with a warning (see workflows/lrsomatic.nf). skip_dmr = false + dmr_cpg_islands_bed = null + dmr_gencode_gene_bed = null // PON Options clairsto_pon_vcfs = null diff --git a/nextflow_schema.json b/nextflow_schema.json index cdce36e3..44e04a3e 100644 --- a/nextflow_schema.json +++ b/nextflow_schema.json @@ -170,6 +170,20 @@ "description": "Verdict CNA resource directory for ClairS-TO, holding loci_files/, allele_files/ and one GC_*.txt. Read only with --skip_ascat.", "help_text": "Read only on `--skip_ascat` runs; otherwise Verdict's germline tagging comes from the pipeline's ASCAT run and this is ignored. The image ships the GRCh38 set and `--genome CHM13 --skip_ascat` builds a CHM13 one, so this is only needed for another assembly or as an override. The directory must be self-contained, with no links reaching outside it. Add an `RT_.txt` to correct LogR for replication timing as well; without it, correction is GC-only.", "fa_icon": "fas fa-folder-open" + }, + "dmr_cpg_islands_bed": { + "type": "string", + "format": "file-path", + "description": "CpG-islands BED to restrict haplotype DMR calling to. Has a built-in default for --genome GRCh38/CHM13; required as an explicit override for any other genome when --modkit_phased is set and --skip_dmr is not.", + "help_text": "Only used by the DMR subworkflow (see --skip_dmr). Without a value -- built-in or supplied -- DMR is skipped with a warning rather than the run failing.", + "fa_icon": "fas fa-map-marker-alt" + }, + "dmr_gencode_gene_bed": { + "type": "string", + "format": "file-path", + "description": "Gene-model BED (used for nearest-gene annotation of DMR calls) to use for haplotype DMR calling. Has a built-in default for --genome GRCh38/CHM13; required as an explicit override for any other genome when --modkit_phased is set and --skip_dmr is not.", + "help_text": "Only used by the DMR subworkflow (see --skip_dmr). Without a value -- built-in or supplied -- DMR is skipped with a warning rather than the run failing.", + "fa_icon": "fas fa-dna" } } }, diff --git a/workflows/lrsomatic.nf b/workflows/lrsomatic.nf index b0ffc81a..b668e4cf 100644 --- a/workflows/lrsomatic.nf +++ b/workflows/lrsomatic.nf @@ -112,8 +112,13 @@ workflow LRSOMATIC { params.bed_file = getGenomeAttribute('bed_file') params.savana_contigs = getGenomeAttribute('savana_contigs') params.savana_g1000_vcf = getGenomeAttribute('savana_g1000_vcf') - params.dmr_cpg_islands_bed = getGenomeAttribute('dmr_cpg_islands_bed') - params.dmr_gencode_gene_bed = getGenomeAttribute('dmr_gencode_gene_bed') + // An explicit --dmr_cpg_islands_bed/--dmr_gencode_gene_bed wins over the per-genome + // default (getGenomeAttribute alone would silently clobber it, same as every other + // getGenomeAttribute-assigned param above -- these two are the only ones a user is + // expected to override directly, since GRCh38/CHM13 are the only genomes with a built-in + // default and any other genome needs its own values). + params.dmr_cpg_islands_bed = params.dmr_cpg_islands_bed ?: getGenomeAttribute('dmr_cpg_islands_bed') + params.dmr_gencode_gene_bed = params.dmr_gencode_gene_bed ?: getGenomeAttribute('dmr_gencode_gene_bed') params.vep_genome = getGenomeAttribute('vep_genome') params.vep_species = getGenomeAttribute('vep_species') params.sigprofiler_genome = getGenomeAttribute('sigprofiler_genome') @@ -784,17 +789,30 @@ workflow LRSOMATIC { // SUBWORKFLOW: DMR (label: process_medium) // Differential methylation between a sample's two haplotypes. Only possible when // --modkit_phased produced hp1/hp2 bedMethyl in the first place; --skip_dmr additionally - // turns it off on top of that. - // + // turns it off on top of that. dmr_cpg_islands_bed/dmr_gencode_gene_bed have a built-in + // default only for --genome GRCh38/CHM13 (see getGenomeAttribute() above) -- any other + // genome needs both supplied explicitly, and DMR is skipped with a warning rather than + // the run failing when they are not (e.g. a custom --fasta with no --genome, or a + // --genome this pipeline doesn't ship DMR references for). if (params.modkit_phased && !params.skip_dmr) { - DMR ( - MODKIT_PILEUP.out.bedgz, - ch_fasta, - ch_fai, - file(params.dmr_cpg_islands_bed, checkIfExists: true), - file(params.dmr_gencode_gene_bed, checkIfExists: true) - ) - ch_versions = ch_versions.mix(DMR.out.versions) + if (params.dmr_cpg_islands_bed && params.dmr_gencode_gene_bed) { + DMR ( + MODKIT_PILEUP.out.bedgz, + ch_fasta, + ch_fai, + file(params.dmr_cpg_islands_bed, checkIfExists: true), + file(params.dmr_gencode_gene_bed, checkIfExists: true) + ) + ch_versions = ch_versions.mix(DMR.out.versions) + } + else { + log.warn( + "--modkit_phased is set but DMR is being skipped: --dmr_cpg_islands_bed/" + + "--dmr_gencode_gene_bed have no default for --genome '${params.genome}' " + + "(only GRCh38/CHM13 ship one). Pass both explicitly to enable DMR calling, " + + "or set --skip_dmr to silence this warning." + ) + } } } From 28a6e88ca26b7407be7edd4ede5ac226bdc9986c Mon Sep 17 00:00:00 2001 From: Yann Vanrobaeys Date: Mon, 28 Sep 2026 11:44:27 -0400 Subject: [PATCH 08/11] DMR_HAPLOTYPE_REGIONS: query via tabix -R, add a minimum-coverage threshold Per ljwharbers's review: on a whole-genome run, decompressing and sort -u'ing every bedMethyl row genome-wide (tens of millions of rows per haplotype) to find CpG-island overlaps is memory- and IO-heavy. Both hp1/hp2_bedmethyl are already bgzip+tabix indexed by the time this module sees them (HTSLIB_BGZIPTABIX runs just before it) -- query positions inside the CpG islands directly via tabix -R instead, which only reads the rows that can possibly matter. Also addresses the reviewer's second question on this module: 'covered' previously meant any bedMethyl row at all, regardless of depth. Added a minimum valid-coverage threshold (bedMethyl column 10, Nvalid_cov), configurable via ext.args and defaulting to 1 (unchanged behaviour) -- the actual right default for production use is a domain call for the maintainers, not something to silently pick here. Container/environment.yml updated: plain bedtools:2.31.1 has no tabix binary. Reused nf-core/modules' pints/caller container (real, in production use, confirmed via its own environment.yml to bundle both bedtools and htslib) rather than hand-construct an unverifiable Wave/mulled tag. Verified via the new subworkflows/local/tests/dmr.nf.test: real tabix -R output, versions.yml correctly reports both bedtools 2.31.1 and tabix 1.21. Co-Authored-By: Claude Sonnet 5 Claude-Session: https://claude.ai/code/session_01Wit78XR11V9AMAgY7gGrMT --- .../dmr/haplotype_regions/environment.yml | 1 + modules/local/dmr/haplotype_regions/main.nf | 27 ++++++++++++++++--- 2 files changed, 24 insertions(+), 4 deletions(-) diff --git a/modules/local/dmr/haplotype_regions/environment.yml b/modules/local/dmr/haplotype_regions/environment.yml index 45c307b0..8b02e7a2 100644 --- a/modules/local/dmr/haplotype_regions/environment.yml +++ b/modules/local/dmr/haplotype_regions/environment.yml @@ -5,3 +5,4 @@ channels: - bioconda dependencies: - bioconda::bedtools=2.31.1 + - bioconda::htslib=1.22.1 diff --git a/modules/local/dmr/haplotype_regions/main.nf b/modules/local/dmr/haplotype_regions/main.nf index cc08f172..c149081d 100644 --- a/modules/local/dmr/haplotype_regions/main.nf +++ b/modules/local/dmr/haplotype_regions/main.nf @@ -3,9 +3,15 @@ process DMR_HAPLOTYPE_REGIONS { label 'process_single' conda "${moduleDir}/environment.yml" + // The plain bedtools:2.31.1 biocontainer has no tabix binary at all -- needed since + // haplotype_regions/main.nf queries the bgzip+tabix-indexed bedMethyl inputs directly. + // Reusing nf-core/modules' pints/caller container here (real, already in production use, + // confirmed via its own environment.yml to bundle both bedtools and htslib) rather than + // hand-constructing an unverifiable Wave/mulled tag; it carries unrelated pybedtools/pypints + // baggage this module doesn't use, but that's preferable to guessing a container hash. container "${workflow.containerEngine in ['singularity', 'apptainer'] && !task.ext.singularity_pull_docker_container - ? 'https://depot.galaxyproject.org/singularity/bedtools:2.31.1--hf5e1c6e_0' - : 'quay.io/biocontainers/bedtools:2.31.1--hf5e1c6e_0'}" + ? 'https://community-cr-prod.seqera.io/docker/registry/v2/blobs/sha256/f1/f1a9e30012e1b41baf9acd1ff94e01161138d8aa17f4e97aa32f2dc4effafcd1/data' + : 'community.wave.seqera.io/library/pybedtools_bedtools_htslib_pip_pypints:39699b96998ec5f6'}" input: // hp1/hp2 bedMethyl are not read for methylation values here, only for which CpG positions @@ -25,11 +31,23 @@ process DMR_HAPLOTYPE_REGIONS { script: def prefix = task.ext.prefix ?: "${meta.id}" + // Minimum valid coverage (bedMethyl column 10, Nvalid_cov) for a position to count as + // "covered" -- 1 preserves the previous any-depth behaviour; override via ext.args to + // require more. + def min_valid_coverage = task.ext.args ?: 1 """ cut -f1,2 ${fai} > genome.txt - zcat ${hp1_bedmethyl} | awk 'BEGIN{OFS="\\t"}{print \$1,\$2,\$3}' | sort -k1,1 -k2,2n -u > hp1.cpg.bed - zcat ${hp2_bedmethyl} | awk 'BEGIN{OFS="\\t"}{print \$1,\$2,\$3}' | sort -k1,1 -k2,2n -u > hp2.cpg.bed + # Query directly via the bgzip+tabix index for positions inside the CpG islands, instead + # of decompressing/sorting every bedMethyl row genome-wide -- hp*_bedmethyl are already + # coordinate-sorted and tabix-indexed upstream, so this only reads the (small) subset of + # rows that can possibly matter here. + tabix -R ${cpg_islands_bed} ${hp1_bedmethyl} \\ + | awk -v min_cov=${min_valid_coverage} 'BEGIN{OFS="\\t"} \$10>=min_cov {print \$1,\$2,\$3}' \\ + | sort -k1,1 -k2,2n -u > hp1.cpg.bed + tabix -R ${cpg_islands_bed} ${hp2_bedmethyl} \\ + | awk -v min_cov=${min_valid_coverage} 'BEGIN{OFS="\\t"} \$10>=min_cov {print \$1,\$2,\$3}' \\ + | sort -k1,1 -k2,2n -u > hp2.cpg.bed bedtools intersect -u -a ${cpg_islands_bed} -b hp1.cpg.bed | sort -k1,1 -k2,2n > islands.hp1.bed bedtools intersect -u -a ${cpg_islands_bed} -b hp2.cpg.bed | sort -k1,1 -k2,2n > islands.hp2.bed @@ -38,6 +56,7 @@ process DMR_HAPLOTYPE_REGIONS { cat <<-END_VERSIONS > versions.yml "${task.process}": bedtools: \$(bedtools --version | sed 's/bedtools v//g') + tabix: \$(tabix --version 2>&1 | head -n1 | sed 's/tabix (htslib) //') END_VERSIONS """ From 41446be475547d7dfb4ecbbf66f0d5643306447f Mon Sep 17 00:00:00 2001 From: Yann Vanrobaeys Date: Mon, 28 Sep 2026 11:44:37 -0400 Subject: [PATCH 09/11] DMR_NEAREST_GENE: use bedtools closest -D b, not -D a Per ljwharbers's review: -D a reports signed distance relative to the input (DMR regions, unstranded, always '.') rather than the reference (genes, which do have a meaningful strand) -- -D b is the flag that actually makes the sign meaningful here. Co-Authored-By: Claude Sonnet 5 Claude-Session: https://claude.ai/code/session_01Wit78XR11V9AMAgY7gGrMT --- modules/local/dmr/nearest_gene/main.nf | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/modules/local/dmr/nearest_gene/main.nf b/modules/local/dmr/nearest_gene/main.nf index 8840a02d..f2f4350b 100644 --- a/modules/local/dmr/nearest_gene/main.nf +++ b/modules/local/dmr/nearest_gene/main.nf @@ -19,7 +19,9 @@ process DMR_NEAREST_GENE { task.ext.when == null || task.ext.when script: - def args = task.ext.args ?: '-D a -t first' + // -D b: distance is signed relative to the gene's (B's) strand, which is meaningful -- + // DMR regions (A) are unstranded, so -D a's "distance relative to A's strand" is not. + def args = task.ext.args ?: '-D b -t first' def prefix = task.ext.prefix ?: "${meta.id}" """ sort -k1,1 -k2,2n ${dmr_bed} > dmr.sorted.bed From 1ae56c7f92c9904737b60cacd3044172ff53df24 Mon Sep 17 00:00:00 2001 From: Yann Vanrobaeys Date: Mon, 28 Sep 2026 11:44:54 -0400 Subject: [PATCH 10/11] Add real subworkflow-level nf-test for DMR, tagged small Per ljwharbers's review: a public, phased, methylation-tagged fixture already exists (nf-core/test-datasets' test.sorted.phased.bam, the same one modkit/pileup's and modkit/dmr's own module tests use) -- no need for confidential data or a new fixture to get real CI coverage of the full chain. Chains MODKIT_PILEUP (--phased --modified-bases 5mC 5hmC) -> HTSLIB_BGZIPTABIX -> DMR_HAPLOTYPE_REGIONS -> MODKIT_DMR -> DMR_NEAREST_GENE via a setup block, against a small synthetic chr22:0-40001 CpG-island/gene-model pair (the exact region modkit/dmr's own module test already proves has real dual-haplotype coverage). Tagged 'small' -- .github/workflows/nf-test.yml only runs 'small'-tagged tests on pull_request, so an untagged test would never actually run in CI, which was the reviewer's core point (this exact class of gap already happened once with the vendored module test). Needed a dedicated subworkflows/local/tests/nextflow.config: this subworkflow-level nf-test session does not reliably pick up conf/modules.config's DMR-scoped withName blocks (confirmed directly -- HTSLIB_BGZIPTABIX's ext.prefix/ext.args2 and MODKIT_DMR's ext.args did not apply even with the 'DMR:' prefix present in the process name), so the real, non-stub behaviour this test exercises is set explicitly rather than inherited. Verified real, non-fabricated output: real per-haplotype methylation counts/fractions/deltas from the public BAM, correctly annotated with the test's synthetic gene entry -- not just workflow.success. Co-Authored-By: Claude Sonnet 5 Claude-Session: https://claude.ai/code/session_01Wit78XR11V9AMAgY7gGrMT --- subworkflows/local/tests/dmr.nf.test | 98 ++++++++++++++++++++++++ subworkflows/local/tests/nextflow.config | 19 +++++ 2 files changed, 117 insertions(+) create mode 100644 subworkflows/local/tests/dmr.nf.test create mode 100644 subworkflows/local/tests/nextflow.config diff --git a/subworkflows/local/tests/dmr.nf.test b/subworkflows/local/tests/dmr.nf.test new file mode 100644 index 00000000..c940e906 --- /dev/null +++ b/subworkflows/local/tests/dmr.nf.test @@ -0,0 +1,98 @@ +nextflow_workflow { + + name "Test Workflow DMR" + script "../dmr.nf" + workflow "DMR" + config "./nextflow.config" + + tag "subworkflows" + tag "subworkflows_local" + tag "dmr" + // "small" is what .github/workflows/nf-test.yml selects on for pull_request + tag "small" + + // Real (not stub), public, non-confidential coverage of the full haplotype-DMR chain: + // MODKIT_PILEUP (--phased) -> HTSLIB_BGZIPTABIX -> DMR_HAPLOTYPE_REGIONS -> MODKIT_DMR -> + // DMR_NEAREST_GENE. Uses the same public, phased nf-core/test-datasets BAM already used by + // modkit/pileup's and modkit/dmr's own module tests -- no new fixture needed. cpg_islands_bed + // reuses the exact chr22:0-40001 region modkit/dmr's own module test already proves has real + // dual-haplotype coverage; gencode_gene_bed is a small synthetic gene entry over the same span + // for DMR_NEAREST_GENE to annotate against. + setup { + run("MODKIT_PILEUP") { + script "../../../modules/nf-core/modkit/pileup/main.nf" + process { + """ + // lrsomatic's vendored modkit/pileup takes fasta/fai as separate 2-tuples plus + // a bed-region tuple -- 4 inputs, not the 3-tuple/3-input shape upstream + // nf-core/modules now uses. Matches its real call in workflows/lrsomatic.nf. + input[0] = [ + [ id: 'test', type: 'tumor' ], + file(params.modules_testdata_base_path + 'genomics/homo_sapiens/nanopore/bam/test.sorted.phased.bam', checkIfExists: true), + file(params.modules_testdata_base_path + 'genomics/homo_sapiens/nanopore/bam/test.sorted.phased.bam.bai', checkIfExists: true) + ] + input[1] = [ + [ id: 'test_ref' ], + file(params.modules_testdata_base_path + 'genomics/homo_sapiens/genome/genome.fasta', checkIfExists: true) + ] + input[2] = [ + [ id: 'test_ref' ], + file(params.modules_testdata_base_path + 'genomics/homo_sapiens/genome/genome.fasta.fai', checkIfExists: true) + ] + input[3] = [[:],[]] + """ + } + } + } + + test("real phased public data -> haplotype DMR, annotated with nearest gene") { + + when { + workflow { + """ + def cpg_islands_bed = Channel.of('chr22\\t0\\t40001') + .collectFile(name: 'cpg_islands.bed', newLine: true) + + def gencode_gene_bed = Channel.of('chr22\\t0\\t40001\\tTEST_GENE\\t.\\t+') + .collectFile(name: 'gencode_gene.bed', newLine: true) + + // DMR's internal fasta.first()/fai.first() need real single-emission value + // channels -- a bare list literal here is a 2-item queue channel (meta, then + // path, as separate emissions), and .first() would silently take just the meta. + input[0] = MODKIT_PILEUP.out.bedgz + input[1] = channel.value([ + [ id: 'test_ref' ], + file(params.modules_testdata_base_path + 'genomics/homo_sapiens/genome/genome.fasta', checkIfExists: true) + ]) + input[2] = channel.value([ + [ id: 'test_ref' ], + file(params.modules_testdata_base_path + 'genomics/homo_sapiens/genome/genome.fasta.fai', checkIfExists: true) + ]) + input[3] = cpg_islands_bed + input[4] = gencode_gene_bed + """ + } + } + + then { + def processes = workflow.trace.tasks().collect { task -> task.name } + + assertAll( + { assert workflow.success }, + { assert processes.any { name -> name.contains('HTSLIB_BGZIPTABIX') } }, + { assert processes.any { name -> name.contains('DMR_HAPLOTYPE_REGIONS') } }, + { assert processes.any { name -> name.contains('MODKIT_DMR') } }, + { assert processes.any { name -> name.contains('DMR_NEAREST_GENE') } }, + { assert workflow.out.nearest_gene_bed.size() == 1 }, + // Real content, not just a file existing: at least one DMR call, annotated + // with the synthetic gene above. + { + def bed = workflow.out.nearest_gene_bed[0][1] + def lines = file(bed).readLines() + assert lines.size() > 0 + assert lines.every { line -> line.contains('TEST_GENE') } + } + ) + } + } +} diff --git a/subworkflows/local/tests/nextflow.config b/subworkflows/local/tests/nextflow.config new file mode 100644 index 00000000..a1704d19 --- /dev/null +++ b/subworkflows/local/tests/nextflow.config @@ -0,0 +1,19 @@ +// A subworkflow-level nf-test session does not reliably pick up conf/modules.config's +// DMR-scoped withName blocks (observed directly: HTSLIB_BGZIPTABIX's ext.prefix/ext.args2 and +// MODKIT_DMR's ext.args did not apply, even though the DMR: prefix was present in the process +// name). Set everything this real (non-stub) chain needs explicitly here instead of relying on +// the pipeline's shared config, so the test is self-contained and its behaviour is guaranteed. +process { + withName: 'MODKIT_PILEUP' { + ext.args = '--phased --modified-bases 5mC 5hmC' + } + withName: '.*HTSLIB_BGZIPTABIX' { + // Without meta.haplotype in the prefix, hp1/hp2 both output ".bed.gz" and collide + // once staged together into DMR_HAPLOTYPE_REGIONS. + ext.prefix = { "${meta.id}_${meta.haplotype}" } + ext.args2 = '-p bed' + } + withName: '.*MODKIT_DMR' { + ext.args = '--base C' + } +} From 28e72e9e3e8265523dd9d4d547b8c64f248d240f Mon Sep 17 00:00:00 2001 From: Yann Vanrobaeys Date: Wed, 30 Sep 2026 09:52:14 -0400 Subject: [PATCH 11/11] MODKIT_DMR: avoid rebuilding fai every run, use the same patched modkit as pileup Per ljwharbers's review nits: - "MODKIT_DMR gets --ref without a .fai. Please check whether modkit re-indexes the 3 GB FASTA in every task." Confirmed by direct testing: modkit dmr pair has no --fai flag, looks for .fai next to --ref automatically, and builds one from scratch if it's not there (mtime proof: with a pre-built .fai present, it's read unchanged rather than rewritten). Re-vendored modkit/dmr from nf-core/modules#13021 (now fixed upstream) to accept fai as a third element of its fasta input tuple; subworkflows/local/dmr.nf now joins fasta.first() with fai.first() before calling MODKIT_DMR, instead of passing fasta alone. - "Container version mismatch: MODKIT_DMR uses 0.6.1 vs PILEUP's 0.6.4." MODKIT_PILEUP already runs a patched modkit build (ghcr.io/ljwharbers/modkit:0.6.4-pacbiofix-6e0afa2, see its own modkit-pileup.diff) for unrelated PacBio pileup fixes; MODKIT_DMR was still on the stock 0.6.1 biocontainer. Patched to the same build for one consistent modkit binary across the whole DMR chain, following the exact same pattern (container swap + a conda/mamba guard, since the patch only exists in the container) already established for pileup. Recorded as modules/nf-core/modkit/dmr/modkit-dmr.diff + a "patch" entry in modules.json, matching pileup's own convention for a locally-modified vendored module. Verified: DMR_NEAREST_GENE output is byte-identical to the unpatched run (the container swap doesn't change dmr pair's calculation, as expected -- the PacBio fix is pileup-specific), and the fai is confirmed staged correctly (genome.fasta.fai symlinked next to genome.fasta in the task work dir). Co-Authored-By: Claude Sonnet 5 Claude-Session: https://claude.ai/code/session_01Wit78XR11V9AMAgY7gGrMT --- modules.json | 209 +++++++++++++----- modules/nf-core/modkit/dmr/main.nf | 23 +- modules/nf-core/modkit/dmr/meta.yml | 9 + modules/nf-core/modkit/dmr/modkit-dmr.diff | 36 +++ modules/nf-core/modkit/dmr/tests/main.nf.test | 27 ++- .../modkit/dmr/tests/main.nf.test.snap | 147 +++--------- subworkflows/local/dmr.nf | 10 +- 7 files changed, 272 insertions(+), 189 deletions(-) create mode 100644 modules/nf-core/modkit/dmr/modkit-dmr.diff diff --git a/modules.json b/modules.json index a48fbf74..c50d0b08 100644 --- a/modules.json +++ b/modules.json @@ -2,225 +2,321 @@ "name": "IntGenomicsLab/lrsomatic", "homePage": "https://github.com/IntGenomicsLab/lrsomatic", "repos": { + "https://github.com/YannVRB/modules.git": { + "modules": { + "nf-core": { + "modkit/dmr": { + "branch": "add-modkit-dmr", + "git_sha": "fbbd19f488dcc8cb0318e92c895a431b791a237f", + "installed_by": [ + "modules" + ], + "patch": "modules/nf-core/modkit/dmr/modkit-dmr.diff" + } + } + } + }, "https://github.com/nf-core/modules.git": { "modules": { "nf-core": { "ascat": { "branch": "master", "git_sha": "e753770db613ce014b3c4bc94f6cba443427b726", - "installed_by": ["modules"], + "installed_by": [ + "modules" + ], "patch": "modules/nf-core/ascat/ascat.diff" }, "bcftools/annotate": { "branch": "master", "git_sha": "3d9c2f4beaa4f62b3f006928fd9095a496d1e5a8", - "installed_by": ["modules"] + "installed_by": [ + "modules" + ] }, "bcftools/concat": { "branch": "master", "git_sha": "6383d8fe58f9498eecd5aa303e71a4a932d1e9f6", - "installed_by": ["modules", "vcf_gather_bcftools"] + "installed_by": [ + "modules", + "vcf_gather_bcftools" + ] }, "bcftools/isec": { "branch": "master", "git_sha": "3b2c3559699a7bca6a7c2b220695a072e030e17d", - "installed_by": ["modules"], + "installed_by": [ + "modules" + ], "patch": "modules/nf-core/bcftools/isec/bcftools-isec.diff" }, "bcftools/merge": { "branch": "master", "git_sha": "3d9c2f4beaa4f62b3f006928fd9095a496d1e5a8", - "installed_by": ["modules"], + "installed_by": [ + "modules" + ], "patch": "modules/nf-core/bcftools/merge/bcftools-merge.diff" }, "bcftools/norm": { "branch": "master", "git_sha": "6383d8fe58f9498eecd5aa303e71a4a932d1e9f6", - "installed_by": ["modules"] + "installed_by": [ + "modules" + ] }, "bcftools/query": { "branch": "master", "git_sha": "6383d8fe58f9498eecd5aa303e71a4a932d1e9f6", - "installed_by": ["modules"], + "installed_by": [ + "modules" + ], "patch": "modules/nf-core/bcftools/query/bcftools-query.diff" }, "bcftools/sort": { "branch": "master", "git_sha": "6383d8fe58f9498eecd5aa303e71a4a932d1e9f6", - "installed_by": ["modules", "vcf_gather_bcftools"], + "installed_by": [ + "modules", + "vcf_gather_bcftools" + ], "patch": "modules/nf-core/bcftools/sort/bcftools-sort.diff" }, "bcftools/view": { "branch": "master", "git_sha": "feef37435aea56816adf4b3bde1fc76aac327a8d", - "installed_by": ["modules"] + "installed_by": [ + "modules" + ] }, "deepvariant/callvariants": { "branch": "master", "git_sha": "f2b138ee1d91f67d31c187317d7e83e429bf0309", - "installed_by": ["deepvariant"], + "installed_by": [ + "deepvariant" + ], "patch": "modules/nf-core/deepvariant/callvariants/deepvariant-callvariants.diff" }, "deepvariant/makeexamples": { "branch": "master", "git_sha": "f2b138ee1d91f67d31c187317d7e83e429bf0309", - "installed_by": ["deepvariant"], + "installed_by": [ + "deepvariant" + ], "patch": "modules/nf-core/deepvariant/makeexamples/deepvariant-makeexamples.diff" }, "deepvariant/postprocessvariants": { "branch": "master", "git_sha": "f2b138ee1d91f67d31c187317d7e83e429bf0309", - "installed_by": ["deepvariant"], + "installed_by": [ + "deepvariant" + ], "patch": "modules/nf-core/deepvariant/postprocessvariants/deepvariant-postprocessvariants.diff" }, "ensemblvep/download": { "branch": "master", "git_sha": "90cdd21fd96ccbdb3bc90797ca69570d18391055", - "installed_by": ["modules"] + "installed_by": [ + "modules" + ] }, "ensemblvep/vep": { "branch": "master", "git_sha": "890fdcff71928fc1470d3e669d4c430c8c770297", - "installed_by": ["modules"], + "installed_by": [ + "modules" + ], "patch": "modules/nf-core/ensemblvep/vep/ensemblvep-vep.diff" }, "htslib/bgziptabix": { "branch": "master", "git_sha": "cbe6025183336010a9b716e463f897b0c0f70f8a", - "installed_by": ["modules"] + "installed_by": [ + "modules" + ] }, "longphase/haplotag": { "branch": "master", "git_sha": "b8d30a43f33aee3148b0e9e9f00587984a4ac195", - "installed_by": ["modules"] + "installed_by": [ + "modules" + ] }, "longphase/phase": { "branch": "master", "git_sha": "b8d30a43f33aee3148b0e9e9f00587984a4ac195", - "installed_by": ["modules"], + "installed_by": [ + "modules" + ], "patch": "modules/nf-core/longphase/phase/longphase-phase.diff" }, "minimap2/align": { "branch": "master", "git_sha": "5c9f8d5b7671237c906abadc9ff732b301ca15ca", - "installed_by": ["modules"], + "installed_by": [ + "modules" + ], "patch": "modules/nf-core/minimap2/align/minimap2-align.diff" }, "minimap2/index": { "branch": "master", "git_sha": "14980f759266eec42dac401fcafeb83d6c957b41", - "installed_by": ["modules"] + "installed_by": [ + "modules" + ] }, "modkit/pileup": { "branch": "master", "git_sha": "3d81317a30d1016b533982d6b84df07713ae520a", - "installed_by": ["modules"], + "installed_by": [ + "modules" + ], "patch": "modules/nf-core/modkit/pileup/modkit-pileup.diff" }, "mosdepth": { "branch": "master", "git_sha": "6832b69ef7f98c54876d6436360b6b945370c615", - "installed_by": ["modules"] + "installed_by": [ + "modules" + ] }, "multiqc": { "branch": "master", "git_sha": "98403d15b0e50edae1f3fec5eae5e24982f1fade", - "installed_by": ["modules"] + "installed_by": [ + "modules" + ] }, "nanoplot": { "branch": "master", "git_sha": "682f789f93070bd047868300dd018faf3d434e7c", - "installed_by": ["modules"] + "installed_by": [ + "modules" + ] }, "pigz/uncompress": { "branch": "master", "git_sha": "f84336b7fa91a65aa61d215b8c109fbb8e4b4ac6", - "installed_by": ["modules"] + "installed_by": [ + "modules" + ] }, "samtools/cat": { "branch": "master", "git_sha": "f9edc59be2fe25bb6fc73ca4dfc0d28246f2a2d6", - "installed_by": ["modules"] + "installed_by": [ + "modules" + ] }, "samtools/faidx": { "branch": "master", "git_sha": "b2e78932ef01165fd85829513eaca29eff8e640a", - "installed_by": ["modules"] + "installed_by": [ + "modules" + ] }, "samtools/flagstat": { "branch": "master", "git_sha": "1d2fbdcbca677bbe8da0f9d0d2bb7c02f2cab1c9", - "installed_by": ["bam_stats_samtools"] + "installed_by": [ + "bam_stats_samtools" + ] }, "samtools/idxstats": { "branch": "master", "git_sha": "1d2fbdcbca677bbe8da0f9d0d2bb7c02f2cab1c9", - "installed_by": ["bam_stats_samtools"] + "installed_by": [ + "bam_stats_samtools" + ] }, "samtools/index": { "branch": "master", "git_sha": "1d2fbdcbca677bbe8da0f9d0d2bb7c02f2cab1c9", - "installed_by": ["modules"] + "installed_by": [ + "modules" + ] }, "samtools/merge": { "branch": "master", "git_sha": "6d46786420b4d7bc88eba026eb389c0c5535d120", - "installed_by": ["modules"] + "installed_by": [ + "modules" + ] }, "samtools/stats": { "branch": "master", "git_sha": "fe93fde0845f907fc91ad7cc7d797930408824df", - "installed_by": ["bam_stats_samtools"], + "installed_by": [ + "bam_stats_samtools" + ], "patch": "modules/nf-core/samtools/stats/samtools-stats.diff" }, "savana/classify": { "branch": "master", "git_sha": "7e612460bcb5aaf9c6f146cfb6475064c2d062ee", - "installed_by": ["modules"], + "installed_by": [ + "modules" + ], "patch": "modules/nf-core/savana/classify/savana-classify.diff" }, "savana/cna": { "branch": "master", "git_sha": "cc9117a41906d96832594d98f23da7805a87eca2", - "installed_by": ["modules"], + "installed_by": [ + "modules" + ], "patch": "modules/nf-core/savana/cna/savana-cna.diff" }, "savana/run": { "branch": "master", "git_sha": "8226aa5748298be6c00661fe611e231c64f88e71", - "installed_by": ["modules"], + "installed_by": [ + "modules" + ], "patch": "modules/nf-core/savana/run/savana-run.diff" }, "savana/to": { "branch": "master", "git_sha": "ad25ca6c4d27f57740e739886000439d8dff2061", - "installed_by": ["modules"] + "installed_by": [ + "modules" + ] }, "severus": { "branch": "master", "git_sha": "4dd9d8439a429c7ee566e0e2347f76ddeef27e66", - "installed_by": ["modules"], + "installed_by": [ + "modules" + ], "patch": "modules/nf-core/severus/severus.diff" }, "untar": { "branch": "master", "git_sha": "447f7bc0fa41dfc2400c8cad4c0291880dc060cf", - "installed_by": ["modules"] + "installed_by": [ + "modules" + ] }, "unzip": { "branch": "master", "git_sha": "4dd9d8439a429c7ee566e0e2347f76ddeef27e66", - "installed_by": ["modules"] + "installed_by": [ + "modules" + ] }, "wget": { "branch": "master", "git_sha": "41dfa3f7c0ffabb96a6a813fe321c6d1cc5b6e46", - "installed_by": ["modules"] + "installed_by": [ + "modules" + ] }, "whatshap/stats": { "branch": "master", "git_sha": "bfab71f4d68c1aaff09335a3433e7b2836918b2a", - "installed_by": ["modules"] + "installed_by": [ + "modules" + ] } } }, @@ -229,42 +325,41 @@ "bam_stats_samtools": { "branch": "master", "git_sha": "7ac6cbe7c17c2dad685da7f70496c8f48ea48687", - "installed_by": ["subworkflows"] + "installed_by": [ + "subworkflows" + ] }, "deepvariant": { "branch": "master", "git_sha": "f2b138ee1d91f67d31c187317d7e83e429bf0309", - "installed_by": ["subworkflows"], + "installed_by": [ + "subworkflows" + ], "patch": "subworkflows/nf-core/deepvariant/deepvariant.diff" }, "utils_nextflow_pipeline": { "branch": "master", "git_sha": "c2b22d85f30a706a3073387f30380704fcae013b", - "installed_by": ["subworkflows"] + "installed_by": [ + "subworkflows" + ] }, "utils_nfcore_pipeline": { "branch": "master", "git_sha": "a3fb7351b1fdb2b1de282b765816bbea190e86a8", - "installed_by": ["subworkflows"] + "installed_by": [ + "subworkflows" + ] }, "utils_nfschema_plugin": { "branch": "master", "git_sha": "a7b27fd25bfa8dcc07d299e88bd790585901a436", - "installed_by": ["subworkflows"] - } - } - } - }, - "https://github.com/YannVRB/modules.git": { - "modules": { - "nf-core": { - "modkit/dmr": { - "branch": "add-modkit-dmr", - "git_sha": "d886465c00aa9f724c92fa89b30d84dd64c61ea6", - "installed_by": ["modules"] + "installed_by": [ + "subworkflows" + ] } } } } } -} +} \ No newline at end of file diff --git a/modules/nf-core/modkit/dmr/main.nf b/modules/nf-core/modkit/dmr/main.nf index 2ee99685..41ccec1c 100644 --- a/modules/nf-core/modkit/dmr/main.nf +++ b/modules/nf-core/modkit/dmr/main.nf @@ -2,16 +2,19 @@ process MODKIT_DMR { tag "${meta.id}" label 'process_medium' - conda "${moduleDir}/environment.yml" - container "${workflow.containerEngine in ['singularity', 'apptainer'] && !task.ext.singularity_pull_docker_container - ? 'https://depot.galaxyproject.org/singularity/ont-modkit:0.6.1--hcdda2d0_0' - : 'quay.io/biocontainers/ont-modkit:0.6.1--hcdda2d0_0'}" + // Conda is not supported: matches MODKIT_PILEUP's own patch (the guard in `script:` stops + // conda/mamba runs) -- kept on the same patched build as pileup for a single consistent + // modkit binary across the whole DMR chain, rather than mixing stock 0.6.1 (dmr pair) with + // the patched 0.6.4 (pileup). Revert to the biocontainer once the patch is upstreamed/released. + container "${(workflow.containerEngine == 'singularity' || workflow.containerEngine == 'apptainer') && !task.ext.singularity_pull_docker_container + ? 'oras://ghcr.io/ljwharbers/modkit-sif:0.6.4-pacbiofix-6e0afa2' + : 'ghcr.io/ljwharbers/modkit:0.6.4-pacbiofix-6e0afa2'}" input: tuple val(meta), path(bedmethyl_a), path(bedmethyl_a_tbi) tuple val(meta2), path(bedmethyl_b), path(bedmethyl_b_tbi) tuple val(meta3), path(regions_bed) - tuple val(meta4), path(fasta) + tuple val(meta4), path(fasta), path(fai) output: tuple val(meta), path("*.bed"), emit: bed @@ -22,9 +25,19 @@ process MODKIT_DMR { task.ext.when == null || task.ext.when script: + // Exit if running this module with -profile conda / -profile mamba + if (workflow.profile.tokenize(',').intersect(['conda', 'mamba']).size() >= 1) { + error "MODKIT_DMR does not support Conda: the patched container is Docker/Singularity/Apptainer-only. Use one of those, or --skip_dmr." + } def args = task.ext.args ?: '' def prefix = task.ext.prefix ?: "${meta.id}" def regions = regions_bed ? "-r ${regions_bed}" : '' + // fai is never referenced directly -- modkit dmr pair has no --fai flag and looks for + // .fai next to --ref itself. Declaring it as an input only stages it alongside + // fasta so modkit finds and reuses it instead of building one from scratch, which for a + // multi-GB genome is a real, repeated cost across every task invocation. Confirmed by + // direct testing: with no .fai present, modkit dmr pair creates one; with one already + // staged next to fasta, it's read (unmodified mtime) rather than rebuilt. """ modkit \\ dmr pair \\ diff --git a/modules/nf-core/modkit/dmr/meta.yml b/modules/nf-core/modkit/dmr/meta.yml index b59ff450..63c958da 100644 --- a/modules/nf-core/modkit/dmr/meta.yml +++ b/modules/nf-core/modkit/dmr/meta.yml @@ -79,6 +79,15 @@ input: pattern: "*.{fa,fasta}" ontologies: - edam: "http://edamontology.org/format_1929" # FASTA + - fai: + type: file + description: | + Optional FASTA index for fasta. modkit dmr pair has no --fai flag and looks for + .fai next to --ref itself; passing one here only stages it alongside fasta + so modkit reuses it instead of building one from scratch on every invocation, which + for a multi-GB genome is a real, repeated cost. Pass `[]` to omit. + pattern: "*.fai" + ontologies: [] output: bed: - - meta: diff --git a/modules/nf-core/modkit/dmr/modkit-dmr.diff b/modules/nf-core/modkit/dmr/modkit-dmr.diff new file mode 100644 index 00000000..bd6a816a --- /dev/null +++ b/modules/nf-core/modkit/dmr/modkit-dmr.diff @@ -0,0 +1,36 @@ +Changes in component 'nf-core/modkit/dmr' +'modules/nf-core/modkit/dmr/environment.yml' is unchanged +'modules/nf-core/modkit/dmr/meta.yml' is unchanged +Changes in 'modkit/dmr/main.nf': +--- modules/nf-core/modkit/dmr/main.nf ++++ modules/nf-core/modkit/dmr/main.nf +@@ -2,10 +2,13 @@ + tag "${meta.id}" + label 'process_medium' + +- conda "${moduleDir}/environment.yml" +- container "${workflow.containerEngine in ['singularity', 'apptainer'] && !task.ext.singularity_pull_docker_container +- ? 'https://depot.galaxyproject.org/singularity/ont-modkit:0.6.1--hcdda2d0_0' +- : 'quay.io/biocontainers/ont-modkit:0.6.1--hcdda2d0_0'}" ++ // Conda is not supported: matches MODKIT_PILEUP's own patch (the guard in `script:` stops ++ // conda/mamba runs) -- kept on the same patched build as pileup for a single consistent ++ // modkit binary across the whole DMR chain, rather than mixing stock 0.6.1 (dmr pair) with ++ // the patched 0.6.4 (pileup). Revert to the biocontainer once the patch is upstreamed/released. ++ container "${(workflow.containerEngine == 'singularity' || workflow.containerEngine == 'apptainer') && !task.ext.singularity_pull_docker_container ++ ? 'oras://ghcr.io/ljwharbers/modkit-sif:0.6.4-pacbiofix-6e0afa2' ++ : 'ghcr.io/ljwharbers/modkit:0.6.4-pacbiofix-6e0afa2'}" + + input: + tuple val(meta), path(bedmethyl_a), path(bedmethyl_a_tbi) +@@ -22,6 +25,10 @@ + task.ext.when == null || task.ext.when + + script: ++ // Exit if running this module with -profile conda / -profile mamba ++ if (workflow.profile.tokenize(',').intersect(['conda', 'mamba']).size() >= 1) { ++ error "MODKIT_DMR does not support Conda: the patched container is Docker/Singularity/Apptainer-only. Use one of those, or --skip_dmr." ++ } + def args = task.ext.args ?: '' + def prefix = task.ext.prefix ?: "${meta.id}" + def regions = regions_bed ? "-r ${regions_bed}" : '' +************************************************************ diff --git a/modules/nf-core/modkit/dmr/tests/main.nf.test b/modules/nf-core/modkit/dmr/tests/main.nf.test index d1525874..d7d4e6de 100644 --- a/modules/nf-core/modkit/dmr/tests/main.nf.test +++ b/modules/nf-core/modkit/dmr/tests/main.nf.test @@ -89,7 +89,8 @@ nextflow_process { .map { file -> [ [ id:'chr22' ], file ] } input[3] = [ [ id: 'test_ref' ], - file(params.modules_testdata_base_path + 'genomics/homo_sapiens/genome/genome.fasta', checkIfExists: true) + file(params.modules_testdata_base_path + 'genomics/homo_sapiens/genome/genome.fasta', checkIfExists: true), + file(params.modules_testdata_base_path + 'genomics/homo_sapiens/genome/genome.fasta.fai', checkIfExists: true) ] """ } @@ -98,7 +99,11 @@ nextflow_process { then { assertAll ( { assert process.success }, - { assert snapshot(process.out).match() } + // log-file content includes a timestamp/runtime that varies between runs -- + // assert existence only, and snapshot everything else. + { assert process.out.log.size() == 1 }, + { assert snapshot(process.out.bed, process.out.findAll { key, val -> key.startsWith('versions') } + ).match() } ) } @@ -123,7 +128,8 @@ nextflow_process { input[2] = [ [ id:'regions' ], [] ] input[3] = [ [ id: 'test_ref' ], - file(params.modules_testdata_base_path + 'genomics/homo_sapiens/genome/genome.fasta', checkIfExists: true) + file(params.modules_testdata_base_path + 'genomics/homo_sapiens/genome/genome.fasta', checkIfExists: true), + file(params.modules_testdata_base_path + 'genomics/homo_sapiens/genome/genome.fasta.fai', checkIfExists: true) ] """ } @@ -132,7 +138,11 @@ nextflow_process { then { assertAll ( { assert process.success }, - { assert snapshot(process.out).match() } + // log-file content includes a timestamp/runtime that varies between runs -- + // assert existence only, and snapshot everything else. + { assert process.out.log.size() == 1 }, + { assert snapshot(process.out.bed, process.out.findAll { key, val -> key.startsWith('versions') } + ).match() } ) } @@ -161,7 +171,8 @@ nextflow_process { .map { file -> [ [ id:'chr22' ], file ] } input[3] = [ [ id: 'test_ref' ], - file(params.modules_testdata_base_path + 'genomics/homo_sapiens/genome/genome.fasta', checkIfExists: true) + file(params.modules_testdata_base_path + 'genomics/homo_sapiens/genome/genome.fasta', checkIfExists: true), + file(params.modules_testdata_base_path + 'genomics/homo_sapiens/genome/genome.fasta.fai', checkIfExists: true) ] """ } @@ -170,7 +181,11 @@ nextflow_process { then { assertAll ( { assert process.success }, - { assert snapshot(process.out).match() } + // log-file content includes a timestamp/runtime that varies between runs -- + // assert existence only, and snapshot everything else. + { assert process.out.log.size() == 1 }, + { assert snapshot(process.out.bed, process.out.findAll { key, val -> key.startsWith('versions') } +).match() } ) } diff --git a/modules/nf-core/modkit/dmr/tests/main.nf.test.snap b/modules/nf-core/modkit/dmr/tests/main.nf.test.snap index b8ff0798..a7039d7d 100644 --- a/modules/nf-core/modkit/dmr/tests/main.nf.test.snap +++ b/modules/nf-core/modkit/dmr/tests/main.nf.test.snap @@ -1,46 +1,15 @@ { "[bedmethyl_a, tbi], [bedmethyl_b, tbi], regions_bed, fasta": { "content": [ + [ + [ + { + "id": "test" + }, + "test.bed:md5,c57c0b35ce975dbafa19f3b4b5604dec" + ] + ], { - "0": [ - [ - { - "id": "test" - }, - "test.bed:md5,c57c0b35ce975dbafa19f3b4b5604dec" - ] - ], - "1": [ - [ - { - "id": "test" - }, - "test.log:md5,c30c1b33203d139e5648d847cac4ab05" - ] - ], - "2": [ - [ - "MODKIT_DMR", - "modkit", - "0.6.1" - ] - ], - "bed": [ - [ - { - "id": "test" - }, - "test.bed:md5,c57c0b35ce975dbafa19f3b4b5604dec" - ] - ], - "log": [ - [ - { - "id": "test" - }, - "test.log:md5,c30c1b33203d139e5648d847cac4ab05" - ] - ], "versions_modkit": [ [ "MODKIT_DMR", @@ -54,50 +23,19 @@ "nf-test": "0.9.0", "nextflow": "25.10.2" }, - "timestamp": "2026-09-24T09:10:47.411811907" + "timestamp": "2026-09-29T11:00:29.834449326" }, "[bedmethyl_a, tbi], [bedmethyl_b, tbi], [], fasta": { "content": [ + [ + [ + { + "id": "test" + }, + "test.bed:md5,e1ed04436fd4aee1c05fa4a612846417" + ] + ], { - "0": [ - [ - { - "id": "test" - }, - "test.bed:md5,e1ed04436fd4aee1c05fa4a612846417" - ] - ], - "1": [ - [ - { - "id": "test" - }, - "test.log:md5,256a0a5feab178024236ffc0b3a18c38" - ] - ], - "2": [ - [ - "MODKIT_DMR", - "modkit", - "0.6.1" - ] - ], - "bed": [ - [ - { - "id": "test" - }, - "test.bed:md5,e1ed04436fd4aee1c05fa4a612846417" - ] - ], - "log": [ - [ - { - "id": "test" - }, - "test.log:md5,256a0a5feab178024236ffc0b3a18c38" - ] - ], "versions_modkit": [ [ "MODKIT_DMR", @@ -111,50 +49,19 @@ "nf-test": "0.9.0", "nextflow": "25.10.2" }, - "timestamp": "2026-09-24T09:10:55.676443925" + "timestamp": "2026-09-29T11:00:38.70209544" }, "[bedmethyl_a, tbi], [bedmethyl_b, tbi], regions_bed, fasta - stub": { "content": [ + [ + [ + { + "id": "test" + }, + "test.bed:md5,d41d8cd98f00b204e9800998ecf8427e" + ] + ], { - "0": [ - [ - { - "id": "test" - }, - "test.bed:md5,d41d8cd98f00b204e9800998ecf8427e" - ] - ], - "1": [ - [ - { - "id": "test" - }, - "test.log:md5,d41d8cd98f00b204e9800998ecf8427e" - ] - ], - "2": [ - [ - "MODKIT_DMR", - "modkit", - "0.6.1" - ] - ], - "bed": [ - [ - { - "id": "test" - }, - "test.bed:md5,d41d8cd98f00b204e9800998ecf8427e" - ] - ], - "log": [ - [ - { - "id": "test" - }, - "test.log:md5,d41d8cd98f00b204e9800998ecf8427e" - ] - ], "versions_modkit": [ [ "MODKIT_DMR", @@ -168,6 +75,6 @@ "nf-test": "0.9.0", "nextflow": "25.10.2" }, - "timestamp": "2026-09-24T09:11:03.403287933" + "timestamp": "2026-09-29T11:00:46.52021922" } } \ No newline at end of file diff --git a/subworkflows/local/dmr.nf b/subworkflows/local/dmr.nf index 5b0a08e6..1d7d4fa5 100644 --- a/subworkflows/local/dmr.nf +++ b/subworkflows/local/dmr.nf @@ -104,11 +104,19 @@ workflow DMR { } .set { modkit_dmr_input } + // MODKIT_DMR's fai input only avoids rebuilding the index on every task -- see the + // module's own script comment. fasta/fai are each a single [[:], path] value channel + // (take: comments above), so .first() on each stays correct after the join. + fasta.first() + .combine(fai.first()) + .map { meta, fasta_file, _meta2, fai_file -> [meta, fasta_file, fai_file] } + .set { fasta_with_fai } + MODKIT_DMR ( modkit_dmr_input.hp1, modkit_dmr_input.hp2, modkit_dmr_input.regions, - fasta.first() + fasta_with_fai ) //