From 83fff3e2075ea17fc0a0bc396eae9431c0b85f34 Mon Sep 17 00:00:00 2001 From: ljwharbers Date: Wed, 23 Sep 2026 14:30:02 +0200 Subject: [PATCH 01/12] Carry aligned and haplotagged reads as CRAM (prototype) MINIMAP2_ALIGN (patched) writes a reference-compressed CRAM + .crai when bam_format is 'cram'; SAMTOOLS_MERGE gets the reference and the joins follow the cram/crai outputs; LONGPHASE_HAPLOTAG writes CRAM (--cram) and PHASING_HAPLOTYPING joins on bai or crai. ASCAT and CRAMINO_POST now get the FASTA so they can decode CRAM. CLAIRSTO drops its intermediate haplotagged copy of the tumor reads (as large as the input) once done. Co-Authored-By: Claude Opus 5.5 --- conf/modules.config | 3 +- modules/local/clairsto/main.nf | 4 ++ modules/local/cramino/main.nf | 4 +- modules/local/cramino/meta.yml | 9 +++++ modules/nf-core/minimap2/align/main.nf | 15 +++++--- .../minimap2/align/minimap2-align.diff | 38 ++++++++++++++++++- subworkflows/local/phasing_haplotyping.nf | 4 +- workflows/lrsomatic.nf | 38 ++++++++++--------- 8 files changed, 87 insertions(+), 28 deletions(-) diff --git a/conf/modules.config b/conf/modules.config index 8b20e4fd..053c5c06 100644 --- a/conf/modules.config +++ b/conf/modules.config @@ -413,7 +413,8 @@ process { ext.prefix = { "${meta.id}_${meta.type}" } ext.args = { [ - params.longphase_tag_supplementary ? "--tagSupplementary" : '' + params.longphase_tag_supplementary ? "--tagSupplementary" : '', + "--cram" ].join(' ').trim() } publishDir = [ diff --git a/modules/local/clairsto/main.nf b/modules/local/clairsto/main.nf index 05cb64d4..fd3da6a0 100644 --- a/modules/local/clairsto/main.nf +++ b/modules/local/clairsto/main.nf @@ -71,6 +71,10 @@ process CLAIRSTO { cp -- "\$src" "${prefix}_Tumor_\${table}.txt" fi done + + # The intermediate haplotagged copy of the tumor reads (phased_bam_output/) is as large as the input + # and nothing reads it after ClairS-TO exits; --remove_intermediate_dir would also take cna_output above + rm -rf tmp/phasing_output/phased_bam_output """ stub: diff --git a/modules/local/cramino/main.nf b/modules/local/cramino/main.nf index e93d100f..5e68a3d8 100644 --- a/modules/local/cramino/main.nf +++ b/modules/local/cramino/main.nf @@ -9,6 +9,7 @@ process CRAMINO { input: tuple val(meta), path(bam) + tuple val(meta2), path(fasta) // [[:], []] for BAM/uBAM input output: tuple val(meta), path("*.txt"), emit: txt @@ -21,9 +22,10 @@ process CRAMINO { script: def args = task.ext.args ?: '' def prefix = task.ext.prefix ?: "${meta.id}" + def reference = fasta ? "--reference ${fasta}" : '' """ - cramino $args $bam --arrow ${prefix}.arrow > ${prefix}_cramino.txt + cramino $args $reference $bam --arrow ${prefix}.arrow > ${prefix}_cramino.txt """ stub: diff --git a/modules/local/cramino/meta.yml b/modules/local/cramino/meta.yml index 93fbea6c..6a52d6f4 100644 --- a/modules/local/cramino/meta.yml +++ b/modules/local/cramino/meta.yml @@ -33,6 +33,15 @@ input: - edam: "http://edamontology.org/format_25722" - edam: "http://edamontology.org/format_2573" - edam: "http://edamontology.org/format_3462" + - - meta2: + type: map + description: | + Groovy Map containing reference information + e.g. `[ id:'genome' ]` + - fasta: + type: file + description: Reference FASTA to decode CRAM input; empty for BAM/uBAM + pattern: "*.{fa,fasta,fna}" output: txt: diff --git a/modules/nf-core/minimap2/align/main.nf b/modules/nf-core/minimap2/align/main.nf index df55071a..d16a0262 100644 --- a/modules/nf-core/minimap2/align/main.nf +++ b/modules/nf-core/minimap2/align/main.nf @@ -19,7 +19,8 @@ process MINIMAP2_ALIGN { output: tuple val(meta), path("*.paf") , optional: true, emit: paf tuple val(meta), path("*.bam") , optional: true, emit: bam - tuple val(meta), path("*.bam.${bam_index_extension}"), optional: true, emit: index + tuple val(meta), path("*.cram") , optional: true, emit: cram + tuple val(meta), path("*.{bam,cram}.${bam_index_extension}"), optional: true, emit: index tuple val("${task.process}"), val("minimap2"), eval("minimap2 --version"), topic: versions, emit: versions_minimap2 when: @@ -31,8 +32,11 @@ process MINIMAP2_ALIGN { def args3 = task.ext.args3 ?: '' def args4 = task.ext.args4 ?: '' def prefix = task.ext.prefix ?: "${meta.id}" - def bam_index = bam_index_extension ? "${prefix}.bam##idx##${prefix}.bam.${bam_index_extension} --write-index" : "${prefix}.bam" - def bam_output = bam_format ? "-a | samtools sort -@ ${task.cpus-1} -o ${bam_index} ${args2}" : "-o ${prefix}.paf" + // bam_format 'cram' writes a reference-compressed CRAM instead of a BAM + def out_ext = bam_format == 'cram' ? 'cram' : 'bam' + def cram_opts = bam_format == 'cram' ? "-O cram --reference ${reference}" : '' + def bam_index = bam_index_extension ? "${prefix}.${out_ext}##idx##${prefix}.${out_ext}.${bam_index_extension} --write-index" : "${prefix}.${out_ext}" + def bam_output = bam_format ? "-a | samtools sort -@ ${task.cpus-1} ${cram_opts} -o ${bam_index} ${args2}" : "-o ${prefix}.paf" def cigar_paf = cigar_paf_format && !bam_format ? "-c" : '' def set_cigar_bam = cigar_bam && bam_format ? "-L" : '' def bam_input = "${reads.extension}".matches('sam|bam|cram') @@ -53,8 +57,9 @@ process MINIMAP2_ALIGN { stub: def prefix = task.ext.prefix ?: "${meta.id}" - def output_file = bam_format ? "${prefix}.bam" : "${prefix}.paf" - def bam_index = bam_index_extension ? "touch ${prefix}.bam.${bam_index_extension}" : "" + def out_ext = bam_format == 'cram' ? 'cram' : 'bam' + def output_file = bam_format ? "${prefix}.${out_ext}" : "${prefix}.paf" + def bam_index = bam_index_extension ? "touch ${prefix}.${out_ext}.${bam_index_extension}" : "" def bam_input = "${reads.extension}".matches('sam|bam|cram') if(bam_input && !reference) { error("Error: minimap2/align BAM input mode requires reference!") diff --git a/modules/nf-core/minimap2/align/minimap2-align.diff b/modules/nf-core/minimap2/align/minimap2-align.diff index 967bb654..feef5f4b 100644 --- a/modules/nf-core/minimap2/align/minimap2-align.diff +++ b/modules/nf-core/minimap2/align/minimap2-align.diff @@ -1,4 +1,5 @@ Changes in component 'nf-core/minimap2/align' +'modules/nf-core/minimap2/align/environment.yml' is unchanged 'modules/nf-core/minimap2/align/meta.yml' is unchanged Changes in 'minimap2/align/main.nf': --- modules/nf-core/minimap2/align/main.nf @@ -11,8 +12,43 @@ Changes in 'minimap2/align/main.nf': // Note: the versions here need to match the versions used in the mulled container below and minimap2/index conda "${moduleDir}/environment.yml" +@@ -19,7 +19,8 @@ + output: + tuple val(meta), path("*.paf") , optional: true, emit: paf + tuple val(meta), path("*.bam") , optional: true, emit: bam +- tuple val(meta), path("*.bam.${bam_index_extension}"), optional: true, emit: index ++ tuple val(meta), path("*.cram") , optional: true, emit: cram ++ tuple val(meta), path("*.{bam,cram}.${bam_index_extension}"), optional: true, emit: index + tuple val("${task.process}"), val("minimap2"), eval("minimap2 --version"), topic: versions, emit: versions_minimap2 + + when: +@@ -31,8 +32,11 @@ + def args3 = task.ext.args3 ?: '' + def args4 = task.ext.args4 ?: '' + def prefix = task.ext.prefix ?: "${meta.id}" +- def bam_index = bam_index_extension ? "${prefix}.bam##idx##${prefix}.bam.${bam_index_extension} --write-index" : "${prefix}.bam" +- def bam_output = bam_format ? "-a | samtools sort -@ ${task.cpus-1} -o ${bam_index} ${args2}" : "-o ${prefix}.paf" ++ // bam_format 'cram' writes a reference-compressed CRAM instead of a BAM ++ def out_ext = bam_format == 'cram' ? 'cram' : 'bam' ++ def cram_opts = bam_format == 'cram' ? "-O cram --reference ${reference}" : '' ++ def bam_index = bam_index_extension ? "${prefix}.${out_ext}##idx##${prefix}.${out_ext}.${bam_index_extension} --write-index" : "${prefix}.${out_ext}" ++ def bam_output = bam_format ? "-a | samtools sort -@ ${task.cpus-1} ${cram_opts} -o ${bam_index} ${args2}" : "-o ${prefix}.paf" + def cigar_paf = cigar_paf_format && !bam_format ? "-c" : '' + def set_cigar_bam = cigar_bam && bam_format ? "-L" : '' + def bam_input = "${reads.extension}".matches('sam|bam|cram') +@@ -53,8 +57,9 @@ + + stub: + def prefix = task.ext.prefix ?: "${meta.id}" +- def output_file = bam_format ? "${prefix}.bam" : "${prefix}.paf" +- def bam_index = bam_index_extension ? "touch ${prefix}.bam.${bam_index_extension}" : "" ++ def out_ext = bam_format == 'cram' ? 'cram' : 'bam' ++ def output_file = bam_format ? "${prefix}.${out_ext}" : "${prefix}.paf" ++ def bam_index = bam_index_extension ? "touch ${prefix}.${out_ext}.${bam_index_extension}" : "" + def bam_input = "${reads.extension}".matches('sam|bam|cram') + if(bam_input && !reference) { + error("Error: minimap2/align BAM input mode requires reference!") -'modules/nf-core/minimap2/align/environment.yml' is unchanged 'modules/nf-core/minimap2/align/tests/main.nf.test' is unchanged 'modules/nf-core/minimap2/align/tests/main.nf.test.snap' is unchanged ************************************************************ diff --git a/subworkflows/local/phasing_haplotyping.nf b/subworkflows/local/phasing_haplotyping.nf index e80561c8..868db60f 100644 --- a/subworkflows/local/phasing_haplotyping.nf +++ b/subworkflows/local/phasing_haplotyping.nf @@ -392,13 +392,13 @@ workflow PHASING_HAPLOTYPING { // // MODULE: SAMTOOLS_INDEX (label: process_medium) // Input: [meta, bam] -- haplotagged BAM - // Output: .bai -- [meta, bai] + // Output: .bai / .crai -- [meta, index] -- whichever matches the haplotag output format // SAMTOOLS_INDEX ( tumor_normal_hapbams_ch ) tumor_normal_hapbams_ch - .join(SAMTOOLS_INDEX.out.bai) + .join(SAMTOOLS_INDEX.out.bai.mix(SAMTOOLS_INDEX.out.crai)) .set{ tumor_normal_hapbams_ch } // tumor_normal_hapbams_ch (final): [meta, bam, bai] -- haplotagged BAM with index diff --git a/workflows/lrsomatic.nf b/workflows/lrsomatic.nf index 37bd1c4c..00bad778 100644 --- a/workflows/lrsomatic.nf +++ b/workflows/lrsomatic.nf @@ -316,7 +316,7 @@ workflow LRSOMATIC { // Output: cramino_pre.out.arrow -- [meta, arrow_file] (feather format stats) // - CRAMINO_PRE( ch_samplesheet ) + CRAMINO_PRE( ch_samplesheet, [[:], []] ) if (!params.skip_nanoplot) { @@ -504,22 +504,23 @@ workflow LRSOMATIC { // MODULE: MINIMAP2_ALIGN (label: process_high) -- once per replicate; @RG encodes sample, type and replicate // Input: [meta_with_replicate, bam] -- unaligned BAM per replicate // ch_fasta -- [[:], fasta] - // sort_bam=true, cigar_paf_format='bai', cigar_bam='', split_prefix='' - // Output: .bam -- [meta_with_replicate, bam] -- coordinate-sorted aligned BAM - // .index -- [meta_with_replicate, bai] -- BAM index + // bam_format='cram', bam_index_extension='crai', cigar_paf_format='', cigar_bam='' + // Output: .cram -- [meta_with_replicate, cram] -- coordinate-sorted, reference-compressed alignments + // .index -- [meta_with_replicate, crai] -- CRAM index // MINIMAP2_ALIGN ( ch_ubams, ch_fasta, - true, - 'bai', + 'cram', + 'crai', "", "" ) - // Join BAM with index, drop the replicate field, group per sample; single-replicate samples skip SAMTOOLS_MERGE - MINIMAP2_ALIGN.out.bam + // Join CRAM with index, drop the replicate field, group per sample; single-replicate samples skip SAMTOOLS_MERGE. + // Downstream tuples keep the [meta, bam, bai] names; every consumer reads CRAM through the same slots. + MINIMAP2_ALIGN.out.cram .join(MINIMAP2_ALIGN.out.index) .map { meta, bam, bai -> def new_meta = meta.subMap('id', @@ -555,22 +556,23 @@ workflow LRSOMATIC { // // MODULE: SAMTOOLS_MERGE (label: process_low) -- replicate identity survives via the unique @RG lines - // Input: [meta, [bam...], [bai...]] -- grouped replicate BAMs + indices - // Output: .bam -- [meta, bam] -- merged BAM + // Input: [meta, [cram...], [crai...]] -- grouped replicate CRAMs + indices + // [meta, fasta, fai, gzi] -- reference to decode the inputs and encode the merged CRAM + // Output: .cram -- [meta, cram] -- merged CRAM (output format follows the input extension) // SAMTOOLS_MERGE( ch_aligned_split.multiple, - [[],[],[],[]] + ch_fasta.join(ch_fai).map { meta, fasta, fai -> [meta, fasta, fai, []] }.first() ) - // Index the merged BAM to produce a BAI (SAMTOOLS_MERGE does not create BAI inline) - SAMTOOLS_INDEX_MERGE(SAMTOOLS_MERGE.out.bam) + // Index the merged CRAM (SAMTOOLS_MERGE does not index inline) + SAMTOOLS_INDEX_MERGE(SAMTOOLS_MERGE.out.cram) - // Combine single-replicate and merged paths into a unified [meta, bam, bai] channel + // Combine single-replicate and merged paths into a unified [meta, cram, crai] channel ch_single_indexed .mix( - SAMTOOLS_MERGE.out.bam - .join(SAMTOOLS_INDEX_MERGE.out.bai) + SAMTOOLS_MERGE.out.cram + .join(SAMTOOLS_INDEX_MERGE.out.crai) ) .set { ch_index_minimap } // ch_index_minimap: [meta, bam, bai] -- one aligned BAM + index per sample (all replicates merged) @@ -681,7 +683,7 @@ workflow LRSOMATIC { allele_files, loci_files, [], - [], + ch_fasta.map { _meta, fasta -> fasta }, // alleleCounter -r, to decode CRAM gc_file, rt_file ) @@ -1028,7 +1030,7 @@ workflow LRSOMATIC { // Output: .arrow -- [meta, arrow_file] -- alignment statistics in feather format // - CRAMINO_POST ( ch_minimap_bam ) + CRAMINO_POST ( ch_minimap_bam, ch_fasta ) ch_cramino_post_txt = CRAMINO_POST.out.txt From 5dff4f116dcf3e90a1830793754d5c557f5016c0 Mon Sep 17 00:00:00 2001 From: ljwharbers Date: Wed, 23 Sep 2026 14:41:34 +0200 Subject: [PATCH 02/12] Drop the uBAM's RG tag before alignment samtools fastq -T '*' carried the basecaller's RG tag through minimap2 -y, next to the @RG added by -R, so every record had two RG tags. BAM keeps both (readers see the first), but CRAM keeps one read group per record: the last, the basecaller's, which has no @RG line in the aligned header. Co-Authored-By: Claude Opus 5.5 --- conf/modules.config | 3 +++ 1 file changed, 3 insertions(+) diff --git a/conf/modules.config b/conf/modules.config index 053c5c06..84c7a61d 100644 --- a/conf/modules.config +++ b/conf/modules.config @@ -306,6 +306,9 @@ process { "-R '${rg}'" ].join(' ').trim() } + // Drop the uBAM's own RG tag: -T '*' would carry it through -y next to the @RG from -R, and a record + // with two RG tags keeps only the last one in CRAM -- the basecaller's, which has no @RG line here + ext.args3 = "-x RG" ext.args4 = "-T '*'" publishDir = [ enabled: false From 1569c5e494a770fc01b44c69592654c77148b550 Mon Sep 17 00:00:00 2001 From: ljwharbers Date: Wed, 23 Sep 2026 15:00:37 +0200 Subject: [PATCH 03/12] Write CRAM 3.0 with the reference embedded Probe on the test-profile tasks (CRAM input, reference not reachable through the header, no EBI fallback): - Clair3 and ClairS-TO's Verdict alleleCounter bundle an htslib that cannot decode CRAM 3.1 slices; Clair3 then calls nothing and still exits 0. CRAM 3.0 decodes fine. - ClairS (internal mpileup), Severus and SAVANA's read counting open the reads without the reference; ClairS then writes empty VCFs and exits 0. embed_ref=1 makes every CRAM decodable without one, at ~1-3% size. longphase haplotag's own --cram cannot embed the reference, so with ext.cram it writes into a FIFO that samtools encodes (ext.args2). Co-Authored-By: Claude Opus 5.5 --- conf/modules.config | 11 +++- modules.json | 3 +- .../haplotag/longphase-haplotag.diff | 57 +++++++++++++++++++ modules/nf-core/longphase/haplotag/main.nf | 20 ++++++- 4 files changed, 86 insertions(+), 5 deletions(-) create mode 100644 modules/nf-core/longphase/haplotag/longphase-haplotag.diff diff --git a/conf/modules.config b/conf/modules.config index 84c7a61d..297d0c63 100644 --- a/conf/modules.config +++ b/conf/modules.config @@ -308,6 +308,10 @@ process { } // Drop the uBAM's own RG tag: -T '*' would carry it through -y next to the @RG from -R, and a record // with two RG tags keeps only the last one in CRAM -- the basecaller's, which has no @RG line here + // CRAM 3.0, not 3.1: the htslib bundled in Clair3 and in ClairS-TO's Verdict (1.11) cannot decode + // 3.1 slices and then calls nothing without failing. embed_ref=1: ClairS, ClairS-TO, Severus and + // SAVANA's read counting open the reads without passing the reference (~1-3% larger files) + ext.args2 = '--output-fmt-option version=3.0 --output-fmt-option embed_ref=1' ext.args3 = "-x RG" ext.args4 = "-T '*'" publishDir = [ @@ -317,6 +321,7 @@ process { withName: '.*:SAMTOOLS_MERGE' { ext.prefix = { "${meta.id}_${meta.type}_mapped" } + ext.args = '--output-fmt-option version=3.0 --output-fmt-option embed_ref=1' publishDir = [ enabled: false ] @@ -416,10 +421,12 @@ process { ext.prefix = { "${meta.id}_${meta.type}" } ext.args = { [ - params.longphase_tag_supplementary ? "--tagSupplementary" : '', - "--cram" + params.longphase_tag_supplementary ? "--tagSupplementary" : '' ].join(' ').trim() } + // CRAM through a FIFO, so the reference can be embedded (longphase's own --cram cannot) + ext.cram = true + ext.args2 = '--output-fmt-option version=3.0 --output-fmt-option embed_ref=1' publishDir = [ path: { "${params.outdir}/${meta.id}/bamfiles" }, mode: params.publish_dir_mode, diff --git a/modules.json b/modules.json index ab8746ac..111b1396 100644 --- a/modules.json +++ b/modules.json @@ -87,7 +87,8 @@ "longphase/haplotag": { "branch": "master", "git_sha": "b8d30a43f33aee3148b0e9e9f00587984a4ac195", - "installed_by": ["modules"] + "installed_by": ["modules"], + "patch": "modules/nf-core/longphase/haplotag/longphase-haplotag.diff" }, "longphase/phase": { "branch": "master", diff --git a/modules/nf-core/longphase/haplotag/longphase-haplotag.diff b/modules/nf-core/longphase/haplotag/longphase-haplotag.diff new file mode 100644 index 00000000..cdffb54d --- /dev/null +++ b/modules/nf-core/longphase/haplotag/longphase-haplotag.diff @@ -0,0 +1,57 @@ +Changes in component 'nf-core/longphase/haplotag' +'modules/nf-core/longphase/haplotag/environment.yml' is unchanged +'modules/nf-core/longphase/haplotag/meta.yml' is unchanged +Changes in 'longphase/haplotag/main.nf': +--- modules/nf-core/longphase/haplotag/main.nf ++++ modules/nf-core/longphase/haplotag/main.nf +@@ -23,11 +23,27 @@ + + script: + def args = task.ext.args ?: '' ++ def args2 = task.ext.args2 ?: '' + def prefix = task.ext.prefix ?: "${meta.id}" + def sv_file = svs ? "--sv-file ${svs}" : "" + def mod_file = mods ? "--mod-file ${mods}" : "" ++ // ext.cram: longphase writes its BAM into a FIFO that samtools encodes as CRAM (options from ext.args2, ++ // e.g. embed_ref=1, which longphase's own --cram cannot set); no intermediate BAM touches the disk ++ def cram_out = task.ext.cram ?: false ++ def cram_start = cram_out ? """ ++ mkfifo ${prefix}.bam ++ samtools view --threads ${task.cpus} -C -T ${fasta} ${args2} -o ${prefix}.cram ${prefix}.bam & ++ conv=\$! ++ trap 'kill \$conv 2>/dev/null || true' EXIT ++ """ : '' ++ def cram_end = cram_out ? """ ++ wait \$conv ++ trap - EXIT ++ rm ${prefix}.bam ++ """ : '' + + """ ++ ${cram_start} + longphase \\ + haplotag \\ + $args \\ +@@ -38,7 +54,7 @@ + --bam ${bam} \\ + ${sv_file} \\ + ${mod_file} +- ++ ${cram_end} + if [ -f "${prefix}.out" ]; then + mv ${prefix}.out ${prefix}.log + fi +@@ -47,7 +63,7 @@ + stub: + def args = task.ext.args ?: '' + def prefix = task.ext.prefix ?: "${meta.id}" +- def suffix = args.contains('--cram') ? "cram" : "bam" ++ def suffix = args.contains('--cram') || task.ext.cram ? "cram" : "bam" + def log = args.contains('--log') ? "touch ${prefix}.log" : '' + """ + touch ${prefix}.${suffix} + +'modules/nf-core/longphase/haplotag/tests/nextflow.config' is unchanged +'modules/nf-core/longphase/haplotag/tests/main.nf.test' is unchanged +'modules/nf-core/longphase/haplotag/tests/main.nf.test.snap' is unchanged +************************************************************ diff --git a/modules/nf-core/longphase/haplotag/main.nf b/modules/nf-core/longphase/haplotag/main.nf index 7eb84669..55862c98 100644 --- a/modules/nf-core/longphase/haplotag/main.nf +++ b/modules/nf-core/longphase/haplotag/main.nf @@ -23,11 +23,27 @@ process LONGPHASE_HAPLOTAG { script: def args = task.ext.args ?: '' + def args2 = task.ext.args2 ?: '' def prefix = task.ext.prefix ?: "${meta.id}" def sv_file = svs ? "--sv-file ${svs}" : "" def mod_file = mods ? "--mod-file ${mods}" : "" + // ext.cram: longphase writes its BAM into a FIFO that samtools encodes as CRAM (options from ext.args2, + // e.g. embed_ref=1, which longphase's own --cram cannot set); no intermediate BAM touches the disk + def cram_out = task.ext.cram ?: false + def cram_start = cram_out ? """ + mkfifo ${prefix}.bam + samtools view --threads ${task.cpus} -C -T ${fasta} ${args2} -o ${prefix}.cram ${prefix}.bam & + conv=\$! + trap 'kill \$conv 2>/dev/null || true' EXIT + """ : '' + def cram_end = cram_out ? """ + wait \$conv + trap - EXIT + rm ${prefix}.bam + """ : '' """ + ${cram_start} longphase \\ haplotag \\ $args \\ @@ -38,7 +54,7 @@ process LONGPHASE_HAPLOTAG { --bam ${bam} \\ ${sv_file} \\ ${mod_file} - + ${cram_end} if [ -f "${prefix}.out" ]; then mv ${prefix}.out ${prefix}.log fi @@ -47,7 +63,7 @@ process LONGPHASE_HAPLOTAG { stub: def args = task.ext.args ?: '' def prefix = task.ext.prefix ?: "${meta.id}" - def suffix = args.contains('--cram') ? "cram" : "bam" + def suffix = args.contains('--cram') || task.ext.cram ? "cram" : "bam" def log = args.contains('--log') ? "touch ${prefix}.log" : '' """ touch ${prefix}.${suffix} From 5481adbfb7cefd6772eafb0937f53aebede0cd79 Mon Sep 17 00:00:00 2001 From: ljwharbers Date: Wed, 23 Sep 2026 15:56:21 +0200 Subject: [PATCH 04/12] Document the CRAM outputs Co-Authored-By: Claude Opus 5.5 --- CHANGELOG.md | 1 + docs/output.md | 27 +++++++++++++++------------ 2 files changed, 16 insertions(+), 12 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 3f6e2a92..c872ab2a 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -20,6 +20,7 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 ### `Changed` +- Aligned and haplotagged reads are now CRAM 3.0 with the reference embedded instead of BAM, from `MINIMAP2_ALIGN` onwards; `bamfiles/` holds `_.cram` + `.crai`. On chr20 of three real samples CRAM is 34 % (ONT), 56 % (older PacBio) and 73 % (Revio) smaller, with identical records. Version 3.0 because Clair3 and ClairS-TO's Verdict bundle an htslib that cannot decode 3.1 and then call nothing without failing; the embedded reference (1-3 % of the file) because ClairS, ClairS-TO, Severus and SAVANA open the reads without the FASTA. `ASCAT` and `CRAMINO_POST` now get the FASTA; `CLAIRSTO` deletes its intermediate haplotagged copy of the tumour reads; the uBAM's own `RG` tag is dropped before alignment, since CRAM keeps one read group per record and kept the basecaller's (@ljwharbers). - [#201](https://github.com/IntGenomicsLab/lrsomatic/pull/201) - `CLAIRSTO` and `CLAIRSTO_VERDICT_TAG` now pull the fork image from Docker Hub: `oras://docker.io/ljwharbers/clairs-to-sif:0.5.1-verdict-chm13-c0687e8-flat` under Singularity/Apptainer and `docker.io/ljwharbers/clairs-to:0.5.1-verdict-chm13-c0687e8-flat` otherwise, instead of `ghcr.io/ljwharbers/clairs-to`. The `-cpu` SIF on ghcr failed with `PROTOCOL_ERROR` on slow links: ghcr redirects every blob download to an Azure URL that expires at the next 5-minute mark and resets a stream still open then, and Apptainer resumes neither an `oras://` nor a `docker://` download. Docker Hub's download URLs are valid for 50 minutes and only checked when the request starts. `-flat` is the same software copied into an empty image in a few layers (3.3 GB instead of 7 GB); the software and its outputs are unchanged. `docs/usage.md` describes `pullTimeout`, Docker Hub's anonymous pull limit, pre-pulling, and how to recover the remaining `oras://ghcr.io` SIFs resumably (@ljwharbers). - [#199](https://github.com/IntGenomicsLab/lrsomatic/pull/199) - `CLAIRSTO` and `CLAIRSTO_VERDICT_TAG` now run the `-cpu` rebuild of the fork image (`0.5.1-verdict-chm13-c0687e8-cpu`), which swaps PyTorch's CUDA build for the CPU build of the same version. The software is otherwise unchanged, but the Apptainer SIF drops from 6.53 GB to 3.46 GB. The old image could not be pulled on a normal VSC link: Apptainer fetches an `oras://` SIF as a single unresumable stream, and the signed blob URL ghcr redirects to expires on a 15-minute wall-clock boundary, so 6.53 GB needed 7.3 MB/s sustained and was otherwise cut mid-transfer with `PROTOCOL_ERROR` (@ljwharbers). - [#197](https://github.com/IntGenomicsLab/lrsomatic/pull/197) - `CLAIRSTO` now runs `ghcr.io/ljwharbers/clairs-to:0.5.1-verdict-chm13-c0687e8` (ClairS-TO 0.5.1) instead of `docker.io/hkubal/clairs-to:v0.4.2`: a fork that lets Verdict read its CNA resources from `--cna_resource_dir`, fixes four places where Verdict's Python port of ASCAT departed from R, and disables Verdict with a warning when its resources cannot be read. **GRCh38 results move as well as CHM13 ones.** Revert to the upstream image once HKU-BAL/ClairS-TO carries these changes. The module also selects the SIF under `-profile apptainer` and sets explicit output prefixes (@ljwharbers). diff --git a/docs/output.md b/docs/output.md index 495ab48c..235163f7 100644 --- a/docs/output.md +++ b/docs/output.md @@ -152,18 +152,21 @@ The pipeline produces per-sample output directories. Two modes exist depending o ``` ├── bamfiles -│ ├── sample_normal.bam -│ ├── sample_normal.bam.bai -│ ├── sample_tumor.bam -│ ├── sample_tumor.bam.bai -``` - -| File | Description | -| ----------------------- | ---------------------------------------------------------------------------------------------------- | -| `sample_normal.bam` | Aligned and haplotagged bam file (with methylation and nucleosome predictions) for the normal sample | -| `sample_normal.bam.bai` | index file for the normal bam file | -| ` sample_tumor.bam` | Aligned and haplotagged bam file (with methylation and nucleosome predictions) for the tumor sample | -| `sample_tumor.bam.bai` | index file for the tumor bam file | +│ ├── sample_normal.cram +│ ├── sample_normal.cram.crai +│ ├── sample_tumor.cram +│ ├── sample_tumor.cram.crai +``` + +| File | Description | +| ------------------------- | ----------------------------------------------------------------------------------------------------- | +| `sample_normal.cram` | Aligned and haplotagged CRAM file (with methylation and nucleosome predictions) for the normal sample | +| `sample_normal.cram.crai` | index file for the normal CRAM file | +| `sample_tumor.cram` | Aligned and haplotagged CRAM file (with methylation and nucleosome predictions) for the tumor sample | +| `sample_tumor.cram.crai` | index file for the tumor CRAM file | + +The CRAM files are version 3.0 with the reference embedded, so they can be read without the reference FASTA +(`samtools view sample_tumor.cram`); `samtools view -b` turns one back into a BAM. From 77bcf5477cf795d95818c2e1423f6ac8f1c32479 Mon Sep 17 00:00:00 2001 From: ljwharbers Date: Thu, 24 Sep 2026 07:23:40 +0200 Subject: [PATCH 05/12] Give ASCAT no FASTA again The ASCAT in the module's image has no ref.fasta argument, so passing the FASTA fails ascat.prepareHTS outright; the CRAMs embed their reference, so alleleCounter decodes them without one. Co-Authored-By: Claude Opus 5.5 --- workflows/lrsomatic.nf | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/workflows/lrsomatic.nf b/workflows/lrsomatic.nf index 00bad778..f8ae62ce 100644 --- a/workflows/lrsomatic.nf +++ b/workflows/lrsomatic.nf @@ -683,7 +683,7 @@ workflow LRSOMATIC { allele_files, loci_files, [], - ch_fasta.map { _meta, fasta -> fasta }, // alleleCounter -r, to decode CRAM + [], // no fasta: the image's ASCAT has no ref.fasta argument; the CRAMs embed their reference gc_file, rt_file ) From 28965de211a88375d3b638f3b3ad50c9d187d820 Mon Sep 17 00:00:00 2001 From: ljwharbers Date: Thu, 24 Sep 2026 11:48:58 +0200 Subject: [PATCH 06/12] Find ClairS-TO's intermediate directory under its sample-named path 0.5.1 writes tmp_/, so the cleanup missed the 162 GiB haplotagged copy of BL1_ont's tumour reads. Co-Authored-By: Claude Opus 5.5 --- modules/local/clairsto/main.nf | 7 ++++--- 1 file changed, 4 insertions(+), 3 deletions(-) diff --git a/modules/local/clairsto/main.nf b/modules/local/clairsto/main.nf index fd3da6a0..3313f59a 100644 --- a/modules/local/clairsto/main.nf +++ b/modules/local/clairsto/main.nf @@ -72,9 +72,10 @@ process CLAIRSTO { fi done - # The intermediate haplotagged copy of the tumor reads (phased_bam_output/) is as large as the input - # and nothing reads it after ClairS-TO exits; --remove_intermediate_dir would also take cna_output above - rm -rf tmp/phasing_output/phased_bam_output + # The intermediate haplotagged copy of the tumor reads (phased_bam_output/, always BAM, so larger than a + # CRAM input) is unused once ClairS-TO exits; --remove_intermediate_dir would also take cna_output above. + # 0.5.1 names the directory tmp_. + rm -rf tmp*/phasing_output/phased_bam_output """ stub: From 51323c881839465613a24f97471bb44caec9e29e Mon Sep 17 00:00:00 2001 From: ljwharbers Date: Fri, 25 Sep 2026 15:50:54 +0200 Subject: [PATCH 07/12] Add --aligned_format (cram, default, or bam) MINIMAP2_ALIGN, SAMTOOLS_MERGE and LONGPHASE_HAPLOTAG write CRAM 3.0 with the reference embedded only when aligned_format is cram; bam keeps BAM and .bai throughout. The merge/index joins take whichever output the format produced. The uBAM RG-tag fix applies to both formats. Tests: clair_only (the replicate samplesheet) runs --aligned_format bam and keeps its merged-reads MD5 check; the other pipeline tests expect CRAM. Snapshots not yet regenerated. Co-Authored-By: Claude Opus 5.5 --- CHANGELOG.md | 2 +- conf/modules.config | 14 +++++++------- docs/output.md | 3 ++- docs/usage.md | 7 +++++++ nextflow.config | 1 + nextflow_schema.json | 8 ++++++++ tests/clair_only.nf.test | 2 ++ tests/consensus.nf.test | 12 ++++++------ tests/deep_only.nf.test | 12 ++++++------ tests/default.nf.test | 8 ++++---- tests/union.nf.test | 12 ++++++------ workflows/lrsomatic.nf | 36 ++++++++++++++++++++---------------- 12 files changed, 70 insertions(+), 47 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index c872ab2a..3b3a4f29 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -20,7 +20,7 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 ### `Changed` -- Aligned and haplotagged reads are now CRAM 3.0 with the reference embedded instead of BAM, from `MINIMAP2_ALIGN` onwards; `bamfiles/` holds `_.cram` + `.crai`. On chr20 of three real samples CRAM is 34 % (ONT), 56 % (older PacBio) and 73 % (Revio) smaller, with identical records. Version 3.0 because Clair3 and ClairS-TO's Verdict bundle an htslib that cannot decode 3.1 and then call nothing without failing; the embedded reference (1-3 % of the file) because ClairS, ClairS-TO, Severus and SAVANA open the reads without the FASTA. `ASCAT` and `CRAMINO_POST` now get the FASTA; `CLAIRSTO` deletes its intermediate haplotagged copy of the tumour reads; the uBAM's own `RG` tag is dropped before alignment, since CRAM keeps one read group per record and kept the basecaller's (@ljwharbers). +- Aligned and haplotagged reads are now CRAM 3.0 with the reference embedded instead of BAM, from `MINIMAP2_ALIGN` onwards; `bamfiles/` holds `_.cram` + `.crai`. The new `--aligned_format` parameter (`cram` by default, or `bam`) keeps BAM throughout for tools that cannot read CRAM. On chr20 of three real samples CRAM is 34 % (ONT), 56 % (older PacBio) and 73 % (Revio) smaller, with identical records. Version 3.0 because Clair3 and ClairS-TO's Verdict bundle an htslib that cannot decode 3.1 and then call nothing without failing; the embedded reference (1-3 % of the file) because ClairS, ClairS-TO, Severus and SAVANA open the reads without the FASTA. `ASCAT` and `CRAMINO_POST` now get the FASTA; `CLAIRSTO` deletes its intermediate haplotagged copy of the tumour reads; the uBAM's own `RG` tag is dropped before alignment, since CRAM keeps one read group per record and kept the basecaller's (@ljwharbers). - [#201](https://github.com/IntGenomicsLab/lrsomatic/pull/201) - `CLAIRSTO` and `CLAIRSTO_VERDICT_TAG` now pull the fork image from Docker Hub: `oras://docker.io/ljwharbers/clairs-to-sif:0.5.1-verdict-chm13-c0687e8-flat` under Singularity/Apptainer and `docker.io/ljwharbers/clairs-to:0.5.1-verdict-chm13-c0687e8-flat` otherwise, instead of `ghcr.io/ljwharbers/clairs-to`. The `-cpu` SIF on ghcr failed with `PROTOCOL_ERROR` on slow links: ghcr redirects every blob download to an Azure URL that expires at the next 5-minute mark and resets a stream still open then, and Apptainer resumes neither an `oras://` nor a `docker://` download. Docker Hub's download URLs are valid for 50 minutes and only checked when the request starts. `-flat` is the same software copied into an empty image in a few layers (3.3 GB instead of 7 GB); the software and its outputs are unchanged. `docs/usage.md` describes `pullTimeout`, Docker Hub's anonymous pull limit, pre-pulling, and how to recover the remaining `oras://ghcr.io` SIFs resumably (@ljwharbers). - [#199](https://github.com/IntGenomicsLab/lrsomatic/pull/199) - `CLAIRSTO` and `CLAIRSTO_VERDICT_TAG` now run the `-cpu` rebuild of the fork image (`0.5.1-verdict-chm13-c0687e8-cpu`), which swaps PyTorch's CUDA build for the CPU build of the same version. The software is otherwise unchanged, but the Apptainer SIF drops from 6.53 GB to 3.46 GB. The old image could not be pulled on a normal VSC link: Apptainer fetches an `oras://` SIF as a single unresumable stream, and the signed blob URL ghcr redirects to expires on a 15-minute wall-clock boundary, so 6.53 GB needed 7.3 MB/s sustained and was otherwise cut mid-transfer with `PROTOCOL_ERROR` (@ljwharbers). - [#197](https://github.com/IntGenomicsLab/lrsomatic/pull/197) - `CLAIRSTO` now runs `ghcr.io/ljwharbers/clairs-to:0.5.1-verdict-chm13-c0687e8` (ClairS-TO 0.5.1) instead of `docker.io/hkubal/clairs-to:v0.4.2`: a fork that lets Verdict read its CNA resources from `--cna_resource_dir`, fixes four places where Verdict's Python port of ASCAT departed from R, and disables Verdict with a warning when its resources cannot be read. **GRCh38 results move as well as CHM13 ones.** Revert to the upstream image once HKU-BAL/ClairS-TO carries these changes. The module also selects the SIF under `-profile apptainer` and sets explicit output prefixes (@ljwharbers). diff --git a/conf/modules.config b/conf/modules.config index 297d0c63..b24e6c49 100644 --- a/conf/modules.config +++ b/conf/modules.config @@ -306,12 +306,12 @@ process { "-R '${rg}'" ].join(' ').trim() } + // --aligned_format cram: CRAM 3.0, not 3.1, because the htslib bundled in Clair3 and in ClairS-TO's + // Verdict (1.11) cannot decode 3.1 slices and then calls nothing without failing; embed_ref=1 because + // ClairS, ClairS-TO, Severus and SAVANA's read counting open the reads without the reference (1-7 % larger) + ext.args2 = { params.aligned_format == 'cram' ? '--output-fmt-option version=3.0 --output-fmt-option embed_ref=1' : '' } // Drop the uBAM's own RG tag: -T '*' would carry it through -y next to the @RG from -R, and a record // with two RG tags keeps only the last one in CRAM -- the basecaller's, which has no @RG line here - // CRAM 3.0, not 3.1: the htslib bundled in Clair3 and in ClairS-TO's Verdict (1.11) cannot decode - // 3.1 slices and then calls nothing without failing. embed_ref=1: ClairS, ClairS-TO, Severus and - // SAVANA's read counting open the reads without passing the reference (~1-3% larger files) - ext.args2 = '--output-fmt-option version=3.0 --output-fmt-option embed_ref=1' ext.args3 = "-x RG" ext.args4 = "-T '*'" publishDir = [ @@ -321,7 +321,7 @@ process { withName: '.*:SAMTOOLS_MERGE' { ext.prefix = { "${meta.id}_${meta.type}_mapped" } - ext.args = '--output-fmt-option version=3.0 --output-fmt-option embed_ref=1' + ext.args = { params.aligned_format == 'cram' ? '--output-fmt-option version=3.0 --output-fmt-option embed_ref=1' : '' } publishDir = [ enabled: false ] @@ -424,8 +424,8 @@ process { params.longphase_tag_supplementary ? "--tagSupplementary" : '' ].join(' ').trim() } - // CRAM through a FIFO, so the reference can be embedded (longphase's own --cram cannot) - ext.cram = true + // --aligned_format cram: CRAM through a FIFO, so the reference can be embedded (longphase's own --cram cannot) + ext.cram = { params.aligned_format == 'cram' } ext.args2 = '--output-fmt-option version=3.0 --output-fmt-option embed_ref=1' publishDir = [ path: { "${params.outdir}/${meta.id}/bamfiles" }, diff --git a/docs/output.md b/docs/output.md index 235163f7..dddacdfd 100644 --- a/docs/output.md +++ b/docs/output.md @@ -166,7 +166,8 @@ The pipeline produces per-sample output directories. Two modes exist depending o | `sample_tumor.cram.crai` | index file for the tumor CRAM file | The CRAM files are version 3.0 with the reference embedded, so they can be read without the reference FASTA -(`samtools view sample_tumor.cram`); `samtools view -b` turns one back into a BAM. +(`samtools view sample_tumor.cram`); `samtools view -b` turns one back into a BAM. With `--aligned_format bam` +this directory holds `sample_{normal,tumor}.bam` and `.bam.bai` instead. diff --git a/docs/usage.md b/docs/usage.md index c9fe4d26..560493b0 100644 --- a/docs/usage.md +++ b/docs/usage.md @@ -240,6 +240,13 @@ opt-in. See [VEP plugins](#vep-plugins) for sizes, licence terms and per-assembl | `--minimap2_ont_model` | specifies which model to use minimap2 with for ONT samples. Default = `null` | | `--minimap2_pb_model` | specifies which model to use minimap2 with for PacBio samples. Default = `null` | | `--save_secondary_alignment` | A boolean to specify if secondary alignments are kept in aligned bam file. Default = `true` | +| `--aligned_format` | `cram` or `bam`: format of the aligned and haplotagged reads. Default = `cram` (see below) | + +`--aligned_format cram` writes CRAM 3.0 with the reference embedded, from alignment onwards and in `bamfiles/`. On +long reads this is a third (ONT) to two thirds (PacBio Revio) smaller than BAM, with identical records and the same +calls from every tool in the pipeline. Because the reference is embedded, the files can be read without the FASTA. +Use `--aligned_format bam` if a tool you run on the results cannot read CRAM, or convert afterwards with +`samtools view -b -o sample_tumor.bam sample_tumor.cram`. #### ASCAT Options diff --git a/nextflow.config b/nextflow.config index 696d5161..ff7abc81 100644 --- a/nextflow.config +++ b/nextflow.config @@ -100,6 +100,7 @@ params { minimap2_ont_model = null minimap2_pb_model = null save_secondary_alignment = true + aligned_format = 'cram' // Fibertools options autocorrelation = null diff --git a/nextflow_schema.json b/nextflow_schema.json index 713e2baa..e2f6cfa0 100644 --- a/nextflow_schema.json +++ b/nextflow_schema.json @@ -191,6 +191,14 @@ "type": "boolean", "default": true, "description": "Whether to save secondary alignments" + }, + "aligned_format": { + "type": "string", + "default": "cram", + "enum": ["cram", "bam"], + "description": "File format of the aligned and haplotagged reads, in the work directory and in `bamfiles/`.", + "help_text": "`cram` writes CRAM 3.0 with the reference embedded: 33-70 % smaller than BAM on long reads, readable without the FASTA, and decoded by every tool in the pipeline. `bam` writes BAM with `.bai` indexes, for downstream tools that cannot read CRAM.", + "fa_icon": "fas fa-file-archive" } } }, diff --git a/tests/clair_only.nf.test b/tests/clair_only.nf.test index 13f6d77c..5276dd52 100644 --- a/tests/clair_only.nf.test +++ b/tests/clair_only.nf.test @@ -13,6 +13,8 @@ nextflow_pipeline { input = "https://raw.githubusercontent.com/IntGenomicsLab/test-datasets/refs/heads/main/samplesheets/test_sheet_2.csv" germline_var_keep = 'clair' somatic_var_keep = 'clair' + // The one test on BAM (the other pipeline tests run the CRAM default), incl. sample4's merge + aligned_format = 'bam' } } diff --git a/tests/consensus.nf.test b/tests/consensus.nf.test index 9689f212..efdfe685 100644 --- a/tests/consensus.nf.test +++ b/tests/consensus.nf.test @@ -66,13 +66,13 @@ nextflow_pipeline { // ── BAM files ──────────────────────────────────────────────── { ['sample1', 'sample2'].each { s -> - assert file("$launchDir/output/${s}/bamfiles/${s}_tumor.bam").exists() - assert file("$launchDir/output/${s}/bamfiles/${s}_tumor.bam.bai").exists() - assert file("$launchDir/output/${s}/bamfiles/${s}_normal.bam").exists() - assert file("$launchDir/output/${s}/bamfiles/${s}_normal.bam.bai").exists() + assert file("$launchDir/output/${s}/bamfiles/${s}_tumor.cram").exists() + assert file("$launchDir/output/${s}/bamfiles/${s}_tumor.cram.crai").exists() + assert file("$launchDir/output/${s}/bamfiles/${s}_normal.cram").exists() + assert file("$launchDir/output/${s}/bamfiles/${s}_normal.cram.crai").exists() } - assert file("$launchDir/output/sample3/bamfiles/sample3_tumor.bam").exists() - assert file("$launchDir/output/sample3/bamfiles/sample3_tumor.bam.bai").exists() + assert file("$launchDir/output/sample3/bamfiles/sample3_tumor.cram").exists() + assert file("$launchDir/output/sample3/bamfiles/sample3_tumor.cram.crai").exists() }, // ── Severus ────────────────────────────────────────────────── diff --git a/tests/deep_only.nf.test b/tests/deep_only.nf.test index 5da64375..208e3d17 100644 --- a/tests/deep_only.nf.test +++ b/tests/deep_only.nf.test @@ -56,13 +56,13 @@ nextflow_pipeline { // ── BAM files ──────────────────────────────────────────────── { ['sample1', 'sample2'].each { s -> - assert file("$launchDir/output/${s}/bamfiles/${s}_tumor.bam").exists() - assert file("$launchDir/output/${s}/bamfiles/${s}_tumor.bam.bai").exists() - assert file("$launchDir/output/${s}/bamfiles/${s}_normal.bam").exists() - assert file("$launchDir/output/${s}/bamfiles/${s}_normal.bam.bai").exists() + assert file("$launchDir/output/${s}/bamfiles/${s}_tumor.cram").exists() + assert file("$launchDir/output/${s}/bamfiles/${s}_tumor.cram.crai").exists() + assert file("$launchDir/output/${s}/bamfiles/${s}_normal.cram").exists() + assert file("$launchDir/output/${s}/bamfiles/${s}_normal.cram.crai").exists() } - assert file("$launchDir/output/sample3/bamfiles/sample3_tumor.bam").exists() - assert file("$launchDir/output/sample3/bamfiles/sample3_tumor.bam.bai").exists() + assert file("$launchDir/output/sample3/bamfiles/sample3_tumor.cram").exists() + assert file("$launchDir/output/sample3/bamfiles/sample3_tumor.cram.crai").exists() }, // ── Severus SV outputs for paired samples ──────────────────── diff --git a/tests/default.nf.test b/tests/default.nf.test index 7c473eba..fb39bfe9 100644 --- a/tests/default.nf.test +++ b/tests/default.nf.test @@ -39,10 +39,10 @@ nextflow_pipeline { assert file("$launchDir/output/sample2/variants/clairs/snvs.vcf.gz.tbi").exists() assert file("$launchDir/output/sample2/variants/severus/somatic_SVs/severus_somatic.vcf.gz").exists() assert file("$launchDir/output/sample2/variants/savana/sample2.classified.somatic.vcf").exists() - assert file("$launchDir/output/sample1/bamfiles/sample1_normal.bam").exists() - assert file("$launchDir/output/sample1/bamfiles/sample1_tumor.bam").exists() - assert file("$launchDir/output/sample1/bamfiles/sample1_normal.bam.bai").exists() - assert file("$launchDir/output/sample1/bamfiles/sample1_tumor.bam.bai").exists() + assert file("$launchDir/output/sample1/bamfiles/sample1_normal.cram").exists() + assert file("$launchDir/output/sample1/bamfiles/sample1_tumor.cram").exists() + assert file("$launchDir/output/sample1/bamfiles/sample1_normal.cram.crai").exists() + assert file("$launchDir/output/sample1/bamfiles/sample1_tumor.cram.crai").exists() assert file("$launchDir/output/sample3/variants/clairsto/indel.vcf.gz").exists() assert file("$launchDir/output/sample3/variants/clairsto/snv.vcf.gz").exists() assert file("$launchDir/output/sample3/variants/clairsto/somatic.vcf.gz").exists() diff --git a/tests/union.nf.test b/tests/union.nf.test index 2fe8925b..a5627b9e 100644 --- a/tests/union.nf.test +++ b/tests/union.nf.test @@ -66,13 +66,13 @@ nextflow_pipeline { // ── BAM files ──────────────────────────────────────────────── { ['sample1', 'sample2'].each { s -> - assert file("$launchDir/output/${s}/bamfiles/${s}_tumor.bam").exists() - assert file("$launchDir/output/${s}/bamfiles/${s}_tumor.bam.bai").exists() - assert file("$launchDir/output/${s}/bamfiles/${s}_normal.bam").exists() - assert file("$launchDir/output/${s}/bamfiles/${s}_normal.bam.bai").exists() + assert file("$launchDir/output/${s}/bamfiles/${s}_tumor.cram").exists() + assert file("$launchDir/output/${s}/bamfiles/${s}_tumor.cram.crai").exists() + assert file("$launchDir/output/${s}/bamfiles/${s}_normal.cram").exists() + assert file("$launchDir/output/${s}/bamfiles/${s}_normal.cram.crai").exists() } - assert file("$launchDir/output/sample3/bamfiles/sample3_tumor.bam").exists() - assert file("$launchDir/output/sample3/bamfiles/sample3_tumor.bam.bai").exists() + assert file("$launchDir/output/sample3/bamfiles/sample3_tumor.cram").exists() + assert file("$launchDir/output/sample3/bamfiles/sample3_tumor.cram.crai").exists() }, // ── Severus ────────────────────────────────────────────────── diff --git a/workflows/lrsomatic.nf b/workflows/lrsomatic.nf index f8ae62ce..1a743132 100644 --- a/workflows/lrsomatic.nf +++ b/workflows/lrsomatic.nf @@ -504,23 +504,27 @@ workflow LRSOMATIC { // MODULE: MINIMAP2_ALIGN (label: process_high) -- once per replicate; @RG encodes sample, type and replicate // Input: [meta_with_replicate, bam] -- unaligned BAM per replicate // ch_fasta -- [[:], fasta] - // bam_format='cram', bam_index_extension='crai', cigar_paf_format='', cigar_bam='' - // Output: .cram -- [meta_with_replicate, cram] -- coordinate-sorted, reference-compressed alignments - // .index -- [meta_with_replicate, crai] -- CRAM index + // bam_format=params.aligned_format ('cram' or 'bam'), bam_index_extension='crai' or 'bai' + // Output: .cram / .bam -- [meta_with_replicate, alignments] -- coordinate-sorted; CRAM is reference-compressed + // .index -- [meta_with_replicate, crai or bai] // + def aligned_index = params.aligned_format == 'cram' ? 'crai' : 'bai' + MINIMAP2_ALIGN ( ch_ubams, ch_fasta, - 'cram', - 'crai', + params.aligned_format, + aligned_index, "", "" ) - // Join CRAM with index, drop the replicate field, group per sample; single-replicate samples skip SAMTOOLS_MERGE. - // Downstream tuples keep the [meta, bam, bai] names; every consumer reads CRAM through the same slots. - MINIMAP2_ALIGN.out.cram + // Join alignments with index, drop the replicate field, group per sample; single-replicate samples skip SAMTOOLS_MERGE. + // Downstream tuples keep the [meta, bam, bai] names whichever format --aligned_format picks; every consumer + // reads CRAM through the same slots. + MINIMAP2_ALIGN.out.bam + .mix(MINIMAP2_ALIGN.out.cram) .join(MINIMAP2_ALIGN.out.index) .map { meta, bam, bai -> def new_meta = meta.subMap('id', @@ -556,23 +560,23 @@ workflow LRSOMATIC { // // MODULE: SAMTOOLS_MERGE (label: process_low) -- replicate identity survives via the unique @RG lines - // Input: [meta, [cram...], [crai...]] -- grouped replicate CRAMs + indices - // [meta, fasta, fai, gzi] -- reference to decode the inputs and encode the merged CRAM - // Output: .cram -- [meta, cram] -- merged CRAM (output format follows the input extension) + // Input: [meta, [cram...], [crai...]] -- grouped replicate alignments + indices + // [meta, fasta, fai, gzi] -- reference to decode CRAM inputs and encode a merged CRAM + // Output: .cram / .bam -- [meta, alignments] -- merged (output format follows the input extension) // SAMTOOLS_MERGE( ch_aligned_split.multiple, ch_fasta.join(ch_fai).map { meta, fasta, fai -> [meta, fasta, fai, []] }.first() ) - // Index the merged CRAM (SAMTOOLS_MERGE does not index inline) - SAMTOOLS_INDEX_MERGE(SAMTOOLS_MERGE.out.cram) + // Index the merged file (SAMTOOLS_MERGE does not index inline) + SAMTOOLS_INDEX_MERGE(SAMTOOLS_MERGE.out.bam.mix(SAMTOOLS_MERGE.out.cram)) - // Combine single-replicate and merged paths into a unified [meta, cram, crai] channel + // Combine single-replicate and merged paths into a unified [meta, alignments, index] channel ch_single_indexed .mix( - SAMTOOLS_MERGE.out.cram - .join(SAMTOOLS_INDEX_MERGE.out.crai) + SAMTOOLS_MERGE.out.bam.mix(SAMTOOLS_MERGE.out.cram) + .join(SAMTOOLS_INDEX_MERGE.out.bai.mix(SAMTOOLS_INDEX_MERGE.out.crai)) ) .set { ch_index_minimap } // ch_index_minimap: [meta, bam, bai] -- one aligned BAM + index per sample (all replicates merged) From 2cd3e5675f4822ab4991d8bddfd01644552d0158 Mon Sep 17 00:00:00 2001 From: ljwharbers Date: Fri, 25 Sep 2026 17:06:10 +0200 Subject: [PATCH 08/12] Regenerate pipeline-test snapshots for the CRAM default All five pipeline tests pass on apptainer (default, consensus, deep_only, union on CRAM; clair_only on --aligned_format bam, snapshot unchanged). The only drift is bamfiles/ (.bam/.bai -> .cram/.crai) and two files whose header names the input file: samtools stats' command line and Severus' breakpoints_double.csv column names. Co-Authored-By: Claude Opus 5.5 --- tests/consensus.nf.test.snap | 64 ++++++++++++++++++------------------ tests/deep_only.nf.test.snap | 64 ++++++++++++++++++------------------ tests/default.nf.test.snap | 64 ++++++++++++++++++------------------ tests/union.nf.test.snap | 64 ++++++++++++++++++------------------ 4 files changed, 128 insertions(+), 128 deletions(-) diff --git a/tests/consensus.nf.test.snap b/tests/consensus.nf.test.snap index 6bf3a01d..1502715d 100644 --- a/tests/consensus.nf.test.snap +++ b/tests/consensus.nf.test.snap @@ -249,10 +249,10 @@ "pipeline_info/lrsomatic_software_mqc_versions.yml", "sample1", "sample1/bamfiles", - "sample1/bamfiles/sample1_normal.bam", - "sample1/bamfiles/sample1_normal.bam.bai", - "sample1/bamfiles/sample1_tumor.bam", - "sample1/bamfiles/sample1_tumor.bam.bai", + "sample1/bamfiles/sample1_normal.cram", + "sample1/bamfiles/sample1_normal.cram.crai", + "sample1/bamfiles/sample1_tumor.cram", + "sample1/bamfiles/sample1_tumor.cram.crai", "sample1/qc", "sample1/qc/normal", "sample1/qc/normal/cramino_aln", @@ -383,10 +383,10 @@ "sample1/vep/somatic/sample1_SOMATIC_VEP.vcf.gz_summary.html", "sample2", "sample2/bamfiles", - "sample2/bamfiles/sample2_normal.bam", - "sample2/bamfiles/sample2_normal.bam.bai", - "sample2/bamfiles/sample2_tumor.bam", - "sample2/bamfiles/sample2_tumor.bam.bai", + "sample2/bamfiles/sample2_normal.cram", + "sample2/bamfiles/sample2_normal.cram.crai", + "sample2/bamfiles/sample2_tumor.cram", + "sample2/bamfiles/sample2_tumor.cram.crai", "sample2/qc", "sample2/qc/normal", "sample2/qc/normal/cramino_aln", @@ -516,8 +516,8 @@ "sample2/vep/somatic/sample2_SOMATIC_VEP.vcf.gz_summary.html", "sample3", "sample3/bamfiles", - "sample3/bamfiles/sample3_tumor.bam", - "sample3/bamfiles/sample3_tumor.bam.bai", + "sample3/bamfiles/sample3_tumor.cram", + "sample3/bamfiles/sample3_tumor.cram.crai", "sample3/qc", "sample3/qc/tumor", "sample3/qc/tumor/cramino_aln", @@ -607,64 +607,64 @@ "sample3/vep/somatic/sample3_SOMATIC_VEP.vcf.gz_summary.html" ], [ - "sample1_normal.bam:md5,93dd8ac8b67eb4eb4bf27e09c8f5f99b", - "sample1_normal.bam.bai:md5,75402ef1cc35229cc131155d9ec973e0", - "sample1_tumor.bam:md5,69cba03cad51bcc1d1ee8c48da042527", - "sample1_tumor.bam.bai:md5,5d633ed05021ad81ce24b1f18cbf38b4", + "sample1_normal.cram:md5,adb39d72c207de22db29bc8594dc696c", + "sample1_normal.cram.crai:md5,e88b4621eef80d6bcbce315e77740b3d", + "sample1_tumor.cram:md5,36141d43a038f8074e0b1020e35e7dfd", + "sample1_tumor.cram.crai:md5,fb1b4557202e5b3aa606c5fb5396df1c", "sample1_normal.flagstat:md5,1c41ea9923945501eb7e41f83a90502d", "sample1_normal.idxstats:md5,902e503387799123ea59255e3fca172c", - "sample1_normal.stats:md5,a8b3fba9c54efbc0934d6eacc1807140", + "sample1_normal.stats:md5,1fee9dfe18bc8b1c3f62bec6d54ae0b7", "sample1_tumor.flagstat:md5,8ff32d733c62c4910bf185ef24bf27cf", "sample1_tumor.idxstats:md5,2de140e61f9e86c9c10af20dd565cc93", - "sample1_tumor.stats:md5,1c60a1d249d2e503b0678c72e851ea93", + "sample1_tumor.stats:md5,8b33272d01216f2cb0d9039d0176ee1c", "sample1_whatshap_stats.gtf:md5,9f09f9ad1a788384cb8e46a933f77b3b", "sample1_whatshap_stats.log:md5,20135b4e9965a31d3f9bb0df7d2cec90", "sample1_whatshap_stats.tsv:md5,264d2d76a9b8d34ea4933aee325ce36e", "breakpoint_clusters.tsv:md5,d36a70de292ee130ef30da4a58bced18", "breakpoint_clusters_list.tsv:md5,0c0ce62e329f8de492487e8414c30a50", - "breakpoints_double.csv:md5,47cb0e0bbe71abdbf4f40217dfda43f9", + "breakpoints_double.csv:md5,3690b22cd4a7da90ed2f69042db45802", "read_qual.txt:md5,78247dfa2ea336eac0e128eba5e9eef4", "breakpoint_clusters.tsv:md5,d36a70de292ee130ef30da4a58bced18", "breakpoint_clusters_list.tsv:md5,0c0ce62e329f8de492487e8414c30a50", - "sample2_normal.bam:md5,f2ba30d007c521d479c6158e1e22367a", - "sample2_normal.bam.bai:md5,c3096f52115ec1e24c46fedc41f1f3d3", - "sample2_tumor.bam:md5,1c0287d24fa5b25b86e48024f2f55031", - "sample2_tumor.bam.bai:md5,62849cea5a005e3d8dbe8f9edcefaf60", + "sample2_normal.cram:md5,9b16a9fcf7eb8dc0dca4ab90ba74d448", + "sample2_normal.cram.crai:md5,f82584bbff7f33a12d2c935cee286876", + "sample2_tumor.cram:md5,d7fd6fddb51625cf08df4a88605d53cd", + "sample2_tumor.cram.crai:md5,5c3424a6b882ac546a9cb47af9e26b1b", "sample2_normal.flagstat:md5,714d0cc0c213e2640e54a16f3d0e6e7e", "sample2_normal.idxstats:md5,72eb83bb11748dc863fef1a0a5497e4b", - "sample2_normal.stats:md5,20c47cb94f9ac739d69c57be6daf82c5", + "sample2_normal.stats:md5,b00725632868d7181d67c23b8cc1d4af", "sample2_tumor.flagstat:md5,4344a8745efef9cc2a017024218d61c6", "sample2_tumor.idxstats:md5,69467fc02c83a30084736aeea8b785fb", - "sample2_tumor.stats:md5,8635df10132c85a13f2d9878b7cf90a2", + "sample2_tumor.stats:md5,154024d3791a0e2b804ae0cf78ff740e", "sample2_whatshap_stats.gtf:md5,f15fb43f0af73d02fc73b66fdc12d5d8", "sample2_whatshap_stats.log:md5,ca87088fc2f11665eca3fb9c80489085", "sample2_whatshap_stats.tsv:md5,ca53f81e39bf5d46aa4f604216add1f6", "breakpoint_clusters.tsv:md5,d36a70de292ee130ef30da4a58bced18", "breakpoint_clusters_list.tsv:md5,0c0ce62e329f8de492487e8414c30a50", - "breakpoints_double.csv:md5,48baac86492026a4a7947bc708c47e6e", + "breakpoints_double.csv:md5,64b72dd7f32cc27c7138830a00695f16", "read_qual.txt:md5,8b92ff7dc4536188be159b95525511cd", "breakpoint_clusters.tsv:md5,d36a70de292ee130ef30da4a58bced18", "breakpoint_clusters_list.tsv:md5,0c0ce62e329f8de492487e8414c30a50", - "sample3_tumor.bam:md5,3c995151bdd974df5bf61d08606d54d3", - "sample3_tumor.bam.bai:md5,9d5964ef8f44127a7d9162d628e2407d", + "sample3_tumor.cram:md5,687e562438347a2055a2dd1c92e21e32", + "sample3_tumor.cram.crai:md5,c04ca455c57db9f2fbcd904b75dd0b70", "sample3_tumor.flagstat:md5,8ff32d733c62c4910bf185ef24bf27cf", "sample3_tumor.idxstats:md5,2de140e61f9e86c9c10af20dd565cc93", - "sample3_tumor.stats:md5,ecd5ea4fee37379dd5c5ae3e89dfddda", + "sample3_tumor.stats:md5,2535ddb28ef378016f7b4cfa9042c72a", "sample3_whatshap_stats.gtf:md5,46a97067376b06b476d180709bc9e3d8", "sample3_whatshap_stats.log:md5,376254ec9c98f9ba204895e7085516ed", "sample3_whatshap_stats.tsv:md5,f7cc79156f23e884ead18e50b8434dbf", "breakpoint_clusters.tsv:md5,d36a70de292ee130ef30da4a58bced18", "breakpoint_clusters_list.tsv:md5,0c0ce62e329f8de492487e8414c30a50", - "breakpoints_double.csv:md5,56e899f85876cee082788927d0f89c5f", + "breakpoints_double.csv:md5,6ef9c12b0311d3a4a23cab394abfd389", "read_qual.txt:md5,b918430d35354dad1d7f02f21e4cd4ed", "breakpoint_clusters.tsv:md5,d36a70de292ee130ef30da4a58bced18", "breakpoint_clusters_list.tsv:md5,0c0ce62e329f8de492487e8414c30a50" ] ], + "timestamp": "2026-09-25T17:03:30.854619726", "meta": { - "nf-test": "0.9.0", - "nextflow": "26.04.6" - }, - "timestamp": "2026-09-21T11:40:36.642665738" + "nf-test": "0.9.4", + "nextflow": "26.04.3" + } } } \ No newline at end of file diff --git a/tests/deep_only.nf.test.snap b/tests/deep_only.nf.test.snap index 7887aaca..df0838de 100644 --- a/tests/deep_only.nf.test.snap +++ b/tests/deep_only.nf.test.snap @@ -219,10 +219,10 @@ "pipeline_info/lrsomatic_software_mqc_versions.yml", "sample1", "sample1/bamfiles", - "sample1/bamfiles/sample1_normal.bam", - "sample1/bamfiles/sample1_normal.bam.bai", - "sample1/bamfiles/sample1_tumor.bam", - "sample1/bamfiles/sample1_tumor.bam.bai", + "sample1/bamfiles/sample1_normal.cram", + "sample1/bamfiles/sample1_normal.cram.crai", + "sample1/bamfiles/sample1_tumor.cram", + "sample1/bamfiles/sample1_tumor.cram.crai", "sample1/qc", "sample1/qc/normal", "sample1/qc/normal/cramino_aln", @@ -345,10 +345,10 @@ "sample1/vep/somatic/sample1_SOMATIC_VEP.vcf.gz_summary.html", "sample2", "sample2/bamfiles", - "sample2/bamfiles/sample2_normal.bam", - "sample2/bamfiles/sample2_normal.bam.bai", - "sample2/bamfiles/sample2_tumor.bam", - "sample2/bamfiles/sample2_tumor.bam.bai", + "sample2/bamfiles/sample2_normal.cram", + "sample2/bamfiles/sample2_normal.cram.crai", + "sample2/bamfiles/sample2_tumor.cram", + "sample2/bamfiles/sample2_tumor.cram.crai", "sample2/qc", "sample2/qc/normal", "sample2/qc/normal/cramino_aln", @@ -470,8 +470,8 @@ "sample2/vep/somatic/sample2_SOMATIC_VEP.vcf.gz_summary.html", "sample3", "sample3/bamfiles", - "sample3/bamfiles/sample3_tumor.bam", - "sample3/bamfiles/sample3_tumor.bam.bai", + "sample3/bamfiles/sample3_tumor.cram", + "sample3/bamfiles/sample3_tumor.cram.crai", "sample3/qc", "sample3/qc/tumor", "sample3/qc/tumor/cramino_aln", @@ -552,64 +552,64 @@ "sample3/vep/somatic/sample3_SOMATIC_VEP.vcf.gz_summary.html" ], [ - "sample1_normal.bam:md5,c3eec48a64763f8a5cc7307c5fe00773", - "sample1_normal.bam.bai:md5,ddb44b5bfb0f798950d8872e9d60d784", - "sample1_tumor.bam:md5,8173a435625724d6192ff6a0e9e737bb", - "sample1_tumor.bam.bai:md5,1de9940795d293f650a22c6983b2a072", + "sample1_normal.cram:md5,2bf14606e8ebca364400893636b02a2b", + "sample1_normal.cram.crai:md5,b44487404119fe77a330016d07e9f44e", + "sample1_tumor.cram:md5,74154c2553f9e5025bbb5c015392beef", + "sample1_tumor.cram.crai:md5,0914add5fa589fe19ae6f3e329a255ad", "sample1_normal.flagstat:md5,1c41ea9923945501eb7e41f83a90502d", "sample1_normal.idxstats:md5,902e503387799123ea59255e3fca172c", - "sample1_normal.stats:md5,a8b3fba9c54efbc0934d6eacc1807140", + "sample1_normal.stats:md5,1fee9dfe18bc8b1c3f62bec6d54ae0b7", "sample1_tumor.flagstat:md5,8ff32d733c62c4910bf185ef24bf27cf", "sample1_tumor.idxstats:md5,2de140e61f9e86c9c10af20dd565cc93", - "sample1_tumor.stats:md5,1c60a1d249d2e503b0678c72e851ea93", + "sample1_tumor.stats:md5,8b33272d01216f2cb0d9039d0176ee1c", "sample1_whatshap_stats.gtf:md5,e1d0e87353a5f9aed8a9ac4bf7973427", "sample1_whatshap_stats.log:md5,bd6b83a062e22cd3201523dc4c2c13e7", "sample1_whatshap_stats.tsv:md5,7a1508751cb1daa841a577ae25f55586", "breakpoint_clusters.tsv:md5,d36a70de292ee130ef30da4a58bced18", "breakpoint_clusters_list.tsv:md5,0c0ce62e329f8de492487e8414c30a50", - "breakpoints_double.csv:md5,47cb0e0bbe71abdbf4f40217dfda43f9", + "breakpoints_double.csv:md5,3690b22cd4a7da90ed2f69042db45802", "read_qual.txt:md5,78247dfa2ea336eac0e128eba5e9eef4", "breakpoint_clusters.tsv:md5,d36a70de292ee130ef30da4a58bced18", "breakpoint_clusters_list.tsv:md5,0c0ce62e329f8de492487e8414c30a50", - "sample2_normal.bam:md5,cfe03e8510475f54eceea0c0c80db501", - "sample2_normal.bam.bai:md5,02a12e347af31e04ffa589935f89e5ed", - "sample2_tumor.bam:md5,7380c09f2ee3ebe328f1a50671408209", - "sample2_tumor.bam.bai:md5,644083528fd153b24d97ef687417e089", + "sample2_normal.cram:md5,6104c7f3e68c2baafb97d5ad798eeb4f", + "sample2_normal.cram.crai:md5,5b40c42d84817fd9c7583e45f901bbc7", + "sample2_tumor.cram:md5,cae06baf4c23485a17a2a2144d3e6fb4", + "sample2_tumor.cram.crai:md5,ad27f073c22a85ba728beb5d704bf487", "sample2_normal.flagstat:md5,714d0cc0c213e2640e54a16f3d0e6e7e", "sample2_normal.idxstats:md5,72eb83bb11748dc863fef1a0a5497e4b", - "sample2_normal.stats:md5,20c47cb94f9ac739d69c57be6daf82c5", + "sample2_normal.stats:md5,b00725632868d7181d67c23b8cc1d4af", "sample2_tumor.flagstat:md5,4344a8745efef9cc2a017024218d61c6", "sample2_tumor.idxstats:md5,69467fc02c83a30084736aeea8b785fb", - "sample2_tumor.stats:md5,8635df10132c85a13f2d9878b7cf90a2", + "sample2_tumor.stats:md5,154024d3791a0e2b804ae0cf78ff740e", "sample2_whatshap_stats.gtf:md5,af33281699a1d0da83fbe7eaff198d03", "sample2_whatshap_stats.log:md5,bbd9ab2ce07a009d9348a1d78bc6fc70", "sample2_whatshap_stats.tsv:md5,c65436f930c23ddbfd568532d07dce70", "breakpoint_clusters.tsv:md5,d36a70de292ee130ef30da4a58bced18", "breakpoint_clusters_list.tsv:md5,0c0ce62e329f8de492487e8414c30a50", - "breakpoints_double.csv:md5,48baac86492026a4a7947bc708c47e6e", + "breakpoints_double.csv:md5,64b72dd7f32cc27c7138830a00695f16", "read_qual.txt:md5,8b92ff7dc4536188be159b95525511cd", "breakpoint_clusters.tsv:md5,d36a70de292ee130ef30da4a58bced18", "breakpoint_clusters_list.tsv:md5,0c0ce62e329f8de492487e8414c30a50", - "sample3_tumor.bam:md5,965264ef8436cb887ad02c92d40fb50e", - "sample3_tumor.bam.bai:md5,cfc6329667a3c6c66e3c0ca0ace4c6e9", + "sample3_tumor.cram:md5,3c9ba316e44071dc1ebad82c9dc1cba1", + "sample3_tumor.cram.crai:md5,72903144f3633e18a8e4b31f30531268", "sample3_tumor.flagstat:md5,8ff32d733c62c4910bf185ef24bf27cf", "sample3_tumor.idxstats:md5,2de140e61f9e86c9c10af20dd565cc93", - "sample3_tumor.stats:md5,ecd5ea4fee37379dd5c5ae3e89dfddda", + "sample3_tumor.stats:md5,2535ddb28ef378016f7b4cfa9042c72a", "sample3_whatshap_stats.gtf:md5,f47156e18c490ff9a4e6efd04d43acc5", "sample3_whatshap_stats.log:md5,4f7648e763004ab764143cb4f8b6499e", "sample3_whatshap_stats.tsv:md5,4cb58bb3b663aaba23da004d69adab3e", "breakpoint_clusters.tsv:md5,d36a70de292ee130ef30da4a58bced18", "breakpoint_clusters_list.tsv:md5,0c0ce62e329f8de492487e8414c30a50", - "breakpoints_double.csv:md5,56e899f85876cee082788927d0f89c5f", + "breakpoints_double.csv:md5,6ef9c12b0311d3a4a23cab394abfd389", "read_qual.txt:md5,b918430d35354dad1d7f02f21e4cd4ed", "breakpoint_clusters.tsv:md5,d36a70de292ee130ef30da4a58bced18", "breakpoint_clusters_list.tsv:md5,0c0ce62e329f8de492487e8414c30a50" ] ], + "timestamp": "2026-09-25T16:59:23.944263755", "meta": { - "nf-test": "0.9.0", - "nextflow": "26.04.6" - }, - "timestamp": "2026-09-21T11:46:42.562085661" + "nf-test": "0.9.4", + "nextflow": "26.04.3" + } } } \ No newline at end of file diff --git a/tests/default.nf.test.snap b/tests/default.nf.test.snap index 22e72e6f..dc7b24e5 100644 --- a/tests/default.nf.test.snap +++ b/tests/default.nf.test.snap @@ -213,10 +213,10 @@ "pipeline_info/lrsomatic_software_mqc_versions.yml", "sample1", "sample1/bamfiles", - "sample1/bamfiles/sample1_normal.bam", - "sample1/bamfiles/sample1_normal.bam.bai", - "sample1/bamfiles/sample1_tumor.bam", - "sample1/bamfiles/sample1_tumor.bam.bai", + "sample1/bamfiles/sample1_normal.cram", + "sample1/bamfiles/sample1_normal.cram.crai", + "sample1/bamfiles/sample1_tumor.cram", + "sample1/bamfiles/sample1_tumor.cram.crai", "sample1/qc", "sample1/qc/normal", "sample1/qc/normal/cramino_aln", @@ -341,10 +341,10 @@ "sample1/vep/somatic/sample1_SOMATIC_VEP.vcf.gz_summary.html", "sample2", "sample2/bamfiles", - "sample2/bamfiles/sample2_normal.bam", - "sample2/bamfiles/sample2_normal.bam.bai", - "sample2/bamfiles/sample2_tumor.bam", - "sample2/bamfiles/sample2_tumor.bam.bai", + "sample2/bamfiles/sample2_normal.cram", + "sample2/bamfiles/sample2_normal.cram.crai", + "sample2/bamfiles/sample2_tumor.cram", + "sample2/bamfiles/sample2_tumor.cram.crai", "sample2/qc", "sample2/qc/normal", "sample2/qc/normal/cramino_aln", @@ -468,8 +468,8 @@ "sample2/vep/somatic/sample2_SOMATIC_VEP.vcf.gz_summary.html", "sample3", "sample3/bamfiles", - "sample3/bamfiles/sample3_tumor.bam", - "sample3/bamfiles/sample3_tumor.bam.bai", + "sample3/bamfiles/sample3_tumor.cram", + "sample3/bamfiles/sample3_tumor.cram.crai", "sample3/qc", "sample3/qc/tumor", "sample3/qc/tumor/cramino_aln", @@ -553,64 +553,64 @@ "sample3/vep/somatic/sample3_SOMATIC_VEP.vcf.gz_summary.html" ], [ - "sample1_normal.bam:md5,772e41f7cd86c03a22afbe5ec0592a6b", - "sample1_normal.bam.bai:md5,1b501f6a11efe5d2e6f47b7f1523220b", - "sample1_tumor.bam:md5,c8315c80dc92dfb5d874aef3f5dd46fb", - "sample1_tumor.bam.bai:md5,bc35f807be4b93fc795a14d701469367", + "sample1_normal.cram:md5,01ea6c95b7dbb44f512a15eb1267cfdb", + "sample1_normal.cram.crai:md5,1368f01b7e1437c072c3b2df6febe397", + "sample1_tumor.cram:md5,7db2a49e498fd4cf6d1748fbcd423750", + "sample1_tumor.cram.crai:md5,51cfed173b2ccbe7e7c6b9488ac90025", "sample1_normal.flagstat:md5,1c41ea9923945501eb7e41f83a90502d", "sample1_normal.idxstats:md5,902e503387799123ea59255e3fca172c", - "sample1_normal.stats:md5,a8b3fba9c54efbc0934d6eacc1807140", + "sample1_normal.stats:md5,1fee9dfe18bc8b1c3f62bec6d54ae0b7", "sample1_tumor.flagstat:md5,8ff32d733c62c4910bf185ef24bf27cf", "sample1_tumor.idxstats:md5,2de140e61f9e86c9c10af20dd565cc93", - "sample1_tumor.stats:md5,1c60a1d249d2e503b0678c72e851ea93", + "sample1_tumor.stats:md5,8b33272d01216f2cb0d9039d0176ee1c", "sample1_whatshap_stats.gtf:md5,eff050a68e36e778b06e0ec19435c569", "sample1_whatshap_stats.log:md5,76b73731f74fe32ef2d11f6bb0a0f71a", "sample1_whatshap_stats.tsv:md5,f566ae25b3c5a8f7e94b3d6c1b0417f8", "breakpoint_clusters.tsv:md5,d36a70de292ee130ef30da4a58bced18", "breakpoint_clusters_list.tsv:md5,0c0ce62e329f8de492487e8414c30a50", - "breakpoints_double.csv:md5,47cb0e0bbe71abdbf4f40217dfda43f9", + "breakpoints_double.csv:md5,3690b22cd4a7da90ed2f69042db45802", "read_qual.txt:md5,78247dfa2ea336eac0e128eba5e9eef4", "breakpoint_clusters.tsv:md5,d36a70de292ee130ef30da4a58bced18", "breakpoint_clusters_list.tsv:md5,0c0ce62e329f8de492487e8414c30a50", - "sample2_normal.bam:md5,3157bd11ba095a884c7951aafcfcfb1c", - "sample2_normal.bam.bai:md5,edebda44c4383173caea728acde4ac43", - "sample2_tumor.bam:md5,47b2c5f86e0493ba94ff72cea77eeae3", - "sample2_tumor.bam.bai:md5,abf2c290c815f54c2b3f8179f717d9bd", + "sample2_normal.cram:md5,091f7d95a3f2c56a68c044d282198afe", + "sample2_normal.cram.crai:md5,5c7d96951a4ab871d28463a46abe39c8", + "sample2_tumor.cram:md5,c3eb6626b9e47d35290252a9cfe1f974", + "sample2_tumor.cram.crai:md5,e3e355c59bfc6e96d82237348abe7590", "sample2_normal.flagstat:md5,714d0cc0c213e2640e54a16f3d0e6e7e", "sample2_normal.idxstats:md5,72eb83bb11748dc863fef1a0a5497e4b", - "sample2_normal.stats:md5,20c47cb94f9ac739d69c57be6daf82c5", + "sample2_normal.stats:md5,b00725632868d7181d67c23b8cc1d4af", "sample2_tumor.flagstat:md5,4344a8745efef9cc2a017024218d61c6", "sample2_tumor.idxstats:md5,69467fc02c83a30084736aeea8b785fb", - "sample2_tumor.stats:md5,8635df10132c85a13f2d9878b7cf90a2", + "sample2_tumor.stats:md5,154024d3791a0e2b804ae0cf78ff740e", "sample2_whatshap_stats.gtf:md5,4d8f4393e3aebe4e945c0b8236cf3b3e", "sample2_whatshap_stats.log:md5,10bba7bae6dd99b989ece5e5dac7a8f9", "sample2_whatshap_stats.tsv:md5,bb46226e486af9026ab76e014624e903", "breakpoint_clusters.tsv:md5,d36a70de292ee130ef30da4a58bced18", "breakpoint_clusters_list.tsv:md5,0c0ce62e329f8de492487e8414c30a50", - "breakpoints_double.csv:md5,48baac86492026a4a7947bc708c47e6e", + "breakpoints_double.csv:md5,64b72dd7f32cc27c7138830a00695f16", "read_qual.txt:md5,8b92ff7dc4536188be159b95525511cd", "breakpoint_clusters.tsv:md5,d36a70de292ee130ef30da4a58bced18", "breakpoint_clusters_list.tsv:md5,0c0ce62e329f8de492487e8414c30a50", - "sample3_tumor.bam:md5,3c995151bdd974df5bf61d08606d54d3", - "sample3_tumor.bam.bai:md5,9d5964ef8f44127a7d9162d628e2407d", + "sample3_tumor.cram:md5,9b6a5bddb20f330af72fe84767ea708e", + "sample3_tumor.cram.crai:md5,c60a2a5f8365501d7910b5acf653c4a4", "sample3_tumor.flagstat:md5,8ff32d733c62c4910bf185ef24bf27cf", "sample3_tumor.idxstats:md5,2de140e61f9e86c9c10af20dd565cc93", - "sample3_tumor.stats:md5,ecd5ea4fee37379dd5c5ae3e89dfddda", + "sample3_tumor.stats:md5,2535ddb28ef378016f7b4cfa9042c72a", "sample3_whatshap_stats.gtf:md5,46a97067376b06b476d180709bc9e3d8", "sample3_whatshap_stats.log:md5,376254ec9c98f9ba204895e7085516ed", "sample3_whatshap_stats.tsv:md5,f7cc79156f23e884ead18e50b8434dbf", "breakpoint_clusters.tsv:md5,d36a70de292ee130ef30da4a58bced18", "breakpoint_clusters_list.tsv:md5,0c0ce62e329f8de492487e8414c30a50", - "breakpoints_double.csv:md5,56e899f85876cee082788927d0f89c5f", + "breakpoints_double.csv:md5,6ef9c12b0311d3a4a23cab394abfd389", "read_qual.txt:md5,b918430d35354dad1d7f02f21e4cd4ed", "breakpoint_clusters.tsv:md5,d36a70de292ee130ef30da4a58bced18", "breakpoint_clusters_list.tsv:md5,0c0ce62e329f8de492487e8414c30a50" ] ], + "timestamp": "2026-09-25T16:06:17.446805978", "meta": { - "nf-test": "0.9.0", - "nextflow": "26.04.6" - }, - "timestamp": "2026-09-21T11:10:56.589241561" + "nf-test": "0.9.4", + "nextflow": "26.04.3" + } } } \ No newline at end of file diff --git a/tests/union.nf.test.snap b/tests/union.nf.test.snap index 654a82d7..4c8df469 100644 --- a/tests/union.nf.test.snap +++ b/tests/union.nf.test.snap @@ -249,10 +249,10 @@ "pipeline_info/lrsomatic_software_mqc_versions.yml", "sample1", "sample1/bamfiles", - "sample1/bamfiles/sample1_normal.bam", - "sample1/bamfiles/sample1_normal.bam.bai", - "sample1/bamfiles/sample1_tumor.bam", - "sample1/bamfiles/sample1_tumor.bam.bai", + "sample1/bamfiles/sample1_normal.cram", + "sample1/bamfiles/sample1_normal.cram.crai", + "sample1/bamfiles/sample1_tumor.cram", + "sample1/bamfiles/sample1_tumor.cram.crai", "sample1/qc", "sample1/qc/normal", "sample1/qc/normal/cramino_aln", @@ -383,10 +383,10 @@ "sample1/vep/somatic/sample1_SOMATIC_VEP.vcf.gz_summary.html", "sample2", "sample2/bamfiles", - "sample2/bamfiles/sample2_normal.bam", - "sample2/bamfiles/sample2_normal.bam.bai", - "sample2/bamfiles/sample2_tumor.bam", - "sample2/bamfiles/sample2_tumor.bam.bai", + "sample2/bamfiles/sample2_normal.cram", + "sample2/bamfiles/sample2_normal.cram.crai", + "sample2/bamfiles/sample2_tumor.cram", + "sample2/bamfiles/sample2_tumor.cram.crai", "sample2/qc", "sample2/qc/normal", "sample2/qc/normal/cramino_aln", @@ -516,8 +516,8 @@ "sample2/vep/somatic/sample2_SOMATIC_VEP.vcf.gz_summary.html", "sample3", "sample3/bamfiles", - "sample3/bamfiles/sample3_tumor.bam", - "sample3/bamfiles/sample3_tumor.bam.bai", + "sample3/bamfiles/sample3_tumor.cram", + "sample3/bamfiles/sample3_tumor.cram.crai", "sample3/qc", "sample3/qc/tumor", "sample3/qc/tumor/cramino_aln", @@ -607,64 +607,64 @@ "sample3/vep/somatic/sample3_SOMATIC_VEP.vcf.gz_summary.html" ], [ - "sample1_normal.bam:md5,cbfc940a38c74cbe8435c18b9da5dd32", - "sample1_normal.bam.bai:md5,c1498328929d45b2898fa2265b0d617c", - "sample1_tumor.bam:md5,b78866edf991393806d37505d16f7e3d", - "sample1_tumor.bam.bai:md5,f613de14ab19fc3a85403661a4f6188c", + "sample1_normal.cram:md5,e25986c1006876c630f8027bb718f902", + "sample1_normal.cram.crai:md5,9946b4f1cd9d3b507ec4519ef8e7e5d1", + "sample1_tumor.cram:md5,7dc25c3ebaba225087db72a2abeb4ce8", + "sample1_tumor.cram.crai:md5,db9be95f0fdafcda4e8d90e27ec7128d", "sample1_normal.flagstat:md5,1c41ea9923945501eb7e41f83a90502d", "sample1_normal.idxstats:md5,902e503387799123ea59255e3fca172c", - "sample1_normal.stats:md5,a8b3fba9c54efbc0934d6eacc1807140", + "sample1_normal.stats:md5,1fee9dfe18bc8b1c3f62bec6d54ae0b7", "sample1_tumor.flagstat:md5,8ff32d733c62c4910bf185ef24bf27cf", "sample1_tumor.idxstats:md5,2de140e61f9e86c9c10af20dd565cc93", - "sample1_tumor.stats:md5,1c60a1d249d2e503b0678c72e851ea93", + "sample1_tumor.stats:md5,8b33272d01216f2cb0d9039d0176ee1c", "sample1_whatshap_stats.gtf:md5,9ae556e13516dd47d4108acf2104bddb", "sample1_whatshap_stats.log:md5,eaddcf6a1666d4a3c1ad3316dac24139", "sample1_whatshap_stats.tsv:md5,c2773e011c2781160fd9a7741b10546b", "breakpoint_clusters.tsv:md5,d36a70de292ee130ef30da4a58bced18", "breakpoint_clusters_list.tsv:md5,0c0ce62e329f8de492487e8414c30a50", - "breakpoints_double.csv:md5,47cb0e0bbe71abdbf4f40217dfda43f9", + "breakpoints_double.csv:md5,3690b22cd4a7da90ed2f69042db45802", "read_qual.txt:md5,78247dfa2ea336eac0e128eba5e9eef4", "breakpoint_clusters.tsv:md5,d36a70de292ee130ef30da4a58bced18", "breakpoint_clusters_list.tsv:md5,0c0ce62e329f8de492487e8414c30a50", - "sample2_normal.bam:md5,f2ba30d007c521d479c6158e1e22367a", - "sample2_normal.bam.bai:md5,c3096f52115ec1e24c46fedc41f1f3d3", - "sample2_tumor.bam:md5,1c0287d24fa5b25b86e48024f2f55031", - "sample2_tumor.bam.bai:md5,62849cea5a005e3d8dbe8f9edcefaf60", + "sample2_normal.cram:md5,86f9d4dcfe3079660a19db295aca4e1e", + "sample2_normal.cram.crai:md5,5ab4d81db94a4d4ebd74b478848e9f21", + "sample2_tumor.cram:md5,2100b57665cde32e0263219890b7fd72", + "sample2_tumor.cram.crai:md5,e1dac1585ae73065f2d80c899ff93f63", "sample2_normal.flagstat:md5,714d0cc0c213e2640e54a16f3d0e6e7e", "sample2_normal.idxstats:md5,72eb83bb11748dc863fef1a0a5497e4b", - "sample2_normal.stats:md5,20c47cb94f9ac739d69c57be6daf82c5", + "sample2_normal.stats:md5,b00725632868d7181d67c23b8cc1d4af", "sample2_tumor.flagstat:md5,4344a8745efef9cc2a017024218d61c6", "sample2_tumor.idxstats:md5,69467fc02c83a30084736aeea8b785fb", - "sample2_tumor.stats:md5,8635df10132c85a13f2d9878b7cf90a2", + "sample2_tumor.stats:md5,154024d3791a0e2b804ae0cf78ff740e", "sample2_whatshap_stats.gtf:md5,f15fb43f0af73d02fc73b66fdc12d5d8", "sample2_whatshap_stats.log:md5,a6767b3490cafdcbaf3b7114644028de", "sample2_whatshap_stats.tsv:md5,570796e5e291229e8872733425e0b133", "breakpoint_clusters.tsv:md5,d36a70de292ee130ef30da4a58bced18", "breakpoint_clusters_list.tsv:md5,0c0ce62e329f8de492487e8414c30a50", - "breakpoints_double.csv:md5,48baac86492026a4a7947bc708c47e6e", + "breakpoints_double.csv:md5,64b72dd7f32cc27c7138830a00695f16", "read_qual.txt:md5,8b92ff7dc4536188be159b95525511cd", "breakpoint_clusters.tsv:md5,d36a70de292ee130ef30da4a58bced18", "breakpoint_clusters_list.tsv:md5,0c0ce62e329f8de492487e8414c30a50", - "sample3_tumor.bam:md5,116b6944da4aa833a8d21c46b5f5ecfe", - "sample3_tumor.bam.bai:md5,ea9eca53bbaba26d40b791a2ea1aadf6", + "sample3_tumor.cram:md5,3b6761b6f9215baf74f28c955de48805", + "sample3_tumor.cram.crai:md5,2cbfc7f7c40d4ac97e4933abbd14e22b", "sample3_tumor.flagstat:md5,8ff32d733c62c4910bf185ef24bf27cf", "sample3_tumor.idxstats:md5,2de140e61f9e86c9c10af20dd565cc93", - "sample3_tumor.stats:md5,ecd5ea4fee37379dd5c5ae3e89dfddda", + "sample3_tumor.stats:md5,2535ddb28ef378016f7b4cfa9042c72a", "sample3_whatshap_stats.gtf:md5,f47156e18c490ff9a4e6efd04d43acc5", "sample3_whatshap_stats.log:md5,679dcfa209888a9e69a07e4c4e4b049e", "sample3_whatshap_stats.tsv:md5,035d5aa0425ba3fc32d65268b793b424", "breakpoint_clusters.tsv:md5,d36a70de292ee130ef30da4a58bced18", "breakpoint_clusters_list.tsv:md5,0c0ce62e329f8de492487e8414c30a50", - "breakpoints_double.csv:md5,56e899f85876cee082788927d0f89c5f", + "breakpoints_double.csv:md5,6ef9c12b0311d3a4a23cab394abfd389", "read_qual.txt:md5,b918430d35354dad1d7f02f21e4cd4ed", "breakpoint_clusters.tsv:md5,d36a70de292ee130ef30da4a58bced18", "breakpoint_clusters_list.tsv:md5,0c0ce62e329f8de492487e8414c30a50" ] ], + "timestamp": "2026-09-25T17:03:32.469972501", "meta": { - "nf-test": "0.9.0", - "nextflow": "26.04.6" - }, - "timestamp": "2026-09-21T11:54:50.88115842" + "nf-test": "0.9.4", + "nextflow": "26.04.3" + } } } \ No newline at end of file From ef82aa0af8a57de5dd4587efac50eadf3d59e640 Mon Sep 17 00:00:00 2001 From: ljwharbers Date: Mon, 28 Sep 2026 11:05:17 +0200 Subject: [PATCH 09/12] Fill in PR number in CHANGELOG entry Co-Authored-By: Claude Opus 5.5 --- CHANGELOG.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index a5dadf84..99056971 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -20,7 +20,7 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 ### `Changed` -- Aligned and haplotagged reads are now CRAM 3.0 with the reference embedded instead of BAM, from `MINIMAP2_ALIGN` onwards; `bamfiles/` holds `_.cram` + `.crai`. The new `--aligned_format` parameter (`cram` by default, or `bam`) keeps BAM throughout for tools that cannot read CRAM. On chr20 of three real samples CRAM is 34 % (ONT), 56 % (older PacBio) and 73 % (Revio) smaller, with identical records. Version 3.0 because Clair3 and ClairS-TO's Verdict bundle an htslib that cannot decode 3.1 and then call nothing without failing; the embedded reference (1-3 % of the file) because ClairS, ClairS-TO, Severus and SAVANA open the reads without the FASTA. `ASCAT` and `CRAMINO_POST` now get the FASTA; `CLAIRSTO` deletes its intermediate haplotagged copy of the tumour reads; the uBAM's own `RG` tag is dropped before alignment, since CRAM keeps one read group per record and kept the basecaller's (@ljwharbers). +- [#207](https://github.com/IntGenomicsLab/lrsomatic/pull/207) - Aligned and haplotagged reads are now CRAM 3.0 with the reference embedded instead of BAM, from `MINIMAP2_ALIGN` onwards; `bamfiles/` holds `_.cram` + `.crai`. The new `--aligned_format` parameter (`cram` by default, or `bam`) keeps BAM throughout for tools that cannot read CRAM. On chr20 of three real samples CRAM is 34 % (ONT), 56 % (older PacBio) and 73 % (Revio) smaller, with identical records. Version 3.0 because Clair3 and ClairS-TO's Verdict bundle an htslib that cannot decode 3.1 and then call nothing without failing; the embedded reference (1-3 % of the file) because ClairS, ClairS-TO, Severus and SAVANA open the reads without the FASTA. `ASCAT` and `CRAMINO_POST` now get the FASTA; `CLAIRSTO` deletes its intermediate haplotagged copy of the tumour reads; the uBAM's own `RG` tag is dropped before alignment, since CRAM keeps one read group per record and kept the basecaller's (@ljwharbers). - [#201](https://github.com/IntGenomicsLab/lrsomatic/pull/201) - `CLAIRSTO` and `CLAIRSTO_VERDICT_TAG` now pull the fork image from Docker Hub: `oras://docker.io/ljwharbers/clairs-to-sif:0.5.1-verdict-chm13-c0687e8-flat` under Singularity/Apptainer and `docker.io/ljwharbers/clairs-to:0.5.1-verdict-chm13-c0687e8-flat` otherwise, instead of `ghcr.io/ljwharbers/clairs-to`. The `-cpu` SIF on ghcr failed with `PROTOCOL_ERROR` on slow links: ghcr redirects every blob download to an Azure URL that expires at the next 5-minute mark and resets a stream still open then, and Apptainer resumes neither an `oras://` nor a `docker://` download. Docker Hub's download URLs are valid for 50 minutes and only checked when the request starts. `-flat` is the same software copied into an empty image in a few layers (3.3 GB instead of 7 GB); the software and its outputs are unchanged. `docs/usage.md` describes `pullTimeout`, Docker Hub's anonymous pull limit, pre-pulling, and how to recover the remaining `oras://ghcr.io` SIFs resumably (@ljwharbers). - [#199](https://github.com/IntGenomicsLab/lrsomatic/pull/199) - `CLAIRSTO` and `CLAIRSTO_VERDICT_TAG` now run the `-cpu` rebuild of the fork image (`0.5.1-verdict-chm13-c0687e8-cpu`), which swaps PyTorch's CUDA build for the CPU build of the same version. The software is otherwise unchanged, but the Apptainer SIF drops from 6.53 GB to 3.46 GB. The old image could not be pulled on a normal VSC link: Apptainer fetches an `oras://` SIF as a single unresumable stream, and the signed blob URL ghcr redirects to expires on a 15-minute wall-clock boundary, so 6.53 GB needed 7.3 MB/s sustained and was otherwise cut mid-transfer with `PROTOCOL_ERROR` (@ljwharbers). - [#197](https://github.com/IntGenomicsLab/lrsomatic/pull/197) - `CLAIRSTO` now runs `ghcr.io/ljwharbers/clairs-to:0.5.1-verdict-chm13-c0687e8` (ClairS-TO 0.5.1) instead of `docker.io/hkubal/clairs-to:v0.4.2`: a fork that lets Verdict read its CNA resources from `--cna_resource_dir`, fixes four places where Verdict's Python port of ASCAT departed from R, and disables Verdict with a warning when its resources cannot be read. **GRCh38 results move as well as CHM13 ones.** Revert to the upstream image once HKU-BAL/ClairS-TO carries these changes. The module also selects the SIF under `-profile apptainer` and sets explicit output prefixes (@ljwharbers). From e2aca1b992ff004ac7c8778ce1679b3dcdb13a56 Mon Sep 17 00:00:00 2001 From: ljwharbers Date: Mon, 28 Sep 2026 15:30:13 +0200 Subject: [PATCH 10/12] Stop snapshotting CRAM/CRAI md5s in the pipeline tests samtools view --threads lays CRAM containers out differently on every run, so the published CRAM and CRAI bytes never repeat: PR #207 CI got a different set of md5s on each shard and attempt while flagstat, idxstats and stats (the read content) matched. Re-encoding one BAM three times with the module's image confirms it: with 4 threads the .crai changes between identical runs, with 1 thread only the path-derived file ID and @PG differ. Ignore */bamfiles/*.cram{,.crai} in tests/.nftignore (the names stay in stable_name) and drop the 10 md5 entries from the four CRAM snapshots. Verified on Mindwell: default (twice), consensus, deep_only and union pass without --update-snapshot. Co-Authored-By: Claude Opus 5.5 --- tests/.nftignore | 3 +++ tests/consensus.nf.test.snap | 10 ---------- tests/deep_only.nf.test.snap | 10 ---------- tests/default.nf.test.snap | 10 ---------- tests/union.nf.test.snap | 10 ---------- 5 files changed, 3 insertions(+), 40 deletions(-) diff --git a/tests/.nftignore b/tests/.nftignore index dfc63dec..a1d91854 100644 --- a/tests/.nftignore +++ b/tests/.nftignore @@ -34,3 +34,6 @@ pipeline_info/*.{html,json,txt,yml} # samtools merge gives sample4's colliding @PG IDs a random suffix, so this BAM's md5 differs every run (reads asserted in tests/clair_only.nf.test) sample4/bamfiles/sample4_tumor.bam sample4/bamfiles/sample4_tumor.bam.bai +# samtools view --threads lays CRAM containers out differently every run, so the CRAM/CRAI md5s never repeat (reads covered by the flagstat/idxstats/stats md5s) +*/bamfiles/*.cram +*/bamfiles/*.cram.crai diff --git a/tests/consensus.nf.test.snap b/tests/consensus.nf.test.snap index 1502715d..1b6fddc8 100644 --- a/tests/consensus.nf.test.snap +++ b/tests/consensus.nf.test.snap @@ -607,10 +607,6 @@ "sample3/vep/somatic/sample3_SOMATIC_VEP.vcf.gz_summary.html" ], [ - "sample1_normal.cram:md5,adb39d72c207de22db29bc8594dc696c", - "sample1_normal.cram.crai:md5,e88b4621eef80d6bcbce315e77740b3d", - "sample1_tumor.cram:md5,36141d43a038f8074e0b1020e35e7dfd", - "sample1_tumor.cram.crai:md5,fb1b4557202e5b3aa606c5fb5396df1c", "sample1_normal.flagstat:md5,1c41ea9923945501eb7e41f83a90502d", "sample1_normal.idxstats:md5,902e503387799123ea59255e3fca172c", "sample1_normal.stats:md5,1fee9dfe18bc8b1c3f62bec6d54ae0b7", @@ -626,10 +622,6 @@ "read_qual.txt:md5,78247dfa2ea336eac0e128eba5e9eef4", "breakpoint_clusters.tsv:md5,d36a70de292ee130ef30da4a58bced18", "breakpoint_clusters_list.tsv:md5,0c0ce62e329f8de492487e8414c30a50", - "sample2_normal.cram:md5,9b16a9fcf7eb8dc0dca4ab90ba74d448", - "sample2_normal.cram.crai:md5,f82584bbff7f33a12d2c935cee286876", - "sample2_tumor.cram:md5,d7fd6fddb51625cf08df4a88605d53cd", - "sample2_tumor.cram.crai:md5,5c3424a6b882ac546a9cb47af9e26b1b", "sample2_normal.flagstat:md5,714d0cc0c213e2640e54a16f3d0e6e7e", "sample2_normal.idxstats:md5,72eb83bb11748dc863fef1a0a5497e4b", "sample2_normal.stats:md5,b00725632868d7181d67c23b8cc1d4af", @@ -645,8 +637,6 @@ "read_qual.txt:md5,8b92ff7dc4536188be159b95525511cd", "breakpoint_clusters.tsv:md5,d36a70de292ee130ef30da4a58bced18", "breakpoint_clusters_list.tsv:md5,0c0ce62e329f8de492487e8414c30a50", - "sample3_tumor.cram:md5,687e562438347a2055a2dd1c92e21e32", - "sample3_tumor.cram.crai:md5,c04ca455c57db9f2fbcd904b75dd0b70", "sample3_tumor.flagstat:md5,8ff32d733c62c4910bf185ef24bf27cf", "sample3_tumor.idxstats:md5,2de140e61f9e86c9c10af20dd565cc93", "sample3_tumor.stats:md5,2535ddb28ef378016f7b4cfa9042c72a", diff --git a/tests/deep_only.nf.test.snap b/tests/deep_only.nf.test.snap index df0838de..d9061324 100644 --- a/tests/deep_only.nf.test.snap +++ b/tests/deep_only.nf.test.snap @@ -552,10 +552,6 @@ "sample3/vep/somatic/sample3_SOMATIC_VEP.vcf.gz_summary.html" ], [ - "sample1_normal.cram:md5,2bf14606e8ebca364400893636b02a2b", - "sample1_normal.cram.crai:md5,b44487404119fe77a330016d07e9f44e", - "sample1_tumor.cram:md5,74154c2553f9e5025bbb5c015392beef", - "sample1_tumor.cram.crai:md5,0914add5fa589fe19ae6f3e329a255ad", "sample1_normal.flagstat:md5,1c41ea9923945501eb7e41f83a90502d", "sample1_normal.idxstats:md5,902e503387799123ea59255e3fca172c", "sample1_normal.stats:md5,1fee9dfe18bc8b1c3f62bec6d54ae0b7", @@ -571,10 +567,6 @@ "read_qual.txt:md5,78247dfa2ea336eac0e128eba5e9eef4", "breakpoint_clusters.tsv:md5,d36a70de292ee130ef30da4a58bced18", "breakpoint_clusters_list.tsv:md5,0c0ce62e329f8de492487e8414c30a50", - "sample2_normal.cram:md5,6104c7f3e68c2baafb97d5ad798eeb4f", - "sample2_normal.cram.crai:md5,5b40c42d84817fd9c7583e45f901bbc7", - "sample2_tumor.cram:md5,cae06baf4c23485a17a2a2144d3e6fb4", - "sample2_tumor.cram.crai:md5,ad27f073c22a85ba728beb5d704bf487", "sample2_normal.flagstat:md5,714d0cc0c213e2640e54a16f3d0e6e7e", "sample2_normal.idxstats:md5,72eb83bb11748dc863fef1a0a5497e4b", "sample2_normal.stats:md5,b00725632868d7181d67c23b8cc1d4af", @@ -590,8 +582,6 @@ "read_qual.txt:md5,8b92ff7dc4536188be159b95525511cd", "breakpoint_clusters.tsv:md5,d36a70de292ee130ef30da4a58bced18", "breakpoint_clusters_list.tsv:md5,0c0ce62e329f8de492487e8414c30a50", - "sample3_tumor.cram:md5,3c9ba316e44071dc1ebad82c9dc1cba1", - "sample3_tumor.cram.crai:md5,72903144f3633e18a8e4b31f30531268", "sample3_tumor.flagstat:md5,8ff32d733c62c4910bf185ef24bf27cf", "sample3_tumor.idxstats:md5,2de140e61f9e86c9c10af20dd565cc93", "sample3_tumor.stats:md5,2535ddb28ef378016f7b4cfa9042c72a", diff --git a/tests/default.nf.test.snap b/tests/default.nf.test.snap index dc7b24e5..a890840b 100644 --- a/tests/default.nf.test.snap +++ b/tests/default.nf.test.snap @@ -553,10 +553,6 @@ "sample3/vep/somatic/sample3_SOMATIC_VEP.vcf.gz_summary.html" ], [ - "sample1_normal.cram:md5,01ea6c95b7dbb44f512a15eb1267cfdb", - "sample1_normal.cram.crai:md5,1368f01b7e1437c072c3b2df6febe397", - "sample1_tumor.cram:md5,7db2a49e498fd4cf6d1748fbcd423750", - "sample1_tumor.cram.crai:md5,51cfed173b2ccbe7e7c6b9488ac90025", "sample1_normal.flagstat:md5,1c41ea9923945501eb7e41f83a90502d", "sample1_normal.idxstats:md5,902e503387799123ea59255e3fca172c", "sample1_normal.stats:md5,1fee9dfe18bc8b1c3f62bec6d54ae0b7", @@ -572,10 +568,6 @@ "read_qual.txt:md5,78247dfa2ea336eac0e128eba5e9eef4", "breakpoint_clusters.tsv:md5,d36a70de292ee130ef30da4a58bced18", "breakpoint_clusters_list.tsv:md5,0c0ce62e329f8de492487e8414c30a50", - "sample2_normal.cram:md5,091f7d95a3f2c56a68c044d282198afe", - "sample2_normal.cram.crai:md5,5c7d96951a4ab871d28463a46abe39c8", - "sample2_tumor.cram:md5,c3eb6626b9e47d35290252a9cfe1f974", - "sample2_tumor.cram.crai:md5,e3e355c59bfc6e96d82237348abe7590", "sample2_normal.flagstat:md5,714d0cc0c213e2640e54a16f3d0e6e7e", "sample2_normal.idxstats:md5,72eb83bb11748dc863fef1a0a5497e4b", "sample2_normal.stats:md5,b00725632868d7181d67c23b8cc1d4af", @@ -591,8 +583,6 @@ "read_qual.txt:md5,8b92ff7dc4536188be159b95525511cd", "breakpoint_clusters.tsv:md5,d36a70de292ee130ef30da4a58bced18", "breakpoint_clusters_list.tsv:md5,0c0ce62e329f8de492487e8414c30a50", - "sample3_tumor.cram:md5,9b6a5bddb20f330af72fe84767ea708e", - "sample3_tumor.cram.crai:md5,c60a2a5f8365501d7910b5acf653c4a4", "sample3_tumor.flagstat:md5,8ff32d733c62c4910bf185ef24bf27cf", "sample3_tumor.idxstats:md5,2de140e61f9e86c9c10af20dd565cc93", "sample3_tumor.stats:md5,2535ddb28ef378016f7b4cfa9042c72a", diff --git a/tests/union.nf.test.snap b/tests/union.nf.test.snap index 4c8df469..a74edaad 100644 --- a/tests/union.nf.test.snap +++ b/tests/union.nf.test.snap @@ -607,10 +607,6 @@ "sample3/vep/somatic/sample3_SOMATIC_VEP.vcf.gz_summary.html" ], [ - "sample1_normal.cram:md5,e25986c1006876c630f8027bb718f902", - "sample1_normal.cram.crai:md5,9946b4f1cd9d3b507ec4519ef8e7e5d1", - "sample1_tumor.cram:md5,7dc25c3ebaba225087db72a2abeb4ce8", - "sample1_tumor.cram.crai:md5,db9be95f0fdafcda4e8d90e27ec7128d", "sample1_normal.flagstat:md5,1c41ea9923945501eb7e41f83a90502d", "sample1_normal.idxstats:md5,902e503387799123ea59255e3fca172c", "sample1_normal.stats:md5,1fee9dfe18bc8b1c3f62bec6d54ae0b7", @@ -626,10 +622,6 @@ "read_qual.txt:md5,78247dfa2ea336eac0e128eba5e9eef4", "breakpoint_clusters.tsv:md5,d36a70de292ee130ef30da4a58bced18", "breakpoint_clusters_list.tsv:md5,0c0ce62e329f8de492487e8414c30a50", - "sample2_normal.cram:md5,86f9d4dcfe3079660a19db295aca4e1e", - "sample2_normal.cram.crai:md5,5ab4d81db94a4d4ebd74b478848e9f21", - "sample2_tumor.cram:md5,2100b57665cde32e0263219890b7fd72", - "sample2_tumor.cram.crai:md5,e1dac1585ae73065f2d80c899ff93f63", "sample2_normal.flagstat:md5,714d0cc0c213e2640e54a16f3d0e6e7e", "sample2_normal.idxstats:md5,72eb83bb11748dc863fef1a0a5497e4b", "sample2_normal.stats:md5,b00725632868d7181d67c23b8cc1d4af", @@ -645,8 +637,6 @@ "read_qual.txt:md5,8b92ff7dc4536188be159b95525511cd", "breakpoint_clusters.tsv:md5,d36a70de292ee130ef30da4a58bced18", "breakpoint_clusters_list.tsv:md5,0c0ce62e329f8de492487e8414c30a50", - "sample3_tumor.cram:md5,3b6761b6f9215baf74f28c955de48805", - "sample3_tumor.cram.crai:md5,2cbfc7f7c40d4ac97e4933abbd14e22b", "sample3_tumor.flagstat:md5,8ff32d733c62c4910bf185ef24bf27cf", "sample3_tumor.idxstats:md5,2de140e61f9e86c9c10af20dd565cc93", "sample3_tumor.stats:md5,2535ddb28ef378016f7b4cfa9042c72a", From f34209248b71e59d59ff7c5e3995ecddffaa4f42 Mon Sep 17 00:00:00 2001 From: ljwharbers Date: Mon, 28 Sep 2026 20:11:46 +0200 Subject: [PATCH 11/12] Correct the CHANGELOG and give longphase haplotag samtools under conda The CHANGELOG said ASCAT now gets the FASTA, which 77bcf54 reverted, and quoted the CRAM 3.1 size savings and a 1-3 % embed_ref cost. The shipped 3.0 + embed_ref files are 33 % (ONT), 52 % (older PacBio) and 69 % (Revio) smaller, and the embedded reference costs 1-7 %. It now also says that a resumed run realigns, since -x RG changes MINIMAP2_ALIGN in both formats. The patched LONGPHASE_HAPLOTAG pipes longphase's output through samtools view, which its container has but its conda environment did not. Co-Authored-By: Claude Opus 5.5 --- CHANGELOG.md | 2 +- modules/nf-core/longphase/haplotag/environment.yml | 1 + .../nf-core/longphase/haplotag/longphase-haplotag.diff | 10 +++++++++- nextflow_schema.json | 2 +- 4 files changed, 12 insertions(+), 3 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 99056971..f6abb45a 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -20,7 +20,7 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 ### `Changed` -- [#207](https://github.com/IntGenomicsLab/lrsomatic/pull/207) - Aligned and haplotagged reads are now CRAM 3.0 with the reference embedded instead of BAM, from `MINIMAP2_ALIGN` onwards; `bamfiles/` holds `_.cram` + `.crai`. The new `--aligned_format` parameter (`cram` by default, or `bam`) keeps BAM throughout for tools that cannot read CRAM. On chr20 of three real samples CRAM is 34 % (ONT), 56 % (older PacBio) and 73 % (Revio) smaller, with identical records. Version 3.0 because Clair3 and ClairS-TO's Verdict bundle an htslib that cannot decode 3.1 and then call nothing without failing; the embedded reference (1-3 % of the file) because ClairS, ClairS-TO, Severus and SAVANA open the reads without the FASTA. `ASCAT` and `CRAMINO_POST` now get the FASTA; `CLAIRSTO` deletes its intermediate haplotagged copy of the tumour reads; the uBAM's own `RG` tag is dropped before alignment, since CRAM keeps one read group per record and kept the basecaller's (@ljwharbers). +- [#207](https://github.com/IntGenomicsLab/lrsomatic/pull/207) - Aligned and haplotagged reads are now CRAM 3.0 with the reference embedded instead of BAM, from `MINIMAP2_ALIGN` onwards; `bamfiles/` holds `_.cram` + `.crai`. The new `--aligned_format` parameter (`cram` by default, or `bam`) keeps BAM throughout for tools that cannot read CRAM. On chr20 of three real samples CRAM is 33 % (ONT), 52 % (older PacBio) and 69 % (Revio) smaller, with identical records. Version 3.0 because Clair3 and ClairS-TO's Verdict bundle an htslib that cannot decode 3.1 and then call nothing without failing; the embedded reference (1-7 % of the file) because ClairS, ClairS-TO, Severus and SAVANA open the reads without the FASTA. `CRAMINO_POST` and `SAMTOOLS_MERGE` now get the FASTA (`ASCAT` does not: its image has no `ref.fasta` argument, and it reads the embedded reference); `CLAIRSTO` deletes its intermediate haplotagged copy of the tumour reads; the uBAM's own `RG` tag is dropped before alignment, since CRAM keeps one read group per record and kept the basecaller's. That change alters `MINIMAP2_ALIGN` in both formats, so a run resumed from an earlier version realigns every sample, also with `--aligned_format bam` (@ljwharbers). - [#201](https://github.com/IntGenomicsLab/lrsomatic/pull/201) - `CLAIRSTO` and `CLAIRSTO_VERDICT_TAG` now pull the fork image from Docker Hub: `oras://docker.io/ljwharbers/clairs-to-sif:0.5.1-verdict-chm13-c0687e8-flat` under Singularity/Apptainer and `docker.io/ljwharbers/clairs-to:0.5.1-verdict-chm13-c0687e8-flat` otherwise, instead of `ghcr.io/ljwharbers/clairs-to`. The `-cpu` SIF on ghcr failed with `PROTOCOL_ERROR` on slow links: ghcr redirects every blob download to an Azure URL that expires at the next 5-minute mark and resets a stream still open then, and Apptainer resumes neither an `oras://` nor a `docker://` download. Docker Hub's download URLs are valid for 50 minutes and only checked when the request starts. `-flat` is the same software copied into an empty image in a few layers (3.3 GB instead of 7 GB); the software and its outputs are unchanged. `docs/usage.md` describes `pullTimeout`, Docker Hub's anonymous pull limit, pre-pulling, and how to recover the remaining `oras://ghcr.io` SIFs resumably (@ljwharbers). - [#199](https://github.com/IntGenomicsLab/lrsomatic/pull/199) - `CLAIRSTO` and `CLAIRSTO_VERDICT_TAG` now run the `-cpu` rebuild of the fork image (`0.5.1-verdict-chm13-c0687e8-cpu`), which swaps PyTorch's CUDA build for the CPU build of the same version. The software is otherwise unchanged, but the Apptainer SIF drops from 6.53 GB to 3.46 GB. The old image could not be pulled on a normal VSC link: Apptainer fetches an `oras://` SIF as a single unresumable stream, and the signed blob URL ghcr redirects to expires on a 15-minute wall-clock boundary, so 6.53 GB needed 7.3 MB/s sustained and was otherwise cut mid-transfer with `PROTOCOL_ERROR` (@ljwharbers). - [#197](https://github.com/IntGenomicsLab/lrsomatic/pull/197) - `CLAIRSTO` now runs `ghcr.io/ljwharbers/clairs-to:0.5.1-verdict-chm13-c0687e8` (ClairS-TO 0.5.1) instead of `docker.io/hkubal/clairs-to:v0.4.2`: a fork that lets Verdict read its CNA resources from `--cna_resource_dir`, fixes four places where Verdict's Python port of ASCAT departed from R, and disables Verdict with a warning when its resources cannot be read. **GRCh38 results move as well as CHM13 ones.** Revert to the upstream image once HKU-BAL/ClairS-TO carries these changes. The module also selects the SIF under `-profile apptainer` and sets explicit output prefixes (@ljwharbers). diff --git a/modules/nf-core/longphase/haplotag/environment.yml b/modules/nf-core/longphase/haplotag/environment.yml index f436bdae..6d2cb363 100644 --- a/modules/nf-core/longphase/haplotag/environment.yml +++ b/modules/nf-core/longphase/haplotag/environment.yml @@ -7,3 +7,4 @@ channels: dependencies: - bioconda::htslib=1.23.1 - bioconda::longphase=2.0.1 + - bioconda::samtools=1.23.1 diff --git a/modules/nf-core/longphase/haplotag/longphase-haplotag.diff b/modules/nf-core/longphase/haplotag/longphase-haplotag.diff index cdffb54d..e18b54b6 100644 --- a/modules/nf-core/longphase/haplotag/longphase-haplotag.diff +++ b/modules/nf-core/longphase/haplotag/longphase-haplotag.diff @@ -1,5 +1,13 @@ Changes in component 'nf-core/longphase/haplotag' -'modules/nf-core/longphase/haplotag/environment.yml' is unchanged +Changes in 'longphase/haplotag/environment.yml': +--- modules/nf-core/longphase/haplotag/environment.yml ++++ modules/nf-core/longphase/haplotag/environment.yml +@@ -7,3 +7,4 @@ + dependencies: + - bioconda::htslib=1.23.1 + - bioconda::longphase=2.0.1 ++ - bioconda::samtools=1.23.1 + 'modules/nf-core/longphase/haplotag/meta.yml' is unchanged Changes in 'longphase/haplotag/main.nf': --- modules/nf-core/longphase/haplotag/main.nf diff --git a/nextflow_schema.json b/nextflow_schema.json index 49345582..079591f4 100644 --- a/nextflow_schema.json +++ b/nextflow_schema.json @@ -197,7 +197,7 @@ "default": "cram", "enum": ["cram", "bam"], "description": "File format of the aligned and haplotagged reads, in the work directory and in `bamfiles/`.", - "help_text": "`cram` writes CRAM 3.0 with the reference embedded: 33-70 % smaller than BAM on long reads, readable without the FASTA, and decoded by every tool in the pipeline. `bam` writes BAM with `.bai` indexes, for downstream tools that cannot read CRAM.", + "help_text": "`cram` writes CRAM 3.0 with the reference embedded: 33-69 % smaller than BAM on long reads, readable without the FASTA, and decoded by every tool in the pipeline. `bam` writes BAM with `.bai` indexes, for downstream tools that cannot read CRAM.", "fa_icon": "fas fa-file-archive" } } From d5b9d7b95f86150532f4af05f50c5c7954edd031 Mon Sep 17 00:00:00 2001 From: ljwharbers Date: Mon, 28 Sep 2026 20:12:12 +0200 Subject: [PATCH 12/12] Test the CRAM merge in clair_only and run deep_only on BAM clair_only is the only pipeline test with replicates, but it ran --aligned_format bam, so the default CRAM merge (SAMTOOLS_MERGE with the FASTA and version=3.0/embed_ref=1) had no test. It now runs the default. nft-bam's htsjdk decodes the embedded-reference CRAM without a FASTA, and sample4's merged reads MD5 is the one the BAM merge gave (88c8d3cf..., 7272 reads), so that snapshot entry is unchanged. deep_only takes --aligned_format bam instead. Its snapshot md5s are now identical to dev's, and the test drops from 79 to 11 min, since DeepVariant/DeepSomatic make_examples are slow on CRAM only on the thin chr19 test data. consensus and union still run both deep callers on CRAM. Snapshots regenerated with apptainer on Mindwell; the drift is bamfiles/, samtools .stats and Severus breakpoints_double.csv, whose headers name the input file. Co-Authored-By: Claude Opus 5.5 --- nf-test.config | 2 +- tests/.nftignore | 5 +-- tests/clair_only.nf.test | 29 +++++++------ tests/clair_only.nf.test.snap | 76 +++++++++++++++-------------------- tests/deep_only.nf.test | 15 ++++--- tests/deep_only.nf.test.snap | 48 +++++++++++++--------- 6 files changed, 86 insertions(+), 89 deletions(-) diff --git a/nf-test.config b/nf-test.config index d1220b10..7cbb6dd1 100644 --- a/nf-test.config +++ b/nf-test.config @@ -34,7 +34,7 @@ config { // load the necessary plugins plugins { load "nft-utils@0.0.3" - // bam() assertions for sample4's merged BAM (see tests/.nftignore) + // bam() assertions for sample4's merged CRAM (see tests/.nftignore) load "nft-bam@0.7.0" } } diff --git a/tests/.nftignore b/tests/.nftignore index a1d91854..aea9eed9 100644 --- a/tests/.nftignore +++ b/tests/.nftignore @@ -31,9 +31,6 @@ pipeline_info/*.{html,json,txt,yml} */signatures/matrices/** */signatures/assignment/** */report/*.html -# samtools merge gives sample4's colliding @PG IDs a random suffix, so this BAM's md5 differs every run (reads asserted in tests/clair_only.nf.test) -sample4/bamfiles/sample4_tumor.bam -sample4/bamfiles/sample4_tumor.bam.bai -# samtools view --threads lays CRAM containers out differently every run, so the CRAM/CRAI md5s never repeat (reads covered by the flagstat/idxstats/stats md5s) +# samtools view --threads lays CRAM containers out differently every run, so the CRAM/CRAI md5s never repeat (reads covered by the flagstat/idxstats/stats md5s; sample4's merged reads asserted in tests/clair_only.nf.test) */bamfiles/*.cram */bamfiles/*.cram.crai diff --git a/tests/clair_only.nf.test b/tests/clair_only.nf.test index 5276dd52..20395500 100644 --- a/tests/clair_only.nf.test +++ b/tests/clair_only.nf.test @@ -13,8 +13,6 @@ nextflow_pipeline { input = "https://raw.githubusercontent.com/IntGenomicsLab/test-datasets/refs/heads/main/samplesheets/test_sheet_2.csv" germline_var_keep = 'clair' somatic_var_keep = 'clair' - // The one test on BAM (the other pipeline tests run the CRAM default), incl. sample4's merge - aligned_format = 'bam' } } @@ -64,19 +62,19 @@ nextflow_pipeline { } }, - // ── BAM files exist ────────────────────────────────────────── + // ── CRAM files exist ───────────────────────────────────────── { - // Paired samples have both tumor + normal BAMs + // Paired samples have both tumor + normal CRAMs ['sample1', 'sample2'].each { s -> - assert file("$launchDir/output/${s}/bamfiles/${s}_tumor.bam").exists() - assert file("$launchDir/output/${s}/bamfiles/${s}_tumor.bam.bai").exists() - assert file("$launchDir/output/${s}/bamfiles/${s}_normal.bam").exists() - assert file("$launchDir/output/${s}/bamfiles/${s}_normal.bam.bai").exists() + assert file("$launchDir/output/${s}/bamfiles/${s}_tumor.cram").exists() + assert file("$launchDir/output/${s}/bamfiles/${s}_tumor.cram.crai").exists() + assert file("$launchDir/output/${s}/bamfiles/${s}_normal.cram").exists() + assert file("$launchDir/output/${s}/bamfiles/${s}_normal.cram.crai").exists() } - // Tumor-only samples have only tumor BAM + // Tumor-only samples have only tumor CRAM; sample4 merges two replicates, so this covers the CRAM merge ['sample3', 'sample4', 'sample5'].each { s -> - assert file("$launchDir/output/${s}/bamfiles/${s}_tumor.bam").exists() - assert file("$launchDir/output/${s}/bamfiles/${s}_tumor.bam.bai").exists() + assert file("$launchDir/output/${s}/bamfiles/${s}_tumor.cram").exists() + assert file("$launchDir/output/${s}/bamfiles/${s}_tumor.cram.crai").exists() } }, @@ -140,12 +138,13 @@ nextflow_pipeline { } }, - // ── sample4: merged BAM content ────────────────────────────── + // ── sample4: merged CRAM content ───────────────────────────── { - // sample4's merged BAM has an unstable md5 (see tests/.nftignore), so - // assert the reads instead + // sample4's merged CRAM has an unstable md5 (see tests/.nftignore), so + // assert the reads instead. htsjdk decodes it without a FASTA (embed_ref=1), + // and the value is the one the merged BAM gave under --aligned_format bam assert snapshot( - bam("$launchDir/output/sample4/bamfiles/sample4_tumor.bam", stringency: 'silent').getReadsMD5() + bam("$launchDir/output/sample4/bamfiles/sample4_tumor.cram", stringency: 'silent').getReadsMD5() ).match("sample4_merged_reads") }, diff --git a/tests/clair_only.nf.test.snap b/tests/clair_only.nf.test.snap index ec1ff735..35fcc0dc 100644 --- a/tests/clair_only.nf.test.snap +++ b/tests/clair_only.nf.test.snap @@ -3,11 +3,11 @@ "content": [ "88c8d3cf9cb49fdbc53372b2275d5f3e" ], + "timestamp": "2026-09-02T10:24:37.511974063", "meta": { "nf-test": "0.9.4", "nextflow": "26.04.3" - }, - "timestamp": "2026-09-02T10:24:37.511974063" + } }, "-profile test, clair only, extended samplesheet": { "content": [ @@ -229,10 +229,10 @@ "pipeline_info/lrsomatic_software_mqc_versions.yml", "sample1", "sample1/bamfiles", - "sample1/bamfiles/sample1_normal.bam", - "sample1/bamfiles/sample1_normal.bam.bai", - "sample1/bamfiles/sample1_tumor.bam", - "sample1/bamfiles/sample1_tumor.bam.bai", + "sample1/bamfiles/sample1_normal.cram", + "sample1/bamfiles/sample1_normal.cram.crai", + "sample1/bamfiles/sample1_tumor.cram", + "sample1/bamfiles/sample1_tumor.cram.crai", "sample1/qc", "sample1/qc/normal", "sample1/qc/normal/cramino_aln", @@ -357,10 +357,10 @@ "sample1/vep/somatic/sample1_SOMATIC_VEP.vcf.gz_summary.html", "sample2", "sample2/bamfiles", - "sample2/bamfiles/sample2_normal.bam", - "sample2/bamfiles/sample2_normal.bam.bai", - "sample2/bamfiles/sample2_tumor.bam", - "sample2/bamfiles/sample2_tumor.bam.bai", + "sample2/bamfiles/sample2_normal.cram", + "sample2/bamfiles/sample2_normal.cram.crai", + "sample2/bamfiles/sample2_tumor.cram", + "sample2/bamfiles/sample2_tumor.cram.crai", "sample2/qc", "sample2/qc/normal", "sample2/qc/normal/cramino_aln", @@ -484,8 +484,8 @@ "sample2/vep/somatic/sample2_SOMATIC_VEP.vcf.gz_summary.html", "sample3", "sample3/bamfiles", - "sample3/bamfiles/sample3_tumor.bam", - "sample3/bamfiles/sample3_tumor.bam.bai", + "sample3/bamfiles/sample3_tumor.cram", + "sample3/bamfiles/sample3_tumor.cram.crai", "sample3/qc", "sample3/qc/tumor", "sample3/qc/tumor/cramino_aln", @@ -569,8 +569,8 @@ "sample3/vep/somatic/sample3_SOMATIC_VEP.vcf.gz_summary.html", "sample4", "sample4/bamfiles", - "sample4/bamfiles/sample4_tumor.bam", - "sample4/bamfiles/sample4_tumor.bam.bai", + "sample4/bamfiles/sample4_tumor.cram", + "sample4/bamfiles/sample4_tumor.cram.crai", "sample4/qc", "sample4/qc/tumor", "sample4/qc/tumor/cramino_aln", @@ -664,8 +664,8 @@ "sample4/vep/somatic/sample4_SOMATIC_VEP.vcf.gz_summary.html", "sample5", "sample5/bamfiles", - "sample5/bamfiles/sample5_tumor.bam", - "sample5/bamfiles/sample5_tumor.bam.bai", + "sample5/bamfiles/sample5_tumor.cram", + "sample5/bamfiles/sample5_tumor.cram.crai", "sample5/qc", "sample5/qc/tumor", "sample5/qc/tumor/cramino_aln", @@ -749,90 +749,78 @@ "sample5/vep/somatic/sample5_SOMATIC_VEP.vcf.gz_summary.html" ], [ - "sample1_normal.bam:md5,772e41f7cd86c03a22afbe5ec0592a6b", - "sample1_normal.bam.bai:md5,1b501f6a11efe5d2e6f47b7f1523220b", - "sample1_tumor.bam:md5,c8315c80dc92dfb5d874aef3f5dd46fb", - "sample1_tumor.bam.bai:md5,bc35f807be4b93fc795a14d701469367", "sample1_normal.flagstat:md5,1c41ea9923945501eb7e41f83a90502d", "sample1_normal.idxstats:md5,902e503387799123ea59255e3fca172c", - "sample1_normal.stats:md5,a8b3fba9c54efbc0934d6eacc1807140", + "sample1_normal.stats:md5,1fee9dfe18bc8b1c3f62bec6d54ae0b7", "sample1_tumor.flagstat:md5,8ff32d733c62c4910bf185ef24bf27cf", "sample1_tumor.idxstats:md5,2de140e61f9e86c9c10af20dd565cc93", - "sample1_tumor.stats:md5,1c60a1d249d2e503b0678c72e851ea93", + "sample1_tumor.stats:md5,8b33272d01216f2cb0d9039d0176ee1c", "sample1_whatshap_stats.gtf:md5,eff050a68e36e778b06e0ec19435c569", "sample1_whatshap_stats.log:md5,76b73731f74fe32ef2d11f6bb0a0f71a", "sample1_whatshap_stats.tsv:md5,f566ae25b3c5a8f7e94b3d6c1b0417f8", "breakpoint_clusters.tsv:md5,d36a70de292ee130ef30da4a58bced18", "breakpoint_clusters_list.tsv:md5,0c0ce62e329f8de492487e8414c30a50", - "breakpoints_double.csv:md5,47cb0e0bbe71abdbf4f40217dfda43f9", + "breakpoints_double.csv:md5,3690b22cd4a7da90ed2f69042db45802", "read_qual.txt:md5,78247dfa2ea336eac0e128eba5e9eef4", "breakpoint_clusters.tsv:md5,d36a70de292ee130ef30da4a58bced18", "breakpoint_clusters_list.tsv:md5,0c0ce62e329f8de492487e8414c30a50", - "sample2_normal.bam:md5,3157bd11ba095a884c7951aafcfcfb1c", - "sample2_normal.bam.bai:md5,edebda44c4383173caea728acde4ac43", - "sample2_tumor.bam:md5,47b2c5f86e0493ba94ff72cea77eeae3", - "sample2_tumor.bam.bai:md5,abf2c290c815f54c2b3f8179f717d9bd", "sample2_normal.flagstat:md5,714d0cc0c213e2640e54a16f3d0e6e7e", "sample2_normal.idxstats:md5,72eb83bb11748dc863fef1a0a5497e4b", - "sample2_normal.stats:md5,20c47cb94f9ac739d69c57be6daf82c5", + "sample2_normal.stats:md5,b00725632868d7181d67c23b8cc1d4af", "sample2_tumor.flagstat:md5,4344a8745efef9cc2a017024218d61c6", "sample2_tumor.idxstats:md5,69467fc02c83a30084736aeea8b785fb", - "sample2_tumor.stats:md5,8635df10132c85a13f2d9878b7cf90a2", + "sample2_tumor.stats:md5,154024d3791a0e2b804ae0cf78ff740e", "sample2_whatshap_stats.gtf:md5,4d8f4393e3aebe4e945c0b8236cf3b3e", "sample2_whatshap_stats.log:md5,10bba7bae6dd99b989ece5e5dac7a8f9", "sample2_whatshap_stats.tsv:md5,bb46226e486af9026ab76e014624e903", "breakpoint_clusters.tsv:md5,d36a70de292ee130ef30da4a58bced18", "breakpoint_clusters_list.tsv:md5,0c0ce62e329f8de492487e8414c30a50", - "breakpoints_double.csv:md5,48baac86492026a4a7947bc708c47e6e", + "breakpoints_double.csv:md5,64b72dd7f32cc27c7138830a00695f16", "read_qual.txt:md5,8b92ff7dc4536188be159b95525511cd", "breakpoint_clusters.tsv:md5,d36a70de292ee130ef30da4a58bced18", "breakpoint_clusters_list.tsv:md5,0c0ce62e329f8de492487e8414c30a50", - "sample3_tumor.bam:md5,3c995151bdd974df5bf61d08606d54d3", - "sample3_tumor.bam.bai:md5,9d5964ef8f44127a7d9162d628e2407d", "sample3_tumor.flagstat:md5,8ff32d733c62c4910bf185ef24bf27cf", "sample3_tumor.idxstats:md5,2de140e61f9e86c9c10af20dd565cc93", - "sample3_tumor.stats:md5,ecd5ea4fee37379dd5c5ae3e89dfddda", + "sample3_tumor.stats:md5,2535ddb28ef378016f7b4cfa9042c72a", "sample3_whatshap_stats.gtf:md5,46a97067376b06b476d180709bc9e3d8", "sample3_whatshap_stats.log:md5,376254ec9c98f9ba204895e7085516ed", "sample3_whatshap_stats.tsv:md5,f7cc79156f23e884ead18e50b8434dbf", "breakpoint_clusters.tsv:md5,d36a70de292ee130ef30da4a58bced18", "breakpoint_clusters_list.tsv:md5,0c0ce62e329f8de492487e8414c30a50", - "breakpoints_double.csv:md5,56e899f85876cee082788927d0f89c5f", + "breakpoints_double.csv:md5,6ef9c12b0311d3a4a23cab394abfd389", "read_qual.txt:md5,b918430d35354dad1d7f02f21e4cd4ed", "breakpoint_clusters.tsv:md5,d36a70de292ee130ef30da4a58bced18", "breakpoint_clusters_list.tsv:md5,0c0ce62e329f8de492487e8414c30a50", "sample4_tumor.flagstat:md5,5710382ba31b23172ca19f9407f689b4", "sample4_tumor.idxstats:md5,3bf4793a1667f41f0b31d578f99e3955", - "sample4_tumor.stats:md5,b998982ea897721c529959e39b693ec6", + "sample4_tumor.stats:md5,8d6846e65e9ac5ead88ed6f1ac4ead09", "sample4_whatshap_stats.gtf:md5,9ca85642243042d3580d14a1d54b3d6a", "sample4_whatshap_stats.log:md5,b453653f5d83c4aa408862e7141f0965", "sample4_whatshap_stats.tsv:md5,8830262298440c4bd8675d1ff3f601e8", "breakpoint_clusters.tsv:md5,d36a70de292ee130ef30da4a58bced18", "breakpoint_clusters_list.tsv:md5,0c0ce62e329f8de492487e8414c30a50", - "breakpoints_double.csv:md5,e3beaa65b40a5093a7e74cecb0b2cfd2", + "breakpoints_double.csv:md5,4f67549ba79d56fb0c8141cd06cdfa99", "read_qual.txt:md5,42eb25878cc6b823a562406db56b29d7", "breakpoint_clusters.tsv:md5,d36a70de292ee130ef30da4a58bced18", "breakpoint_clusters_list.tsv:md5,0c0ce62e329f8de492487e8414c30a50", - "sample5_tumor.bam:md5,e3b7a4c9d391e9ddda187d7d6bcd1403", - "sample5_tumor.bam.bai:md5,d3ba11453e0eefda8d602e7b4883e82c", "sample5_tumor.flagstat:md5,8ff32d733c62c4910bf185ef24bf27cf", "sample5_tumor.idxstats:md5,2de140e61f9e86c9c10af20dd565cc93", - "sample5_tumor.stats:md5,3f1429e9b2379bf282299d71a8c5d22c", + "sample5_tumor.stats:md5,628088d29060c1aa8cd466e930d68ea4", "sample5_whatshap_stats.gtf:md5,d02a3be9b50b953b32b42e2fedabc0e5", "sample5_whatshap_stats.log:md5,a2e4ed9edca8609fc947041a9fead794", "sample5_whatshap_stats.tsv:md5,860ad30057b783e24dcb41ec581c8b68", "breakpoint_clusters.tsv:md5,d36a70de292ee130ef30da4a58bced18", "breakpoint_clusters_list.tsv:md5,0c0ce62e329f8de492487e8414c30a50", - "breakpoints_double.csv:md5,cc06f0726375cc45f29875b5053831e9", + "breakpoints_double.csv:md5,b12ed2cb2111ae41dd44b7e92c5ed6b9", "read_qual.txt:md5,b918430d35354dad1d7f02f21e4cd4ed", "breakpoint_clusters.tsv:md5,d36a70de292ee130ef30da4a58bced18", "breakpoint_clusters_list.tsv:md5,0c0ce62e329f8de492487e8414c30a50" ] ], + "timestamp": "2026-09-28T20:10:51.882257334", "meta": { - "nf-test": "0.9.0", - "nextflow": "26.04.6" - }, - "timestamp": "2026-09-21T11:27:45.728407577" + "nf-test": "0.9.4", + "nextflow": "26.04.3" + } } } \ No newline at end of file diff --git a/tests/deep_only.nf.test b/tests/deep_only.nf.test index 208e3d17..ffa6a6c7 100644 --- a/tests/deep_only.nf.test +++ b/tests/deep_only.nf.test @@ -12,6 +12,9 @@ nextflow_pipeline { outdir = "$outputDir" germline_var_keep = 'deepvariant' somatic_var_keep = 'deepsomatic' + // The one test on BAM (the others run the CRAM default); DeepVariant/DeepSomatic make_examples are + // 13-14x slower on CRAM on the thin chr19 test data only, so this also keeps the test short + aligned_format = 'bam' } } @@ -56,13 +59,13 @@ nextflow_pipeline { // ── BAM files ──────────────────────────────────────────────── { ['sample1', 'sample2'].each { s -> - assert file("$launchDir/output/${s}/bamfiles/${s}_tumor.cram").exists() - assert file("$launchDir/output/${s}/bamfiles/${s}_tumor.cram.crai").exists() - assert file("$launchDir/output/${s}/bamfiles/${s}_normal.cram").exists() - assert file("$launchDir/output/${s}/bamfiles/${s}_normal.cram.crai").exists() + assert file("$launchDir/output/${s}/bamfiles/${s}_tumor.bam").exists() + assert file("$launchDir/output/${s}/bamfiles/${s}_tumor.bam.bai").exists() + assert file("$launchDir/output/${s}/bamfiles/${s}_normal.bam").exists() + assert file("$launchDir/output/${s}/bamfiles/${s}_normal.bam.bai").exists() } - assert file("$launchDir/output/sample3/bamfiles/sample3_tumor.cram").exists() - assert file("$launchDir/output/sample3/bamfiles/sample3_tumor.cram.crai").exists() + assert file("$launchDir/output/sample3/bamfiles/sample3_tumor.bam").exists() + assert file("$launchDir/output/sample3/bamfiles/sample3_tumor.bam.bai").exists() }, // ── Severus SV outputs for paired samples ──────────────────── diff --git a/tests/deep_only.nf.test.snap b/tests/deep_only.nf.test.snap index d9061324..611d5881 100644 --- a/tests/deep_only.nf.test.snap +++ b/tests/deep_only.nf.test.snap @@ -219,10 +219,10 @@ "pipeline_info/lrsomatic_software_mqc_versions.yml", "sample1", "sample1/bamfiles", - "sample1/bamfiles/sample1_normal.cram", - "sample1/bamfiles/sample1_normal.cram.crai", - "sample1/bamfiles/sample1_tumor.cram", - "sample1/bamfiles/sample1_tumor.cram.crai", + "sample1/bamfiles/sample1_normal.bam", + "sample1/bamfiles/sample1_normal.bam.bai", + "sample1/bamfiles/sample1_tumor.bam", + "sample1/bamfiles/sample1_tumor.bam.bai", "sample1/qc", "sample1/qc/normal", "sample1/qc/normal/cramino_aln", @@ -345,10 +345,10 @@ "sample1/vep/somatic/sample1_SOMATIC_VEP.vcf.gz_summary.html", "sample2", "sample2/bamfiles", - "sample2/bamfiles/sample2_normal.cram", - "sample2/bamfiles/sample2_normal.cram.crai", - "sample2/bamfiles/sample2_tumor.cram", - "sample2/bamfiles/sample2_tumor.cram.crai", + "sample2/bamfiles/sample2_normal.bam", + "sample2/bamfiles/sample2_normal.bam.bai", + "sample2/bamfiles/sample2_tumor.bam", + "sample2/bamfiles/sample2_tumor.bam.bai", "sample2/qc", "sample2/qc/normal", "sample2/qc/normal/cramino_aln", @@ -470,8 +470,8 @@ "sample2/vep/somatic/sample2_SOMATIC_VEP.vcf.gz_summary.html", "sample3", "sample3/bamfiles", - "sample3/bamfiles/sample3_tumor.cram", - "sample3/bamfiles/sample3_tumor.cram.crai", + "sample3/bamfiles/sample3_tumor.bam", + "sample3/bamfiles/sample3_tumor.bam.bai", "sample3/qc", "sample3/qc/tumor", "sample3/qc/tumor/cramino_aln", @@ -552,51 +552,61 @@ "sample3/vep/somatic/sample3_SOMATIC_VEP.vcf.gz_summary.html" ], [ + "sample1_normal.bam:md5,c3eec48a64763f8a5cc7307c5fe00773", + "sample1_normal.bam.bai:md5,ddb44b5bfb0f798950d8872e9d60d784", + "sample1_tumor.bam:md5,8173a435625724d6192ff6a0e9e737bb", + "sample1_tumor.bam.bai:md5,1de9940795d293f650a22c6983b2a072", "sample1_normal.flagstat:md5,1c41ea9923945501eb7e41f83a90502d", "sample1_normal.idxstats:md5,902e503387799123ea59255e3fca172c", - "sample1_normal.stats:md5,1fee9dfe18bc8b1c3f62bec6d54ae0b7", + "sample1_normal.stats:md5,a8b3fba9c54efbc0934d6eacc1807140", "sample1_tumor.flagstat:md5,8ff32d733c62c4910bf185ef24bf27cf", "sample1_tumor.idxstats:md5,2de140e61f9e86c9c10af20dd565cc93", - "sample1_tumor.stats:md5,8b33272d01216f2cb0d9039d0176ee1c", + "sample1_tumor.stats:md5,1c60a1d249d2e503b0678c72e851ea93", "sample1_whatshap_stats.gtf:md5,e1d0e87353a5f9aed8a9ac4bf7973427", "sample1_whatshap_stats.log:md5,bd6b83a062e22cd3201523dc4c2c13e7", "sample1_whatshap_stats.tsv:md5,7a1508751cb1daa841a577ae25f55586", "breakpoint_clusters.tsv:md5,d36a70de292ee130ef30da4a58bced18", "breakpoint_clusters_list.tsv:md5,0c0ce62e329f8de492487e8414c30a50", - "breakpoints_double.csv:md5,3690b22cd4a7da90ed2f69042db45802", + "breakpoints_double.csv:md5,47cb0e0bbe71abdbf4f40217dfda43f9", "read_qual.txt:md5,78247dfa2ea336eac0e128eba5e9eef4", "breakpoint_clusters.tsv:md5,d36a70de292ee130ef30da4a58bced18", "breakpoint_clusters_list.tsv:md5,0c0ce62e329f8de492487e8414c30a50", + "sample2_normal.bam:md5,cfe03e8510475f54eceea0c0c80db501", + "sample2_normal.bam.bai:md5,02a12e347af31e04ffa589935f89e5ed", + "sample2_tumor.bam:md5,7380c09f2ee3ebe328f1a50671408209", + "sample2_tumor.bam.bai:md5,644083528fd153b24d97ef687417e089", "sample2_normal.flagstat:md5,714d0cc0c213e2640e54a16f3d0e6e7e", "sample2_normal.idxstats:md5,72eb83bb11748dc863fef1a0a5497e4b", - "sample2_normal.stats:md5,b00725632868d7181d67c23b8cc1d4af", + "sample2_normal.stats:md5,20c47cb94f9ac739d69c57be6daf82c5", "sample2_tumor.flagstat:md5,4344a8745efef9cc2a017024218d61c6", "sample2_tumor.idxstats:md5,69467fc02c83a30084736aeea8b785fb", - "sample2_tumor.stats:md5,154024d3791a0e2b804ae0cf78ff740e", + "sample2_tumor.stats:md5,8635df10132c85a13f2d9878b7cf90a2", "sample2_whatshap_stats.gtf:md5,af33281699a1d0da83fbe7eaff198d03", "sample2_whatshap_stats.log:md5,bbd9ab2ce07a009d9348a1d78bc6fc70", "sample2_whatshap_stats.tsv:md5,c65436f930c23ddbfd568532d07dce70", "breakpoint_clusters.tsv:md5,d36a70de292ee130ef30da4a58bced18", "breakpoint_clusters_list.tsv:md5,0c0ce62e329f8de492487e8414c30a50", - "breakpoints_double.csv:md5,64b72dd7f32cc27c7138830a00695f16", + "breakpoints_double.csv:md5,48baac86492026a4a7947bc708c47e6e", "read_qual.txt:md5,8b92ff7dc4536188be159b95525511cd", "breakpoint_clusters.tsv:md5,d36a70de292ee130ef30da4a58bced18", "breakpoint_clusters_list.tsv:md5,0c0ce62e329f8de492487e8414c30a50", + "sample3_tumor.bam:md5,965264ef8436cb887ad02c92d40fb50e", + "sample3_tumor.bam.bai:md5,cfc6329667a3c6c66e3c0ca0ace4c6e9", "sample3_tumor.flagstat:md5,8ff32d733c62c4910bf185ef24bf27cf", "sample3_tumor.idxstats:md5,2de140e61f9e86c9c10af20dd565cc93", - "sample3_tumor.stats:md5,2535ddb28ef378016f7b4cfa9042c72a", + "sample3_tumor.stats:md5,ecd5ea4fee37379dd5c5ae3e89dfddda", "sample3_whatshap_stats.gtf:md5,f47156e18c490ff9a4e6efd04d43acc5", "sample3_whatshap_stats.log:md5,4f7648e763004ab764143cb4f8b6499e", "sample3_whatshap_stats.tsv:md5,4cb58bb3b663aaba23da004d69adab3e", "breakpoint_clusters.tsv:md5,d36a70de292ee130ef30da4a58bced18", "breakpoint_clusters_list.tsv:md5,0c0ce62e329f8de492487e8414c30a50", - "breakpoints_double.csv:md5,6ef9c12b0311d3a4a23cab394abfd389", + "breakpoints_double.csv:md5,56e899f85876cee082788927d0f89c5f", "read_qual.txt:md5,b918430d35354dad1d7f02f21e4cd4ed", "breakpoint_clusters.tsv:md5,d36a70de292ee130ef30da4a58bced18", "breakpoint_clusters_list.tsv:md5,0c0ce62e329f8de492487e8414c30a50" ] ], - "timestamp": "2026-09-25T16:59:23.944263755", + "timestamp": "2026-09-28T18:26:06.724830269", "meta": { "nf-test": "0.9.4", "nextflow": "26.04.3"