From ea33d8b25e02e49b352b2132cf068830e67e0ef6 Mon Sep 17 00:00:00 2001 From: piplus2 Date: Wed, 23 Sep 2026 12:32:09 +0200 Subject: [PATCH 1/2] Port subworkflow prepare_genome to nf-core structure Add meta.yml and nf-tests for the local PREPARE_GENOME subworkflow. There is no prepare_genome subworkflow in nf-core/modules to install in its place. The tests cover building every index from a FASTA and a GTF, compressed inputs with a GFF3 annotation and supplied indices, a non fastq source that needs no index, and a stub run. Also: - Take source, aligner, pseudo_aligner and skip_alignment as inputs instead of reading params inside the subworkflow. - Emit salmon_index and suppa_tpm as bare paths whatever the input. An uncompressed --salmon_index or --suppa_tpm was emitted as [ [:], path ], which SALMON_QUANT and SUPPA cannot use, while the .tar.gz and .gz inputs came out as a path. - Give GFFREAD a meta id taken from the GFF file name, so the converted annotation is no longer called null.gtf. - Pass every path to the modules through file(), add the MODULE section headers and format with nextflow lint -format. Generated by Claude Opus 5.5 Co-Authored-By: Claude Opus 5.5 (1M context) --- CHANGELOG.md | 2 + main.nf | 4 + subworkflows/local/prepare_genome/main.nf | 157 +++++------ subworkflows/local/prepare_genome/meta.yml | 149 +++++++++++ .../local/prepare_genome/tests/main.nf.test | 185 +++++++++++++ .../prepare_genome/tests/main.nf.test.snap | 243 ++++++++++++++++++ 6 files changed, 666 insertions(+), 74 deletions(-) create mode 100644 subworkflows/local/prepare_genome/meta.yml create mode 100644 subworkflows/local/prepare_genome/tests/main.nf.test create mode 100644 subworkflows/local/prepare_genome/tests/main.nf.test.snap diff --git a/CHANGELOG.md b/CHANGELOG.md index 4c7a18e..331ef0b 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -64,6 +64,7 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 - #286 - Replace the local `STAR_ALIGN_IGENOMES` module with the nf-core `star/align` module, aliased and pinned in `conf/modules.config` to the STAR 2.6.1d container that reads the AWS iGenomes indices. The `--seq_center` read group tag is set in `ext.args` and now applies to every STAR alignment, not only to the iGenomes path (by @piplus2) - #287 - Replace the local `STAR_GENOMEGENERATE_IGENOMES` module with the nf-core `star/genomegenerate` module, aliased and pinned in `conf/modules.config` to the same STAR 2.6.1d container as `STAR_ALIGN_IGENOMES` (by @piplus2) - #289 - Drop the STAR 2.6.1d pin. A supplied STAR index now goes through the new `STAR_GENOMEPARAMS_UPGRADE` module, which rewrites the `genomeParameters.txt` metadata that STAR 2.7.4a and later validate and copies the binary index files through untouched, so the AWS iGenomes indices work with a single STAR version. This removes the `STAR_ALIGN_IGENOMES` and `STAR_GENOMEGENERATE_IGENOMES` aliases, the legacy `--quantTranscriptomeBan` argument and the `is_aws_igenome` flag (suggested by @pinin4fjords, done by @piplus2) +- #290 - Move the `PREPARE_GENOME` subworkflow to the nf-core subworkflow template. It takes `--source`, `--aligner`, `--pseudo_aligner` and `--skip_alignment` as inputs instead of reading `params`, and the GTF converted from `--gff` is named after the GFF file instead of `null.gtf` (by @piplus2) ### Fixed @@ -106,6 +107,7 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 - #275 - Fix missing yaml from singularity container in `GTFGENEFILTER` (reported by @srira25, fix by @piplus2) - #277 - Fix missing pyyaml from singularity container in `STRAND_JUNCTIONS`, `CLUSTERGROUPS` and `MISOPY_SETTINGS` (by @piplus2) - #281 - Fix `ISOFORMSWITCHANALYZER` writing `common_switch_consequences.pdf` next to the `results` directory instead of inside it, so it was never published (by @piplus2) +- #290 - Fix an uncompressed `--salmon_index` or `--suppa_tpm`, which reached `SALMON_QUANT` and `SUPPA` still wrapped with an empty meta map (by @piplus2) ## v1.0.5 - 2024-11-03 diff --git a/main.nf b/main.nf index b5288e6..60c0349 100644 --- a/main.nf +++ b/main.nf @@ -212,6 +212,10 @@ workflow NFCORE_RNASPLICE { params.gff_dexseq, params.suppa_tpm, params.gencode, + params.source, + params.aligner, + params.pseudo_aligner, + params.skip_alignment, ) // diff --git a/subworkflows/local/prepare_genome/main.nf b/subworkflows/local/prepare_genome/main.nf index d4c4e47..dcbd47c 100644 --- a/subworkflows/local/prepare_genome/main.nf +++ b/subworkflows/local/prepare_genome/main.nf @@ -2,58 +2,63 @@ // Uncompress and prepare reference genome files // -include { GUNZIP as GUNZIP_FASTA } from '../../../modules/nf-core/gunzip' -include { GUNZIP as GUNZIP_GTF } from '../../../modules/nf-core/gunzip' -include { GUNZIP as GUNZIP_GFF } from '../../../modules/nf-core/gunzip' -include { GUNZIP as GUNZIP_TRANSCRIPT_FASTA } from '../../../modules/nf-core/gunzip' -include { GUNZIP as GUNZIP_GFF_DEXSEQ } from '../../../modules/nf-core/gunzip' -include { GUNZIP as GUNZIP_SUPPA_TPM } from '../../../modules/nf-core/gunzip' - -include { UNTAR as UNTAR_STAR_INDEX } from '../../../modules/nf-core/untar' -include { UNTAR as UNTAR_SALMON_INDEX } from '../../../modules/nf-core/untar' - -include { SAMTOOLS_FAIDX } from '../../../modules/nf-core/samtools/faidx' -include { GFFREAD } from '../../../modules/nf-core/gffread' -include { STAR_GENOMEGENERATE } from '../../../modules/nf-core/star/genomegenerate' -include { SALMON_INDEX } from '../../../modules/nf-core/salmon/index' +include { GUNZIP as GUNZIP_FASTA } from '../../../modules/nf-core/gunzip' +include { GUNZIP as GUNZIP_GTF } from '../../../modules/nf-core/gunzip' +include { GUNZIP as GUNZIP_GFF } from '../../../modules/nf-core/gunzip' +include { GUNZIP as GUNZIP_TRANSCRIPT_FASTA } from '../../../modules/nf-core/gunzip' +include { GUNZIP as GUNZIP_GFF_DEXSEQ } from '../../../modules/nf-core/gunzip' +include { GUNZIP as GUNZIP_SUPPA_TPM } from '../../../modules/nf-core/gunzip' + +include { UNTAR as UNTAR_STAR_INDEX } from '../../../modules/nf-core/untar' +include { UNTAR as UNTAR_SALMON_INDEX } from '../../../modules/nf-core/untar' + +include { SAMTOOLS_FAIDX } from '../../../modules/nf-core/samtools/faidx' +include { GFFREAD } from '../../../modules/nf-core/gffread' +include { STAR_GENOMEGENERATE } from '../../../modules/nf-core/star/genomegenerate' +include { SALMON_INDEX } from '../../../modules/nf-core/salmon/index' include { RSEM_PREPAREREFERENCE as MAKE_TRANSCRIPTS_FASTA } from '../../../modules/nf-core/rsem/preparereference' -include { GTFGENEFILTER } from '../../../modules/local/gtfgenefilter' -include { PREPROCESS_TRANSCRIPTS_FASTA_GENCODE } from '../../../modules/local/preprocess_transcripts_fasta_gencode' -include { STAR_GENOMEPARAMS_UPGRADE } from '../../../modules/local/star_genomeparams_upgrade' +include { GTFGENEFILTER } from '../../../modules/local/gtfgenefilter' +include { PREPROCESS_TRANSCRIPTS_FASTA_GENCODE } from '../../../modules/local/preprocess_transcripts_fasta_gencode' +include { STAR_GENOMEPARAMS_UPGRADE } from '../../../modules/local/star_genomeparams_upgrade' workflow PREPARE_GENOME { take: - fasta // file: /path/to/genome.fasta - gtf // file: /path/to/genome.gtf - gff // file: /path/to/genome.gff - transcript_fasta // file: /path/to/transcript.fasta - star_index // directory: /path/to/star/index/ - salmon_index // directory: /path/to/salmon/index/ - gff_dexseq // file: /path/to/dexseq/genome.gff - suppa_tpm // file: /path/to/suppa/quant.tpm - gencode // boolean: whether gene annotation is from gencode + fasta // string: path to the genome FASTA, may be gzipped + gtf // string: path to the GTF annotation, may be gzipped + gff // string: path to the GFF3 annotation, used when no GTF is given, may be gzipped + transcript_fasta // string: path to the transcript FASTA, may be gzipped + star_index // string: path to the STAR index directory, may be a .tar.gz archive + salmon_index // string: path to the Salmon index directory, may be a .tar.gz archive + gff_dexseq // string: path to the flattened DEXSeq GFF annotation, may be gzipped + suppa_tpm // string: path to the SUPPA transcript TPM table, may be gzipped + gencode // boolean: whether the annotation is from GENCODE + source // string: type of input data [fastq, genome_bam, transcriptome_bam, salmon_results] + aligner // string: genome aligner [star, star_salmon] + pseudo_aligner // string: pseudo aligner [salmon] + skip_alignment // boolean: whether the genome alignment is skipped main: // - // Uncompress genome fasta file if required + // MODULE: GUNZIP_FASTA // + if (fasta.endsWith('.gz')) { - GUNZIP_FASTA([[:], fasta]) - ch_fasta = GUNZIP_FASTA.out.gunzip + ch_fasta = GUNZIP_FASTA([[:], file(fasta, checkIfExists: true)]).gunzip } else { ch_fasta = channel.value([[:], file(fasta, checkIfExists: true)]) } // - // Uncompress GTF annotation file or create from GFF3 if required + // MODULE: GUNZIP_GTF, GUNZIP_GFF and GFFREAD // + + // A GTF annotation is used as is, a GFF3 one is converted to GTF if (gtf) { if (gtf.endsWith('.gz')) { - GUNZIP_GTF([[:], gtf]) - ch_gtf = GUNZIP_GTF.out.gunzip + ch_gtf = GUNZIP_GTF([[:], file(gtf, checkIfExists: true)]).gunzip } else { ch_gtf = channel.value([[:], file(gtf, checkIfExists: true)]) @@ -61,29 +66,30 @@ workflow PREPARE_GENOME { } else if (gff) { if (gff.endsWith('.gz')) { - GUNZIP_GFF([[:], gff]) - ch_gff = GUNZIP_GFF.out.gunzip + ch_gff = GUNZIP_GFF([[:], file(gff, checkIfExists: true)]).gunzip } else { ch_gff = channel.value([[:], file(gff, checkIfExists: true)]) } - ch_gtf = GFFREAD(ch_gff, null).gtf + // GFFREAD names its output after `meta.id`, so give it the annotation name + ch_gtf = GFFREAD(ch_gff.map { _meta, gff_file -> [[id: gff_file.baseName], gff_file] }, []).gtf } // - // Uncompress transcript fasta file / create if required + // MODULE: GUNZIP_TRANSCRIPT_FASTA, PREPROCESS_TRANSCRIPTS_FASTA_GENCODE, GTFGENEFILTER and MAKE_TRANSCRIPTS_FASTA // + + // Without a transcript FASTA, one is built from the genome and the annotation, keeping + // only the genes on sequences present in the genome FASTA if (transcript_fasta) { if (transcript_fasta.endsWith('.gz')) { - GUNZIP_TRANSCRIPT_FASTA([[:], transcript_fasta]) - ch_transcript_fasta = GUNZIP_TRANSCRIPT_FASTA.out.gunzip + ch_transcript_fasta = GUNZIP_TRANSCRIPT_FASTA([[:], file(transcript_fasta, checkIfExists: true)]).gunzip } else { ch_transcript_fasta = channel.value([[:], file(transcript_fasta, checkIfExists: true)]) } if (gencode) { - PREPROCESS_TRANSCRIPTS_FASTA_GENCODE(ch_transcript_fasta) - ch_transcript_fasta = PREPROCESS_TRANSCRIPTS_FASTA_GENCODE.out.fasta + ch_transcript_fasta = PREPROCESS_TRANSCRIPTS_FASTA_GENCODE(ch_transcript_fasta).fasta } } else { @@ -95,21 +101,24 @@ workflow PREPARE_GENOME { } // - // Create chromosome sizes file + // MODULE: SAMTOOLS_FAIDX // + SAMTOOLS_FAIDX(ch_fasta.map { meta, fa -> [meta, fa, []] }, true) - ch_fai = SAMTOOLS_FAIDX.out.fai - ch_chrom_sizes = SAMTOOLS_FAIDX.out.sizes // - // Uncompress STAR index or generate from scratch if required + // MODULE: UNTAR_STAR_INDEX, STAR_GENOMEPARAMS_UPGRADE and STAR_GENOMEGENERATE // + ch_star_index = channel.empty() - if (params.source == 'fastq' && !params.skip_alignment && (params.aligner == 'star' || params.aligner == 'star_salmon')) { + if (source == 'fastq' && !skip_alignment && aligner in ['star', 'star_salmon']) { if (star_index) { - def ch_star_index_raw = star_index.endsWith('.tar.gz') - ? UNTAR_STAR_INDEX([[:], star_index]).untar - : channel.value([[:], file(star_index, checkIfExists: true)]) + if (star_index.endsWith('.tar.gz')) { + ch_star_index_raw = UNTAR_STAR_INDEX([[:], file(star_index, checkIfExists: true)]).untar + } + else { + ch_star_index_raw = channel.value([[:], file(star_index, checkIfExists: true)]) + } // A supplied index may have been built with STAR 2.6.x, as the AWS iGenomes ones were, // which STAR 2.7.4a and later refuse to read. `STAR_GENOMEPARAMS_UPGRADE` rewrites the @@ -122,37 +131,37 @@ workflow PREPARE_GENOME { } // - // Uncompress Salmon index or generate from scratch if required + // MODULE: UNTAR_SALMON_INDEX and SALMON_INDEX // + + // `star_salmon` quantifies the STAR transcriptome alignments, which needs no index, so an + // index is only built for the `salmon` pseudo aligner ch_salmon_index = channel.empty() - if (params.source == 'fastq' && (params.pseudo_aligner == 'salmon' || params.aligner == 'star_salmon')) { + if (source == 'fastq' && (pseudo_aligner == 'salmon' || aligner == 'star_salmon')) { if (salmon_index) { if (salmon_index.endsWith('.tar.gz')) { - ch_salmon_index = UNTAR_SALMON_INDEX([[:], salmon_index]).untar.map { _meta, index -> index } + ch_salmon_index = UNTAR_SALMON_INDEX([[:], file(salmon_index, checkIfExists: true)]).untar } else { ch_salmon_index = channel.value([[:], file(salmon_index, checkIfExists: true)]) } } - else { - if (params.pseudo_aligner == 'salmon') { - SALMON_INDEX( - ch_fasta.map { _meta, fa -> fa }, - ch_transcript_fasta.map { _meta, tr -> tr }, - ) - ch_salmon_index = SALMON_INDEX.out.index - } + else if (pseudo_aligner == 'salmon') { + ch_salmon_index = SALMON_INDEX( + ch_fasta.map { _meta, fa -> fa }, + ch_transcript_fasta.map { _meta, tr -> tr }, + ).index.map { index -> [[:], index] } } } // - // Uncompress DEXSeq GFF annotation file if required + // MODULE: GUNZIP_GFF_DEXSEQ // + ch_dexseq_gff = channel.empty() if (gff_dexseq) { if (gff_dexseq.endsWith('.gz')) { - GUNZIP_GFF_DEXSEQ([[:], gff_dexseq]) - ch_dexseq_gff = GUNZIP_GFF_DEXSEQ.out.gunzip + ch_dexseq_gff = GUNZIP_GFF_DEXSEQ([[:], file(gff_dexseq, checkIfExists: true)]).gunzip } else { ch_dexseq_gff = channel.value([[:], file(gff_dexseq, checkIfExists: true)]) @@ -160,13 +169,13 @@ workflow PREPARE_GENOME { } // - // Uncompress SUPPA TPM file if required + // MODULE: GUNZIP_SUPPA_TPM // + ch_suppa_tpm = channel.empty() if (suppa_tpm) { if (suppa_tpm.endsWith('.gz')) { - GUNZIP_SUPPA_TPM([[:], suppa_tpm]) - ch_suppa_tpm = GUNZIP_SUPPA_TPM.out.gunzip.map { _meta, tpm -> tpm } + ch_suppa_tpm = GUNZIP_SUPPA_TPM([[:], file(suppa_tpm, checkIfExists: true)]).gunzip } else { ch_suppa_tpm = channel.value([[:], file(suppa_tpm, checkIfExists: true)]) @@ -174,13 +183,13 @@ workflow PREPARE_GENOME { } emit: - fasta = ch_fasta.map { _meta, fa -> fa } // path: genome.fasta - fai = ch_fai // path: genome.fai - chrom_sizes = ch_chrom_sizes // path: genome.sizes - gtf = ch_gtf.map { _meta, out_gtf -> out_gtf } // path: genome.gtf - transcript_fasta = ch_transcript_fasta.map { _meta, fa -> fa } // path: transcript.fasta - star_index = ch_star_index.map { _meta, index -> index } // path: star/index/ - salmon_index = ch_salmon_index // path: salmon/index/ - dexseq_gff = ch_dexseq_gff.map { _meta, dexseq_gff -> dexseq_gff } // path: dexseq.gff - suppa_tpm = ch_suppa_tpm // path: suppa.tpm + fasta = ch_fasta.map { _meta, fa -> fa } // channel: path(genome.fasta) + fai = SAMTOOLS_FAIDX.out.fai // channel: [ val(meta), path(genome.fasta.fai) ] + chrom_sizes = SAMTOOLS_FAIDX.out.sizes // channel: [ val(meta), path(genome.fasta.sizes) ] + gtf = ch_gtf.map { _meta, out_gtf -> out_gtf } // channel: path(genome.gtf) + transcript_fasta = ch_transcript_fasta.map { _meta, fa -> fa } // channel: path(transcripts.fasta) + star_index = ch_star_index.map { _meta, index -> index } // channel: path(star/index/) + salmon_index = ch_salmon_index.map { _meta, index -> index } // channel: path(salmon/index/) + dexseq_gff = ch_dexseq_gff.map { _meta, dexseq_gff -> dexseq_gff } // channel: path(dexseq.gff) + suppa_tpm = ch_suppa_tpm.map { _meta, tpm -> tpm } // channel: path(suppa.tpm) } diff --git a/subworkflows/local/prepare_genome/meta.yml b/subworkflows/local/prepare_genome/meta.yml new file mode 100644 index 0000000..93788e4 --- /dev/null +++ b/subworkflows/local/prepare_genome/meta.yml @@ -0,0 +1,149 @@ +# yaml-language-server: $schema=https://raw.githubusercontent.com/nf-core/modules/master/subworkflows/yaml-schema.json +name: "prepare_genome" +description: Uncompress the reference genome files and build the genome index, transcript FASTA and aligner indices that were not supplied +keywords: + - genome + - reference + - index + - star + - salmon + - gunzip + - untar +components: + - gunzip + - untar + - samtools/faidx + - gffread + - star/genomegenerate + - salmon/index + - rsem/preparereference + - gtfgenefilter + - preprocess_transcripts_fasta_gencode + - star_genomeparams_upgrade +input: + - fasta: + type: string + description: | + Path to the genome FASTA file, gzipped or not + pattern: "*.{fa,fasta}{,.gz}" + - gtf: + type: string + description: | + Path to the GTF annotation file, gzipped or not. Takes precedence over `gff` + pattern: "*.gtf{,.gz}" + - gff: + type: string + description: | + Path to the GFF3 annotation file, gzipped or not, converted to GTF with gffread when no `gtf` is given + pattern: "*.{gff,gff3}{,.gz}" + - transcript_fasta: + type: string + description: | + Path to the transcript FASTA file, gzipped or not. When not given, it is built from the + genome FASTA and the annotation with RSEM + pattern: "*.{fa,fasta}{,.gz}" + - star_index: + type: string + description: | + Path to the STAR index directory, or a `.tar.gz` archive of it. When not given, it is + built from the genome FASTA and the annotation + - salmon_index: + type: string + description: | + Path to the Salmon index directory, or a `.tar.gz` archive of it. When not given, it is + built from the genome FASTA and the transcript FASTA for the `salmon` pseudo aligner + - gff_dexseq: + type: string + description: | + Path to the flattened DEXSeq GFF annotation, gzipped or not + pattern: "*.gff{,.gz}" + - suppa_tpm: + type: string + description: | + Path to the SUPPA transcript TPM table, gzipped or not + pattern: "*.tpm{,.gz}" + - gencode: + type: boolean + description: | + Whether the annotation is from GENCODE, in which case the transcript FASTA headers are + trimmed to the transcript ID + - source: + type: string + description: | + Type of input data. The aligner indices are only prepared for `fastq` + enum: ["fastq", "genome_bam", "transcriptome_bam", "salmon_results"] + - aligner: + type: string + description: | + Genome aligner, the STAR index is prepared for both values + enum: ["star", "star_salmon"] + - pseudo_aligner: + type: string + description: | + Pseudo aligner, the Salmon index is prepared for `salmon` + enum: ["salmon"] + - skip_alignment: + type: boolean + description: | + Whether the genome alignment is skipped, in which case no STAR index is prepared +output: + - fasta: + type: file + description: | + Channel containing the uncompressed genome FASTA file + Structure: [ path(fasta) ] + pattern: "*.{fa,fasta}" + - fai: + type: file + description: | + Channel containing the genome FASTA index + Structure: [ val(meta), path(fai) ] + pattern: "*.fai" + - chrom_sizes: + type: file + description: | + Channel containing the chromosome sizes of the genome + Structure: [ val(meta), path(sizes) ] + pattern: "*.sizes" + - gtf: + type: file + description: | + Channel containing the uncompressed GTF annotation, or the one converted from the GFF3 annotation + Structure: [ path(gtf) ] + pattern: "*.gtf" + - transcript_fasta: + type: file + description: | + Channel containing the uncompressed transcript FASTA file, or the one built with RSEM + Structure: [ path(fasta) ] + pattern: "*.{fa,fasta}" + - star_index: + type: directory + description: | + Channel containing the STAR index directory, empty when no STAR alignment is run + Structure: [ path(index) ] + - salmon_index: + type: directory + description: | + Channel containing the Salmon index directory, empty when Salmon is not run or no index is needed + Structure: [ path(index) ] + - dexseq_gff: + type: file + description: | + Channel containing the uncompressed flattened DEXSeq GFF annotation, empty when none is given + Structure: [ path(gff) ] + pattern: "*.gff" + - suppa_tpm: + type: file + description: | + Channel containing the uncompressed SUPPA transcript TPM table, empty when none is given + Structure: [ path(tpm) ] + pattern: "*.tpm" +authors: + - "@asmaali98" + - "@bensouthgate" + - "@jma1991" + - "@valentinoruggieri" + - "@piplus2" +maintainers: + - "@piplus2" diff --git a/subworkflows/local/prepare_genome/tests/main.nf.test b/subworkflows/local/prepare_genome/tests/main.nf.test new file mode 100644 index 0000000..20ad445 --- /dev/null +++ b/subworkflows/local/prepare_genome/tests/main.nf.test @@ -0,0 +1,185 @@ +// nf-core subworkflows test prepare_genome +nextflow_workflow { + + name "Test Subworkflow PREPARE_GENOME" + script "../main.nf" + workflow "PREPARE_GENOME" + + tag "subworkflows" + tag "subworkflows/prepare_genome" + tag "gunzip" + tag "untar" + tag "samtools" + tag "samtools/faidx" + tag "gffread" + tag "star" + tag "star/genomegenerate" + tag "salmon" + tag "salmon/index" + tag "rsem" + tag "rsem/preparereference" + tag "gtfgenefilter" + tag "preprocess_transcripts_fasta_gencode" + tag "star_genomeparams_upgrade" + + test("homo_sapiens - fasta and gtf - build indices") { + + when { + workflow { + """ + input[0] = params.modules_testdata_base_path + 'genomics/homo_sapiens/genome/genome.fasta' + input[1] = params.modules_testdata_base_path + 'genomics/homo_sapiens/genome/genome.gtf' + input[2] = null + input[3] = null + input[4] = null + input[5] = null + input[6] = null + input[7] = null + input[8] = false + input[9] = 'fastq' + input[10] = 'star' + input[11] = 'salmon' + input[12] = false + """ + } + } + + then { + assertAll( + { assert workflow.success }, + { assert snapshot( + workflow.out.fasta, + workflow.out.fai, + workflow.out.chrom_sizes, + workflow.out.gtf, + workflow.out.transcript_fasta, + workflow.out.star_index.collect { index -> file(index).list().sort() }, + workflow.out.salmon_index.collect { index -> file(index).name }, + workflow.out.dexseq_gff, + workflow.out.suppa_tpm + ).match() } + ) + } + + } + + test("homo_sapiens - compressed inputs, gff and supplied indices") { + + when { + workflow { + """ + // No test dataset carries an uncompressed Salmon index or a SUPPA TPM table, and + // the subworkflow only passes them through, so small stand-ins are written here + def salmon_index = file("${outputDir}/salmon_index") + salmon_index.mkdirs() + file("${outputDir}/salmon_index/versionInfo.json").text = '{}\\n' + file("${outputDir}/quant.tpm").text = 'sample1\\ttranscript1\\t1.0\\n' + + input[0] = params.modules_testdata_base_path + 'genomics/homo_sapiens/genome/genome.fasta.gz' + input[1] = null + input[2] = params.pipelines_testdata_base_path + 'reference/genes_chrX.gff3' + input[3] = params.modules_testdata_base_path + 'genomics/homo_sapiens/genome/transcriptome.fasta' + input[4] = params.modules_testdata_base_path + 'genomics/homo_sapiens/genome/index/star/star.tar.gz' + input[5] = "${outputDir}/salmon_index" + input[6] = params.pipelines_testdata_base_path + 'testdata/dexseq/genes_chrX.DEXSeq.gff.gz' + input[7] = "${outputDir}/quant.tpm" + input[8] = true + input[9] = 'fastq' + input[10] = 'star_salmon' + input[11] = 'salmon' + input[12] = false + """ + } + } + + then { + assertAll( + { assert workflow.success }, + { assert snapshot( + workflow.out.fasta, + workflow.out.fai, + workflow.out.chrom_sizes, + workflow.out.gtf, + workflow.out.transcript_fasta, + workflow.out.star_index.collect { index -> file(index).list().sort() }, + workflow.out.salmon_index.collect { index -> file(index).name }, + workflow.out.dexseq_gff, + workflow.out.suppa_tpm + ).match() } + ) + } + + } + + test("homo_sapiens - genome_bam source - no indices") { + + when { + workflow { + """ + input[0] = params.modules_testdata_base_path + 'genomics/homo_sapiens/genome/genome.fasta' + input[1] = params.modules_testdata_base_path + 'genomics/homo_sapiens/genome/genome.gtf' + input[2] = null + input[3] = params.modules_testdata_base_path + 'genomics/homo_sapiens/genome/transcriptome.fasta' + input[4] = params.modules_testdata_base_path + 'genomics/homo_sapiens/genome/index/star/star.tar.gz' + input[5] = null + input[6] = null + input[7] = null + input[8] = false + input[9] = 'genome_bam' + input[10] = 'star_salmon' + input[11] = 'salmon' + input[12] = false + """ + } + } + + then { + assertAll( + { assert workflow.success }, + { assert workflow.out.star_index.size() == 0 }, + { assert workflow.out.salmon_index.size() == 0 }, + { assert snapshot( + workflow.out.fasta, + workflow.out.fai, + workflow.out.chrom_sizes, + workflow.out.gtf, + workflow.out.transcript_fasta + ).match() } + ) + } + + } + + test("homo_sapiens - fasta and gtf - build indices - stub") { + + options "-stub" + + when { + workflow { + """ + input[0] = params.modules_testdata_base_path + 'genomics/homo_sapiens/genome/genome.fasta' + input[1] = params.modules_testdata_base_path + 'genomics/homo_sapiens/genome/genome.gtf' + input[2] = null + input[3] = null + input[4] = null + input[5] = null + input[6] = null + input[7] = null + input[8] = false + input[9] = 'fastq' + input[10] = 'star' + input[11] = 'salmon' + input[12] = false + """ + } + } + + then { + assertAll( + { assert workflow.success }, + { assert snapshot(sanitizeOutput(workflow.out)).match() } + ) + } + + } +} diff --git a/subworkflows/local/prepare_genome/tests/main.nf.test.snap b/subworkflows/local/prepare_genome/tests/main.nf.test.snap new file mode 100644 index 0000000..4d655db --- /dev/null +++ b/subworkflows/local/prepare_genome/tests/main.nf.test.snap @@ -0,0 +1,243 @@ +{ + "homo_sapiens - fasta and gtf - build indices": { + "content": [ + [ + "/nf-core/test-datasets/modules/data/genomics/homo_sapiens/genome/genome.fasta" + ], + [ + [ + { + + }, + "genome.fasta.fai:md5,3520cd30e1b100e55f578db9c855f685" + ] + ], + [ + [ + { + + }, + "genome.fasta.sizes:md5,b190587cae0531f3cf25552d8aa674db" + ] + ], + [ + "/nf-core/test-datasets/modules/data/genomics/homo_sapiens/genome/genome.gtf" + ], + [ + "genome.transcripts.fa:md5,050c521a2719c2ae48267c1e65218f29" + ], + [ + [ + "Genome", + "Log.out", + "SA", + "SAindex", + "chrLength.txt", + "chrName.txt", + "chrNameLength.txt", + "chrStart.txt", + "exonGeTrInfo.tab", + "exonInfo.tab", + "geneInfo.tab", + "genomeParameters.txt", + "sjdbInfo.txt", + "sjdbList.fromGTF.out.tab", + "sjdbList.out.tab", + "transcriptInfo.tab" + ] + ], + [ + "salmon" + ], + [ + + ], + [ + + ] + ], + "timestamp": "2026-09-23T12:30:21.989618264", + "meta": { + "nf-test": "0.9.5", + "nextflow": "26.04.1" + } + }, + "homo_sapiens - compressed inputs, gff and supplied indices": { + "content": [ + [ + "genome.fasta:md5,f315020d899597c1b57e5fe9f60f4c3e" + ], + [ + [ + { + + }, + "genome.fasta.fai:md5,3520cd30e1b100e55f578db9c855f685" + ] + ], + [ + [ + { + + }, + "genome.fasta.sizes:md5,b190587cae0531f3cf25552d8aa674db" + ] + ], + [ + "genes_chrX.gtf:md5,0624e70c0b23bae787b4bfd78cb331dd" + ], + [ + "transcriptome.fixed.fa:md5,4e04341f8781429444f53f69beb31392" + ], + [ + [ + "Genome", + "Log.out", + "SA", + "SAindex", + "chrLength.txt", + "chrName.txt", + "chrNameLength.txt", + "chrStart.txt", + "exonGeTrInfo.tab", + "exonInfo.tab", + "geneInfo.tab", + "genomeParameters.txt", + "sjdbInfo.txt", + "sjdbList.fromGTF.out.tab", + "sjdbList.out.tab", + "transcriptInfo.tab" + ] + ], + [ + "salmon_index" + ], + [ + "genes_chrX.DEXSeq.gff:md5,ee8b662ee32c6c563802690491ec0633" + ], + [ + "quant.tpm:md5,d606cb7c509c0f2d4287cd83243201d9" + ] + ], + "timestamp": "2026-09-23T12:30:30.786874171", + "meta": { + "nf-test": "0.9.5", + "nextflow": "26.04.1" + } + }, + "homo_sapiens - fasta and gtf - build indices - stub": { + "content": [ + { + "chrom_sizes": [ + [ + { + + }, + "genome.fasta.sizes:md5,d41d8cd98f00b204e9800998ecf8427e" + ] + ], + "dexseq_gff": [ + + ], + "fai": [ + [ + { + + }, + "genome.fasta.fai:md5,d41d8cd98f00b204e9800998ecf8427e" + ] + ], + "fasta": [ + "/nf-core/test-datasets/modules/data/genomics/homo_sapiens/genome/genome.fasta" + ], + "gtf": [ + "/nf-core/test-datasets/modules/data/genomics/homo_sapiens/genome/genome.gtf" + ], + "salmon_index": [ + [ + "complete_ref_lens.bin:md5,d41d8cd98f00b204e9800998ecf8427e", + "ctable.bin:md5,d41d8cd98f00b204e9800998ecf8427e", + "ctg_offsets.bin:md5,d41d8cd98f00b204e9800998ecf8427e", + "duplicate_clusters.tsv:md5,d41d8cd98f00b204e9800998ecf8427e", + "info.json:md5,d41d8cd98f00b204e9800998ecf8427e", + "mphf.bin:md5,d41d8cd98f00b204e9800998ecf8427e", + "pos.bin:md5,d41d8cd98f00b204e9800998ecf8427e", + "pre_indexing.log:md5,d41d8cd98f00b204e9800998ecf8427e", + "rank.bin:md5,d41d8cd98f00b204e9800998ecf8427e", + "refAccumLengths.bin:md5,d41d8cd98f00b204e9800998ecf8427e", + "ref_indexing.log:md5,d41d8cd98f00b204e9800998ecf8427e", + "reflengths.bin:md5,d41d8cd98f00b204e9800998ecf8427e", + "refseq.bin:md5,d41d8cd98f00b204e9800998ecf8427e", + "seq.bin:md5,d41d8cd98f00b204e9800998ecf8427e", + "versionInfo.json:md5,d41d8cd98f00b204e9800998ecf8427e" + ] + ], + "star_index": [ + [ + "Genome:md5,d41d8cd98f00b204e9800998ecf8427e", + "Log.out:md5,d41d8cd98f00b204e9800998ecf8427e", + "SA:md5,d41d8cd98f00b204e9800998ecf8427e", + "SAindex:md5,d41d8cd98f00b204e9800998ecf8427e", + "chrLength.txt:md5,d41d8cd98f00b204e9800998ecf8427e", + "chrName.txt:md5,d41d8cd98f00b204e9800998ecf8427e", + "chrNameLength.txt:md5,d41d8cd98f00b204e9800998ecf8427e", + "chrStart.txt:md5,d41d8cd98f00b204e9800998ecf8427e", + "exonGeTrInfo.tab:md5,d41d8cd98f00b204e9800998ecf8427e", + "exonInfo.tab:md5,d41d8cd98f00b204e9800998ecf8427e", + "geneInfo.tab:md5,d41d8cd98f00b204e9800998ecf8427e", + "genomeParameters.txt:md5,d41d8cd98f00b204e9800998ecf8427e", + "sjdbInfo.txt:md5,d41d8cd98f00b204e9800998ecf8427e", + "sjdbList.fromGTF.out.tab:md5,d41d8cd98f00b204e9800998ecf8427e", + "sjdbList.out.tab:md5,d41d8cd98f00b204e9800998ecf8427e", + "transcriptInfo.tab:md5,d41d8cd98f00b204e9800998ecf8427e" + ] + ], + "suppa_tpm": [ + + ], + "transcript_fasta": [ + "genome.transcripts.fa:md5,d41d8cd98f00b204e9800998ecf8427e" + ] + } + ], + "timestamp": "2026-09-23T12:30:44.947864209", + "meta": { + "nf-test": "0.9.5", + "nextflow": "26.04.1" + } + }, + "homo_sapiens - genome_bam source - no indices": { + "content": [ + [ + "/nf-core/test-datasets/modules/data/genomics/homo_sapiens/genome/genome.fasta" + ], + [ + [ + { + + }, + "genome.fasta.fai:md5,3520cd30e1b100e55f578db9c855f685" + ] + ], + [ + [ + { + + }, + "genome.fasta.sizes:md5,b190587cae0531f3cf25552d8aa674db" + ] + ], + [ + "/nf-core/test-datasets/modules/data/genomics/homo_sapiens/genome/genome.gtf" + ], + [ + "/nf-core/test-datasets/modules/data/genomics/homo_sapiens/genome/transcriptome.fasta" + ] + ], + "timestamp": "2026-09-23T12:30:37.262506689", + "meta": { + "nf-test": "0.9.5", + "nextflow": "26.04.1" + } + } +} \ No newline at end of file From d4dd27d6b3c8ad7918f9f6775a01edef4fd55dae Mon Sep 17 00:00:00 2001 From: piplus2 Date: Wed, 23 Sep 2026 16:49:15 +0200 Subject: [PATCH 2/2] Cover every gunzip branch in the prepare_genome tests Review feedback on #290. The tests only ran GUNZIP_FASTA and GUNZIP_GFF_DEXSEQ, so a wrong variable in the GTF, GFF3, transcript FASTA or SUPPA TPM branch would have gone unnoticed. The test datasets carry no gzipped copy of those files, so the tests gzip them on the fly: the second test now takes a gzipped GFF3, transcript FASTA and TPM table, and the genome_bam test a gzipped GTF and an uncompressed TPM table, which keeps the uncompressed --suppa_tpm fix covered. The gzipped GFF3 also covers the GFFREAD naming fix on the path that goes through GUNZIP_GFF. The stand-in inputs are written to workDir instead of outputDir, so the output directory holds only outputs. Generated by Claude Opus 5.5 Co-Authored-By: Claude Opus 5.5 (1M context) Claude-Session: https://claude.ai/code/session_013jvD13HCjdAWqMnRvEzVH6 --- .../local/prepare_genome/tests/main.nf.test | 46 ++++-- .../prepare_genome/tests/main.nf.test.snap | 135 +++++++++--------- 2 files changed, 100 insertions(+), 81 deletions(-) diff --git a/subworkflows/local/prepare_genome/tests/main.nf.test b/subworkflows/local/prepare_genome/tests/main.nf.test index 20ad445..aa434ae 100644 --- a/subworkflows/local/prepare_genome/tests/main.nf.test +++ b/subworkflows/local/prepare_genome/tests/main.nf.test @@ -63,26 +63,32 @@ nextflow_workflow { } - test("homo_sapiens - compressed inputs, gff and supplied indices") { + test("homo_sapiens - gzipped inputs, gff and supplied indices") { when { workflow { """ - // No test dataset carries an uncompressed Salmon index or a SUPPA TPM table, and - // the subworkflow only passes them through, so small stand-ins are written here - def salmon_index = file("${outputDir}/salmon_index") - salmon_index.mkdirs() - file("${outputDir}/salmon_index/versionInfo.json").text = '{}\\n' - file("${outputDir}/quant.tpm").text = 'sample1\\ttranscript1\\t1.0\\n' + // The test datasets carry no gzipped GFF3, transcript FASTA or SUPPA TPM table, nor + // an uncompressed Salmon index, so they are written here. The Salmon index and the + // TPM table are only passed through, so small stand-ins do + def gzip = { src, dest -> + def out = new java.util.zip.GZIPOutputStream(file(dest).newOutputStream()) + out.write(file(src).bytes) + out.close() + dest + } + file("${workDir}/salmon_index").mkdirs() + file("${workDir}/salmon_index/versionInfo.json").text = '{}\\n' + file("${workDir}/quant.tpm").text = 'sample1\\ttranscript1\\t1.0\\n' input[0] = params.modules_testdata_base_path + 'genomics/homo_sapiens/genome/genome.fasta.gz' input[1] = null - input[2] = params.pipelines_testdata_base_path + 'reference/genes_chrX.gff3' - input[3] = params.modules_testdata_base_path + 'genomics/homo_sapiens/genome/transcriptome.fasta' + input[2] = gzip.call(params.pipelines_testdata_base_path + 'reference/genes_chrX.gff3', "${workDir}/genes_chrX.gff3.gz") + input[3] = gzip.call(params.modules_testdata_base_path + 'genomics/homo_sapiens/genome/transcriptome.fasta', "${workDir}/transcriptome.fasta.gz") input[4] = params.modules_testdata_base_path + 'genomics/homo_sapiens/genome/index/star/star.tar.gz' - input[5] = "${outputDir}/salmon_index" + input[5] = "${workDir}/salmon_index" input[6] = params.pipelines_testdata_base_path + 'testdata/dexseq/genes_chrX.DEXSeq.gff.gz' - input[7] = "${outputDir}/quant.tpm" + input[7] = gzip.call("${workDir}/quant.tpm", "${workDir}/quant.tpm.gz") input[8] = true input[9] = 'fastq' input[10] = 'star_salmon' @@ -111,19 +117,28 @@ nextflow_workflow { } - test("homo_sapiens - genome_bam source - no indices") { + test("homo_sapiens - genome_bam source - gzipped gtf - no indices") { when { workflow { """ + // The test datasets carry no gzipped GTF or SUPPA TPM table, so they are written here + def gzip = { src, dest -> + def out = new java.util.zip.GZIPOutputStream(file(dest).newOutputStream()) + out.write(file(src).bytes) + out.close() + dest + } + file("${workDir}/quant.tpm").text = 'sample1\\ttranscript1\\t1.0\\n' + input[0] = params.modules_testdata_base_path + 'genomics/homo_sapiens/genome/genome.fasta' - input[1] = params.modules_testdata_base_path + 'genomics/homo_sapiens/genome/genome.gtf' + input[1] = gzip.call(params.modules_testdata_base_path + 'genomics/homo_sapiens/genome/genome.gtf', "${workDir}/genome.gtf.gz") input[2] = null input[3] = params.modules_testdata_base_path + 'genomics/homo_sapiens/genome/transcriptome.fasta' input[4] = params.modules_testdata_base_path + 'genomics/homo_sapiens/genome/index/star/star.tar.gz' input[5] = null input[6] = null - input[7] = null + input[7] = "${workDir}/quant.tpm" input[8] = false input[9] = 'genome_bam' input[10] = 'star_salmon' @@ -143,7 +158,8 @@ nextflow_workflow { workflow.out.fai, workflow.out.chrom_sizes, workflow.out.gtf, - workflow.out.transcript_fasta + workflow.out.transcript_fasta, + workflow.out.suppa_tpm ).match() } ) } diff --git a/subworkflows/local/prepare_genome/tests/main.nf.test.snap b/subworkflows/local/prepare_genome/tests/main.nf.test.snap index 4d655db..8bbe10b 100644 --- a/subworkflows/local/prepare_genome/tests/main.nf.test.snap +++ b/subworkflows/local/prepare_genome/tests/main.nf.test.snap @@ -62,69 +62,6 @@ "nextflow": "26.04.1" } }, - "homo_sapiens - compressed inputs, gff and supplied indices": { - "content": [ - [ - "genome.fasta:md5,f315020d899597c1b57e5fe9f60f4c3e" - ], - [ - [ - { - - }, - "genome.fasta.fai:md5,3520cd30e1b100e55f578db9c855f685" - ] - ], - [ - [ - { - - }, - "genome.fasta.sizes:md5,b190587cae0531f3cf25552d8aa674db" - ] - ], - [ - "genes_chrX.gtf:md5,0624e70c0b23bae787b4bfd78cb331dd" - ], - [ - "transcriptome.fixed.fa:md5,4e04341f8781429444f53f69beb31392" - ], - [ - [ - "Genome", - "Log.out", - "SA", - "SAindex", - "chrLength.txt", - "chrName.txt", - "chrNameLength.txt", - "chrStart.txt", - "exonGeTrInfo.tab", - "exonInfo.tab", - "geneInfo.tab", - "genomeParameters.txt", - "sjdbInfo.txt", - "sjdbList.fromGTF.out.tab", - "sjdbList.out.tab", - "transcriptInfo.tab" - ] - ], - [ - "salmon_index" - ], - [ - "genes_chrX.DEXSeq.gff:md5,ee8b662ee32c6c563802690491ec0633" - ], - [ - "quant.tpm:md5,d606cb7c509c0f2d4287cd83243201d9" - ] - ], - "timestamp": "2026-09-23T12:30:30.786874171", - "meta": { - "nf-test": "0.9.5", - "nextflow": "26.04.1" - } - }, "homo_sapiens - fasta and gtf - build indices - stub": { "content": [ { @@ -206,7 +143,70 @@ "nextflow": "26.04.1" } }, - "homo_sapiens - genome_bam source - no indices": { + "homo_sapiens - gzipped inputs, gff and supplied indices": { + "content": [ + [ + "genome.fasta:md5,f315020d899597c1b57e5fe9f60f4c3e" + ], + [ + [ + { + + }, + "genome.fasta.fai:md5,3520cd30e1b100e55f578db9c855f685" + ] + ], + [ + [ + { + + }, + "genome.fasta.sizes:md5,b190587cae0531f3cf25552d8aa674db" + ] + ], + [ + "genes_chrX.gtf:md5,0624e70c0b23bae787b4bfd78cb331dd" + ], + [ + "transcriptome.fixed.fa:md5,4e04341f8781429444f53f69beb31392" + ], + [ + [ + "Genome", + "Log.out", + "SA", + "SAindex", + "chrLength.txt", + "chrName.txt", + "chrNameLength.txt", + "chrStart.txt", + "exonGeTrInfo.tab", + "exonInfo.tab", + "geneInfo.tab", + "genomeParameters.txt", + "sjdbInfo.txt", + "sjdbList.fromGTF.out.tab", + "sjdbList.out.tab", + "transcriptInfo.tab" + ] + ], + [ + "salmon_index" + ], + [ + "genes_chrX.DEXSeq.gff:md5,ee8b662ee32c6c563802690491ec0633" + ], + [ + "quant.tpm:md5,d606cb7c509c0f2d4287cd83243201d9" + ] + ], + "timestamp": "2026-09-23T16:48:37.469824792", + "meta": { + "nf-test": "0.9.5", + "nextflow": "26.04.1" + } + }, + "homo_sapiens - genome_bam source - gzipped gtf - no indices": { "content": [ [ "/nf-core/test-datasets/modules/data/genomics/homo_sapiens/genome/genome.fasta" @@ -228,13 +228,16 @@ ] ], [ - "/nf-core/test-datasets/modules/data/genomics/homo_sapiens/genome/genome.gtf" + "genome.gtf:md5,50fc877b1c53b36b3b413aff88bda48c" ], [ "/nf-core/test-datasets/modules/data/genomics/homo_sapiens/genome/transcriptome.fasta" + ], + [ + "quant.tpm:md5,d606cb7c509c0f2d4287cd83243201d9" ] ], - "timestamp": "2026-09-23T12:30:37.262506689", + "timestamp": "2026-09-23T16:48:43.731886703", "meta": { "nf-test": "0.9.5", "nextflow": "26.04.1"