diff --git a/CHANGELOG.md b/CHANGELOG.md index 4c7a18e..331ef0b 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -64,6 +64,7 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 - #286 - Replace the local `STAR_ALIGN_IGENOMES` module with the nf-core `star/align` module, aliased and pinned in `conf/modules.config` to the STAR 2.6.1d container that reads the AWS iGenomes indices. The `--seq_center` read group tag is set in `ext.args` and now applies to every STAR alignment, not only to the iGenomes path (by @piplus2) - #287 - Replace the local `STAR_GENOMEGENERATE_IGENOMES` module with the nf-core `star/genomegenerate` module, aliased and pinned in `conf/modules.config` to the same STAR 2.6.1d container as `STAR_ALIGN_IGENOMES` (by @piplus2) - #289 - Drop the STAR 2.6.1d pin. A supplied STAR index now goes through the new `STAR_GENOMEPARAMS_UPGRADE` module, which rewrites the `genomeParameters.txt` metadata that STAR 2.7.4a and later validate and copies the binary index files through untouched, so the AWS iGenomes indices work with a single STAR version. This removes the `STAR_ALIGN_IGENOMES` and `STAR_GENOMEGENERATE_IGENOMES` aliases, the legacy `--quantTranscriptomeBan` argument and the `is_aws_igenome` flag (suggested by @pinin4fjords, done by @piplus2) +- #290 - Move the `PREPARE_GENOME` subworkflow to the nf-core subworkflow template. It takes `--source`, `--aligner`, `--pseudo_aligner` and `--skip_alignment` as inputs instead of reading `params`, and the GTF converted from `--gff` is named after the GFF file instead of `null.gtf` (by @piplus2) ### Fixed @@ -106,6 +107,7 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 - #275 - Fix missing yaml from singularity container in `GTFGENEFILTER` (reported by @srira25, fix by @piplus2) - #277 - Fix missing pyyaml from singularity container in `STRAND_JUNCTIONS`, `CLUSTERGROUPS` and `MISOPY_SETTINGS` (by @piplus2) - #281 - Fix `ISOFORMSWITCHANALYZER` writing `common_switch_consequences.pdf` next to the `results` directory instead of inside it, so it was never published (by @piplus2) +- #290 - Fix an uncompressed `--salmon_index` or `--suppa_tpm`, which reached `SALMON_QUANT` and `SUPPA` still wrapped with an empty meta map (by @piplus2) ## v1.0.5 - 2024-11-03 diff --git a/main.nf b/main.nf index b5288e6..60c0349 100644 --- a/main.nf +++ b/main.nf @@ -212,6 +212,10 @@ workflow NFCORE_RNASPLICE { params.gff_dexseq, params.suppa_tpm, params.gencode, + params.source, + params.aligner, + params.pseudo_aligner, + params.skip_alignment, ) // diff --git a/subworkflows/local/prepare_genome/main.nf b/subworkflows/local/prepare_genome/main.nf index d4c4e47..dcbd47c 100644 --- a/subworkflows/local/prepare_genome/main.nf +++ b/subworkflows/local/prepare_genome/main.nf @@ -2,58 +2,63 @@ // Uncompress and prepare reference genome files // -include { GUNZIP as GUNZIP_FASTA } from '../../../modules/nf-core/gunzip' -include { GUNZIP as GUNZIP_GTF } from '../../../modules/nf-core/gunzip' -include { GUNZIP as GUNZIP_GFF } from '../../../modules/nf-core/gunzip' -include { GUNZIP as GUNZIP_TRANSCRIPT_FASTA } from '../../../modules/nf-core/gunzip' -include { GUNZIP as GUNZIP_GFF_DEXSEQ } from '../../../modules/nf-core/gunzip' -include { GUNZIP as GUNZIP_SUPPA_TPM } from '../../../modules/nf-core/gunzip' - -include { UNTAR as UNTAR_STAR_INDEX } from '../../../modules/nf-core/untar' -include { UNTAR as UNTAR_SALMON_INDEX } from '../../../modules/nf-core/untar' - -include { SAMTOOLS_FAIDX } from '../../../modules/nf-core/samtools/faidx' -include { GFFREAD } from '../../../modules/nf-core/gffread' -include { STAR_GENOMEGENERATE } from '../../../modules/nf-core/star/genomegenerate' -include { SALMON_INDEX } from '../../../modules/nf-core/salmon/index' +include { GUNZIP as GUNZIP_FASTA } from '../../../modules/nf-core/gunzip' +include { GUNZIP as GUNZIP_GTF } from '../../../modules/nf-core/gunzip' +include { GUNZIP as GUNZIP_GFF } from '../../../modules/nf-core/gunzip' +include { GUNZIP as GUNZIP_TRANSCRIPT_FASTA } from '../../../modules/nf-core/gunzip' +include { GUNZIP as GUNZIP_GFF_DEXSEQ } from '../../../modules/nf-core/gunzip' +include { GUNZIP as GUNZIP_SUPPA_TPM } from '../../../modules/nf-core/gunzip' + +include { UNTAR as UNTAR_STAR_INDEX } from '../../../modules/nf-core/untar' +include { UNTAR as UNTAR_SALMON_INDEX } from '../../../modules/nf-core/untar' + +include { SAMTOOLS_FAIDX } from '../../../modules/nf-core/samtools/faidx' +include { GFFREAD } from '../../../modules/nf-core/gffread' +include { STAR_GENOMEGENERATE } from '../../../modules/nf-core/star/genomegenerate' +include { SALMON_INDEX } from '../../../modules/nf-core/salmon/index' include { RSEM_PREPAREREFERENCE as MAKE_TRANSCRIPTS_FASTA } from '../../../modules/nf-core/rsem/preparereference' -include { GTFGENEFILTER } from '../../../modules/local/gtfgenefilter' -include { PREPROCESS_TRANSCRIPTS_FASTA_GENCODE } from '../../../modules/local/preprocess_transcripts_fasta_gencode' -include { STAR_GENOMEPARAMS_UPGRADE } from '../../../modules/local/star_genomeparams_upgrade' +include { GTFGENEFILTER } from '../../../modules/local/gtfgenefilter' +include { PREPROCESS_TRANSCRIPTS_FASTA_GENCODE } from '../../../modules/local/preprocess_transcripts_fasta_gencode' +include { STAR_GENOMEPARAMS_UPGRADE } from '../../../modules/local/star_genomeparams_upgrade' workflow PREPARE_GENOME { take: - fasta // file: /path/to/genome.fasta - gtf // file: /path/to/genome.gtf - gff // file: /path/to/genome.gff - transcript_fasta // file: /path/to/transcript.fasta - star_index // directory: /path/to/star/index/ - salmon_index // directory: /path/to/salmon/index/ - gff_dexseq // file: /path/to/dexseq/genome.gff - suppa_tpm // file: /path/to/suppa/quant.tpm - gencode // boolean: whether gene annotation is from gencode + fasta // string: path to the genome FASTA, may be gzipped + gtf // string: path to the GTF annotation, may be gzipped + gff // string: path to the GFF3 annotation, used when no GTF is given, may be gzipped + transcript_fasta // string: path to the transcript FASTA, may be gzipped + star_index // string: path to the STAR index directory, may be a .tar.gz archive + salmon_index // string: path to the Salmon index directory, may be a .tar.gz archive + gff_dexseq // string: path to the flattened DEXSeq GFF annotation, may be gzipped + suppa_tpm // string: path to the SUPPA transcript TPM table, may be gzipped + gencode // boolean: whether the annotation is from GENCODE + source // string: type of input data [fastq, genome_bam, transcriptome_bam, salmon_results] + aligner // string: genome aligner [star, star_salmon] + pseudo_aligner // string: pseudo aligner [salmon] + skip_alignment // boolean: whether the genome alignment is skipped main: // - // Uncompress genome fasta file if required + // MODULE: GUNZIP_FASTA // + if (fasta.endsWith('.gz')) { - GUNZIP_FASTA([[:], fasta]) - ch_fasta = GUNZIP_FASTA.out.gunzip + ch_fasta = GUNZIP_FASTA([[:], file(fasta, checkIfExists: true)]).gunzip } else { ch_fasta = channel.value([[:], file(fasta, checkIfExists: true)]) } // - // Uncompress GTF annotation file or create from GFF3 if required + // MODULE: GUNZIP_GTF, GUNZIP_GFF and GFFREAD // + + // A GTF annotation is used as is, a GFF3 one is converted to GTF if (gtf) { if (gtf.endsWith('.gz')) { - GUNZIP_GTF([[:], gtf]) - ch_gtf = GUNZIP_GTF.out.gunzip + ch_gtf = GUNZIP_GTF([[:], file(gtf, checkIfExists: true)]).gunzip } else { ch_gtf = channel.value([[:], file(gtf, checkIfExists: true)]) @@ -61,29 +66,30 @@ workflow PREPARE_GENOME { } else if (gff) { if (gff.endsWith('.gz')) { - GUNZIP_GFF([[:], gff]) - ch_gff = GUNZIP_GFF.out.gunzip + ch_gff = GUNZIP_GFF([[:], file(gff, checkIfExists: true)]).gunzip } else { ch_gff = channel.value([[:], file(gff, checkIfExists: true)]) } - ch_gtf = GFFREAD(ch_gff, null).gtf + // GFFREAD names its output after `meta.id`, so give it the annotation name + ch_gtf = GFFREAD(ch_gff.map { _meta, gff_file -> [[id: gff_file.baseName], gff_file] }, []).gtf } // - // Uncompress transcript fasta file / create if required + // MODULE: GUNZIP_TRANSCRIPT_FASTA, PREPROCESS_TRANSCRIPTS_FASTA_GENCODE, GTFGENEFILTER and MAKE_TRANSCRIPTS_FASTA // + + // Without a transcript FASTA, one is built from the genome and the annotation, keeping + // only the genes on sequences present in the genome FASTA if (transcript_fasta) { if (transcript_fasta.endsWith('.gz')) { - GUNZIP_TRANSCRIPT_FASTA([[:], transcript_fasta]) - ch_transcript_fasta = GUNZIP_TRANSCRIPT_FASTA.out.gunzip + ch_transcript_fasta = GUNZIP_TRANSCRIPT_FASTA([[:], file(transcript_fasta, checkIfExists: true)]).gunzip } else { ch_transcript_fasta = channel.value([[:], file(transcript_fasta, checkIfExists: true)]) } if (gencode) { - PREPROCESS_TRANSCRIPTS_FASTA_GENCODE(ch_transcript_fasta) - ch_transcript_fasta = PREPROCESS_TRANSCRIPTS_FASTA_GENCODE.out.fasta + ch_transcript_fasta = PREPROCESS_TRANSCRIPTS_FASTA_GENCODE(ch_transcript_fasta).fasta } } else { @@ -95,21 +101,24 @@ workflow PREPARE_GENOME { } // - // Create chromosome sizes file + // MODULE: SAMTOOLS_FAIDX // + SAMTOOLS_FAIDX(ch_fasta.map { meta, fa -> [meta, fa, []] }, true) - ch_fai = SAMTOOLS_FAIDX.out.fai - ch_chrom_sizes = SAMTOOLS_FAIDX.out.sizes // - // Uncompress STAR index or generate from scratch if required + // MODULE: UNTAR_STAR_INDEX, STAR_GENOMEPARAMS_UPGRADE and STAR_GENOMEGENERATE // + ch_star_index = channel.empty() - if (params.source == 'fastq' && !params.skip_alignment && (params.aligner == 'star' || params.aligner == 'star_salmon')) { + if (source == 'fastq' && !skip_alignment && aligner in ['star', 'star_salmon']) { if (star_index) { - def ch_star_index_raw = star_index.endsWith('.tar.gz') - ? UNTAR_STAR_INDEX([[:], star_index]).untar - : channel.value([[:], file(star_index, checkIfExists: true)]) + if (star_index.endsWith('.tar.gz')) { + ch_star_index_raw = UNTAR_STAR_INDEX([[:], file(star_index, checkIfExists: true)]).untar + } + else { + ch_star_index_raw = channel.value([[:], file(star_index, checkIfExists: true)]) + } // A supplied index may have been built with STAR 2.6.x, as the AWS iGenomes ones were, // which STAR 2.7.4a and later refuse to read. `STAR_GENOMEPARAMS_UPGRADE` rewrites the @@ -122,37 +131,37 @@ workflow PREPARE_GENOME { } // - // Uncompress Salmon index or generate from scratch if required + // MODULE: UNTAR_SALMON_INDEX and SALMON_INDEX // + + // `star_salmon` quantifies the STAR transcriptome alignments, which needs no index, so an + // index is only built for the `salmon` pseudo aligner ch_salmon_index = channel.empty() - if (params.source == 'fastq' && (params.pseudo_aligner == 'salmon' || params.aligner == 'star_salmon')) { + if (source == 'fastq' && (pseudo_aligner == 'salmon' || aligner == 'star_salmon')) { if (salmon_index) { if (salmon_index.endsWith('.tar.gz')) { - ch_salmon_index = UNTAR_SALMON_INDEX([[:], salmon_index]).untar.map { _meta, index -> index } + ch_salmon_index = UNTAR_SALMON_INDEX([[:], file(salmon_index, checkIfExists: true)]).untar } else { ch_salmon_index = channel.value([[:], file(salmon_index, checkIfExists: true)]) } } - else { - if (params.pseudo_aligner == 'salmon') { - SALMON_INDEX( - ch_fasta.map { _meta, fa -> fa }, - ch_transcript_fasta.map { _meta, tr -> tr }, - ) - ch_salmon_index = SALMON_INDEX.out.index - } + else if (pseudo_aligner == 'salmon') { + ch_salmon_index = SALMON_INDEX( + ch_fasta.map { _meta, fa -> fa }, + ch_transcript_fasta.map { _meta, tr -> tr }, + ).index.map { index -> [[:], index] } } } // - // Uncompress DEXSeq GFF annotation file if required + // MODULE: GUNZIP_GFF_DEXSEQ // + ch_dexseq_gff = channel.empty() if (gff_dexseq) { if (gff_dexseq.endsWith('.gz')) { - GUNZIP_GFF_DEXSEQ([[:], gff_dexseq]) - ch_dexseq_gff = GUNZIP_GFF_DEXSEQ.out.gunzip + ch_dexseq_gff = GUNZIP_GFF_DEXSEQ([[:], file(gff_dexseq, checkIfExists: true)]).gunzip } else { ch_dexseq_gff = channel.value([[:], file(gff_dexseq, checkIfExists: true)]) @@ -160,13 +169,13 @@ workflow PREPARE_GENOME { } // - // Uncompress SUPPA TPM file if required + // MODULE: GUNZIP_SUPPA_TPM // + ch_suppa_tpm = channel.empty() if (suppa_tpm) { if (suppa_tpm.endsWith('.gz')) { - GUNZIP_SUPPA_TPM([[:], suppa_tpm]) - ch_suppa_tpm = GUNZIP_SUPPA_TPM.out.gunzip.map { _meta, tpm -> tpm } + ch_suppa_tpm = GUNZIP_SUPPA_TPM([[:], file(suppa_tpm, checkIfExists: true)]).gunzip } else { ch_suppa_tpm = channel.value([[:], file(suppa_tpm, checkIfExists: true)]) @@ -174,13 +183,13 @@ workflow PREPARE_GENOME { } emit: - fasta = ch_fasta.map { _meta, fa -> fa } // path: genome.fasta - fai = ch_fai // path: genome.fai - chrom_sizes = ch_chrom_sizes // path: genome.sizes - gtf = ch_gtf.map { _meta, out_gtf -> out_gtf } // path: genome.gtf - transcript_fasta = ch_transcript_fasta.map { _meta, fa -> fa } // path: transcript.fasta - star_index = ch_star_index.map { _meta, index -> index } // path: star/index/ - salmon_index = ch_salmon_index // path: salmon/index/ - dexseq_gff = ch_dexseq_gff.map { _meta, dexseq_gff -> dexseq_gff } // path: dexseq.gff - suppa_tpm = ch_suppa_tpm // path: suppa.tpm + fasta = ch_fasta.map { _meta, fa -> fa } // channel: path(genome.fasta) + fai = SAMTOOLS_FAIDX.out.fai // channel: [ val(meta), path(genome.fasta.fai) ] + chrom_sizes = SAMTOOLS_FAIDX.out.sizes // channel: [ val(meta), path(genome.fasta.sizes) ] + gtf = ch_gtf.map { _meta, out_gtf -> out_gtf } // channel: path(genome.gtf) + transcript_fasta = ch_transcript_fasta.map { _meta, fa -> fa } // channel: path(transcripts.fasta) + star_index = ch_star_index.map { _meta, index -> index } // channel: path(star/index/) + salmon_index = ch_salmon_index.map { _meta, index -> index } // channel: path(salmon/index/) + dexseq_gff = ch_dexseq_gff.map { _meta, dexseq_gff -> dexseq_gff } // channel: path(dexseq.gff) + suppa_tpm = ch_suppa_tpm.map { _meta, tpm -> tpm } // channel: path(suppa.tpm) } diff --git a/subworkflows/local/prepare_genome/meta.yml b/subworkflows/local/prepare_genome/meta.yml new file mode 100644 index 0000000..93788e4 --- /dev/null +++ b/subworkflows/local/prepare_genome/meta.yml @@ -0,0 +1,149 @@ +# yaml-language-server: $schema=https://raw.githubusercontent.com/nf-core/modules/master/subworkflows/yaml-schema.json +name: "prepare_genome" +description: Uncompress the reference genome files and build the genome index, transcript FASTA and aligner indices that were not supplied +keywords: + - genome + - reference + - index + - star + - salmon + - gunzip + - untar +components: + - gunzip + - untar + - samtools/faidx + - gffread + - star/genomegenerate + - salmon/index + - rsem/preparereference + - gtfgenefilter + - preprocess_transcripts_fasta_gencode + - star_genomeparams_upgrade +input: + - fasta: + type: string + description: | + Path to the genome FASTA file, gzipped or not + pattern: "*.{fa,fasta}{,.gz}" + - gtf: + type: string + description: | + Path to the GTF annotation file, gzipped or not. Takes precedence over `gff` + pattern: "*.gtf{,.gz}" + - gff: + type: string + description: | + Path to the GFF3 annotation file, gzipped or not, converted to GTF with gffread when no `gtf` is given + pattern: "*.{gff,gff3}{,.gz}" + - transcript_fasta: + type: string + description: | + Path to the transcript FASTA file, gzipped or not. When not given, it is built from the + genome FASTA and the annotation with RSEM + pattern: "*.{fa,fasta}{,.gz}" + - star_index: + type: string + description: | + Path to the STAR index directory, or a `.tar.gz` archive of it. When not given, it is + built from the genome FASTA and the annotation + - salmon_index: + type: string + description: | + Path to the Salmon index directory, or a `.tar.gz` archive of it. When not given, it is + built from the genome FASTA and the transcript FASTA for the `salmon` pseudo aligner + - gff_dexseq: + type: string + description: | + Path to the flattened DEXSeq GFF annotation, gzipped or not + pattern: "*.gff{,.gz}" + - suppa_tpm: + type: string + description: | + Path to the SUPPA transcript TPM table, gzipped or not + pattern: "*.tpm{,.gz}" + - gencode: + type: boolean + description: | + Whether the annotation is from GENCODE, in which case the transcript FASTA headers are + trimmed to the transcript ID + - source: + type: string + description: | + Type of input data. The aligner indices are only prepared for `fastq` + enum: ["fastq", "genome_bam", "transcriptome_bam", "salmon_results"] + - aligner: + type: string + description: | + Genome aligner, the STAR index is prepared for both values + enum: ["star", "star_salmon"] + - pseudo_aligner: + type: string + description: | + Pseudo aligner, the Salmon index is prepared for `salmon` + enum: ["salmon"] + - skip_alignment: + type: boolean + description: | + Whether the genome alignment is skipped, in which case no STAR index is prepared +output: + - fasta: + type: file + description: | + Channel containing the uncompressed genome FASTA file + Structure: [ path(fasta) ] + pattern: "*.{fa,fasta}" + - fai: + type: file + description: | + Channel containing the genome FASTA index + Structure: [ val(meta), path(fai) ] + pattern: "*.fai" + - chrom_sizes: + type: file + description: | + Channel containing the chromosome sizes of the genome + Structure: [ val(meta), path(sizes) ] + pattern: "*.sizes" + - gtf: + type: file + description: | + Channel containing the uncompressed GTF annotation, or the one converted from the GFF3 annotation + Structure: [ path(gtf) ] + pattern: "*.gtf" + - transcript_fasta: + type: file + description: | + Channel containing the uncompressed transcript FASTA file, or the one built with RSEM + Structure: [ path(fasta) ] + pattern: "*.{fa,fasta}" + - star_index: + type: directory + description: | + Channel containing the STAR index directory, empty when no STAR alignment is run + Structure: [ path(index) ] + - salmon_index: + type: directory + description: | + Channel containing the Salmon index directory, empty when Salmon is not run or no index is needed + Structure: [ path(index) ] + - dexseq_gff: + type: file + description: | + Channel containing the uncompressed flattened DEXSeq GFF annotation, empty when none is given + Structure: [ path(gff) ] + pattern: "*.gff" + - suppa_tpm: + type: file + description: | + Channel containing the uncompressed SUPPA transcript TPM table, empty when none is given + Structure: [ path(tpm) ] + pattern: "*.tpm" +authors: + - "@asmaali98" + - "@bensouthgate" + - "@jma1991" + - "@valentinoruggieri" + - "@piplus2" +maintainers: + - "@piplus2" diff --git a/subworkflows/local/prepare_genome/tests/main.nf.test b/subworkflows/local/prepare_genome/tests/main.nf.test new file mode 100644 index 0000000..aa434ae --- /dev/null +++ b/subworkflows/local/prepare_genome/tests/main.nf.test @@ -0,0 +1,201 @@ +// nf-core subworkflows test prepare_genome +nextflow_workflow { + + name "Test Subworkflow PREPARE_GENOME" + script "../main.nf" + workflow "PREPARE_GENOME" + + tag "subworkflows" + tag "subworkflows/prepare_genome" + tag "gunzip" + tag "untar" + tag "samtools" + tag "samtools/faidx" + tag "gffread" + tag "star" + tag "star/genomegenerate" + tag "salmon" + tag "salmon/index" + tag "rsem" + tag "rsem/preparereference" + tag "gtfgenefilter" + tag "preprocess_transcripts_fasta_gencode" + tag "star_genomeparams_upgrade" + + test("homo_sapiens - fasta and gtf - build indices") { + + when { + workflow { + """ + input[0] = params.modules_testdata_base_path + 'genomics/homo_sapiens/genome/genome.fasta' + input[1] = params.modules_testdata_base_path + 'genomics/homo_sapiens/genome/genome.gtf' + input[2] = null + input[3] = null + input[4] = null + input[5] = null + input[6] = null + input[7] = null + input[8] = false + input[9] = 'fastq' + input[10] = 'star' + input[11] = 'salmon' + input[12] = false + """ + } + } + + then { + assertAll( + { assert workflow.success }, + { assert snapshot( + workflow.out.fasta, + workflow.out.fai, + workflow.out.chrom_sizes, + workflow.out.gtf, + workflow.out.transcript_fasta, + workflow.out.star_index.collect { index -> file(index).list().sort() }, + workflow.out.salmon_index.collect { index -> file(index).name }, + workflow.out.dexseq_gff, + workflow.out.suppa_tpm + ).match() } + ) + } + + } + + test("homo_sapiens - gzipped inputs, gff and supplied indices") { + + when { + workflow { + """ + // The test datasets carry no gzipped GFF3, transcript FASTA or SUPPA TPM table, nor + // an uncompressed Salmon index, so they are written here. The Salmon index and the + // TPM table are only passed through, so small stand-ins do + def gzip = { src, dest -> + def out = new java.util.zip.GZIPOutputStream(file(dest).newOutputStream()) + out.write(file(src).bytes) + out.close() + dest + } + file("${workDir}/salmon_index").mkdirs() + file("${workDir}/salmon_index/versionInfo.json").text = '{}\\n' + file("${workDir}/quant.tpm").text = 'sample1\\ttranscript1\\t1.0\\n' + + input[0] = params.modules_testdata_base_path + 'genomics/homo_sapiens/genome/genome.fasta.gz' + input[1] = null + input[2] = gzip.call(params.pipelines_testdata_base_path + 'reference/genes_chrX.gff3', "${workDir}/genes_chrX.gff3.gz") + input[3] = gzip.call(params.modules_testdata_base_path + 'genomics/homo_sapiens/genome/transcriptome.fasta', "${workDir}/transcriptome.fasta.gz") + input[4] = params.modules_testdata_base_path + 'genomics/homo_sapiens/genome/index/star/star.tar.gz' + input[5] = "${workDir}/salmon_index" + input[6] = params.pipelines_testdata_base_path + 'testdata/dexseq/genes_chrX.DEXSeq.gff.gz' + input[7] = gzip.call("${workDir}/quant.tpm", "${workDir}/quant.tpm.gz") + input[8] = true + input[9] = 'fastq' + input[10] = 'star_salmon' + input[11] = 'salmon' + input[12] = false + """ + } + } + + then { + assertAll( + { assert workflow.success }, + { assert snapshot( + workflow.out.fasta, + workflow.out.fai, + workflow.out.chrom_sizes, + workflow.out.gtf, + workflow.out.transcript_fasta, + workflow.out.star_index.collect { index -> file(index).list().sort() }, + workflow.out.salmon_index.collect { index -> file(index).name }, + workflow.out.dexseq_gff, + workflow.out.suppa_tpm + ).match() } + ) + } + + } + + test("homo_sapiens - genome_bam source - gzipped gtf - no indices") { + + when { + workflow { + """ + // The test datasets carry no gzipped GTF or SUPPA TPM table, so they are written here + def gzip = { src, dest -> + def out = new java.util.zip.GZIPOutputStream(file(dest).newOutputStream()) + out.write(file(src).bytes) + out.close() + dest + } + file("${workDir}/quant.tpm").text = 'sample1\\ttranscript1\\t1.0\\n' + + input[0] = params.modules_testdata_base_path + 'genomics/homo_sapiens/genome/genome.fasta' + input[1] = gzip.call(params.modules_testdata_base_path + 'genomics/homo_sapiens/genome/genome.gtf', "${workDir}/genome.gtf.gz") + input[2] = null + input[3] = params.modules_testdata_base_path + 'genomics/homo_sapiens/genome/transcriptome.fasta' + input[4] = params.modules_testdata_base_path + 'genomics/homo_sapiens/genome/index/star/star.tar.gz' + input[5] = null + input[6] = null + input[7] = "${workDir}/quant.tpm" + input[8] = false + input[9] = 'genome_bam' + input[10] = 'star_salmon' + input[11] = 'salmon' + input[12] = false + """ + } + } + + then { + assertAll( + { assert workflow.success }, + { assert workflow.out.star_index.size() == 0 }, + { assert workflow.out.salmon_index.size() == 0 }, + { assert snapshot( + workflow.out.fasta, + workflow.out.fai, + workflow.out.chrom_sizes, + workflow.out.gtf, + workflow.out.transcript_fasta, + workflow.out.suppa_tpm + ).match() } + ) + } + + } + + test("homo_sapiens - fasta and gtf - build indices - stub") { + + options "-stub" + + when { + workflow { + """ + input[0] = params.modules_testdata_base_path + 'genomics/homo_sapiens/genome/genome.fasta' + input[1] = params.modules_testdata_base_path + 'genomics/homo_sapiens/genome/genome.gtf' + input[2] = null + input[3] = null + input[4] = null + input[5] = null + input[6] = null + input[7] = null + input[8] = false + input[9] = 'fastq' + input[10] = 'star' + input[11] = 'salmon' + input[12] = false + """ + } + } + + then { + assertAll( + { assert workflow.success }, + { assert snapshot(sanitizeOutput(workflow.out)).match() } + ) + } + + } +} diff --git a/subworkflows/local/prepare_genome/tests/main.nf.test.snap b/subworkflows/local/prepare_genome/tests/main.nf.test.snap new file mode 100644 index 0000000..8bbe10b --- /dev/null +++ b/subworkflows/local/prepare_genome/tests/main.nf.test.snap @@ -0,0 +1,246 @@ +{ + "homo_sapiens - fasta and gtf - build indices": { + "content": [ + [ + "/nf-core/test-datasets/modules/data/genomics/homo_sapiens/genome/genome.fasta" + ], + [ + [ + { + + }, + "genome.fasta.fai:md5,3520cd30e1b100e55f578db9c855f685" + ] + ], + [ + [ + { + + }, + "genome.fasta.sizes:md5,b190587cae0531f3cf25552d8aa674db" + ] + ], + [ + "/nf-core/test-datasets/modules/data/genomics/homo_sapiens/genome/genome.gtf" + ], + [ + "genome.transcripts.fa:md5,050c521a2719c2ae48267c1e65218f29" + ], + [ + [ + "Genome", + "Log.out", + "SA", + "SAindex", + "chrLength.txt", + "chrName.txt", + "chrNameLength.txt", + "chrStart.txt", + "exonGeTrInfo.tab", + "exonInfo.tab", + "geneInfo.tab", + "genomeParameters.txt", + "sjdbInfo.txt", + "sjdbList.fromGTF.out.tab", + "sjdbList.out.tab", + "transcriptInfo.tab" + ] + ], + [ + "salmon" + ], + [ + + ], + [ + + ] + ], + "timestamp": "2026-09-23T12:30:21.989618264", + "meta": { + "nf-test": "0.9.5", + "nextflow": "26.04.1" + } + }, + "homo_sapiens - fasta and gtf - build indices - stub": { + "content": [ + { + "chrom_sizes": [ + [ + { + + }, + "genome.fasta.sizes:md5,d41d8cd98f00b204e9800998ecf8427e" + ] + ], + "dexseq_gff": [ + + ], + "fai": [ + [ + { + + }, + "genome.fasta.fai:md5,d41d8cd98f00b204e9800998ecf8427e" + ] + ], + "fasta": [ + "/nf-core/test-datasets/modules/data/genomics/homo_sapiens/genome/genome.fasta" + ], + "gtf": [ + "/nf-core/test-datasets/modules/data/genomics/homo_sapiens/genome/genome.gtf" + ], + "salmon_index": [ + [ + "complete_ref_lens.bin:md5,d41d8cd98f00b204e9800998ecf8427e", + "ctable.bin:md5,d41d8cd98f00b204e9800998ecf8427e", + "ctg_offsets.bin:md5,d41d8cd98f00b204e9800998ecf8427e", + "duplicate_clusters.tsv:md5,d41d8cd98f00b204e9800998ecf8427e", + "info.json:md5,d41d8cd98f00b204e9800998ecf8427e", + "mphf.bin:md5,d41d8cd98f00b204e9800998ecf8427e", + "pos.bin:md5,d41d8cd98f00b204e9800998ecf8427e", + "pre_indexing.log:md5,d41d8cd98f00b204e9800998ecf8427e", + "rank.bin:md5,d41d8cd98f00b204e9800998ecf8427e", + "refAccumLengths.bin:md5,d41d8cd98f00b204e9800998ecf8427e", + "ref_indexing.log:md5,d41d8cd98f00b204e9800998ecf8427e", + "reflengths.bin:md5,d41d8cd98f00b204e9800998ecf8427e", + "refseq.bin:md5,d41d8cd98f00b204e9800998ecf8427e", + "seq.bin:md5,d41d8cd98f00b204e9800998ecf8427e", + "versionInfo.json:md5,d41d8cd98f00b204e9800998ecf8427e" + ] + ], + "star_index": [ + [ + "Genome:md5,d41d8cd98f00b204e9800998ecf8427e", + "Log.out:md5,d41d8cd98f00b204e9800998ecf8427e", + "SA:md5,d41d8cd98f00b204e9800998ecf8427e", + "SAindex:md5,d41d8cd98f00b204e9800998ecf8427e", + "chrLength.txt:md5,d41d8cd98f00b204e9800998ecf8427e", + "chrName.txt:md5,d41d8cd98f00b204e9800998ecf8427e", + "chrNameLength.txt:md5,d41d8cd98f00b204e9800998ecf8427e", + "chrStart.txt:md5,d41d8cd98f00b204e9800998ecf8427e", + "exonGeTrInfo.tab:md5,d41d8cd98f00b204e9800998ecf8427e", + "exonInfo.tab:md5,d41d8cd98f00b204e9800998ecf8427e", + "geneInfo.tab:md5,d41d8cd98f00b204e9800998ecf8427e", + "genomeParameters.txt:md5,d41d8cd98f00b204e9800998ecf8427e", + "sjdbInfo.txt:md5,d41d8cd98f00b204e9800998ecf8427e", + "sjdbList.fromGTF.out.tab:md5,d41d8cd98f00b204e9800998ecf8427e", + "sjdbList.out.tab:md5,d41d8cd98f00b204e9800998ecf8427e", + "transcriptInfo.tab:md5,d41d8cd98f00b204e9800998ecf8427e" + ] + ], + "suppa_tpm": [ + + ], + "transcript_fasta": [ + "genome.transcripts.fa:md5,d41d8cd98f00b204e9800998ecf8427e" + ] + } + ], + "timestamp": "2026-09-23T12:30:44.947864209", + "meta": { + "nf-test": "0.9.5", + "nextflow": "26.04.1" + } + }, + "homo_sapiens - gzipped inputs, gff and supplied indices": { + "content": [ + [ + "genome.fasta:md5,f315020d899597c1b57e5fe9f60f4c3e" + ], + [ + [ + { + + }, + "genome.fasta.fai:md5,3520cd30e1b100e55f578db9c855f685" + ] + ], + [ + [ + { + + }, + "genome.fasta.sizes:md5,b190587cae0531f3cf25552d8aa674db" + ] + ], + [ + "genes_chrX.gtf:md5,0624e70c0b23bae787b4bfd78cb331dd" + ], + [ + "transcriptome.fixed.fa:md5,4e04341f8781429444f53f69beb31392" + ], + [ + [ + "Genome", + "Log.out", + "SA", + "SAindex", + "chrLength.txt", + "chrName.txt", + "chrNameLength.txt", + "chrStart.txt", + "exonGeTrInfo.tab", + "exonInfo.tab", + "geneInfo.tab", + "genomeParameters.txt", + "sjdbInfo.txt", + "sjdbList.fromGTF.out.tab", + "sjdbList.out.tab", + "transcriptInfo.tab" + ] + ], + [ + "salmon_index" + ], + [ + "genes_chrX.DEXSeq.gff:md5,ee8b662ee32c6c563802690491ec0633" + ], + [ + "quant.tpm:md5,d606cb7c509c0f2d4287cd83243201d9" + ] + ], + "timestamp": "2026-09-23T16:48:37.469824792", + "meta": { + "nf-test": "0.9.5", + "nextflow": "26.04.1" + } + }, + "homo_sapiens - genome_bam source - gzipped gtf - no indices": { + "content": [ + [ + "/nf-core/test-datasets/modules/data/genomics/homo_sapiens/genome/genome.fasta" + ], + [ + [ + { + + }, + "genome.fasta.fai:md5,3520cd30e1b100e55f578db9c855f685" + ] + ], + [ + [ + { + + }, + "genome.fasta.sizes:md5,b190587cae0531f3cf25552d8aa674db" + ] + ], + [ + "genome.gtf:md5,50fc877b1c53b36b3b413aff88bda48c" + ], + [ + "/nf-core/test-datasets/modules/data/genomics/homo_sapiens/genome/transcriptome.fasta" + ], + [ + "quant.tpm:md5,d606cb7c509c0f2d4287cd83243201d9" + ] + ], + "timestamp": "2026-09-23T16:48:43.731886703", + "meta": { + "nf-test": "0.9.5", + "nextflow": "26.04.1" + } + } +} \ No newline at end of file