diff --git a/test/fixtures/salmon/README.md b/test/fixtures/salmon/README.md new file mode 100644 index 000000000..2d935bb72 --- /dev/null +++ b/test/fixtures/salmon/README.md @@ -0,0 +1,3 @@ +# Salmon test fixtures + +`salmon_index.tar.gz` — built with Salmon 2.6.0 by running the `build_salmon_index` task against the existing `reference/GRCh38.chrY_chrM.fa` fixture (chosen because the shared `fastqs/test_R1.fq.gz`/`fastqs/test_R2.fq.gz` reads were simulated from this reference, per their FASTQ headers). diff --git a/test/fixtures/salmon/salmon_index.tar.gz b/test/fixtures/salmon/salmon_index.tar.gz new file mode 100644 index 000000000..39690e80f --- /dev/null +++ b/test/fixtures/salmon/salmon_index.tar.gz @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:507b111d23bc167bb026f2ed907a742ee2563d65431c76f518512eccce0f952d +size 311026 diff --git a/tools/CHANGELOG.md b/tools/CHANGELOG.md index cf5c342d2..46de4b14e 100644 --- a/tools/CHANGELOG.md +++ b/tools/CHANGELOG.md @@ -9,6 +9,8 @@ The format is based on [Keep a Changelog](http://keepachangelog.com/). ### Changed - Enforced non-empty qualifiers on `Array[*]` types that cannot be empty [#328](https://github.com/stjudecloud/workflows/pull/328) +### Added +- Added WDL implementation for Salmon (`build_salmon_index` and `quant` tasks) [#326](https://github.com/stjudecloud/workflows/pull/326) ## 2026 February diff --git a/tools/salmon.wdl b/tools/salmon.wdl new file mode 100644 index 000000000..cedfe1c8a --- /dev/null +++ b/tools/salmon.wdl @@ -0,0 +1,293 @@ +version 1.1 + +task build_salmon_index { + meta { + description: "Builds a Salmon index from a transcriptome FASTA file, for use in quantification" + outputs: { + salmon_index_tar_gz: "A gzipped TAR file containing the Salmon index files." + } + } + + parameter_meta { + transcripts_fasta: "FASTA format file containing the reference transcriptome to index" + decoys_fasta: { + description: "Optional FASTA file containing decoy genome sequences to improve mapping specificity.", + help: "Per Salmon's decoy-aware indexing workflow.", + group: "Common", + } + index_name: { + description: "Name for the output index, in compressed archive format. The suffix `.tar.gz` will be added.", + group: "Common", + } + use_all_cores: { + description: "Use all cores? Recommended for cloud environments.", + group: "Resources", + } + ncpu: { + description: "Number of cores to allocate for task", + group: "Resources", + } + modify_disk_size_gb: { + description: "Add to or subtract from dynamic disk space allocation. Default disk size is determined by the size of the inputs. Specified in GB.", + group: "Resources", + } + modify_memory_gb: { + description: "Add to or subtract from dynamic memory allocation. Default memory is determined by the size of the inputs. Specified in GB.", + group: "Resources", + } + } + + input { + File transcripts_fasta + File? decoys_fasta + String index_name = "salmon_index" + Boolean use_all_cores = false + Int ncpu = 4 + Int modify_disk_size_gb = 0 + Int modify_memory_gb = 0 + } + + String salmon_index_filename = index_name + ".tar.gz" + + Float transcripts_fasta_size = size(transcripts_fasta, "GB") + Int disk_size_gb = ceil(transcripts_fasta_size * 4) + 10 + modify_disk_size_gb + + command <<< + set -euo pipefail + + n_cores=~{ncpu} + if ~{use_all_cores}; then + n_cores=$(nproc) + fi + + (gzip -dcf "~{transcripts_fasta}" > transcripts.fasta 2>/dev/null) || cp "~{transcripts_fasta}" transcripts.fasta + fasta="transcripts.fasta" + + ~{if defined(decoys_fasta) then "(gzip -dcf " + select_first([decoys_fasta]) + " > decoys.fasta 2>/dev/null) || cp " + select_first([decoys_fasta]) + " decoys.fasta" else ""} + ~{if defined(decoys_fasta) then "grep \"^>\" decoys.fasta | cut -d \" \" -f1 | sed \"s/^>//\" > decoys.txt" else ""} + ~{if defined(decoys_fasta) then "cat transcripts.fasta decoys.fasta > combined.fasta" else ""} + ~{if defined(decoys_fasta) then "fasta=combined.fasta" else ""} + + salmon index \ + -t "$fasta" \ + -i "~{index_name}" \ + ~{if defined(decoys_fasta) then "-d decoys.txt" else ""} \ + -p "$n_cores" + + tar -czf "~{salmon_index_filename}" "~{index_name}" + + rm -f transcripts.fasta decoys.fasta combined.fasta + >>> + + output { + File salmon_index_tar_gz = salmon_index_filename + } + + runtime { + cpu: ncpu + memory: "~{ceil(transcripts_fasta_size * 4) + 4 + modify_memory_gb} GB" + disks: "~{disk_size_gb} GB" + container: "quay.io/biocontainers/salmon:2.6.0--hfa8f182_0" + maxRetries: 1 + } +} + +task quant { + meta { + description: "Runs Salmon quant in mapping-based mode to quantify transcript-level expression from RNA-Seq reads, using a pre-built Salmon index" + outputs: { + quant_results_tar_gz: "A gzipped TAR file containing the Salmon quantification output directory, including `quant.sf`.", + quant_sf: "The raw `quant.sf` transcript quantification file, provided directly in addition to the tarballed output for convenience." + } + } + + parameter_meta { + salmon_index_tar_gz: "A gzipped TAR file containing the Salmon index files. Suitable as the output of the `build_salmon_index` task." + read_one_fastqs_gz: "An array of gzipped FASTQ files containing read one information" + read_two_fastqs_gz: { + description: "An array of gzipped FASTQ files containing read two information. Omit for single-end reads.", + group: "Common", + } + lib_type: { + description: "Salmon library type describing the relative orientation and strandedness of paired reads.", + help: "Use `A` to let Salmon auto-detect the library type — recommended for most users.", + group: "Common", + } + prefix: { + description: "Prefix for the Salmon quantification output. The extension `.tar.gz` will be added.", + group: "Common", + } + validate_mappings: { + description: "Validate mappings using an alignment-based verification step.", + group: "Salmon Options", + } + num_bootstraps: { + description: "Salmon has the ability to optionally compute bootstrapped abundance estimates.", + help: "This is done by resampling (with replacement) from the counts assigned to the fragment equivalence classes, and then re-running the optimization procedure for each such sample.", + group: "Salmon Options", + } + incompat_prior: { + description: "This parameter governs the a priori probability that a fragment mapping is nonetheless the correct mapping.", + help: "Specifically, this is for a fragment mapping or aligning to the reference in a manner incompatible with the prescribed library type.", + group: "Salmon Options", + } + range_factorization_bins: { + description: "The range-factorization feature allows using a data-driven likelihood factorization.", + help: "This can improve quantification accuracy on certain classes of difficult transcripts.", + group: "Salmon Options", + } + fld_mean: { + description: "Allows the user to set the expected mean fragment length of the sequencing library.", + help: "Since the empirical fragment length distribution cannot be estimated from the mappings of single-end reads, this is only important when running Salmon with single-end reads.", + group: "Salmon Options", + } + fld_sd: { + description: "Allows the user to set the expected standard deviation of the fragment length distribution.", + help: "Since the empirical fragment length distribution cannot be estimated from the mappings of single-end reads, this is only important when running Salmon with single-end reads.", + group: "Salmon Options", + } + seq_bias: { + description: "Passing this flag will enable it to learn and correct for sequence-specific biases in the input data.", + group: "Salmon Options", + } + gc_bias: { + description: "Passing this flag will enable it to learn and correct for fragment-level GC biases in the input data.", + group: "Salmon Options", + } + pos_bias: { + description: "Passing this flag will enable modeling of a position-specific fragment start distribution.", + group: "Salmon Options", + } + use_em: { + description: "Use the \"standard\" EM algorithm to optimize abundance estimates instead of the variational Bayesian EM algorithm.", + group: "Salmon Options", + } + recover_orphans: { + description: "This flag (which should only be used in conjunction with selective alignment), performs orphan \"rescue\" for reads.", + group: "Salmon Options", + } + hard_filter: { + description: "This flag (which should only be used with selective alignment) turns off soft filtering and range-factorized equivalence classes.", + help: "Removes all but the equally highest scoring mappings from the equivalence class label for each fragment.", + group: "Salmon Options", + } + allow_dovetail: { + description: "Dovetailing mappings and alignments are considered discordant and discarded by default.", + help: "If you wish to consider dovetailing mappings as concordant, you can do so by passing this flag.", + group: "Salmon Options", + } + dump_eq: { + description: "If passed, Salmon will write a file in the auxiliary directory, called eq_classes.txt.", + help: "Contains the equivalence classes and corresponding counts that were computed during quasi-mapping.", + group: "Salmon Options", + } + write_unmapped_names: { + description: "Passing this flag will tell Salmon to write out the names of reads (or mates in paired-end reads) that do not map to the transcriptome.", + group: "Salmon Options", + } + use_all_cores: { + description: "Use all cores? Recommended for cloud environments.", + group: "Resources", + } + ncpu: { + description: "Number of cores to allocate for task", + group: "Resources", + } + modify_disk_size_gb: { + description: "Add to or subtract from dynamic disk space allocation. Default disk size is determined by the size of the inputs. Specified in GB.", + group: "Resources", + } + modify_memory_gb: { + description: "Add to or subtract from dynamic memory allocation. Default memory is determined by the size of the inputs. Specified in GB.", + group: "Resources", + } + } + + input { + File salmon_index_tar_gz + Array[File]+ read_one_fastqs_gz + Array[File]? read_two_fastqs_gz + String lib_type = "A" + String prefix = basename(read_one_fastqs_gz[0], ".fastq.gz") + Boolean validate_mappings = true + Int num_bootstraps = 0 + Float incompat_prior = 0.0 + Int range_factorization_bins = 4 + Int fld_mean = 250 + Int fld_sd = 25 + Boolean seq_bias = false + Boolean gc_bias = false + Boolean pos_bias = false + Boolean use_em = false + Boolean recover_orphans = false + Boolean hard_filter = false + Boolean allow_dovetail = false + Boolean dump_eq = false + Boolean write_unmapped_names = false + Boolean use_all_cores = false + Int ncpu = 4 + Int modify_disk_size_gb = 0 + Int modify_memory_gb = 0 + } + + Array[File] read_twos = select_first([read_two_fastqs_gz, []]) + + Float read_one_size = size(read_one_fastqs_gz, "GB") + Float read_two_size = size(read_twos, "GB") + Float index_size = size(salmon_index_tar_gz, "GB") + Int disk_size_gb = ceil((read_one_size + read_two_size + index_size) * 3) + 10 + modify_disk_size_gb + Int memory_gb = ceil(index_size * 4) + 8 + modify_memory_gb + + command <<< + set -euo pipefail + + n_cores=~{ncpu} + if ~{use_all_cores}; then + n_cores=$(nproc) + fi + + mkdir salmon_index + tar -xzf "~{salmon_index_tar_gz}" -C salmon_index --strip-components 1 + + # shellcheck disable=SC2086 + # shellcheck disable=SC2086 + salmon quant \ + -i salmon_index \ + -l "~{lib_type}" \ + ~{if length(read_twos) > 0 then "-1 " + sep(" ", squote(read_one_fastqs_gz)) + " -2 " + sep(" ", squote(read_twos)) else "-r " + sep(" ", squote(read_one_fastqs_gz))} \ + ~{if validate_mappings then "--validateMappings" else ""} \ + -p "$n_cores" \ + --numBootstraps ~{num_bootstraps} \ + --incompatPrior ~{incompat_prior} \ + --rangeFactorizationBins ~{range_factorization_bins} \ + ~{if length(read_twos) == 0 then "--fldMean " + fld_mean else ""} \ + ~{if length(read_twos) == 0 then "--fldSD " + fld_sd else ""} \ + ~{if seq_bias then "--seqBias" else ""} \ + ~{if gc_bias then "--gcBias" else ""} \ + ~{if pos_bias then "--posBias" else ""} \ + ~{if use_em then "--useEM" else ""} \ + ~{if recover_orphans then "--recoverOrphans" else ""} \ + ~{if hard_filter then "--hardFilter" else ""} \ + ~{if allow_dovetail then "--allowDovetail" else ""} \ + ~{if dump_eq then "--dumpEq" else ""} \ + ~{if write_unmapped_names then "--writeUnmappedNames" else ""} \ + -o "~{prefix}" + + cp "~{prefix}/quant.sf" "~{prefix}.quant.sf" + + tar -czf "~{prefix}.tar.gz" "~{prefix}" + >>> + + output { + File quant_results_tar_gz = prefix + ".tar.gz" + File quant_sf = prefix + ".quant.sf" + } + + runtime { + cpu: ncpu + memory: "~{memory_gb} GB" + disks: "~{disk_size_gb} GB" + container: "quay.io/biocontainers/salmon:2.6.0--hfa8f182_0" + maxRetries: 1 + } +} diff --git a/tools/test/salmon.yaml b/tools/test/salmon.yaml new file mode 100644 index 000000000..5fba13c95 --- /dev/null +++ b/tools/test/salmon.yaml @@ -0,0 +1,43 @@ +build_salmon_index: + - name: builds_index_successfully + inputs: + transcripts_fasta: + - reference/gencode.v50.BCR_ABL1.transcripts.fa.gz + assertions: + outputs: + salmon_index_tar_gz: + - Name: salmon_index.tar.gz + - name: builds_index_with_decoys + inputs: + transcripts_fasta: + - reference/gencode.v50.BCR_ABL1.transcripts.fa.gz + decoys_fasta: + - reference/GRCh38.chrY_chrM.fa + assertions: + outputs: + salmon_index_tar_gz: + - Name: salmon_index.tar.gz + +quant: + - name: quantifies_paired_end_reads + inputs: + salmon_index_tar_gz: + - salmon/salmon_index.tar.gz + read_one_fastqs_gz: + - - fastqs/test_R1.fq.gz + read_two_fastqs_gz: + - - fastqs/test_R2.fq.gz + assertions: + outputs: + quant_results_tar_gz: + - Name: test_R1.fq.gz.tar.gz + - name: quantifies_single_end_reads + inputs: + salmon_index_tar_gz: + - salmon/salmon_index.tar.gz + read_one_fastqs_gz: + - - fastqs/test_R1.fq.gz + assertions: + outputs: + quant_results_tar_gz: + - Name: test_R1.fq.gz.tar.gz \ No newline at end of file