diff --git a/cspell.json b/cspell.json index 1c7e5f46b..9c85392d4 100644 --- a/cspell.json +++ b/cspell.json @@ -93,6 +93,7 @@ "excludelist", "bklist", "probelist", + "incompat", // genomics / sequencing terms "phred", "mapq", @@ -111,6 +112,7 @@ "exonic", "intragenic", "transcriptomic", + "transcriptome", "endedness", "unstranded", "Unstranded", diff --git a/sprocket.toml b/sprocket.toml index f1fec3163..46c78d84d 100644 --- a/sprocket.toml +++ b/sprocket.toml @@ -1,5 +1,4 @@ [check] -all_lint_rules = true except = ["ContainerUri", "TodoComment", "UnusedInput"] deny_notes = true diff --git a/test/fixtures/salmon/README.md b/test/fixtures/salmon/README.md new file mode 100644 index 000000000..672730891 --- /dev/null +++ b/test/fixtures/salmon/README.md @@ -0,0 +1,3 @@ +# Salmon test fixtures + +`salmon_index.tar.gz` - built with Salmon 2.6.0 by running the `index` task against the existing `reference/gencode.v50.BCR_ABL1.transcripts.fa.gz` fixture (a small protein-coding transcriptome subset covering the BCR and ABL1 genes). diff --git a/test/fixtures/salmon/salmon_index.tar.gz b/test/fixtures/salmon/salmon_index.tar.gz new file mode 100644 index 000000000..39690e80f --- /dev/null +++ b/test/fixtures/salmon/salmon_index.tar.gz @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:507b111d23bc167bb026f2ed907a742ee2563d65431c76f518512eccce0f952d +size 311026 diff --git a/tools/CHANGELOG.md b/tools/CHANGELOG.md index cf5c342d2..156e5170e 100644 --- a/tools/CHANGELOG.md +++ b/tools/CHANGELOG.md @@ -4,6 +4,12 @@ All notable changes to this project will be documented in this file. The format is based on [Keep a Changelog](http://keepachangelog.com/). +## 2026 September + +### Added + +- Added WDL implementation for Salmon (`index` and `quant` tasks) [#326](https://github.com/stjudecloud/workflows/pull/326) + ## 2026 August ### Changed diff --git a/tools/salmon.wdl b/tools/salmon.wdl new file mode 100644 index 000000000..b230b76a1 --- /dev/null +++ b/tools/salmon.wdl @@ -0,0 +1,292 @@ +version 1.3 + +task index { + meta { + description: "Builds a Salmon index from a transcriptome FASTA file, for use in quantification" + outputs: { + index_tar_gz: "A gzipped TAR file containing the Salmon index files." + } + } + + parameter_meta { + transcripts_fasta: "FASTA format file containing the reference transcriptome to index. Can be gzipped." + decoys_fasta: { + description: "Optional FASTA file containing decoy genome sequences to improve mapping specificity. Can be gzipped.", + group: "Common", + } + index_name: { + description: "Name for the output index, in compressed archive format. The suffix `.tar.gz` will be added.", + group: "Common", + } + use_all_cores: { + description: "Use all cores? Recommended for cloud environments.", + group: "Resources", + } + ncpu: { + description: "Number of cores to allocate for task", + group: "Resources", + } + modify_disk_size_gb: { + description: "Add to or subtract from dynamic disk space allocation. Default disk size is determined by the size of the inputs. Specified in GB.", + group: "Resources", + } + modify_memory_gb: { + description: "Add to or subtract from dynamic memory allocation. Default memory is determined by the size of the inputs. Specified in GB.", + group: "Resources", + } + } + + input { + File transcripts_fasta + File? decoys_fasta + String index_name = "salmon_index" + Boolean use_all_cores = false + Int ncpu = 4 + Int modify_disk_size_gb = 0 + Int modify_memory_gb = 0 + } + + String salmon_index_filename = index_name + ".tar.gz" + + Float transcripts_fasta_size = size(transcripts_fasta, "GB") + Float decoys_fasta_size = size(decoys_fasta, "GB") + Int disk_size_gb = ceil(transcripts_fasta_size * 4) + ceil(decoys_fasta_size * 4) + 10 + modify_disk_size_gb + + command <<< + set -euo pipefail + + n_cores=~{ncpu} + if ~{use_all_cores}; then + n_cores=$(nproc) + fi + + transcripts_name="~{basename(transcripts_fasta, ".gz")}" + gunzip -c "~{transcripts_fasta}" > "$transcripts_name" || ln -sf "~{transcripts_fasta}" "$transcripts_name" + fasta="$transcripts_name" + + decoys_name="~{if defined(decoys_fasta) then basename(select_first([decoys_fasta]), ".gz") else ""}" + if [ -n "$decoys_name" ]; then + gunzip -c "~{decoys_fasta}" > "$decoys_name" || ln -sf "~{decoys_fasta}" "$decoys_name" + grep "^>" "$decoys_name" | cut -d " " -f1 | sed "s/^>//" > decoys.txt + cat "$transcripts_name" "$decoys_name" > combined.fasta + fasta=combined.fasta + fi + + salmon index \ + -t "$fasta" \ + -i "~{index_name}" \ + ~{if defined(decoys_fasta) then "-d decoys.txt" else ""} \ + -p "$n_cores" + + tar -czf "~{salmon_index_filename}" "~{index_name}" + + rm -f "$transcripts_name" "$decoys_name" combined.fasta + >>> + + output { + File index_tar_gz = salmon_index_filename + } + + requirements { + cpu: ncpu + # Based on n of 1 test with GRCh38 and GENCODE v50. + memory: "~{if defined(decoys_fasta) then 52 else 5} + 4 + modify_memory_gb} GB" + disks: "~{disk_size_gb} GB" + container: "quay.io/biocontainers/salmon:2.6.0--hfa8f182_0" + maxRetries: 1 + } +} + +task quant { + meta { + description: "Runs `salmon quant` in mapping-based mode to quantify transcript-level expression from RNA-Seq reads, using a pre-built Salmon index." + outputs: { + quant_results_tar_gz: "A gzipped TAR file containing the Salmon quantification output directory, including `quant.sf`.", + quant_sf: "The raw `quant.sf` file, renamed to `.quant.sf`, provided alongside the tarballed output." + } + } + + parameter_meta { + index_tar_gz: "A gzipped TAR file containing the Salmon index files. Suitable as the output of the `index` task." + read_one_fastqs_gz: "An array of gzipped FASTQ files containing read one information" + read_two_fastqs_gz: { + description: "An array of gzipped FASTQ files containing read two information. Omit for single-end reads.", + group: "Common", + } + lib_type: { + description: "Salmon library type describing the relative orientation and strandedness of paired reads.", + help: "Use `A` to let Salmon auto-detect the library type - recommended for most users.", + external_help: "https://combine-lab.github.io/salmon/guides/library-types/", + group: "Common", + } + prefix: { + description: "Prefix for the Salmon quantification output. The extension `.tar.gz` will be added.", + group: "Common", + } + num_bootstraps: { + description: "Compute bootstrapped abundance estimates.", + help: "This is done by resampling (with replacement) from the counts assigned to the fragment equivalence classes, and then re-running the optimization procedure for each such sample.", + group: "Salmon Options", + } + incompat_prior: { + description: "This parameter governs the a priori probability that a fragment mapping is nonetheless the correct mapping.", + help: "Specifically, this is for a fragment mapping or aligning to the reference in a manner incompatible with the prescribed library type.", + group: "Salmon Options", + } + range_factorization_bins: { + description: "The range-factorization feature allows using a data-driven likelihood factorization.", + help: "This can improve quantification accuracy on certain classes of difficult transcripts.", + group: "Salmon Options", + } + fld_mean: { + description: "Allows the user to set the expected mean fragment length of the sequencing library.", + help: "Since the empirical fragment length distribution cannot be estimated from the mappings of single-end reads, this is only important when running Salmon with single-end reads.", + group: "Salmon Options", + } + fld_sd: { + description: "Allows the user to set the expected standard deviation of the fragment length distribution.", + help: "Since the empirical fragment length distribution cannot be estimated from the mappings of single-end reads, this is only important when running Salmon with single-end reads.", + group: "Salmon Options", + } + seq_bias: { + description: "Passing this flag will enable it to learn and correct for sequence-specific biases in the input data.", + group: "Salmon Options", + } + gc_bias: { + description: "Passing this flag will enable it to learn and correct for fragment-level GC biases in the input data.", + group: "Salmon Options", + } + pos_bias: { + description: "Passing this flag will enable modeling of a position-specific fragment start distribution.", + group: "Salmon Options", + } + use_em: { + description: "Use the \"standard\" EM algorithm to optimize abundance estimates instead of the variational Bayesian EM algorithm.", + group: "Salmon Options", + } + recover_orphans: { + description: "This flag enables orphan \"rescue\" for reads.", + group: "Salmon Options", + } + hard_filter: { + description: "This flag turns off soft filtering and range-factorized equivalence classes.", + help: "Removes all but the equally highest scoring mappings from the equivalence class label for each fragment.", + group: "Salmon Options", + } + allow_dovetail: { + description: "Dovetailing mappings and alignments are considered discordant and discarded by default.", + help: "If you wish to consider dovetailing mappings as concordant, you can do so by passing this flag.", + group: "Salmon Options", + } + dump_eq: { + description: "If passed, Salmon will write a file in the auxiliary directory, called eq_classes.txt.", + help: "Contains the equivalence classes and corresponding counts that were computed during quasi-mapping.", + group: "Salmon Options", + } + write_unmapped_names: { + description: "Passing this flag will tell Salmon to write out the names of reads (or mates in paired-end reads) that do not map to the transcriptome.", + group: "Salmon Options", + } + use_all_cores: { + description: "Use all cores? Recommended for cloud environments.", + group: "Resources", + } + ncpu: { + description: "Number of cores to allocate for task", + group: "Resources", + } + modify_disk_size_gb: { + description: "Add to or subtract from dynamic disk space allocation. Default disk size is determined by the size of the inputs. Specified in GB.", + group: "Resources", + } + modify_memory_gb: { + description: "Add to or subtract from dynamic memory allocation. Default memory is determined by the size of the inputs. Specified in GB.", + group: "Resources", + } + } + + input { + File index_tar_gz + Array[File]+ read_one_fastqs_gz + Array[File]? read_two_fastqs_gz + String lib_type = "A" + String prefix = sub(basename(read_one_fastqs_gz[0]), "(([_.][rR](?:ead)?[12])((?:[_.-][^_.-]*?)*?))?\\.(fastq|fq)(\\.gz)?$", "") + Int num_bootstraps = 0 + Float incompat_prior = 0.0 + Int range_factorization_bins = 4 + Int fld_mean = 250 + Int fld_sd = 25 + Boolean seq_bias = false + Boolean gc_bias = false + Boolean pos_bias = false + Boolean use_em = false + Boolean recover_orphans = false + Boolean hard_filter = false + Boolean allow_dovetail = false + Boolean dump_eq = false + Boolean write_unmapped_names = false + Boolean use_all_cores = false + Int ncpu = 4 + Int modify_disk_size_gb = 0 + Int modify_memory_gb = 0 + } + + Array[File] read_twos = select_first([read_two_fastqs_gz, []]) + + Float read_one_size = size(read_one_fastqs_gz, "GB") + Float read_two_size = size(read_twos, "GB") + Float index_size = size(index_tar_gz, "GB") + Int disk_size_gb = ceil((read_one_size + read_two_size + index_size) * 3) + 10 + modify_disk_size_gb + Int memory_gb = ceil(index_size * 4) + 8 + modify_memory_gb + + command <<< + set -euo pipefail + + n_cores=~{ncpu} + if ~{use_all_cores}; then + n_cores=$(nproc) + fi + + mkdir salmon_index + tar -xzf "~{index_tar_gz}" -C salmon_index --strip-components 1 + + # shellcheck disable=SC2086 + salmon quant \ + -i salmon_index \ + -l "~{lib_type}" \ + ~{if length(read_twos) > 0 then "-1 " + sep(" ", squote(read_one_fastqs_gz)) + " -2 " + sep(" ", squote(read_twos)) else "-r " + sep(" ", squote(read_one_fastqs_gz))} \ + -p "$n_cores" \ + --numBootstraps ~{num_bootstraps} \ + --incompatPrior ~{incompat_prior} \ + --rangeFactorizationBins ~{range_factorization_bins} \ + ~{if length(read_twos) == 0 then "--fldMean " + fld_mean else ""} \ + ~{if length(read_twos) == 0 then "--fldSD " + fld_sd else ""} \ + ~{if seq_bias then "--seqBias" else ""} \ + ~{if gc_bias then "--gcBias" else ""} \ + ~{if pos_bias then "--posBias" else ""} \ + ~{if use_em then "--useEM" else ""} \ + ~{if recover_orphans then "--recoverOrphans" else ""} \ + ~{if hard_filter then "--hardFilter" else ""} \ + ~{if allow_dovetail then "--allowDovetail" else ""} \ + ~{if dump_eq then "--dumpEq" else ""} \ + ~{if write_unmapped_names then "--writeUnmappedNames" else ""} \ + -o "~{prefix}" + + cp "~{prefix}/quant.sf" "~{prefix}.quant.sf" + + tar -czf "~{prefix}.tar.gz" "~{prefix}" + >>> + + output { + File quant_results_tar_gz = prefix + ".tar.gz" + File quant_sf = prefix + ".quant.sf" + } + + requirements { + cpu: ncpu + memory: "~{memory_gb} GB" + disks: "~{disk_size_gb} GB" + container: "quay.io/biocontainers/salmon:2.6.0--hfa8f182_0" + maxRetries: 1 + } +} diff --git a/tools/test/salmon.yaml b/tools/test/salmon.yaml new file mode 100644 index 000000000..057e061cb --- /dev/null +++ b/tools/test/salmon.yaml @@ -0,0 +1,48 @@ +index: + - name: builds_index_successfully + tags: + - reference + inputs: + transcripts_fasta: + - reference/gencode.v50.BCR_ABL1.transcripts.fa.gz + assertions: + outputs: + index_tar_gz: + - Name: salmon_index.tar.gz + - name: builds_index_with_decoys + tags: + - reference + - slow + inputs: + transcripts_fasta: + - reference/gencode.v50.BCR_ABL1.transcripts.fa.gz + decoys_fasta: + - reference/GRCh38.chrY_chrM.fa + assertions: + outputs: + index_tar_gz: + - Name: salmon_index.tar.gz + +quant: + - name: quantifies_paired_end_reads + inputs: + index_tar_gz: + - salmon/salmon_index.tar.gz + read_one_fastqs_gz: + - - fastqs/test_R1.fq.gz + read_two_fastqs_gz: + - - fastqs/test_R2.fq.gz + assertions: + outputs: + quant_results_tar_gz: + - Name: test.tar.gz + - name: quantifies_single_end_reads + inputs: + index_tar_gz: + - salmon/salmon_index.tar.gz + read_one_fastqs_gz: + - - fastqs/test_R1.fq.gz + assertions: + outputs: + quant_results_tar_gz: + - Name: test.tar.gz