Merge branch 'temp-update' into 'dev'

Temp update

See merge request epi2melabs/workflows/wf-transcriptomes!167
This commit is contained in:
Sam Nicholls 2024-05-10 08:42:13 +00:00
commit 39cc58ae9e
6 changed files with 359 additions and 99 deletions

View File

@ -3,6 +3,7 @@ import argparse
import glob import glob
import importlib import importlib
import os import os
import sys
from .util import _log_level, get_main_logger # noqa: ABS101 from .util import _log_level, get_main_logger # noqa: ABS101
@ -11,15 +12,17 @@ __version__ = "0.0.1"
_package_name = "workflow_glue" _package_name = "workflow_glue"
def get_components(): def get_components(allowed_components=None):
"""Find a list of workflow command scripts.""" """Find a list of workflow command scripts."""
logger = get_main_logger(_package_name) logger = get_main_logger(_package_name)
path = os.path.dirname(os.path.abspath(__file__)) path = os.path.dirname(os.path.abspath(__file__))
components = list() components = dict()
for fname in glob.glob(os.path.join(path, "*.py")): for fname in glob.glob(os.path.join(path, "*.py")):
name = os.path.splitext(os.path.basename(fname))[0] name = os.path.splitext(os.path.basename(fname))[0]
if name in ("__init__", "util"): if name in ("__init__", "util"):
continue continue
if allowed_components is not None and name not in allowed_components:
continue
# leniently attempt to import module # leniently attempt to import module
try: try:
@ -34,7 +37,7 @@ def get_components():
try: try:
req = "main", "argparser" req = "main", "argparser"
if all(callable(getattr(mod, x)) for x in req): if all(callable(getattr(mod, x)) for x in req):
components.append(name) components[name] = mod
except Exception: except Exception:
pass pass
return components return components
@ -42,6 +45,8 @@ def get_components():
def cli(): def cli():
"""Run workflow entry points.""" """Run workflow entry points."""
logger = get_main_logger(_package_name)
logger.info("Bootstrapping CLI.")
parser = argparse.ArgumentParser( parser = argparse.ArgumentParser(
'wf-glue', 'wf-glue',
parents=[_log_level()], parents=[_log_level()],
@ -56,16 +61,21 @@ def cli():
help='additional help', dest='command') help='additional help', dest='command')
subparsers.required = True subparsers.required = True
# all component demos, plus some others # importing everything can take time, try to shortcut
components = [ if len(sys.argv) > 1:
f'{_package_name}.{comp}' for comp in get_components()] components = get_components(allowed_components=[sys.argv[1]])
for module in components: if not sys.argv[1] in components:
mod = importlib.import_module(module) logger.warn("Importing all modules, this may take some time.")
p = subparsers.add_parser( components = get_components()
module.split(".")[-1], parents=[mod.argparser()]) else:
p.set_defaults(func=mod.main) components = get_components()
# add all module parsers to main CLI
for name, module in components.items():
p = subparsers.add_parser(
name.split(".")[-1], parents=[module.argparser()])
p.set_defaults(func=module.main)
logger = get_main_logger(_package_name)
args = parser.parse_args() args = parser.parse_args()
logger.info("Starting entrypoint.") logger.info("Starting entrypoint.")

View File

@ -8,12 +8,6 @@ import pysam
from .util import get_named_logger, wf_parser # noqa: ABS101 from .util import get_named_logger, wf_parser # noqa: ABS101
def get_sq_hd_lines(xam_file):
"""Extract the `@SQ` and `@HD` lines from the header of a XAM file."""
alignments = pysam.AlignmentFile(xam_file, check_sq=False)
return alignments.header["SQ"], alignments.header["HD"]
def main(args): def main(args):
"""Run the entry point.""" """Run the entry point."""
logger = get_named_logger("checkBamHdr") logger = get_named_logger("checkBamHdr")
@ -33,11 +27,14 @@ def main(args):
mixed_headers = False mixed_headers = False
sorted_xam = False sorted_xam = False
for xam_file in target_files: for xam_file in target_files:
sq_lines, hd_lines = get_sq_hd_lines(xam_file) # get the `@SQ` and `@HD` lines in the header
with pysam.AlignmentFile(xam_file, check_sq=False) as f:
sq_lines = f.header.get("SQ")
hd_lines = f.header.get("HD")
# Check if it is sorted. # Check if it is sorted.
# When there is more than one BAM, merging/sorting # When there is more than one BAM, merging/sorting
# will happen regardless of this flag. # will happen regardless of this flag.
if hd_lines.get('SO') == 'coordinate': if hd_lines is not None and hd_lines.get('SO') == 'coordinate':
sorted_xam = True sorted_xam = True
if first_sq_lines is None: if first_sq_lines is None:
# this is the first file # this is the first file

View File

@ -0,0 +1,166 @@
"""Extract unique values for a key from XAM RG DS headers or FASTX comments.
Use pysam to read the description tag of XAM read group header(s) to
collect values for a given key and check the expected cardinality from
one or more XAM; or equivalently the comments of one or more FASTX to
do the same.
Use for example, to extract a single basecaller configuration name in
order to match the input data to suitable models for downstream tools
without troubling the user. No guarantee is made for ordering.
"""
import os
import sys
import pysam
from .util import wf_parser # noqa: ABS101
# This is not my ideal way to raise nice user-facing errors, but embedding them in the
# Nextflow process itself is (a) confusing for users; as the error log includes echo
# commands which may or may not actually be printed and (b) a footgun for developers;
# who may inadvertently mishandle catching and then (re)returning a bad exit code.
# This also rather neatly keeps intended messaging for users near the code that will
# cause errors to be raised.
def get_extended_errmsg(key, expected_cardinality):
"""Get an applicable extended error message for a given key and cardinality."""
if key == "basecall_model" and expected_cardinality == "zero-or-one":
return """
################################################################################
# INPUT DATA PROBLEM
Your input data contains reads basecalled with more than one basecaller model.
Our workflows automatically select appropriate configuration and models for
downstream tools for a given basecaller model. This cannot be done reliably when
reads with different basecaller models are mixed in the same data set.
## Next steps
To use this workflow you must separate your input files, making sure all reads
are have been basecalled with the same basecaller model.
################################################################################
"""
def path_to_lofn(input_path):
"""Convert the input path to a list of one or more files to be checked."""
if os.path.isdir(input_path):
return [
os.path.join(root, f)
for (root, dirnames, filenames) in os.walk(input_path)
for f in filenames
]
else:
return [input_path]
def xam_extract_ds_key(xam_lofn, key):
"""Extract the set of values for a given key from all RG DS tags."""
entries = set()
for xam_fn in xam_lofn:
with pysam.AlignmentFile(xam_fn, check_sq=False) as xam:
for read_group in xam.header.get("RG", []):
for ds_kv in read_group.get("DS", "").split():
k, v = ds_kv.split("=", 1)
if k == key:
entries.add(v)
return entries
def fastx_extract_ds_key(fastx_lofn, key, stop_after=0):
"""Extract the set of values for a given key from all FASTQ tags."""
entries = set()
for fastx_fn in fastx_lofn:
with pysam.FastxFile(fastx_fn) as fastx:
for i, read in enumerate(fastx):
if stop_after > 0 and i > stop_after:
break
for ds_kv in read.comment.split():
k, v = ds_kv.split("=", 1)
if k == key:
entries.add(v)
return entries
def check_cardinality(obs, desired_cardinality):
"""Return whether the observed cardinality meets the desired cardinality."""
cardinality_lookup = {
0: ["zero", "zero-or-one", "zero-or-more"],
1: ["zero-or-one", "zero-or-more", "one", "one-or-more"],
2: ["zero-or-more", "one-or-more", "more-than-one"],
}
if obs > 1:
obs = 2
return desired_cardinality in cardinality_lookup[obs]
def main(args):
"""Script entrypoint.
Extracts values using the XAM or FASTX extractor and checks the set
is of the right cardinality. Prints a message to stdout (or stderr)
and exits appropriately.
"""
if args.xam:
extractor = xam_extract_ds_key
input_path = args.xam
elif args.fastx:
extractor = fastx_extract_ds_key
input_path = args.fastx
lofn = path_to_lofn(input_path)
entries = extractor(lofn, args.key)
if not check_cardinality(len(entries), args.cardinality):
sys.stdout.write(args.sep.join(entries) + '\n')
extended_error = get_extended_errmsg(args.key, args.cardinality)
if args.explode_obviously and extended_error:
sys.stderr.write(extended_error)
else:
sys.stderr.write(
f"Required {args.cardinality} {args.key} but found {len(entries)}\n"
)
sys.exit(os.EX_DATAERR)
sys.stdout.write(args.sep.join(entries) + '\n')
sys.exit(os.EX_OK)
def argparser():
"""Argument parser for entrypoint."""
parser = wf_parser("get_ds_records")
parser.add_argument("--key", required=True)
input_arg = parser.add_mutually_exclusive_group(required=True)
input_arg.add_argument("--xam", help="Path to a single XAM or folder of XAM")
input_arg.add_argument("--fastx", help="Path to a single FASTX or folder of FASTX")
parser.add_argument(
"--sep",
default="\n",
help=(
"Value separator to use if more than one element is printed to stdout."
),
)
parser.add_argument(
"--cardinality",
default="zero-or-more",
choices=[
"zero",
"zero-or-one",
"zero-or-more",
"one",
"one-or-more",
"more-than-one",
],
help=(
"Expected cardinality of entries. Will exit EX_DATAERR if the wrong "
"number of elements are found. Defaults to zero-or-more."
),
)
parser.add_argument(
"--explode_obviously",
action="store_true",
help=(
"If appropriate, print a more fulsome user facing error if the expected"
"cardinality has been violated."
)
)
return parser

View File

@ -124,6 +124,7 @@ def add_number_of_reads_to_meta(ch, String input_type_format) {
* - "fastcat_extra_args": string with extra arguments to pass to `fastcat` * - "fastcat_extra_args": string with extra arguments to pass to `fastcat`
* - "required_sample_types": list of required sample types in the sample sheet * - "required_sample_types": list of required sample types in the sample sheet
* - "watch_path": boolean whether to use `watchPath` and run in streaming mode * - "watch_path": boolean whether to use `watchPath` and run in streaming mode
* - "fastq_chunk": null or a number of reads to place into chunked FASTQ files
* @return: channel of `[Map(alias, barcode, type, ...), Path|null, Path|null]`. * @return: channel of `[Map(alias, barcode, type, ...), Path|null, Path|null]`.
* The first element is a map with metadata, the second is the path to the * The first element is a map with metadata, the second is the path to the
* `.fastq.gz` file with the (potentially concatenated) sequences and the third is * `.fastq.gz` file with the (potentially concatenated) sequences and the third is
@ -134,7 +135,15 @@ def add_number_of_reads_to_meta(ch, String input_type_format) {
def fastq_ingress(Map arguments) def fastq_ingress(Map arguments)
{ {
// check arguments // check arguments
Map margs = parse_arguments("fastq_ingress", arguments, ["fastcat_extra_args": ""]) Map margs = parse_arguments(
"fastq_ingress", arguments,
[
"fastcat_extra_args": "",
"fastq_chunk": null,
"per_read_stats": false
]
)
margs["fastq_chunk"] ?= 0 // cant pass null through channel
ArrayList fq_extensions = [".fastq", ".fastq.gz", ".fq", ".fq.gz"] ArrayList fq_extensions = [".fastq", ".fastq.gz", ".fq", ".fq.gz"]
@ -144,22 +153,52 @@ def fastq_ingress(Map arguments)
def ch_result def ch_result
if (margs.stats) { if (margs.stats) {
// run fastcat regardless of input type // run fastcat regardless of input type
ch_result = fastcat(input.files.mix(input.dirs), margs["fastcat_extra_args"]) ch_result = fastcat(input.files.mix(input.dirs), margs, "FASTQ")
} else { } else {
// run `fastcat` only on directories and rename / compress single files // run `fastcat` only on directories and rename / compress single files
ch_result = fastcat(input.dirs, margs["fastcat_extra_args"]) ch_dir = fastcat(input.dirs, margs, "FASTQ")
| mix( .map { meta, path, stats -> [meta, path] }
input.files def ch_file
| move_or_compress_fq_file if (margs["fastq_chunk"] > 0) {
ch_file = split_fq_file(input.files, margs["fastq_chunk"])
} else {
ch_file = move_or_compress_fq_file(input.files)
}
ch_result = ch_dir
| mix(ch_file)
| map { meta, path -> [meta, path, null] } | map { meta, path -> [meta, path, null] }
)
} }
// add sample sheet entries without barcode dirs to the results channel and extract // TODO: xam_ingress mixes in a .no_files channel here. Do we need to do the same?
// the run IDs into the metamaps before returning
ch_result = ch_result.mix(input.missing.map { [*it, null] }) // The above may have returned a channel with multiple fastqs if chunking
ch_result_run_IDs = add_run_IDs_to_meta(ch_result) // is enabled. Flatten this and add a groupKey to meta information which
// add number of reads after potential filtering under the field n_seqs // states the number of sibling files. This can be later used as the key
return add_number_of_reads_to_meta(ch_result_run_IDs, "fastq") // for .groupTuple() on a channel in order to get all results for a sample
// We don't decorate "alias" with a count because that messes up downstream
// serialisation.
// Mix in the missing files from the sample sheet
// Add in a unique key for every emission
def ch_spread_result = ch_result
.mix (input.missing.map { meta, files -> [meta, files, null] })
.map { meta, files, stats ->
// new `arity: '1..*'` would be nice here
files = files instanceof List ? files : [files]
new_keys = [
"group_key": groupKey(meta["alias"], files.size()),
"n_fastq": files.size()]
grp_index = (0..<files.size()).collect()
[meta + new_keys, files, grp_index, stats]
}
.transpose(by: [1, 2]) // spread multiple fastq files into separate emissions
.map { meta, files, grp_i, stats ->
new_keys = [
"group_index": "${meta["alias"]}_${grp_i}"]
[meta + new_keys, files, stats]
}
def ch_final = add_number_of_reads_to_meta(
add_run_IDs_to_meta(ch_spread_result), "fastq")
return ch_final
} }
@ -197,10 +236,16 @@ def xam_ingress(Map arguments)
{ {
// check arguments // check arguments
Map margs = parse_arguments( Map margs = parse_arguments(
"xam_ingress", "xam_ingress", arguments,
arguments, [
["keep_unaligned": false, "return_fastq": false, "fastcat_extra_args": ""] "keep_unaligned": false,
"return_fastq": false,
"fastcat_extra_args": "",
"fastq_chunk": null,
"per_read_stats": false
]
) )
margs["fastq_chunk"] ?= 0 // cant pass null through channel
// we only accept BAM or uBAM for now (i.e. no SAM or CRAM) // we only accept BAM or uBAM for now (i.e. no SAM or CRAM)
ArrayList xam_extensions = [".bam", ".ubam"] ArrayList xam_extensions = [".bam", ".ubam"]
@ -231,7 +276,7 @@ def xam_ingress(Map arguments)
[meta + [xai_fn: xai_fn], paths] [meta + [xai_fn: xai_fn], paths]
} }
| checkBamHeaders | checkBamHeaders
| map { meta, paths, is_unaligned_env, mixed_headers_env, is_sorted_env -> | map { meta, paths, is_unaligned_env, mixed_headers_env, is_sorted_env, ds_basecaller_env, ds_runids_env ->
// convert the env. variables from strings ('0' or '1') into bools // convert the env. variables from strings ('0' or '1') into bools
boolean is_unaligned = is_unaligned_env as int as boolean boolean is_unaligned = is_unaligned_env as int as boolean
boolean mixed_headers = mixed_headers_env as int as boolean boolean mixed_headers = mixed_headers_env as int as boolean
@ -243,7 +288,15 @@ def xam_ingress(Map arguments)
// add `is_unaligned` to the metamap (note the use of `+` to create a copy of // add `is_unaligned` to the metamap (note the use of `+` to create a copy of
// `meta` to avoid modifying every item in the channel; // `meta` to avoid modifying every item in the channel;
// https://github.com/nextflow-io/nextflow/issues/2660) // https://github.com/nextflow-io/nextflow/issues/2660)
[meta + [is_unaligned: is_unaligned, is_sorted: is_sorted], paths] [
meta + [
is_unaligned: is_unaligned,
is_sorted: is_sorted,
ds_runids: ds_runids_env.tokenize(','),
ds_basecall_models: ds_basecaller_env.tokenize(','),
],
paths
]
} }
| branch { meta, paths -> | branch { meta, paths ->
// set `paths` to `null` for uBAM samples if unallowed (they will be added to // set `paths` to `null` for uBAM samples if unallowed (they will be added to
@ -285,17 +338,37 @@ def xam_ingress(Map arguments)
ch_result.to_merge, ch_result.to_merge,
ch_result.to_catsort ch_result.to_catsort
) )
// TODO: this is largely similar to fastq_ingress, should be refactored
// input.missing: sample sheet entries without barcode dirs // input.missing: sample sheet entries without barcode dirs
ch_result = input.missing def ch_spread_result = input.missing
| mix(ch_result.no_files) .mix(ch_result.no_files) // TODO: we don't have this in fastq_ingress?
| map { [*it, null] } .map { meta, files -> [meta, files, null] }
| mix(bamToFastq(ch_to_fastq, margs["fastcat_extra_args"])) .mix(
| map{ fastcat(ch_to_fastq, margs, "BAM")
meta, path, stats -> )
[meta.findAll { it.key !in ['xai_fn', 'is_sorted'] }, path, stats] .map { meta, files, stats ->
} // new `arity: '1..*'` would be nice here
return add_number_of_reads_to_meta(add_run_IDs_to_meta(ch_result), "fastq") files = files instanceof List ? files : [files]
new_keys = [
"group_key": groupKey(meta["alias"], files.size()),
"n_fastq": files.size()]
grp_index = (0..<files.size()).collect()
[meta + new_keys, files, grp_index, stats]
}
.transpose(by: [1, 2]) // spread multiple fastq files into separate emissions
.map { meta, files, grp_i, stats ->
new_keys = [
"group_index": "${meta["alias"]}_${grp_i}"]
[meta + new_keys, files, stats]
}
.map { meta, path, stats ->
[meta.findAll { it.key !in ['xai_fn', 'is_sorted'] }, path, stats]
}
def ch_final = add_number_of_reads_to_meta(
add_run_IDs_to_meta(ch_spread_result), "fastq")
return ch_final
} }
// deal with samples with few-enough files for `samtools merge` first // deal with samples with few-enough files for `samtools merge` first
@ -373,7 +446,7 @@ def xam_ingress(Map arguments)
has_reads: path has_reads: path
is_null: true is_null: true
} }
ch_bamstats = bamstats(ch_result.has_reads) ch_bamstats = bamstats(ch_result.has_reads, margs)
// the channel comes from xam_ingress also have the BAM index in it. // the channel comes from xam_ingress also have the BAM index in it.
// Handle this by placing them in a nested array, maintaining the structure // Handle this by placing them in a nested array, maintaining the structure
@ -425,40 +498,57 @@ def xam_ingress(Map arguments)
return ch_result return ch_result
} }
process bamToFastq {
process fastcat {
label "ingress" label "ingress"
label "wf_common" label "wf_common"
cpus 4 cpus 4
memory "2 GB" memory "2 GB"
input: input:
tuple val(meta), path(bams, stageAs: "input_dir/reads*.bam") tuple val(meta), path(input_src, stageAs: "input_src")
val extra_args val fcargs
output: tuple val(meta), path("seqs.fastq.gz"), path("fastcat_stats") val src
output:
tuple val(meta),
path("fastq_chunks/*.fastq.gz"), // TODO: change this to use new arity: '1..*'
path("fastcat_stats")
script: script:
Integer lines_per_chunk = fcargs["fastq_chunk"] != 0 ? fcargs["fastq_chunk"] * 4 : null
def input_src = src == "FASTQ"
? "input_src"
: """<(
samtools cat -b <(find . -name 'input_src*') | \
samtools fastq - -n -T '*' -o - -0 -
)"""
def stats_args = fcargs["per_read_stats"] ? "-r >(bgzip -c > fastcat_stats/per-read-stats.tsv.gz)" : ""
""" """
mkdir fastcat_stats mkdir fastcat_stats
mkdir fastq_chunks
# Save file as compressed fastq # Save file as compressed fastq
fastcat \ fastcat \
-s ${meta["alias"]} \ -s ${meta["alias"]} \
-r >(bgzip -c > fastcat_stats/per-read-stats.tsv.gz) \
-f fastcat_stats/per-file-stats.tsv \ -f fastcat_stats/per-file-stats.tsv \
-i fastcat_stats/per-file-runids.txt \
--histograms histograms \ --histograms histograms \
$extra_args \ $stats_args \
<( ${fcargs["fastcat_extra_args"]} \
samtools cat -b <(find input_dir -name 'reads*.bam') | \ $input_src \
samtools fastq - -n -T '*' -o - -0 - | if [ "${fcargs["fastq_chunk"]}" = "0" ]; then
) \ bgzip -@ $task.cpus > fastq_chunks/seqs.fastq.gz
| bgzip -c > seqs.fastq.gz else
split -l $lines_per_chunk -d --additional-suffix=.fastq.gz --filter='bgzip -@ $task.cpus > \$FILE' - fastq_chunks/seqs_;
fi
mv histograms/* fastcat_stats mv histograms/* fastcat_stats
# extract the run IDs and number of sequences (n_seqs) from the per-read stats # get n_seqs from per-file stats - need to sum them up
csvtk freq -tf runid fastcat_stats/per-read-stats.tsv.gz \ awk 'NR==1{for (i=1; i<=NF; i++) {ix[\$i] = i}} NR>1 {c+=\$ix["n_seqs"]} END{print c}' \
| csvtk del-header \ fastcat_stats/per-file-stats.tsv > fastcat_stats/n_seqs
| tee >(cut -f 1 | sort > "fastcat_stats/run_ids") \ # get unique run IDs
| awk 'BEGIN{n=0}; {n+=\$2}; END{print n}' > "fastcat_stats/n_seqs" awk 'NR==1{for (i=1; i<=NF; i++) {ix[\$i] = i}} NR>1 {print \$ix["run_id"]}' \
""" fastcat_stats/per-file-runids.txt | sort | uniq > fastcat_stats/run_ids
"""
} }
process checkBamHeaders { process checkBamHeaders {
@ -468,19 +558,21 @@ process checkBamHeaders {
memory "2 GB" memory "2 GB"
input: tuple val(meta), path("input_dir/reads*.bam") input: tuple val(meta), path("input_dir/reads*.bam")
output: output:
// set the two env variables by `eval`-ing the output of the python script
// checking the XAM headers
tuple( tuple(
val(meta), val(meta),
path("input_dir/reads*.bam", includeInputs: true), path("input_dir/reads*.bam", includeInputs: true),
env(IS_UNALIGNED), env(IS_UNALIGNED),
env(MIXED_HEADERS), env(MIXED_HEADERS),
env(IS_SORTED), env(IS_SORTED),
env(DS_BASECALL_MODELS),
env(DS_RUNIDS),
) )
script: script:
""" """
workflow-glue check_bam_headers_in_dir input_dir > env.vars workflow-glue check_bam_headers_in_dir input_dir > env.vars
source env.vars source env.vars
DS_RUNIDS=\$(workflow-glue get_ds_records --xam input_dir --key runid --cardinality zero-or-more --sep ',')
DS_BASECALL_MODELS=\$(workflow-glue get_ds_records --xam input_dir --key basecall_model --cardinality zero-or-one --sep ',' --explode_obviously)
""" """
} }
@ -562,6 +654,7 @@ process bamstats {
memory "4 GB" memory "4 GB"
input: input:
tuple val(meta), path("reads.bam"), path("reads.bam.bai") tuple val(meta), path("reads.bam"), path("reads.bam.bai")
val bsargs
output: output:
tuple val(meta), tuple val(meta),
path("reads.bam"), path("reads.bam"),
@ -569,17 +662,22 @@ process bamstats {
path("bamstats_results") path("bamstats_results")
script: script:
def bamstats_threads = Math.max(1, task.cpus - 1) def bamstats_threads = Math.max(1, task.cpus - 1)
def per_read_stats_arg = bsargs["per_read_stats"] ? "| bgzip > bamstats_results/bamstats.readstats.tsv.gz" : " > /dev/null"
""" """
mkdir bamstats_results mkdir bamstats_results
bamstats reads.bam -s $meta.alias -u \ bamstats reads.bam -s $meta.alias -u \
-f bamstats_results/bamstats.flagstat.tsv -t $bamstats_threads \ -f bamstats_results/bamstats.flagstat.tsv -t $bamstats_threads \
-i bamstats_results/bamstats.runids.txt \
--histograms histograms \ --histograms histograms \
| bgzip > bamstats_results/bamstats.readstats.tsv.gz $per_read_stats_arg
mv histograms/* bamstats_results/ mv histograms/* bamstats_results/
# extract the run IDs from the per-read stats # get n_seqs from flagstats - need to sum them up
csvtk cut -tf runid bamstats_results/bamstats.readstats.tsv.gz \ awk 'NR==1{for (i=1; i<=NF; i++) {ix[\$i] = i}} NR>1 {c+=\$ix["total"]} END{print c}' \
| csvtk del-header | sort | uniq > bamstats_results/run_ids bamstats_results/bamstats.flagstat.tsv > bamstats_results/n_seqs
# get unique run IDs
awk 'NR==1{for (i=1; i<=NF; i++) {ix[\$i] = i}} NR>1 {print \$ix["run_id"]}' \
bamstats_results/bamstats.runids.txt | sort | uniq > bamstats_results/run_ids
""" """
} }
/** /**
@ -701,38 +799,25 @@ process move_or_compress_fq_file {
} }
process fastcat { process split_fq_file {
label "ingress" label "ingress"
label "wf_common" label "wf_common"
cpus 3 cpus 1
memory "2 GB" memory "2 GB"
input: input:
tuple val(meta), path("input") // don't stage `input` with a literal because we check the file extension
val extra_args tuple val(meta), path(input)
val fastq_chunk
output: output:
tuple val(meta), tuple val(meta), path("fastq_chunks/*.fastq.gz") // TODO: change this to use new arity: '1..*'
path("seqs.fastq.gz"),
path("fastcat_stats")
script: script:
String out = "seqs.fastq.gz" String cat = input.name.endsWith('.gz') ? "zcat" : "cat"
String fastcat_stats_outdir = "fastcat_stats" Integer lines_per_chunk = fastq_chunk * 4
""" """
mkdir $fastcat_stats_outdir mkdir fastq_chunks
fastcat \ $cat "$input" \
-s ${meta["alias"]} \ | split -l $lines_per_chunk -d --additional-suffix=.fastq.gz --filter='bgzip \
-r >(bgzip -c > $fastcat_stats_outdir/per-read-stats.tsv.gz) \ > \$FILE' - fastq_chunks/seqs_
-f $fastcat_stats_outdir/per-file-stats.tsv \
--histograms histograms \
$extra_args \
input \
| bgzip > $out
mv histograms/* $fastcat_stats_outdir
# extract the run IDs and number of sequences (n_seqs) from the per-read stats
csvtk freq -tf runid $fastcat_stats_outdir/per-read-stats.tsv.gz \
| csvtk del-header \
| tee >(cut -f 1 | sort > "$fastcat_stats_outdir/run_ids") \
| awk 'BEGIN{n=0}; {n+=\$2}; END{print n}' > "$fastcat_stats_outdir/n_seqs"
""" """
} }
@ -754,7 +839,8 @@ Map parse_arguments(String func_name, Map arguments, Map extra_kwargs=[:]) {
"analyse_unclassified": false, "analyse_unclassified": false,
"stats": true, "stats": true,
"required_sample_types": [], "required_sample_types": [],
"watch_path": false "watch_path": false,
"per_read_stats": false
] ]
ArgumentParser parser = new ArgumentParser( ArgumentParser parser = new ArgumentParser(
args: required_args, args: required_args,

View File

@ -815,7 +815,8 @@ workflow {
"sample_sheet":params.sample_sheet, "sample_sheet":params.sample_sheet,
"analyse_unclassified":params.analyse_unclassified, "analyse_unclassified":params.analyse_unclassified,
"stats": true, "stats": true,
"fastcat_extra_args": ""]) "fastcat_extra_args": "",
"per_read_stats": true])
pipeline(reads, ref_genome, ref_annotation, pipeline(reads, ref_genome, ref_annotation,
jaffal_refBase, params.jaffal_genome, params.jaffal_annotation, jaffal_refBase, params.jaffal_genome, params.jaffal_annotation,

View File

@ -105,7 +105,7 @@ params {
] ]
agent = null agent = null
container_sha = "shae7c9f184996a384e99be68e790f0612f0c732867" container_sha = "shae7c9f184996a384e99be68e790f0612f0c732867"
common_sha = "sha645176f98b8780851f9c476a064d44c2ae76ddf6" common_sha = "sha91cd87900c86f05bf36d8c77b841b8fda5ecf3aa"
} }
} }