Merge branch 'CW-3174' into 'dev'
No ref annotation workflow still completes Closes CW-3174 See merge request epi2melabs/workflows/wf-transcriptomes!154
This commit is contained in:
commit
84dbafd1c9
@ -52,7 +52,7 @@ docker-run:
|
|||||||
"fusions", "differential_expression", "isoforms",
|
"fusions", "differential_expression", "isoforms",
|
||||||
"only_differential_expression", "differential_expression_gff3",
|
"only_differential_expression", "differential_expression_gff3",
|
||||||
"ncbi_gzip", "ncbi_no_gene_id", "ensembl_with_versions",
|
"ncbi_gzip", "ncbi_no_gene_id", "ensembl_with_versions",
|
||||||
"differential_expression_mouse"
|
"differential_expression_mouse", "no_ref_annotation"
|
||||||
]
|
]
|
||||||
rules:
|
rules:
|
||||||
# NOTE As we're overriding the rules block for the included docker-run
|
# NOTE As we're overriding the rules block for the included docker-run
|
||||||
@ -66,6 +66,12 @@ docker-run:
|
|||||||
NF_WORKFLOW_OPTS: "--fastq ERR6053095_chr20.fastq --transcriptome-source reference-guided \
|
NF_WORKFLOW_OPTS: "--fastq ERR6053095_chr20.fastq --transcriptome-source reference-guided \
|
||||||
--ref_genome chr20/hg38_chr20.fa --ref_annotation chr20/gencode.v22.annotation.chr20.gtf --pychopper_backend phmm"
|
--ref_genome chr20/hg38_chr20.fa --ref_annotation chr20/gencode.v22.annotation.chr20.gtf --pychopper_backend phmm"
|
||||||
NF_IGNORE_PROCESSES: preprocess_reads,merge_transcriptomes,decompress_annotation,decompress_ref,decompress_transcriptome,preprocess_ref_transcriptome
|
NF_IGNORE_PROCESSES: preprocess_reads,merge_transcriptomes,decompress_annotation,decompress_ref,decompress_transcriptome,preprocess_ref_transcriptome
|
||||||
|
- if: $MATRIX_NAME == "no_ref_annotation"
|
||||||
|
variables:
|
||||||
|
NF_BEFORE_SCRIPT: wget -O test_data.tar.gz https://ont-exd-int-s3-euwst1-epi2me-labs.s3.amazonaws.com/wf-isoforms/wf-isoforms_test_data.tar.gz && tar -xzvf test_data.tar.gz
|
||||||
|
NF_WORKFLOW_OPTS: "--fastq ERR6053095_chr20.fastq --transcriptome-source reference-guided \
|
||||||
|
--ref_genome chr20/hg38_chr20.fa"
|
||||||
|
NF_IGNORE_PROCESSES: run_gffcompare,preprocess_reads,merge_transcriptomes,decompress_annotation,decompress_ref,decompress_transcriptome,preprocess_ref_transcriptome
|
||||||
- if: $MATRIX_NAME == "fusions"
|
- if: $MATRIX_NAME == "fusions"
|
||||||
variables:
|
variables:
|
||||||
NF_BEFORE_SCRIPT: wget -O test_data.tar.gz https://ont-exd-int-s3-euwst1-epi2me-labs.s3.amazonaws.com/wf-isoforms/wf-isoforms_test_data.tar.gz && tar -xzvf test_data.tar.gz
|
NF_BEFORE_SCRIPT: wget -O test_data.tar.gz https://ont-exd-int-s3-euwst1-epi2me-labs.s3.amazonaws.com/wf-isoforms/wf-isoforms_test_data.tar.gz && tar -xzvf test_data.tar.gz
|
||||||
|
|||||||
@ -16,6 +16,7 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
|
|||||||
- Add gene name column to the de_analysis counts TSV files.
|
- Add gene name column to the de_analysis counts TSV files.
|
||||||
### Fixed
|
### Fixed
|
||||||
- Mapping stage using a single thread only.
|
- Mapping stage using a single thread only.
|
||||||
|
- When no `--ref_annotation` is provided the workflow will still run but the output transcripts will not be annotated. However `--de_analysis` mode still requires a `--ref_annotation`.
|
||||||
|
|
||||||
## [v1.0.0]
|
## [v1.0.0]
|
||||||
### Added
|
### Added
|
||||||
|
|||||||
75
main.nf
75
main.nf
@ -323,8 +323,6 @@ process run_gffcompare{
|
|||||||
"""
|
"""
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
process get_transcriptome{
|
process get_transcriptome{
|
||||||
/*
|
/*
|
||||||
Write out a transcriptome file based on the query gff annotations.
|
Write out a transcriptome file based on the query gff annotations.
|
||||||
@ -340,6 +338,9 @@ process get_transcriptome{
|
|||||||
script:
|
script:
|
||||||
def transcriptome = "${sample_id}_transcriptome.fas"
|
def transcriptome = "${sample_id}_transcriptome.fas"
|
||||||
def merged_transcriptome = "${sample_id}_merged_transcriptome.fas"
|
def merged_transcriptome = "${sample_id}_merged_transcriptome.fas"
|
||||||
|
// if no ref_annotation gffcmp_dir will be optional file
|
||||||
|
// so skip getting transcriptome FASTA from the annotated files.
|
||||||
|
if (params.ref_annotation){
|
||||||
"""
|
"""
|
||||||
gffread -g ${reference_seq} -w ${transcriptome} ${transcripts_gff}
|
gffread -g ${reference_seq} -w ${transcriptome} ${transcripts_gff}
|
||||||
if [ "\$(ls -A $gffcmp_dir)" ];
|
if [ "\$(ls -A $gffcmp_dir)" ];
|
||||||
@ -347,6 +348,11 @@ process get_transcriptome{
|
|||||||
gffread -F -g ${reference_seq} -w ${merged_transcriptome} $gffcmp_dir/str_merged.annotated.gtf
|
gffread -F -g ${reference_seq} -w ${merged_transcriptome} $gffcmp_dir/str_merged.annotated.gtf
|
||||||
fi
|
fi
|
||||||
"""
|
"""
|
||||||
|
} else {
|
||||||
|
"""
|
||||||
|
gffread -g ${reference_seq} -w ${transcriptome} ${transcripts_gff}
|
||||||
|
"""
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
process merge_transcriptomes {
|
process merge_transcriptomes {
|
||||||
@ -397,11 +403,11 @@ process makeReport {
|
|||||||
// for DE analysis, a `gene_name` column will be added to
|
// for DE analysis, a `gene_name` column will be added to
|
||||||
// `de_report/results_dge.tsv`
|
// `de_report/results_dge.tsv`
|
||||||
path "results_dge.tsv", emit: de_analysis, optional: true
|
path "results_dge.tsv", emit: de_analysis, optional: true
|
||||||
script:
|
shell:
|
||||||
// Convert the sample_id arrayList.
|
// Convert the sample_id arrayList.
|
||||||
sids = new BlankSeparatedList(sample_ids)
|
sids = new BlankSeparatedList(sample_ids)
|
||||||
def report_name = "wf-transcriptomes-report.html"
|
report_name = "wf-transcriptomes-report.html"
|
||||||
"""
|
'''
|
||||||
if [ -f "de_report/OPTIONAL_FILE" ]; then
|
if [ -f "de_report/OPTIONAL_FILE" ]; then
|
||||||
dereport=""
|
dereport=""
|
||||||
else
|
else
|
||||||
@ -409,10 +415,14 @@ process makeReport {
|
|||||||
mv de_report/*.g*f* de_report/stringtie_merged.gtf
|
mv de_report/*.g*f* de_report/stringtie_merged.gtf
|
||||||
fi
|
fi
|
||||||
if [ -f "gff_annotation/OPTIONAL_FILE" ]; then
|
if [ -f "gff_annotation/OPTIONAL_FILE" ]; then
|
||||||
OPT_GFF=""
|
OPT_GFF_ANNOTATION=""
|
||||||
else
|
else
|
||||||
OPT_GFF="--gffcompare_dir ${gffcmp_dir} --gff_annotation gff_annotation/*"
|
OPT_GFF_ANNOTATION="--gff_annotation gff_annotation/*"
|
||||||
|
fi
|
||||||
|
if [ -f "OPTIONAL_FILE" ]; then
|
||||||
|
OPT_GFFCMP_DIR=""
|
||||||
|
else
|
||||||
|
OPT_GFFCMP_DIR="--gffcompare_dir !{gffcmp_dir}"
|
||||||
fi
|
fi
|
||||||
if [ -f "jaffal_csv/OPTIONAL_FILE" ]; then
|
if [ -f "jaffal_csv/OPTIONAL_FILE" ]; then
|
||||||
OPT_JAFFAL_CSV=""
|
OPT_JAFFAL_CSV=""
|
||||||
@ -429,18 +439,19 @@ process makeReport {
|
|||||||
else
|
else
|
||||||
OPT_PC_REPORT="--pychop_report pychopper_report/*"
|
OPT_PC_REPORT="--pychop_report pychopper_report/*"
|
||||||
fi
|
fi
|
||||||
workflow-glue report --report $report_name \
|
workflow-glue report --report !{report_name} \
|
||||||
--versions $versions \
|
--versions !{versions} \
|
||||||
--params params.json \
|
--params params.json \
|
||||||
\$OPT_ALN \
|
${OPT_ALN} \
|
||||||
\$OPT_PC_REPORT \
|
${OPT_PC_REPORT} \
|
||||||
--sample_ids $sids \
|
--sample_ids !{sids} \
|
||||||
--stats per_read_stats/* \
|
--stats per_read_stats/* \
|
||||||
\$OPT_GFF \
|
${OPT_GFF_ANNOTATION} \
|
||||||
--isoform_table_nrows $params.isoform_table_nrows \
|
${OPT_GFFCMP_DIR} \
|
||||||
\$OPT_JAFFAL_CSV \
|
--isoform_table_nrows !{params.isoform_table_nrows} \
|
||||||
\$dereport
|
${OPT_JAFFAL_CSV} \
|
||||||
"""
|
${dereport}
|
||||||
|
'''
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
@ -582,18 +593,28 @@ workflow pipeline {
|
|||||||
assemble_transcripts(split_bam.out.bundles.flatMap(map_sample_ids_cls).combine(ref_annotation),use_ref_ann)
|
assemble_transcripts(split_bam.out.bundles.flatMap(map_sample_ids_cls).combine(ref_annotation),use_ref_ann)
|
||||||
|
|
||||||
merge_gff_bundles(assemble_transcripts.out.gff_bundles.groupTuple())
|
merge_gff_bundles(assemble_transcripts.out.gff_bundles.groupTuple())
|
||||||
run_gffcompare(merge_gff_bundles.out.gff, ref_annotation)
|
// only run gffcompare if ref annotation provided. Otherwise create optional files and channels
|
||||||
|
if (params.ref_annotation){
|
||||||
|
run_gffcompare(merge_gff_bundles.out.gff, ref_annotation)
|
||||||
|
gff_compare_dir = run_gffcompare.out.gffcmp_dir
|
||||||
|
gff_compare = run_gffcompare.out.gffcmp_dir.map{ it -> it[1]}.collect()
|
||||||
|
// create per sample gff tuples with gff compare directories
|
||||||
|
gff_tuple = merge_gff_bundles.out.gff
|
||||||
|
.join(gff_compare_dir)
|
||||||
|
} else {
|
||||||
|
// create per sample gff tuples with optional files as no ref_annotation
|
||||||
|
optional_channel = Channel.fromPath("$projectDir/data/OPTIONAL_FILE")
|
||||||
|
gff_tuple = merge_gff_bundles.out.gff.combine(optional_channel)
|
||||||
|
gff_compare = OPTIONAL_FILE
|
||||||
|
}
|
||||||
// For reference based assembly, there is only one reference
|
// For reference based assembly, there is only one reference
|
||||||
// So map this reference to all sample_ids
|
// So map this reference to all sample_ids
|
||||||
seq_for_transcriptome_build = sample_ids.flatten().combine(ref_genome)
|
seq_for_transcriptome_build = sample_ids.flatten().combine(ref_genome)
|
||||||
|
|
||||||
|
|
||||||
get_transcriptome(
|
get_transcriptome(
|
||||||
merge_gff_bundles.out.gff
|
gff_tuple
|
||||||
.join(run_gffcompare.out.gffcmp_dir)
|
|
||||||
.join(seq_for_transcriptome_build))
|
.join(seq_for_transcriptome_build))
|
||||||
|
|
||||||
gff_compare = run_gffcompare.out.gffcmp_dir.map{ it -> it[1]}.collect()
|
|
||||||
merge_gff = merge_gff_bundles.out.gff.map{ it -> it[1]}.collect()
|
merge_gff = merge_gff_bundles.out.gff.map{ it -> it[1]}.collect()
|
||||||
results = results.concat(assembly.bam.map {sample_id, bam, bai -> [bam, bai]}.flatten())
|
results = results.concat(assembly.bam.map {sample_id, bam, bai -> [bam, bai]}.flatten())
|
||||||
}
|
}
|
||||||
@ -754,7 +775,7 @@ workflow {
|
|||||||
}
|
}
|
||||||
if (params.de_analysis){
|
if (params.de_analysis){
|
||||||
if (!params.ref_annotation){
|
if (!params.ref_annotation){
|
||||||
error = "You must provide a reference annotation."
|
error = "When running in --de_analysis mode you must provide a reference annotation."
|
||||||
}
|
}
|
||||||
if (!params.sample_sheet){
|
if (!params.sample_sheet){
|
||||||
error = "You must provide a sample_sheet with at least alias and condition columns."
|
error = "You must provide a sample_sheet with at least alias and condition columns."
|
||||||
@ -762,6 +783,10 @@ workflow {
|
|||||||
if (params.containsKey("condition_sheet")) {
|
if (params.containsKey("condition_sheet")) {
|
||||||
error = "Condition sheets have been deprecated. Please add a 'condition' column to your sample sheet instead. Check the quickstart for more information."
|
error = "Condition sheets have been deprecated. Please add a 'condition' column to your sample sheet instead. Check the quickstart for more information."
|
||||||
}
|
}
|
||||||
|
} else{
|
||||||
|
if (!params.ref_annotation){
|
||||||
|
log.info("Warning: As no --ref_annotation was provided, the output transcripts will not be annotated.")
|
||||||
|
}
|
||||||
}
|
}
|
||||||
if (error){
|
if (error){
|
||||||
throw new Exception(error)
|
throw new Exception(error)
|
||||||
|
|||||||
Loading…
Reference in New Issue
Block a user