diff --git a/.gitlab-ci.yml b/.gitlab-ci.yml index 18eb2ce..eef3466 100644 --- a/.gitlab-ci.yml +++ b/.gitlab-ci.yml @@ -52,7 +52,7 @@ docker-run: "fusions", "differential_expression", "isoforms", "only_differential_expression", "differential_expression_gff3", "ncbi_gzip", "ncbi_no_gene_id", "ensembl_with_versions", - "differential_expression_mouse" + "differential_expression_mouse", "no_ref_annotation" ] rules: # NOTE As we're overriding the rules block for the included docker-run @@ -66,6 +66,12 @@ docker-run: NF_WORKFLOW_OPTS: "--fastq ERR6053095_chr20.fastq --transcriptome-source reference-guided \ --ref_genome chr20/hg38_chr20.fa --ref_annotation chr20/gencode.v22.annotation.chr20.gtf --pychopper_backend phmm" NF_IGNORE_PROCESSES: preprocess_reads,merge_transcriptomes,decompress_annotation,decompress_ref,decompress_transcriptome,preprocess_ref_transcriptome + - if: $MATRIX_NAME == "no_ref_annotation" + variables: + NF_BEFORE_SCRIPT: wget -O test_data.tar.gz https://ont-exd-int-s3-euwst1-epi2me-labs.s3.amazonaws.com/wf-isoforms/wf-isoforms_test_data.tar.gz && tar -xzvf test_data.tar.gz + NF_WORKFLOW_OPTS: "--fastq ERR6053095_chr20.fastq --transcriptome-source reference-guided \ + --ref_genome chr20/hg38_chr20.fa" + NF_IGNORE_PROCESSES: run_gffcompare,preprocess_reads,merge_transcriptomes,decompress_annotation,decompress_ref,decompress_transcriptome,preprocess_ref_transcriptome - if: $MATRIX_NAME == "fusions" variables: NF_BEFORE_SCRIPT: wget -O test_data.tar.gz https://ont-exd-int-s3-euwst1-epi2me-labs.s3.amazonaws.com/wf-isoforms/wf-isoforms_test_data.tar.gz && tar -xzvf test_data.tar.gz diff --git a/CHANGELOG.md b/CHANGELOG.md index 56ca0fa..7a38111 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -16,6 +16,7 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 - Add gene name column to the de_analysis counts TSV files. ### Fixed - Mapping stage using a single thread only. +- When no `--ref_annotation` is provided the workflow will still run but the output transcripts will not be annotated. However `--de_analysis` mode still requires a `--ref_annotation`. ## [v1.0.0] ### Added diff --git a/main.nf b/main.nf index d566b5b..8b08afe 100644 --- a/main.nf +++ b/main.nf @@ -323,8 +323,6 @@ process run_gffcompare{ """ } - - process get_transcriptome{ /* Write out a transcriptome file based on the query gff annotations. @@ -340,6 +338,9 @@ process get_transcriptome{ script: def transcriptome = "${sample_id}_transcriptome.fas" def merged_transcriptome = "${sample_id}_merged_transcriptome.fas" + // if no ref_annotation gffcmp_dir will be optional file + // so skip getting transcriptome FASTA from the annotated files. + if (params.ref_annotation){ """ gffread -g ${reference_seq} -w ${transcriptome} ${transcripts_gff} if [ "\$(ls -A $gffcmp_dir)" ]; @@ -347,6 +348,11 @@ process get_transcriptome{ gffread -F -g ${reference_seq} -w ${merged_transcriptome} $gffcmp_dir/str_merged.annotated.gtf fi """ + } else { + """ + gffread -g ${reference_seq} -w ${transcriptome} ${transcripts_gff} + """ + } } process merge_transcriptomes { @@ -397,11 +403,11 @@ process makeReport { // for DE analysis, a `gene_name` column will be added to // `de_report/results_dge.tsv` path "results_dge.tsv", emit: de_analysis, optional: true - script: + shell: // Convert the sample_id arrayList. sids = new BlankSeparatedList(sample_ids) - def report_name = "wf-transcriptomes-report.html" - """ + report_name = "wf-transcriptomes-report.html" + ''' if [ -f "de_report/OPTIONAL_FILE" ]; then dereport="" else @@ -409,10 +415,14 @@ process makeReport { mv de_report/*.g*f* de_report/stringtie_merged.gtf fi if [ -f "gff_annotation/OPTIONAL_FILE" ]; then - OPT_GFF="" + OPT_GFF_ANNOTATION="" else - OPT_GFF="--gffcompare_dir ${gffcmp_dir} --gff_annotation gff_annotation/*" - + OPT_GFF_ANNOTATION="--gff_annotation gff_annotation/*" + fi + if [ -f "OPTIONAL_FILE" ]; then + OPT_GFFCMP_DIR="" + else + OPT_GFFCMP_DIR="--gffcompare_dir !{gffcmp_dir}" fi if [ -f "jaffal_csv/OPTIONAL_FILE" ]; then OPT_JAFFAL_CSV="" @@ -429,18 +439,19 @@ process makeReport { else OPT_PC_REPORT="--pychop_report pychopper_report/*" fi - workflow-glue report --report $report_name \ - --versions $versions \ + workflow-glue report --report !{report_name} \ + --versions !{versions} \ --params params.json \ - \$OPT_ALN \ - \$OPT_PC_REPORT \ - --sample_ids $sids \ + ${OPT_ALN} \ + ${OPT_PC_REPORT} \ + --sample_ids !{sids} \ --stats per_read_stats/* \ - \$OPT_GFF \ - --isoform_table_nrows $params.isoform_table_nrows \ - \$OPT_JAFFAL_CSV \ - \$dereport - """ + ${OPT_GFF_ANNOTATION} \ + ${OPT_GFFCMP_DIR} \ + --isoform_table_nrows !{params.isoform_table_nrows} \ + ${OPT_JAFFAL_CSV} \ + ${dereport} + ''' } @@ -582,18 +593,28 @@ workflow pipeline { assemble_transcripts(split_bam.out.bundles.flatMap(map_sample_ids_cls).combine(ref_annotation),use_ref_ann) merge_gff_bundles(assemble_transcripts.out.gff_bundles.groupTuple()) - run_gffcompare(merge_gff_bundles.out.gff, ref_annotation) + // only run gffcompare if ref annotation provided. Otherwise create optional files and channels + if (params.ref_annotation){ + run_gffcompare(merge_gff_bundles.out.gff, ref_annotation) + gff_compare_dir = run_gffcompare.out.gffcmp_dir + gff_compare = run_gffcompare.out.gffcmp_dir.map{ it -> it[1]}.collect() + // create per sample gff tuples with gff compare directories + gff_tuple = merge_gff_bundles.out.gff + .join(gff_compare_dir) + } else { + // create per sample gff tuples with optional files as no ref_annotation + optional_channel = Channel.fromPath("$projectDir/data/OPTIONAL_FILE") + gff_tuple = merge_gff_bundles.out.gff.combine(optional_channel) + gff_compare = OPTIONAL_FILE + } // For reference based assembly, there is only one reference // So map this reference to all sample_ids seq_for_transcriptome_build = sample_ids.flatten().combine(ref_genome) - - get_transcriptome( - merge_gff_bundles.out.gff - .join(run_gffcompare.out.gffcmp_dir) + gff_tuple .join(seq_for_transcriptome_build)) - gff_compare = run_gffcompare.out.gffcmp_dir.map{ it -> it[1]}.collect() + merge_gff = merge_gff_bundles.out.gff.map{ it -> it[1]}.collect() results = results.concat(assembly.bam.map {sample_id, bam, bai -> [bam, bai]}.flatten()) } @@ -754,7 +775,7 @@ workflow { } if (params.de_analysis){ if (!params.ref_annotation){ - error = "You must provide a reference annotation." + error = "When running in --de_analysis mode you must provide a reference annotation." } if (!params.sample_sheet){ error = "You must provide a sample_sheet with at least alias and condition columns." @@ -762,6 +783,10 @@ workflow { if (params.containsKey("condition_sheet")) { error = "Condition sheets have been deprecated. Please add a 'condition' column to your sample sheet instead. Check the quickstart for more information." } + } else{ + if (!params.ref_annotation){ + log.info("Warning: As no --ref_annotation was provided, the output transcripts will not be annotated.") + } } if (error){ throw new Exception(error)