From 956c80573ab940c39471f470dbffcb3ad74e7e51 Mon Sep 17 00:00:00 2001 From: Neil Horner Date: Mon, 1 Aug 2022 12:01:01 +0000 Subject: [PATCH] Fusions --- .gitlab-ci.yml | 7 +- .gitmodules | 0 .pre-commit-config.yaml | 5 +- CHANGELOG.md | 8 +- Dockerfile | 5 + README.md | 109 ++++++++++++++--- bin/ping.py | 2 +- bin/report.py | 61 ++++++++-- docs/header.md | 2 +- docs/intro.md | 14 ++- docs/quickstart.md | 92 ++++++++++++--- download_jaffal_references.sh | 5 + environment.yaml | 10 +- evaluation/run_evaluation_dmel.sh | 59 ---------- main.nf | 80 ++++++++----- nextflow.config | 23 ++-- nextflow_schema.json | 43 ++++++- subworkflows/JAFFAL/build_jaffal_ref.sh | 2 + subworkflows/JAFFAL/gene_fusions.nf | 40 +++++++ subworkflows/JAFFAL/install_jaffa.sh | 21 ++++ subworkflows/JAFFAL/install_linux64.sh | 111 ++++++++++++++++++ subworkflows/JAFFAL/tools.groovy | 10 ++ .../denovo_assembly.nf | 0 .../reference_assembly.nf | 0 24 files changed, 555 insertions(+), 154 deletions(-) create mode 100644 .gitmodules create mode 100644 download_jaffal_references.sh delete mode 100644 evaluation/run_evaluation_dmel.sh create mode 100644 subworkflows/JAFFAL/build_jaffal_ref.sh create mode 100644 subworkflows/JAFFAL/gene_fusions.nf create mode 100755 subworkflows/JAFFAL/install_jaffa.sh create mode 100755 subworkflows/JAFFAL/install_linux64.sh create mode 100644 subworkflows/JAFFAL/tools.groovy rename denovo_assembly.nf => subworkflows/denovo_assembly.nf (100%) rename reference_assembly.nf => subworkflows/reference_assembly.nf (100%) diff --git a/.gitlab-ci.yml b/.gitlab-ci.yml index 70c55df..21547a2 100644 --- a/.gitlab-ci.yml +++ b/.gitlab-ci.yml @@ -8,5 +8,8 @@ variables: # The workflow should define `--out_dir`, the CI template sets this. # Only common file inputs and option values need to be given here # (not things such as -profile) - NF_WORKFLOW_OPTS: "--fastq test_data/fastq \ - --ref_genome test_data/SIRV_150601a.fasta --ref_annotation test_data/SIRV_isoforms.gtf" + NF_BEFORE_SCRIPT: | + wget -O test_data.tar.gz https://ont-exd-int-s3-euwst1-epi2me-labs.s3.amazonaws.com/wf-isoforms/wf-isoforms_test_data.tar.gz && tar -xzvf test_data.tar.gz + NF_WORKFLOW_OPTS: "--fastq ERR6053095_chr20.fastq \ + --ref_genome chr20/hg38_chr20.fa --ref_annotation chr20/gencode.v22.annotation.chr20.gtf \ + --jaffal_refBase chr20/ --jaffal_genome hg38_chr20 --jaffal_annotation genCode22" diff --git a/.gitmodules b/.gitmodules new file mode 100644 index 0000000..e69de29 diff --git a/.pre-commit-config.yaml b/.pre-commit-config.yaml index ecd70f9..14b9aa2 100644 --- a/.pre-commit-config.yaml +++ b/.pre-commit-config.yaml @@ -23,4 +23,7 @@ repos: - id: flake8 additional_dependencies: - flake8-import-order==0.18.1 - entry: flake8 bin --import-order-style google --statistics \ No newline at end of file + - flake8-docstrings==1.6.0 + - flake8-rst-docstrings==0.2.5 + - flake8-forbid-visual-indent==0.0.2 + entry: flake8 bin --import-order-style google --statistics diff --git a/CHANGELOG.md b/CHANGELOG.md index 42db068..3c36ff2 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -6,13 +6,17 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 ## [unreleased] ### Changed -- Skip unnecessary conversion to fasta from fastq -- Fastqingress metadata map + ## [v0.1.4] +### Added +- JAFFAL fusion detectoion subworkflow ### Changed - Args parser for fastqingress - Set out_dir option type to ensure output is written to correct directory on Windows +- Skip unnecessary conversion to fasta from fastq +- Fastqingress metadata map +- Changed workflow name to wf-transcriptomes ## [v0.1.3] ### Changed diff --git a/Dockerfile b/Dockerfile index 49c9148..a6fca02 100644 --- a/Dockerfile +++ b/Dockerfile @@ -15,6 +15,11 @@ RUN \ && rm -rf $CONDA_DIR/lib/python3.*/site-packages/pip \ && find $CONDA_DIR -name '__pycache__' -type d -exec rm -rf '{}' '+' + USER $WF_UID WORKDIR $HOME +# Install JAFFA +ADD subworkflows $HOME/subworkflows +RUN /bin/sh -c $HOME/subworkflows/JAFFAL/install_jaffa.sh + diff --git a/README.md b/README.md index 7274f20..07edbcd 100644 --- a/README.md +++ b/README.md @@ -1,4 +1,4 @@ -# wf-isoforms +# wf-transcriptomes This repository contains a [nextflow](https://www.nextflow.io/) workflow for assembly and annotation of transcripts from Oxford Nanopore cDNA or direct RNA reads. @@ -16,14 +16,17 @@ cDNA reads are initially preprocessed by [pychopper](https://github.com/epi2me-l for the identification of full-length reads, as well as trimming and orientation correction (This step is omitted for direct RNA reads). -### Reference-aided approach + +### Transcript assembly + +#### Reference-aided transcript assembly approach * Full length reads are mapped to a supplied reference genome using [minimap2](https://github.com/lh3/minimap2) * Transcripts are assembled by [stringtie](http://ccb.jhu.edu/software/stringtie) in long read mode (with or without a guide reference annotation) to generate the GFF annotation. * The annotation generated by the pipeline is compared to the reference annotation. using [gffcompare](http://ccb.jhu.edu/software/stringtie/gffcompare.shtml) -### de novo-based approach (experimental!) +#### de novo-based transcript assembly (experimental!) * Sequence clusters are generated using [isONclust2](https://github.com/nanoporetech/isONclust2) * If a reference genome is supplied, cluster quality metrics are determined by comparing with clusters generated from a minimap2 alignment. @@ -33,11 +36,17 @@ using [gffcompare](http://ccb.jhu.edu/software/stringtie/gffcompare.shtml) * Transcripts are assembled by stringtie as for the reference-based approach. * __Note__: This approach is currently not supported with direct RNA reads. +### Fusion gene detection +Fusion gene detection is performed using [JAFFA](https://github.com/Oshlack/JAFFA), with the JAFFAL extension for use +with ONT long reads. + ### Workflow inputs - Directory containing cDNA/direct RNA reads. Or a directory containing subdirectories each with reads from different samples (in fastq/fastq.gz format) - Reference genome in fasta format (required for reference-based assembly). -- Optional reference annotation in GFF2/3 format.## Quickstart +- Optional reference annotation in GFF2/3 format. +- For fusion detection, JAFFAL reference files (see Quickstart) +## Quickstart The workflow uses [nextflow](https://www.nextflow.io/) to manage compute and software resources, as such nextflow will need to be installed before attempting @@ -54,31 +63,31 @@ It is not required to clone or download the git repository in order to run the w For more information on running EPI2ME Labs workflows [visit out website](https://labs.epi2me.io/wfindex). -**Workflow options** +### Workflow options To obtain the workflow, having installed `nextflow`, users can run: ``` -nextflow run epi2me-labs/wf-isoforms --help +nextflow run epi2me-labs/wf-transcriptomes --help ``` to see the options for the workflow. +**Download demonstration data** +A small test dataset is provided for the purposes of testing the workflow software. It consists of reads, reference, +and annotations from human chromosome 20 only. +It can be downloaded using: +```shell +wget -O test_data.tar.gz https://ont-exd-int-s3-euwst1-epi2me-labs.s3.amazonaws.com/wf-isoforms/wf-isoforms_test_data.tar.gz +tar -xzvf test_data.tar.gz +``` -**Example execution of a workflow for reference-based transcript assembly** - -This example uses a synthetic SIRV dataset, so we need to tell minimap2 about the non-canonical splice junctions with ---minimap2_opts '-uf --splice-flank=no' +**Example execution of a workflow for reference-based transcript assembly and fusion detection** ``` OUTPUT=~/output; -nextflow run wf-isoforms/ --fastq test_data/fastq --ref_genome test_data/SIRV_150601a.fasta --ref_annotation test_data/SIRV_isofroms.gtf ---minimap2_opts '-uf --splice-flank=no' --out_dir outdir -w workspace_dir -profile conda -resume -``` - -``` -# To evaluate the workflow on a larger Drosophila dataset -./evaluation/run_evaluation_dmel.sh outdir +nexflow run epi2me-labs/wf-transcriptomes --fastq ERR6053095_chr20.fastq --ref_genome chr20/hg38_chr20.fa --ref_annotation chr20/gencode.v22.annotation.chr20.gtf \ + --jaffal_refBase chr20/ --jaffal_genome hg38_chr20 --jaffal_annotation genCode22" --out_dir outdir -w workspace_dir -profile conda -resume ``` **Example workflow for denovo transcript assembly** @@ -91,7 +100,7 @@ A full list of options can be seen in nextflow_schema.json. Below are some commo - Threshold for including isoforms into interactive table `transcript_table_cov_thresh = 50` - Run the denovo pipeline `denovo = true` (default false) -- To run the workflow with direct RNA reads `--direct_rna` (skips the pychopper step). +- To run the workflow with direct RNA reads `--direct_rna` (this just skips the pychopper step). Pychopper and minimap2 can take options via `minimap2_opts` and `pychopper_opts`, for example: @@ -102,12 +111,74 @@ Pychopper and minimap2 can take options via `minimap2_opts` and `pychopper_opts` - pychopper needs to know which cDNA synthesis kit used - SQK-PCS109: use `pychopper_opts = '-k PCS109'` (default) - SQK-PCS110: use `pychopper_opts = '-k PCS110'` + - SQK-PCS11: use `pychopper_opts = '-k PCS111'` - pychopper can use one of two available backends for identifying primers in the raw reads - nhmmscan `pychopper opts = '-m phmm'` - edlib `pychopper opts = '-m edlib'` -__Note__: edlib is set by default in the config as it's quite a lot faster. However it may be less sensitive than nhmmscan. +__Note__: edlib is set by default in the config as it's quite a lot faster. However, it may be less sensitive than nhmmscan. + +### Fusion detection + +JAFFAL from the [JAFFA](https://github.com/Oshlack/JAFFA) +package is used to identify potential fusion transcripts. To get this this working, there are a couple of things that need doing first. + +**Install JAFFA** + +to install JAFFA and it's dependencies run the folllowing: +```shell +cd wf-transcriptomes/ +./subworkflows/JAFFAL/install_jaffa.sh +``` + +**Prepare JAFFAL reference data** + +To use pre-processed reference files for the hg38 genome and GENCODE v22 annotation (as used in the JFFAAL paper), +do: +```shell +mkdir jaffal_data_dir +cd jaffal_data_dir/ +wf-transcriptomes/download_jaffal_references.sh +```` + +To use alternative genome and annotation files, they should be prepared as described +[here](https://github.com/Oshlack/JAFFA/wiki/FAQandTroubleshooting#how-can-i-generate-the-reference-files-for-a-non-supported-genome) + +**Specifying the location of the JAFFA code and reference directories** + +`--jaffal_dir` +This is the directory made by running install_jaffa.sh as shown above + +`--jaffal_refBase` +The directory containing the reference data prepared for use with JAFFAL + + +**JAFFAL annotation and genome files** + +The prepared JAFFAL reference files will look something like `hg38_chr20_genCode22.fa`. To enable JAFFAL to find these +files `--jaffal_genome` should be set to `hg38_chr20` and `--jaffal_annotation` to `genCode22` + + +__JAFFAL Notes__: +g++ must be installed. JAFFAL is not currently working on Mac M1 (osx-arm64 architecture). If there are no fusion transcripts +detected, the workflow will terminate with an error at the JAFFAL stage. If this happens, +skip the JAFFAL stage by omitting ` --jaffal_refBase` + + +## Workflow outputs +* an HTML report document detailing the primary findings of the workflow. +* for each sample: + * [gffcomapre](https://ccb.jhu.edu/software/stringtie/gffcompare.shtml) output directories + * read_aln_stats.tsv - alignment summary statistics + * transcriptome.fas - the assembled transcriptome + * merged_transcritptome.fas - annotated, assembled transcriptome + * [jaffal](https://github.com/Oshlack/JAFFA) ooutput directories + +### Fusion detection outputs +in `${out_dir}/jaffal_output_${sample_id}` you will find: +* jaffa_results.csv - the csv results summary file +* jaffa_results.fasta - fusion transcritpt sequences ## Useful links * [nextflow](https://www.nextflow.io/) diff --git a/bin/ping.py b/bin/ping.py index 689ff54..de8c9a0 100755 --- a/bin/ping.py +++ b/bin/ping.py @@ -53,7 +53,7 @@ def main(): hostname=args.hostname, opsys=args.opsys ).send_workflow_ping( - workflow='wf-isoforms', + workflow='wf-transcriptomes', message=args.message, revision=args.revision, commit=args.commit, diff --git a/bin/report.py b/bin/report.py index f027b71..07a83c9 100755 --- a/bin/report.py +++ b/bin/report.py @@ -400,10 +400,10 @@ def gff_compare_plots(report, gffcompare_outdirs: Path, sample_ids): tracking = df_track.groupby("Overlaps").count().reset_index() tracking.Overlaps = tracking.Overlaps.map(names) tracking['Percent'] = tracking.Count * 100 / tracking.Count.sum() - tracking = tracking.sort_values("Overlaps") + tracking = tracking.sort_values("Count", ascending=False) track_bar = bars.simple_hbar( - tracking['Overlaps'].values.tolist(), - tracking['Percent'].values.tolist(), + list(reversed(tracking['Overlaps'].values.tolist())), + list(reversed(tracking['Percent'].values.tolist())), colors=Colors.cerulean, title=id_) tracking_dfs.append(tracking) @@ -426,6 +426,7 @@ def gff_compare_plots(report, gffcompare_outdirs: Path, sample_ids): track_table = DataTable( columns=cols, source=ColumnDataSource(tracking), index_position=None, width=500) + tabs.append(Panel( child=gridplot([track_bar, track_table], ncols=2), title=id_) ) @@ -602,7 +603,7 @@ def transcript_table(report, df_tmaps, covr_threshold): # drop some columns for the big table and do some filtering section.markdown(''' - ### Query transcript table + ### Isoforms table Low coverage transcripts are removed to speed up the table viewing.
Coverage threshold can be set with the parameter @@ -650,6 +651,8 @@ def transcript_table(report, df_tmaps, covr_threshold): df['parent gene iso num'] = df.apply( lambda x: gb.loc[(x.ref_gene_id, x.sample_id), 'num_isoforms'], axis=1) + # Uncalssified transcritps should not be lumped togetehr + df.loc[df.class_code == 'u', 'parent gene iso num'] = None df.sort_values('parent gene iso num', inplace=True, ascending=True) @@ -795,6 +798,43 @@ def seq_stats_tabs(report, sample_ids, stats): section.plot(Tabs(tabs=tabs)) +def jaffal_table(report, sample_ids, result_csvs): + """Make a table of fusion transcripts identified by JAFFAL.""" + cols = [ + 'sample_id', 'fusion genes', 'chrom1', 'chrom2', 'spanning reads', + 'classification', 'known'] + dfs = [] + for csv, sid in zip(result_csvs, sample_ids): + df = pd.read_csv(csv) + df['sample_id'] = sid + sid_col = df.pop('sample_id') + df.insert(0, 'sample_id', sid_col) + dfs.append(df) + + df = pd.concat(dfs) + df = df[cols] + df['chroms'] = df.chrom1.astype(str) + ':' + df.chrom2.astype(str) + df.rename(columns={ + 'spanning reads': 'nreads', + 'fusion genes': 'genes'}, inplace=True) + df.drop(columns=['chrom1', 'chrom2'], inplace=True) + section = report.add_section() + section.markdown(""" + ### JAFFAL fusion transcript summary + + This table summarizes putative fusion transcripts identified + by [JAFFAL](https://github.com/Oshlack/JAFFA/). + + * genes: the gene symbols of the fusion partners + * nreads: The number of reads supporting the fusion + * classification: JAFFAL's classification + * known: whether this fusion is in the given set of known gene fusions + * chroms: the respective, original chromosome location of the two partner + genes + """) + section.table(df) + + def main(): """Run the entry point.""" parser = argparse.ArgumentParser() @@ -833,15 +873,17 @@ def main(): parser.add_argument( "--cluster_qc_dirs", required=False, type=str, default=None, nargs='*', help="Directory with various cluster quality csvs") + parser.add_argument( + "--jaffal_csv", required=False, type=str, default=None, nargs='*', + help="Path to JAFFAL results csv") parser.add_argument('--denovo', dest='denovo', action='store_true') args = parser.parse_args() - print('denovo', args.denovo) sample_ids = args.sample_ids report = WFReport( - "Transcript isoform report", "wf-isoforms", + "Transcript isoform report", "wf-transcriptomes", revision=args.revision, commit=args.commit) # Add reads summary section @@ -853,7 +895,9 @@ def main(): section.markdown(''' ### Read mapping summary - Output of [seqkit](https://bioinf.shenwei.me/seqkit/) bam -s''') + Summary of minimap2 mapping from + [seqkit](https://bioinf.shenwei.me/seqkit/) + `seqkit bam -s`''') section.table(df_aln_stats) @@ -877,6 +921,9 @@ def main(): if args.cluster_qc_dirs is not None: cluster_quality(args.cluster_qc_dirs, report, sample_ids) + if args.jaffal_csv is not None: + jaffal_table(report, sample_ids, args.jaffal_csv) + # Arguments and software versions report.add_section( section=scomponents.version_table(args.versions)) diff --git a/docs/header.md b/docs/header.md index 2affd15..31f01a7 100644 --- a/docs/header.md +++ b/docs/header.md @@ -1,4 +1,4 @@ -# wf-isoforms +# wf-transcriptomes This repository contains a [nextflow](https://www.nextflow.io/) workflow for assembly and annotation of transcripts from Oxford Nanopore cDNA or direct RNA reads. diff --git a/docs/intro.md b/docs/intro.md index c532336..19fac27 100644 --- a/docs/intro.md +++ b/docs/intro.md @@ -8,14 +8,17 @@ cDNA reads are initially preprocessed by [pychopper](https://github.com/epi2me-l for the identification of full-length reads, as well as trimming and orientation correction (This step is omitted for direct RNA reads). -### Reference-aided approach + +### Transcript assembly + +#### Reference-aided transcript assembly approach * Full length reads are mapped to a supplied reference genome using [minimap2](https://github.com/lh3/minimap2) * Transcripts are assembled by [stringtie](http://ccb.jhu.edu/software/stringtie) in long read mode (with or without a guide reference annotation) to generate the GFF annotation. * The annotation generated by the pipeline is compared to the reference annotation. using [gffcompare](http://ccb.jhu.edu/software/stringtie/gffcompare.shtml) -### de novo-based approach (experimental!) +#### de novo-based transcript assembly (experimental!) * Sequence clusters are generated using [isONclust2](https://github.com/nanoporetech/isONclust2) * If a reference genome is supplied, cluster quality metrics are determined by comparing with clusters generated from a minimap2 alignment. @@ -25,8 +28,13 @@ using [gffcompare](http://ccb.jhu.edu/software/stringtie/gffcompare.shtml) * Transcripts are assembled by stringtie as for the reference-based approach. * __Note__: This approach is currently not supported with direct RNA reads. +### Fusion gene detection +Fusion gene detection is performed using [JAFFA](https://github.com/Oshlack/JAFFA), with the JAFFAL extension for use +with ONT long reads. + ### Workflow inputs - Directory containing cDNA/direct RNA reads. Or a directory containing subdirectories each with reads from different samples (in fastq/fastq.gz format) - Reference genome in fasta format (required for reference-based assembly). -- Optional reference annotation in GFF2/3 format. \ No newline at end of file +- Optional reference annotation in GFF2/3 format. +- For fusion detection, JAFFAL reference files (see Quickstart) diff --git a/docs/quickstart.md b/docs/quickstart.md index ff1bff0..0076331 100644 --- a/docs/quickstart.md +++ b/docs/quickstart.md @@ -15,31 +15,31 @@ It is not required to clone or download the git repository in order to run the w For more information on running EPI2ME Labs workflows [visit out website](https://labs.epi2me.io/wfindex). -**Workflow options** +### Workflow options To obtain the workflow, having installed `nextflow`, users can run: ``` -nextflow run epi2me-labs/wf-isoforms --help +nextflow run epi2me-labs/wf-transcriptomes --help ``` to see the options for the workflow. +**Download demonstration data** +A small test dataset is provided for the purposes of testing the workflow software. It consists of reads, reference, +and annotations from human chromosome 20 only. +It can be downloaded using: +```shell +wget -O test_data.tar.gz https://ont-exd-int-s3-euwst1-epi2me-labs.s3.amazonaws.com/wf-isoforms/wf-isoforms_test_data.tar.gz +tar -xzvf test_data.tar.gz +``` -**Example execution of a workflow for reference-based transcript assembly** - -This example uses a synthetic SIRV dataset, so we need to tell minimap2 about the non-canonical splice junctions with ---minimap2_opts '-uf --splice-flank=no' +**Example execution of a workflow for reference-based transcript assembly and fusion detection** ``` OUTPUT=~/output; -nextflow run wf-isoforms/ --fastq test_data/fastq --ref_genome test_data/SIRV_150601a.fasta --ref_annotation test_data/SIRV_isofroms.gtf ---minimap2_opts '-uf --splice-flank=no' --out_dir outdir -w workspace_dir -profile conda -resume -``` - -``` -# To evaluate the workflow on a larger Drosophila dataset -./evaluation/run_evaluation_dmel.sh outdir +nexflow run epi2me-labs/wf-transcriptomes --fastq ERR6053095_chr20.fastq --ref_genome chr20/hg38_chr20.fa --ref_annotation chr20/gencode.v22.annotation.chr20.gtf \ + --jaffal_refBase chr20/ --jaffal_genome hg38_chr20 --jaffal_annotation genCode22" --out_dir outdir -w workspace_dir -profile conda -resume ``` **Example workflow for denovo transcript assembly** @@ -52,7 +52,7 @@ A full list of options can be seen in nextflow_schema.json. Below are some commo - Threshold for including isoforms into interactive table `transcript_table_cov_thresh = 50` - Run the denovo pipeline `denovo = true` (default false) -- To run the workflow with direct RNA reads `--direct_rna` (skips the pychopper step). +- To run the workflow with direct RNA reads `--direct_rna` (this just skips the pychopper step). Pychopper and minimap2 can take options via `minimap2_opts` and `pychopper_opts`, for example: @@ -63,9 +63,71 @@ Pychopper and minimap2 can take options via `minimap2_opts` and `pychopper_opts` - pychopper needs to know which cDNA synthesis kit used - SQK-PCS109: use `pychopper_opts = '-k PCS109'` (default) - SQK-PCS110: use `pychopper_opts = '-k PCS110'` + - SQK-PCS11: use `pychopper_opts = '-k PCS111'` - pychopper can use one of two available backends for identifying primers in the raw reads - nhmmscan `pychopper opts = '-m phmm'` - edlib `pychopper opts = '-m edlib'` -__Note__: edlib is set by default in the config as it's quite a lot faster. However it may be less sensitive than nhmmscan. +__Note__: edlib is set by default in the config as it's quite a lot faster. However, it may be less sensitive than nhmmscan. + +### Fusion detection + +JAFFAL from the [JAFFA](https://github.com/Oshlack/JAFFA) +package is used to identify potential fusion transcripts. To get this this working, there are a couple of things that need doing first. + +**Install JAFFA** + +to install JAFFA and it's dependencies run the folllowing: +```shell +cd wf-transcriptomes/ +./subworkflows/JAFFAL/install_jaffa.sh +``` + +**Prepare JAFFAL reference data** + +To use pre-processed reference files for the hg38 genome and GENCODE v22 annotation (as used in the JFFAAL paper), +do: +```shell +mkdir jaffal_data_dir +cd jaffal_data_dir/ +wf-transcriptomes/download_jaffal_references.sh +```` + +To use alternative genome and annotation files, they should be prepared as described +[here](https://github.com/Oshlack/JAFFA/wiki/FAQandTroubleshooting#how-can-i-generate-the-reference-files-for-a-non-supported-genome) + +**Specifying the location of the JAFFA code and reference directories** + +`--jaffal_dir` +This is the directory made by running install_jaffa.sh as shown above + +`--jaffal_refBase` +The directory containing the reference data prepared for use with JAFFAL + + +**JAFFAL annotation and genome files** + +The prepared JAFFAL reference files will look something like `hg38_chr20_genCode22.fa`. To enable JAFFAL to find these +files `--jaffal_genome` should be set to `hg38_chr20` and `--jaffal_annotation` to `genCode22` + + +__JAFFAL Notes__: +g++ must be installed. JAFFAL is not currently working on Mac M1 (osx-arm64 architecture). If there are no fusion transcripts +detected, the workflow will terminate with an error at the JAFFAL stage. If this happens, +skip the JAFFAL stage by omitting ` --jaffal_refBase` + + +## Workflow outputs +* an HTML report document detailing the primary findings of the workflow. +* for each sample: + * [gffcomapre](https://ccb.jhu.edu/software/stringtie/gffcompare.shtml) output directories + * read_aln_stats.tsv - alignment summary statistics + * transcriptome.fas - the assembled transcriptome + * merged_transcritptome.fas - annotated, assembled transcriptome + * [jaffal](https://github.com/Oshlack/JAFFA) ooutput directories + +### Fusion detection outputs +in `${out_dir}/jaffal_output_${sample_id}` you will find: +* jaffa_results.csv - the csv results summary file +* jaffa_results.fasta - fusion transcritpt sequences diff --git a/download_jaffal_references.sh b/download_jaffal_references.sh new file mode 100644 index 0000000..ec12648 --- /dev/null +++ b/download_jaffal_references.sh @@ -0,0 +1,5 @@ +#!/bin/sh + +#Download the data. We should we move the data out of Figshare? +wget -O JAFFA_REFERENCE_FILES_HG38_GENCODE22.V2.tar.gz https://figshare.com/ndownloader/files/25410494 +tar -zxvf JAFFA_REFERENCE_FILES_HG38_GENCODE22.V2.tar.gz \ No newline at end of file diff --git a/environment.yaml b/environment.yaml index 74769a4..815c298 100644 --- a/environment.yaml +++ b/environment.yaml @@ -1,8 +1,8 @@ -name: epi2melabs-wf-isoforms +name: epi2melabs-wf-transcriptomes channels: - epi2melabs - - bioconda - conda-forge + - bioconda - defaults dependencies: - python==3.8.* @@ -27,4 +27,8 @@ dependencies: - parallel - scikit-learn==1.0.2 - natsort - - graphviz \ No newline at end of file +# Fusion detection dependencies +# - bpipe=0.9.9.2 + - java-jdk + - r-base + - gxx \ No newline at end of file diff --git a/evaluation/run_evaluation_dmel.sh b/evaluation/run_evaluation_dmel.sh deleted file mode 100644 index 5da39ea..0000000 --- a/evaluation/run_evaluation_dmel.sh +++ /dev/null @@ -1,59 +0,0 @@ -#!/usr/bin/env bash - -# Usage: ./run_evaluation_dmel.sh pathto/outputdir - -# See the isONcorrect paper https://www.nature.com/articles/s41467-020-20340-8 where this dataset is described - - -if [[ "$#" -lt 1 ]]; then - echo "usage: run_evaluation_dmel.sh [nextflow.config]" - exit 1 -fi - -if [[ "$#" -eq 1 ]]; then - config='' -fi - -if [[ "$#" -eq 2 ]]; then - config="-c $2"; -fi - -OUTDIR=$1; - -FASTQ_URL="http://ftp.sra.ebi.ac.uk/vol1/fastq/ERR358/005/ERR3588905/ERR3588905_1.fastq.gz" -REF_URL="http://ftp.ensembl.org/pub/release-99/fasta/drosophila_melanogaster/dna/Drosophila_melanogaster.BDGP6.28.dna.toplevel.fa.gz" -GFF_URL="http://ftp.ensembl.org/pub/release-99/gff3/drosophila_melanogaster/Drosophila_melanogaster.BDGP6.28.99.gff3.gz" - -DATA_DIR="$OUTDIR/data" -READS_DIR="$DATA_DIR/reads" -FASTQ="$READS_DIR/ERR3588905_1.fastq.gz" -REF="$DATA_DIR/Drosophila_melanogaster.BDGP6.28.dna.toplevel.fa" -GFF="$DATA_DIR/Drosophila_melanogaster.BDGP6.28.99.gff3" - -mkdir -p $READS_DIR - -if [ ! -f $REF ]; -then (echo "downloading reference genome"; cd $DATA_DIR; curl -L -C - -O $REF_URL); gzip -d ${REF}.gz -fi - -if [ ! -f $GFF ]; -then - (echo "downloading reference annotation"; cd $DATA_DIR; curl -L -C - -O $GFF_URL); gzip -d ${GFF}.gz -fi - -if [ ! -f $FASTQ ]; -then (echo "downloading reads"; cd $READS_DIR; curl -L -C - -O $FASTQ_URL); gzip -d ${FASTQ}.gz -fi - - -OUT_REF="$OUTDIR/ref" -OUT_DENOVO="$OUTDIR/denovo" - - -nextflow run ../ --fastq $READS_DIR $config \ ---ref_genome $REF --ref_annotation $GFF -profile local --out_dir $OUT_REF --minimap2_opts '-uf --splice-flank=no' \ --w $OUT_REF/workspace -resume; - -echo "Doing de novo evaluation" -nextflow run ../ --fastq $READS_DIR $config --denovo -profile local --out_dir $OUT_DENOVO \ --w $OUT_DENOVO/workspace -resume; diff --git a/main.nf b/main.nf index fbb548a..5f82738 100644 --- a/main.nf +++ b/main.nf @@ -12,8 +12,9 @@ nextflow.enable.dsl = 2 include { fastq_ingress } from './lib/fastqingress' include { start_ping; end_ping } from './lib/ping' -include { reference_assembly } from './reference_assembly' -include { denovo_assembly } from './denovo_assembly' +include { reference_assembly } from './subworkflows/reference_assembly' +include { denovo_assembly } from './subworkflows/denovo_assembly' +include { gene_fusions } from './subworkflows/JAFFAL/gene_fusions' process summariseConcatReads { @@ -83,11 +84,11 @@ process preprocess_reads { input: tuple val(sample_id), path(input_reads) output: - tuple val(sample_id), path("${sample_id}_full_length_reads.fq"), emit: full_len_reads + tuple val(sample_id), path("${sample_id}_full_length_reads.fastq"), emit: full_len_reads path '*.tsv', emit: report script: """ - pychopper -t ${params.threads} ${params.pychopper_opts} ${input_reads} ${sample_id}_full_length_reads.fq + pychopper -t ${params.threads} ${params.pychopper_opts} ${input_reads} ${sample_id}_full_length_reads.fastq mv pychopper.tsv ${sample_id}_pychopper.tsv generate_pychopper_stats.py --data ${sample_id}_pychopper.tsv --output . @@ -285,13 +286,14 @@ process makeReport { path(seq_summaries), path(aln_stats), path(gffcmp_dir), - path(gff_annotation) + path(gff_annotation), + path(jaffal_csv) output: - path("wf-isoforms-*.html"), emit: report + path("wf-transcriptomes-*.html"), emit: report script: // Convert the sample_id arrayList. sids = new BlankSeparatedList(sample_ids) - def report_name = "wf-isoforms-report.html" + def report_name = "wf-transcriptomes-report.html" def OPT_ALN = denovo ? '' : "--alignment_stats ${aln_stats}" def OPT_DENOVO = denovo ? "--denovo" : '' def OPT_PC_REPORT = pychopper_report.name.startsWith('OPTIONAL_FILE') ? '' : "--pychop_report ${pychopper_report}" @@ -306,6 +308,7 @@ process makeReport { --gffcompare_dir $gffcmp_dir \ --gff_annotation $gff_annotation \ --transcript_table_cov_thresh $params.transcript_table_cov_thresh \ + --jaffal_csv $jaffal_csv $OPT_DENOVO """ } @@ -332,6 +335,9 @@ workflow pipeline { reads ref_genome ref_annotation + jaffal_refBase + jaffal_genome + jaffal_annotation main: map_sample_ids_cls = {it -> /* Harmonize tuples @@ -400,6 +406,10 @@ workflow pipeline { seq_for_transcriptome_build = sample_ids.flatten().combine(Channel.fromPath(params.ref_genome)) } + if (jaffal_refBase){ + gene_fusions(full_len_reads, jaffal_refBase, jaffal_genome, jaffal_annotation) + } + makeReport( software_versions, workflow_params, @@ -409,6 +419,7 @@ workflow pipeline { .join(m.stats) .join(run_gffcompare.out.gffcmp_dir) .join(merge_gff_bundles.out.gff) + .join(gene_fusions.out.results_csv) .toList().transpose().toList()) report = makeReport.out.report @@ -452,6 +463,11 @@ workflow pipeline { .map {it -> it[1]} .concat(makeReport.out.report) } + if (params.jaffal_refBase){ + results = results + .concat(gene_fusions.out.results + .map {it -> it[1]}) + } emit: results @@ -466,52 +482,60 @@ workflow { fastq = file(params.fastq, type: "file") - if (!fastq.exists()) { - println("--fastq: File doesn't exist, check path.") - exit 1 + error = null + + if (!fastq.exists()) { + error = "--fastq: File doesn't exist, check path." } if (!params.denovo && !params.ref_genome){ - println("--ref_genome must be supplied unless doing de novo assembly (--denovo)") - exit 1 + error = "--ref_genome must be supplied unless doing de novo assembly (--denovo)" } - if (params.ref_genome){ ref_genome = file(params.ref_genome, type: "file") if (!ref_genome.exists()) { - println("--ref_genome: File doesn't exist, check path.") - exit 1 + error = "--ref_genome: File doesn't exist, check path." } }else { ref_genome = file("$projectDir/data/OPTIONAL_FILE") } if (params.denovo && params.ref_annotation) { - println("Reference annotation with de denovo assembly is not supported") - exit 1 + error = "Reference annotation with de denovo assembly is not supported" } if (params.ref_annotation){ ref_annotation = file(params.ref_annotation, type: "file") if (!ref_annotation.exists()) { - println("--annotation: File doesn't exist, check path.") - exit 1 + error = "--annotation: File doesn't exist, check path." } }else{ ref_annotation = file("$projectDir/data/OPTIONAL_FILE") } + if (params.jaffal_refBase){ + jaffal_refBase = file(params.jaffal_refBase, type: "dir") + if (!jaffal_refBase.exists()) { + error = "--jaffa_refBase: Directory doesn't exist, check path." + } + }else{ + jaffal_refBase = null + } - reads = fastq_ingress([ - "input":params.fastq, - "sample":params.sample, - "sample_sheet":params.sample_sheet, - "sanitize": params.sanitize_fastq, - "output":params.out_dir]) + if (error){ + println(error) + }else{ + reads = fastq_ingress([ + "input":params.fastq, + "sample":params.sample, + "sample_sheet":params.sample_sheet, + "sanitize": params.sanitize_fastq, + "output":params.out_dir]) - pipeline(reads, ref_genome, ref_annotation) + pipeline(reads, ref_genome, ref_annotation, jaffal_refBase, params.jaffal_genome, params.jaffal_annotation) - output(pipeline.out.results) + output(pipeline.out.results) - end_ping(pipeline.out.telemetry) + end_ping(pipeline.out.telemetry) + } } diff --git a/nextflow.config b/nextflow.config index 74be708..eb361f5 100644 --- a/nextflow.config +++ b/nextflow.config @@ -119,21 +119,30 @@ params { // Minimum probability for i consecutive minimizers to be different between read and representative: min_prob_no_hits = 0.1 + ////// Fusion detection parameters + jaffal_refBase = null + jaffal_genome = "hg38" + jaffal_annotation = "genCode22" + // The default location of the JAFFA src directory when running in EPI2ME-Labs environment + // This needs overriding if running elsewhere + jaffal_dir = "/home/epi2melabs/JAFFA" + wf { example_cmd = [ "--fastq test_data/fastq", "--ref_genome test_data/SIRV_150601a.fasta", - "--ref_annotation test_data/SIRV_isofroms.gtf" + "--ref_annotation test_data/SIRV_isofroms.gtf", + "--jaffal_refBase chr20/", + "--jaffal_genome hg38", + "--jaffal_annotation genCode22" ] } - - } manifest { - name = 'epi2me-labs/wf-isoforms' + name = 'epi2me-labs/wf-transcriptomes' author = 'Oxford Nanopore Technologies' - homePage = 'https://github.com/epi2me-labs/wf-isoforms' + homePage = 'https://github.com/epi2me-labs/wf-transcriptomes' description = 'RNA/cDNA isoform analysis workflow' mainScript = 'main.nf' nextflowVersion = '>=20.10.0' @@ -151,7 +160,7 @@ executor { // other profiles may override. process { withLabel:isoforms { - container = "ontresearch/wf-isoforms:${params.wfversion}" + container = "ontresearch/wf-transcriptomes:${params.wfversion}" } shell = ['/bin/bash', '-euo', 'pipefail'] } @@ -201,7 +210,7 @@ profiles { queue = "${params.aws_queue}" memory = '8G' withLabel:isoforms { - container = "${params.aws_image_prefix}-wf-isoforms:${params.wfversion}" + container = "${params.aws_image_prefix}-wf-transcriptomes:${params.wfversion}" } shell = ['/bin/bash', '-euo', 'pipefail'] } diff --git a/nextflow_schema.json b/nextflow_schema.json index b8aa3db..4d0d2c0 100644 --- a/nextflow_schema.json +++ b/nextflow_schema.json @@ -1,9 +1,9 @@ { "$schema": "http://json-schema.org/draft-07/schema", "$id": "https://raw.githubusercontent.com/./master/nextflow_schema.json", - "title": "epi2me-labs/wf-isoforms", + "title": "epi2me-labs/wf-transcriptomes", "description": "Isoform detection and characterisation.", - "url": "https://github.com/epi2me-labs/wf-isoforms", + "url": "https://github.com/epi2me-labs/wf-transcriptomes", "type": "object", "definitions": { "basic_input_output_options": { @@ -14,13 +14,13 @@ "properties": { "out_dir": { "type": "string", - "default": "output", "format": "directory-path", + "default": "output", "description": "Directory for output of all user-facing files." }, "fastq": { "type": "string", - "format": "path", + "format": "file-path", "demo_data": "${projectDir}/test_data/fastq", "description": "A fastq file or directory containing fastq input files or directories of input files.", "help_text": "If directories named \\\"barcode*\\\" are found under the `--fastq` directory the data is assumed to be multiplex and each barcode directory will be processed independently. If `.fastq(.gz)` files are found under the `--fastq` directory the sample is assumed to not be multiplexed. In this second case `--samples` should be a simple name rather than a CSV file." @@ -99,7 +99,7 @@ "reference_wf_options": { "title": "Options for reference-based workflow", "type": "object", - "description": "Parameters that are used solely for the referenc-guided workflow", + "description": "Parameters that are used solely for the reference-guided workflow", "properties": { "plot_gffcmp_stats": { "type": "boolean", @@ -219,6 +219,34 @@ } } }, + "fusion_detection_options": { + "title": "Gene fusion detection options", + "type": "object", + "description": "Parameters for gene fusion detection", + "properties": { + "jaffal_refBase": { + "type": "string", + "format": "path", + "description": "JAFFAl reference genome directory" + }, + "jaffal_genome": { + "type": "string", + "description": "Genome reference prefix. e.g. hg38", + "default": "hg38" + }, + "jaffal_annotation": { + "type": "string", + "description": "Annotation prefix", + "default": "genCode22" + }, + "jaffal_dir": { + "type": "string", + "format": "path", + "description": "Path to JAFFAL git code directory. Defaults is epi2me-labs container location", + "default": "/home/epi2melabs/JAFFA" + } + } + }, "meta_data": { "title": "Meta Data", "type": "object", @@ -266,6 +294,9 @@ { "$ref": "#/definitions/denovo_wf_options" }, + { + "$ref": "#/definitions/fusion_detection_options" + }, { "$ref": "#/definitions/meta_data" }, @@ -299,7 +330,7 @@ } }, "docs": { - "intro": "## Introduction\n\nThis workflow identifies RNA isoforms using either cDNA or direct RNA (dRNA) \nOxford Nanopore reads.\n\n### Preprocesing\ncDNA reads are initially preprocessed by [pychopper](https://github.com/epi2me-labs/pychopper) \nfor the identification of full-length reads, as well as trimming and orientation correction (This step is omitted for \n direct RNA reads).\n\n### Reference-aided approach\n* Full length reads are mapped to a supplied reference genome using [minimap2](https://github.com/lh3/minimap2)\n* Transcripts are assembled by [stringtie](http://ccb.jhu.edu/software/stringtie) \nin long read mode (with or without a guide reference annotation) to generate the GFF annotation.\n* The annotation generated by the pipeline is compared to the reference annotation. \nusing [gffcompare](http://ccb.jhu.edu/software/stringtie/gffcompare.shtml)\n\n### de novo-based approach (experimental!)\n* Sequence clusters are generated using [isONclust2](https://github.com/nanoporetech/isONclust2)\n * If a reference genome is supplied, cluster quality metrics are determined by comparing \n with clusters generated from a minimap2 alignment.\n* A consensus sequence for each cluster is generated using [spoa](https://github.com/rvaser/spoa)\n* Three rounds of polishing using racon and minimap2 to give a final polished CDS for each gene.\n* Full-length reads are then mapped to these polished CDS.\n* Transcripts are assembled by stringtie as for the reference-based approach.\n* __Note__: This approach is currently not supported with direct RNA reads.\n\n### Workflow inputs\n- Directory containing cDNA/direct RNA reads. Or a directory containing subdirectories each with reads from different samples\n (in fastq/fastq.gz format)\n- Reference genome in fasta format (required for reference-based assembly).\n- Optional reference annotation in GFF2/3 format.", + "intro": "## Introduction\n\nThis workflow identifies RNA isoforms using either cDNA or direct RNA (dRNA) \nOxford Nanopore reads.\n\n### Preprocesing\ncDNA reads are initially preprocessed by [pychopper](https://github.com/epi2me-labs/pychopper) \nfor the identification of full-length reads, as well as trimming and orientation correction (This step is omitted for \n direct RNA reads).\n\n\n### Transcript assembly\n\n#### Reference-aided transcript assembly approach\n* Full length reads are mapped to a supplied reference genome using [minimap2](https://github.com/lh3/minimap2)\n* Transcripts are assembled by [stringtie](http://ccb.jhu.edu/software/stringtie) \nin long read mode (with or without a guide reference annotation) to generate the GFF annotation.\n* The annotation generated by the pipeline is compared to the reference annotation. \nusing [gffcompare](http://ccb.jhu.edu/software/stringtie/gffcompare.shtml)\n\n#### de novo-based transcript assembly (experimental!)\n* Sequence clusters are generated using [isONclust2](https://github.com/nanoporetech/isONclust2)\n * If a reference genome is supplied, cluster quality metrics are determined by comparing \n with clusters generated from a minimap2 alignment.\n* A consensus sequence for each cluster is generated using [spoa](https://github.com/rvaser/spoa)\n* Three rounds of polishing using racon and minimap2 to give a final polished CDS for each gene.\n* Full-length reads are then mapped to these polished CDS.\n* Transcripts are assembled by stringtie as for the reference-based approach.\n* __Note__: This approach is currently not supported with direct RNA reads.\n\n### Fusion gene detection\nFusion gene detection is performed using [JAFFA](https://github.com/Oshlack/JAFFA), with the JAFFAL extension for use \nwith ONT long reads. \n\n### Workflow inputs\n- Directory containing cDNA/direct RNA reads. Or a directory containing subdirectories each with reads from different samples\n (in fastq/fastq.gz format)\n- Reference genome in fasta format (required for reference-based assembly).\n- Optional reference annotation in GFF2/3 format.\n- For fusion detection, JAFFAL reference files (see Quickstart) \n", "links": "## Useful links\n\n* [nextflow](https://www.nextflow.io/)\n* [docker](https://www.docker.com/products/docker-desktop)\n* [Singularity](https://sylabs.io/singularity/)\n* [conda](https://docs.conda.io/en/latest/miniconda.html)\n* [racon](https://github.com/isovic/racon)\n* [spoa](https://github.com/rvaser/spoa)\n* [inONclust](https://github.com/ksahlin/isONclust)\n* [isONclust2](https://github.com/nanoporetech/isONclust2)" } } \ No newline at end of file diff --git a/subworkflows/JAFFAL/build_jaffal_ref.sh b/subworkflows/JAFFAL/build_jaffal_ref.sh new file mode 100644 index 0000000..13f4793 --- /dev/null +++ b/subworkflows/JAFFAL/build_jaffal_ref.sh @@ -0,0 +1,2 @@ +#!/bin/sh + diff --git a/subworkflows/JAFFAL/gene_fusions.nf b/subworkflows/JAFFAL/gene_fusions.nf new file mode 100644 index 0000000..5b8f550 --- /dev/null +++ b/subworkflows/JAFFAL/gene_fusions.nf @@ -0,0 +1,40 @@ + +process jaffal{ + label "isoforms" + input: + tuple val(sample_id), path(fastq) + path refBase + val genome + val annotation + output: + tuple val(sample_id), path("jaffal_output_$sample_id"), emit: results + tuple val(sample_id), path("jaffal_output_$sample_id/*jaffa_results.csv"), emit: results_csv + script: + """ + JAFFAOUT=jaffal_output_$sample_id + $params.jaffal_dir/tools/bin/bpipe run \ + -n $params.threads \ + -p jaffa_output="\$JAFFAOUT/" \ + -p refBase=$refBase \ + -p genome=$genome \ + -p annotation=$annotation \ + -p fastqInputFormat="*.fastq" \ + $params.jaffal_dir/JAFFAL.groovy \ + $fastq + mv "\$JAFFAOUT/jaffa_results.csv" "\$JAFFAOUT/${sample_id}_jaffa_results.csv" + """ +} + +// workflow module +workflow gene_fusions { + take: + fastq + refBase + genome + annotation + main: + jaffal(fastq, refBase, genome, annotation) + emit: + results_csv = jaffal.out.results_csv + results = jaffal.out.results +} diff --git a/subworkflows/JAFFAL/install_jaffa.sh b/subworkflows/JAFFAL/install_jaffa.sh new file mode 100755 index 0000000..15ded88 --- /dev/null +++ b/subworkflows/JAFFAL/install_jaffa.sh @@ -0,0 +1,21 @@ +#!/bin/bash +set -e + +git clone https://github.com/Oshlack/JAFFA.git && +cd JAFFA +git checkout 24b1c3b + +cp ../subworkflows/JAFFAL/install_linux64.sh . +./install_linux64.sh + +# JAFFA uses tools.groovy to locate binaries. We modify it as we only need a small subset or they are already +# included in our env + +# Tools to be compiled from src/ +bin=$(realpath tools/bin) +echo "//Tools built locally" >> tools.groovy +declare -a tools=("reformat" "extract_seq_from_fasta" "make_simple_read_table" "process_transcriptome_align_table" "make_3_gene_fusion_table") +for b in "${tools[@]}"; do + echo "$b=\"$bin/$b\"" >> tools.groovy +done +echo "minimap2=\"minimap2\"" >> tools.groovy diff --git a/subworkflows/JAFFAL/install_linux64.sh b/subworkflows/JAFFAL/install_linux64.sh new file mode 100755 index 0000000..af5d472 --- /dev/null +++ b/subworkflows/JAFFAL/install_linux64.sh @@ -0,0 +1,111 @@ +#!/bin/bash + +# 21/06/22: This script has been modified to install only those applications needed for epi2melabs/wf-transcriptomes + +## This script will install the tools required for the JAFFA pipeline. +## It will fetched each tool from the web and placed into the tools/ subdirectory. +## Paths to all installed tools can be found in the file tools.groovy at the +## end of execution of this script. These paths can be changed if a different +## version of software is required. Note that R must be installed manually +## +## Last Modified: Sep. 2021 by Nadia Davidson + +mkdir -p tools/bin +cd tools + +#a list of which programs need to be installed +commands="bpipe reformat extract_seq_from_fasta make_simple_read_table process_transcriptome_align_table make_3_gene_fusion_table dedupe" + +#installation methods +function bpipe_install { + wget -O bpipe-0.9.9.2.tar.gz https://github.com/ssadedin/bpipe/releases/download/0.9.9.2/bpipe-0.9.9.2.tar.gz + tar -zxvf bpipe-0.9.9.2.tar.gz ; rm bpipe-0.9.9.2.tar.gz + ln -s $PWD/bpipe-0.9.9.2/bin/* $PWD/bin/ +} + +function make_3_gene_fusion_table_install { + g++ -std=c++11 -O3 -o bin/make_3_gene_fusion_table ../src/make_3_gene_fusion_table.c++ +} + +function extract_seq_from_fasta_install { + g++ -std=c++11 -O3 -o bin/extract_seq_from_fasta ../src/extract_seq_from_fasta.c++ +} + +function make_simple_read_table_install { + g++ -std=c++11 -O3 -o bin/make_simple_read_table ../src/make_simple_read_table.c++ +} + +function process_transcriptome_align_table_install { + g++ -std=c++11 -O3 -o bin/process_transcriptome_align_table ../src/process_transcriptome_align_table.c++ +} + +function make_count_table_install { + g++ -O3 -o bin/make_count_table ../src/make_count_table.c++ +} + +function dedupe_install { + wget --no-check-certificate https://sourceforge.net/projects/bbmap/files/BBMap_36.59.tar.gz + tar -zxvf BBMap_36.59.tar.gz + rm BBMap_36.59.tar.gz + for script in `ls $PWD/bbmap/*.sh` ; do + s=`basename $script` + s_pre=`echo $s | sed 's/.sh//g'` + echo "$PWD/bbmap/$s \$@" > $PWD/bin/$s_pre + chmod +x $PWD/bin/$s_pre + done +} + +#function bypass_genomic_alignment_install { +# g++ -std=c++11 -O3 -o bin/bypass_genomic_alignment ../src/bypass_genomic_alignment.c++ +#} + +#Check if the version of gcc is >= 4.9 +gcc_version=`gcc -dumpversion` +gcc_check=`echo -e "$gcc_version\n4.9" | sort -n | tail -n1` +if [[ $gcc_chek = "4.9" ]] +then + echo "Your version of gcc is $gcc_version." + echo "gcc must be >= 4.9 to install JAFFA. Exiting..." + exit 1 +fi + +echo "gcc check passed" + +echo "// Path to tools used by the JAFFA pipeline" > ../tools.groovy + +for c in $commands ; do + c_path=`which $PWD/bin/$c 2>/dev/null` + if [ -z $c_path ] ; then + echo "$c not found, fetching it" + ${c}_install + c_path=`which $PWD/bin/$c 2>/dev/null` + fi + echo "$c=\"$c_path\"" >> ../tools.groovy +done + +#finally check that R is install +R_path=`which R 2>/dev/null` +if [ -z $R_path ] ; then + echo "R not found!" + echo "Please go to http://www.r-project.org/ and follow the installation instructions." + echo "Note that the IRanges R package must be installed." +fi +echo "R=\"$R_path\"" >> ../tools.groovy + +#loop through commands to check they are all installed +echo "Checking that all required tools were installed:" +Final_message="All commands installed successfully!" +for c in $commands ; do + c_path=`which $PWD/bin/$c 2>/dev/null` + if [ -z $c_path ] ; then + echo -n "WARNING: $c could not be found!!!! " + echo "You will need to download and install $c manually, then add its path to tools.groovy" + Final_message="WARNING: One or more command did not install successfully. See warning messages above. \ + You will need to correct this before running JAFFA." + else + echo "$c looks like it has been installed" + fi +done +echo "**********************************************************" +echo $Final_message + diff --git a/subworkflows/JAFFAL/tools.groovy b/subworkflows/JAFFAL/tools.groovy new file mode 100644 index 0000000..934e16b --- /dev/null +++ b/subworkflows/JAFFAL/tools.groovy @@ -0,0 +1,10 @@ +// Path to tools used by the JAFFA pipeline + +// Conda-installable tools +bpipe="bpipe" +//trimmomatic="trimmomatic" +R="/usr/bin/R" +minimap2="minimap2" +dedupe="dedupe" + + diff --git a/denovo_assembly.nf b/subworkflows/denovo_assembly.nf similarity index 100% rename from denovo_assembly.nf rename to subworkflows/denovo_assembly.nf diff --git a/reference_assembly.nf b/subworkflows/reference_assembly.nf similarity index 100% rename from reference_assembly.nf rename to subworkflows/reference_assembly.nf