diff --git a/.gitlab-ci.yml b/.gitlab-ci.yml
index 70c55df..21547a2 100644
--- a/.gitlab-ci.yml
+++ b/.gitlab-ci.yml
@@ -8,5 +8,8 @@ variables:
# The workflow should define `--out_dir`, the CI template sets this.
# Only common file inputs and option values need to be given here
# (not things such as -profile)
- NF_WORKFLOW_OPTS: "--fastq test_data/fastq \
- --ref_genome test_data/SIRV_150601a.fasta --ref_annotation test_data/SIRV_isoforms.gtf"
+ NF_BEFORE_SCRIPT: |
+ wget -O test_data.tar.gz https://ont-exd-int-s3-euwst1-epi2me-labs.s3.amazonaws.com/wf-isoforms/wf-isoforms_test_data.tar.gz && tar -xzvf test_data.tar.gz
+ NF_WORKFLOW_OPTS: "--fastq ERR6053095_chr20.fastq \
+ --ref_genome chr20/hg38_chr20.fa --ref_annotation chr20/gencode.v22.annotation.chr20.gtf \
+ --jaffal_refBase chr20/ --jaffal_genome hg38_chr20 --jaffal_annotation genCode22"
diff --git a/.gitmodules b/.gitmodules
new file mode 100644
index 0000000..e69de29
diff --git a/.pre-commit-config.yaml b/.pre-commit-config.yaml
index ecd70f9..14b9aa2 100644
--- a/.pre-commit-config.yaml
+++ b/.pre-commit-config.yaml
@@ -23,4 +23,7 @@ repos:
- id: flake8
additional_dependencies:
- flake8-import-order==0.18.1
- entry: flake8 bin --import-order-style google --statistics
\ No newline at end of file
+ - flake8-docstrings==1.6.0
+ - flake8-rst-docstrings==0.2.5
+ - flake8-forbid-visual-indent==0.0.2
+ entry: flake8 bin --import-order-style google --statistics
diff --git a/CHANGELOG.md b/CHANGELOG.md
index 42db068..3c36ff2 100644
--- a/CHANGELOG.md
+++ b/CHANGELOG.md
@@ -6,13 +6,17 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
## [unreleased]
### Changed
-- Skip unnecessary conversion to fasta from fastq
-- Fastqingress metadata map
+
## [v0.1.4]
+### Added
+- JAFFAL fusion detectoion subworkflow
### Changed
- Args parser for fastqingress
- Set out_dir option type to ensure output is written to correct directory on Windows
+- Skip unnecessary conversion to fasta from fastq
+- Fastqingress metadata map
+- Changed workflow name to wf-transcriptomes
## [v0.1.3]
### Changed
diff --git a/Dockerfile b/Dockerfile
index 49c9148..a6fca02 100644
--- a/Dockerfile
+++ b/Dockerfile
@@ -15,6 +15,11 @@ RUN \
&& rm -rf $CONDA_DIR/lib/python3.*/site-packages/pip \
&& find $CONDA_DIR -name '__pycache__' -type d -exec rm -rf '{}' '+'
+
USER $WF_UID
WORKDIR $HOME
+# Install JAFFA
+ADD subworkflows $HOME/subworkflows
+RUN /bin/sh -c $HOME/subworkflows/JAFFAL/install_jaffa.sh
+
diff --git a/README.md b/README.md
index 7274f20..07edbcd 100644
--- a/README.md
+++ b/README.md
@@ -1,4 +1,4 @@
-# wf-isoforms
+# wf-transcriptomes
This repository contains a [nextflow](https://www.nextflow.io/) workflow
for assembly and annotation of transcripts from Oxford Nanopore cDNA or direct RNA reads.
@@ -16,14 +16,17 @@ cDNA reads are initially preprocessed by [pychopper](https://github.com/epi2me-l
for the identification of full-length reads, as well as trimming and orientation correction (This step is omitted for
direct RNA reads).
-### Reference-aided approach
+
+### Transcript assembly
+
+#### Reference-aided transcript assembly approach
* Full length reads are mapped to a supplied reference genome using [minimap2](https://github.com/lh3/minimap2)
* Transcripts are assembled by [stringtie](http://ccb.jhu.edu/software/stringtie)
in long read mode (with or without a guide reference annotation) to generate the GFF annotation.
* The annotation generated by the pipeline is compared to the reference annotation.
using [gffcompare](http://ccb.jhu.edu/software/stringtie/gffcompare.shtml)
-### de novo-based approach (experimental!)
+#### de novo-based transcript assembly (experimental!)
* Sequence clusters are generated using [isONclust2](https://github.com/nanoporetech/isONclust2)
* If a reference genome is supplied, cluster quality metrics are determined by comparing
with clusters generated from a minimap2 alignment.
@@ -33,11 +36,17 @@ using [gffcompare](http://ccb.jhu.edu/software/stringtie/gffcompare.shtml)
* Transcripts are assembled by stringtie as for the reference-based approach.
* __Note__: This approach is currently not supported with direct RNA reads.
+### Fusion gene detection
+Fusion gene detection is performed using [JAFFA](https://github.com/Oshlack/JAFFA), with the JAFFAL extension for use
+with ONT long reads.
+
### Workflow inputs
- Directory containing cDNA/direct RNA reads. Or a directory containing subdirectories each with reads from different samples
(in fastq/fastq.gz format)
- Reference genome in fasta format (required for reference-based assembly).
-- Optional reference annotation in GFF2/3 format.## Quickstart
+- Optional reference annotation in GFF2/3 format.
+- For fusion detection, JAFFAL reference files (see Quickstart)
+## Quickstart
The workflow uses [nextflow](https://www.nextflow.io/) to manage compute and
software resources, as such nextflow will need to be installed before attempting
@@ -54,31 +63,31 @@ It is not required to clone or download the git repository in order to run the w
For more information on running EPI2ME Labs workflows [visit out website](https://labs.epi2me.io/wfindex).
-**Workflow options**
+### Workflow options
To obtain the workflow, having installed `nextflow`, users can run:
```
-nextflow run epi2me-labs/wf-isoforms --help
+nextflow run epi2me-labs/wf-transcriptomes --help
```
to see the options for the workflow.
+**Download demonstration data**
+A small test dataset is provided for the purposes of testing the workflow software. It consists of reads, reference,
+and annotations from human chromosome 20 only.
+It can be downloaded using:
+```shell
+wget -O test_data.tar.gz https://ont-exd-int-s3-euwst1-epi2me-labs.s3.amazonaws.com/wf-isoforms/wf-isoforms_test_data.tar.gz
+tar -xzvf test_data.tar.gz
+```
-**Example execution of a workflow for reference-based transcript assembly**
-
-This example uses a synthetic SIRV dataset, so we need to tell minimap2 about the non-canonical splice junctions with
---minimap2_opts '-uf --splice-flank=no'
+**Example execution of a workflow for reference-based transcript assembly and fusion detection**
```
OUTPUT=~/output;
-nextflow run wf-isoforms/ --fastq test_data/fastq --ref_genome test_data/SIRV_150601a.fasta --ref_annotation test_data/SIRV_isofroms.gtf
---minimap2_opts '-uf --splice-flank=no' --out_dir outdir -w workspace_dir -profile conda -resume
-```
-
-```
-# To evaluate the workflow on a larger Drosophila dataset
-./evaluation/run_evaluation_dmel.sh outdir
+nexflow run epi2me-labs/wf-transcriptomes --fastq ERR6053095_chr20.fastq --ref_genome chr20/hg38_chr20.fa --ref_annotation chr20/gencode.v22.annotation.chr20.gtf \
+ --jaffal_refBase chr20/ --jaffal_genome hg38_chr20 --jaffal_annotation genCode22" --out_dir outdir -w workspace_dir -profile conda -resume
```
**Example workflow for denovo transcript assembly**
@@ -91,7 +100,7 @@ A full list of options can be seen in nextflow_schema.json. Below are some commo
- Threshold for including isoforms into interactive table `transcript_table_cov_thresh = 50`
- Run the denovo pipeline `denovo = true` (default false)
-- To run the workflow with direct RNA reads `--direct_rna` (skips the pychopper step).
+- To run the workflow with direct RNA reads `--direct_rna` (this just skips the pychopper step).
Pychopper and minimap2 can take options via `minimap2_opts` and `pychopper_opts`, for example:
@@ -102,12 +111,74 @@ Pychopper and minimap2 can take options via `minimap2_opts` and `pychopper_opts`
- pychopper needs to know which cDNA synthesis kit used
- SQK-PCS109: use `pychopper_opts = '-k PCS109'` (default)
- SQK-PCS110: use `pychopper_opts = '-k PCS110'`
+ - SQK-PCS11: use `pychopper_opts = '-k PCS111'`
- pychopper can use one of two available backends for identifying primers in the raw reads
- nhmmscan `pychopper opts = '-m phmm'`
- edlib `pychopper opts = '-m edlib'`
-__Note__: edlib is set by default in the config as it's quite a lot faster. However it may be less sensitive than nhmmscan.
+__Note__: edlib is set by default in the config as it's quite a lot faster. However, it may be less sensitive than nhmmscan.
+
+### Fusion detection
+
+JAFFAL from the [JAFFA](https://github.com/Oshlack/JAFFA)
+package is used to identify potential fusion transcripts. To get this this working, there are a couple of things that need doing first.
+
+**Install JAFFA**
+
+to install JAFFA and it's dependencies run the folllowing:
+```shell
+cd wf-transcriptomes/
+./subworkflows/JAFFAL/install_jaffa.sh
+```
+
+**Prepare JAFFAL reference data**
+
+To use pre-processed reference files for the hg38 genome and GENCODE v22 annotation (as used in the JFFAAL paper),
+do:
+```shell
+mkdir jaffal_data_dir
+cd jaffal_data_dir/
+wf-transcriptomes/download_jaffal_references.sh
+````
+
+To use alternative genome and annotation files, they should be prepared as described
+[here](https://github.com/Oshlack/JAFFA/wiki/FAQandTroubleshooting#how-can-i-generate-the-reference-files-for-a-non-supported-genome)
+
+**Specifying the location of the JAFFA code and reference directories**
+
+`--jaffal_dir`
+This is the directory made by running install_jaffa.sh as shown above
+
+`--jaffal_refBase`
+The directory containing the reference data prepared for use with JAFFAL
+
+
+**JAFFAL annotation and genome files**
+
+The prepared JAFFAL reference files will look something like `hg38_chr20_genCode22.fa`. To enable JAFFAL to find these
+files `--jaffal_genome` should be set to `hg38_chr20` and `--jaffal_annotation` to `genCode22`
+
+
+__JAFFAL Notes__:
+g++ must be installed. JAFFAL is not currently working on Mac M1 (osx-arm64 architecture). If there are no fusion transcripts
+detected, the workflow will terminate with an error at the JAFFAL stage. If this happens,
+skip the JAFFAL stage by omitting ` --jaffal_refBase`
+
+
+## Workflow outputs
+* an HTML report document detailing the primary findings of the workflow.
+* for each sample:
+ * [gffcomapre](https://ccb.jhu.edu/software/stringtie/gffcompare.shtml) output directories
+ * read_aln_stats.tsv - alignment summary statistics
+ * transcriptome.fas - the assembled transcriptome
+ * merged_transcritptome.fas - annotated, assembled transcriptome
+ * [jaffal](https://github.com/Oshlack/JAFFA) ooutput directories
+
+### Fusion detection outputs
+in `${out_dir}/jaffal_output_${sample_id}` you will find:
+* jaffa_results.csv - the csv results summary file
+* jaffa_results.fasta - fusion transcritpt sequences
## Useful links
* [nextflow](https://www.nextflow.io/)
diff --git a/bin/ping.py b/bin/ping.py
index 689ff54..de8c9a0 100755
--- a/bin/ping.py
+++ b/bin/ping.py
@@ -53,7 +53,7 @@ def main():
hostname=args.hostname,
opsys=args.opsys
).send_workflow_ping(
- workflow='wf-isoforms',
+ workflow='wf-transcriptomes',
message=args.message,
revision=args.revision,
commit=args.commit,
diff --git a/bin/report.py b/bin/report.py
index f027b71..07a83c9 100755
--- a/bin/report.py
+++ b/bin/report.py
@@ -400,10 +400,10 @@ def gff_compare_plots(report, gffcompare_outdirs: Path, sample_ids):
tracking = df_track.groupby("Overlaps").count().reset_index()
tracking.Overlaps = tracking.Overlaps.map(names)
tracking['Percent'] = tracking.Count * 100 / tracking.Count.sum()
- tracking = tracking.sort_values("Overlaps")
+ tracking = tracking.sort_values("Count", ascending=False)
track_bar = bars.simple_hbar(
- tracking['Overlaps'].values.tolist(),
- tracking['Percent'].values.tolist(),
+ list(reversed(tracking['Overlaps'].values.tolist())),
+ list(reversed(tracking['Percent'].values.tolist())),
colors=Colors.cerulean, title=id_)
tracking_dfs.append(tracking)
@@ -426,6 +426,7 @@ def gff_compare_plots(report, gffcompare_outdirs: Path, sample_ids):
track_table = DataTable(
columns=cols, source=ColumnDataSource(tracking),
index_position=None, width=500)
+
tabs.append(Panel(
child=gridplot([track_bar, track_table], ncols=2), title=id_)
)
@@ -602,7 +603,7 @@ def transcript_table(report, df_tmaps, covr_threshold):
# drop some columns for the big table and do some filtering
section.markdown('''
- ### Query transcript table
+ ### Isoforms table
Low coverage transcripts are removed to speed up the table viewing.
Coverage threshold can be set with the parameter
@@ -650,6 +651,8 @@ def transcript_table(report, df_tmaps, covr_threshold):
df['parent gene iso num'] = df.apply(
lambda x: gb.loc[(x.ref_gene_id, x.sample_id), 'num_isoforms'], axis=1)
+ # Uncalssified transcritps should not be lumped togetehr
+ df.loc[df.class_code == 'u', 'parent gene iso num'] = None
df.sort_values('parent gene iso num', inplace=True, ascending=True)
@@ -795,6 +798,43 @@ def seq_stats_tabs(report, sample_ids, stats):
section.plot(Tabs(tabs=tabs))
+def jaffal_table(report, sample_ids, result_csvs):
+ """Make a table of fusion transcripts identified by JAFFAL."""
+ cols = [
+ 'sample_id', 'fusion genes', 'chrom1', 'chrom2', 'spanning reads',
+ 'classification', 'known']
+ dfs = []
+ for csv, sid in zip(result_csvs, sample_ids):
+ df = pd.read_csv(csv)
+ df['sample_id'] = sid
+ sid_col = df.pop('sample_id')
+ df.insert(0, 'sample_id', sid_col)
+ dfs.append(df)
+
+ df = pd.concat(dfs)
+ df = df[cols]
+ df['chroms'] = df.chrom1.astype(str) + ':' + df.chrom2.astype(str)
+ df.rename(columns={
+ 'spanning reads': 'nreads',
+ 'fusion genes': 'genes'}, inplace=True)
+ df.drop(columns=['chrom1', 'chrom2'], inplace=True)
+ section = report.add_section()
+ section.markdown("""
+ ### JAFFAL fusion transcript summary
+
+ This table summarizes putative fusion transcripts identified
+ by [JAFFAL](https://github.com/Oshlack/JAFFA/).
+
+ * genes: the gene symbols of the fusion partners
+ * nreads: The number of reads supporting the fusion
+ * classification: JAFFAL's classification
+ * known: whether this fusion is in the given set of known gene fusions
+ * chroms: the respective, original chromosome location of the two partner
+ genes
+ """)
+ section.table(df)
+
+
def main():
"""Run the entry point."""
parser = argparse.ArgumentParser()
@@ -833,15 +873,17 @@ def main():
parser.add_argument(
"--cluster_qc_dirs", required=False, type=str, default=None, nargs='*',
help="Directory with various cluster quality csvs")
+ parser.add_argument(
+ "--jaffal_csv", required=False, type=str, default=None, nargs='*',
+ help="Path to JAFFAL results csv")
parser.add_argument('--denovo', dest='denovo', action='store_true')
args = parser.parse_args()
- print('denovo', args.denovo)
sample_ids = args.sample_ids
report = WFReport(
- "Transcript isoform report", "wf-isoforms",
+ "Transcript isoform report", "wf-transcriptomes",
revision=args.revision, commit=args.commit)
# Add reads summary section
@@ -853,7 +895,9 @@ def main():
section.markdown('''
### Read mapping summary
- Output of [seqkit](https://bioinf.shenwei.me/seqkit/) bam -s''')
+ Summary of minimap2 mapping from
+ [seqkit](https://bioinf.shenwei.me/seqkit/)
+ `seqkit bam -s`''')
section.table(df_aln_stats)
@@ -877,6 +921,9 @@ def main():
if args.cluster_qc_dirs is not None:
cluster_quality(args.cluster_qc_dirs, report, sample_ids)
+ if args.jaffal_csv is not None:
+ jaffal_table(report, sample_ids, args.jaffal_csv)
+
# Arguments and software versions
report.add_section(
section=scomponents.version_table(args.versions))
diff --git a/docs/header.md b/docs/header.md
index 2affd15..31f01a7 100644
--- a/docs/header.md
+++ b/docs/header.md
@@ -1,4 +1,4 @@
-# wf-isoforms
+# wf-transcriptomes
This repository contains a [nextflow](https://www.nextflow.io/) workflow
for assembly and annotation of transcripts from Oxford Nanopore cDNA or direct RNA reads.
diff --git a/docs/intro.md b/docs/intro.md
index c532336..19fac27 100644
--- a/docs/intro.md
+++ b/docs/intro.md
@@ -8,14 +8,17 @@ cDNA reads are initially preprocessed by [pychopper](https://github.com/epi2me-l
for the identification of full-length reads, as well as trimming and orientation correction (This step is omitted for
direct RNA reads).
-### Reference-aided approach
+
+### Transcript assembly
+
+#### Reference-aided transcript assembly approach
* Full length reads are mapped to a supplied reference genome using [minimap2](https://github.com/lh3/minimap2)
* Transcripts are assembled by [stringtie](http://ccb.jhu.edu/software/stringtie)
in long read mode (with or without a guide reference annotation) to generate the GFF annotation.
* The annotation generated by the pipeline is compared to the reference annotation.
using [gffcompare](http://ccb.jhu.edu/software/stringtie/gffcompare.shtml)
-### de novo-based approach (experimental!)
+#### de novo-based transcript assembly (experimental!)
* Sequence clusters are generated using [isONclust2](https://github.com/nanoporetech/isONclust2)
* If a reference genome is supplied, cluster quality metrics are determined by comparing
with clusters generated from a minimap2 alignment.
@@ -25,8 +28,13 @@ using [gffcompare](http://ccb.jhu.edu/software/stringtie/gffcompare.shtml)
* Transcripts are assembled by stringtie as for the reference-based approach.
* __Note__: This approach is currently not supported with direct RNA reads.
+### Fusion gene detection
+Fusion gene detection is performed using [JAFFA](https://github.com/Oshlack/JAFFA), with the JAFFAL extension for use
+with ONT long reads.
+
### Workflow inputs
- Directory containing cDNA/direct RNA reads. Or a directory containing subdirectories each with reads from different samples
(in fastq/fastq.gz format)
- Reference genome in fasta format (required for reference-based assembly).
-- Optional reference annotation in GFF2/3 format.
\ No newline at end of file
+- Optional reference annotation in GFF2/3 format.
+- For fusion detection, JAFFAL reference files (see Quickstart)
diff --git a/docs/quickstart.md b/docs/quickstart.md
index ff1bff0..0076331 100644
--- a/docs/quickstart.md
+++ b/docs/quickstart.md
@@ -15,31 +15,31 @@ It is not required to clone or download the git repository in order to run the w
For more information on running EPI2ME Labs workflows [visit out website](https://labs.epi2me.io/wfindex).
-**Workflow options**
+### Workflow options
To obtain the workflow, having installed `nextflow`, users can run:
```
-nextflow run epi2me-labs/wf-isoforms --help
+nextflow run epi2me-labs/wf-transcriptomes --help
```
to see the options for the workflow.
+**Download demonstration data**
+A small test dataset is provided for the purposes of testing the workflow software. It consists of reads, reference,
+and annotations from human chromosome 20 only.
+It can be downloaded using:
+```shell
+wget -O test_data.tar.gz https://ont-exd-int-s3-euwst1-epi2me-labs.s3.amazonaws.com/wf-isoforms/wf-isoforms_test_data.tar.gz
+tar -xzvf test_data.tar.gz
+```
-**Example execution of a workflow for reference-based transcript assembly**
-
-This example uses a synthetic SIRV dataset, so we need to tell minimap2 about the non-canonical splice junctions with
---minimap2_opts '-uf --splice-flank=no'
+**Example execution of a workflow for reference-based transcript assembly and fusion detection**
```
OUTPUT=~/output;
-nextflow run wf-isoforms/ --fastq test_data/fastq --ref_genome test_data/SIRV_150601a.fasta --ref_annotation test_data/SIRV_isofroms.gtf
---minimap2_opts '-uf --splice-flank=no' --out_dir outdir -w workspace_dir -profile conda -resume
-```
-
-```
-# To evaluate the workflow on a larger Drosophila dataset
-./evaluation/run_evaluation_dmel.sh outdir
+nexflow run epi2me-labs/wf-transcriptomes --fastq ERR6053095_chr20.fastq --ref_genome chr20/hg38_chr20.fa --ref_annotation chr20/gencode.v22.annotation.chr20.gtf \
+ --jaffal_refBase chr20/ --jaffal_genome hg38_chr20 --jaffal_annotation genCode22" --out_dir outdir -w workspace_dir -profile conda -resume
```
**Example workflow for denovo transcript assembly**
@@ -52,7 +52,7 @@ A full list of options can be seen in nextflow_schema.json. Below are some commo
- Threshold for including isoforms into interactive table `transcript_table_cov_thresh = 50`
- Run the denovo pipeline `denovo = true` (default false)
-- To run the workflow with direct RNA reads `--direct_rna` (skips the pychopper step).
+- To run the workflow with direct RNA reads `--direct_rna` (this just skips the pychopper step).
Pychopper and minimap2 can take options via `minimap2_opts` and `pychopper_opts`, for example:
@@ -63,9 +63,71 @@ Pychopper and minimap2 can take options via `minimap2_opts` and `pychopper_opts`
- pychopper needs to know which cDNA synthesis kit used
- SQK-PCS109: use `pychopper_opts = '-k PCS109'` (default)
- SQK-PCS110: use `pychopper_opts = '-k PCS110'`
+ - SQK-PCS11: use `pychopper_opts = '-k PCS111'`
- pychopper can use one of two available backends for identifying primers in the raw reads
- nhmmscan `pychopper opts = '-m phmm'`
- edlib `pychopper opts = '-m edlib'`
-__Note__: edlib is set by default in the config as it's quite a lot faster. However it may be less sensitive than nhmmscan.
+__Note__: edlib is set by default in the config as it's quite a lot faster. However, it may be less sensitive than nhmmscan.
+
+### Fusion detection
+
+JAFFAL from the [JAFFA](https://github.com/Oshlack/JAFFA)
+package is used to identify potential fusion transcripts. To get this this working, there are a couple of things that need doing first.
+
+**Install JAFFA**
+
+to install JAFFA and it's dependencies run the folllowing:
+```shell
+cd wf-transcriptomes/
+./subworkflows/JAFFAL/install_jaffa.sh
+```
+
+**Prepare JAFFAL reference data**
+
+To use pre-processed reference files for the hg38 genome and GENCODE v22 annotation (as used in the JFFAAL paper),
+do:
+```shell
+mkdir jaffal_data_dir
+cd jaffal_data_dir/
+wf-transcriptomes/download_jaffal_references.sh
+````
+
+To use alternative genome and annotation files, they should be prepared as described
+[here](https://github.com/Oshlack/JAFFA/wiki/FAQandTroubleshooting#how-can-i-generate-the-reference-files-for-a-non-supported-genome)
+
+**Specifying the location of the JAFFA code and reference directories**
+
+`--jaffal_dir`
+This is the directory made by running install_jaffa.sh as shown above
+
+`--jaffal_refBase`
+The directory containing the reference data prepared for use with JAFFAL
+
+
+**JAFFAL annotation and genome files**
+
+The prepared JAFFAL reference files will look something like `hg38_chr20_genCode22.fa`. To enable JAFFAL to find these
+files `--jaffal_genome` should be set to `hg38_chr20` and `--jaffal_annotation` to `genCode22`
+
+
+__JAFFAL Notes__:
+g++ must be installed. JAFFAL is not currently working on Mac M1 (osx-arm64 architecture). If there are no fusion transcripts
+detected, the workflow will terminate with an error at the JAFFAL stage. If this happens,
+skip the JAFFAL stage by omitting ` --jaffal_refBase`
+
+
+## Workflow outputs
+* an HTML report document detailing the primary findings of the workflow.
+* for each sample:
+ * [gffcomapre](https://ccb.jhu.edu/software/stringtie/gffcompare.shtml) output directories
+ * read_aln_stats.tsv - alignment summary statistics
+ * transcriptome.fas - the assembled transcriptome
+ * merged_transcritptome.fas - annotated, assembled transcriptome
+ * [jaffal](https://github.com/Oshlack/JAFFA) ooutput directories
+
+### Fusion detection outputs
+in `${out_dir}/jaffal_output_${sample_id}` you will find:
+* jaffa_results.csv - the csv results summary file
+* jaffa_results.fasta - fusion transcritpt sequences
diff --git a/download_jaffal_references.sh b/download_jaffal_references.sh
new file mode 100644
index 0000000..ec12648
--- /dev/null
+++ b/download_jaffal_references.sh
@@ -0,0 +1,5 @@
+#!/bin/sh
+
+#Download the data. We should we move the data out of Figshare?
+wget -O JAFFA_REFERENCE_FILES_HG38_GENCODE22.V2.tar.gz https://figshare.com/ndownloader/files/25410494
+tar -zxvf JAFFA_REFERENCE_FILES_HG38_GENCODE22.V2.tar.gz
\ No newline at end of file
diff --git a/environment.yaml b/environment.yaml
index 74769a4..815c298 100644
--- a/environment.yaml
+++ b/environment.yaml
@@ -1,8 +1,8 @@
-name: epi2melabs-wf-isoforms
+name: epi2melabs-wf-transcriptomes
channels:
- epi2melabs
- - bioconda
- conda-forge
+ - bioconda
- defaults
dependencies:
- python==3.8.*
@@ -27,4 +27,8 @@ dependencies:
- parallel
- scikit-learn==1.0.2
- natsort
- - graphviz
\ No newline at end of file
+# Fusion detection dependencies
+# - bpipe=0.9.9.2
+ - java-jdk
+ - r-base
+ - gxx
\ No newline at end of file
diff --git a/evaluation/run_evaluation_dmel.sh b/evaluation/run_evaluation_dmel.sh
deleted file mode 100644
index 5da39ea..0000000
--- a/evaluation/run_evaluation_dmel.sh
+++ /dev/null
@@ -1,59 +0,0 @@
-#!/usr/bin/env bash
-
-# Usage: ./run_evaluation_dmel.sh pathto/outputdir
-
-# See the isONcorrect paper https://www.nature.com/articles/s41467-020-20340-8 where this dataset is described
-
-
-if [[ "$#" -lt 1 ]]; then
- echo "usage: run_evaluation_dmel.sh [nextflow.config]"
- exit 1
-fi
-
-if [[ "$#" -eq 1 ]]; then
- config=''
-fi
-
-if [[ "$#" -eq 2 ]]; then
- config="-c $2";
-fi
-
-OUTDIR=$1;
-
-FASTQ_URL="http://ftp.sra.ebi.ac.uk/vol1/fastq/ERR358/005/ERR3588905/ERR3588905_1.fastq.gz"
-REF_URL="http://ftp.ensembl.org/pub/release-99/fasta/drosophila_melanogaster/dna/Drosophila_melanogaster.BDGP6.28.dna.toplevel.fa.gz"
-GFF_URL="http://ftp.ensembl.org/pub/release-99/gff3/drosophila_melanogaster/Drosophila_melanogaster.BDGP6.28.99.gff3.gz"
-
-DATA_DIR="$OUTDIR/data"
-READS_DIR="$DATA_DIR/reads"
-FASTQ="$READS_DIR/ERR3588905_1.fastq.gz"
-REF="$DATA_DIR/Drosophila_melanogaster.BDGP6.28.dna.toplevel.fa"
-GFF="$DATA_DIR/Drosophila_melanogaster.BDGP6.28.99.gff3"
-
-mkdir -p $READS_DIR
-
-if [ ! -f $REF ];
-then (echo "downloading reference genome"; cd $DATA_DIR; curl -L -C - -O $REF_URL); gzip -d ${REF}.gz
-fi
-
-if [ ! -f $GFF ];
-then
- (echo "downloading reference annotation"; cd $DATA_DIR; curl -L -C - -O $GFF_URL); gzip -d ${GFF}.gz
-fi
-
-if [ ! -f $FASTQ ];
-then (echo "downloading reads"; cd $READS_DIR; curl -L -C - -O $FASTQ_URL); gzip -d ${FASTQ}.gz
-fi
-
-
-OUT_REF="$OUTDIR/ref"
-OUT_DENOVO="$OUTDIR/denovo"
-
-
-nextflow run ../ --fastq $READS_DIR $config \
---ref_genome $REF --ref_annotation $GFF -profile local --out_dir $OUT_REF --minimap2_opts '-uf --splice-flank=no' \
--w $OUT_REF/workspace -resume;
-
-echo "Doing de novo evaluation"
-nextflow run ../ --fastq $READS_DIR $config --denovo -profile local --out_dir $OUT_DENOVO \
--w $OUT_DENOVO/workspace -resume;
diff --git a/main.nf b/main.nf
index fbb548a..5f82738 100644
--- a/main.nf
+++ b/main.nf
@@ -12,8 +12,9 @@ nextflow.enable.dsl = 2
include { fastq_ingress } from './lib/fastqingress'
include { start_ping; end_ping } from './lib/ping'
-include { reference_assembly } from './reference_assembly'
-include { denovo_assembly } from './denovo_assembly'
+include { reference_assembly } from './subworkflows/reference_assembly'
+include { denovo_assembly } from './subworkflows/denovo_assembly'
+include { gene_fusions } from './subworkflows/JAFFAL/gene_fusions'
process summariseConcatReads {
@@ -83,11 +84,11 @@ process preprocess_reads {
input:
tuple val(sample_id), path(input_reads)
output:
- tuple val(sample_id), path("${sample_id}_full_length_reads.fq"), emit: full_len_reads
+ tuple val(sample_id), path("${sample_id}_full_length_reads.fastq"), emit: full_len_reads
path '*.tsv', emit: report
script:
"""
- pychopper -t ${params.threads} ${params.pychopper_opts} ${input_reads} ${sample_id}_full_length_reads.fq
+ pychopper -t ${params.threads} ${params.pychopper_opts} ${input_reads} ${sample_id}_full_length_reads.fastq
mv pychopper.tsv ${sample_id}_pychopper.tsv
generate_pychopper_stats.py --data ${sample_id}_pychopper.tsv --output .
@@ -285,13 +286,14 @@ process makeReport {
path(seq_summaries),
path(aln_stats),
path(gffcmp_dir),
- path(gff_annotation)
+ path(gff_annotation),
+ path(jaffal_csv)
output:
- path("wf-isoforms-*.html"), emit: report
+ path("wf-transcriptomes-*.html"), emit: report
script:
// Convert the sample_id arrayList.
sids = new BlankSeparatedList(sample_ids)
- def report_name = "wf-isoforms-report.html"
+ def report_name = "wf-transcriptomes-report.html"
def OPT_ALN = denovo ? '' : "--alignment_stats ${aln_stats}"
def OPT_DENOVO = denovo ? "--denovo" : ''
def OPT_PC_REPORT = pychopper_report.name.startsWith('OPTIONAL_FILE') ? '' : "--pychop_report ${pychopper_report}"
@@ -306,6 +308,7 @@ process makeReport {
--gffcompare_dir $gffcmp_dir \
--gff_annotation $gff_annotation \
--transcript_table_cov_thresh $params.transcript_table_cov_thresh \
+ --jaffal_csv $jaffal_csv
$OPT_DENOVO
"""
}
@@ -332,6 +335,9 @@ workflow pipeline {
reads
ref_genome
ref_annotation
+ jaffal_refBase
+ jaffal_genome
+ jaffal_annotation
main:
map_sample_ids_cls = {it ->
/* Harmonize tuples
@@ -400,6 +406,10 @@ workflow pipeline {
seq_for_transcriptome_build = sample_ids.flatten().combine(Channel.fromPath(params.ref_genome))
}
+ if (jaffal_refBase){
+ gene_fusions(full_len_reads, jaffal_refBase, jaffal_genome, jaffal_annotation)
+ }
+
makeReport(
software_versions,
workflow_params,
@@ -409,6 +419,7 @@ workflow pipeline {
.join(m.stats)
.join(run_gffcompare.out.gffcmp_dir)
.join(merge_gff_bundles.out.gff)
+ .join(gene_fusions.out.results_csv)
.toList().transpose().toList())
report = makeReport.out.report
@@ -452,6 +463,11 @@ workflow pipeline {
.map {it -> it[1]}
.concat(makeReport.out.report)
}
+ if (params.jaffal_refBase){
+ results = results
+ .concat(gene_fusions.out.results
+ .map {it -> it[1]})
+ }
emit:
results
@@ -466,52 +482,60 @@ workflow {
fastq = file(params.fastq, type: "file")
- if (!fastq.exists()) {
- println("--fastq: File doesn't exist, check path.")
- exit 1
+ error = null
+
+ if (!fastq.exists()) {
+ error = "--fastq: File doesn't exist, check path."
}
if (!params.denovo && !params.ref_genome){
- println("--ref_genome must be supplied unless doing de novo assembly (--denovo)")
- exit 1
+ error = "--ref_genome must be supplied unless doing de novo assembly (--denovo)"
}
-
if (params.ref_genome){
ref_genome = file(params.ref_genome, type: "file")
if (!ref_genome.exists()) {
- println("--ref_genome: File doesn't exist, check path.")
- exit 1
+ error = "--ref_genome: File doesn't exist, check path."
}
}else {
ref_genome = file("$projectDir/data/OPTIONAL_FILE")
}
if (params.denovo && params.ref_annotation) {
- println("Reference annotation with de denovo assembly is not supported")
- exit 1
+ error = "Reference annotation with de denovo assembly is not supported"
}
if (params.ref_annotation){
ref_annotation = file(params.ref_annotation, type: "file")
if (!ref_annotation.exists()) {
- println("--annotation: File doesn't exist, check path.")
- exit 1
+ error = "--annotation: File doesn't exist, check path."
}
}else{
ref_annotation = file("$projectDir/data/OPTIONAL_FILE")
}
+ if (params.jaffal_refBase){
+ jaffal_refBase = file(params.jaffal_refBase, type: "dir")
+ if (!jaffal_refBase.exists()) {
+ error = "--jaffa_refBase: Directory doesn't exist, check path."
+ }
+ }else{
+ jaffal_refBase = null
+ }
- reads = fastq_ingress([
- "input":params.fastq,
- "sample":params.sample,
- "sample_sheet":params.sample_sheet,
- "sanitize": params.sanitize_fastq,
- "output":params.out_dir])
+ if (error){
+ println(error)
+ }else{
+ reads = fastq_ingress([
+ "input":params.fastq,
+ "sample":params.sample,
+ "sample_sheet":params.sample_sheet,
+ "sanitize": params.sanitize_fastq,
+ "output":params.out_dir])
- pipeline(reads, ref_genome, ref_annotation)
+ pipeline(reads, ref_genome, ref_annotation, jaffal_refBase, params.jaffal_genome, params.jaffal_annotation)
- output(pipeline.out.results)
+ output(pipeline.out.results)
- end_ping(pipeline.out.telemetry)
+ end_ping(pipeline.out.telemetry)
+ }
}
diff --git a/nextflow.config b/nextflow.config
index 74be708..eb361f5 100644
--- a/nextflow.config
+++ b/nextflow.config
@@ -119,21 +119,30 @@ params {
// Minimum probability for i consecutive minimizers to be different between read and representative:
min_prob_no_hits = 0.1
+ ////// Fusion detection parameters
+ jaffal_refBase = null
+ jaffal_genome = "hg38"
+ jaffal_annotation = "genCode22"
+ // The default location of the JAFFA src directory when running in EPI2ME-Labs environment
+ // This needs overriding if running elsewhere
+ jaffal_dir = "/home/epi2melabs/JAFFA"
+
wf {
example_cmd = [
"--fastq test_data/fastq",
"--ref_genome test_data/SIRV_150601a.fasta",
- "--ref_annotation test_data/SIRV_isofroms.gtf"
+ "--ref_annotation test_data/SIRV_isofroms.gtf",
+ "--jaffal_refBase chr20/",
+ "--jaffal_genome hg38",
+ "--jaffal_annotation genCode22"
]
}
-
-
}
manifest {
- name = 'epi2me-labs/wf-isoforms'
+ name = 'epi2me-labs/wf-transcriptomes'
author = 'Oxford Nanopore Technologies'
- homePage = 'https://github.com/epi2me-labs/wf-isoforms'
+ homePage = 'https://github.com/epi2me-labs/wf-transcriptomes'
description = 'RNA/cDNA isoform analysis workflow'
mainScript = 'main.nf'
nextflowVersion = '>=20.10.0'
@@ -151,7 +160,7 @@ executor {
// other profiles may override.
process {
withLabel:isoforms {
- container = "ontresearch/wf-isoforms:${params.wfversion}"
+ container = "ontresearch/wf-transcriptomes:${params.wfversion}"
}
shell = ['/bin/bash', '-euo', 'pipefail']
}
@@ -201,7 +210,7 @@ profiles {
queue = "${params.aws_queue}"
memory = '8G'
withLabel:isoforms {
- container = "${params.aws_image_prefix}-wf-isoforms:${params.wfversion}"
+ container = "${params.aws_image_prefix}-wf-transcriptomes:${params.wfversion}"
}
shell = ['/bin/bash', '-euo', 'pipefail']
}
diff --git a/nextflow_schema.json b/nextflow_schema.json
index b8aa3db..4d0d2c0 100644
--- a/nextflow_schema.json
+++ b/nextflow_schema.json
@@ -1,9 +1,9 @@
{
"$schema": "http://json-schema.org/draft-07/schema",
"$id": "https://raw.githubusercontent.com/./master/nextflow_schema.json",
- "title": "epi2me-labs/wf-isoforms",
+ "title": "epi2me-labs/wf-transcriptomes",
"description": "Isoform detection and characterisation.",
- "url": "https://github.com/epi2me-labs/wf-isoforms",
+ "url": "https://github.com/epi2me-labs/wf-transcriptomes",
"type": "object",
"definitions": {
"basic_input_output_options": {
@@ -14,13 +14,13 @@
"properties": {
"out_dir": {
"type": "string",
- "default": "output",
"format": "directory-path",
+ "default": "output",
"description": "Directory for output of all user-facing files."
},
"fastq": {
"type": "string",
- "format": "path",
+ "format": "file-path",
"demo_data": "${projectDir}/test_data/fastq",
"description": "A fastq file or directory containing fastq input files or directories of input files.",
"help_text": "If directories named \\\"barcode*\\\" are found under the `--fastq` directory the data is assumed to be multiplex and each barcode directory will be processed independently. If `.fastq(.gz)` files are found under the `--fastq` directory the sample is assumed to not be multiplexed. In this second case `--samples` should be a simple name rather than a CSV file."
@@ -99,7 +99,7 @@
"reference_wf_options": {
"title": "Options for reference-based workflow",
"type": "object",
- "description": "Parameters that are used solely for the referenc-guided workflow",
+ "description": "Parameters that are used solely for the reference-guided workflow",
"properties": {
"plot_gffcmp_stats": {
"type": "boolean",
@@ -219,6 +219,34 @@
}
}
},
+ "fusion_detection_options": {
+ "title": "Gene fusion detection options",
+ "type": "object",
+ "description": "Parameters for gene fusion detection",
+ "properties": {
+ "jaffal_refBase": {
+ "type": "string",
+ "format": "path",
+ "description": "JAFFAl reference genome directory"
+ },
+ "jaffal_genome": {
+ "type": "string",
+ "description": "Genome reference prefix. e.g. hg38",
+ "default": "hg38"
+ },
+ "jaffal_annotation": {
+ "type": "string",
+ "description": "Annotation prefix",
+ "default": "genCode22"
+ },
+ "jaffal_dir": {
+ "type": "string",
+ "format": "path",
+ "description": "Path to JAFFAL git code directory. Defaults is epi2me-labs container location",
+ "default": "/home/epi2melabs/JAFFA"
+ }
+ }
+ },
"meta_data": {
"title": "Meta Data",
"type": "object",
@@ -266,6 +294,9 @@
{
"$ref": "#/definitions/denovo_wf_options"
},
+ {
+ "$ref": "#/definitions/fusion_detection_options"
+ },
{
"$ref": "#/definitions/meta_data"
},
@@ -299,7 +330,7 @@
}
},
"docs": {
- "intro": "## Introduction\n\nThis workflow identifies RNA isoforms using either cDNA or direct RNA (dRNA) \nOxford Nanopore reads.\n\n### Preprocesing\ncDNA reads are initially preprocessed by [pychopper](https://github.com/epi2me-labs/pychopper) \nfor the identification of full-length reads, as well as trimming and orientation correction (This step is omitted for \n direct RNA reads).\n\n### Reference-aided approach\n* Full length reads are mapped to a supplied reference genome using [minimap2](https://github.com/lh3/minimap2)\n* Transcripts are assembled by [stringtie](http://ccb.jhu.edu/software/stringtie) \nin long read mode (with or without a guide reference annotation) to generate the GFF annotation.\n* The annotation generated by the pipeline is compared to the reference annotation. \nusing [gffcompare](http://ccb.jhu.edu/software/stringtie/gffcompare.shtml)\n\n### de novo-based approach (experimental!)\n* Sequence clusters are generated using [isONclust2](https://github.com/nanoporetech/isONclust2)\n * If a reference genome is supplied, cluster quality metrics are determined by comparing \n with clusters generated from a minimap2 alignment.\n* A consensus sequence for each cluster is generated using [spoa](https://github.com/rvaser/spoa)\n* Three rounds of polishing using racon and minimap2 to give a final polished CDS for each gene.\n* Full-length reads are then mapped to these polished CDS.\n* Transcripts are assembled by stringtie as for the reference-based approach.\n* __Note__: This approach is currently not supported with direct RNA reads.\n\n### Workflow inputs\n- Directory containing cDNA/direct RNA reads. Or a directory containing subdirectories each with reads from different samples\n (in fastq/fastq.gz format)\n- Reference genome in fasta format (required for reference-based assembly).\n- Optional reference annotation in GFF2/3 format.",
+ "intro": "## Introduction\n\nThis workflow identifies RNA isoforms using either cDNA or direct RNA (dRNA) \nOxford Nanopore reads.\n\n### Preprocesing\ncDNA reads are initially preprocessed by [pychopper](https://github.com/epi2me-labs/pychopper) \nfor the identification of full-length reads, as well as trimming and orientation correction (This step is omitted for \n direct RNA reads).\n\n\n### Transcript assembly\n\n#### Reference-aided transcript assembly approach\n* Full length reads are mapped to a supplied reference genome using [minimap2](https://github.com/lh3/minimap2)\n* Transcripts are assembled by [stringtie](http://ccb.jhu.edu/software/stringtie) \nin long read mode (with or without a guide reference annotation) to generate the GFF annotation.\n* The annotation generated by the pipeline is compared to the reference annotation. \nusing [gffcompare](http://ccb.jhu.edu/software/stringtie/gffcompare.shtml)\n\n#### de novo-based transcript assembly (experimental!)\n* Sequence clusters are generated using [isONclust2](https://github.com/nanoporetech/isONclust2)\n * If a reference genome is supplied, cluster quality metrics are determined by comparing \n with clusters generated from a minimap2 alignment.\n* A consensus sequence for each cluster is generated using [spoa](https://github.com/rvaser/spoa)\n* Three rounds of polishing using racon and minimap2 to give a final polished CDS for each gene.\n* Full-length reads are then mapped to these polished CDS.\n* Transcripts are assembled by stringtie as for the reference-based approach.\n* __Note__: This approach is currently not supported with direct RNA reads.\n\n### Fusion gene detection\nFusion gene detection is performed using [JAFFA](https://github.com/Oshlack/JAFFA), with the JAFFAL extension for use \nwith ONT long reads. \n\n### Workflow inputs\n- Directory containing cDNA/direct RNA reads. Or a directory containing subdirectories each with reads from different samples\n (in fastq/fastq.gz format)\n- Reference genome in fasta format (required for reference-based assembly).\n- Optional reference annotation in GFF2/3 format.\n- For fusion detection, JAFFAL reference files (see Quickstart) \n",
"links": "## Useful links\n\n* [nextflow](https://www.nextflow.io/)\n* [docker](https://www.docker.com/products/docker-desktop)\n* [Singularity](https://sylabs.io/singularity/)\n* [conda](https://docs.conda.io/en/latest/miniconda.html)\n* [racon](https://github.com/isovic/racon)\n* [spoa](https://github.com/rvaser/spoa)\n* [inONclust](https://github.com/ksahlin/isONclust)\n* [isONclust2](https://github.com/nanoporetech/isONclust2)"
}
}
\ No newline at end of file
diff --git a/subworkflows/JAFFAL/build_jaffal_ref.sh b/subworkflows/JAFFAL/build_jaffal_ref.sh
new file mode 100644
index 0000000..13f4793
--- /dev/null
+++ b/subworkflows/JAFFAL/build_jaffal_ref.sh
@@ -0,0 +1,2 @@
+#!/bin/sh
+
diff --git a/subworkflows/JAFFAL/gene_fusions.nf b/subworkflows/JAFFAL/gene_fusions.nf
new file mode 100644
index 0000000..5b8f550
--- /dev/null
+++ b/subworkflows/JAFFAL/gene_fusions.nf
@@ -0,0 +1,40 @@
+
+process jaffal{
+ label "isoforms"
+ input:
+ tuple val(sample_id), path(fastq)
+ path refBase
+ val genome
+ val annotation
+ output:
+ tuple val(sample_id), path("jaffal_output_$sample_id"), emit: results
+ tuple val(sample_id), path("jaffal_output_$sample_id/*jaffa_results.csv"), emit: results_csv
+ script:
+ """
+ JAFFAOUT=jaffal_output_$sample_id
+ $params.jaffal_dir/tools/bin/bpipe run \
+ -n $params.threads \
+ -p jaffa_output="\$JAFFAOUT/" \
+ -p refBase=$refBase \
+ -p genome=$genome \
+ -p annotation=$annotation \
+ -p fastqInputFormat="*.fastq" \
+ $params.jaffal_dir/JAFFAL.groovy \
+ $fastq
+ mv "\$JAFFAOUT/jaffa_results.csv" "\$JAFFAOUT/${sample_id}_jaffa_results.csv"
+ """
+}
+
+// workflow module
+workflow gene_fusions {
+ take:
+ fastq
+ refBase
+ genome
+ annotation
+ main:
+ jaffal(fastq, refBase, genome, annotation)
+ emit:
+ results_csv = jaffal.out.results_csv
+ results = jaffal.out.results
+}
diff --git a/subworkflows/JAFFAL/install_jaffa.sh b/subworkflows/JAFFAL/install_jaffa.sh
new file mode 100755
index 0000000..15ded88
--- /dev/null
+++ b/subworkflows/JAFFAL/install_jaffa.sh
@@ -0,0 +1,21 @@
+#!/bin/bash
+set -e
+
+git clone https://github.com/Oshlack/JAFFA.git &&
+cd JAFFA
+git checkout 24b1c3b
+
+cp ../subworkflows/JAFFAL/install_linux64.sh .
+./install_linux64.sh
+
+# JAFFA uses tools.groovy to locate binaries. We modify it as we only need a small subset or they are already
+# included in our env
+
+# Tools to be compiled from src/
+bin=$(realpath tools/bin)
+echo "//Tools built locally" >> tools.groovy
+declare -a tools=("reformat" "extract_seq_from_fasta" "make_simple_read_table" "process_transcriptome_align_table" "make_3_gene_fusion_table")
+for b in "${tools[@]}"; do
+ echo "$b=\"$bin/$b\"" >> tools.groovy
+done
+echo "minimap2=\"minimap2\"" >> tools.groovy
diff --git a/subworkflows/JAFFAL/install_linux64.sh b/subworkflows/JAFFAL/install_linux64.sh
new file mode 100755
index 0000000..af5d472
--- /dev/null
+++ b/subworkflows/JAFFAL/install_linux64.sh
@@ -0,0 +1,111 @@
+#!/bin/bash
+
+# 21/06/22: This script has been modified to install only those applications needed for epi2melabs/wf-transcriptomes
+
+## This script will install the tools required for the JAFFA pipeline.
+## It will fetched each tool from the web and placed into the tools/ subdirectory.
+## Paths to all installed tools can be found in the file tools.groovy at the
+## end of execution of this script. These paths can be changed if a different
+## version of software is required. Note that R must be installed manually
+##
+## Last Modified: Sep. 2021 by Nadia Davidson
+
+mkdir -p tools/bin
+cd tools
+
+#a list of which programs need to be installed
+commands="bpipe reformat extract_seq_from_fasta make_simple_read_table process_transcriptome_align_table make_3_gene_fusion_table dedupe"
+
+#installation methods
+function bpipe_install {
+ wget -O bpipe-0.9.9.2.tar.gz https://github.com/ssadedin/bpipe/releases/download/0.9.9.2/bpipe-0.9.9.2.tar.gz
+ tar -zxvf bpipe-0.9.9.2.tar.gz ; rm bpipe-0.9.9.2.tar.gz
+ ln -s $PWD/bpipe-0.9.9.2/bin/* $PWD/bin/
+}
+
+function make_3_gene_fusion_table_install {
+ g++ -std=c++11 -O3 -o bin/make_3_gene_fusion_table ../src/make_3_gene_fusion_table.c++
+}
+
+function extract_seq_from_fasta_install {
+ g++ -std=c++11 -O3 -o bin/extract_seq_from_fasta ../src/extract_seq_from_fasta.c++
+}
+
+function make_simple_read_table_install {
+ g++ -std=c++11 -O3 -o bin/make_simple_read_table ../src/make_simple_read_table.c++
+}
+
+function process_transcriptome_align_table_install {
+ g++ -std=c++11 -O3 -o bin/process_transcriptome_align_table ../src/process_transcriptome_align_table.c++
+}
+
+function make_count_table_install {
+ g++ -O3 -o bin/make_count_table ../src/make_count_table.c++
+}
+
+function dedupe_install {
+ wget --no-check-certificate https://sourceforge.net/projects/bbmap/files/BBMap_36.59.tar.gz
+ tar -zxvf BBMap_36.59.tar.gz
+ rm BBMap_36.59.tar.gz
+ for script in `ls $PWD/bbmap/*.sh` ; do
+ s=`basename $script`
+ s_pre=`echo $s | sed 's/.sh//g'`
+ echo "$PWD/bbmap/$s \$@" > $PWD/bin/$s_pre
+ chmod +x $PWD/bin/$s_pre
+ done
+}
+
+#function bypass_genomic_alignment_install {
+# g++ -std=c++11 -O3 -o bin/bypass_genomic_alignment ../src/bypass_genomic_alignment.c++
+#}
+
+#Check if the version of gcc is >= 4.9
+gcc_version=`gcc -dumpversion`
+gcc_check=`echo -e "$gcc_version\n4.9" | sort -n | tail -n1`
+if [[ $gcc_chek = "4.9" ]]
+then
+ echo "Your version of gcc is $gcc_version."
+ echo "gcc must be >= 4.9 to install JAFFA. Exiting..."
+ exit 1
+fi
+
+echo "gcc check passed"
+
+echo "// Path to tools used by the JAFFA pipeline" > ../tools.groovy
+
+for c in $commands ; do
+ c_path=`which $PWD/bin/$c 2>/dev/null`
+ if [ -z $c_path ] ; then
+ echo "$c not found, fetching it"
+ ${c}_install
+ c_path=`which $PWD/bin/$c 2>/dev/null`
+ fi
+ echo "$c=\"$c_path\"" >> ../tools.groovy
+done
+
+#finally check that R is install
+R_path=`which R 2>/dev/null`
+if [ -z $R_path ] ; then
+ echo "R not found!"
+ echo "Please go to http://www.r-project.org/ and follow the installation instructions."
+ echo "Note that the IRanges R package must be installed."
+fi
+echo "R=\"$R_path\"" >> ../tools.groovy
+
+#loop through commands to check they are all installed
+echo "Checking that all required tools were installed:"
+Final_message="All commands installed successfully!"
+for c in $commands ; do
+ c_path=`which $PWD/bin/$c 2>/dev/null`
+ if [ -z $c_path ] ; then
+ echo -n "WARNING: $c could not be found!!!! "
+ echo "You will need to download and install $c manually, then add its path to tools.groovy"
+ Final_message="WARNING: One or more command did not install successfully. See warning messages above. \
+ You will need to correct this before running JAFFA."
+ else
+ echo "$c looks like it has been installed"
+ fi
+done
+echo "**********************************************************"
+echo $Final_message
+
diff --git a/subworkflows/JAFFAL/tools.groovy b/subworkflows/JAFFAL/tools.groovy
new file mode 100644
index 0000000..934e16b
--- /dev/null
+++ b/subworkflows/JAFFAL/tools.groovy
@@ -0,0 +1,10 @@
+// Path to tools used by the JAFFA pipeline
+
+// Conda-installable tools
+bpipe="bpipe"
+//trimmomatic="trimmomatic"
+R="/usr/bin/R"
+minimap2="minimap2"
+dedupe="dedupe"
+
+
diff --git a/denovo_assembly.nf b/subworkflows/denovo_assembly.nf
similarity index 100%
rename from denovo_assembly.nf
rename to subworkflows/denovo_assembly.nf
diff --git a/reference_assembly.nf b/subworkflows/reference_assembly.nf
similarity index 100%
rename from reference_assembly.nf
rename to subworkflows/reference_assembly.nf