This commit is contained in:
Neil Horner 2022-08-01 12:01:01 +00:00
parent be28906ab2
commit 956c80573a
24 changed files with 555 additions and 154 deletions

View File

@ -8,5 +8,8 @@ variables:
# The workflow should define `--out_dir`, the CI template sets this.
# Only common file inputs and option values need to be given here
# (not things such as -profile)
NF_WORKFLOW_OPTS: "--fastq test_data/fastq \
--ref_genome test_data/SIRV_150601a.fasta --ref_annotation test_data/SIRV_isoforms.gtf"
NF_BEFORE_SCRIPT: |
wget -O test_data.tar.gz https://ont-exd-int-s3-euwst1-epi2me-labs.s3.amazonaws.com/wf-isoforms/wf-isoforms_test_data.tar.gz && tar -xzvf test_data.tar.gz
NF_WORKFLOW_OPTS: "--fastq ERR6053095_chr20.fastq \
--ref_genome chr20/hg38_chr20.fa --ref_annotation chr20/gencode.v22.annotation.chr20.gtf \
--jaffal_refBase chr20/ --jaffal_genome hg38_chr20 --jaffal_annotation genCode22"

0
.gitmodules vendored Normal file
View File

View File

@ -23,4 +23,7 @@ repos:
- id: flake8
additional_dependencies:
- flake8-import-order==0.18.1
- flake8-docstrings==1.6.0
- flake8-rst-docstrings==0.2.5
- flake8-forbid-visual-indent==0.0.2
entry: flake8 bin --import-order-style google --statistics

View File

@ -6,13 +6,17 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
## [unreleased]
### Changed
- Skip unnecessary conversion to fasta from fastq
- Fastqingress metadata map
## [v0.1.4]
### Added
- JAFFAL fusion detectoion subworkflow
### Changed
- Args parser for fastqingress
- Set out_dir option type to ensure output is written to correct directory on Windows
- Skip unnecessary conversion to fasta from fastq
- Fastqingress metadata map
- Changed workflow name to wf-transcriptomes
## [v0.1.3]
### Changed

View File

@ -15,6 +15,11 @@ RUN \
&& rm -rf $CONDA_DIR/lib/python3.*/site-packages/pip \
&& find $CONDA_DIR -name '__pycache__' -type d -exec rm -rf '{}' '+'
USER $WF_UID
WORKDIR $HOME
# Install JAFFA
ADD subworkflows $HOME/subworkflows
RUN /bin/sh -c $HOME/subworkflows/JAFFAL/install_jaffa.sh

109
README.md
View File

@ -1,4 +1,4 @@
# wf-isoforms
# wf-transcriptomes
This repository contains a [nextflow](https://www.nextflow.io/) workflow
for assembly and annotation of transcripts from Oxford Nanopore cDNA or direct RNA reads.
@ -16,14 +16,17 @@ cDNA reads are initially preprocessed by [pychopper](https://github.com/epi2me-l
for the identification of full-length reads, as well as trimming and orientation correction (This step is omitted for
direct RNA reads).
### Reference-aided approach
### Transcript assembly
#### Reference-aided transcript assembly approach
* Full length reads are mapped to a supplied reference genome using [minimap2](https://github.com/lh3/minimap2)
* Transcripts are assembled by [stringtie](http://ccb.jhu.edu/software/stringtie)
in long read mode (with or without a guide reference annotation) to generate the GFF annotation.
* The annotation generated by the pipeline is compared to the reference annotation.
using [gffcompare](http://ccb.jhu.edu/software/stringtie/gffcompare.shtml)
### de novo-based approach (experimental!)
#### de novo-based transcript assembly (experimental!)
* Sequence clusters are generated using [isONclust2](https://github.com/nanoporetech/isONclust2)
* If a reference genome is supplied, cluster quality metrics are determined by comparing
with clusters generated from a minimap2 alignment.
@ -33,11 +36,17 @@ using [gffcompare](http://ccb.jhu.edu/software/stringtie/gffcompare.shtml)
* Transcripts are assembled by stringtie as for the reference-based approach.
* __Note__: This approach is currently not supported with direct RNA reads.
### Fusion gene detection
Fusion gene detection is performed using [JAFFA](https://github.com/Oshlack/JAFFA), with the JAFFAL extension for use
with ONT long reads.
### Workflow inputs
- Directory containing cDNA/direct RNA reads. Or a directory containing subdirectories each with reads from different samples
(in fastq/fastq.gz format)
- Reference genome in fasta format (required for reference-based assembly).
- Optional reference annotation in GFF2/3 format.## Quickstart
- Optional reference annotation in GFF2/3 format.
- For fusion detection, JAFFAL reference files (see Quickstart)
## Quickstart
The workflow uses [nextflow](https://www.nextflow.io/) to manage compute and
software resources, as such nextflow will need to be installed before attempting
@ -54,31 +63,31 @@ It is not required to clone or download the git repository in order to run the w
For more information on running EPI2ME Labs workflows [visit out website](https://labs.epi2me.io/wfindex).
**Workflow options**
### Workflow options
To obtain the workflow, having installed `nextflow`, users can run:
```
nextflow run epi2me-labs/wf-isoforms --help
nextflow run epi2me-labs/wf-transcriptomes --help
```
to see the options for the workflow.
**Download demonstration data**
A small test dataset is provided for the purposes of testing the workflow software. It consists of reads, reference,
and annotations from human chromosome 20 only.
It can be downloaded using:
```shell
wget -O test_data.tar.gz https://ont-exd-int-s3-euwst1-epi2me-labs.s3.amazonaws.com/wf-isoforms/wf-isoforms_test_data.tar.gz
tar -xzvf test_data.tar.gz
```
**Example execution of a workflow for reference-based transcript assembly**
This example uses a synthetic SIRV dataset, so we need to tell minimap2 about the non-canonical splice junctions with
--minimap2_opts '-uf --splice-flank=no'
**Example execution of a workflow for reference-based transcript assembly and fusion detection**
```
OUTPUT=~/output;
nextflow run wf-isoforms/ --fastq test_data/fastq --ref_genome test_data/SIRV_150601a.fasta --ref_annotation test_data/SIRV_isofroms.gtf
--minimap2_opts '-uf --splice-flank=no' --out_dir outdir -w workspace_dir -profile conda -resume
```
```
# To evaluate the workflow on a larger Drosophila dataset
./evaluation/run_evaluation_dmel.sh outdir
nexflow run epi2me-labs/wf-transcriptomes --fastq ERR6053095_chr20.fastq --ref_genome chr20/hg38_chr20.fa --ref_annotation chr20/gencode.v22.annotation.chr20.gtf \
--jaffal_refBase chr20/ --jaffal_genome hg38_chr20 --jaffal_annotation genCode22" --out_dir outdir -w workspace_dir -profile conda -resume
```
**Example workflow for denovo transcript assembly**
@ -91,7 +100,7 @@ A full list of options can be seen in nextflow_schema.json. Below are some commo
- Threshold for including isoforms into interactive table `transcript_table_cov_thresh = 50`
- Run the denovo pipeline `denovo = true` (default false)
- To run the workflow with direct RNA reads `--direct_rna` (skips the pychopper step).
- To run the workflow with direct RNA reads `--direct_rna` (this just skips the pychopper step).
Pychopper and minimap2 can take options via `minimap2_opts` and `pychopper_opts`, for example:
@ -102,12 +111,74 @@ Pychopper and minimap2 can take options via `minimap2_opts` and `pychopper_opts`
- pychopper needs to know which cDNA synthesis kit used
- SQK-PCS109: use `pychopper_opts = '-k PCS109'` (default)
- SQK-PCS110: use `pychopper_opts = '-k PCS110'`
- SQK-PCS11: use `pychopper_opts = '-k PCS111'`
- pychopper can use one of two available backends for identifying primers in the raw reads
- nhmmscan `pychopper opts = '-m phmm'`
- edlib `pychopper opts = '-m edlib'`
__Note__: edlib is set by default in the config as it's quite a lot faster. However it may be less sensitive than nhmmscan.
__Note__: edlib is set by default in the config as it's quite a lot faster. However, it may be less sensitive than nhmmscan.
### Fusion detection
JAFFAL from the [JAFFA](https://github.com/Oshlack/JAFFA)
package is used to identify potential fusion transcripts. To get this this working, there are a couple of things that need doing first.
**Install JAFFA**
to install JAFFA and it's dependencies run the folllowing:
```shell
cd wf-transcriptomes/
./subworkflows/JAFFAL/install_jaffa.sh
```
**Prepare JAFFAL reference data**
To use pre-processed reference files for the hg38 genome and GENCODE v22 annotation (as used in the JFFAAL paper),
do:
```shell
mkdir jaffal_data_dir
cd jaffal_data_dir/
wf-transcriptomes/download_jaffal_references.sh
````
To use alternative genome and annotation files, they should be prepared as described
[here](https://github.com/Oshlack/JAFFA/wiki/FAQandTroubleshooting#how-can-i-generate-the-reference-files-for-a-non-supported-genome)
**Specifying the location of the JAFFA code and reference directories**
`--jaffal_dir`
This is the directory made by running install_jaffa.sh as shown above
`--jaffal_refBase`
The directory containing the reference data prepared for use with JAFFAL
**JAFFAL annotation and genome files**
The prepared JAFFAL reference files will look something like `hg38_chr20_genCode22.fa`. To enable JAFFAL to find these
files `--jaffal_genome` should be set to `hg38_chr20` and `--jaffal_annotation` to `genCode22`
__JAFFAL Notes__:
g++ must be installed. JAFFAL is not currently working on Mac M1 (osx-arm64 architecture). If there are no fusion transcripts
detected, the workflow will terminate with an error at the JAFFAL stage. If this happens,
skip the JAFFAL stage by omitting ` --jaffal_refBase`
## Workflow outputs
* an HTML report document detailing the primary findings of the workflow.
* for each sample:
* [gffcomapre](https://ccb.jhu.edu/software/stringtie/gffcompare.shtml) output directories
* read_aln_stats.tsv - alignment summary statistics
* transcriptome.fas - the assembled transcriptome
* merged_transcritptome.fas - annotated, assembled transcriptome
* [jaffal](https://github.com/Oshlack/JAFFA) ooutput directories
### Fusion detection outputs
in `${out_dir}/jaffal_output_${sample_id}` you will find:
* jaffa_results.csv - the csv results summary file
* jaffa_results.fasta - fusion transcritpt sequences
## Useful links
* [nextflow](https://www.nextflow.io/)

View File

@ -53,7 +53,7 @@ def main():
hostname=args.hostname,
opsys=args.opsys
).send_workflow_ping(
workflow='wf-isoforms',
workflow='wf-transcriptomes',
message=args.message,
revision=args.revision,
commit=args.commit,

View File

@ -400,10 +400,10 @@ def gff_compare_plots(report, gffcompare_outdirs: Path, sample_ids):
tracking = df_track.groupby("Overlaps").count().reset_index()
tracking.Overlaps = tracking.Overlaps.map(names)
tracking['Percent'] = tracking.Count * 100 / tracking.Count.sum()
tracking = tracking.sort_values("Overlaps")
tracking = tracking.sort_values("Count", ascending=False)
track_bar = bars.simple_hbar(
tracking['Overlaps'].values.tolist(),
tracking['Percent'].values.tolist(),
list(reversed(tracking['Overlaps'].values.tolist())),
list(reversed(tracking['Percent'].values.tolist())),
colors=Colors.cerulean, title=id_)
tracking_dfs.append(tracking)
@ -426,6 +426,7 @@ def gff_compare_plots(report, gffcompare_outdirs: Path, sample_ids):
track_table = DataTable(
columns=cols, source=ColumnDataSource(tracking),
index_position=None, width=500)
tabs.append(Panel(
child=gridplot([track_bar, track_table], ncols=2), title=id_)
)
@ -602,7 +603,7 @@ def transcript_table(report, df_tmaps, covr_threshold):
# drop some columns for the big table and do some filtering
section.markdown('''
### Query transcript table
### Isoforms table
Low coverage transcripts are removed to speed up the table viewing. <br>
Coverage threshold can be set with the parameter
@ -650,6 +651,8 @@ def transcript_table(report, df_tmaps, covr_threshold):
df['parent gene iso num'] = df.apply(
lambda x: gb.loc[(x.ref_gene_id, x.sample_id), 'num_isoforms'], axis=1)
# Uncalssified transcritps should not be lumped togetehr
df.loc[df.class_code == 'u', 'parent gene iso num'] = None
df.sort_values('parent gene iso num', inplace=True, ascending=True)
@ -795,6 +798,43 @@ def seq_stats_tabs(report, sample_ids, stats):
section.plot(Tabs(tabs=tabs))
def jaffal_table(report, sample_ids, result_csvs):
"""Make a table of fusion transcripts identified by JAFFAL."""
cols = [
'sample_id', 'fusion genes', 'chrom1', 'chrom2', 'spanning reads',
'classification', 'known']
dfs = []
for csv, sid in zip(result_csvs, sample_ids):
df = pd.read_csv(csv)
df['sample_id'] = sid
sid_col = df.pop('sample_id')
df.insert(0, 'sample_id', sid_col)
dfs.append(df)
df = pd.concat(dfs)
df = df[cols]
df['chroms'] = df.chrom1.astype(str) + ':' + df.chrom2.astype(str)
df.rename(columns={
'spanning reads': 'nreads',
'fusion genes': 'genes'}, inplace=True)
df.drop(columns=['chrom1', 'chrom2'], inplace=True)
section = report.add_section()
section.markdown("""
### JAFFAL fusion transcript summary
This table summarizes putative fusion transcripts identified
by [JAFFAL](https://github.com/Oshlack/JAFFA/).
* genes: the gene symbols of the fusion partners
* nreads: The number of reads supporting the fusion
* classification: JAFFAL's classification
* known: whether this fusion is in the given set of known gene fusions
* chroms: the respective, original chromosome location of the two partner
genes
""")
section.table(df)
def main():
"""Run the entry point."""
parser = argparse.ArgumentParser()
@ -833,15 +873,17 @@ def main():
parser.add_argument(
"--cluster_qc_dirs", required=False, type=str, default=None, nargs='*',
help="Directory with various cluster quality csvs")
parser.add_argument(
"--jaffal_csv", required=False, type=str, default=None, nargs='*',
help="Path to JAFFAL results csv")
parser.add_argument('--denovo', dest='denovo', action='store_true')
args = parser.parse_args()
print('denovo', args.denovo)
sample_ids = args.sample_ids
report = WFReport(
"Transcript isoform report", "wf-isoforms",
"Transcript isoform report", "wf-transcriptomes",
revision=args.revision, commit=args.commit)
# Add reads summary section
@ -853,7 +895,9 @@ def main():
section.markdown('''
### Read mapping summary
Output of [seqkit](https://bioinf.shenwei.me/seqkit/) bam -s''')
Summary of minimap2 mapping from
[seqkit](https://bioinf.shenwei.me/seqkit/)
`seqkit bam -s`''')
section.table(df_aln_stats)
@ -877,6 +921,9 @@ def main():
if args.cluster_qc_dirs is not None:
cluster_quality(args.cluster_qc_dirs, report, sample_ids)
if args.jaffal_csv is not None:
jaffal_table(report, sample_ids, args.jaffal_csv)
# Arguments and software versions
report.add_section(
section=scomponents.version_table(args.versions))

View File

@ -1,4 +1,4 @@
# wf-isoforms
# wf-transcriptomes
This repository contains a [nextflow](https://www.nextflow.io/) workflow
for assembly and annotation of transcripts from Oxford Nanopore cDNA or direct RNA reads.

View File

@ -8,14 +8,17 @@ cDNA reads are initially preprocessed by [pychopper](https://github.com/epi2me-l
for the identification of full-length reads, as well as trimming and orientation correction (This step is omitted for
direct RNA reads).
### Reference-aided approach
### Transcript assembly
#### Reference-aided transcript assembly approach
* Full length reads are mapped to a supplied reference genome using [minimap2](https://github.com/lh3/minimap2)
* Transcripts are assembled by [stringtie](http://ccb.jhu.edu/software/stringtie)
in long read mode (with or without a guide reference annotation) to generate the GFF annotation.
* The annotation generated by the pipeline is compared to the reference annotation.
using [gffcompare](http://ccb.jhu.edu/software/stringtie/gffcompare.shtml)
### de novo-based approach (experimental!)
#### de novo-based transcript assembly (experimental!)
* Sequence clusters are generated using [isONclust2](https://github.com/nanoporetech/isONclust2)
* If a reference genome is supplied, cluster quality metrics are determined by comparing
with clusters generated from a minimap2 alignment.
@ -25,8 +28,13 @@ using [gffcompare](http://ccb.jhu.edu/software/stringtie/gffcompare.shtml)
* Transcripts are assembled by stringtie as for the reference-based approach.
* __Note__: This approach is currently not supported with direct RNA reads.
### Fusion gene detection
Fusion gene detection is performed using [JAFFA](https://github.com/Oshlack/JAFFA), with the JAFFAL extension for use
with ONT long reads.
### Workflow inputs
- Directory containing cDNA/direct RNA reads. Or a directory containing subdirectories each with reads from different samples
(in fastq/fastq.gz format)
- Reference genome in fasta format (required for reference-based assembly).
- Optional reference annotation in GFF2/3 format.
- For fusion detection, JAFFAL reference files (see Quickstart)

View File

@ -15,31 +15,31 @@ It is not required to clone or download the git repository in order to run the w
For more information on running EPI2ME Labs workflows [visit out website](https://labs.epi2me.io/wfindex).
**Workflow options**
### Workflow options
To obtain the workflow, having installed `nextflow`, users can run:
```
nextflow run epi2me-labs/wf-isoforms --help
nextflow run epi2me-labs/wf-transcriptomes --help
```
to see the options for the workflow.
**Download demonstration data**
A small test dataset is provided for the purposes of testing the workflow software. It consists of reads, reference,
and annotations from human chromosome 20 only.
It can be downloaded using:
```shell
wget -O test_data.tar.gz https://ont-exd-int-s3-euwst1-epi2me-labs.s3.amazonaws.com/wf-isoforms/wf-isoforms_test_data.tar.gz
tar -xzvf test_data.tar.gz
```
**Example execution of a workflow for reference-based transcript assembly**
This example uses a synthetic SIRV dataset, so we need to tell minimap2 about the non-canonical splice junctions with
--minimap2_opts '-uf --splice-flank=no'
**Example execution of a workflow for reference-based transcript assembly and fusion detection**
```
OUTPUT=~/output;
nextflow run wf-isoforms/ --fastq test_data/fastq --ref_genome test_data/SIRV_150601a.fasta --ref_annotation test_data/SIRV_isofroms.gtf
--minimap2_opts '-uf --splice-flank=no' --out_dir outdir -w workspace_dir -profile conda -resume
```
```
# To evaluate the workflow on a larger Drosophila dataset
./evaluation/run_evaluation_dmel.sh outdir
nexflow run epi2me-labs/wf-transcriptomes --fastq ERR6053095_chr20.fastq --ref_genome chr20/hg38_chr20.fa --ref_annotation chr20/gencode.v22.annotation.chr20.gtf \
--jaffal_refBase chr20/ --jaffal_genome hg38_chr20 --jaffal_annotation genCode22" --out_dir outdir -w workspace_dir -profile conda -resume
```
**Example workflow for denovo transcript assembly**
@ -52,7 +52,7 @@ A full list of options can be seen in nextflow_schema.json. Below are some commo
- Threshold for including isoforms into interactive table `transcript_table_cov_thresh = 50`
- Run the denovo pipeline `denovo = true` (default false)
- To run the workflow with direct RNA reads `--direct_rna` (skips the pychopper step).
- To run the workflow with direct RNA reads `--direct_rna` (this just skips the pychopper step).
Pychopper and minimap2 can take options via `minimap2_opts` and `pychopper_opts`, for example:
@ -63,9 +63,71 @@ Pychopper and minimap2 can take options via `minimap2_opts` and `pychopper_opts`
- pychopper needs to know which cDNA synthesis kit used
- SQK-PCS109: use `pychopper_opts = '-k PCS109'` (default)
- SQK-PCS110: use `pychopper_opts = '-k PCS110'`
- SQK-PCS11: use `pychopper_opts = '-k PCS111'`
- pychopper can use one of two available backends for identifying primers in the raw reads
- nhmmscan `pychopper opts = '-m phmm'`
- edlib `pychopper opts = '-m edlib'`
__Note__: edlib is set by default in the config as it's quite a lot faster. However it may be less sensitive than nhmmscan.
__Note__: edlib is set by default in the config as it's quite a lot faster. However, it may be less sensitive than nhmmscan.
### Fusion detection
JAFFAL from the [JAFFA](https://github.com/Oshlack/JAFFA)
package is used to identify potential fusion transcripts. To get this this working, there are a couple of things that need doing first.
**Install JAFFA**
to install JAFFA and it's dependencies run the folllowing:
```shell
cd wf-transcriptomes/
./subworkflows/JAFFAL/install_jaffa.sh
```
**Prepare JAFFAL reference data**
To use pre-processed reference files for the hg38 genome and GENCODE v22 annotation (as used in the JFFAAL paper),
do:
```shell
mkdir jaffal_data_dir
cd jaffal_data_dir/
wf-transcriptomes/download_jaffal_references.sh
````
To use alternative genome and annotation files, they should be prepared as described
[here](https://github.com/Oshlack/JAFFA/wiki/FAQandTroubleshooting#how-can-i-generate-the-reference-files-for-a-non-supported-genome)
**Specifying the location of the JAFFA code and reference directories**
`--jaffal_dir`
This is the directory made by running install_jaffa.sh as shown above
`--jaffal_refBase`
The directory containing the reference data prepared for use with JAFFAL
**JAFFAL annotation and genome files**
The prepared JAFFAL reference files will look something like `hg38_chr20_genCode22.fa`. To enable JAFFAL to find these
files `--jaffal_genome` should be set to `hg38_chr20` and `--jaffal_annotation` to `genCode22`
__JAFFAL Notes__:
g++ must be installed. JAFFAL is not currently working on Mac M1 (osx-arm64 architecture). If there are no fusion transcripts
detected, the workflow will terminate with an error at the JAFFAL stage. If this happens,
skip the JAFFAL stage by omitting ` --jaffal_refBase`
## Workflow outputs
* an HTML report document detailing the primary findings of the workflow.
* for each sample:
* [gffcomapre](https://ccb.jhu.edu/software/stringtie/gffcompare.shtml) output directories
* read_aln_stats.tsv - alignment summary statistics
* transcriptome.fas - the assembled transcriptome
* merged_transcritptome.fas - annotated, assembled transcriptome
* [jaffal](https://github.com/Oshlack/JAFFA) ooutput directories
### Fusion detection outputs
in `${out_dir}/jaffal_output_${sample_id}` you will find:
* jaffa_results.csv - the csv results summary file
* jaffa_results.fasta - fusion transcritpt sequences

View File

@ -0,0 +1,5 @@
#!/bin/sh
#Download the data. We should we move the data out of Figshare?
wget -O JAFFA_REFERENCE_FILES_HG38_GENCODE22.V2.tar.gz https://figshare.com/ndownloader/files/25410494
tar -zxvf JAFFA_REFERENCE_FILES_HG38_GENCODE22.V2.tar.gz

View File

@ -1,8 +1,8 @@
name: epi2melabs-wf-isoforms
name: epi2melabs-wf-transcriptomes
channels:
- epi2melabs
- bioconda
- conda-forge
- bioconda
- defaults
dependencies:
- python==3.8.*
@ -27,4 +27,8 @@ dependencies:
- parallel
- scikit-learn==1.0.2
- natsort
- graphviz
# Fusion detection dependencies
# - bpipe=0.9.9.2
- java-jdk
- r-base
- gxx

View File

@ -1,59 +0,0 @@
#!/usr/bin/env bash
# Usage: ./run_evaluation_dmel.sh pathto/outputdir
# See the isONcorrect paper https://www.nature.com/articles/s41467-020-20340-8 where this dataset is described
if [[ "$#" -lt 1 ]]; then
echo "usage: run_evaluation_dmel.sh <outdir> [nextflow.config]"
exit 1
fi
if [[ "$#" -eq 1 ]]; then
config=''
fi
if [[ "$#" -eq 2 ]]; then
config="-c $2";
fi
OUTDIR=$1;
FASTQ_URL="http://ftp.sra.ebi.ac.uk/vol1/fastq/ERR358/005/ERR3588905/ERR3588905_1.fastq.gz"
REF_URL="http://ftp.ensembl.org/pub/release-99/fasta/drosophila_melanogaster/dna/Drosophila_melanogaster.BDGP6.28.dna.toplevel.fa.gz"
GFF_URL="http://ftp.ensembl.org/pub/release-99/gff3/drosophila_melanogaster/Drosophila_melanogaster.BDGP6.28.99.gff3.gz"
DATA_DIR="$OUTDIR/data"
READS_DIR="$DATA_DIR/reads"
FASTQ="$READS_DIR/ERR3588905_1.fastq.gz"
REF="$DATA_DIR/Drosophila_melanogaster.BDGP6.28.dna.toplevel.fa"
GFF="$DATA_DIR/Drosophila_melanogaster.BDGP6.28.99.gff3"
mkdir -p $READS_DIR
if [ ! -f $REF ];
then (echo "downloading reference genome"; cd $DATA_DIR; curl -L -C - -O $REF_URL); gzip -d ${REF}.gz
fi
if [ ! -f $GFF ];
then
(echo "downloading reference annotation"; cd $DATA_DIR; curl -L -C - -O $GFF_URL); gzip -d ${GFF}.gz
fi
if [ ! -f $FASTQ ];
then (echo "downloading reads"; cd $READS_DIR; curl -L -C - -O $FASTQ_URL); gzip -d ${FASTQ}.gz
fi
OUT_REF="$OUTDIR/ref"
OUT_DENOVO="$OUTDIR/denovo"
nextflow run ../ --fastq $READS_DIR $config \
--ref_genome $REF --ref_annotation $GFF -profile local --out_dir $OUT_REF --minimap2_opts '-uf --splice-flank=no' \
-w $OUT_REF/workspace -resume;
echo "Doing de novo evaluation"
nextflow run ../ --fastq $READS_DIR $config --denovo -profile local --out_dir $OUT_DENOVO \
-w $OUT_DENOVO/workspace -resume;

62
main.nf
View File

@ -12,8 +12,9 @@ nextflow.enable.dsl = 2
include { fastq_ingress } from './lib/fastqingress'
include { start_ping; end_ping } from './lib/ping'
include { reference_assembly } from './reference_assembly'
include { denovo_assembly } from './denovo_assembly'
include { reference_assembly } from './subworkflows/reference_assembly'
include { denovo_assembly } from './subworkflows/denovo_assembly'
include { gene_fusions } from './subworkflows/JAFFAL/gene_fusions'
process summariseConcatReads {
@ -83,11 +84,11 @@ process preprocess_reads {
input:
tuple val(sample_id), path(input_reads)
output:
tuple val(sample_id), path("${sample_id}_full_length_reads.fq"), emit: full_len_reads
tuple val(sample_id), path("${sample_id}_full_length_reads.fastq"), emit: full_len_reads
path '*.tsv', emit: report
script:
"""
pychopper -t ${params.threads} ${params.pychopper_opts} ${input_reads} ${sample_id}_full_length_reads.fq
pychopper -t ${params.threads} ${params.pychopper_opts} ${input_reads} ${sample_id}_full_length_reads.fastq
mv pychopper.tsv ${sample_id}_pychopper.tsv
generate_pychopper_stats.py --data ${sample_id}_pychopper.tsv --output .
@ -285,13 +286,14 @@ process makeReport {
path(seq_summaries),
path(aln_stats),
path(gffcmp_dir),
path(gff_annotation)
path(gff_annotation),
path(jaffal_csv)
output:
path("wf-isoforms-*.html"), emit: report
path("wf-transcriptomes-*.html"), emit: report
script:
// Convert the sample_id arrayList.
sids = new BlankSeparatedList(sample_ids)
def report_name = "wf-isoforms-report.html"
def report_name = "wf-transcriptomes-report.html"
def OPT_ALN = denovo ? '' : "--alignment_stats ${aln_stats}"
def OPT_DENOVO = denovo ? "--denovo" : ''
def OPT_PC_REPORT = pychopper_report.name.startsWith('OPTIONAL_FILE') ? '' : "--pychop_report ${pychopper_report}"
@ -306,6 +308,7 @@ process makeReport {
--gffcompare_dir $gffcmp_dir \
--gff_annotation $gff_annotation \
--transcript_table_cov_thresh $params.transcript_table_cov_thresh \
--jaffal_csv $jaffal_csv
$OPT_DENOVO
"""
}
@ -332,6 +335,9 @@ workflow pipeline {
reads
ref_genome
ref_annotation
jaffal_refBase
jaffal_genome
jaffal_annotation
main:
map_sample_ids_cls = {it ->
/* Harmonize tuples
@ -400,6 +406,10 @@ workflow pipeline {
seq_for_transcriptome_build = sample_ids.flatten().combine(Channel.fromPath(params.ref_genome))
}
if (jaffal_refBase){
gene_fusions(full_len_reads, jaffal_refBase, jaffal_genome, jaffal_annotation)
}
makeReport(
software_versions,
workflow_params,
@ -409,6 +419,7 @@ workflow pipeline {
.join(m.stats)
.join(run_gffcompare.out.gffcmp_dir)
.join(merge_gff_bundles.out.gff)
.join(gene_fusions.out.results_csv)
.toList().transpose().toList())
report = makeReport.out.report
@ -452,6 +463,11 @@ workflow pipeline {
.map {it -> it[1]}
.concat(makeReport.out.report)
}
if (params.jaffal_refBase){
results = results
.concat(gene_fusions.out.results
.map {it -> it[1]})
}
emit:
results
@ -466,42 +482,49 @@ workflow {
fastq = file(params.fastq, type: "file")
error = null
if (!fastq.exists()) {
println("--fastq: File doesn't exist, check path.")
exit 1
error = "--fastq: File doesn't exist, check path."
}
if (!params.denovo && !params.ref_genome){
println("--ref_genome must be supplied unless doing de novo assembly (--denovo)")
exit 1
error = "--ref_genome must be supplied unless doing de novo assembly (--denovo)"
}
if (params.ref_genome){
ref_genome = file(params.ref_genome, type: "file")
if (!ref_genome.exists()) {
println("--ref_genome: File doesn't exist, check path.")
exit 1
error = "--ref_genome: File doesn't exist, check path."
}
}else {
ref_genome = file("$projectDir/data/OPTIONAL_FILE")
}
if (params.denovo && params.ref_annotation) {
println("Reference annotation with de denovo assembly is not supported")
exit 1
error = "Reference annotation with de denovo assembly is not supported"
}
if (params.ref_annotation){
ref_annotation = file(params.ref_annotation, type: "file")
if (!ref_annotation.exists()) {
println("--annotation: File doesn't exist, check path.")
exit 1
error = "--annotation: File doesn't exist, check path."
}
}else{
ref_annotation = file("$projectDir/data/OPTIONAL_FILE")
}
if (params.jaffal_refBase){
jaffal_refBase = file(params.jaffal_refBase, type: "dir")
if (!jaffal_refBase.exists()) {
error = "--jaffa_refBase: Directory doesn't exist, check path."
}
}else{
jaffal_refBase = null
}
if (error){
println(error)
}else{
reads = fastq_ingress([
"input":params.fastq,
"sample":params.sample,
@ -509,9 +532,10 @@ workflow {
"sanitize": params.sanitize_fastq,
"output":params.out_dir])
pipeline(reads, ref_genome, ref_annotation)
pipeline(reads, ref_genome, ref_annotation, jaffal_refBase, params.jaffal_genome, params.jaffal_annotation)
output(pipeline.out.results)
end_ping(pipeline.out.telemetry)
}
}

View File

@ -119,21 +119,30 @@ params {
// Minimum probability for i consecutive minimizers to be different between read and representative:
min_prob_no_hits = 0.1
////// Fusion detection parameters
jaffal_refBase = null
jaffal_genome = "hg38"
jaffal_annotation = "genCode22"
// The default location of the JAFFA src directory when running in EPI2ME-Labs environment
// This needs overriding if running elsewhere
jaffal_dir = "/home/epi2melabs/JAFFA"
wf {
example_cmd = [
"--fastq test_data/fastq",
"--ref_genome test_data/SIRV_150601a.fasta",
"--ref_annotation test_data/SIRV_isofroms.gtf"
"--ref_annotation test_data/SIRV_isofroms.gtf",
"--jaffal_refBase chr20/",
"--jaffal_genome hg38",
"--jaffal_annotation genCode22"
]
}
}
manifest {
name = 'epi2me-labs/wf-isoforms'
name = 'epi2me-labs/wf-transcriptomes'
author = 'Oxford Nanopore Technologies'
homePage = 'https://github.com/epi2me-labs/wf-isoforms'
homePage = 'https://github.com/epi2me-labs/wf-transcriptomes'
description = 'RNA/cDNA isoform analysis workflow'
mainScript = 'main.nf'
nextflowVersion = '>=20.10.0'
@ -151,7 +160,7 @@ executor {
// other profiles may override.
process {
withLabel:isoforms {
container = "ontresearch/wf-isoforms:${params.wfversion}"
container = "ontresearch/wf-transcriptomes:${params.wfversion}"
}
shell = ['/bin/bash', '-euo', 'pipefail']
}
@ -201,7 +210,7 @@ profiles {
queue = "${params.aws_queue}"
memory = '8G'
withLabel:isoforms {
container = "${params.aws_image_prefix}-wf-isoforms:${params.wfversion}"
container = "${params.aws_image_prefix}-wf-transcriptomes:${params.wfversion}"
}
shell = ['/bin/bash', '-euo', 'pipefail']
}

View File

@ -1,9 +1,9 @@
{
"$schema": "http://json-schema.org/draft-07/schema",
"$id": "https://raw.githubusercontent.com/./master/nextflow_schema.json",
"title": "epi2me-labs/wf-isoforms",
"title": "epi2me-labs/wf-transcriptomes",
"description": "Isoform detection and characterisation.",
"url": "https://github.com/epi2me-labs/wf-isoforms",
"url": "https://github.com/epi2me-labs/wf-transcriptomes",
"type": "object",
"definitions": {
"basic_input_output_options": {
@ -14,13 +14,13 @@
"properties": {
"out_dir": {
"type": "string",
"default": "output",
"format": "directory-path",
"default": "output",
"description": "Directory for output of all user-facing files."
},
"fastq": {
"type": "string",
"format": "path",
"format": "file-path",
"demo_data": "${projectDir}/test_data/fastq",
"description": "A fastq file or directory containing fastq input files or directories of input files.",
"help_text": "If directories named \\\"barcode*\\\" are found under the `--fastq` directory the data is assumed to be multiplex and each barcode directory will be processed independently. If `.fastq(.gz)` files are found under the `--fastq` directory the sample is assumed to not be multiplexed. In this second case `--samples` should be a simple name rather than a CSV file."
@ -99,7 +99,7 @@
"reference_wf_options": {
"title": "Options for reference-based workflow",
"type": "object",
"description": "Parameters that are used solely for the referenc-guided workflow",
"description": "Parameters that are used solely for the reference-guided workflow",
"properties": {
"plot_gffcmp_stats": {
"type": "boolean",
@ -219,6 +219,34 @@
}
}
},
"fusion_detection_options": {
"title": "Gene fusion detection options",
"type": "object",
"description": "Parameters for gene fusion detection",
"properties": {
"jaffal_refBase": {
"type": "string",
"format": "path",
"description": "JAFFAl reference genome directory"
},
"jaffal_genome": {
"type": "string",
"description": "Genome reference prefix. e.g. hg38",
"default": "hg38"
},
"jaffal_annotation": {
"type": "string",
"description": "Annotation prefix",
"default": "genCode22"
},
"jaffal_dir": {
"type": "string",
"format": "path",
"description": "Path to JAFFAL git code directory. Defaults is epi2me-labs container location",
"default": "/home/epi2melabs/JAFFA"
}
}
},
"meta_data": {
"title": "Meta Data",
"type": "object",
@ -266,6 +294,9 @@
{
"$ref": "#/definitions/denovo_wf_options"
},
{
"$ref": "#/definitions/fusion_detection_options"
},
{
"$ref": "#/definitions/meta_data"
},
@ -299,7 +330,7 @@
}
},
"docs": {
"intro": "## Introduction\n\nThis workflow identifies RNA isoforms using either cDNA or direct RNA (dRNA) \nOxford Nanopore reads.\n\n### Preprocesing\ncDNA reads are initially preprocessed by [pychopper](https://github.com/epi2me-labs/pychopper) \nfor the identification of full-length reads, as well as trimming and orientation correction (This step is omitted for \n direct RNA reads).\n\n### Reference-aided approach\n* Full length reads are mapped to a supplied reference genome using [minimap2](https://github.com/lh3/minimap2)\n* Transcripts are assembled by [stringtie](http://ccb.jhu.edu/software/stringtie) \nin long read mode (with or without a guide reference annotation) to generate the GFF annotation.\n* The annotation generated by the pipeline is compared to the reference annotation. \nusing [gffcompare](http://ccb.jhu.edu/software/stringtie/gffcompare.shtml)\n\n### de novo-based approach (experimental!)\n* Sequence clusters are generated using [isONclust2](https://github.com/nanoporetech/isONclust2)\n * If a reference genome is supplied, cluster quality metrics are determined by comparing \n with clusters generated from a minimap2 alignment.\n* A consensus sequence for each cluster is generated using [spoa](https://github.com/rvaser/spoa)\n* Three rounds of polishing using racon and minimap2 to give a final polished CDS for each gene.\n* Full-length reads are then mapped to these polished CDS.\n* Transcripts are assembled by stringtie as for the reference-based approach.\n* __Note__: This approach is currently not supported with direct RNA reads.\n\n### Workflow inputs\n- Directory containing cDNA/direct RNA reads. Or a directory containing subdirectories each with reads from different samples\n (in fastq/fastq.gz format)\n- Reference genome in fasta format (required for reference-based assembly).\n- Optional reference annotation in GFF2/3 format.",
"intro": "## Introduction\n\nThis workflow identifies RNA isoforms using either cDNA or direct RNA (dRNA) \nOxford Nanopore reads.\n\n### Preprocesing\ncDNA reads are initially preprocessed by [pychopper](https://github.com/epi2me-labs/pychopper) \nfor the identification of full-length reads, as well as trimming and orientation correction (This step is omitted for \n direct RNA reads).\n\n\n### Transcript assembly\n\n#### Reference-aided transcript assembly approach\n* Full length reads are mapped to a supplied reference genome using [minimap2](https://github.com/lh3/minimap2)\n* Transcripts are assembled by [stringtie](http://ccb.jhu.edu/software/stringtie) \nin long read mode (with or without a guide reference annotation) to generate the GFF annotation.\n* The annotation generated by the pipeline is compared to the reference annotation. \nusing [gffcompare](http://ccb.jhu.edu/software/stringtie/gffcompare.shtml)\n\n#### de novo-based transcript assembly (experimental!)\n* Sequence clusters are generated using [isONclust2](https://github.com/nanoporetech/isONclust2)\n * If a reference genome is supplied, cluster quality metrics are determined by comparing \n with clusters generated from a minimap2 alignment.\n* A consensus sequence for each cluster is generated using [spoa](https://github.com/rvaser/spoa)\n* Three rounds of polishing using racon and minimap2 to give a final polished CDS for each gene.\n* Full-length reads are then mapped to these polished CDS.\n* Transcripts are assembled by stringtie as for the reference-based approach.\n* __Note__: This approach is currently not supported with direct RNA reads.\n\n### Fusion gene detection\nFusion gene detection is performed using [JAFFA](https://github.com/Oshlack/JAFFA), with the JAFFAL extension for use \nwith ONT long reads. \n\n### Workflow inputs\n- Directory containing cDNA/direct RNA reads. Or a directory containing subdirectories each with reads from different samples\n (in fastq/fastq.gz format)\n- Reference genome in fasta format (required for reference-based assembly).\n- Optional reference annotation in GFF2/3 format.\n- For fusion detection, JAFFAL reference files (see Quickstart) \n",
"links": "## Useful links\n\n* [nextflow](https://www.nextflow.io/)\n* [docker](https://www.docker.com/products/docker-desktop)\n* [Singularity](https://sylabs.io/singularity/)\n* [conda](https://docs.conda.io/en/latest/miniconda.html)\n* [racon](https://github.com/isovic/racon)\n* [spoa](https://github.com/rvaser/spoa)\n* [inONclust](https://github.com/ksahlin/isONclust)\n* [isONclust2](https://github.com/nanoporetech/isONclust2)"
}
}

View File

@ -0,0 +1,2 @@
#!/bin/sh

View File

@ -0,0 +1,40 @@
process jaffal{
label "isoforms"
input:
tuple val(sample_id), path(fastq)
path refBase
val genome
val annotation
output:
tuple val(sample_id), path("jaffal_output_$sample_id"), emit: results
tuple val(sample_id), path("jaffal_output_$sample_id/*jaffa_results.csv"), emit: results_csv
script:
"""
JAFFAOUT=jaffal_output_$sample_id
$params.jaffal_dir/tools/bin/bpipe run \
-n $params.threads \
-p jaffa_output="\$JAFFAOUT/" \
-p refBase=$refBase \
-p genome=$genome \
-p annotation=$annotation \
-p fastqInputFormat="*.fastq" \
$params.jaffal_dir/JAFFAL.groovy \
$fastq
mv "\$JAFFAOUT/jaffa_results.csv" "\$JAFFAOUT/${sample_id}_jaffa_results.csv"
"""
}
// workflow module
workflow gene_fusions {
take:
fastq
refBase
genome
annotation
main:
jaffal(fastq, refBase, genome, annotation)
emit:
results_csv = jaffal.out.results_csv
results = jaffal.out.results
}

View File

@ -0,0 +1,21 @@
#!/bin/bash
set -e
git clone https://github.com/Oshlack/JAFFA.git &&
cd JAFFA
git checkout 24b1c3b
cp ../subworkflows/JAFFAL/install_linux64.sh .
./install_linux64.sh
# JAFFA uses tools.groovy to locate binaries. We modify it as we only need a small subset or they are already
# included in our env
# Tools to be compiled from src/
bin=$(realpath tools/bin)
echo "//Tools built locally" >> tools.groovy
declare -a tools=("reformat" "extract_seq_from_fasta" "make_simple_read_table" "process_transcriptome_align_table" "make_3_gene_fusion_table")
for b in "${tools[@]}"; do
echo "$b=\"$bin/$b\"" >> tools.groovy
done
echo "minimap2=\"minimap2\"" >> tools.groovy

View File

@ -0,0 +1,111 @@
#!/bin/bash
# 21/06/22: This script has been modified to install only those applications needed for epi2melabs/wf-transcriptomes
## This script will install the tools required for the JAFFA pipeline.
## It will fetched each tool from the web and placed into the tools/ subdirectory.
## Paths to all installed tools can be found in the file tools.groovy at the
## end of execution of this script. These paths can be changed if a different
## version of software is required. Note that R must be installed manually
##
## Last Modified: Sep. 2021 by Nadia Davidson
mkdir -p tools/bin
cd tools
#a list of which programs need to be installed
commands="bpipe reformat extract_seq_from_fasta make_simple_read_table process_transcriptome_align_table make_3_gene_fusion_table dedupe"
#installation methods
function bpipe_install {
wget -O bpipe-0.9.9.2.tar.gz https://github.com/ssadedin/bpipe/releases/download/0.9.9.2/bpipe-0.9.9.2.tar.gz
tar -zxvf bpipe-0.9.9.2.tar.gz ; rm bpipe-0.9.9.2.tar.gz
ln -s $PWD/bpipe-0.9.9.2/bin/* $PWD/bin/
}
function make_3_gene_fusion_table_install {
g++ -std=c++11 -O3 -o bin/make_3_gene_fusion_table ../src/make_3_gene_fusion_table.c++
}
function extract_seq_from_fasta_install {
g++ -std=c++11 -O3 -o bin/extract_seq_from_fasta ../src/extract_seq_from_fasta.c++
}
function make_simple_read_table_install {
g++ -std=c++11 -O3 -o bin/make_simple_read_table ../src/make_simple_read_table.c++
}
function process_transcriptome_align_table_install {
g++ -std=c++11 -O3 -o bin/process_transcriptome_align_table ../src/process_transcriptome_align_table.c++
}
function make_count_table_install {
g++ -O3 -o bin/make_count_table ../src/make_count_table.c++
}
function dedupe_install {
wget --no-check-certificate https://sourceforge.net/projects/bbmap/files/BBMap_36.59.tar.gz
tar -zxvf BBMap_36.59.tar.gz
rm BBMap_36.59.tar.gz
for script in `ls $PWD/bbmap/*.sh` ; do
s=`basename $script`
s_pre=`echo $s | sed 's/.sh//g'`
echo "$PWD/bbmap/$s \$@" > $PWD/bin/$s_pre
chmod +x $PWD/bin/$s_pre
done
}
#function bypass_genomic_alignment_install {
# g++ -std=c++11 -O3 -o bin/bypass_genomic_alignment ../src/bypass_genomic_alignment.c++
#}
#Check if the version of gcc is >= 4.9
gcc_version=`gcc -dumpversion`
gcc_check=`echo -e "$gcc_version\n4.9" | sort -n | tail -n1`
if [[ $gcc_chek = "4.9" ]]
then
echo "Your version of gcc is $gcc_version."
echo "gcc must be >= 4.9 to install JAFFA. Exiting..."
exit 1
fi
echo "gcc check passed"
echo "// Path to tools used by the JAFFA pipeline" > ../tools.groovy
for c in $commands ; do
c_path=`which $PWD/bin/$c 2>/dev/null`
if [ -z $c_path ] ; then
echo "$c not found, fetching it"
${c}_install
c_path=`which $PWD/bin/$c 2>/dev/null`
fi
echo "$c=\"$c_path\"" >> ../tools.groovy
done
#finally check that R is install
R_path=`which R 2>/dev/null`
if [ -z $R_path ] ; then
echo "R not found!"
echo "Please go to http://www.r-project.org/ and follow the installation instructions."
echo "Note that the IRanges R package must be installed."
fi
echo "R=\"$R_path\"" >> ../tools.groovy
#loop through commands to check they are all installed
echo "Checking that all required tools were installed:"
Final_message="All commands installed successfully!"
for c in $commands ; do
c_path=`which $PWD/bin/$c 2>/dev/null`
if [ -z $c_path ] ; then
echo -n "WARNING: $c could not be found!!!! "
echo "You will need to download and install $c manually, then add its path to tools.groovy"
Final_message="WARNING: One or more command did not install successfully. See warning messages above. \
You will need to correct this before running JAFFA."
else
echo "$c looks like it has been installed"
fi
done
echo "**********************************************************"
echo $Final_message

View File

@ -0,0 +1,10 @@
// Path to tools used by the JAFFA pipeline
// Conda-installable tools
bpipe="bpipe"
//trimmomatic="trimmomatic"
R="/usr/bin/R"
minimap2="minimap2"
dedupe="dedupe"