Merge branch '649-b' into 'dev'

CW-649

See merge request epi2melabs/workflows/wf-transcriptomes!72
This commit is contained in:
Neil Horner 2022-08-23 19:15:47 +00:00
commit 835385e6af
18 changed files with 1122 additions and 54 deletions

View File

@ -4,12 +4,52 @@ include:
file: "wf-containers.yaml" file: "wf-containers.yaml"
variables: variables:
# Workflow inputs given to nextflow. NF_BEFORE_SCRIPT: wget -O test_data.tar.gz https://ont-exd-int-s3-euwst1-epi2me-labs.s3.amazonaws.com/wf-isoforms/wf-isoforms_test_data.tar.gz && tar -xzvf test_data.tar.gz
# The workflow should define `--out_dir`, the CI template sets this. NF_WORKFLOW_OPTS: "--fastq ERR6053095_chr20.fastq \
# Only common file inputs and option values need to be given here --ref_genome chr20/hg38_chr20.fa --ref_annotation chr20/gencode.v22.annotation.chr20.gtf \
# (not things such as -profile) --jaffal_refBase chr20/ --jaffal_genome hg38_chr20 --jaffal_annotation genCode22"
NF_BEFORE_SCRIPT: | NF_IGNORE_PROCESSES: preprocess_reads,merge_transcriptomes
wget -O test_data.tar.gz https://ont-exd-int-s3-euwst1-epi2me-labs.s3.amazonaws.com/wf-isoforms/wf-isoforms_test_data.tar.gz && tar -xzvf test_data.tar.gz
NF_WORKFLOW_OPTS: "--fastq ERR6053095_chr20.fastq \ docker-run:
--ref_genome chr20/hg38_chr20.fa --ref_annotation chr20/gencode.v22.annotation.chr20.gtf \
--jaffal_refBase chr20/ --jaffal_genome hg38_chr20 --jaffal_annotation genCode22" # Remove this directive in downstream templates
tags: [large_ram] # no need for big ram
# Define a 1D job matrix to inject a variable named MATRIX_NAME into
# the CI environment, we can use the value of MATRIX_NAME to determine
# which options to apply as part of the rules block below
# NOTE There is a slightly cleaner way to define this matrix to include
# the variables, but it is broken when using long strings! See CW-756
parallel:
matrix:
- MATRIX_NAME: [
"fusions", "differential_expression", "isoforms"
]
rules:
# NOTE As we're overriding the rules block for the included docker-run
# we must redefine this CI_COMMIT_BRANCH rule to prevent docker-run
# being incorrectly scheduled for "detached merge request pipelines" etc.
- if: ($CI_COMMIT_BRANCH == null || $CI_COMMIT_BRANCH == "dev-template")
when: never
- if: $MATRIX_NAME == "isoforms"
variables:
NF_BEFORE_SCRIPT: wget -O test_data.tar.gz https://ont-exd-int-s3-euwst1-epi2me-labs.s3.amazonaws.com/wf-isoforms/wf-isoforms_test_data.tar.gz && tar -xzvf test_data.tar.gz
NF_WORKFLOW_OPTS: "--fastq ERR6053095_chr20.fastq \
--ref_genome chr20/hg38_chr20.fa --ref_annotation chr20/gencode.v22.annotation.chr20.gtf"
NF_IGNORE_PROCESSES: preprocess_reads,merge_transcriptomes
- if: $MATRIX_NAME == "fusions"
variables:
NF_BEFORE_SCRIPT: wget -O test_data.tar.gz https://ont-exd-int-s3-euwst1-epi2me-labs.s3.amazonaws.com/wf-isoforms/wf-isoforms_test_data.tar.gz && tar -xzvf test_data.tar.gz
NF_WORKFLOW_OPTS: "--fastq ERR6053095_chr20.fastq \
--ref_genome chr20/hg38_chr20.fa --ref_annotation chr20/gencode.v22.annotation.chr20.gtf \
--jaffal_refBase chr20/ --jaffal_genome hg38_chr20 --jaffal_annotation genCode22"
NF_IGNORE_PROCESSES: preprocess_reads,merge_transcriptomes
- if: $MATRIX_NAME == "differential_expression"
variables:
NF_BEFORE_SCRIPT: wget -O differential_expression.tar.gz https://ont-exd-int-s3-euwst1-epi2me-labs.s3.amazonaws.com/wf-isoforms/wf-isoforms_differential_expression.tar.gz && tar -xzvf differential_expression.tar.gz
NF_WORKFLOW_OPTS: "--fastq differential_expression_dataset/fastq \
--de_analysis \
--ref_genome differential_expression_dataset/hg38_chr20.fa \
--ref_annotation differential_expression_dataset/gencode.v22.annotation.chr20.gtf \
--direct_rna"
NF_IGNORE_PROCESSES: preprocess_reads,merge_transcriptomes

View File

@ -5,8 +5,8 @@ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.0.0/),
and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html). and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html).
## [unreleased] ## [unreleased]
### Changed ### Added
- Differential transcript and gene expression subworkflow
## [v0.1.4] ## [v0.1.4]
### Added ### Added

View File

@ -1,10 +1,10 @@
ARG BASEIMAGE=ontresearch/base-workflow-image:v0.1.1 ARG BASEIMAGE=ontresearch/base-workflow-image:v0.2.0
FROM $BASEIMAGE FROM $BASEIMAGE
ARG ENVFILE=environment.yaml ARG ENVFILE=environment.yaml
COPY $ENVFILE $HOME/environment.yaml COPY $ENVFILE $HOME/environment.yaml
RUN \ RUN \
. $CONDA_DIR/etc/profile.d/mamba.sh \ . $CONDA_DIR/etc/profile.d/micromamba.sh \
&& micromamba activate \ && micromamba activate \
&& micromamba install -n base --file $HOME/environment.yaml \ && micromamba install -n base --file $HOME/environment.yaml \
&& micromamba clean --all --yes \ && micromamba clean --all --yes \
@ -15,7 +15,6 @@ RUN \
&& rm -rf $CONDA_DIR/lib/python3.*/site-packages/pip \ && rm -rf $CONDA_DIR/lib/python3.*/site-packages/pip \
&& find $CONDA_DIR -name '__pycache__' -type d -exec rm -rf '{}' '+' && find $CONDA_DIR -name '__pycache__' -type d -exec rm -rf '{}' '+'
USER $WF_UID USER $WF_UID
WORKDIR $HOME WORKDIR $HOME

View File

@ -40,11 +40,19 @@ using [gffcompare](http://ccb.jhu.edu/software/stringtie/gffcompare.shtml)
Fusion gene detection is performed using [JAFFA](https://github.com/Oshlack/JAFFA), with the JAFFAL extension for use Fusion gene detection is performed using [JAFFA](https://github.com/Oshlack/JAFFA), with the JAFFAL extension for use
with ONT long reads. with ONT long reads.
### Differential expression analysis
* Differential expression is done using the transcripts output by the workflow.
* A non redundant transcriptome is found using the merge function in [stringtie](http://ccb.jhu.edu/software/stringtie).
* The reads are then aligned to the transcriptome using minimap2 in a splice-aware manner.
* [salmon](https://github.com/COMBINE-lab/salmon) is used for transcript quantification.
* R packages [edgeR](https://bioconductor.org/packages/release/bioc/html/edgeR.html) and [stageR](https://bioconductor.org/packages/release/bioc/html/stageR.html) are used for differential expression analysis.
* [DEXSeq](https://bioconductor.org/packages/release/bioc/html/DEXSeq.html) is then used for differential transcript usage analysis.
### Workflow inputs ### Workflow inputs
- Directory containing cDNA/direct RNA reads. Or a directory containing subdirectories each with reads from different samples - Directory containing cDNA/direct RNA reads. Or a directory containing subdirectories each with reads from different samples
(in fastq/fastq.gz format) (in fastq/fastq.gz format)
- Reference genome in fasta format (required for reference-based assembly). - Reference genome in fasta format (required for reference-based assembly).
- Optional reference annotation in GFF2/3 format. - Optional reference annotation in GFF2/3 format (required for differential expression analysis `--de_analysis`).
- For fusion detection, JAFFAL reference files (see Quickstart) - For fusion detection, JAFFAL reference files (see Quickstart)
## Quickstart ## Quickstart
@ -164,6 +172,25 @@ g++ must be installed. JAFFAL is not currently working on Mac M1 (osx-arm64 arch
detected, the workflow will terminate with an error at the JAFFAL stage. If this happens, detected, the workflow will terminate with an error at the JAFFAL stage. If this happens,
skip the JAFFAL stage by omitting ` --jaffal_refBase` skip the JAFFAL stage by omitting ` --jaffal_refBase`
### Differential Expression
Differential Expression requires at least 2 replicates of each sample to compare. You can see an example condition_sheet.tsv in test_data.
**Example workflow for differential expression transcript assembly**
Download differential expression data set
`wget -O differential_expression.tar.gz https://ont-exd-int-s3-euwst1-epi2me-labs.s3.amazonaws.com/wf-isoforms/wf-isoforms_differential_expression.tar.gz && tar -xzvf differential_expression.tar.gz`
Run the cmd
```
OUTPUT=~/output;
nexflow run epi2me-labs/wf-transcriptomes --fastq differential_expression_dataset/fastq --de_analysis \
--ref_genome differential_expression_dataset/hg38_chr20.fa \
--ref_annotation differential_expression_dataset/gencode.v22.annotation.chr20.gtf \
--direct_rna
```
## Workflow outputs ## Workflow outputs
* an HTML report document detailing the primary findings of the workflow. * an HTML report document detailing the primary findings of the workflow.
@ -179,6 +206,9 @@ skip the JAFFAL stage by omitting ` --jaffal_refBase`
in `${out_dir}/jaffal_output_${sample_id}` you will find: in `${out_dir}/jaffal_output_${sample_id}` you will find:
* jaffa_results.csv - the csv results summary file * jaffa_results.csv - the csv results summary file
* jaffa_results.fasta - fusion transcritpt sequences * jaffa_results.fasta - fusion transcritpt sequences
### Differential Expression outputs
* dtu_plots.pdf - a pdf with differntial transcript usage plots
## Useful links ## Useful links
* [nextflow](https://www.nextflow.io/) * [nextflow](https://www.nextflow.io/)

139
bin/de_analysis.R Executable file
View File

@ -0,0 +1,139 @@
#!/usr/bin/env Rscript
suppressMessages(library("DRIMSeq"))
suppressMessages(library("GenomicFeatures"))
cat("Loading counts, conditions and parameters.\n")
cts <- as.matrix(read.csv("merged/all_counts.tsv", sep="\t", row.names="Reference", stringsAsFactors=FALSE))
# Set up sample data frame:
coldata <- read.csv("de_analysis/coldata.tsv", row.names="sample", sep=",", stringsAsFactors=TRUE)
coldata$sample_id <- rownames(coldata)
coldata$condition <- factor(coldata$condition, levels=rev(levels(coldata$condition)))
de_params <- read.csv("de_analysis/de_params.tsv", sep="\t", stringsAsFactors=FALSE)
cat("Loading annotation database.\n")
#txdb <- makeTxDbFromGFF(de_params$Annotation[[1]])
txdb <- makeTxDbFromGFF("annotation.gtf")
txdf <- select(txdb, keys(txdb,"GENEID"), "TXNAME", "GENEID")
tab <- table(txdf$GENEID)
txdf$ntx<- tab[match(txdf$GENEID, names(tab))]
strip_version<-function(x) {
tmp<-data.frame(strsplit(x,".", fixed=TRUE), stringsAsFactors=FALSE)
tmp<-as.vector(tmp[1,])
colnames(tmp) <- c()
rownames(tmp) <- c()
return(tmp)
}
#rownames(cts) <- strip_version(rownames(cts))
cts <- cts[rownames(cts) %in% txdf$TXNAME, ] # FIXME: filter for transcripts which are in the annotation. Why they are not all there?
# Reorder transcript/gene database to match input counts:
txdf <- txdf[match(rownames(cts), txdf$TXNAME), ]
rownames(txdf) <- NULL
# Create counts data frame:
counts<-data.frame(gene_id=txdf$GENEID, feature_id=txdf$TXNAME, cts)
cat("Filtering counts using DRIMSeq.\n")
print(coldata)
d <- dmDSdata(counts=counts, samples=coldata)
trs_cts_unfiltered <- counts(d)
d <- dmFilter(d, min_samps_gene_expr = de_params$min_samps_gene_expr[[1]], min_samps_feature_expr = de_params$min_samps_feature_expr[[1]],
min_gene_expr = de_params$min_gene_expr[[1]], min_feature_expr = de_params$min_feature_expr[[1]])
cat("Building model matrix.\n")
design <- model.matrix(~condition, data=DRIMSeq::samples(d))
suppressMessages(library("dplyr"))
# Sum transcript counts into gene counts:
cat("Sum transcript counts into gene counts.\n")
trs_cts <- counts(d)
write.table(trs_cts, file="merged/all_counts_filtered.tsv",sep="\t")
gene_cts <- trs_cts_unfiltered %>% dplyr::select(c(1, 3:ncol(trs_cts))) %>% group_by(gene_id) %>% summarise_all(tibble::lst(sum)) %>% data.frame()
rownames(gene_cts) <- gene_cts$gene_id
gene_cts$gene_id <- NULL
write.table(gene_cts, file="merged/all_gene_counts.tsv",sep="\t")
# Differential gene expression using edgeR:
suppressMessages(library("edgeR"))
cat("Running differential gene expression analysis using edgeR.\n")
y <- DGEList(gene_cts)
y <- calcNormFactors(y)
y <- estimateDisp(y,design)
fit <- glmQLFit(y,design)
qlf <- glmQLFTest(fit)
edger_res <- topTags(qlf, n=nrow(y), sort.by="PValue")[[1]]
pdf("de_analysis/results_dge.pdf")
plotMD(qlf)
abline(h=c(-1,1), col="blue")
plotQLDisp(fit)
write.table(as.data.frame(edger_res), file="de_analysis/results_dge.tsv", sep="\t")
# Differential transcript usage using DEXSeq:
suppressMessages(library("DEXSeq"))
cat("Running differential transcript usage analysis using DEXSeq.\n")
sample.data<-DRIMSeq::samples(d)
count.data <- round(as.matrix(counts(d)[,-c(1:2)]))
dxd <- DEXSeqDataSet(countData=count.data, sampleData=sample.data, design=~sample + exon + condition:exon, featureID=trs_cts$feature_id, groupID=trs_cts$gene_id)
dxd <- estimateSizeFactors(dxd)
dxd <- estimateDispersions(dxd)
dxd <- testForDEU(dxd, reducedModel=~sample + exon)
dxd <- estimateExonFoldChanges( dxd, fitExpToVar="condition")
dxr <- DEXSeqResults(dxd, independentFiltering=FALSE)
dev.off()
pdf("de_analysis/results_dtu.pdf")
plotMA(dxr, cex=0.8, alpha=0.05)
plotDispEsts(dxd)
qval <- perGeneQValue(dxr)
dxr.g<-data.frame(gene=names(qval), qval)
dxr.g <- dxr.g[order(dxr.g$qval),]
dxr_out <- as.data.frame(dxr[,c("featureID", "groupID", "pvalue")])
dxr_out <- dxr_out[order(dxr$pvalue),]
write.table(dxr.g, file="de_analysis/results_dtu_gene.tsv", sep="\t")
write.table(dxr_out, file="de_analysis/results_dtu_transcript.tsv", sep="\t")
# and writing out some of the DEXSeq metrics to accompany EPI2ME Labs tutorial
colnames(dxr)[grep("log2fold", colnames(dxr))] <- "log2fold"
MADTUdata <- data.frame(dxr)[order(dxr$padj),c("exonBaseMean", "log2fold", "pvalue", "padj")]
MADTUdata$exonBaseMean <- log2(MADTUdata$exonBaseMean)
colnames(MADTUdata)[which(colnames(MADTUdata)=="exonBaseMean")] <- "Log2MeanExon"
colnames(MADTUdata)[which(colnames(MADTUdata)=="log2fold")] <- "Log2FC"
write.table(MADTUdata, file="de_analysis/results_dexseq.tsv", sep="\t")
# stageR analysis of DEXSeq results:
cat("stageR analysis\n")
library(stageR)
cat("Running stageR analysis on the differential transcript usage results.\n")
pConfirmation <- matrix(dxr$pvalue, ncol=1)
dimnames(pConfirmation) <- list(dxr$featureID, "transcript")
pScreen <- qval
tx2gene <- as.data.frame(dxr[,c("featureID", "groupID")])
stageRObj <- stageRTx(pScreen=pScreen, pConfirmation=pConfirmation, pScreenAdjusted=TRUE, tx2gene=tx2gene)
# note: the choice of 0.05 here means you can *only* threshold at 5% OFDR later
stageRObj <- stageWiseAdjustment(stageRObj, method="dtu", alpha=0.10)
suppressWarnings({dex.padj <- getAdjustedPValues(stageRObj, order=FALSE, onlySignificantGenes=FALSE)})
# dex.padj <- dex.padj[,-1]
write.table(dex.padj, file="de_analysis/results_dtu_stageR.tsv", sep="\t")

367
bin/de_plots.py Executable file
View File

@ -0,0 +1,367 @@
#!/usr/bin/env python
"""Create de report section."""
from glob import glob
import os
from aplanat import hist, points
from aplanat.bars import boxplot_series
from aplanat.util import Colors
import numpy as np
import pandas as pd
def parse_seqkit(fname):
"""Get seqkit columns."""
cols = {
'Read': str, 'Ref': str, 'MapQual': int, 'Acc': float, 'ReadLen': int,
'ReadAln': int, 'ReadCov': float, 'MeanQual': float,
'IsSec': bool, 'IsSup': bool}
df = pd.read_csv(fname, sep="\t", dtype=cols, usecols=cols.keys())
df['Clipped'] = df['ReadLen'] - df['ReadAln']
df['Type'] = 'Primary'
df.loc[df['IsSec'], 'Type'] = 'Secondary'
df.loc[df['IsSup'], 'Type'] = 'Supplementary'
df["fname"] = os.path.basename(fname).rstrip(".seqkit.stats")
return df
def number_of_alignments(df, field_name):
"""Group alignments for summary table."""
grouped = df.groupby('fname').agg(**{
field_name: ('Read', 'size'),
})
return grouped.transpose()
def create_summary_table(df):
"""Create summary table."""
all = number_of_alignments(df, "Read mappings")
primary = number_of_alignments(df.loc[df['Type'] == 'Primary'], "Primary")
secondary = number_of_alignments(
df.loc[df['Type'] == 'Secondary'], "Secondary")
supplementary = number_of_alignments(
df.loc[df['Type'] == 'Supplementary'], "Supplementary")
avg_acc = df.loc[df['Type'] == 'Primary'].groupby(
'fname').agg(**{"Median Qscore": ('MeanQual', 'median'), }).transpose()
avg_mapq = df.loc[df['Type'] == 'Primary'].groupby(
'fname').agg(**{"Median MAPQ": ('MapQual', 'median'), }).transpose()
return pd.concat([
all, primary, secondary, supplementary,
avg_acc, avg_mapq])
def dtu_table(gene_id, dtu_file, alignment_stats, condition_sheet):
"""Create DTU table and plot."""
dtu_results = pd.read_csv(dtu_file, sep='\t')
f = open(condition_sheet)
df = pd.read_csv(f, sep='\t')
treated_df = df.loc[df['condition'] == "treated"]
untreated_df = df.loc[df['condition'] == "untreated"]
treated_samples = treated_df['sample'].tolist()
control_samples = untreated_df['sample'].tolist()
# parmaterise these
control_name = "condition1"
treated_name = "condition2"
table = dtu_results.loc[(dtu_results["geneID"] == gene_id)]
msg = "Gene ID \"{}\" does not exist in the dataset, please select another"
assert not table.empty, msg.format(gene_id)
alignment_stats[alignment_stats["Ref"].isin([gene_id])]
alignment_stats["Transcript"] = alignment_stats["Ref"].apply(
lambda x: x.split(".")[0])
gene_alignments = alignment_stats[alignment_stats["Transcript"].isin(
table["txID"])]
gene_alignments = gene_alignments.loc[(
gene_alignments["Type"] == "Primary")]
gene_alignments["condition"] = gene_alignments.apply(
lambda x: control_name if x["fname"] in
control_samples else treated_name, axis=1)
groups = gene_alignments.groupby(
["Transcript", "fname", "condition"]).agg(
**{"Transcript_count": ("Read", "size")})
df = None
for gene_id, group in gene_alignments.groupby("Transcript"):
temp = group.groupby("fname").agg(**{
gene_id: ("Read", "size")
}).transpose()
if df is None:
df = temp
else:
df = pd.concat([df, temp])
df = df.reset_index().rename(columns={"index": "Transcript_ID"})
table = table.rename(columns={
"txID": "Transcript_ID",
"gene": "p_gene",
"transcript": "p_transcript"
})
gene_table = pd.merge(df, table)
gene_table = gene_table.set_index(["geneID", "Transcript_ID"])
file_names = set(control_samples.keys())
file_names.update(treated_samples.keys())
gene_table = gene_table.fillna(0)
table = groups.reset_index()[
["condition", "Transcript", "Transcript_count"]]
table['transcript, condition'] = \
table['Transcript'].astype(str) + ', ' + table['condition'].astype(str)
repeats = groups.groupby(level=['Transcript', 'condition']).size()
min_rep, max_rep = min(repeats), max(repeats)
plot = boxplot_series(
table['transcript, condition'], table['Transcript_count'],
x_axis_label='Transcript, condition',
y_axis_label='Transcript count',
height=200, width=200,
title="Transcript counts (from {}-{} replicates)".format(
min_rep, max_rep))
plot.xaxis.major_label_orientation = 3.1452/2
if min_rep < 7:
for renderer in plot.renderers:
renderer.glyph.line_alpha = 0.2
try:
renderer.glyph.fill_alpha = 0.2
except Exception:
pass
plot.circle(
table['transcript, condition'], table['Transcript_count'],
fill_color='black', line_color='black')
return (table, plot)
def pool_csvs(folder):
"""Concat seqkit stats."""
files = glob(folder + "/*.seqkit.stats")
dfs = [parse_seqkit(f) for f in files]
return pd.concat(dfs)
def abundance_histogram(filtered_counts, gene_counts, section):
"""Create plot for abundance of transcripts across all samples."""
section.markdown("""
Histogram showing the abundance of transcript
counts for genes identified in the analysis.
""")
filtered_count_file = filtered_counts
gene_count_file = gene_counts
transcripts_per_gene = pd.read_csv(
filtered_count_file, sep='\t',
usecols=['gene_id', 'feature_id']).groupby(['gene_id']).agg(['count'])
transcripts_per_gene.columns = transcripts_per_gene.columns.droplevel()
gene_ids = pd.read_csv(
gene_count_file, sep='\t',
usecols=[0]).index.values.tolist()
singletons = [
gene_id for gene_id in gene_ids if gene_id not in transcripts_per_gene.index.values.tolist()] # noqa
singletons = pd.DataFrame(index=singletons, columns=['count']).fillna(1)
transcripts_per_gene = pd.concat([singletons, transcripts_per_gene])
transcript_plot = hist.histogram(
[transcripts_per_gene['count'].tolist()],
binwidth=1, colors=[Colors.cerulean])
transcript_plot.xaxis.axis_label = "Number of isoforms per gene (n)"
transcript_plot.yaxis.axis_label = "Number of occurences"
section.markdown("### Transcripts per gene")
section.plot(transcript_plot)
def dexseq_section(dexseq_file, section, id_dic):
"""Add gene isoforms table and plot."""
section.markdown("### Differential Isoform usage")
dexseq_caption = '''Table showing gene isoforms, ranked by adjusted
p-value, from the DEXSeq analysis. Information shown includes the log2 fold
change between experimental conditions, the log-scaled transcript
abundance and the false discovery corrected p-value (FDR).
This table has not been filtered
for genes that satisfy statistical or magnitudinal thresholds'''
section.markdown(dexseq_caption)
dexseq_results = pd.read_csv(dexseq_file, sep='\t')
dexseq_results.index.name = "gene_id:trancript_id"
dexseq_results.index = dexseq_results.index.map(
lambda x: id_dic[x.split(':')[0]] + ':' + str(x.split(':')[1]))
dexseq_pvals = dexseq_results.sort_values(by='pvalue', ascending=True)
section.table(dexseq_results.loc[dexseq_pvals.index], index=True)
section.markdown("""
The figure below presents the MA plot from the DEXSeq analysis.
M is the log2 ratio of isoform transcript abundance between conditions.
A is the log2 transformed mean abundance value.
Transcripts that satisfy the logFC and FDR corrected p-value
thresholds defined are shaded as 'Up-' or 'Down-' regulated.""")
pval_limit = 0.01
up = dexseq_results.loc[
(dexseq_results["Log2FC"] > 0) & (
dexseq_results['pvalue'] < pval_limit)]
down = dexseq_results.loc[
(dexseq_results["Log2FC"] <= 0) & (
dexseq_results['pvalue'] < pval_limit)]
not_sig = dexseq_results.loc[(dexseq_results["pvalue"] >= pval_limit)]
dexseq_plot = points.points(
x_datas=[
up["Log2MeanExon"],
down["Log2MeanExon"],
not_sig["Log2MeanExon"],
],
y_datas=[
up["Log2FC"],
down["Log2FC"],
not_sig["Log2FC"],
],
title="Average copy per million (CPM) vs Log-fold change (LFC)",
colors=["red", "blue", "black"],
xlim=[0, 5],
names=["Up", "Down", "NotSig"]
)
dexseq_plot.xaxis.axis_label = "A (log2 transformed mean exon read counts)"
dexseq_plot.yaxis.axis_label = """
M (log2 transformed differential abundance)
"""
dexseq_results_caption = "### Dexseq results"
section.markdown(dexseq_results_caption)
section.plot(dexseq_plot)
def dtu_section(dtu_file, section, gt_dic, ge_dic):
"""Plot dtu section."""
dtu_results = pd.read_csv(dtu_file, sep='\t')
dtu_results["gene_name"] = dtu_results["txID"].apply(lambda x: gt_dic[x])
dtu_results["geneID"] = dtu_results["geneID"].apply(lambda x: ge_dic[x])
dtu_pvals = dtu_results.sort_values(by='gene', ascending=True)
dtu_caption = '''Table showing gene and transcript identifiers
and their FDR corrected probabilities
for the genes and their isoforms that have been
identified as showing DTU using the R packages DEXSeq and StageR.
This list has been shortened requiring that both gene and transcript
must satisfy the p-value
threshold'''
section.markdown(dtu_caption)
section.table(dtu_results.loc[dtu_pvals.index])
def dge_section(dge_file, section, ids_dic):
"""Create DGE table and plot."""
section.markdown('### Differential gene expression')
dge_results = pd.read_csv(dge_file, sep='\t')
dge_pvals = dge_results.sort_values(by='FDR', ascending=True)
dge_results[['logFC', 'logCPM', 'F']] = dge_results[
['logFC', 'logCPM', 'F']].round(2)
dge_caption = """
Table showing the genes from the edgeR analysis.
Information shown includes the log2 fold change between
experimental conditions, the log-scaled counts per million measure of abundance
and the false discovery corrected p-value (FDR). This table has not been
filtered for genes that satisfy statistical or magnitudinal thresholds"""
section.markdown(dge_caption)
dge_results.index = dge_results.index.map(lambda x: ids_dic[x])
dge_pvals.index = dge_pvals.index.map(lambda x: ids_dic[x])
section.table(dge_results.loc[dge_pvals.index], index=True)
dge = pd.read_csv(dge_file, sep="\t")
section.markdown("""
This plot visualises differences in measurements between the
two experimental conditions. M is the log2 ratio of gene expression
calculated between the conditions.
A is a log2 transformed mean expression value.
The figure below presents the MA figure from this edgeR analysis.
Genes that satisfy the logFC and FDR corrected p-value thresholds
defined are shaded as 'Up-' or 'Down-' regulated.
""")
pval_limit = 0.01
up = dge.loc[(dge["logFC"] > 0) & (dge['PValue'] < pval_limit)]
down = dge.loc[(dge["logFC"] <= 0) & (dge['PValue'] < pval_limit)]
not_sig = dge.loc[(dge["PValue"] >= pval_limit)]
logcpm_vs_logfc = points.points(
x_datas=[
up["logCPM"],
down["logCPM"],
not_sig["logCPM"],
],
y_datas=[
up["logFC"],
down["logFC"],
not_sig["logFC"],
],
title="Average copy per million (CPM) vs Log-fold change (LFC)",
colors=["blue", "red", "black"],
names=["Up", "Down", "NotSig"]
)
logcpm_vs_logfc.xaxis.axis_label = "Average log CPM"
logcpm_vs_logfc.yaxis.axis_label = "Log-fold change"
logcpm_caption = """### Results of the edgeR Analysis."""
section.markdown(logcpm_caption)
section.plot(logcpm_vs_logfc)
def salmon_table(salmon_counts, section):
"""Create salmon counts summary table."""
salmon_counts = pd.read_csv(salmon_counts, sep='\t')
salmon_counts.set_index("Reference", drop=True, append=False, inplace=True)
salmon_size_top = salmon_counts.sum(axis=1).sort_values(ascending=False)
salmon_counts = salmon_counts.applymap(np.int64)
salmon_count_caption = """
Table showing the annotated Transcripts Per Million
identified by Minimap2 mapping and Salmon transcript
detection with the highest
number of mapped reads"""
section.markdown("### Transcripts Per Million ")
section.markdown(salmon_count_caption, "salmon-head-caption")
section.table(
salmon_counts.loc[salmon_size_top.index].head(n=100), index=True)
def get_translations(gtf):
"""Create dic with gene_name and gene_references."""
fn = open(gtf).readlines()
gene_txid = {}
gene_geid = {}
for i in fn:
if i.startswith("#"):
continue
try:
gene_name = i.split("gene_name")[1].split(";")[0]
except Exception:
gene_name = i.split("gene_id")[1].split(";")[0]
try:
gene_reference = i.split("ref_gene_id")[1].split(";")[0]
except Exception:
gene_reference = i.split("gene_id")[1].split(";")[0]
try:
transcript_id = i.split("transcript_id")[1].split(";")[0]
except Exception:
transcript_id = "unknown"
transcript_id = transcript_id.replace("\"", "").strip()
gene_id = i.split("gene_id")[1].split(";")[0]
gene_id = gene_id.replace("\"", "").strip()
gene_name = gene_name.replace("\"", "").strip()
gene_reference = gene_reference.replace("\"", "").strip()
gene_txid[transcript_id] = gene_name
gene_geid[gene_id] = gene_reference
return gene_txid, gene_geid
def de_section(
stringtie, dge, dexseq, dtu,
tpm, report):
"""Differential expression sections."""
section = report.add_section()
"""Add Differential expression section."""
section.markdown("# Differential expression.")
section.markdown("""
This section shows differential gene expression
and differential isoform usage. Salmon was used to
assign reads to individual annotated isoforms defined by
the GTF-format annotation.
These counts were used to perform a statistical analysis to identify
the genes and isoforms that show differences in abundance between
the experimental conditions.
""")
section.markdown("### Alignment summary stats")
alignment_stats = pool_csvs("seqkit")
alignment_summary_df = create_summary_table(alignment_stats)
alignment_summary_df = alignment_summary_df.fillna(0).applymap(np.int64)
section.table(alignment_summary_df, key='alignment-stats', index=True)
salmon_table(tpm, section)
gene_txid, gene_name = get_translations(stringtie)
dge_section(dge, section, gene_name)
dexseq_section(dexseq, section, gene_name)
dtu_section(dtu, section, gene_txid, gene_name)
# missing dtu plots at the moment as too many
section.markdown("""
### View dtu_plots.pdf file to see plots of differential isoform usage
""")

58
bin/merge_count_tsvs.py Executable file
View File

@ -0,0 +1,58 @@
#!/usr/bin/env python
"""Merge salmon output count files."""
import argparse
from functools import reduce
import numpy as np
import pandas as pd
# Parse command line arguments:
parser = argparse.ArgumentParser(
description="""Merge tab separated files on a given field using pandas.""")
parser.add_argument(
'-j', metavar='join', type=str, help="Join type (outer).", default="outer")
parser.add_argument(
'-f', metavar='field', type=str,
help="Join on this field (Reference).", default="Reference")
parser.add_argument(
'-o', metavar='out_tsv', type=str,
help="Output tsv (merge_tsvs.tsv).", default="merge_tsvs.tsv")
parser.add_argument(
'-z', action="store_true",
help="Fill NA values with zero.", default=False)
parser.add_argument(
'tsvs', metavar='input_tsvs', nargs='*',
type=str, help="Input tab separated files.")
parser.add_argument(
'-tpm', type=bool, nargs='*', default=False,
help="TPM instead of counts")
if __name__ == '__main__':
args = parser.parse_args()
dfs = {x: pd.read_csv(x, sep="\t") for x in args.tsvs}
ndfs = []
for x, df in dfs.items():
# Transform counts to integers:
df = df.rename(columns={'NumReads': 'Count', 'Name': 'Reference'})
if args.tpm:
df = df.rename(columns={'TPM': 'Count', 'Name': 'Reference'})
df.Count = np.array(df.Count, dtype=int)
# Take only non-zero counts:
df = df[df.Count > 0]
df = df[["Reference", "Count"]]
df = df.sort_values(by=["Count"], ascending=False)
name = x.split('.')[0]
df = df.rename(columns={'Count': name})
ndfs.append(df)
dfs = ndfs
df_merged = reduce(lambda left, right: pd.merge(
left, right, on=args.f, how=args.j), dfs)
if args.z:
df_merged = df_merged.fillna(0)
df_merged.to_csv(args.o, sep="\t", index=False)

57
bin/plot_dtu_results.R Executable file
View File

@ -0,0 +1,57 @@
#!/usr/bin/env Rscript
suppressMessages(library(dplyr))
suppressMessages(library(ggplot2))
suppressMessages(library(tidyr))
# Set up sample data frame:
coldata <- read.csv("de_analysis/coldata.tsv", row.names="sample", sep=",")
coldata$sample_id <- rownames(coldata)
coldata$condition <- factor(coldata$condition, levels=rev(levels(coldata$condition)))
coldata$type <-NULL
coldata$patient <-NULL
# Read stageR results:
stageR <- read.csv("de_analysis/results_dtu_stageR.tsv", sep="\t")
names(stageR) <- c("gene_id", "transcript_id", "p_gene", "p_transcript");
# Read filtered counts:
counts <- read.csv("merged/all_counts_filtered.tsv", sep="\t");
names(counts)[2]<-"transcript_id"
# Join counts and stageR results:
df <- counts %>% left_join(stageR, by = c("gene_id", "transcript_id"))
df <- df[order(df$p_gene),]
scols <- setdiff(names(df),c("gene_id", "transcript_id", "p_gene", "p_transcript"))
# Normalise counts:
for(sc in scols){
df[sc] <- df[sc] / sum(df[sc])
}
# Melt data frame:
tdf <- df %>% gather(key='sample', value='norm_count',-gene_id, -transcript_id, -p_gene, -p_transcript)
# Add sample group column:
sampleToGroup<-function(x){
return(coldata[x,]$condition)
}
tdf$group <- sampleToGroup(tdf$sample)
# Filter for significant genes:
sig_level <- 0.05
genes <- as.character(tdf[which(tdf$p_gene < sig_level),]$gene_id)
genes <- unique(genes)
pdf("de_analysis/dtu_plots.pdf")
for(gene in genes){
gdf<-tdf[which(tdf$gene_id==gene),]
p_gene <- unique(gdf$p_gene)
p <- ggplot(gdf, aes(x=transcript_id, y=norm_count)) + geom_bar(stat="identity", aes(fill=sample), position="dodge")
p <- p + facet_wrap(~ group) + coord_flip()
p <- p + ggtitle(paste(gene," : p_value=",p_gene,sep=""))
print(p)
}

View File

@ -4,6 +4,7 @@
import argparse import argparse
from collections import Counter, defaultdict, OrderedDict from collections import Counter, defaultdict, OrderedDict
import math import math
import os
from pathlib import Path from pathlib import Path
from aplanat import bars, hist from aplanat import bars, hist
@ -17,6 +18,7 @@ from bokeh.models.widgets import DataTable, TableColumn
from bokeh.palettes import Category10_10 from bokeh.palettes import Category10_10
from bokeh.plotting import figure from bokeh.plotting import figure
from bokeh.transform import dodge from bokeh.transform import dodge
import de_plots
import gffutils import gffutils
import numpy as np import numpy as np
import pandas as pd import pandas as pd
@ -816,6 +818,22 @@ def jaffal_table(report, result_csv):
section.table(df) section.table(df)
def de_section(report):
"""Make differential transcript expression section."""
dexseq = os.path.join("de_report", "results_dexseq.tsv")
dge = os.path.join("de_report", "results_dge.tsv")
dtu = os.path.join("de_report", "results_dtu_stageR.tsv")
stringtie = os.path.join("de_report", "stringtie_merged.gtf")
tpm = os.path.join("de_report", "tpm_counts.tsv")
de_plots.de_section(
stringtie=stringtie,
dexseq=dexseq,
dge=dge,
dtu=dtu,
tpm=tpm,
report=report)
def main(): def main():
"""Run the entry point.""" """Run the entry point."""
parser = argparse.ArgumentParser() parser = argparse.ArgumentParser()
@ -857,6 +875,12 @@ def main():
parser.add_argument( parser.add_argument(
"--jaffal_csv", required=False, type=str, default=None, "--jaffal_csv", required=False, type=str, default=None,
help="Path to JAFFAL results csv") help="Path to JAFFAL results csv")
parser.add_argument(
"--de_report", required=False, type=str, default=None,
help="Differential expression report optional")
parser.add_argument(
"--de_stats", required=False, type=str, default=None, nargs='*',
help="Differential expression report optional")
parser.add_argument('--denovo', dest='denovo', action='store_true') parser.add_argument('--denovo', dest='denovo', action='store_true')
args = parser.parse_args() args = parser.parse_args()
@ -902,6 +926,9 @@ def main():
if args.cluster_qc_dirs is not None: if args.cluster_qc_dirs is not None:
cluster_quality(args.cluster_qc_dirs, report, sample_ids) cluster_quality(args.cluster_qc_dirs, report, sample_ids)
if args.de_report:
de_section(report)
if args.jaffal_csv: if args.jaffal_csv:
jaffal_table(report, args.jaffal_csv) jaffal_table(report, args.jaffal_csv)

View File

@ -32,9 +32,17 @@ using [gffcompare](http://ccb.jhu.edu/software/stringtie/gffcompare.shtml)
Fusion gene detection is performed using [JAFFA](https://github.com/Oshlack/JAFFA), with the JAFFAL extension for use Fusion gene detection is performed using [JAFFA](https://github.com/Oshlack/JAFFA), with the JAFFAL extension for use
with ONT long reads. with ONT long reads.
### Differential expression analysis
* Differential expression is done using the transcripts output by the workflow.
* A non redundant transcriptome is found using the merge function in [stringtie](http://ccb.jhu.edu/software/stringtie).
* The reads are then aligned to the transcriptome using minimap2 in a splice-aware manner.
* [salmon](https://github.com/COMBINE-lab/salmon) is used for transcript quantification.
* R packages [edgeR](https://bioconductor.org/packages/release/bioc/html/edgeR.html) and [stageR](https://bioconductor.org/packages/release/bioc/html/stageR.html) are used for differential expression analysis.
* [DEXSeq](https://bioconductor.org/packages/release/bioc/html/DEXSeq.html) is then used for differential transcript usage analysis.
### Workflow inputs ### Workflow inputs
- Directory containing cDNA/direct RNA reads. Or a directory containing subdirectories each with reads from different samples - Directory containing cDNA/direct RNA reads. Or a directory containing subdirectories each with reads from different samples
(in fastq/fastq.gz format) (in fastq/fastq.gz format)
- Reference genome in fasta format (required for reference-based assembly). - Reference genome in fasta format (required for reference-based assembly).
- Optional reference annotation in GFF2/3 format. - Optional reference annotation in GFF2/3 format (required for differential expression analysis `--de_analysis`).
- For fusion detection, JAFFAL reference files (see Quickstart) - For fusion detection, JAFFAL reference files (see Quickstart)

View File

@ -116,6 +116,25 @@ g++ must be installed. JAFFAL is not currently working on Mac M1 (osx-arm64 arch
detected, the workflow will terminate with an error at the JAFFAL stage. If this happens, detected, the workflow will terminate with an error at the JAFFAL stage. If this happens,
skip the JAFFAL stage by omitting ` --jaffal_refBase` skip the JAFFAL stage by omitting ` --jaffal_refBase`
### Differential Expression
Differential Expression requires at least 2 replicates of each sample to compare. You can see an example condition_sheet.tsv in test_data.
**Example workflow for differential expression transcript assembly**
Download differential expression data set
`wget -O differential_expression.tar.gz https://ont-exd-int-s3-euwst1-epi2me-labs.s3.amazonaws.com/wf-isoforms/wf-isoforms_differential_expression.tar.gz && tar -xzvf differential_expression.tar.gz`
Run the cmd
```
OUTPUT=~/output;
nexflow run epi2me-labs/wf-transcriptomes --fastq differential_expression_dataset/fastq --de_analysis \
--ref_genome differential_expression_dataset/hg38_chr20.fa \
--ref_annotation differential_expression_dataset/gencode.v22.annotation.chr20.gtf \
--direct_rna
```
## Workflow outputs ## Workflow outputs
* an HTML report document detailing the primary findings of the workflow. * an HTML report document detailing the primary findings of the workflow.
@ -131,3 +150,6 @@ skip the JAFFAL stage by omitting ` --jaffal_refBase`
in `${out_dir}/jaffal_output_${sample_id}` you will find: in `${out_dir}/jaffal_output_${sample_id}` you will find:
* jaffa_results.csv - the csv results summary file * jaffa_results.csv - the csv results summary file
* jaffa_results.fasta - fusion transcritpt sequences * jaffa_results.fasta - fusion transcritpt sequences
### Differential Expression outputs
* dtu_plots.pdf - a pdf with differntial transcript usage plots

View File

@ -16,7 +16,7 @@ dependencies:
- gffread==0.12.7 - gffread==0.12.7
- gffcompare==0.11.2 - gffcompare==0.11.2
- gffutils==0.10.1 - gffutils==0.10.1
- seqkit==2.1.0 - seqkit==2.2.0
- stringtie==2.1.1 - stringtie==2.1.1
- curl - curl
- pysam==0.17.0 - pysam==0.17.0
@ -31,4 +31,13 @@ dependencies:
# - bpipe=0.9.9.2 # - bpipe=0.9.9.2
- java-jdk - java-jdk
- r-base - r-base
- gxx - gxx
# Differential expression dependencies
- bioconductor-genomicfeatures
- bioconductor-drimseq
- bioconductor-edger
- bioconductor-dexseq
- bioconductor-stager
- r-dplyr
- r-tidyr
- salmon==1.9.0

View File

@ -301,7 +301,12 @@ def barcode_in_range(path, min_barcode, max_barcode)
{ {
pattern = ~/barcode(\d+)/ pattern = ~/barcode(\d+)/
matcher = "${path}" =~ pattern matcher = "${path}" =~ pattern
value = matcher[0][1].toInteger() def value = null
try{
value = matcher[0][1].toInteger()
}catch(ArrayIndexOutOfBoundsException ex){
print("${path} is not a barcoded directory")
}
valid = ((value >= min_barcode) && (value <= max_barcode)) valid = ((value >= min_barcode) && (value <= max_barcode))
return valid return valid
} }

133
main.nf
View File

@ -15,6 +15,8 @@ include { start_ping; end_ping } from './lib/ping'
include { reference_assembly } from './subworkflows/reference_assembly' include { reference_assembly } from './subworkflows/reference_assembly'
include { denovo_assembly } from './subworkflows/denovo_assembly' include { denovo_assembly } from './subworkflows/denovo_assembly'
include { gene_fusions } from './subworkflows/JAFFAL/gene_fusions' include { gene_fusions } from './subworkflows/JAFFAL/gene_fusions'
include { differential_expression } from './subworkflows/differential_expression'
process summariseConcatReads { process summariseConcatReads {
@ -181,7 +183,7 @@ process assemble_transcripts{
def prefix = bam.name.split(/\./)[0] def prefix = bam.name.split(/\./)[0]
""" """
stringtie --rf ${G_FLAG} -L -v -p ${params.threads} ${params.stringtie_opts} \ stringtie --rf ${G_FLAG} -L -v -p ${task.cpus} ${params.stringtie_opts} \
-o ${prefix}.gff -l ${prefix} ${bam} 2>/dev/null -o ${prefix}.gff -l ${prefix} ${bam} 2>/dev/null
""" """
} }
@ -225,6 +227,7 @@ process run_gffcompare{
path ref_annotation path ref_annotation
output: output:
tuple val(sample_id), path("${sample_id}_gffcompare"), emit: gffcmp_dir tuple val(sample_id), path("${sample_id}_gffcompare"), emit: gffcmp_dir
path ("${sample_id}_annotated.gtf"), emit: gtf, optional: true
script: script:
def out_dir = "${sample_id}_gffcompare" def out_dir = "${sample_id}_gffcompare"
@ -245,6 +248,7 @@ process run_gffcompare{
mv *.tmap $out_dir mv *.tmap $out_dir
mv *.refmap $out_dir mv *.refmap $out_dir
cp ${out_dir}/str_merged.annotated.gtf ${sample_id}_annotated.gtf
""" """
} }
} }
@ -273,6 +277,29 @@ process get_transcriptome{
""" """
} }
process merge_transcriptomes {
// Merge the transcriptomes from all samples
label 'isoforms'
input:
path "query_annotations/*"
path ref_annotation
path ref_genome
output:
path "non_redundant.fasta", emit: fasta
path "stringtie.gtf", emit: gtf
"""
stringtie --merge -G $ref_annotation -p ${task.cpus} -o stringtie.gtf query_annotations/*
seqkit subseq --feature "transcript" --gtf-tag "transcript_id" --gtf stringtie.gtf $ref_genome > temp_transcriptome.fasta
seqkit rmdup -s < temp_transcriptome.fasta > temp_del_repeats.fasta
cat temp_del_repeats.fasta | sed 's/>.* />/' | sed -e 's/_[0-9]* \\[/ \\[/' > temp_rm_empty_seq.fasta
awk 'BEGIN {RS = ">" ; FS = "\\n" ; ORS = ""} \$2 {print ">"\$0}' temp_rm_empty_seq.fasta > non_redundant.fasta
rm temp_transcriptome.fasta
rm temp_del_repeats.fasta
rm temp_rm_empty_seq.fasta
"""
}
process makeReport { process makeReport {
label "isoforms" label "isoforms"
@ -288,6 +315,8 @@ process makeReport {
path(aln_stats), path(aln_stats),
path(gffcmp_dir), path(gffcmp_dir),
path(gff_annotation) path(gff_annotation)
path "de_report/*"
path "seqkit/*"
output: output:
path("wf-transcriptomes-*.html"), emit: report path("wf-transcriptomes-*.html"), emit: report
script: script:
@ -298,8 +327,15 @@ process makeReport {
def OPT_DENOVO = denovo ? "--denovo" : '' def OPT_DENOVO = denovo ? "--denovo" : ''
def OPT_PC_REPORT = pychopper_report.name.startsWith('OPTIONAL_FILE') ? '' : "--pychop_report ${pychopper_report}" def OPT_PC_REPORT = pychopper_report.name.startsWith('OPTIONAL_FILE') ? '' : "--pychop_report ${pychopper_report}"
def OPT_JAFFAL_CSV = jaffal_csv.name.startsWith('OPTIONAL_FILE') ? '' : "--jaffal_csv ${jaffal_csv}" def OPT_JAFFAL_CSV = jaffal_csv.name.startsWith('OPTIONAL_FILE') ? '' : "--jaffal_csv ${jaffal_csv}"
""" """
if [ -e "de_report/OPTIONAL_FILE" ]; then
dereport=""
else
dereport="--de_report true --de_stats "seqkit/*""
mv de_report/*.gtf de_report/stringtie_merged.gtf
fi
report.py --report $report_name \ report.py --report $report_name \
--versions $versions \ --versions $versions \
--params params.json \ --params params.json \
@ -311,7 +347,9 @@ process makeReport {
--gff_annotation $gff_annotation \ --gff_annotation $gff_annotation \
--isoform_table_nrows $params.isoform_table_nrows \ --isoform_table_nrows $params.isoform_table_nrows \
$OPT_JAFFAL_CSV \ $OPT_JAFFAL_CSV \
$OPT_DENOVO $OPT_DENOVO \
\$dereport
""" """
} }
@ -340,6 +378,8 @@ workflow pipeline {
jaffal_refBase jaffal_refBase
jaffal_genome jaffal_genome
jaffal_annotation jaffal_annotation
condition_sheet
ref_transcriptome
main: main:
map_sample_ids_cls = {it -> map_sample_ids_cls = {it ->
/* Harmonize tuples /* Harmonize tuples
@ -381,15 +421,15 @@ workflow pipeline {
if (params.denovo){ if (params.denovo){
println("Doing de novo assembly") println("Doing de novo assembly")
m = denovo_assembly(full_len_reads, ref_genome) assembly = denovo_assembly(full_len_reads, ref_genome)
} else { } else {
build_minimap_index(ref_genome) build_minimap_index(ref_genome)
println("Doing reference based transcript analysis") println("Doing reference based transcript analysis")
m = reference_assembly(build_minimap_index.out.index, ref_genome, full_len_reads) assembly = reference_assembly(build_minimap_index.out.index, ref_genome, full_len_reads)
} }
split_bam(m.bam) split_bam(assembly.bam)
assemble_transcripts(split_bam.out.bundles.flatMap(map_sample_ids_cls), ref_annotation) assemble_transcripts(split_bam.out.bundles.flatMap(map_sample_ids_cls), ref_annotation)
@ -401,13 +441,13 @@ workflow pipeline {
if (params.denovo){ if (params.denovo){
// Use the per-sample, de novo-assembled CDS // Use the per-sample, de novo-assembled CDS
seq_for_transcriptome_build = m.cds seq_for_transcriptome_build = assembly.cds
}else { }else {
// For reference based assembly, there is only one reference // For reference based assembly, there is only one reference
// So map this reference to all sample_ids // So map this reference to all sample_ids
seq_for_transcriptome_build = sample_ids.flatten().combine(Channel.fromPath(params.ref_genome)) seq_for_transcriptome_build = sample_ids.flatten().combine(Channel.fromPath(params.ref_genome))
} }
if (jaffal_refBase){ if (jaffal_refBase){
gene_fusions(full_len_reads, jaffal_refBase, jaffal_genome, jaffal_annotation) gene_fusions(full_len_reads, jaffal_refBase, jaffal_genome, jaffal_annotation)
jaffal_out = gene_fusions.out.results_csv.collectFile(keepHeader: true, name: 'jaffal.csv') jaffal_out = gene_fusions.out.results_csv.collectFile(keepHeader: true, name: 'jaffal.csv')
@ -415,6 +455,31 @@ workflow pipeline {
jaffal_out = file("$projectDir/data/OPTIONAL_FILE_1") jaffal_out = file("$projectDir/data/OPTIONAL_FILE_1")
} }
get_transcriptome(
merge_gff_bundles.out.gff
.join(run_gffcompare.out.gffcmp_dir)
.join(seq_for_transcriptome_build))
if (params.de_analysis){
if (!params.ref_transcriptome){
merge_transcriptomes(run_gffcompare.output.gtf.collect(), ref_annotation, ref_genome)
transcriptome = merge_transcriptomes.out.fasta
gtf = merge_transcriptomes.out.gtf
}
else {
transcriptome = ref_transcriptome
gtf = Channel.fromPath(ref_annotation)
}
de = differential_expression(transcriptome, summariseConcatReads.out.input_reads, condition_sheet, gtf)
de_report = de.all_de
count_transcripts_file = de.count_transcripts
dtu_plots = de.dtu_plots
} else{
de_report = file("$projectDir/data/OPTIONAL_FILE")
count_transcripts_file = file("$projectDir/data/OPTIONAL_FILE")
}
makeReport( makeReport(
software_versions, software_versions,
workflow_params, workflow_params,
@ -422,42 +487,41 @@ workflow pipeline {
pychopper_report, pychopper_report,
jaffal_out, jaffal_out,
summariseConcatReads.out.summary summariseConcatReads.out.summary
.join(m.stats) .join(assembly.stats)
.join(run_gffcompare.out.gffcmp_dir) .join(run_gffcompare.out.gffcmp_dir)
.join(merge_gff_bundles.out.gff) .join(merge_gff_bundles.out.gff)
.toList().transpose().toList()) .toList().transpose().toList(),
de_report,
count_transcripts_file)
report = makeReport.out.report report = makeReport.out.report
get_transcriptome(
merge_gff_bundles.out.gff
.join(run_gffcompare.out.gffcmp_dir)
.join(seq_for_transcriptome_build))
if (use_ref_ann){ if (use_ref_ann){
results = run_gffcompare.output.gffcmp_dir results = run_gffcompare.output.gffcmp_dir
.concat( .concat(
m.stats, assembly.stats,
get_transcriptome.out.flatMap(map_sample_ids_cls)) get_transcriptome.out.transcriptome.flatMap(map_sample_ids_cls))
.map {it -> it[1]} .map {it -> it[1]}
.concat(makeReport.out.report) .concat(makeReport.out.report)
} }
if (!use_ref_ann && !params.denovo){ if (!use_ref_ann && !params.denovo){
results = m.stats results = assembly.stats
.concat( .concat(
get_transcriptome.out.flatMap(map_sample_ids_cls)) get_transcriptome.out.transcriptome.flatMap(map_sample_ids_cls))
.map {it -> it[1]} .map {it -> it[1]}
.concat(makeReport.out.report) .concat(makeReport.out.report)
} }
if (params.denovo){ if (params.denovo){
results = m.cds results = assembly.cds
.concat(m.stats, .concat(assembly.stats,
seq_for_transcriptome_build, seq_for_transcriptome_build,
get_transcriptome.out.flatMap(map_sample_ids_cls), get_transcriptome.out.transcriptome.flatMap(map_sample_ids_cls),
merge_gff_bundles.out.gff, merge_gff_bundles.out.gff,
m.opt_qual_ch.flatMap { assembly.opt_qual_ch.flatMap {
it -> it ->
l = [] l = []
for (x in it[1..-1]){ for (x in it[1..-1]){
@ -474,6 +538,10 @@ workflow pipeline {
.map {it -> it[1]}) .map {it -> it[1]})
} }
if (params.de_analysis){
results = results.concat(de.dtu_plots)
}
emit: emit:
results results
telemetry = workflow_params telemetry = workflow_params
@ -518,7 +586,7 @@ workflow {
}else{ }else{
ref_annotation = file("$projectDir/data/OPTIONAL_FILE") ref_annotation = file("$projectDir/data/OPTIONAL_FILE")
} }
if (params.jaffal_refBase){ if (params.jaffal_refBase){
jaffal_refBase = file(params.jaffal_refBase, type: "dir") jaffal_refBase = file(params.jaffal_refBase, type: "dir")
if (!jaffal_refBase.exists()) { if (!jaffal_refBase.exists()) {
error = "--jaffa_refBase: Directory doesn't exist, check path." error = "--jaffa_refBase: Directory doesn't exist, check path."
@ -526,7 +594,22 @@ workflow {
}else{ }else{
jaffal_refBase = null jaffal_refBase = null
} }
ref_transcriptome = file("$projectDir/data/OPTIONAL_FILE")
if (params.ref_transcriptome){
ref_transcriptome = file(params.ref_transcriptome, type:"file")
}
if (params.de_analysis){
if (!params.ref_annotation){
error = "You must provide a reference annotation."
}
if (!params.condition_sheet){
error = "You must provide a condition_sheet or set de_analysis to false."
}
condition_sheet = file(params.condition_sheet, type:"file")
} else{
condition_sheet = file("$projectDir/data/OPTIONAL_FILE")
}
if (error){ if (error){
println(error) println(error)
}else{ }else{
@ -537,7 +620,9 @@ workflow {
"sanitize": params.sanitize_fastq, "sanitize": params.sanitize_fastq,
"output":params.out_dir]) "output":params.out_dir])
pipeline(reads, ref_genome, ref_annotation, jaffal_refBase, params.jaffal_genome, params.jaffal_annotation) pipeline(reads, ref_genome, ref_annotation,
jaffal_refBase, params.jaffal_genome, params.jaffal_annotation,
condition_sheet, ref_transcriptome)
output(pipeline.out.results) output(pipeline.out.results)

View File

@ -127,6 +127,16 @@ params {
// This needs overriding if running elsewhere // This needs overriding if running elsewhere
jaffal_dir = "/home/epi2melabs/JAFFA" jaffal_dir = "/home/epi2melabs/JAFFA"
// de options
de_analysis = false
condition_sheet = "test_data/condition_sheet.tsv"
ref_transcriptome = null
min_samps_gene_expr = 3
min_samps_feature_expr = 1
min_gene_expr = 10
min_feature_expr = 3
wf { wf {
example_cmd = [ example_cmd = [
"--fastq test_data/fastq", "--fastq test_data/fastq",
@ -189,18 +199,19 @@ profiles {
// profile using conda environments // profile using conda environments
conda { conda {
docker.enabled = false docker.enabled = false
process { process {
withLabel:isoforms { withLabel:isoforms {
conda = "${projectDir}/environment.yaml" conda = "${projectDir}/environment.yaml"
}
shell = ['/bin/bash', '-euo', 'pipefail']
}
conda {
cacheDir = ""
useMamba = true
} }
shell = ['/bin/bash', '-euo', 'pipefail']
} }
conda {
enabled = true // required for 22.08
cacheDir = ""
useMamba = true
}
}
// Using AWS batch. // Using AWS batch.
// May need to set aws.region and aws.batch.cliPath // May need to set aws.region and aws.batch.cliPath

View File

@ -247,6 +247,51 @@
} }
} }
}, },
"differential_expression_options": {
"title": "Differential expression options",
"type": "object",
"description": "",
"default": "",
"properties": {
"de_analysis": {
"type": "boolean",
"description": "Run DE anaylsis",
"help_text": "Running this requires you to provide at least two replicates for a control and treated sample as well as a condition sheet param."
},
"condition_sheet": {
"type": "string",
"format": "file-path",
"description": "tsv with (sample, condition, type)",
"default": "null"
},
"ref_transcriptome": {
"type": "string",
"default": "null",
"format": "file-path",
"description": "Transcriptome reference file"
},
"min_gene_expr": {
"type": "integer",
"default": 10,
"description": "Minimum gene counts"
},
"min_feature_expr": {
"type": "integer",
"default": 3,
"description": "Minimum transcript counts"
},
"min_samps_feature_expr": {
"type": "integer",
"default": 1,
"description": "Transcripts expressed in minimum this many samples"
},
"min_samps_gene_expr": {
"type": "integer",
"description": "Genes expressed in minimum this many samples",
"default": 3
}
}
},
"meta_data": { "meta_data": {
"title": "Meta Data", "title": "Meta Data",
"type": "object", "type": "object",
@ -297,6 +342,9 @@
{ {
"$ref": "#/definitions/fusion_detection_options" "$ref": "#/definitions/fusion_detection_options"
}, },
{
"$ref": "#/definitions/differential_expression_options"
},
{ {
"$ref": "#/definitions/meta_data" "$ref": "#/definitions/meta_data"
}, },
@ -330,7 +378,7 @@
} }
}, },
"docs": { "docs": {
"intro": "## Introduction\n\nThis workflow identifies RNA isoforms using either cDNA or direct RNA (dRNA) \nOxford Nanopore reads.\n\n### Preprocesing\ncDNA reads are initially preprocessed by [pychopper](https://github.com/epi2me-labs/pychopper) \nfor the identification of full-length reads, as well as trimming and orientation correction (This step is omitted for \n direct RNA reads).\n\n\n### Transcript assembly\n\n#### Reference-aided transcript assembly approach\n* Full length reads are mapped to a supplied reference genome using [minimap2](https://github.com/lh3/minimap2)\n* Transcripts are assembled by [stringtie](http://ccb.jhu.edu/software/stringtie) \nin long read mode (with or without a guide reference annotation) to generate the GFF annotation.\n* The annotation generated by the pipeline is compared to the reference annotation. \nusing [gffcompare](http://ccb.jhu.edu/software/stringtie/gffcompare.shtml)\n\n#### de novo-based transcript assembly (experimental!)\n* Sequence clusters are generated using [isONclust2](https://github.com/nanoporetech/isONclust2)\n * If a reference genome is supplied, cluster quality metrics are determined by comparing \n with clusters generated from a minimap2 alignment.\n* A consensus sequence for each cluster is generated using [spoa](https://github.com/rvaser/spoa)\n* Three rounds of polishing using racon and minimap2 to give a final polished CDS for each gene.\n* Full-length reads are then mapped to these polished CDS.\n* Transcripts are assembled by stringtie as for the reference-based approach.\n* __Note__: This approach is currently not supported with direct RNA reads.\n\n### Fusion gene detection\nFusion gene detection is performed using [JAFFA](https://github.com/Oshlack/JAFFA), with the JAFFAL extension for use \nwith ONT long reads. \n\n### Workflow inputs\n- Directory containing cDNA/direct RNA reads. Or a directory containing subdirectories each with reads from different samples\n (in fastq/fastq.gz format)\n- Reference genome in fasta format (required for reference-based assembly).\n- Optional reference annotation in GFF2/3 format.\n- For fusion detection, JAFFAL reference files (see Quickstart) \n", "intro": "## Introduction\n\nThis workflow identifies RNA isoforms using either cDNA or direct RNA (dRNA) \nOxford Nanopore reads.\n\n### Preprocesing\ncDNA reads are initially preprocessed by [pychopper](https://github.com/epi2me-labs/pychopper) \nfor the identification of full-length reads, as well as trimming and orientation correction (This step is omitted for \n direct RNA reads).\n\n\n### Transcript assembly\n\n#### Reference-aided transcript assembly approach\n* Full length reads are mapped to a supplied reference genome using [minimap2](https://github.com/lh3/minimap2)\n* Transcripts are assembled by [stringtie](http://ccb.jhu.edu/software/stringtie) \nin long read mode (with or without a guide reference annotation) to generate the GFF annotation.\n* The annotation generated by the pipeline is compared to the reference annotation. \nusing [gffcompare](http://ccb.jhu.edu/software/stringtie/gffcompare.shtml)\n\n#### de novo-based transcript assembly (experimental!)\n* Sequence clusters are generated using [isONclust2](https://github.com/nanoporetech/isONclust2)\n * If a reference genome is supplied, cluster quality metrics are determined by comparing \n with clusters generated from a minimap2 alignment.\n* A consensus sequence for each cluster is generated using [spoa](https://github.com/rvaser/spoa)\n* Three rounds of polishing using racon and minimap2 to give a final polished CDS for each gene.\n* Full-length reads are then mapped to these polished CDS.\n* Transcripts are assembled by stringtie as for the reference-based approach.\n* __Note__: This approach is currently not supported with direct RNA reads.\n\n### Fusion gene detection\nFusion gene detection is performed using [JAFFA](https://github.com/Oshlack/JAFFA), with the JAFFAL extension for use \nwith ONT long reads. \n\n### Differential expression analysis\n* Differential expression is done using the transcripts output by the workflow.\n* A non redundant transcriptome is found using the merge function in [stringtie](http://ccb.jhu.edu/software/stringtie).\n* The reads are then aligned to the transcriptome using minimap2 in a splice-aware manner.\n* [salmon](https://github.com/COMBINE-lab/salmon) is used for transcript quantification.\n* R packages [edgeR](https://bioconductor.org/packages/release/bioc/html/edgeR.html) and [stageR](https://bioconductor.org/packages/release/bioc/html/stageR.html) are used for differential expression analysis.\n* [DEXSeq](https://bioconductor.org/packages/release/bioc/html/DEXSeq.html) is then used for differential transcript usage analysis.\n\n### Workflow inputs\n- Directory containing cDNA/direct RNA reads. Or a directory containing subdirectories each with reads from different samples\n (in fastq/fastq.gz format)\n- Reference genome in fasta format (required for reference-based assembly).\n- Optional reference annotation in GFF2/3 format (required for differential expression analysis `--de_analysis`).\n- For fusion detection, JAFFAL reference files (see Quickstart) \n",
"links": "## Useful links\n\n* [nextflow](https://www.nextflow.io/)\n* [docker](https://www.docker.com/products/docker-desktop)\n* [Singularity](https://sylabs.io/singularity/)\n* [conda](https://docs.conda.io/en/latest/miniconda.html)\n* [racon](https://github.com/isovic/racon)\n* [spoa](https://github.com/rvaser/spoa)\n* [inONclust](https://github.com/ksahlin/isONclust)\n* [isONclust2](https://github.com/nanoporetech/isONclust2)" "links": "## Useful links\n\n* [nextflow](https://www.nextflow.io/)\n* [docker](https://www.docker.com/products/docker-desktop)\n* [Singularity](https://sylabs.io/singularity/)\n* [conda](https://docs.conda.io/en/latest/miniconda.html)\n* [racon](https://github.com/isovic/racon)\n* [spoa](https://github.com/rvaser/spoa)\n* [inONclust](https://github.com/ksahlin/isONclust)\n* [isONclust2](https://github.com/nanoporetech/isONclust2)"
} }
} }

View File

@ -0,0 +1,156 @@
process count_transcripts {
// Count transcripts using Salmon.
// library type is specified as forward stranded (-l SF) as it should have either been through pychopper or come from direct RNA reads.
label "isoforms"
input:
tuple val(sample_id), path(bam)
path ref_transcriptome
output:
path "*transcript_counts.tsv", emit: counts
path "*seqkit.stats", emit: seqkit_stats
"""
salmon quant --noErrorModel -p $params.threads -t $ref_transcriptome -l SF -a $bam -o counts
mv counts/quant.sf "${sample_id}".transcript_counts.tsv
seqkit bam "$bam" 2> "${sample_id}".seqkit.stats
"""
}
process mergeCounts {
label "isoforms"
input:
path counts
output:
path "all_counts.tsv"
"""
merge_count_tsvs.py -z -o all_counts.tsv $counts
"""
}
process mergeTPM {
label "isoforms"
input:
path counts
output:
path "tpm_counts.tsv"
"""
merge_count_tsvs.py -z -o tpm_counts.tsv $counts -tpm
"""
}
process deAnalysis {
label "isoforms"
input:
path condition_sheet
path merged_tsv
path annotation
output:
path "de_analysis/results_dtu_stageR.tsv", emit: stageR
path "merged/all_counts_filtered.tsv", emit: flt_counts
path "merged/all_gene_counts.tsv", emit: gene_counts
path "de_analysis/results_dge.tsv", emit: dge
path "de_analysis/results_dexseq.tsv", emit: dexseq
"""
cp $annotation annotation.gtf
echo \$(realpath annotation.gtf)
echo Annotation\$'\t'min_samps_gene_expr\$'\t'min_samps_feature_expr\$'\t'min_gene_expr\$'\t'min_feature_expr > params.tsv
echo \$(realpath $params.ref_annotation)\$'\t'$params.min_samps_gene_expr\$'\t'\
$params.min_samps_feature_expr\$'\t'$params.min_gene_expr\$'\t'$params.min_feature_expr >> params.tsv
mkdir merged
mkdir de_analysis
mv $merged_tsv merged/all_counts.tsv
mv params.tsv de_analysis/de_params.tsv
mv $condition_sheet de_analysis/coldata.tsv
de_analysis.R
"""
}
process plotResults {
label "isoforms"
input:
path flt_count
path res_dtu
path condition_sheet
output:
path "de_analysis/dtu_plots.pdf", emit: dtu_plots
path "condition_sheet.tsv", emit: condition_sheet_tsv
"""
mkdir merged
mkdir de_analysis
mv $res_dtu de_analysis/results_dtu_stageR.tsv
mv $condition_sheet de_analysis/coldata.tsv
mv $flt_count merged/all_counts_filtered.tsv
plot_dtu_results.R
cp de_analysis/coldata.tsv condition_sheet.tsv
"""
}
process build_minimap_index_transcriptome{
/*
Build minimap index from reference genome
*/
label "isoforms"
cpus params.threads
input:
path reference
output:
path "genome_index.mmi", emit: index
script:
"""
minimap2 -t ${params.threads} ${params.minimap_index_opts} -I 1000G -d "genome_index.mmi" ${reference}
"""
}
process map_transcriptome{
/*
Map reads to reference using minimap2.
Filter reads by mapping quality.
Filter internally-primed reads.
*/
label "isoforms"
cpus params.threads
input:
tuple val(sample_id), path (fastq_reads)
file index
file transcript_reference
output:
tuple val(sample_id), path("${sample_id}_reads_aln_sorted.bam"), emit: bam
"""
minimap2 -t ${params.threads} -ax splice -uf -p 1.0 $index $fastq_reads\
| samtools view -Sb > output.bam
samtools sort -@ ${params.threads} output.bam -o "${sample_id}"_reads_aln_sorted.bam
samtools index "${sample_id}"_reads_aln_sorted.bam
"""
}
workflow differential_expression {
take:
ref_transcriptome
full_len_reads
condition_sheet
ref_annotation
main:
t_index = build_minimap_index_transcriptome(ref_transcriptome)
mapped = map_transcriptome(full_len_reads, t_index, ref_transcriptome)
count_transcripts(mapped.bam, ref_transcriptome)
merged = mergeCounts(count_transcripts.out.counts.collect())
merged_TPM = mergeTPM(count_transcripts.out.counts.collect())
analysis = deAnalysis(condition_sheet, merged, ref_annotation)
plotResults(analysis.flt_counts, analysis.stageR, condition_sheet)
de_report = analysis.flt_counts.combine(analysis.gene_counts).combine(analysis.dge).combine(analysis.dexseq).combine(
analysis.stageR).combine(plotResults.out.condition_sheet_tsv).combine(merged).combine(
ref_annotation).combine(merged_TPM)
count_transcripts_file = count_transcripts.out.seqkit_stats.collect()
emit:
all_de = de_report
count_transcripts = count_transcripts_file
dtu_plots = plotResults.out.dtu_plots
}

View File

@ -0,0 +1,7 @@
sample,condition,type
barcode01,untreated,single-read
barcode02,untreated,single-read
barcode03,untreated,single-read
barcode04,treated,single-read
barcode05,treated,single-read
barcode06,treated,single-read
1 sample,condition,type
2 barcode01,untreated,single-read
3 barcode02,untreated,single-read
4 barcode03,untreated,single-read
5 barcode04,treated,single-read
6 barcode05,treated,single-read
7 barcode06,treated,single-read