Merge branch '649-b' into 'dev'
CW-649 See merge request epi2melabs/workflows/wf-transcriptomes!72
This commit is contained in:
commit
835385e6af
@ -4,12 +4,52 @@ include:
|
|||||||
file: "wf-containers.yaml"
|
file: "wf-containers.yaml"
|
||||||
|
|
||||||
variables:
|
variables:
|
||||||
# Workflow inputs given to nextflow.
|
NF_BEFORE_SCRIPT: wget -O test_data.tar.gz https://ont-exd-int-s3-euwst1-epi2me-labs.s3.amazonaws.com/wf-isoforms/wf-isoforms_test_data.tar.gz && tar -xzvf test_data.tar.gz
|
||||||
# The workflow should define `--out_dir`, the CI template sets this.
|
NF_WORKFLOW_OPTS: "--fastq ERR6053095_chr20.fastq \
|
||||||
# Only common file inputs and option values need to be given here
|
--ref_genome chr20/hg38_chr20.fa --ref_annotation chr20/gencode.v22.annotation.chr20.gtf \
|
||||||
# (not things such as -profile)
|
--jaffal_refBase chr20/ --jaffal_genome hg38_chr20 --jaffal_annotation genCode22"
|
||||||
NF_BEFORE_SCRIPT: |
|
NF_IGNORE_PROCESSES: preprocess_reads,merge_transcriptomes
|
||||||
wget -O test_data.tar.gz https://ont-exd-int-s3-euwst1-epi2me-labs.s3.amazonaws.com/wf-isoforms/wf-isoforms_test_data.tar.gz && tar -xzvf test_data.tar.gz
|
|
||||||
NF_WORKFLOW_OPTS: "--fastq ERR6053095_chr20.fastq \
|
docker-run:
|
||||||
--ref_genome chr20/hg38_chr20.fa --ref_annotation chr20/gencode.v22.annotation.chr20.gtf \
|
|
||||||
--jaffal_refBase chr20/ --jaffal_genome hg38_chr20 --jaffal_annotation genCode22"
|
# Remove this directive in downstream templates
|
||||||
|
tags: [large_ram] # no need for big ram
|
||||||
|
|
||||||
|
# Define a 1D job matrix to inject a variable named MATRIX_NAME into
|
||||||
|
# the CI environment, we can use the value of MATRIX_NAME to determine
|
||||||
|
# which options to apply as part of the rules block below
|
||||||
|
# NOTE There is a slightly cleaner way to define this matrix to include
|
||||||
|
# the variables, but it is broken when using long strings! See CW-756
|
||||||
|
parallel:
|
||||||
|
matrix:
|
||||||
|
- MATRIX_NAME: [
|
||||||
|
"fusions", "differential_expression", "isoforms"
|
||||||
|
]
|
||||||
|
rules:
|
||||||
|
# NOTE As we're overriding the rules block for the included docker-run
|
||||||
|
# we must redefine this CI_COMMIT_BRANCH rule to prevent docker-run
|
||||||
|
# being incorrectly scheduled for "detached merge request pipelines" etc.
|
||||||
|
- if: ($CI_COMMIT_BRANCH == null || $CI_COMMIT_BRANCH == "dev-template")
|
||||||
|
when: never
|
||||||
|
- if: $MATRIX_NAME == "isoforms"
|
||||||
|
variables:
|
||||||
|
NF_BEFORE_SCRIPT: wget -O test_data.tar.gz https://ont-exd-int-s3-euwst1-epi2me-labs.s3.amazonaws.com/wf-isoforms/wf-isoforms_test_data.tar.gz && tar -xzvf test_data.tar.gz
|
||||||
|
NF_WORKFLOW_OPTS: "--fastq ERR6053095_chr20.fastq \
|
||||||
|
--ref_genome chr20/hg38_chr20.fa --ref_annotation chr20/gencode.v22.annotation.chr20.gtf"
|
||||||
|
NF_IGNORE_PROCESSES: preprocess_reads,merge_transcriptomes
|
||||||
|
- if: $MATRIX_NAME == "fusions"
|
||||||
|
variables:
|
||||||
|
NF_BEFORE_SCRIPT: wget -O test_data.tar.gz https://ont-exd-int-s3-euwst1-epi2me-labs.s3.amazonaws.com/wf-isoforms/wf-isoforms_test_data.tar.gz && tar -xzvf test_data.tar.gz
|
||||||
|
NF_WORKFLOW_OPTS: "--fastq ERR6053095_chr20.fastq \
|
||||||
|
--ref_genome chr20/hg38_chr20.fa --ref_annotation chr20/gencode.v22.annotation.chr20.gtf \
|
||||||
|
--jaffal_refBase chr20/ --jaffal_genome hg38_chr20 --jaffal_annotation genCode22"
|
||||||
|
NF_IGNORE_PROCESSES: preprocess_reads,merge_transcriptomes
|
||||||
|
- if: $MATRIX_NAME == "differential_expression"
|
||||||
|
variables:
|
||||||
|
NF_BEFORE_SCRIPT: wget -O differential_expression.tar.gz https://ont-exd-int-s3-euwst1-epi2me-labs.s3.amazonaws.com/wf-isoforms/wf-isoforms_differential_expression.tar.gz && tar -xzvf differential_expression.tar.gz
|
||||||
|
NF_WORKFLOW_OPTS: "--fastq differential_expression_dataset/fastq \
|
||||||
|
--de_analysis \
|
||||||
|
--ref_genome differential_expression_dataset/hg38_chr20.fa \
|
||||||
|
--ref_annotation differential_expression_dataset/gencode.v22.annotation.chr20.gtf \
|
||||||
|
--direct_rna"
|
||||||
|
NF_IGNORE_PROCESSES: preprocess_reads,merge_transcriptomes
|
||||||
|
|||||||
@ -5,8 +5,8 @@ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.0.0/),
|
|||||||
and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html).
|
and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html).
|
||||||
|
|
||||||
## [unreleased]
|
## [unreleased]
|
||||||
### Changed
|
### Added
|
||||||
|
- Differential transcript and gene expression subworkflow
|
||||||
|
|
||||||
## [v0.1.4]
|
## [v0.1.4]
|
||||||
### Added
|
### Added
|
||||||
|
|||||||
@ -1,10 +1,10 @@
|
|||||||
ARG BASEIMAGE=ontresearch/base-workflow-image:v0.1.1
|
ARG BASEIMAGE=ontresearch/base-workflow-image:v0.2.0
|
||||||
FROM $BASEIMAGE
|
FROM $BASEIMAGE
|
||||||
ARG ENVFILE=environment.yaml
|
ARG ENVFILE=environment.yaml
|
||||||
|
|
||||||
COPY $ENVFILE $HOME/environment.yaml
|
COPY $ENVFILE $HOME/environment.yaml
|
||||||
RUN \
|
RUN \
|
||||||
. $CONDA_DIR/etc/profile.d/mamba.sh \
|
. $CONDA_DIR/etc/profile.d/micromamba.sh \
|
||||||
&& micromamba activate \
|
&& micromamba activate \
|
||||||
&& micromamba install -n base --file $HOME/environment.yaml \
|
&& micromamba install -n base --file $HOME/environment.yaml \
|
||||||
&& micromamba clean --all --yes \
|
&& micromamba clean --all --yes \
|
||||||
@ -15,7 +15,6 @@ RUN \
|
|||||||
&& rm -rf $CONDA_DIR/lib/python3.*/site-packages/pip \
|
&& rm -rf $CONDA_DIR/lib/python3.*/site-packages/pip \
|
||||||
&& find $CONDA_DIR -name '__pycache__' -type d -exec rm -rf '{}' '+'
|
&& find $CONDA_DIR -name '__pycache__' -type d -exec rm -rf '{}' '+'
|
||||||
|
|
||||||
|
|
||||||
USER $WF_UID
|
USER $WF_UID
|
||||||
WORKDIR $HOME
|
WORKDIR $HOME
|
||||||
|
|
||||||
|
|||||||
32
README.md
32
README.md
@ -40,11 +40,19 @@ using [gffcompare](http://ccb.jhu.edu/software/stringtie/gffcompare.shtml)
|
|||||||
Fusion gene detection is performed using [JAFFA](https://github.com/Oshlack/JAFFA), with the JAFFAL extension for use
|
Fusion gene detection is performed using [JAFFA](https://github.com/Oshlack/JAFFA), with the JAFFAL extension for use
|
||||||
with ONT long reads.
|
with ONT long reads.
|
||||||
|
|
||||||
|
### Differential expression analysis
|
||||||
|
* Differential expression is done using the transcripts output by the workflow.
|
||||||
|
* A non redundant transcriptome is found using the merge function in [stringtie](http://ccb.jhu.edu/software/stringtie).
|
||||||
|
* The reads are then aligned to the transcriptome using minimap2 in a splice-aware manner.
|
||||||
|
* [salmon](https://github.com/COMBINE-lab/salmon) is used for transcript quantification.
|
||||||
|
* R packages [edgeR](https://bioconductor.org/packages/release/bioc/html/edgeR.html) and [stageR](https://bioconductor.org/packages/release/bioc/html/stageR.html) are used for differential expression analysis.
|
||||||
|
* [DEXSeq](https://bioconductor.org/packages/release/bioc/html/DEXSeq.html) is then used for differential transcript usage analysis.
|
||||||
|
|
||||||
### Workflow inputs
|
### Workflow inputs
|
||||||
- Directory containing cDNA/direct RNA reads. Or a directory containing subdirectories each with reads from different samples
|
- Directory containing cDNA/direct RNA reads. Or a directory containing subdirectories each with reads from different samples
|
||||||
(in fastq/fastq.gz format)
|
(in fastq/fastq.gz format)
|
||||||
- Reference genome in fasta format (required for reference-based assembly).
|
- Reference genome in fasta format (required for reference-based assembly).
|
||||||
- Optional reference annotation in GFF2/3 format.
|
- Optional reference annotation in GFF2/3 format (required for differential expression analysis `--de_analysis`).
|
||||||
- For fusion detection, JAFFAL reference files (see Quickstart)
|
- For fusion detection, JAFFAL reference files (see Quickstart)
|
||||||
## Quickstart
|
## Quickstart
|
||||||
|
|
||||||
@ -164,6 +172,25 @@ g++ must be installed. JAFFAL is not currently working on Mac M1 (osx-arm64 arch
|
|||||||
detected, the workflow will terminate with an error at the JAFFAL stage. If this happens,
|
detected, the workflow will terminate with an error at the JAFFAL stage. If this happens,
|
||||||
skip the JAFFAL stage by omitting ` --jaffal_refBase`
|
skip the JAFFAL stage by omitting ` --jaffal_refBase`
|
||||||
|
|
||||||
|
### Differential Expression
|
||||||
|
|
||||||
|
Differential Expression requires at least 2 replicates of each sample to compare. You can see an example condition_sheet.tsv in test_data.
|
||||||
|
|
||||||
|
**Example workflow for differential expression transcript assembly**
|
||||||
|
|
||||||
|
Download differential expression data set
|
||||||
|
|
||||||
|
`wget -O differential_expression.tar.gz https://ont-exd-int-s3-euwst1-epi2me-labs.s3.amazonaws.com/wf-isoforms/wf-isoforms_differential_expression.tar.gz && tar -xzvf differential_expression.tar.gz`
|
||||||
|
|
||||||
|
Run the cmd
|
||||||
|
|
||||||
|
```
|
||||||
|
OUTPUT=~/output;
|
||||||
|
nexflow run epi2me-labs/wf-transcriptomes --fastq differential_expression_dataset/fastq --de_analysis \
|
||||||
|
--ref_genome differential_expression_dataset/hg38_chr20.fa \
|
||||||
|
--ref_annotation differential_expression_dataset/gencode.v22.annotation.chr20.gtf \
|
||||||
|
--direct_rna
|
||||||
|
```
|
||||||
|
|
||||||
## Workflow outputs
|
## Workflow outputs
|
||||||
* an HTML report document detailing the primary findings of the workflow.
|
* an HTML report document detailing the primary findings of the workflow.
|
||||||
@ -179,6 +206,9 @@ skip the JAFFAL stage by omitting ` --jaffal_refBase`
|
|||||||
in `${out_dir}/jaffal_output_${sample_id}` you will find:
|
in `${out_dir}/jaffal_output_${sample_id}` you will find:
|
||||||
* jaffa_results.csv - the csv results summary file
|
* jaffa_results.csv - the csv results summary file
|
||||||
* jaffa_results.fasta - fusion transcritpt sequences
|
* jaffa_results.fasta - fusion transcritpt sequences
|
||||||
|
|
||||||
|
### Differential Expression outputs
|
||||||
|
* dtu_plots.pdf - a pdf with differntial transcript usage plots
|
||||||
## Useful links
|
## Useful links
|
||||||
|
|
||||||
* [nextflow](https://www.nextflow.io/)
|
* [nextflow](https://www.nextflow.io/)
|
||||||
|
|||||||
139
bin/de_analysis.R
Executable file
139
bin/de_analysis.R
Executable file
@ -0,0 +1,139 @@
|
|||||||
|
#!/usr/bin/env Rscript
|
||||||
|
|
||||||
|
suppressMessages(library("DRIMSeq"))
|
||||||
|
suppressMessages(library("GenomicFeatures"))
|
||||||
|
|
||||||
|
cat("Loading counts, conditions and parameters.\n")
|
||||||
|
cts <- as.matrix(read.csv("merged/all_counts.tsv", sep="\t", row.names="Reference", stringsAsFactors=FALSE))
|
||||||
|
|
||||||
|
# Set up sample data frame:
|
||||||
|
coldata <- read.csv("de_analysis/coldata.tsv", row.names="sample", sep=",", stringsAsFactors=TRUE)
|
||||||
|
|
||||||
|
coldata$sample_id <- rownames(coldata)
|
||||||
|
coldata$condition <- factor(coldata$condition, levels=rev(levels(coldata$condition)))
|
||||||
|
|
||||||
|
de_params <- read.csv("de_analysis/de_params.tsv", sep="\t", stringsAsFactors=FALSE)
|
||||||
|
|
||||||
|
cat("Loading annotation database.\n")
|
||||||
|
#txdb <- makeTxDbFromGFF(de_params$Annotation[[1]])
|
||||||
|
txdb <- makeTxDbFromGFF("annotation.gtf")
|
||||||
|
txdf <- select(txdb, keys(txdb,"GENEID"), "TXNAME", "GENEID")
|
||||||
|
tab <- table(txdf$GENEID)
|
||||||
|
txdf$ntx<- tab[match(txdf$GENEID, names(tab))]
|
||||||
|
|
||||||
|
strip_version<-function(x) {
|
||||||
|
tmp<-data.frame(strsplit(x,".", fixed=TRUE), stringsAsFactors=FALSE)
|
||||||
|
tmp<-as.vector(tmp[1,])
|
||||||
|
colnames(tmp) <- c()
|
||||||
|
rownames(tmp) <- c()
|
||||||
|
return(tmp)
|
||||||
|
}
|
||||||
|
|
||||||
|
#rownames(cts) <- strip_version(rownames(cts))
|
||||||
|
|
||||||
|
cts <- cts[rownames(cts) %in% txdf$TXNAME, ] # FIXME: filter for transcripts which are in the annotation. Why they are not all there?
|
||||||
|
|
||||||
|
# Reorder transcript/gene database to match input counts:
|
||||||
|
txdf <- txdf[match(rownames(cts), txdf$TXNAME), ]
|
||||||
|
rownames(txdf) <- NULL
|
||||||
|
|
||||||
|
# Create counts data frame:
|
||||||
|
counts<-data.frame(gene_id=txdf$GENEID, feature_id=txdf$TXNAME, cts)
|
||||||
|
|
||||||
|
cat("Filtering counts using DRIMSeq.\n")
|
||||||
|
print(coldata)
|
||||||
|
d <- dmDSdata(counts=counts, samples=coldata)
|
||||||
|
trs_cts_unfiltered <- counts(d)
|
||||||
|
|
||||||
|
d <- dmFilter(d, min_samps_gene_expr = de_params$min_samps_gene_expr[[1]], min_samps_feature_expr = de_params$min_samps_feature_expr[[1]],
|
||||||
|
min_gene_expr = de_params$min_gene_expr[[1]], min_feature_expr = de_params$min_feature_expr[[1]])
|
||||||
|
|
||||||
|
cat("Building model matrix.\n")
|
||||||
|
design <- model.matrix(~condition, data=DRIMSeq::samples(d))
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
suppressMessages(library("dplyr"))
|
||||||
|
|
||||||
|
# Sum transcript counts into gene counts:
|
||||||
|
cat("Sum transcript counts into gene counts.\n")
|
||||||
|
trs_cts <- counts(d)
|
||||||
|
write.table(trs_cts, file="merged/all_counts_filtered.tsv",sep="\t")
|
||||||
|
|
||||||
|
gene_cts <- trs_cts_unfiltered %>% dplyr::select(c(1, 3:ncol(trs_cts))) %>% group_by(gene_id) %>% summarise_all(tibble::lst(sum)) %>% data.frame()
|
||||||
|
rownames(gene_cts) <- gene_cts$gene_id
|
||||||
|
gene_cts$gene_id <- NULL
|
||||||
|
write.table(gene_cts, file="merged/all_gene_counts.tsv",sep="\t")
|
||||||
|
|
||||||
|
# Differential gene expression using edgeR:
|
||||||
|
suppressMessages(library("edgeR"))
|
||||||
|
cat("Running differential gene expression analysis using edgeR.\n")
|
||||||
|
|
||||||
|
y <- DGEList(gene_cts)
|
||||||
|
y <- calcNormFactors(y)
|
||||||
|
y <- estimateDisp(y,design)
|
||||||
|
fit <- glmQLFit(y,design)
|
||||||
|
qlf <- glmQLFTest(fit)
|
||||||
|
edger_res <- topTags(qlf, n=nrow(y), sort.by="PValue")[[1]]
|
||||||
|
|
||||||
|
pdf("de_analysis/results_dge.pdf")
|
||||||
|
plotMD(qlf)
|
||||||
|
abline(h=c(-1,1), col="blue")
|
||||||
|
plotQLDisp(fit)
|
||||||
|
|
||||||
|
write.table(as.data.frame(edger_res), file="de_analysis/results_dge.tsv", sep="\t")
|
||||||
|
|
||||||
|
# Differential transcript usage using DEXSeq:
|
||||||
|
suppressMessages(library("DEXSeq"))
|
||||||
|
cat("Running differential transcript usage analysis using DEXSeq.\n")
|
||||||
|
|
||||||
|
sample.data<-DRIMSeq::samples(d)
|
||||||
|
count.data <- round(as.matrix(counts(d)[,-c(1:2)]))
|
||||||
|
dxd <- DEXSeqDataSet(countData=count.data, sampleData=sample.data, design=~sample + exon + condition:exon, featureID=trs_cts$feature_id, groupID=trs_cts$gene_id)
|
||||||
|
dxd <- estimateSizeFactors(dxd)
|
||||||
|
dxd <- estimateDispersions(dxd)
|
||||||
|
dxd <- testForDEU(dxd, reducedModel=~sample + exon)
|
||||||
|
dxd <- estimateExonFoldChanges( dxd, fitExpToVar="condition")
|
||||||
|
dxr <- DEXSeqResults(dxd, independentFiltering=FALSE)
|
||||||
|
|
||||||
|
dev.off()
|
||||||
|
pdf("de_analysis/results_dtu.pdf")
|
||||||
|
plotMA(dxr, cex=0.8, alpha=0.05)
|
||||||
|
plotDispEsts(dxd)
|
||||||
|
|
||||||
|
qval <- perGeneQValue(dxr)
|
||||||
|
dxr.g<-data.frame(gene=names(qval), qval)
|
||||||
|
dxr.g <- dxr.g[order(dxr.g$qval),]
|
||||||
|
|
||||||
|
dxr_out <- as.data.frame(dxr[,c("featureID", "groupID", "pvalue")])
|
||||||
|
dxr_out <- dxr_out[order(dxr$pvalue),]
|
||||||
|
|
||||||
|
write.table(dxr.g, file="de_analysis/results_dtu_gene.tsv", sep="\t")
|
||||||
|
write.table(dxr_out, file="de_analysis/results_dtu_transcript.tsv", sep="\t")
|
||||||
|
|
||||||
|
# and writing out some of the DEXSeq metrics to accompany EPI2ME Labs tutorial
|
||||||
|
colnames(dxr)[grep("log2fold", colnames(dxr))] <- "log2fold"
|
||||||
|
MADTUdata <- data.frame(dxr)[order(dxr$padj),c("exonBaseMean", "log2fold", "pvalue", "padj")]
|
||||||
|
MADTUdata$exonBaseMean <- log2(MADTUdata$exonBaseMean)
|
||||||
|
colnames(MADTUdata)[which(colnames(MADTUdata)=="exonBaseMean")] <- "Log2MeanExon"
|
||||||
|
colnames(MADTUdata)[which(colnames(MADTUdata)=="log2fold")] <- "Log2FC"
|
||||||
|
write.table(MADTUdata, file="de_analysis/results_dexseq.tsv", sep="\t")
|
||||||
|
|
||||||
|
# stageR analysis of DEXSeq results:
|
||||||
|
cat("stageR analysis\n")
|
||||||
|
library(stageR)
|
||||||
|
|
||||||
|
cat("Running stageR analysis on the differential transcript usage results.\n")
|
||||||
|
pConfirmation <- matrix(dxr$pvalue, ncol=1)
|
||||||
|
|
||||||
|
dimnames(pConfirmation) <- list(dxr$featureID, "transcript")
|
||||||
|
pScreen <- qval
|
||||||
|
tx2gene <- as.data.frame(dxr[,c("featureID", "groupID")])
|
||||||
|
|
||||||
|
stageRObj <- stageRTx(pScreen=pScreen, pConfirmation=pConfirmation, pScreenAdjusted=TRUE, tx2gene=tx2gene)
|
||||||
|
# note: the choice of 0.05 here means you can *only* threshold at 5% OFDR later
|
||||||
|
stageRObj <- stageWiseAdjustment(stageRObj, method="dtu", alpha=0.10)
|
||||||
|
suppressWarnings({dex.padj <- getAdjustedPValues(stageRObj, order=FALSE, onlySignificantGenes=FALSE)})
|
||||||
|
|
||||||
|
# dex.padj <- dex.padj[,-1]
|
||||||
|
write.table(dex.padj, file="de_analysis/results_dtu_stageR.tsv", sep="\t")
|
||||||
367
bin/de_plots.py
Executable file
367
bin/de_plots.py
Executable file
@ -0,0 +1,367 @@
|
|||||||
|
#!/usr/bin/env python
|
||||||
|
"""Create de report section."""
|
||||||
|
|
||||||
|
from glob import glob
|
||||||
|
import os
|
||||||
|
|
||||||
|
from aplanat import hist, points
|
||||||
|
from aplanat.bars import boxplot_series
|
||||||
|
from aplanat.util import Colors
|
||||||
|
import numpy as np
|
||||||
|
import pandas as pd
|
||||||
|
|
||||||
|
|
||||||
|
def parse_seqkit(fname):
|
||||||
|
"""Get seqkit columns."""
|
||||||
|
cols = {
|
||||||
|
'Read': str, 'Ref': str, 'MapQual': int, 'Acc': float, 'ReadLen': int,
|
||||||
|
'ReadAln': int, 'ReadCov': float, 'MeanQual': float,
|
||||||
|
'IsSec': bool, 'IsSup': bool}
|
||||||
|
df = pd.read_csv(fname, sep="\t", dtype=cols, usecols=cols.keys())
|
||||||
|
df['Clipped'] = df['ReadLen'] - df['ReadAln']
|
||||||
|
df['Type'] = 'Primary'
|
||||||
|
df.loc[df['IsSec'], 'Type'] = 'Secondary'
|
||||||
|
df.loc[df['IsSup'], 'Type'] = 'Supplementary'
|
||||||
|
df["fname"] = os.path.basename(fname).rstrip(".seqkit.stats")
|
||||||
|
return df
|
||||||
|
|
||||||
|
|
||||||
|
def number_of_alignments(df, field_name):
|
||||||
|
"""Group alignments for summary table."""
|
||||||
|
grouped = df.groupby('fname').agg(**{
|
||||||
|
field_name: ('Read', 'size'),
|
||||||
|
})
|
||||||
|
return grouped.transpose()
|
||||||
|
|
||||||
|
|
||||||
|
def create_summary_table(df):
|
||||||
|
"""Create summary table."""
|
||||||
|
all = number_of_alignments(df, "Read mappings")
|
||||||
|
primary = number_of_alignments(df.loc[df['Type'] == 'Primary'], "Primary")
|
||||||
|
secondary = number_of_alignments(
|
||||||
|
df.loc[df['Type'] == 'Secondary'], "Secondary")
|
||||||
|
supplementary = number_of_alignments(
|
||||||
|
df.loc[df['Type'] == 'Supplementary'], "Supplementary")
|
||||||
|
avg_acc = df.loc[df['Type'] == 'Primary'].groupby(
|
||||||
|
'fname').agg(**{"Median Qscore": ('MeanQual', 'median'), }).transpose()
|
||||||
|
avg_mapq = df.loc[df['Type'] == 'Primary'].groupby(
|
||||||
|
'fname').agg(**{"Median MAPQ": ('MapQual', 'median'), }).transpose()
|
||||||
|
return pd.concat([
|
||||||
|
all, primary, secondary, supplementary,
|
||||||
|
avg_acc, avg_mapq])
|
||||||
|
|
||||||
|
|
||||||
|
def dtu_table(gene_id, dtu_file, alignment_stats, condition_sheet):
|
||||||
|
"""Create DTU table and plot."""
|
||||||
|
dtu_results = pd.read_csv(dtu_file, sep='\t')
|
||||||
|
f = open(condition_sheet)
|
||||||
|
df = pd.read_csv(f, sep='\t')
|
||||||
|
treated_df = df.loc[df['condition'] == "treated"]
|
||||||
|
untreated_df = df.loc[df['condition'] == "untreated"]
|
||||||
|
treated_samples = treated_df['sample'].tolist()
|
||||||
|
control_samples = untreated_df['sample'].tolist()
|
||||||
|
# parmaterise these
|
||||||
|
control_name = "condition1"
|
||||||
|
treated_name = "condition2"
|
||||||
|
table = dtu_results.loc[(dtu_results["geneID"] == gene_id)]
|
||||||
|
msg = "Gene ID \"{}\" does not exist in the dataset, please select another"
|
||||||
|
assert not table.empty, msg.format(gene_id)
|
||||||
|
alignment_stats[alignment_stats["Ref"].isin([gene_id])]
|
||||||
|
alignment_stats["Transcript"] = alignment_stats["Ref"].apply(
|
||||||
|
lambda x: x.split(".")[0])
|
||||||
|
gene_alignments = alignment_stats[alignment_stats["Transcript"].isin(
|
||||||
|
table["txID"])]
|
||||||
|
gene_alignments = gene_alignments.loc[(
|
||||||
|
gene_alignments["Type"] == "Primary")]
|
||||||
|
gene_alignments["condition"] = gene_alignments.apply(
|
||||||
|
lambda x: control_name if x["fname"] in
|
||||||
|
control_samples else treated_name, axis=1)
|
||||||
|
groups = gene_alignments.groupby(
|
||||||
|
["Transcript", "fname", "condition"]).agg(
|
||||||
|
**{"Transcript_count": ("Read", "size")})
|
||||||
|
df = None
|
||||||
|
for gene_id, group in gene_alignments.groupby("Transcript"):
|
||||||
|
temp = group.groupby("fname").agg(**{
|
||||||
|
gene_id: ("Read", "size")
|
||||||
|
}).transpose()
|
||||||
|
if df is None:
|
||||||
|
df = temp
|
||||||
|
else:
|
||||||
|
df = pd.concat([df, temp])
|
||||||
|
df = df.reset_index().rename(columns={"index": "Transcript_ID"})
|
||||||
|
table = table.rename(columns={
|
||||||
|
"txID": "Transcript_ID",
|
||||||
|
"gene": "p_gene",
|
||||||
|
"transcript": "p_transcript"
|
||||||
|
})
|
||||||
|
gene_table = pd.merge(df, table)
|
||||||
|
gene_table = gene_table.set_index(["geneID", "Transcript_ID"])
|
||||||
|
file_names = set(control_samples.keys())
|
||||||
|
file_names.update(treated_samples.keys())
|
||||||
|
gene_table = gene_table.fillna(0)
|
||||||
|
table = groups.reset_index()[
|
||||||
|
["condition", "Transcript", "Transcript_count"]]
|
||||||
|
table['transcript, condition'] = \
|
||||||
|
table['Transcript'].astype(str) + ', ' + table['condition'].astype(str)
|
||||||
|
repeats = groups.groupby(level=['Transcript', 'condition']).size()
|
||||||
|
min_rep, max_rep = min(repeats), max(repeats)
|
||||||
|
plot = boxplot_series(
|
||||||
|
table['transcript, condition'], table['Transcript_count'],
|
||||||
|
x_axis_label='Transcript, condition',
|
||||||
|
y_axis_label='Transcript count',
|
||||||
|
height=200, width=200,
|
||||||
|
title="Transcript counts (from {}-{} replicates)".format(
|
||||||
|
min_rep, max_rep))
|
||||||
|
plot.xaxis.major_label_orientation = 3.1452/2
|
||||||
|
if min_rep < 7:
|
||||||
|
for renderer in plot.renderers:
|
||||||
|
renderer.glyph.line_alpha = 0.2
|
||||||
|
try:
|
||||||
|
renderer.glyph.fill_alpha = 0.2
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
plot.circle(
|
||||||
|
table['transcript, condition'], table['Transcript_count'],
|
||||||
|
fill_color='black', line_color='black')
|
||||||
|
return (table, plot)
|
||||||
|
|
||||||
|
|
||||||
|
def pool_csvs(folder):
|
||||||
|
"""Concat seqkit stats."""
|
||||||
|
files = glob(folder + "/*.seqkit.stats")
|
||||||
|
dfs = [parse_seqkit(f) for f in files]
|
||||||
|
return pd.concat(dfs)
|
||||||
|
|
||||||
|
|
||||||
|
def abundance_histogram(filtered_counts, gene_counts, section):
|
||||||
|
"""Create plot for abundance of transcripts across all samples."""
|
||||||
|
section.markdown("""
|
||||||
|
Histogram showing the abundance of transcript
|
||||||
|
counts for genes identified in the analysis.
|
||||||
|
""")
|
||||||
|
filtered_count_file = filtered_counts
|
||||||
|
gene_count_file = gene_counts
|
||||||
|
transcripts_per_gene = pd.read_csv(
|
||||||
|
filtered_count_file, sep='\t',
|
||||||
|
usecols=['gene_id', 'feature_id']).groupby(['gene_id']).agg(['count'])
|
||||||
|
transcripts_per_gene.columns = transcripts_per_gene.columns.droplevel()
|
||||||
|
gene_ids = pd.read_csv(
|
||||||
|
gene_count_file, sep='\t',
|
||||||
|
usecols=[0]).index.values.tolist()
|
||||||
|
singletons = [
|
||||||
|
gene_id for gene_id in gene_ids if gene_id not in transcripts_per_gene.index.values.tolist()] # noqa
|
||||||
|
singletons = pd.DataFrame(index=singletons, columns=['count']).fillna(1)
|
||||||
|
transcripts_per_gene = pd.concat([singletons, transcripts_per_gene])
|
||||||
|
transcript_plot = hist.histogram(
|
||||||
|
[transcripts_per_gene['count'].tolist()],
|
||||||
|
binwidth=1, colors=[Colors.cerulean])
|
||||||
|
transcript_plot.xaxis.axis_label = "Number of isoforms per gene (n)"
|
||||||
|
transcript_plot.yaxis.axis_label = "Number of occurences"
|
||||||
|
section.markdown("### Transcripts per gene")
|
||||||
|
section.plot(transcript_plot)
|
||||||
|
|
||||||
|
|
||||||
|
def dexseq_section(dexseq_file, section, id_dic):
|
||||||
|
"""Add gene isoforms table and plot."""
|
||||||
|
section.markdown("### Differential Isoform usage")
|
||||||
|
dexseq_caption = '''Table showing gene isoforms, ranked by adjusted
|
||||||
|
p-value, from the DEXSeq analysis. Information shown includes the log2 fold
|
||||||
|
change between experimental conditions, the log-scaled transcript
|
||||||
|
abundance and the false discovery corrected p-value (FDR).
|
||||||
|
This table has not been filtered
|
||||||
|
for genes that satisfy statistical or magnitudinal thresholds'''
|
||||||
|
section.markdown(dexseq_caption)
|
||||||
|
dexseq_results = pd.read_csv(dexseq_file, sep='\t')
|
||||||
|
dexseq_results.index.name = "gene_id:trancript_id"
|
||||||
|
dexseq_results.index = dexseq_results.index.map(
|
||||||
|
lambda x: id_dic[x.split(':')[0]] + ':' + str(x.split(':')[1]))
|
||||||
|
dexseq_pvals = dexseq_results.sort_values(by='pvalue', ascending=True)
|
||||||
|
section.table(dexseq_results.loc[dexseq_pvals.index], index=True)
|
||||||
|
section.markdown("""
|
||||||
|
The figure below presents the MA plot from the DEXSeq analysis.
|
||||||
|
M is the log2 ratio of isoform transcript abundance between conditions.
|
||||||
|
A is the log2 transformed mean abundance value.
|
||||||
|
Transcripts that satisfy the logFC and FDR corrected p-value
|
||||||
|
thresholds defined are shaded as 'Up-' or 'Down-' regulated.""")
|
||||||
|
pval_limit = 0.01
|
||||||
|
up = dexseq_results.loc[
|
||||||
|
(dexseq_results["Log2FC"] > 0) & (
|
||||||
|
dexseq_results['pvalue'] < pval_limit)]
|
||||||
|
down = dexseq_results.loc[
|
||||||
|
(dexseq_results["Log2FC"] <= 0) & (
|
||||||
|
dexseq_results['pvalue'] < pval_limit)]
|
||||||
|
not_sig = dexseq_results.loc[(dexseq_results["pvalue"] >= pval_limit)]
|
||||||
|
|
||||||
|
dexseq_plot = points.points(
|
||||||
|
x_datas=[
|
||||||
|
up["Log2MeanExon"],
|
||||||
|
down["Log2MeanExon"],
|
||||||
|
not_sig["Log2MeanExon"],
|
||||||
|
],
|
||||||
|
y_datas=[
|
||||||
|
up["Log2FC"],
|
||||||
|
down["Log2FC"],
|
||||||
|
not_sig["Log2FC"],
|
||||||
|
],
|
||||||
|
title="Average copy per million (CPM) vs Log-fold change (LFC)",
|
||||||
|
colors=["red", "blue", "black"],
|
||||||
|
xlim=[0, 5],
|
||||||
|
names=["Up", "Down", "NotSig"]
|
||||||
|
)
|
||||||
|
dexseq_plot.xaxis.axis_label = "A (log2 transformed mean exon read counts)"
|
||||||
|
dexseq_plot.yaxis.axis_label = """
|
||||||
|
M (log2 transformed differential abundance)
|
||||||
|
"""
|
||||||
|
dexseq_results_caption = "### Dexseq results"
|
||||||
|
section.markdown(dexseq_results_caption)
|
||||||
|
section.plot(dexseq_plot)
|
||||||
|
|
||||||
|
|
||||||
|
def dtu_section(dtu_file, section, gt_dic, ge_dic):
|
||||||
|
"""Plot dtu section."""
|
||||||
|
dtu_results = pd.read_csv(dtu_file, sep='\t')
|
||||||
|
dtu_results["gene_name"] = dtu_results["txID"].apply(lambda x: gt_dic[x])
|
||||||
|
dtu_results["geneID"] = dtu_results["geneID"].apply(lambda x: ge_dic[x])
|
||||||
|
dtu_pvals = dtu_results.sort_values(by='gene', ascending=True)
|
||||||
|
dtu_caption = '''Table showing gene and transcript identifiers
|
||||||
|
and their FDR corrected probabilities
|
||||||
|
for the genes and their isoforms that have been
|
||||||
|
identified as showing DTU using the R packages DEXSeq and StageR.
|
||||||
|
This list has been shortened requiring that both gene and transcript
|
||||||
|
must satisfy the p-value
|
||||||
|
threshold'''
|
||||||
|
section.markdown(dtu_caption)
|
||||||
|
section.table(dtu_results.loc[dtu_pvals.index])
|
||||||
|
|
||||||
|
|
||||||
|
def dge_section(dge_file, section, ids_dic):
|
||||||
|
"""Create DGE table and plot."""
|
||||||
|
section.markdown('### Differential gene expression')
|
||||||
|
dge_results = pd.read_csv(dge_file, sep='\t')
|
||||||
|
dge_pvals = dge_results.sort_values(by='FDR', ascending=True)
|
||||||
|
dge_results[['logFC', 'logCPM', 'F']] = dge_results[
|
||||||
|
['logFC', 'logCPM', 'F']].round(2)
|
||||||
|
dge_caption = """
|
||||||
|
Table showing the genes from the edgeR analysis.
|
||||||
|
Information shown includes the log2 fold change between
|
||||||
|
experimental conditions, the log-scaled counts per million measure of abundance
|
||||||
|
and the false discovery corrected p-value (FDR). This table has not been
|
||||||
|
filtered for genes that satisfy statistical or magnitudinal thresholds"""
|
||||||
|
section.markdown(dge_caption)
|
||||||
|
dge_results.index = dge_results.index.map(lambda x: ids_dic[x])
|
||||||
|
dge_pvals.index = dge_pvals.index.map(lambda x: ids_dic[x])
|
||||||
|
section.table(dge_results.loc[dge_pvals.index], index=True)
|
||||||
|
dge = pd.read_csv(dge_file, sep="\t")
|
||||||
|
section.markdown("""
|
||||||
|
This plot visualises differences in measurements between the
|
||||||
|
two experimental conditions. M is the log2 ratio of gene expression
|
||||||
|
calculated between the conditions.
|
||||||
|
A is a log2 transformed mean expression value.
|
||||||
|
The figure below presents the MA figure from this edgeR analysis.
|
||||||
|
Genes that satisfy the logFC and FDR corrected p-value thresholds
|
||||||
|
defined are shaded as 'Up-' or 'Down-' regulated.
|
||||||
|
""")
|
||||||
|
pval_limit = 0.01
|
||||||
|
up = dge.loc[(dge["logFC"] > 0) & (dge['PValue'] < pval_limit)]
|
||||||
|
down = dge.loc[(dge["logFC"] <= 0) & (dge['PValue'] < pval_limit)]
|
||||||
|
not_sig = dge.loc[(dge["PValue"] >= pval_limit)]
|
||||||
|
logcpm_vs_logfc = points.points(
|
||||||
|
x_datas=[
|
||||||
|
up["logCPM"],
|
||||||
|
down["logCPM"],
|
||||||
|
not_sig["logCPM"],
|
||||||
|
],
|
||||||
|
y_datas=[
|
||||||
|
up["logFC"],
|
||||||
|
down["logFC"],
|
||||||
|
not_sig["logFC"],
|
||||||
|
],
|
||||||
|
title="Average copy per million (CPM) vs Log-fold change (LFC)",
|
||||||
|
colors=["blue", "red", "black"],
|
||||||
|
names=["Up", "Down", "NotSig"]
|
||||||
|
)
|
||||||
|
|
||||||
|
logcpm_vs_logfc.xaxis.axis_label = "Average log CPM"
|
||||||
|
logcpm_vs_logfc.yaxis.axis_label = "Log-fold change"
|
||||||
|
logcpm_caption = """### Results of the edgeR Analysis."""
|
||||||
|
section.markdown(logcpm_caption)
|
||||||
|
section.plot(logcpm_vs_logfc)
|
||||||
|
|
||||||
|
|
||||||
|
def salmon_table(salmon_counts, section):
|
||||||
|
"""Create salmon counts summary table."""
|
||||||
|
salmon_counts = pd.read_csv(salmon_counts, sep='\t')
|
||||||
|
salmon_counts.set_index("Reference", drop=True, append=False, inplace=True)
|
||||||
|
salmon_size_top = salmon_counts.sum(axis=1).sort_values(ascending=False)
|
||||||
|
salmon_counts = salmon_counts.applymap(np.int64)
|
||||||
|
salmon_count_caption = """
|
||||||
|
Table showing the annotated Transcripts Per Million
|
||||||
|
identified by Minimap2 mapping and Salmon transcript
|
||||||
|
detection with the highest
|
||||||
|
number of mapped reads"""
|
||||||
|
section.markdown("### Transcripts Per Million ")
|
||||||
|
section.markdown(salmon_count_caption, "salmon-head-caption")
|
||||||
|
section.table(
|
||||||
|
salmon_counts.loc[salmon_size_top.index].head(n=100), index=True)
|
||||||
|
|
||||||
|
|
||||||
|
def get_translations(gtf):
|
||||||
|
"""Create dic with gene_name and gene_references."""
|
||||||
|
fn = open(gtf).readlines()
|
||||||
|
gene_txid = {}
|
||||||
|
gene_geid = {}
|
||||||
|
for i in fn:
|
||||||
|
if i.startswith("#"):
|
||||||
|
continue
|
||||||
|
try:
|
||||||
|
gene_name = i.split("gene_name")[1].split(";")[0]
|
||||||
|
except Exception:
|
||||||
|
gene_name = i.split("gene_id")[1].split(";")[0]
|
||||||
|
try:
|
||||||
|
gene_reference = i.split("ref_gene_id")[1].split(";")[0]
|
||||||
|
except Exception:
|
||||||
|
gene_reference = i.split("gene_id")[1].split(";")[0]
|
||||||
|
try:
|
||||||
|
transcript_id = i.split("transcript_id")[1].split(";")[0]
|
||||||
|
except Exception:
|
||||||
|
transcript_id = "unknown"
|
||||||
|
transcript_id = transcript_id.replace("\"", "").strip()
|
||||||
|
gene_id = i.split("gene_id")[1].split(";")[0]
|
||||||
|
gene_id = gene_id.replace("\"", "").strip()
|
||||||
|
gene_name = gene_name.replace("\"", "").strip()
|
||||||
|
gene_reference = gene_reference.replace("\"", "").strip()
|
||||||
|
gene_txid[transcript_id] = gene_name
|
||||||
|
gene_geid[gene_id] = gene_reference
|
||||||
|
return gene_txid, gene_geid
|
||||||
|
|
||||||
|
|
||||||
|
def de_section(
|
||||||
|
stringtie, dge, dexseq, dtu,
|
||||||
|
tpm, report):
|
||||||
|
"""Differential expression sections."""
|
||||||
|
section = report.add_section()
|
||||||
|
"""Add Differential expression section."""
|
||||||
|
section.markdown("# Differential expression.")
|
||||||
|
section.markdown("""
|
||||||
|
This section shows differential gene expression
|
||||||
|
and differential isoform usage. Salmon was used to
|
||||||
|
assign reads to individual annotated isoforms defined by
|
||||||
|
the GTF-format annotation.
|
||||||
|
These counts were used to perform a statistical analysis to identify
|
||||||
|
the genes and isoforms that show differences in abundance between
|
||||||
|
the experimental conditions.
|
||||||
|
""")
|
||||||
|
section.markdown("### Alignment summary stats")
|
||||||
|
alignment_stats = pool_csvs("seqkit")
|
||||||
|
alignment_summary_df = create_summary_table(alignment_stats)
|
||||||
|
alignment_summary_df = alignment_summary_df.fillna(0).applymap(np.int64)
|
||||||
|
section.table(alignment_summary_df, key='alignment-stats', index=True)
|
||||||
|
salmon_table(tpm, section)
|
||||||
|
gene_txid, gene_name = get_translations(stringtie)
|
||||||
|
dge_section(dge, section, gene_name)
|
||||||
|
dexseq_section(dexseq, section, gene_name)
|
||||||
|
dtu_section(dtu, section, gene_txid, gene_name)
|
||||||
|
# missing dtu plots at the moment as too many
|
||||||
|
section.markdown("""
|
||||||
|
### View dtu_plots.pdf file to see plots of differential isoform usage
|
||||||
|
""")
|
||||||
58
bin/merge_count_tsvs.py
Executable file
58
bin/merge_count_tsvs.py
Executable file
@ -0,0 +1,58 @@
|
|||||||
|
#!/usr/bin/env python
|
||||||
|
"""Merge salmon output count files."""
|
||||||
|
|
||||||
|
import argparse
|
||||||
|
from functools import reduce
|
||||||
|
|
||||||
|
import numpy as np
|
||||||
|
import pandas as pd
|
||||||
|
|
||||||
|
# Parse command line arguments:
|
||||||
|
parser = argparse.ArgumentParser(
|
||||||
|
description="""Merge tab separated files on a given field using pandas.""")
|
||||||
|
parser.add_argument(
|
||||||
|
'-j', metavar='join', type=str, help="Join type (outer).", default="outer")
|
||||||
|
parser.add_argument(
|
||||||
|
'-f', metavar='field', type=str,
|
||||||
|
help="Join on this field (Reference).", default="Reference")
|
||||||
|
parser.add_argument(
|
||||||
|
'-o', metavar='out_tsv', type=str,
|
||||||
|
help="Output tsv (merge_tsvs.tsv).", default="merge_tsvs.tsv")
|
||||||
|
parser.add_argument(
|
||||||
|
'-z', action="store_true",
|
||||||
|
help="Fill NA values with zero.", default=False)
|
||||||
|
parser.add_argument(
|
||||||
|
'tsvs', metavar='input_tsvs', nargs='*',
|
||||||
|
type=str, help="Input tab separated files.")
|
||||||
|
parser.add_argument(
|
||||||
|
'-tpm', type=bool, nargs='*', default=False,
|
||||||
|
help="TPM instead of counts")
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == '__main__':
|
||||||
|
args = parser.parse_args()
|
||||||
|
|
||||||
|
dfs = {x: pd.read_csv(x, sep="\t") for x in args.tsvs}
|
||||||
|
|
||||||
|
ndfs = []
|
||||||
|
for x, df in dfs.items():
|
||||||
|
# Transform counts to integers:
|
||||||
|
df = df.rename(columns={'NumReads': 'Count', 'Name': 'Reference'})
|
||||||
|
if args.tpm:
|
||||||
|
df = df.rename(columns={'TPM': 'Count', 'Name': 'Reference'})
|
||||||
|
df.Count = np.array(df.Count, dtype=int)
|
||||||
|
# Take only non-zero counts:
|
||||||
|
df = df[df.Count > 0]
|
||||||
|
df = df[["Reference", "Count"]]
|
||||||
|
df = df.sort_values(by=["Count"], ascending=False)
|
||||||
|
name = x.split('.')[0]
|
||||||
|
df = df.rename(columns={'Count': name})
|
||||||
|
ndfs.append(df)
|
||||||
|
dfs = ndfs
|
||||||
|
|
||||||
|
df_merged = reduce(lambda left, right: pd.merge(
|
||||||
|
left, right, on=args.f, how=args.j), dfs)
|
||||||
|
if args.z:
|
||||||
|
df_merged = df_merged.fillna(0)
|
||||||
|
|
||||||
|
df_merged.to_csv(args.o, sep="\t", index=False)
|
||||||
57
bin/plot_dtu_results.R
Executable file
57
bin/plot_dtu_results.R
Executable file
@ -0,0 +1,57 @@
|
|||||||
|
#!/usr/bin/env Rscript
|
||||||
|
|
||||||
|
suppressMessages(library(dplyr))
|
||||||
|
suppressMessages(library(ggplot2))
|
||||||
|
suppressMessages(library(tidyr))
|
||||||
|
|
||||||
|
# Set up sample data frame:
|
||||||
|
coldata <- read.csv("de_analysis/coldata.tsv", row.names="sample", sep=",")
|
||||||
|
coldata$sample_id <- rownames(coldata)
|
||||||
|
coldata$condition <- factor(coldata$condition, levels=rev(levels(coldata$condition)))
|
||||||
|
coldata$type <-NULL
|
||||||
|
coldata$patient <-NULL
|
||||||
|
|
||||||
|
# Read stageR results:
|
||||||
|
stageR <- read.csv("de_analysis/results_dtu_stageR.tsv", sep="\t")
|
||||||
|
names(stageR) <- c("gene_id", "transcript_id", "p_gene", "p_transcript");
|
||||||
|
|
||||||
|
# Read filtered counts:
|
||||||
|
counts <- read.csv("merged/all_counts_filtered.tsv", sep="\t");
|
||||||
|
names(counts)[2]<-"transcript_id"
|
||||||
|
|
||||||
|
# Join counts and stageR results:
|
||||||
|
df <- counts %>% left_join(stageR, by = c("gene_id", "transcript_id"))
|
||||||
|
df <- df[order(df$p_gene),]
|
||||||
|
|
||||||
|
scols <- setdiff(names(df),c("gene_id", "transcript_id", "p_gene", "p_transcript"))
|
||||||
|
|
||||||
|
# Normalise counts:
|
||||||
|
for(sc in scols){
|
||||||
|
df[sc] <- df[sc] / sum(df[sc])
|
||||||
|
}
|
||||||
|
|
||||||
|
# Melt data frame:
|
||||||
|
tdf <- df %>% gather(key='sample', value='norm_count',-gene_id, -transcript_id, -p_gene, -p_transcript)
|
||||||
|
|
||||||
|
# Add sample group column:
|
||||||
|
sampleToGroup<-function(x){
|
||||||
|
return(coldata[x,]$condition)
|
||||||
|
}
|
||||||
|
|
||||||
|
tdf$group <- sampleToGroup(tdf$sample)
|
||||||
|
|
||||||
|
# Filter for significant genes:
|
||||||
|
sig_level <- 0.05
|
||||||
|
genes <- as.character(tdf[which(tdf$p_gene < sig_level),]$gene_id)
|
||||||
|
genes <- unique(genes)
|
||||||
|
|
||||||
|
pdf("de_analysis/dtu_plots.pdf")
|
||||||
|
|
||||||
|
for(gene in genes){
|
||||||
|
gdf<-tdf[which(tdf$gene_id==gene),]
|
||||||
|
p_gene <- unique(gdf$p_gene)
|
||||||
|
p <- ggplot(gdf, aes(x=transcript_id, y=norm_count)) + geom_bar(stat="identity", aes(fill=sample), position="dodge")
|
||||||
|
p <- p + facet_wrap(~ group) + coord_flip()
|
||||||
|
p <- p + ggtitle(paste(gene," : p_value=",p_gene,sep=""))
|
||||||
|
print(p)
|
||||||
|
}
|
||||||
@ -4,6 +4,7 @@
|
|||||||
import argparse
|
import argparse
|
||||||
from collections import Counter, defaultdict, OrderedDict
|
from collections import Counter, defaultdict, OrderedDict
|
||||||
import math
|
import math
|
||||||
|
import os
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
|
|
||||||
from aplanat import bars, hist
|
from aplanat import bars, hist
|
||||||
@ -17,6 +18,7 @@ from bokeh.models.widgets import DataTable, TableColumn
|
|||||||
from bokeh.palettes import Category10_10
|
from bokeh.palettes import Category10_10
|
||||||
from bokeh.plotting import figure
|
from bokeh.plotting import figure
|
||||||
from bokeh.transform import dodge
|
from bokeh.transform import dodge
|
||||||
|
import de_plots
|
||||||
import gffutils
|
import gffutils
|
||||||
import numpy as np
|
import numpy as np
|
||||||
import pandas as pd
|
import pandas as pd
|
||||||
@ -816,6 +818,22 @@ def jaffal_table(report, result_csv):
|
|||||||
section.table(df)
|
section.table(df)
|
||||||
|
|
||||||
|
|
||||||
|
def de_section(report):
|
||||||
|
"""Make differential transcript expression section."""
|
||||||
|
dexseq = os.path.join("de_report", "results_dexseq.tsv")
|
||||||
|
dge = os.path.join("de_report", "results_dge.tsv")
|
||||||
|
dtu = os.path.join("de_report", "results_dtu_stageR.tsv")
|
||||||
|
stringtie = os.path.join("de_report", "stringtie_merged.gtf")
|
||||||
|
tpm = os.path.join("de_report", "tpm_counts.tsv")
|
||||||
|
de_plots.de_section(
|
||||||
|
stringtie=stringtie,
|
||||||
|
dexseq=dexseq,
|
||||||
|
dge=dge,
|
||||||
|
dtu=dtu,
|
||||||
|
tpm=tpm,
|
||||||
|
report=report)
|
||||||
|
|
||||||
|
|
||||||
def main():
|
def main():
|
||||||
"""Run the entry point."""
|
"""Run the entry point."""
|
||||||
parser = argparse.ArgumentParser()
|
parser = argparse.ArgumentParser()
|
||||||
@ -857,6 +875,12 @@ def main():
|
|||||||
parser.add_argument(
|
parser.add_argument(
|
||||||
"--jaffal_csv", required=False, type=str, default=None,
|
"--jaffal_csv", required=False, type=str, default=None,
|
||||||
help="Path to JAFFAL results csv")
|
help="Path to JAFFAL results csv")
|
||||||
|
parser.add_argument(
|
||||||
|
"--de_report", required=False, type=str, default=None,
|
||||||
|
help="Differential expression report optional")
|
||||||
|
parser.add_argument(
|
||||||
|
"--de_stats", required=False, type=str, default=None, nargs='*',
|
||||||
|
help="Differential expression report optional")
|
||||||
parser.add_argument('--denovo', dest='denovo', action='store_true')
|
parser.add_argument('--denovo', dest='denovo', action='store_true')
|
||||||
|
|
||||||
args = parser.parse_args()
|
args = parser.parse_args()
|
||||||
@ -902,6 +926,9 @@ def main():
|
|||||||
if args.cluster_qc_dirs is not None:
|
if args.cluster_qc_dirs is not None:
|
||||||
cluster_quality(args.cluster_qc_dirs, report, sample_ids)
|
cluster_quality(args.cluster_qc_dirs, report, sample_ids)
|
||||||
|
|
||||||
|
if args.de_report:
|
||||||
|
de_section(report)
|
||||||
|
|
||||||
if args.jaffal_csv:
|
if args.jaffal_csv:
|
||||||
jaffal_table(report, args.jaffal_csv)
|
jaffal_table(report, args.jaffal_csv)
|
||||||
|
|
||||||
|
|||||||
@ -32,9 +32,17 @@ using [gffcompare](http://ccb.jhu.edu/software/stringtie/gffcompare.shtml)
|
|||||||
Fusion gene detection is performed using [JAFFA](https://github.com/Oshlack/JAFFA), with the JAFFAL extension for use
|
Fusion gene detection is performed using [JAFFA](https://github.com/Oshlack/JAFFA), with the JAFFAL extension for use
|
||||||
with ONT long reads.
|
with ONT long reads.
|
||||||
|
|
||||||
|
### Differential expression analysis
|
||||||
|
* Differential expression is done using the transcripts output by the workflow.
|
||||||
|
* A non redundant transcriptome is found using the merge function in [stringtie](http://ccb.jhu.edu/software/stringtie).
|
||||||
|
* The reads are then aligned to the transcriptome using minimap2 in a splice-aware manner.
|
||||||
|
* [salmon](https://github.com/COMBINE-lab/salmon) is used for transcript quantification.
|
||||||
|
* R packages [edgeR](https://bioconductor.org/packages/release/bioc/html/edgeR.html) and [stageR](https://bioconductor.org/packages/release/bioc/html/stageR.html) are used for differential expression analysis.
|
||||||
|
* [DEXSeq](https://bioconductor.org/packages/release/bioc/html/DEXSeq.html) is then used for differential transcript usage analysis.
|
||||||
|
|
||||||
### Workflow inputs
|
### Workflow inputs
|
||||||
- Directory containing cDNA/direct RNA reads. Or a directory containing subdirectories each with reads from different samples
|
- Directory containing cDNA/direct RNA reads. Or a directory containing subdirectories each with reads from different samples
|
||||||
(in fastq/fastq.gz format)
|
(in fastq/fastq.gz format)
|
||||||
- Reference genome in fasta format (required for reference-based assembly).
|
- Reference genome in fasta format (required for reference-based assembly).
|
||||||
- Optional reference annotation in GFF2/3 format.
|
- Optional reference annotation in GFF2/3 format (required for differential expression analysis `--de_analysis`).
|
||||||
- For fusion detection, JAFFAL reference files (see Quickstart)
|
- For fusion detection, JAFFAL reference files (see Quickstart)
|
||||||
|
|||||||
@ -116,6 +116,25 @@ g++ must be installed. JAFFAL is not currently working on Mac M1 (osx-arm64 arch
|
|||||||
detected, the workflow will terminate with an error at the JAFFAL stage. If this happens,
|
detected, the workflow will terminate with an error at the JAFFAL stage. If this happens,
|
||||||
skip the JAFFAL stage by omitting ` --jaffal_refBase`
|
skip the JAFFAL stage by omitting ` --jaffal_refBase`
|
||||||
|
|
||||||
|
### Differential Expression
|
||||||
|
|
||||||
|
Differential Expression requires at least 2 replicates of each sample to compare. You can see an example condition_sheet.tsv in test_data.
|
||||||
|
|
||||||
|
**Example workflow for differential expression transcript assembly**
|
||||||
|
|
||||||
|
Download differential expression data set
|
||||||
|
|
||||||
|
`wget -O differential_expression.tar.gz https://ont-exd-int-s3-euwst1-epi2me-labs.s3.amazonaws.com/wf-isoforms/wf-isoforms_differential_expression.tar.gz && tar -xzvf differential_expression.tar.gz`
|
||||||
|
|
||||||
|
Run the cmd
|
||||||
|
|
||||||
|
```
|
||||||
|
OUTPUT=~/output;
|
||||||
|
nexflow run epi2me-labs/wf-transcriptomes --fastq differential_expression_dataset/fastq --de_analysis \
|
||||||
|
--ref_genome differential_expression_dataset/hg38_chr20.fa \
|
||||||
|
--ref_annotation differential_expression_dataset/gencode.v22.annotation.chr20.gtf \
|
||||||
|
--direct_rna
|
||||||
|
```
|
||||||
|
|
||||||
## Workflow outputs
|
## Workflow outputs
|
||||||
* an HTML report document detailing the primary findings of the workflow.
|
* an HTML report document detailing the primary findings of the workflow.
|
||||||
@ -131,3 +150,6 @@ skip the JAFFAL stage by omitting ` --jaffal_refBase`
|
|||||||
in `${out_dir}/jaffal_output_${sample_id}` you will find:
|
in `${out_dir}/jaffal_output_${sample_id}` you will find:
|
||||||
* jaffa_results.csv - the csv results summary file
|
* jaffa_results.csv - the csv results summary file
|
||||||
* jaffa_results.fasta - fusion transcritpt sequences
|
* jaffa_results.fasta - fusion transcritpt sequences
|
||||||
|
|
||||||
|
### Differential Expression outputs
|
||||||
|
* dtu_plots.pdf - a pdf with differntial transcript usage plots
|
||||||
|
|||||||
@ -16,7 +16,7 @@ dependencies:
|
|||||||
- gffread==0.12.7
|
- gffread==0.12.7
|
||||||
- gffcompare==0.11.2
|
- gffcompare==0.11.2
|
||||||
- gffutils==0.10.1
|
- gffutils==0.10.1
|
||||||
- seqkit==2.1.0
|
- seqkit==2.2.0
|
||||||
- stringtie==2.1.1
|
- stringtie==2.1.1
|
||||||
- curl
|
- curl
|
||||||
- pysam==0.17.0
|
- pysam==0.17.0
|
||||||
@ -31,4 +31,13 @@ dependencies:
|
|||||||
# - bpipe=0.9.9.2
|
# - bpipe=0.9.9.2
|
||||||
- java-jdk
|
- java-jdk
|
||||||
- r-base
|
- r-base
|
||||||
- gxx
|
- gxx
|
||||||
|
# Differential expression dependencies
|
||||||
|
- bioconductor-genomicfeatures
|
||||||
|
- bioconductor-drimseq
|
||||||
|
- bioconductor-edger
|
||||||
|
- bioconductor-dexseq
|
||||||
|
- bioconductor-stager
|
||||||
|
- r-dplyr
|
||||||
|
- r-tidyr
|
||||||
|
- salmon==1.9.0
|
||||||
@ -301,7 +301,12 @@ def barcode_in_range(path, min_barcode, max_barcode)
|
|||||||
{
|
{
|
||||||
pattern = ~/barcode(\d+)/
|
pattern = ~/barcode(\d+)/
|
||||||
matcher = "${path}" =~ pattern
|
matcher = "${path}" =~ pattern
|
||||||
value = matcher[0][1].toInteger()
|
def value = null
|
||||||
|
try{
|
||||||
|
value = matcher[0][1].toInteger()
|
||||||
|
}catch(ArrayIndexOutOfBoundsException ex){
|
||||||
|
print("${path} is not a barcoded directory")
|
||||||
|
}
|
||||||
valid = ((value >= min_barcode) && (value <= max_barcode))
|
valid = ((value >= min_barcode) && (value <= max_barcode))
|
||||||
return valid
|
return valid
|
||||||
}
|
}
|
||||||
|
|||||||
133
main.nf
133
main.nf
@ -15,6 +15,8 @@ include { start_ping; end_ping } from './lib/ping'
|
|||||||
include { reference_assembly } from './subworkflows/reference_assembly'
|
include { reference_assembly } from './subworkflows/reference_assembly'
|
||||||
include { denovo_assembly } from './subworkflows/denovo_assembly'
|
include { denovo_assembly } from './subworkflows/denovo_assembly'
|
||||||
include { gene_fusions } from './subworkflows/JAFFAL/gene_fusions'
|
include { gene_fusions } from './subworkflows/JAFFAL/gene_fusions'
|
||||||
|
include { differential_expression } from './subworkflows/differential_expression'
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
process summariseConcatReads {
|
process summariseConcatReads {
|
||||||
@ -181,7 +183,7 @@ process assemble_transcripts{
|
|||||||
def prefix = bam.name.split(/\./)[0]
|
def prefix = bam.name.split(/\./)[0]
|
||||||
|
|
||||||
"""
|
"""
|
||||||
stringtie --rf ${G_FLAG} -L -v -p ${params.threads} ${params.stringtie_opts} \
|
stringtie --rf ${G_FLAG} -L -v -p ${task.cpus} ${params.stringtie_opts} \
|
||||||
-o ${prefix}.gff -l ${prefix} ${bam} 2>/dev/null
|
-o ${prefix}.gff -l ${prefix} ${bam} 2>/dev/null
|
||||||
"""
|
"""
|
||||||
}
|
}
|
||||||
@ -225,6 +227,7 @@ process run_gffcompare{
|
|||||||
path ref_annotation
|
path ref_annotation
|
||||||
output:
|
output:
|
||||||
tuple val(sample_id), path("${sample_id}_gffcompare"), emit: gffcmp_dir
|
tuple val(sample_id), path("${sample_id}_gffcompare"), emit: gffcmp_dir
|
||||||
|
path ("${sample_id}_annotated.gtf"), emit: gtf, optional: true
|
||||||
script:
|
script:
|
||||||
def out_dir = "${sample_id}_gffcompare"
|
def out_dir = "${sample_id}_gffcompare"
|
||||||
|
|
||||||
@ -245,6 +248,7 @@ process run_gffcompare{
|
|||||||
|
|
||||||
mv *.tmap $out_dir
|
mv *.tmap $out_dir
|
||||||
mv *.refmap $out_dir
|
mv *.refmap $out_dir
|
||||||
|
cp ${out_dir}/str_merged.annotated.gtf ${sample_id}_annotated.gtf
|
||||||
"""
|
"""
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@ -273,6 +277,29 @@ process get_transcriptome{
|
|||||||
"""
|
"""
|
||||||
}
|
}
|
||||||
|
|
||||||
|
process merge_transcriptomes {
|
||||||
|
// Merge the transcriptomes from all samples
|
||||||
|
label 'isoforms'
|
||||||
|
input:
|
||||||
|
path "query_annotations/*"
|
||||||
|
path ref_annotation
|
||||||
|
path ref_genome
|
||||||
|
output:
|
||||||
|
path "non_redundant.fasta", emit: fasta
|
||||||
|
path "stringtie.gtf", emit: gtf
|
||||||
|
"""
|
||||||
|
stringtie --merge -G $ref_annotation -p ${task.cpus} -o stringtie.gtf query_annotations/*
|
||||||
|
seqkit subseq --feature "transcript" --gtf-tag "transcript_id" --gtf stringtie.gtf $ref_genome > temp_transcriptome.fasta
|
||||||
|
seqkit rmdup -s < temp_transcriptome.fasta > temp_del_repeats.fasta
|
||||||
|
cat temp_del_repeats.fasta | sed 's/>.* />/' | sed -e 's/_[0-9]* \\[/ \\[/' > temp_rm_empty_seq.fasta
|
||||||
|
awk 'BEGIN {RS = ">" ; FS = "\\n" ; ORS = ""} \$2 {print ">"\$0}' temp_rm_empty_seq.fasta > non_redundant.fasta
|
||||||
|
rm temp_transcriptome.fasta
|
||||||
|
rm temp_del_repeats.fasta
|
||||||
|
rm temp_rm_empty_seq.fasta
|
||||||
|
"""
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
process makeReport {
|
process makeReport {
|
||||||
|
|
||||||
label "isoforms"
|
label "isoforms"
|
||||||
@ -288,6 +315,8 @@ process makeReport {
|
|||||||
path(aln_stats),
|
path(aln_stats),
|
||||||
path(gffcmp_dir),
|
path(gffcmp_dir),
|
||||||
path(gff_annotation)
|
path(gff_annotation)
|
||||||
|
path "de_report/*"
|
||||||
|
path "seqkit/*"
|
||||||
output:
|
output:
|
||||||
path("wf-transcriptomes-*.html"), emit: report
|
path("wf-transcriptomes-*.html"), emit: report
|
||||||
script:
|
script:
|
||||||
@ -298,8 +327,15 @@ process makeReport {
|
|||||||
def OPT_DENOVO = denovo ? "--denovo" : ''
|
def OPT_DENOVO = denovo ? "--denovo" : ''
|
||||||
def OPT_PC_REPORT = pychopper_report.name.startsWith('OPTIONAL_FILE') ? '' : "--pychop_report ${pychopper_report}"
|
def OPT_PC_REPORT = pychopper_report.name.startsWith('OPTIONAL_FILE') ? '' : "--pychop_report ${pychopper_report}"
|
||||||
def OPT_JAFFAL_CSV = jaffal_csv.name.startsWith('OPTIONAL_FILE') ? '' : "--jaffal_csv ${jaffal_csv}"
|
def OPT_JAFFAL_CSV = jaffal_csv.name.startsWith('OPTIONAL_FILE') ? '' : "--jaffal_csv ${jaffal_csv}"
|
||||||
|
|
||||||
"""
|
"""
|
||||||
|
if [ -e "de_report/OPTIONAL_FILE" ]; then
|
||||||
|
dereport=""
|
||||||
|
else
|
||||||
|
dereport="--de_report true --de_stats "seqkit/*""
|
||||||
|
mv de_report/*.gtf de_report/stringtie_merged.gtf
|
||||||
|
fi
|
||||||
|
|
||||||
report.py --report $report_name \
|
report.py --report $report_name \
|
||||||
--versions $versions \
|
--versions $versions \
|
||||||
--params params.json \
|
--params params.json \
|
||||||
@ -311,7 +347,9 @@ process makeReport {
|
|||||||
--gff_annotation $gff_annotation \
|
--gff_annotation $gff_annotation \
|
||||||
--isoform_table_nrows $params.isoform_table_nrows \
|
--isoform_table_nrows $params.isoform_table_nrows \
|
||||||
$OPT_JAFFAL_CSV \
|
$OPT_JAFFAL_CSV \
|
||||||
$OPT_DENOVO
|
$OPT_DENOVO \
|
||||||
|
\$dereport
|
||||||
|
|
||||||
"""
|
"""
|
||||||
}
|
}
|
||||||
|
|
||||||
@ -340,6 +378,8 @@ workflow pipeline {
|
|||||||
jaffal_refBase
|
jaffal_refBase
|
||||||
jaffal_genome
|
jaffal_genome
|
||||||
jaffal_annotation
|
jaffal_annotation
|
||||||
|
condition_sheet
|
||||||
|
ref_transcriptome
|
||||||
main:
|
main:
|
||||||
map_sample_ids_cls = {it ->
|
map_sample_ids_cls = {it ->
|
||||||
/* Harmonize tuples
|
/* Harmonize tuples
|
||||||
@ -381,15 +421,15 @@ workflow pipeline {
|
|||||||
|
|
||||||
if (params.denovo){
|
if (params.denovo){
|
||||||
println("Doing de novo assembly")
|
println("Doing de novo assembly")
|
||||||
m = denovo_assembly(full_len_reads, ref_genome)
|
assembly = denovo_assembly(full_len_reads, ref_genome)
|
||||||
|
|
||||||
} else {
|
} else {
|
||||||
build_minimap_index(ref_genome)
|
build_minimap_index(ref_genome)
|
||||||
println("Doing reference based transcript analysis")
|
println("Doing reference based transcript analysis")
|
||||||
m = reference_assembly(build_minimap_index.out.index, ref_genome, full_len_reads)
|
assembly = reference_assembly(build_minimap_index.out.index, ref_genome, full_len_reads)
|
||||||
}
|
}
|
||||||
|
|
||||||
split_bam(m.bam)
|
split_bam(assembly.bam)
|
||||||
|
|
||||||
assemble_transcripts(split_bam.out.bundles.flatMap(map_sample_ids_cls), ref_annotation)
|
assemble_transcripts(split_bam.out.bundles.flatMap(map_sample_ids_cls), ref_annotation)
|
||||||
|
|
||||||
@ -401,13 +441,13 @@ workflow pipeline {
|
|||||||
|
|
||||||
if (params.denovo){
|
if (params.denovo){
|
||||||
// Use the per-sample, de novo-assembled CDS
|
// Use the per-sample, de novo-assembled CDS
|
||||||
seq_for_transcriptome_build = m.cds
|
seq_for_transcriptome_build = assembly.cds
|
||||||
}else {
|
}else {
|
||||||
// For reference based assembly, there is only one reference
|
// For reference based assembly, there is only one reference
|
||||||
// So map this reference to all sample_ids
|
// So map this reference to all sample_ids
|
||||||
seq_for_transcriptome_build = sample_ids.flatten().combine(Channel.fromPath(params.ref_genome))
|
seq_for_transcriptome_build = sample_ids.flatten().combine(Channel.fromPath(params.ref_genome))
|
||||||
}
|
}
|
||||||
|
|
||||||
if (jaffal_refBase){
|
if (jaffal_refBase){
|
||||||
gene_fusions(full_len_reads, jaffal_refBase, jaffal_genome, jaffal_annotation)
|
gene_fusions(full_len_reads, jaffal_refBase, jaffal_genome, jaffal_annotation)
|
||||||
jaffal_out = gene_fusions.out.results_csv.collectFile(keepHeader: true, name: 'jaffal.csv')
|
jaffal_out = gene_fusions.out.results_csv.collectFile(keepHeader: true, name: 'jaffal.csv')
|
||||||
@ -415,6 +455,31 @@ workflow pipeline {
|
|||||||
jaffal_out = file("$projectDir/data/OPTIONAL_FILE_1")
|
jaffal_out = file("$projectDir/data/OPTIONAL_FILE_1")
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
|
get_transcriptome(
|
||||||
|
merge_gff_bundles.out.gff
|
||||||
|
.join(run_gffcompare.out.gffcmp_dir)
|
||||||
|
.join(seq_for_transcriptome_build))
|
||||||
|
|
||||||
|
if (params.de_analysis){
|
||||||
|
|
||||||
|
if (!params.ref_transcriptome){
|
||||||
|
merge_transcriptomes(run_gffcompare.output.gtf.collect(), ref_annotation, ref_genome)
|
||||||
|
transcriptome = merge_transcriptomes.out.fasta
|
||||||
|
gtf = merge_transcriptomes.out.gtf
|
||||||
|
}
|
||||||
|
else {
|
||||||
|
transcriptome = ref_transcriptome
|
||||||
|
gtf = Channel.fromPath(ref_annotation)
|
||||||
|
}
|
||||||
|
de = differential_expression(transcriptome, summariseConcatReads.out.input_reads, condition_sheet, gtf)
|
||||||
|
de_report = de.all_de
|
||||||
|
count_transcripts_file = de.count_transcripts
|
||||||
|
dtu_plots = de.dtu_plots
|
||||||
|
} else{
|
||||||
|
de_report = file("$projectDir/data/OPTIONAL_FILE")
|
||||||
|
count_transcripts_file = file("$projectDir/data/OPTIONAL_FILE")
|
||||||
|
}
|
||||||
makeReport(
|
makeReport(
|
||||||
software_versions,
|
software_versions,
|
||||||
workflow_params,
|
workflow_params,
|
||||||
@ -422,42 +487,41 @@ workflow pipeline {
|
|||||||
pychopper_report,
|
pychopper_report,
|
||||||
jaffal_out,
|
jaffal_out,
|
||||||
summariseConcatReads.out.summary
|
summariseConcatReads.out.summary
|
||||||
.join(m.stats)
|
.join(assembly.stats)
|
||||||
.join(run_gffcompare.out.gffcmp_dir)
|
.join(run_gffcompare.out.gffcmp_dir)
|
||||||
.join(merge_gff_bundles.out.gff)
|
.join(merge_gff_bundles.out.gff)
|
||||||
.toList().transpose().toList())
|
.toList().transpose().toList(),
|
||||||
|
de_report,
|
||||||
|
count_transcripts_file)
|
||||||
|
|
||||||
report = makeReport.out.report
|
report = makeReport.out.report
|
||||||
|
|
||||||
get_transcriptome(
|
|
||||||
merge_gff_bundles.out.gff
|
|
||||||
.join(run_gffcompare.out.gffcmp_dir)
|
|
||||||
.join(seq_for_transcriptome_build))
|
|
||||||
|
|
||||||
if (use_ref_ann){
|
if (use_ref_ann){
|
||||||
results = run_gffcompare.output.gffcmp_dir
|
results = run_gffcompare.output.gffcmp_dir
|
||||||
.concat(
|
.concat(
|
||||||
m.stats,
|
assembly.stats,
|
||||||
get_transcriptome.out.flatMap(map_sample_ids_cls))
|
get_transcriptome.out.transcriptome.flatMap(map_sample_ids_cls))
|
||||||
.map {it -> it[1]}
|
.map {it -> it[1]}
|
||||||
.concat(makeReport.out.report)
|
.concat(makeReport.out.report)
|
||||||
|
|
||||||
}
|
}
|
||||||
if (!use_ref_ann && !params.denovo){
|
if (!use_ref_ann && !params.denovo){
|
||||||
results = m.stats
|
results = assembly.stats
|
||||||
.concat(
|
.concat(
|
||||||
get_transcriptome.out.flatMap(map_sample_ids_cls))
|
get_transcriptome.out.transcriptome.flatMap(map_sample_ids_cls))
|
||||||
.map {it -> it[1]}
|
.map {it -> it[1]}
|
||||||
.concat(makeReport.out.report)
|
.concat(makeReport.out.report)
|
||||||
|
|
||||||
}
|
}
|
||||||
if (params.denovo){
|
if (params.denovo){
|
||||||
results = m.cds
|
results = assembly.cds
|
||||||
.concat(m.stats,
|
.concat(assembly.stats,
|
||||||
seq_for_transcriptome_build,
|
seq_for_transcriptome_build,
|
||||||
get_transcriptome.out.flatMap(map_sample_ids_cls),
|
get_transcriptome.out.transcriptome.flatMap(map_sample_ids_cls),
|
||||||
merge_gff_bundles.out.gff,
|
merge_gff_bundles.out.gff,
|
||||||
m.opt_qual_ch.flatMap {
|
assembly.opt_qual_ch.flatMap {
|
||||||
it ->
|
it ->
|
||||||
l = []
|
l = []
|
||||||
for (x in it[1..-1]){
|
for (x in it[1..-1]){
|
||||||
@ -474,6 +538,10 @@ workflow pipeline {
|
|||||||
.map {it -> it[1]})
|
.map {it -> it[1]})
|
||||||
}
|
}
|
||||||
|
|
||||||
|
if (params.de_analysis){
|
||||||
|
results = results.concat(de.dtu_plots)
|
||||||
|
}
|
||||||
|
|
||||||
emit:
|
emit:
|
||||||
results
|
results
|
||||||
telemetry = workflow_params
|
telemetry = workflow_params
|
||||||
@ -518,7 +586,7 @@ workflow {
|
|||||||
}else{
|
}else{
|
||||||
ref_annotation = file("$projectDir/data/OPTIONAL_FILE")
|
ref_annotation = file("$projectDir/data/OPTIONAL_FILE")
|
||||||
}
|
}
|
||||||
if (params.jaffal_refBase){
|
if (params.jaffal_refBase){
|
||||||
jaffal_refBase = file(params.jaffal_refBase, type: "dir")
|
jaffal_refBase = file(params.jaffal_refBase, type: "dir")
|
||||||
if (!jaffal_refBase.exists()) {
|
if (!jaffal_refBase.exists()) {
|
||||||
error = "--jaffa_refBase: Directory doesn't exist, check path."
|
error = "--jaffa_refBase: Directory doesn't exist, check path."
|
||||||
@ -526,7 +594,22 @@ workflow {
|
|||||||
}else{
|
}else{
|
||||||
jaffal_refBase = null
|
jaffal_refBase = null
|
||||||
}
|
}
|
||||||
|
ref_transcriptome = file("$projectDir/data/OPTIONAL_FILE")
|
||||||
|
if (params.ref_transcriptome){
|
||||||
|
ref_transcriptome = file(params.ref_transcriptome, type:"file")
|
||||||
|
}
|
||||||
|
if (params.de_analysis){
|
||||||
|
if (!params.ref_annotation){
|
||||||
|
error = "You must provide a reference annotation."
|
||||||
|
}
|
||||||
|
if (!params.condition_sheet){
|
||||||
|
|
||||||
|
error = "You must provide a condition_sheet or set de_analysis to false."
|
||||||
|
}
|
||||||
|
condition_sheet = file(params.condition_sheet, type:"file")
|
||||||
|
} else{
|
||||||
|
condition_sheet = file("$projectDir/data/OPTIONAL_FILE")
|
||||||
|
}
|
||||||
if (error){
|
if (error){
|
||||||
println(error)
|
println(error)
|
||||||
}else{
|
}else{
|
||||||
@ -537,7 +620,9 @@ workflow {
|
|||||||
"sanitize": params.sanitize_fastq,
|
"sanitize": params.sanitize_fastq,
|
||||||
"output":params.out_dir])
|
"output":params.out_dir])
|
||||||
|
|
||||||
pipeline(reads, ref_genome, ref_annotation, jaffal_refBase, params.jaffal_genome, params.jaffal_annotation)
|
pipeline(reads, ref_genome, ref_annotation,
|
||||||
|
jaffal_refBase, params.jaffal_genome, params.jaffal_annotation,
|
||||||
|
condition_sheet, ref_transcriptome)
|
||||||
|
|
||||||
output(pipeline.out.results)
|
output(pipeline.out.results)
|
||||||
|
|
||||||
|
|||||||
@ -127,6 +127,16 @@ params {
|
|||||||
// This needs overriding if running elsewhere
|
// This needs overriding if running elsewhere
|
||||||
jaffal_dir = "/home/epi2melabs/JAFFA"
|
jaffal_dir = "/home/epi2melabs/JAFFA"
|
||||||
|
|
||||||
|
// de options
|
||||||
|
de_analysis = false
|
||||||
|
condition_sheet = "test_data/condition_sheet.tsv"
|
||||||
|
ref_transcriptome = null
|
||||||
|
min_samps_gene_expr = 3
|
||||||
|
min_samps_feature_expr = 1
|
||||||
|
min_gene_expr = 10
|
||||||
|
min_feature_expr = 3
|
||||||
|
|
||||||
|
|
||||||
wf {
|
wf {
|
||||||
example_cmd = [
|
example_cmd = [
|
||||||
"--fastq test_data/fastq",
|
"--fastq test_data/fastq",
|
||||||
@ -189,18 +199,19 @@ profiles {
|
|||||||
|
|
||||||
// profile using conda environments
|
// profile using conda environments
|
||||||
conda {
|
conda {
|
||||||
docker.enabled = false
|
docker.enabled = false
|
||||||
process {
|
process {
|
||||||
withLabel:isoforms {
|
withLabel:isoforms {
|
||||||
conda = "${projectDir}/environment.yaml"
|
conda = "${projectDir}/environment.yaml"
|
||||||
}
|
|
||||||
shell = ['/bin/bash', '-euo', 'pipefail']
|
|
||||||
}
|
|
||||||
conda {
|
|
||||||
cacheDir = ""
|
|
||||||
useMamba = true
|
|
||||||
}
|
}
|
||||||
|
shell = ['/bin/bash', '-euo', 'pipefail']
|
||||||
}
|
}
|
||||||
|
conda {
|
||||||
|
enabled = true // required for 22.08
|
||||||
|
cacheDir = ""
|
||||||
|
useMamba = true
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// Using AWS batch.
|
// Using AWS batch.
|
||||||
// May need to set aws.region and aws.batch.cliPath
|
// May need to set aws.region and aws.batch.cliPath
|
||||||
|
|||||||
@ -247,6 +247,51 @@
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
},
|
},
|
||||||
|
"differential_expression_options": {
|
||||||
|
"title": "Differential expression options",
|
||||||
|
"type": "object",
|
||||||
|
"description": "",
|
||||||
|
"default": "",
|
||||||
|
"properties": {
|
||||||
|
"de_analysis": {
|
||||||
|
"type": "boolean",
|
||||||
|
"description": "Run DE anaylsis",
|
||||||
|
"help_text": "Running this requires you to provide at least two replicates for a control and treated sample as well as a condition sheet param."
|
||||||
|
},
|
||||||
|
"condition_sheet": {
|
||||||
|
"type": "string",
|
||||||
|
"format": "file-path",
|
||||||
|
"description": "tsv with (sample, condition, type)",
|
||||||
|
"default": "null"
|
||||||
|
},
|
||||||
|
"ref_transcriptome": {
|
||||||
|
"type": "string",
|
||||||
|
"default": "null",
|
||||||
|
"format": "file-path",
|
||||||
|
"description": "Transcriptome reference file"
|
||||||
|
},
|
||||||
|
"min_gene_expr": {
|
||||||
|
"type": "integer",
|
||||||
|
"default": 10,
|
||||||
|
"description": "Minimum gene counts"
|
||||||
|
},
|
||||||
|
"min_feature_expr": {
|
||||||
|
"type": "integer",
|
||||||
|
"default": 3,
|
||||||
|
"description": "Minimum transcript counts"
|
||||||
|
},
|
||||||
|
"min_samps_feature_expr": {
|
||||||
|
"type": "integer",
|
||||||
|
"default": 1,
|
||||||
|
"description": "Transcripts expressed in minimum this many samples"
|
||||||
|
},
|
||||||
|
"min_samps_gene_expr": {
|
||||||
|
"type": "integer",
|
||||||
|
"description": "Genes expressed in minimum this many samples",
|
||||||
|
"default": 3
|
||||||
|
}
|
||||||
|
}
|
||||||
|
},
|
||||||
"meta_data": {
|
"meta_data": {
|
||||||
"title": "Meta Data",
|
"title": "Meta Data",
|
||||||
"type": "object",
|
"type": "object",
|
||||||
@ -297,6 +342,9 @@
|
|||||||
{
|
{
|
||||||
"$ref": "#/definitions/fusion_detection_options"
|
"$ref": "#/definitions/fusion_detection_options"
|
||||||
},
|
},
|
||||||
|
{
|
||||||
|
"$ref": "#/definitions/differential_expression_options"
|
||||||
|
},
|
||||||
{
|
{
|
||||||
"$ref": "#/definitions/meta_data"
|
"$ref": "#/definitions/meta_data"
|
||||||
},
|
},
|
||||||
@ -330,7 +378,7 @@
|
|||||||
}
|
}
|
||||||
},
|
},
|
||||||
"docs": {
|
"docs": {
|
||||||
"intro": "## Introduction\n\nThis workflow identifies RNA isoforms using either cDNA or direct RNA (dRNA) \nOxford Nanopore reads.\n\n### Preprocesing\ncDNA reads are initially preprocessed by [pychopper](https://github.com/epi2me-labs/pychopper) \nfor the identification of full-length reads, as well as trimming and orientation correction (This step is omitted for \n direct RNA reads).\n\n\n### Transcript assembly\n\n#### Reference-aided transcript assembly approach\n* Full length reads are mapped to a supplied reference genome using [minimap2](https://github.com/lh3/minimap2)\n* Transcripts are assembled by [stringtie](http://ccb.jhu.edu/software/stringtie) \nin long read mode (with or without a guide reference annotation) to generate the GFF annotation.\n* The annotation generated by the pipeline is compared to the reference annotation. \nusing [gffcompare](http://ccb.jhu.edu/software/stringtie/gffcompare.shtml)\n\n#### de novo-based transcript assembly (experimental!)\n* Sequence clusters are generated using [isONclust2](https://github.com/nanoporetech/isONclust2)\n * If a reference genome is supplied, cluster quality metrics are determined by comparing \n with clusters generated from a minimap2 alignment.\n* A consensus sequence for each cluster is generated using [spoa](https://github.com/rvaser/spoa)\n* Three rounds of polishing using racon and minimap2 to give a final polished CDS for each gene.\n* Full-length reads are then mapped to these polished CDS.\n* Transcripts are assembled by stringtie as for the reference-based approach.\n* __Note__: This approach is currently not supported with direct RNA reads.\n\n### Fusion gene detection\nFusion gene detection is performed using [JAFFA](https://github.com/Oshlack/JAFFA), with the JAFFAL extension for use \nwith ONT long reads. \n\n### Workflow inputs\n- Directory containing cDNA/direct RNA reads. Or a directory containing subdirectories each with reads from different samples\n (in fastq/fastq.gz format)\n- Reference genome in fasta format (required for reference-based assembly).\n- Optional reference annotation in GFF2/3 format.\n- For fusion detection, JAFFAL reference files (see Quickstart) \n",
|
"intro": "## Introduction\n\nThis workflow identifies RNA isoforms using either cDNA or direct RNA (dRNA) \nOxford Nanopore reads.\n\n### Preprocesing\ncDNA reads are initially preprocessed by [pychopper](https://github.com/epi2me-labs/pychopper) \nfor the identification of full-length reads, as well as trimming and orientation correction (This step is omitted for \n direct RNA reads).\n\n\n### Transcript assembly\n\n#### Reference-aided transcript assembly approach\n* Full length reads are mapped to a supplied reference genome using [minimap2](https://github.com/lh3/minimap2)\n* Transcripts are assembled by [stringtie](http://ccb.jhu.edu/software/stringtie) \nin long read mode (with or without a guide reference annotation) to generate the GFF annotation.\n* The annotation generated by the pipeline is compared to the reference annotation. \nusing [gffcompare](http://ccb.jhu.edu/software/stringtie/gffcompare.shtml)\n\n#### de novo-based transcript assembly (experimental!)\n* Sequence clusters are generated using [isONclust2](https://github.com/nanoporetech/isONclust2)\n * If a reference genome is supplied, cluster quality metrics are determined by comparing \n with clusters generated from a minimap2 alignment.\n* A consensus sequence for each cluster is generated using [spoa](https://github.com/rvaser/spoa)\n* Three rounds of polishing using racon and minimap2 to give a final polished CDS for each gene.\n* Full-length reads are then mapped to these polished CDS.\n* Transcripts are assembled by stringtie as for the reference-based approach.\n* __Note__: This approach is currently not supported with direct RNA reads.\n\n### Fusion gene detection\nFusion gene detection is performed using [JAFFA](https://github.com/Oshlack/JAFFA), with the JAFFAL extension for use \nwith ONT long reads. \n\n### Differential expression analysis\n* Differential expression is done using the transcripts output by the workflow.\n* A non redundant transcriptome is found using the merge function in [stringtie](http://ccb.jhu.edu/software/stringtie).\n* The reads are then aligned to the transcriptome using minimap2 in a splice-aware manner.\n* [salmon](https://github.com/COMBINE-lab/salmon) is used for transcript quantification.\n* R packages [edgeR](https://bioconductor.org/packages/release/bioc/html/edgeR.html) and [stageR](https://bioconductor.org/packages/release/bioc/html/stageR.html) are used for differential expression analysis.\n* [DEXSeq](https://bioconductor.org/packages/release/bioc/html/DEXSeq.html) is then used for differential transcript usage analysis.\n\n### Workflow inputs\n- Directory containing cDNA/direct RNA reads. Or a directory containing subdirectories each with reads from different samples\n (in fastq/fastq.gz format)\n- Reference genome in fasta format (required for reference-based assembly).\n- Optional reference annotation in GFF2/3 format (required for differential expression analysis `--de_analysis`).\n- For fusion detection, JAFFAL reference files (see Quickstart) \n",
|
||||||
"links": "## Useful links\n\n* [nextflow](https://www.nextflow.io/)\n* [docker](https://www.docker.com/products/docker-desktop)\n* [Singularity](https://sylabs.io/singularity/)\n* [conda](https://docs.conda.io/en/latest/miniconda.html)\n* [racon](https://github.com/isovic/racon)\n* [spoa](https://github.com/rvaser/spoa)\n* [inONclust](https://github.com/ksahlin/isONclust)\n* [isONclust2](https://github.com/nanoporetech/isONclust2)"
|
"links": "## Useful links\n\n* [nextflow](https://www.nextflow.io/)\n* [docker](https://www.docker.com/products/docker-desktop)\n* [Singularity](https://sylabs.io/singularity/)\n* [conda](https://docs.conda.io/en/latest/miniconda.html)\n* [racon](https://github.com/isovic/racon)\n* [spoa](https://github.com/rvaser/spoa)\n* [inONclust](https://github.com/ksahlin/isONclust)\n* [isONclust2](https://github.com/nanoporetech/isONclust2)"
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
156
subworkflows/differential_expression.nf
Normal file
156
subworkflows/differential_expression.nf
Normal file
@ -0,0 +1,156 @@
|
|||||||
|
process count_transcripts {
|
||||||
|
// Count transcripts using Salmon.
|
||||||
|
// library type is specified as forward stranded (-l SF) as it should have either been through pychopper or come from direct RNA reads.
|
||||||
|
label "isoforms"
|
||||||
|
input:
|
||||||
|
tuple val(sample_id), path(bam)
|
||||||
|
path ref_transcriptome
|
||||||
|
output:
|
||||||
|
path "*transcript_counts.tsv", emit: counts
|
||||||
|
path "*seqkit.stats", emit: seqkit_stats
|
||||||
|
"""
|
||||||
|
salmon quant --noErrorModel -p $params.threads -t $ref_transcriptome -l SF -a $bam -o counts
|
||||||
|
mv counts/quant.sf "${sample_id}".transcript_counts.tsv
|
||||||
|
seqkit bam "$bam" 2> "${sample_id}".seqkit.stats
|
||||||
|
"""
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
process mergeCounts {
|
||||||
|
label "isoforms"
|
||||||
|
input:
|
||||||
|
path counts
|
||||||
|
output:
|
||||||
|
path "all_counts.tsv"
|
||||||
|
"""
|
||||||
|
merge_count_tsvs.py -z -o all_counts.tsv $counts
|
||||||
|
"""
|
||||||
|
}
|
||||||
|
|
||||||
|
process mergeTPM {
|
||||||
|
label "isoforms"
|
||||||
|
input:
|
||||||
|
path counts
|
||||||
|
output:
|
||||||
|
path "tpm_counts.tsv"
|
||||||
|
"""
|
||||||
|
merge_count_tsvs.py -z -o tpm_counts.tsv $counts -tpm
|
||||||
|
"""
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
process deAnalysis {
|
||||||
|
label "isoforms"
|
||||||
|
input:
|
||||||
|
path condition_sheet
|
||||||
|
path merged_tsv
|
||||||
|
path annotation
|
||||||
|
output:
|
||||||
|
path "de_analysis/results_dtu_stageR.tsv", emit: stageR
|
||||||
|
path "merged/all_counts_filtered.tsv", emit: flt_counts
|
||||||
|
path "merged/all_gene_counts.tsv", emit: gene_counts
|
||||||
|
path "de_analysis/results_dge.tsv", emit: dge
|
||||||
|
path "de_analysis/results_dexseq.tsv", emit: dexseq
|
||||||
|
|
||||||
|
"""
|
||||||
|
cp $annotation annotation.gtf
|
||||||
|
echo \$(realpath annotation.gtf)
|
||||||
|
echo Annotation\$'\t'min_samps_gene_expr\$'\t'min_samps_feature_expr\$'\t'min_gene_expr\$'\t'min_feature_expr > params.tsv
|
||||||
|
echo \$(realpath $params.ref_annotation)\$'\t'$params.min_samps_gene_expr\$'\t'\
|
||||||
|
$params.min_samps_feature_expr\$'\t'$params.min_gene_expr\$'\t'$params.min_feature_expr >> params.tsv
|
||||||
|
mkdir merged
|
||||||
|
mkdir de_analysis
|
||||||
|
mv $merged_tsv merged/all_counts.tsv
|
||||||
|
mv params.tsv de_analysis/de_params.tsv
|
||||||
|
mv $condition_sheet de_analysis/coldata.tsv
|
||||||
|
de_analysis.R
|
||||||
|
"""
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
process plotResults {
|
||||||
|
label "isoforms"
|
||||||
|
input:
|
||||||
|
path flt_count
|
||||||
|
path res_dtu
|
||||||
|
path condition_sheet
|
||||||
|
output:
|
||||||
|
path "de_analysis/dtu_plots.pdf", emit: dtu_plots
|
||||||
|
path "condition_sheet.tsv", emit: condition_sheet_tsv
|
||||||
|
"""
|
||||||
|
mkdir merged
|
||||||
|
mkdir de_analysis
|
||||||
|
mv $res_dtu de_analysis/results_dtu_stageR.tsv
|
||||||
|
mv $condition_sheet de_analysis/coldata.tsv
|
||||||
|
mv $flt_count merged/all_counts_filtered.tsv
|
||||||
|
plot_dtu_results.R
|
||||||
|
cp de_analysis/coldata.tsv condition_sheet.tsv
|
||||||
|
|
||||||
|
"""
|
||||||
|
}
|
||||||
|
|
||||||
|
process build_minimap_index_transcriptome{
|
||||||
|
/*
|
||||||
|
Build minimap index from reference genome
|
||||||
|
*/
|
||||||
|
label "isoforms"
|
||||||
|
cpus params.threads
|
||||||
|
input:
|
||||||
|
path reference
|
||||||
|
output:
|
||||||
|
path "genome_index.mmi", emit: index
|
||||||
|
script:
|
||||||
|
"""
|
||||||
|
minimap2 -t ${params.threads} ${params.minimap_index_opts} -I 1000G -d "genome_index.mmi" ${reference}
|
||||||
|
|
||||||
|
"""
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
process map_transcriptome{
|
||||||
|
/*
|
||||||
|
Map reads to reference using minimap2.
|
||||||
|
Filter reads by mapping quality.
|
||||||
|
Filter internally-primed reads.
|
||||||
|
*/
|
||||||
|
label "isoforms"
|
||||||
|
cpus params.threads
|
||||||
|
|
||||||
|
input:
|
||||||
|
tuple val(sample_id), path (fastq_reads)
|
||||||
|
file index
|
||||||
|
file transcript_reference
|
||||||
|
output:
|
||||||
|
tuple val(sample_id), path("${sample_id}_reads_aln_sorted.bam"), emit: bam
|
||||||
|
"""
|
||||||
|
minimap2 -t ${params.threads} -ax splice -uf -p 1.0 $index $fastq_reads\
|
||||||
|
| samtools view -Sb > output.bam
|
||||||
|
samtools sort -@ ${params.threads} output.bam -o "${sample_id}"_reads_aln_sorted.bam
|
||||||
|
samtools index "${sample_id}"_reads_aln_sorted.bam
|
||||||
|
"""
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
workflow differential_expression {
|
||||||
|
take:
|
||||||
|
ref_transcriptome
|
||||||
|
full_len_reads
|
||||||
|
condition_sheet
|
||||||
|
ref_annotation
|
||||||
|
main:
|
||||||
|
t_index = build_minimap_index_transcriptome(ref_transcriptome)
|
||||||
|
mapped = map_transcriptome(full_len_reads, t_index, ref_transcriptome)
|
||||||
|
count_transcripts(mapped.bam, ref_transcriptome)
|
||||||
|
merged = mergeCounts(count_transcripts.out.counts.collect())
|
||||||
|
merged_TPM = mergeTPM(count_transcripts.out.counts.collect())
|
||||||
|
analysis = deAnalysis(condition_sheet, merged, ref_annotation)
|
||||||
|
plotResults(analysis.flt_counts, analysis.stageR, condition_sheet)
|
||||||
|
de_report = analysis.flt_counts.combine(analysis.gene_counts).combine(analysis.dge).combine(analysis.dexseq).combine(
|
||||||
|
analysis.stageR).combine(plotResults.out.condition_sheet_tsv).combine(merged).combine(
|
||||||
|
ref_annotation).combine(merged_TPM)
|
||||||
|
count_transcripts_file = count_transcripts.out.seqkit_stats.collect()
|
||||||
|
emit:
|
||||||
|
all_de = de_report
|
||||||
|
count_transcripts = count_transcripts_file
|
||||||
|
dtu_plots = plotResults.out.dtu_plots
|
||||||
|
}
|
||||||
7
test_data/condition_sheet.tsv
Normal file
7
test_data/condition_sheet.tsv
Normal file
@ -0,0 +1,7 @@
|
|||||||
|
sample,condition,type
|
||||||
|
barcode01,untreated,single-read
|
||||||
|
barcode02,untreated,single-read
|
||||||
|
barcode03,untreated,single-read
|
||||||
|
barcode04,treated,single-read
|
||||||
|
barcode05,treated,single-read
|
||||||
|
barcode06,treated,single-read
|
||||||
|
Loading…
Reference in New Issue
Block a user