Epi2melabs pychopper

This commit is contained in:
Neil Horner 2022-06-01 12:08:00 +00:00
parent 2f2130b443
commit d97e84cdc1
9 changed files with 22 additions and 21 deletions

View File

@ -4,9 +4,10 @@ All notable changes to this project will be documented in this file.
The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.0.0/),
and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html).
## [unreleased]
## [v0.1.3]
### Changed
- Better help text on cli
- Use EPI2ME Labs-maintained version of pychopper
## [v0.1.2]
### Added

View File

@ -12,7 +12,7 @@ This workflow identifies RNA isoforms using either cDNA or direct RNA (dRNA)
Oxford Nanopore reads.
### Preprocesing
cDNA reads are initially preprocessed by [pychopper](https://github.com/nanoporetech/pychopper)
cDNA reads are initially preprocessed by [pychopper](https://github.com/epi2me-labs/pychopper)
for the identification of full-length reads, as well as trimming and orientation correction (This step is omitted for
direct RNA reads).

View File

@ -498,7 +498,7 @@ def gff_compare_plots(report, gffcompare_outdirs: Path, sample_ids):
def pychopper_plots(report, pychop_report):
"""Make plots from pychopper.cdna_classifier.py.
"""Make plots from pychopper output.
:param report: aplanat WFReport
:param pychop_report: path to pychopper stats file
@ -507,7 +507,7 @@ def pychopper_plots(report, pychop_report):
section.markdown('''
### Pychopper summary statisitcs
The following plots summarize the output of [cdna_classifier.py]
The following plots summarize [pychopper] output
(https://github.com/nanoporetech/pychopper)
* **Pr.found**: Reads with primers found in correct orientation at

View File

@ -4,7 +4,7 @@ This workflow identifies RNA isoforms using either cDNA or direct RNA (dRNA)
Oxford Nanopore reads.
### Preprocesing
cDNA reads are initially preprocessed by [pychopper](https://github.com/nanoporetech/pychopper)
cDNA reads are initially preprocessed by [pychopper](https://github.com/epi2me-labs/pychopper)
for the identification of full-length reads, as well as trimming and orientation correction (This step is omitted for
direct RNA reads).

View File

@ -6,16 +6,16 @@ channels:
- defaults
dependencies:
- python==3.8.*
- aplanat >=0.6.4
- aplanat>=0.6.4
- epi2melabs
- minimap2 ==2.24
- samtools ==1.14
- bedtools ==2.30.0
- pychopper==2.5.0
- minimap2==2.24
- samtools==1.14
- bedtools==2.30.0
- pychopper>=2.7.0
- pandas==1.3.5
- gffread==0.12.7
- gffcompare==0.11.2
- gffutils=0.10.1
- gffutils==0.10.1
- seqkit==2.1.0
- stringtie==2.1.1
- curl

12
main.nf
View File

@ -74,7 +74,7 @@ process getParams {
process preprocess_reads {
/*
Concatenate reads from a sample directory.
Optionally classify, trim, and orient cDNA reads using cdna_classifier from pychopper
Optionally classify, trim, and orient cDNA reads using pychopper
*/
label "isoforms"
@ -87,13 +87,13 @@ process preprocess_reads {
path '*.tsv', emit: report
script:
"""
cdna_classifier.py -t ${params.threads} ${params.pychopper_opts} ${input_reads} ${sample_id}_full_length_reads.fq
mv cdna_classifier_report.tsv ${sample_id}_cdna_classifier_report.tsv
generate_pychopper_stats.py --data ${sample_id}_cdna_classifier_report.tsv --output .
pychopper -t ${params.threads} ${params.pychopper_opts} ${input_reads} ${sample_id}_full_length_reads.fq
mv pychopper.tsv ${sample_id}_pychopper.tsv
generate_pychopper_stats.py --data ${sample_id}_pychopper.tsv --output .
# Add sample id column
sed "1s/\$/\tsample_id/; 1 ! s/\$/\t${sample_id}/" ${sample_id}_cdna_classifier_report.tsv > tmp
mv tmp ${sample_id}_cdna_classifier_report.tsv
sed "1s/\$/\tsample_id/; 1 ! s/\$/\t${sample_id}/" ${sample_id}_pychopper.tsv > tmp
mv tmp ${sample_id}_pychopper.tsv
"""
}

View File

@ -23,7 +23,7 @@ params {
sample = null
sample_sheet = null
sanitize_fastq = false
wfversion = "v0.1.2"
wfversion = "v0.1.3"
aws_image_prefix = null
aws_queue = null
report_name = "report"

View File

@ -280,7 +280,7 @@
},
"wfversion": {
"type": "string",
"default": "v0.1.2",
"default": "v0.1.3",
"hidden": true
},
"monochrome_logs": {
@ -295,7 +295,7 @@
}
},
"docs": {
"intro": "## Introduction\n\nThis workflow identifies RNA isoforms using either cDNA or direct RNA (dRNA) \nOxford Nanopore reads.\n\n### Preprocesing\ncDNA reads are initially preprocessed by [pychopper](https://github.com/nanoporetech/pychopper) \nfor the identification of full-length reads, as well as trimming and orientation correction (This step is omitted for \n direct RNA reads).\n\n### Reference-aided approach\n* Full length reads are mapped to a supplied reference genome using [minimap2](https://github.com/lh3/minimap2)\n* Transcripts are assembled by [stringtie](http://ccb.jhu.edu/software/stringtie) \nin long read mode (with or without a guide reference annotation) to generate the GFF annotation.\n* The annotation generated by the pipeline is compared to the reference annotation. \nusing [gffcompare](http://ccb.jhu.edu/software/stringtie/gffcompare.shtml)\n\n### de novo-based approach (experimental!)\n* Sequence clusters are generated using [isONclust2](https://github.com/nanoporetech/isONclust2)\n * If a reference genome is supplied, cluster quality metrics are determined by comparing \n with clusters generated from a minimap2 alignment.\n* A consensus sequence for each cluster is generated using [spoa](https://github.com/rvaser/spoa)\n* Three rounds of polishing using racon and minimap2 to give a final polished CDS for each gene.\n* Full-length reads are then mapped to these polished CDS.\n* Transcripts are assembled by stringtie as for the reference-based approach.\n* __Note__: This approach is currently not supported with direct RNA reads.\n\n### Workflow inputs\n- Directory containing cDNA/direct RNA reads. Or a directory containing subdirectories each with reads from different samples\n (in fastq/fastq.gz format)\n- Reference genome in fasta format (required for reference-based assembly).\n- Optional reference annotation in GFF2/3 format.",
"intro": "## Introduction\n\nThis workflow identifies RNA isoforms using either cDNA or direct RNA (dRNA) \nOxford Nanopore reads.\n\n### Preprocesing\ncDNA reads are initially preprocessed by [pychopper](https://github.com/epi2me-labs/pychopper) \nfor the identification of full-length reads, as well as trimming and orientation correction (This step is omitted for \n direct RNA reads).\n\n### Reference-aided approach\n* Full length reads are mapped to a supplied reference genome using [minimap2](https://github.com/lh3/minimap2)\n* Transcripts are assembled by [stringtie](http://ccb.jhu.edu/software/stringtie) \nin long read mode (with or without a guide reference annotation) to generate the GFF annotation.\n* The annotation generated by the pipeline is compared to the reference annotation. \nusing [gffcompare](http://ccb.jhu.edu/software/stringtie/gffcompare.shtml)\n\n### de novo-based approach (experimental!)\n* Sequence clusters are generated using [isONclust2](https://github.com/nanoporetech/isONclust2)\n * If a reference genome is supplied, cluster quality metrics are determined by comparing \n with clusters generated from a minimap2 alignment.\n* A consensus sequence for each cluster is generated using [spoa](https://github.com/rvaser/spoa)\n* Three rounds of polishing using racon and minimap2 to give a final polished CDS for each gene.\n* Full-length reads are then mapped to these polished CDS.\n* Transcripts are assembled by stringtie as for the reference-based approach.\n* __Note__: This approach is currently not supported with direct RNA reads.\n\n### Workflow inputs\n- Directory containing cDNA/direct RNA reads. Or a directory containing subdirectories each with reads from different samples\n (in fastq/fastq.gz format)\n- Reference genome in fasta format (required for reference-based assembly).\n- Optional reference annotation in GFF2/3 format.",
"links": "## Useful links\n\n* [nextflow](https://www.nextflow.io/)\n* [docker](https://www.docker.com/products/docker-desktop)\n* [Singularity](https://sylabs.io/singularity/)\n* [conda](https://docs.conda.io/en/latest/miniconda.html)\n* [racon](https://github.com/isovic/racon)\n* [spoa](https://github.com/rvaser/spoa)\n* [inONclust](https://github.com/ksahlin/isONclust)\n* [isONclust2](https://github.com/nanoporetech/isONclust2)"
}
}

View File

@ -2,7 +2,7 @@ process map_reads{
/*
Map reads to reference using minimap2.
Filter reads by mapping quality.
Filter reads where length of poly(A) > max_poly_run at either ends of the read (defined by poly_context)
Filter internally-primed reads.
*/
label "isoforms"
cpus params.threads