diff --git a/CHANGELOG.md b/CHANGELOG.md index 53f6ce1..29ac54d 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -4,9 +4,10 @@ All notable changes to this project will be documented in this file. The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.0.0/), and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html). -## [unreleased] +## [v0.1.3] ### Changed - Better help text on cli +- Use EPI2ME Labs-maintained version of pychopper ## [v0.1.2] ### Added diff --git a/README.md b/README.md index 301afb6..7274f20 100644 --- a/README.md +++ b/README.md @@ -12,7 +12,7 @@ This workflow identifies RNA isoforms using either cDNA or direct RNA (dRNA) Oxford Nanopore reads. ### Preprocesing -cDNA reads are initially preprocessed by [pychopper](https://github.com/nanoporetech/pychopper) +cDNA reads are initially preprocessed by [pychopper](https://github.com/epi2me-labs/pychopper) for the identification of full-length reads, as well as trimming and orientation correction (This step is omitted for direct RNA reads). diff --git a/bin/report.py b/bin/report.py index bbc2024..f027b71 100755 --- a/bin/report.py +++ b/bin/report.py @@ -498,7 +498,7 @@ def gff_compare_plots(report, gffcompare_outdirs: Path, sample_ids): def pychopper_plots(report, pychop_report): - """Make plots from pychopper.cdna_classifier.py. + """Make plots from pychopper output. :param report: aplanat WFReport :param pychop_report: path to pychopper stats file @@ -507,7 +507,7 @@ def pychopper_plots(report, pychop_report): section.markdown(''' ### Pychopper summary statisitcs - The following plots summarize the output of [cdna_classifier.py] + The following plots summarize [pychopper] output (https://github.com/nanoporetech/pychopper) * **Pr.found**: Reads with primers found in correct orientation at diff --git a/docs/intro.md b/docs/intro.md index 9a8935b..c532336 100644 --- a/docs/intro.md +++ b/docs/intro.md @@ -4,7 +4,7 @@ This workflow identifies RNA isoforms using either cDNA or direct RNA (dRNA) Oxford Nanopore reads. ### Preprocesing -cDNA reads are initially preprocessed by [pychopper](https://github.com/nanoporetech/pychopper) +cDNA reads are initially preprocessed by [pychopper](https://github.com/epi2me-labs/pychopper) for the identification of full-length reads, as well as trimming and orientation correction (This step is omitted for direct RNA reads). diff --git a/environment.yaml b/environment.yaml index b097162..29e02e5 100644 --- a/environment.yaml +++ b/environment.yaml @@ -6,16 +6,16 @@ channels: - defaults dependencies: - python==3.8.* - - aplanat >=0.6.4 + - aplanat>=0.6.4 - epi2melabs - - minimap2 ==2.24 - - samtools ==1.14 - - bedtools ==2.30.0 - - pychopper==2.5.0 + - minimap2==2.24 + - samtools==1.14 + - bedtools==2.30.0 + - pychopper>=2.7.0 - pandas==1.3.5 - gffread==0.12.7 - gffcompare==0.11.2 - - gffutils=0.10.1 + - gffutils==0.10.1 - seqkit==2.1.0 - stringtie==2.1.1 - curl diff --git a/main.nf b/main.nf index 2f230fd..0f892ae 100644 --- a/main.nf +++ b/main.nf @@ -74,7 +74,7 @@ process getParams { process preprocess_reads { /* Concatenate reads from a sample directory. - Optionally classify, trim, and orient cDNA reads using cdna_classifier from pychopper + Optionally classify, trim, and orient cDNA reads using pychopper */ label "isoforms" @@ -87,13 +87,13 @@ process preprocess_reads { path '*.tsv', emit: report script: """ - cdna_classifier.py -t ${params.threads} ${params.pychopper_opts} ${input_reads} ${sample_id}_full_length_reads.fq - mv cdna_classifier_report.tsv ${sample_id}_cdna_classifier_report.tsv - generate_pychopper_stats.py --data ${sample_id}_cdna_classifier_report.tsv --output . + pychopper -t ${params.threads} ${params.pychopper_opts} ${input_reads} ${sample_id}_full_length_reads.fq + mv pychopper.tsv ${sample_id}_pychopper.tsv + generate_pychopper_stats.py --data ${sample_id}_pychopper.tsv --output . # Add sample id column - sed "1s/\$/\tsample_id/; 1 ! s/\$/\t${sample_id}/" ${sample_id}_cdna_classifier_report.tsv > tmp - mv tmp ${sample_id}_cdna_classifier_report.tsv + sed "1s/\$/\tsample_id/; 1 ! s/\$/\t${sample_id}/" ${sample_id}_pychopper.tsv > tmp + mv tmp ${sample_id}_pychopper.tsv """ } diff --git a/nextflow.config b/nextflow.config index a942659..a6f3263 100644 --- a/nextflow.config +++ b/nextflow.config @@ -23,7 +23,7 @@ params { sample = null sample_sheet = null sanitize_fastq = false - wfversion = "v0.1.2" + wfversion = "v0.1.3" aws_image_prefix = null aws_queue = null report_name = "report" diff --git a/nextflow_schema.json b/nextflow_schema.json index 4e14e9f..1b69ba1 100644 --- a/nextflow_schema.json +++ b/nextflow_schema.json @@ -280,7 +280,7 @@ }, "wfversion": { "type": "string", - "default": "v0.1.2", + "default": "v0.1.3", "hidden": true }, "monochrome_logs": { @@ -295,7 +295,7 @@ } }, "docs": { - "intro": "## Introduction\n\nThis workflow identifies RNA isoforms using either cDNA or direct RNA (dRNA) \nOxford Nanopore reads.\n\n### Preprocesing\ncDNA reads are initially preprocessed by [pychopper](https://github.com/nanoporetech/pychopper) \nfor the identification of full-length reads, as well as trimming and orientation correction (This step is omitted for \n direct RNA reads).\n\n### Reference-aided approach\n* Full length reads are mapped to a supplied reference genome using [minimap2](https://github.com/lh3/minimap2)\n* Transcripts are assembled by [stringtie](http://ccb.jhu.edu/software/stringtie) \nin long read mode (with or without a guide reference annotation) to generate the GFF annotation.\n* The annotation generated by the pipeline is compared to the reference annotation. \nusing [gffcompare](http://ccb.jhu.edu/software/stringtie/gffcompare.shtml)\n\n### de novo-based approach (experimental!)\n* Sequence clusters are generated using [isONclust2](https://github.com/nanoporetech/isONclust2)\n * If a reference genome is supplied, cluster quality metrics are determined by comparing \n with clusters generated from a minimap2 alignment.\n* A consensus sequence for each cluster is generated using [spoa](https://github.com/rvaser/spoa)\n* Three rounds of polishing using racon and minimap2 to give a final polished CDS for each gene.\n* Full-length reads are then mapped to these polished CDS.\n* Transcripts are assembled by stringtie as for the reference-based approach.\n* __Note__: This approach is currently not supported with direct RNA reads.\n\n### Workflow inputs\n- Directory containing cDNA/direct RNA reads. Or a directory containing subdirectories each with reads from different samples\n (in fastq/fastq.gz format)\n- Reference genome in fasta format (required for reference-based assembly).\n- Optional reference annotation in GFF2/3 format.", + "intro": "## Introduction\n\nThis workflow identifies RNA isoforms using either cDNA or direct RNA (dRNA) \nOxford Nanopore reads.\n\n### Preprocesing\ncDNA reads are initially preprocessed by [pychopper](https://github.com/epi2me-labs/pychopper) \nfor the identification of full-length reads, as well as trimming and orientation correction (This step is omitted for \n direct RNA reads).\n\n### Reference-aided approach\n* Full length reads are mapped to a supplied reference genome using [minimap2](https://github.com/lh3/minimap2)\n* Transcripts are assembled by [stringtie](http://ccb.jhu.edu/software/stringtie) \nin long read mode (with or without a guide reference annotation) to generate the GFF annotation.\n* The annotation generated by the pipeline is compared to the reference annotation. \nusing [gffcompare](http://ccb.jhu.edu/software/stringtie/gffcompare.shtml)\n\n### de novo-based approach (experimental!)\n* Sequence clusters are generated using [isONclust2](https://github.com/nanoporetech/isONclust2)\n * If a reference genome is supplied, cluster quality metrics are determined by comparing \n with clusters generated from a minimap2 alignment.\n* A consensus sequence for each cluster is generated using [spoa](https://github.com/rvaser/spoa)\n* Three rounds of polishing using racon and minimap2 to give a final polished CDS for each gene.\n* Full-length reads are then mapped to these polished CDS.\n* Transcripts are assembled by stringtie as for the reference-based approach.\n* __Note__: This approach is currently not supported with direct RNA reads.\n\n### Workflow inputs\n- Directory containing cDNA/direct RNA reads. Or a directory containing subdirectories each with reads from different samples\n (in fastq/fastq.gz format)\n- Reference genome in fasta format (required for reference-based assembly).\n- Optional reference annotation in GFF2/3 format.", "links": "## Useful links\n\n* [nextflow](https://www.nextflow.io/)\n* [docker](https://www.docker.com/products/docker-desktop)\n* [Singularity](https://sylabs.io/singularity/)\n* [conda](https://docs.conda.io/en/latest/miniconda.html)\n* [racon](https://github.com/isovic/racon)\n* [spoa](https://github.com/rvaser/spoa)\n* [inONclust](https://github.com/ksahlin/isONclust)\n* [isONclust2](https://github.com/nanoporetech/isONclust2)" } } \ No newline at end of file diff --git a/reference_assembly.nf b/reference_assembly.nf index ae58d79..0e5dc08 100644 --- a/reference_assembly.nf +++ b/reference_assembly.nf @@ -2,7 +2,7 @@ process map_reads{ /* Map reads to reference using minimap2. Filter reads by mapping quality. - Filter reads where length of poly(A) > max_poly_run at either ends of the read (defined by poly_context) + Filter internally-primed reads. */ label "isoforms" cpus params.threads