Assembly of long, error-prone reads using repeat graphs. Kolmogorov, M., Yuan, J., Lin, Y., & Pevzner, P. A. Nature Biotechnology, 37(5):540–546, May, 2019. doi abstract bibtex Accurate genome assembly is hampered by repetitive regions. Although long single molecule sequencing reads are better able to resolve genomic repeats than short-read data, most long-read assembly algorithms do not provide the repeat characterization necessary for producing optimal assemblies. Here, we present Flye, a long-read assembly algorithm that generates arbitrary paths in an unknown repeat graph, called disjointigs, and constructs an accurate repeat graph from these error-riddled disjointigs. We benchmark Flye against five state-of-the-art assemblers and show that it generates better or comparable assemblies, while being an order of magnitude faster. Flye nearly doubled the contiguity of the human genome assembly (as measured by the NGA50 assembly quality metric) compared with existing assemblers.
@article{kolmogorov_assembly_2019,
title = {Assembly of long, error-prone reads using repeat graphs},
volume = {37},
issn = {1546-1696},
doi = {10.1038/s41587-019-0072-8},
abstract = {Accurate genome assembly is hampered by repetitive regions. Although long single molecule sequencing reads are better able to resolve genomic repeats than short-read data, most long-read assembly algorithms do not provide the repeat characterization necessary for producing optimal assemblies. Here, we present Flye, a long-read assembly algorithm that generates arbitrary paths in an unknown repeat graph, called disjointigs, and constructs an accurate repeat graph from these error-riddled disjointigs. We benchmark Flye against five state-of-the-art assemblers and show that it generates better or comparable assemblies, while being an order of magnitude faster. Flye nearly doubled the contiguity of the human genome assembly (as measured by the NGA50 assembly quality metric) compared with existing assemblers.},
language = {eng},
number = {5},
journal = {Nature Biotechnology},
author = {Kolmogorov, Mikhail and Yuan, Jeffrey and Lin, Yu and Pevzner, Pavel A.},
month = may,
year = {2019},
pmid = {30936562},
keywords = {Algorithms, Software, Genomics, Genome, Bacterial, Humans, Sequence Analysis, DNA, Genome, Human, Repetitive Sequences, Nucleic Acid, High-Throughput Nucleotide Sequencing, Molecular Sequence Annotation},
pages = {540--546},
file = {Version soumise:C\:\\Users\\qcarrade\\Zotero\\storage\\PNG86DC7\\Kolmogorov et al. - 2019 - Assembly of long, error-prone reads using repeat g.pdf:application/pdf},
}
Downloads: 0
{"_id":"57o2K8tkvzxCNNt2Y","bibbaseid":"kolmogorov-yuan-lin-pevzner-assemblyoflongerrorpronereadsusingrepeatgraphs-2019","author_short":["Kolmogorov, M.","Yuan, J.","Lin, Y.","Pevzner, P. A."],"bibdata":{"bibtype":"article","type":"article","title":"Assembly of long, error-prone reads using repeat graphs","volume":"37","issn":"1546-1696","doi":"10.1038/s41587-019-0072-8","abstract":"Accurate genome assembly is hampered by repetitive regions. Although long single molecule sequencing reads are better able to resolve genomic repeats than short-read data, most long-read assembly algorithms do not provide the repeat characterization necessary for producing optimal assemblies. Here, we present Flye, a long-read assembly algorithm that generates arbitrary paths in an unknown repeat graph, called disjointigs, and constructs an accurate repeat graph from these error-riddled disjointigs. We benchmark Flye against five state-of-the-art assemblers and show that it generates better or comparable assemblies, while being an order of magnitude faster. Flye nearly doubled the contiguity of the human genome assembly (as measured by the NGA50 assembly quality metric) compared with existing assemblers.","language":"eng","number":"5","journal":"Nature Biotechnology","author":[{"propositions":[],"lastnames":["Kolmogorov"],"firstnames":["Mikhail"],"suffixes":[]},{"propositions":[],"lastnames":["Yuan"],"firstnames":["Jeffrey"],"suffixes":[]},{"propositions":[],"lastnames":["Lin"],"firstnames":["Yu"],"suffixes":[]},{"propositions":[],"lastnames":["Pevzner"],"firstnames":["Pavel","A."],"suffixes":[]}],"month":"May","year":"2019","pmid":"30936562","keywords":"Algorithms, Software, Genomics, Genome, Bacterial, Humans, Sequence Analysis, DNA, Genome, Human, Repetitive Sequences, Nucleic Acid, High-Throughput Nucleotide Sequencing, Molecular Sequence Annotation","pages":"540–546","file":"Version soumise:C\\:\\\\Users\\\\qcarrade\\\\Zotero\\\\storage\\¶NG86DC7\\\\Kolmogorov et al. - 2019 - Assembly of long, error-prone reads using repeat g.pdf:application/pdf","bibtex":"@article{kolmogorov_assembly_2019,\n\ttitle = {Assembly of long, error-prone reads using repeat graphs},\n\tvolume = {37},\n\tissn = {1546-1696},\n\tdoi = {10.1038/s41587-019-0072-8},\n\tabstract = {Accurate genome assembly is hampered by repetitive regions. Although long single molecule sequencing reads are better able to resolve genomic repeats than short-read data, most long-read assembly algorithms do not provide the repeat characterization necessary for producing optimal assemblies. Here, we present Flye, a long-read assembly algorithm that generates arbitrary paths in an unknown repeat graph, called disjointigs, and constructs an accurate repeat graph from these error-riddled disjointigs. We benchmark Flye against five state-of-the-art assemblers and show that it generates better or comparable assemblies, while being an order of magnitude faster. Flye nearly doubled the contiguity of the human genome assembly (as measured by the NGA50 assembly quality metric) compared with existing assemblers.},\n\tlanguage = {eng},\n\tnumber = {5},\n\tjournal = {Nature Biotechnology},\n\tauthor = {Kolmogorov, Mikhail and Yuan, Jeffrey and Lin, Yu and Pevzner, Pavel A.},\n\tmonth = may,\n\tyear = {2019},\n\tpmid = {30936562},\n\tkeywords = {Algorithms, Software, Genomics, Genome, Bacterial, Humans, Sequence Analysis, DNA, Genome, Human, Repetitive Sequences, Nucleic Acid, High-Throughput Nucleotide Sequencing, Molecular Sequence Annotation},\n\tpages = {540--546},\n\tfile = {Version soumise:C\\:\\\\Users\\\\qcarrade\\\\Zotero\\\\storage\\\\PNG86DC7\\\\Kolmogorov et al. - 2019 - Assembly of long, error-prone reads using repeat g.pdf:application/pdf},\n}\n\n","author_short":["Kolmogorov, M.","Yuan, J.","Lin, Y.","Pevzner, P. A."],"key":"kolmogorov_assembly_2019","id":"kolmogorov_assembly_2019","bibbaseid":"kolmogorov-yuan-lin-pevzner-assemblyoflongerrorpronereadsusingrepeatgraphs-2019","role":"author","urls":{},"keyword":["Algorithms","Software","Genomics","Genome","Bacterial","Humans","Sequence Analysis","DNA","Genome","Human","Repetitive Sequences","Nucleic Acid","High-Throughput Nucleotide Sequencing","Molecular Sequence Annotation"],"metadata":{"authorlinks":{}},"html":""},"bibtype":"article","biburl":"https://bibbase.org/network/files/WuqK2wWLHYLdqzePj","dataSources":["J6aT4umod2y9wFwS7","FCxvnsDatnWhcZ8mn","ohyeaCmZPewmvBHRd"],"keywords":["algorithms","software","genomics","genome","bacterial","humans","sequence analysis","dna","genome","human","repetitive sequences","nucleic acid","high-throughput nucleotide sequencing","molecular sequence annotation"],"search_terms":["assembly","long","error","prone","reads","using","repeat","graphs","kolmogorov","yuan","lin","pevzner"],"title":"Assembly of long, error-prone reads using repeat graphs","year":2019}