Autocycler: long-read consensus assembly for bacterial genomes

preprint OA: closed
📄 Open PDF Full text JSON View at publisher

Abstract

Motivation Long-read sequencing enables complete bacterial genome assemblies, but individual assemblers are imperfect and often produce sequence-level and structural errors. Consensus assembly using Trycycler can improve accuracy, but its lack of automation limits scalability. There is a need for an automated method to generate high-quality consensus bacterial genome assemblies from long-read data. Results We present Autocycler, a command-line tool for generating accurate bacterial genome assemblies by combining multiple alternative long-read assemblies of the same genome. Without requiring user input, Autocycler builds a compacted De Bruijn graph from the input assemblies, clusters and filters contigs, trims overlaps and resolves consensus sequences by selecting the most common variant at each locus. It also supports manual curation when desired, allowing users to refine assemblies in challenging or important cases. In our evaluation using Oxford Nanopore Technologies reads from five bacterial isolates, Autocycler outperformed individual assemblers, automated pipelines and other consensus tools, producing assemblies with lower error rates and improved structural accuracy. Availability and implementation Autocycler is implemented in Rust, open-source and freely available at github.com/rrwick/Autocycler . It runs on Linux and macOS and is extensively documented.
Full text 37,428 characters · extracted from preprint-html · click to expand
Autocycler: long-read consensus assembly for bacterial genomes | bioRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-M677548'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search New Results Autocycler: long-read consensus assembly for bacterial genomes View ORCID Profile Ryan R Wick , View ORCID Profile Benjamin P Howden , View ORCID Profile Timothy P Stinear doi: https://doi.org/10.1101/2025.05.12.653612 Ryan R Wick 1 Department of Microbiology and Immunology, The University of Melbourne at the Peter Doherty Institute for Infection and Immunity , Melbourne, Victoria, Australia 2 Centre for Pathogen Genomics, The University of Melbourne , Parkville, Victoria, Australia Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Ryan R Wick For correspondence: rrwick{at}gmail.com ryan.wick{at}unimelb.edu.au Benjamin P Howden 1 Department of Microbiology and Immunology, The University of Melbourne at the Peter Doherty Institute for Infection and Immunity , Melbourne, Victoria, Australia 2 Centre for Pathogen Genomics, The University of Melbourne , Parkville, Victoria, Australia Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Benjamin P Howden Timothy P Stinear 1 Department of Microbiology and Immunology, The University of Melbourne at the Peter Doherty Institute for Infection and Immunity , Melbourne, Victoria, Australia 2 Centre for Pathogen Genomics, The University of Melbourne , Parkville, Victoria, Australia Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Timothy P Stinear Abstract Full Text Info/History Metrics Supplementary material Data/Code Preview PDF Abstract Motivation Long-read sequencing enables complete bacterial genome assemblies, but individual assemblers are imperfect and often produce sequence-level and structural errors. Consensus assembly using Trycycler can improve accuracy, but its lack of automation limits scalability. There is a need for an automated method to generate high-quality consensus bacterial genome assemblies from long-read data. Results We present Autocycler, a command-line tool for generating accurate bacterial genome assemblies by combining multiple alternative long-read assemblies of the same genome. Without requiring user input, Autocycler builds a compacted De Bruijn graph from the input assemblies, clusters and filters contigs, trims overlaps and resolves consensus sequences by selecting the most common variant at each locus. It also supports manual curation when desired, allowing users to refine assemblies in challenging or important cases. In our evaluation using Oxford Nanopore Technologies reads from five bacterial isolates, Autocycler outperformed individual assemblers, automated pipelines and other consensus tools, producing assemblies with lower error rates and improved structural accuracy. Availability and implementation Autocycler is implemented in Rust, open-source and freely available at github.com/rrwick/Autocycler . It runs on Linux and macOS and is extensively documented. 1. Introduction Complete genome assemblies are essential for resolving bacterial genome structure and fully characterising accessory elements such as plasmids and prophages. 1 , 2 Accurate assemblies reduce the risk of errors in downstream analyses such as comparative genomics, annotation and studies of genome dynamics. Long-read sequencing platforms, such as those from Oxford Nanopore Technologies (ONT), have made complete assemblies of bacterial genomes widely achievable. Long reads can span repetitive elements, allowing assemblers to resolve structural complexity that short reads (e.g. from Illumina platforms) cannot. 3 For most bacterial genomes and high-quality read sets, long-read assemblers can assemble each replicon into a single contig. 4 In practice, however, long-read assemblers are imperfect, and different tools produce different assemblies from the same input read set. Common problems include: incomplete or overlapping circularisation, missing small plasmids, duplicated small plasmids and spurious extra contigs from repeats or contamination. 5 , 6 No single assembler is reliably the best across all datasets. Consensus assembly offers a solution. By combining multiple alternative assemblies of the same genome (e.g. those produced by different assemblers or read subsets) consistent sequences can be distinguished from assembler-specific errors. 7 , 8 The software Trycycler put this idea into practice for bacterial genomes. 9 Compared to assemblies produced by a single tool, Trycycler assemblies usually contain fewer errors, more reliable circularisation and a more complete and less contaminated representation of the genome. 9 Trycycler, however, relies on human interventions and decision-making for several key steps. While this design offers flexibility and control, it limits scalability. As bacterial genomics increasingly involves large datasets of hundreds or thousands of genomes, there is a need for automated methods that can generate high-quality consensus assemblies without manual interventions. Here we present Autocycler, a fully automated command-line tool to generate consensus long-read assemblies of bacterial genomes. Like Trycycler, it combines multiple input assemblies to produce a high-quality consensus. Unlike Trycycler, Autocycler is designed to run to completion without user input. It also supports manual intervention for cases where careful output curation is warranted. For most bacterial genomes and read sets with sufficient depth and read length, Autocycler can produce complete assemblies automatically. In more difficult cases, such as genomes with large repeats, genomic heterogeneity or unusual structures like linear replicons, users can step in to refine the output. 2. Implementation Autocycler constructs a consensus bacterial genome assembly by combining multiple alternative assemblies of the same genome ( Figure 1 ). It is designed to run fully automatically and produces intermediate files and metrics for every step, allowing the process to be inspected or curated if needed. In addition to the main consensus assembly workflow, Autocycler includes commands to assist with upstream and downstream tasks. Download figure Open in new tab Fig. 1. Overview of the Autocycler workflow. By following only the blue steps on the left, Autocycler can produce consensus genome assemblies with no human intervention. The optional orange steps on the right can be used when accuracy is critical or when Autocycler’s metrics (gathered using the autocycler table command) indicate potential issues. The autocycler subsample command creates read subsets for generating input assemblies. By dividing a single long-read set into minimally overlapping subsets, users can generate assemblies that are more independent of one another. Additionally, the subsampling process reduces very high-depth read sets to medium-depth subsets, which often assemble more cleanly. Input assemblies can be generated using any long-read assembler, and Autocycler provides helper scripts for several common ones. Ideally, each replicon in the genome (e.g. chromosome or plasmid) should be assembled as a single contig. While some fragmentation is tolerated, Autocycler relies on the assumption that most input assemblies are complete. If all input assemblies are fragmented, Autocycler will not be able to produce a complete consensus. The autocycler compress command builds a compacted De Bruijn graph from the input assemblies. This graph requires much less disk space than the inputs, as shared regions are collapsed together. Importantly, it stores the full path of each input sequence, allowing the original assemblies to be recovered later using autocycler decompress . The graph representation also provides an efficient structure for manipulating and comparing sequences in downstream steps. In the autocycler cluster step, input contigs are grouped based on similarity using UPGMA clustering. Autocycler then applies quality control filters to exclude low-confidence clusters (e.g. those with too few contigs or contained within other clusters), ideally leaving one high-quality cluster for each replicon in the genome. This step also outputs a Newick-format tree, which users can inspect to decide whether to override the automatic clustering. For each cluster, autocycler trim processes the input contigs to remove unwanted sequence. It looks for both circular overlaps, where the start of a contig overlaps with its end, and hairpin overlaps, where the start or end of a contig extends past the hairpin to the opposite strand. It also handles cases where small plasmids are fully duplicated within a single contig. After trimming, sequences with lengths that deviate too far from the cluster median are discarded, leaving a set of consistent contigs for consensus generation. The autocycler resolve command generates a consensus sequence for each cluster. It begins by identifying anchors: sequences that appear exactly once in each contig. These anchors serve as a scaffold for constructing bridges, which represent the most common paths between anchors in the input contigs. Autocycler first applies unambiguous bridges and then iteratively resolves ambiguous cases by selecting the most supported paths, ideally producing a single consensus sequence for the cluster. In cases of structural heterogeneity, such as phase-variable loci or assembly inconsistencies, Autocycler includes intermediate output for user inspection. Once each cluster has been resolved, the autocycler combine command merges them into a final consensus assembly in both FASTA and GFA formats. Autocycler also produces detailed metrics at every step of the pipeline, saved in YAML format (both human- and machine-readable). The autocycler table command can be used to gather metrics from many assemblies, making it easy to track success and identify samples that require further attention. Autocycler is implemented in Rust and is fast, deterministic and resource-efficient. Most of the computational time in an Autocycler workflow is spent generating input assemblies. Autocycler itself typically completes in minutes and requires only modest resources. It runs on both Linux and macOS, although Linux is preferred due to broader compatibility with long-read assemblers. Extensive documentation, including worked examples and guidance for manual curation, is available online at github.com/rrwick/Autocycler/wiki. 3. Evaluation 3.1 Methods Long-read sequencing of 84 diverse bacterial isolates was performed using an Oxford Nanopore Technologies PromethION 2 Solo using the Rapid Barcoding Kit 96 V14 (SQK-RBK114.96). Reads were basecalled with Dorado v0.9.5 10 using the [email protected] model and filtered to retain reads with mean quality ≥ 10. Five isolates were selected, each from a different genus: Enterobacter hormaechei, Klebsiella pneumoniae, Listeria innocua, Providencia rettgeri and Shigella flexneri . Short-read Illumina sequencing was available for all samples and was used to polish the reference genomes (Table S1). Selection was based on high read depth and preliminary assessments showing no evidence of heterogeneity or divergence between the Illumina and ONT datasets. For each genome, we followed our previously published method to generate a high-accuracy reference assembly. 11 Briefly, the ONT reads were assembled with Trycycler v0.5.5 9 and the resulting genome was polished using Medaka v2.0.1 12 , Polypolish v0.6.0 13 and Pypolca v0.3.1. 14 The resulting assemblies were highly accurate and used as ground truth. ONT reads for each genome were divided into six non-overlapping 50 × subsets for a total of 30 read sets. Each read set was assembled using the following long-read assemblers: Canu v2.3 15 , Flye v2.9.5 16 , LJA v0.2 17 , metaMDBG v1.1 18 , miniasm v0.3 19 , NECAT v0.0.1 20 , NextDenovo v2.5.2 21 , Raven v1.8.3 22 and wtdbg2 v2.5 23 . Each of these tools was run via the helper scripts included with Autocycler, which included extra processing for Canu (overlap-trimming and repeat/bubble removal), metaMDBG (low-depth contig removal), miniasm (polishing with Minipolish 24 ) and NextDenovo (polishing with NextPolish 25 ). In addition, we assembled each read set with the Dragonflye v1.2.1 26 and Hybracter v0.11.2 27 long-read assembly pipelines and the consensus assembly tool MAECI 28 (commit f1eb3d7). For Autocycler, we produced an automated assembly (using its autocycler full.sh script) and a manually curated assembly. We also attempted to evaluate MAC2.0 29 , but it did not perform correctly on complete bacterial genomes, producing outputs with duplicated sequences. Other consensus assembly tools, such as quickmerge 30 and Metassembler 8 , are older and were primarily designed to improve contiguity in fragmented eukaryotic assemblies. These tools are not suitable for refining complete bacterial genomes from long-read data and were therefore excluded from this comparison. 31 Each assembly was compared to its corresponding ground-truth reference using a custom script ( assess_assembly.py ) that aligns the assembly to the reference sequence with minimap2 v2.28 32 and quantifies accuracy metrics including sequence errors (substitutions and indels), missing bases and extra bases. We also assessed assembly accuracy with Inspector v1.3.1 33 and CRAQ v1.0.9 34 , which evaluate assemblies based on read alignments rather than a reference. Full commands are provided in the supplementary data. 3.2 Results Among the single-tool assemblers, Canu and Flye consistently produced the fewest sequence-level errors, typically with fewer than 10 substitutions and indel errors per assembly ( Figure 2A , Table S2). All other long-read assemblers had higher error rates. Across all tools, structural inaccuracies were common ( Figure 2B ), with assemblies often missing genomic elements (e.g. small plasmids) or containing spurious extra sequence (e.g. duplicated ends of circular contigs). Download figure Open in new tab Fig. 2. Assembler benchmarking results from the assess_assembly.py script. (A) Sequence errors (substitutions and indels); (B) Sequence errors and structural assembly errors (missing and extra bases). Lower values on the y-axes indicate better assembly accuracy. Results are coloured by category: individual long-read assembly tools (blue), long-read assembly pipelines (orange) and consensus assembly tools (green). Autocycler results are shown separately for fully automated and manually curated assemblies. Boxplot whiskers extend to the minimum and maximum values. The y-axes use a pseudo-logarithmic scale that accommodates zeros. Of the long-read assembly pipelines, Hybracter outperformed Dragonflye. By integrating Plassembler 35 , Hybracter improves plasmid recovery and avoids structural errors such as duplication of plasmids. However, for the Enterobacter and Klebsiella genomes, Hybracter showed elevated error rates in large plasmids, likely due to errors introduced during plasmid assembly by Unicycler 36 within Plassembler (Table S3). Dragonflye performed worse than Flye alone, likely due to its default use of Racon polishing (instead of Flye’s internal polisher) and the --nano-raw option which is suboptimal for modern ONT reads. With adjusted parameters ( --racon 0 --opts ‘-i 1’ --nanohq ), Dragonflye?s performance matched that of Flye (Table S4).’MAECI was the only consensus assembly tool tested apart from Autocycler. Despite incorporating three input assemblers (Canu, Flye and wtdbg2), MAECI did not consistently outperform Canu or Flye. When run in a fully automated manner, Autocycler produced assemblies with the lowest sequence error counts of any method (median: 3.5 errors per assembly, range: 0−11). It was also structurally accurate in most cases, successfully recovering all replicons for four of the five genomes. This is in part because autocycler_full.sh uses Plassembler when generating input assemblies. However, the smallest plasmid of the Enterobacter genome (2.5 kbp) was occasionally missed. In manually curated runs, the missing plasmid could be identified by inspecting the clustering tree (output by autocycler cluster ) and included in the final assembly by overriding the default clustering. These curated Autocycler assemblies had no structural errors reported by assess_assembly.py , Inspector or CRAQ (Table S2). 4. Discussion and conclusions Consensus assembly offers a clear accuracy advantage over single-tool assembly. In our benchmarking, even the best-performing individual assemblers (Canu and Flye) consistently made avoidable errors. Consensus approaches mitigate such issues by averaging over multiple inputs, reducing both small-scale errors and structural inaccuracies. Trycycler 9 provided a robust framework for generating consensus bacterial genome assemblies, but it requires substantial user intervention, limiting its scalability. Autocycler brings the benefits of consensus assembly into a fully automated workflow, enabling accurate bacterial genome assembly at scale. Autocycler does not guarantee perfect results, as its consensus reflects the input assemblies. If the inputs are fragmented (e.g. due to a repeat longer than the read length), Autocycler cannot produce a fully resolved consensus. When most inputs share the same error, that error can persist into the final assembly. In our evaluation, there were typically fewer than 10 sequence errors (substitutions and indels) per Autocycler assembly (Table S5), most commonly homopolymer-length errors resulting from systematic basecalling issues in ONT reads. 37 The frequency of these errors depends on factors such as pore type (R10.4.1 is more accurate than R9.4.1), basecalling model (sup is more accurate than hac or fast) and bacterial strain. To address these errors, short-read polishing can be applied after Autocycler, yielding hybrid assemblies with maximal accuracy. 14 The only structural error observed in automated Autocycler assemblies in this study was the omission of a small plasmid, which many long-read assemblers fail to recover. 6 , 38 Autocycler supports manual intervention at key steps in its pipeline, enabling users to review the data and apply their own judgement to correct such issues. This flexible design allows Autocycler to function as both a scalable automated tool and as a framework for high-accuracy reference genome assembly. Although not evaluated in this study, linear replicons pose additional challenges for genome assembly. Assemblers may erroneously extend hairpin ends or terminate open ends inconsistently. While Autocycler includes logic to detect and trim hairpin overlaps, full resolution of linear sequences still frequently requires manual intervention. Improved support for such cases remains an area for improvement for both long-read assemblers and Autocycler. Autocycler is open-source, well documented and easy to install. It requires only modest system resources (excluding input assembly generation) and provides intermediate outputs to support transparency and manual curation. It fills a key gap in the current assembly tool landscape: existing consensus tools either underperform or do not scale, while long-read assembly pipelines rely on a single assembler and inherit its limitations. Because it relies on multiple inputs, an Autocycler-based pipeline is more computationally intensive than using a single assembler, but it often yields better assemblies. We therefore recommend Autocycler for long-read bacterial genome projects where maximum assembly accuracy is required. 5. Competing interests The authors declare that there are no conflicts of interest. 6. Author contributions statement RRW conceived and programmed Autocycler with input from TPS. RRW performed the benchmarks and analysed the data. RRW, BPH and TPS wrote and reviewed the manuscript. 7. Funding RRW is supported by an ARC Discovery Early Career Researcher Award (DE250100677). TPS is supported by an NHMRC Research Fellowship (APP1105525) and ARC Discovery Project (DP240102465). BPH is supported by an NHMRC Research Fellowship (APP1196103). 8. Data availability Supplementary figures, tables and methods are available at github.com/rrwick/Autocycler-paper . Assemblies, reference genomes and read sets used in the analysis are available at figshare.unimelb.edu.au/projects/Autocycler/247142 . 9. Acknowledgements This research was performed in part at the Centre for Pathogen Genomics Innovation Hub, Department of Microbiology and Immunology, University of Melbourne at the Peter Doherty Institute for Infection and Immunity. This paper acknowledges the PulseNet Asia-Pacific team at the Centre for Pathogen Genomics and the Microbiological Diagnostic Unit Public Health Laboratory (MDU PHL) for contributing data and isolates to the study. Support for PulseNet Asia-Pacific is funded by the US Centers for Disease Control and Prevention (CDC) Global Antimicrobial Resistance Laboratory and Response Network through the Association of Public Health Laboratories. MDU PHL is funded by the Victorian Government, Australia. We thank the Autocycler alpha testers for their valuable feedback prior to the tool’s public release: Alex Krause, Bogdan Iorga, Dan Whiley, Danielle Ingle, Erin Young, George Bouras, Josh Zhang, Mariel Beiers, Marko Verce, Matthew Croxen, Mona Taouk, Munazzah Maqbool, Oliver Schwengers, Sarah Baines, Steve Baeyen, Sudaraka Mallawaarachchi, Tatum Mortimer, Tue Sparholt Jørgensen, Tung Trinh and Ying Xu. Funder Information Declared Australian Research Council, https://ror.org/05mmh0f86 , DE250100677 National Health and Medical Research Council , APP1105525 , APP1196103 Footnotes https://github.com/rrwick/Autocycler-paper https://figshare.unimelb.edu.au/projects/Autocycler/247142 References 1. ↵ Arun Gonzales Decano , Catherine Ludden , Theresa Feltwell , Kim Judge , Julian Parkhill , and Tim Downing . Complete assembly of Escherichia coli sequence type 131 genomes using long reads demonstrates antibiotic resistance gene variation within diverse plasmid and chromosomal contexts . mSphere , June 2019 . doi: 10.1128/mSphere.00130-19 . OpenUrl Abstract / FREE Full Text 2. ↵ Emily L. Gulliver , Vicki Adams , Vanessa Rossetto Marcelino , Jodee Gould , Emily L. Rutten , David R. Powell , Remy B. Young , Gemma L. D’Adamo , Jamia Hemphill , Sean M. Solari , Sarah A. Revitt-Mills , Samantha Munn , Thanavit Jirapanjawat , Chris Greening , Jennifer C. Boer , Katie L. Flanagan , Magne Kaldhusdal , Magdalena Plebanski , Katherine B. Gibney , Robert J. Moore , Julian I. Rood , and Samuel C. Forster . Extensive genome analysis identifies novel plasmid families in Clostridium perfringens . Microbial Genomics , April 2023 . doi: 10.1099/mgen.0.000995 . OpenUrl CrossRef 3. ↵ Sergey Koren , Gregory P. Harhay , Timothy P. L. Smith , James L. Bono , Dayna M. Harhay , Scott D. Mcvey , Diana Radune , Nicholas H. Bergman , and Adam M. Phillippy . Reducing assembly complexity of microbial genomes with single-molecule sequencing . Genome Biology , September 2013 . doi: 10.1186/gb-2013-14-9-r101 . OpenUrl CrossRef PubMed 4. ↵ Ryan R. Wick and Kathryn E. Holt . Benchmarking of long-read assemblers for prokaryote whole genome sequencing . F1000Research , December 2019 . doi: 10.12688/f1000research.21782.4 . OpenUrl CrossRef PubMed 5. ↵ Ian Boostrom , Edward A. R. Portal , Owen B. Spiller , Timothy R. Walsh , and Kirsty Sands . Comparing long-read assemblers to explore the potential of a sustainable low-cost, low-infrastructure approach to sequence antimicrobial resistant bacteria with Oxford Nanopore sequencing . Frontiers in Microbiology , March 2022 . doi: 10.3389/fmicb.2022.796465 . OpenUrl CrossRef 6. ↵ Jared Johnson , Marty Soehnlen , and Heather M. Blankenship . Long read genome assemblers struggle with small plasmids . Microbial Genomics , May 2023 . doi: 10.1099/mgen.0.001024 . OpenUrl CrossRef 7. ↵ Shin-Hung Lin and Yu-Chieh Liao . CISA: Contig integrator for sequence assembly of bacterial genomes . PLoS ONE , March 2013 . doi: 10.1371/journal.pone.0060843 . OpenUrl CrossRef PubMed 8. ↵ Alejandro Hernandez Wences and Michael C. Schatz . Metassembler: merging and optimizing de novo genome assemblies . Genome Biology , December 2015 . doi: 10.1186/s13059-015-0764-4 . OpenUrl CrossRef PubMed 9. ↵ Ryan R. Wick , Louise M. Judd , Louise T. Cerdeira , Jane Hawkey , Guillaume Méric , Ben Vezina , Kelly L. Wyres , and Kathryn E. Holt . Trycycler: consensus long-read assemblies for bacterial genomes . Genome Biology , December 2021 . doi: 10.1186/s13059-021-02483-z . OpenUrl CrossRef PubMed 10. ↵ Oxford Nanopore Technologies PLC . Dorado : Oxford Nanopore’s basecaller . https://github.com/nanoporetech/dorado , 2025 . Version 0.9.5. 11. ↵ Ryan R. Wick , Louise M. Judd , and Kathryn E. Holt . Assembling the perfect bacterial genome using Oxford Nanopore and Illumina sequencing . PLOS Computational Biology , March 2023 . doi: 10.1371/journal.pcbi.1010905 . OpenUrl CrossRef PubMed 12. ↵ Oxford Nanopore Technologies PLC . Medaka : Sequence correction provided by ONT Research . https://github.com/nanoporetech/medaka , 2024 . Version 2.0.1. 13. ↵ Ryan R. Wick and Kathryn E. Holt . Polypolish: short-read polishing of long-read bacterial genome assemblies . PLOS Computational Biology , January 2022 . doi: 10.1371/journal.pcbi.1009802 . OpenUrl CrossRef PubMed 14. ↵ George Bouras , Louise M. Judd , Robert A. Edwards , Sarah Vreugde , Timothy P. Stinear , and Ryan R. Wick . How low can you go? Short-read polishing of Oxford Nanopore bacterial genome assemblies . Microbial Genomics , June 2024 . doi: 10.1099/mgen.0.001254 . OpenUrl CrossRef 15. ↵ Sergey Koren , Brian P. Walenz , Konstantin Berlin , Jason R. Miller , Nicholas H. Bergman , and Adam M. Phillippy . Canu: scalable and accurate long-read assembly via adaptive k-mer weighting and repeat separation . Genome Research , May 2017 . doi: 10.1101/gr.215087.116 . OpenUrl Abstract / FREE Full Text 16. ↵ Mikhail Kolmogorov , Jeffrey Yuan , Yu Lin , and Pavel A. Pevzner . Assembly of long, error-prone reads using repeat graphs . Nature Biotechnology , May 2019 . doi: 10.1038/s41587-019-0072-8 . OpenUrl CrossRef 17. ↵ Anton Bankevich , Andrey V. Bzikadze , Mikhail Kolmogorov , Dmitry Antipov , and Pavel A. Pevzner . Multiplex de Bruijn graphs enable genome assembly from long, high-fidelity reads . Nature Biotechnology , July 2022 . doi: 10.1038/s41587-022-01220-6 . OpenUrl CrossRef 18. ↵ Gäetan Benoit , Sébastien Raguideau , Robert James , Adam M. Phillippy , Rayan Chikhi , and Christopher Quince . High-quality metagenome assembly from long accurate reads with metaMDBG . Nature Biotechnology , September 2024 . doi: 10.1038/s41587-023-01983-6 . OpenUrl CrossRef 19. ↵ Heng Li . Minimap and miniasm: fast mapping and de novo assembly for noisy long sequences . Bioinformatics , July 2016 . doi: 10.1093/bioinformatics/btw152 . OpenUrl CrossRef PubMed 20. ↵ Ying Chen , Fan Nie , Shang-Qian Xie , Ying-Feng Zheng , Qi Dai , Thomas Bray , Yao-Xin Wang , Jian-Feng Xing , Zhi-Jian Huang , De-Peng Wang , Li-Juan He , Feng Luo , Jian-Xin Wang , Yi-Zhi Liu , and Chuan-Le Xiao . Efficient assembly of nanopore reads via highly accurate and intact error correction . Nature Communications , December 2021 . doi: 10.1038/s41467-020-20236-7 . OpenUrl CrossRef PubMed 21. ↵ Jiang Hu , Zhuo Wang , Zongyi Sun , Benxia Hu , Adeola Oluwakemi Ayoola , Fan Liang , Jingjing Li , José R. Sandoval , David N. Cooper , Kai Ye , Jue Ruan , Chuan-Le Xiao , Depeng Wang , Dong-Dong Wu , and Sheng Wang . NextDenovo: an efficient error correction and accurate assembly tool for noisy long reads . Genome Biology , April 2024 . doi: 10.1186/s13059-024-03252-4 . OpenUrl CrossRef PubMed 22. ↵ Robert Vaser and Mile Šikić . Time- and memory-efficient genome assembly with Raven . Nature Computational Science , May 2021 . doi: 10.1038/s43588-021-00073-4 . OpenUrl CrossRef PubMed 23. ↵ Jue Ruan and Heng Li . Fast and accurate long-read assembly with wtdbg2 . Nature Methods , February 2020 . doi: 10.1038/s41592-019-0669-3 . OpenUrl CrossRef PubMed 24. ↵ Ryan R. Wick . Minipolish: A tool for Racon polishing of miniasm assemblies . https://github.com/rrwick/Minipolish , 2020 . xVersion 0.1.3. 25. ↵ Jiang Hu , Junpeng Fan , Zongyi Sun , and Shanlin Liu . NextPolish: a fast and efficient genome polishing tool for long-read assembly . Bioinformatics , April 2020 . doi: 10.1093/bioinformatics/btz891 . OpenUrl CrossRef 26. ↵ Robert A. Petit III . . Dragonflye: Assemble bacterial isolate genomes from nanopore reads . https://github.com/rpetit3/dragonflye , 2024 . xVersion 1.2.1. 27. ↵ George Bouras , Ghais Houtak , Ryan R. Wick , Vijini Mallawaarachchi , Michael J. Roach , Bhavya Papudeshi , Louise M. Judd , Anna E. Sheppard , Robert A. Edwards , and Sarah Vreugde . Hybracter: enabling scalable, automated, complete and accurate bacterial genome assemblies . Microbial Genomics , May 2024 . doi: 10.1099/mgen.0.001244 . OpenUrl CrossRef PubMed 28. ↵ Jidong Lang . MAECI: A pipeline for generating consensus sequence with nanopore sequencing long-read assembly and error correction . PLOS ONE , May 2022 . doi: 10.1371/journal.pone.0267066 . OpenUrl CrossRef 29. ↵ Li Tang , Min Li , Fang-Xiang Wu , Yi Pan , and Jianxin Wang . MAC: Merging assemblies by using adjacency algebraic model and classification . Frontiers in Genetics , January 2020 . doi: 10.3389/fgene.2019.01396 . OpenUrl CrossRef PubMed 30. ↵ Mahul Chakraborty , James G. Baldwin-Brown , Anthony D. Long , and J. J. Emerson . Contiguous and accurate de novo assembly of metazoan genomes with modest long read coverage . Nucleic Acids Research , July 2016 . doi: 10.1093/nar/gkw654 . OpenUrl CrossRef PubMed 31. ↵ Hind Alhakami , Hamid Mirebrahim , and Stefano Lonardi . A comparative evaluation of genome assembly reconciliation tools . Genome Biology , December 2017 . doi: 10.1186/s13059-017-1213-3 . OpenUrl CrossRef PubMed 32. ↵ Heng Li . Minimap2: pairwise alignment for nucleotide sequences . Bioinformatics , September 2018 . doi: 10.1093/bioinformatics/bty191 . OpenUrl CrossRef PubMed 33. ↵ Yu Chen , Yixin Zhang , Amy Y. Wang , Min Gao , and Zechen Chong . Accurate long-read de novo assembly evaluation with Inspector . Genome Biology , November 2021 . doi: 10.1186/s13059-021-02527-4 . OpenUrl CrossRef PubMed 34. ↵ Kunpeng Li , Peng Xu , Jinpeng Wang , Xin Yi , and Yuannian Jiao . Identification of errors in draft genome assemblies at single-nucleotide resolution for quality assessment and improvement . Nature Communications , October 2023 . doi: 10.1038/s41467-023-42336-w . OpenUrl CrossRef PubMed 35. ↵ George Bouras , Anna E. Sheppard , Vijini Mallawaarachchi , and Sarah Vreugde . Plassembler: an automated bacterial plasmid assembly tool . Bioinformatics , July 2023 . doi: 10.1093/bioinformatics/btad409 . OpenUrl CrossRef 36. ↵ Ryan R. Wick , Louise M. Judd , Claire L. Gorrie , and Kathryn E. Holt . Unicycler: resolving bacterial genome assemblies from short and long sequencing reads . PLOS Computational Biology , June 2017 . doi: 10.1371/journal.pcbi.1005595 . OpenUrl CrossRef PubMed 37. ↵ Mantas Sereika , Rasmus Hansen Kirkegaard , Søren Michael Karst , Thomas Yssing Michaelsen , Emil Aarre Sørensen , Rasmus Dam Wollenberg , and Mads Albertsen . Oxford Nanopore R10.4 long-read sequencing enables the generation of near-finished bacterial genomes from pure cultures and metagenomes without short-read or reference polishing . Nature Methods , July 2022 . doi: 10.1038/s41592-022-01539-7 . OpenUrl CrossRef PubMed 38. ↵ Nicole Lerminiaux , Ken Fakharuddin , Michael R. Mulvey , and Laura Mataseje . Do we still need Illumina sequencing data? Evaluating Oxford Nanopore Technologies R10.4.1 flow cells and the Rapid v14 library prep kit for Gram negative bacteria whole genome assemblies . Canadian Journal of Microbiology , May 2024 . doi: 10.1139/cjm-2023-0175 . OpenUrl CrossRef PubMed View the discussion thread. Back to top Previous Next Posted May 15, 2025. Download PDF Supplementary Material Data/Code Email Thank you for your interest in spreading the word about bioRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following Autocycler: long-read consensus assembly for bacterial genomes Message Subject (Your Name) has forwarded a page to you from bioRxiv Message Body (Your Name) thought you would like to see this page from the bioRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share Autocycler: long-read consensus assembly for bacterial genomes Ryan R Wick , Benjamin P Howden , Timothy P Stinear bioRxiv 2025.05.12.653612; doi: https://doi.org/10.1101/2025.05.12.653612 Share This Article: Copy Citation Tools Autocycler: long-read consensus assembly for bacterial genomes Ryan R Wick , Benjamin P Howden , Timothy P Stinear bioRxiv 2025.05.12.653612; doi: https://doi.org/10.1101/2025.05.12.653612 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Bioinformatics Subject Areas All Articles Animal Behavior and Cognition (7629) Biochemistry (17660) Bioengineering (13881) Bioinformatics (41911) Biophysics (21436) Cancer Biology (18578) Cell Biology (25482) Clinical Trials (138) Developmental Biology (13371) Ecology (19887) Epidemiology (2067) Evolutionary Biology (24302) Genetics (15599) Genomics (22483) Immunology (17728) Microbiology (40364) Molecular Biology (17163) Neuroscience (88537) Paleontology (666) Pathology (2830) Pharmacology and Toxicology (4821) Physiology (7637) Plant Biology (15129) Scientific Communication and Education (2045) Synthetic Biology (4290) Systems Biology (9817) Zoology (2269)

Text is read by the "Ask this paper" AI Q&A widget below. Extraction quality varies by source — PMC NXML preserves structure cleanly, OA-HTML may include some navigation residue, and OA-PDF can have broken hyphenation. The publisher copy (via DOI) is the canonical version.

My notes (saved in your browser only)

Ask this paper AI returns verbatim quotes from the full text · source: preprint-html

Answers must be backed by verbatim quotes from this paper's full text. Hallucinated quotes are dropped automatically; if no verbatim passage answers the question, we say so. How this works

Citation neighborhood (no data yet)

We don't have any in-corpus citations linked to this paper yet. This is a recent paper (2025) — citers typically take a year or two to land, and the OpenAlex reference graph may still be filling in.

Source provenance

europepmc
last seen: 2026-05-20T01:45:00.602351+00:00