Full text
39,433 characters
· extracted from
preprint-html
· click to expand
Bacterial genome reduction as a result of short read sequence assembly | bioRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-M677548'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search New Results Bacterial genome reduction as a result of short read sequence assembly Charles H.D. Williamson , Andrew Sanchez , Adam Vazquez , Joshua Gutman , Jason W. Sahl doi: https://doi.org/10.1101/091314 Charles H.D. Williamson Find this author on Google Scholar Find this author on PubMed Search for this author on this site Andrew Sanchez Find this author on Google Scholar Find this author on PubMed Search for this author on this site Adam Vazquez Find this author on Google Scholar Find this author on PubMed Search for this author on this site Joshua Gutman Find this author on Google Scholar Find this author on PubMed Search for this author on this site Jason W. Sahl Find this author on Google Scholar Find this author on PubMed Search for this author on this site Abstract Full Text Info/History Metrics Supplementary material Preview PDF Abstract High-throughput comparative genomics has changed our view of bacterial evolution and relatedness. Many genomic comparisons, especially those regarding the accessory genome that is variably conserved across strains in a species, are performed using assembled genomes. For completed genomes, an assumption is made that the entire genome was incorporated into the genome assembly, while for draft assemblies, often constructed from short sequence reads, an assumption is made that genome assembly is an approximation of the entire genome. To understand the potential effects of short read assemblies on the estimation of the complete genome, we downloaded all completed bacterial genomes from GenBank, simulated short reads, assembled the simulated short reads and compared the resulting assembly to the completed assembly. Although most simulated assemblies demonstrated little reduction, others were reduced by as much as 25%, which was correlated with the repeat structure of the genome. A comparative analysis of lost coding region sequences demonstrated that up to 48 CDSs or up to ~112,000 bases of coding region sequence, were missing from some draft assemblies compared to their finished counterparts. Although this effect was observed to some extent in 32% of genomes, only minimal effects were observed on pan-genome statistics when using simulated draft genome assemblies. The benefits and limitations of using draft genome assemblies should be fully realized before interpreting data from assembly-based comparative analyses. Introduction Advances in DNA sequencing technologies have allowed for large-scale whole genome sequencing of bacterial genomes. Short read technologies, such as those employed on the Illumina sequencing platforms, have facilitated high-throughput analyses of organisms for the purposes of comparative genomics ( 1 ), phylogeography ( 2 ), and association of genomic attributes with antimicrobial resistance ( 3 ). While reference-guided methods, including the identification of single nucleotide polymorphisms (SNPs), are important for understanding population genetics ( 4 ), many analyses are typically performed with assembled genomes making genome assembly an important and standard method in the analysis of bacterial organisms. Studies that rely on assembled genomes include analyzing the conservation of genomic features within a set of isolates and estimating core and pan-genomes. Core and pan-genome analyses, introduced by Tettelin and colleagues ( 5 ), have been applied to many bacterial species ( 6 ), and a number of tools have been developed to calculate and analyze the pan-genome ( 7 – 13 ). All of these tools rely on assembled genomes (or protein/nucleotide sequences from assemblies) as input. Most of the assembled genomes currently available in public databases are draft assemblies. Of the approximately 80,000 bacterial genomes available from NCBI on November 1, 2016, less than 6000 are complete. As assemblies generated from short read sequencing data have become an integral part of many research projects, potential limitations of this type of data must be considered. For instance, contaminating reads can be incorporated into assemblies ( 14 – 16 ) requiring post-assembly screening and quality control. Additionally, genome assemblies generated from short read technologies are typically fragmented due to the inability of short reads (and insert regions) to span large repeat regions of a genome ( 17 ), which often breaks assemblies into multiple contigs. This fragmentation can drop genomic regions from an assembly, which look like missing regions in comparative analyses. In this study, we evaluated how well assemblies generated from short read data estimate complete bacterial genomes. Methods and Materials Complete Genomes Used . We downloaded (September 16, 2016) all bacterial genomes from GenBank, then filtered the genomes to only include completed assemblies (n=5676). We then filtered out genomes that contained >10 non-nucleotide characters (non A,T,G,C), which could indicate problems with genome assembly (n=203). A complete list of genomes (n=5473) used in this study is shown in Table S1. Read simulation . Paired end illumina reads were simulated for each complete genome using ART ( 18 ) vMountRainier with the following parameters: -ss MSv3 - l 250 -f 75 -m 300 -s 30. Genomes were then assembled with SPAdes v3.7.1 ( 19 ) using the following parameters: -t 4 -k 21,33,55,77,99,127 -cov-cutoff auto - careful -1 pair1 -2 pair2. Following assembly, genomes were polished with Pilon 2 v.1.7, using the following parameters: -threads 4 -fix all,amb. Contigs shorter than 200bp were filtered from the assembly to stay consistent with GenBank standards. The genome assembly was automated with the UGAP assembly pipeline ( https://github.com/jasonsahl/UGAP ), which was run using the Slurm management system on a high-performance computing (HPC) cluster at Northern Arizona University. In order to identify how well the simulated reads represented the completed genomes, we mapped the reads to the completed genome with BWA-MEM ( 20 ). The per base coverage was calculated with the GenomeCoverageBed method in BEDTOOLS ( 21 ). The number of bases with a minimum coverage of 1 was then divided by the total number of bases in the completed genome to calculate the percent coverage of simulated reads across each genome. Genome validation . In addition to simulated reads, we also analyzed a set of 49 complete, or near complete, genomes that have been assembled separately with both Illumina and PacBio sequencing platforms (Table S2). To test the ability of ART to simulate representative short sequencing reads, we ran the Illumina reads through SPAdes using the same parameters as with the simulated reads. Genome size calculation. For each genome, we summed the entire sequence length across all sequences with a Python script ( https://gist.github.com/jasonsahl/64d88d2858a915ee730b5f86e305e5d4 ). We divided the size of the simulated assembly by the size of the completed assembly to determine the amount of the genome retained. Repeat characterization . To identify the percentage of the genome associated with repeated regions, we aligned each genome against itself with NUCmer ( 22 ). We then divided the number of bases in repeated regions by the total length of the genome to characterize the repeat percentage. The identification of repeat regions was facilitated by methods implemented in the NASP pipeline ( 4 ). Using default parameters, NUCmer is unable to detect repeats shorter than 21 nucleotides. Multi-locus Sequence Typing comparisons . The sequence type of E. coli and S. aureus assemblies was identified using the PubMLST system and a custom script ( https://github.com/jasonsahl/mlst_blast.git ). Each allele was assigned if an exact match to the database was observed. Comparative genomics. To identify the impact of regions collapsed or lost during the genome assembly using simulated reads, a large-scale Blast Score Ratio (LS-BSR) ( 12 , 23 ) analysis was performed. Coding regions were predicted from the completed genome and the simulated genome with Prodigal ( 24 ). All coding regions were clustered with USEARCH ( 25 ) at an ID of 0.9 and aligned against both genomes with BLAT ( 26 ). The BSR values were then compared between the simulated and the completed genome to identify the number and combined length of regions that had a BSR value > 0.8 (~80% peptide identity over 100% of the peptide length) in the completed genome and a BSR value < 3 0.4 in the simulated genome. These regions represent those that were lost from the assembly and could confound comparative analysis using genome assemblies of short read sequence data. Publicly available genomes . To characterize the quality of all genomes from a single species in public databases, all Escherichia coli genome assemblies (n=4842) were downloaded on September 16, 2016. Genomes were assessed for contig number and assembly size. Results Extent of genome reduction using simulated, short read assemblies . In order to understand the effects of short read assembly on the retention of sequence from bacterial genomes, we downloaded all completed genomes from GenBank with fewer than 10 ambiguities (n=5473) (Table S1) and simulated paired-end Illumina MiSeq reads with ART ( 18 ) at an average coverage of 75x. We assembled all genomes with SPAdes as it performs well compared to other assemblers ( 27 ), it recovers larger portions of reference genomes than other short read assembly algorithms ( 28 ), and we wanted to keep the assembly algorithm constant. The sizes of the complete and the simulated genomes were compared to understand the extent of reduction due to assembly problems. While the vast majority of the genome was recovered in most cases, some genomes showed significant reduction due to short read assembly ( Figure 1 , Table S3). The maximum percentage of observed genome reduction was approximately 25% in Orientia tsutsugamushi, which has been described as having one of the most duplicated genomes ( 29 ). In some cases, the simulated genome assembly was slightly larger than the complete genome (maximum of ~0.76% larger), which may be due to the presence of contigs in the simulated genome that should have been merged during assembly. Download figure Open in new tab Figure 1: Frequency plot of the number of genomes against the extent of genome reduction. We then calculated the breadth of coverage of the completed genome, at a minimum depth of 1x, with simulated reads (Table S3). The breadth of coverage was meant to estimate how well the simulated reads represented the complete genome. Breadth of coverage values range from approximately 73% to 100%. A correlation of breadth of coverage and genome reduction (correlation coefficient=0.76) demonstrates that different methods (genome assembly and short read mapping) return a similar result ( Figure 2 , Table S3). Download figure Open in new tab Figure 2: Breadth of coverage of simulated sequencing reads across complete genomes compared to the genome reduction of simulated genome assemblies compared to completed genomes. Genome reduction using actual sequence data . To confirm that genome reduction wasn‘t solely due to the short read simulation, a set of 49 complete or near complete Burkholderia genomes ( 30 ) was compared to the same isolates where the genomes were also sequenced on the Illumina MiSeq platform. When the genome reduction percentages were compared between real and simulated reads, similar results were observed (correlation coefficient=0.50) ( Table 1 ). In some cases, the Illumina assembly was larger than the completed genome, which may be due to bleed over between multiplexed samples on the same sequencing run ( 31 ) or assembly error. This analysis demonstrates that the simulated short reads should be generally representative of the extent of genome reduction across other species. View this table: View inline View popup Download powerpoint Table 1: Correlations between simulated and true assemblies Repeat structure of all genomes . In order to understand the repeat structure of each genome, NUCmer self-alignments were performed on all genomes and the summed repeat regions were divided by the entire genome length. The results demonstrate that several of the genomes with a high level of reduction were also highly repetitive ( Figure 3 ). In general, genomes with a low level of repeats also had a low level of reduction. The inability to span repeats largely explains the reduction in genome size following genome assembly. As mentioned above, genome reduction is correlated to breadth of coverage (short read mapping), which highlights the limitations of short reads in resolving repeats using independent approaches. Download figure Open in new tab Figure 3: A plot of the % of the genome that is repetitive against the % of the genome that is reduced Draft genome assembly effects on comparative genomics . The potential effects of a reduced genome on comparative genomics was investigated using LS-BSR. The number and length of regions that were missing from the simulated genome was calculated (Table S3). In 3729 of 5473 queried genomes, there were no coding regions (CDSs) that were missing from the simulated genome compared to the completed genome, despite seeing simulated genome assembly sizes that were up to 16% smaller than the completed genome. Of all simulated genomes, 780 were missing more than one CDS identified in the complete genome. The maximum number of CDSs missing from a simulated genome compared to a completed genome was 48, while the maximum length of coding region sequence lost in any genome was approximately 112,000 nucleotides. Reads were aligned to CDSs identified in complete genome assemblies but missing from simulated genome assemblies to determine if short read alignment could be used to verify the presence or absence of CDSs in a genome. The breadth of coverage was determined at a minimum depth of 1X as described above. In two test cases (GCA_000017805.1, missing 48 CDSs in the simulated genome; GCA_000147815.3, missing 8 CDSs corresponding to ~112,000 nucleotides), all missing CDSs were at least partially covered by simulated reads (minimum of ~46% coverage breadth). Mapped reads provided 100% breadth of coverage for 50 of the 56 CDSs evaluated for both genomes, which suggests that read mapping is a valuable method for confirming the presence/absence of potentially missing genomic features. Draft genome assembly effects on pan-genome calculations. The effect of genome reduction on core and pan genome calculations was identified in an analysis of Escherichia coli , Staphylococcus aureus , and Salmonella enterica , species for which numerous (>100) complete genomes are available. In each case, the core genome was calculated with LS-BSR for coding region sequences with a BSR value of > 0.8 across all genomes tested; in each case, the average core genome was calculated across 10 replicates at each level of sampling. The core genome results demonstrate the simulated and completed genomes generally return a consistent core genome size ( Figure 4 ). Additionally, the pan-genome size was slightly larger using simulated reads, which is likely a result of fragmented coding regions that appear to be separate sequences during the clustering step in LS-BSR. The same general trends were observed across each species. Download figure Open in new tab Figure 4: Comparative pan-genome plots for species with a large number of complete genomes. The plots either demonstrate the accumulation of coding regions in the pan-genome (upper lines) or reduction of coding regions in the core genome (lower lines). MLST comparisons between complete and simulated genome assemblies . The relationships between bacterial isolates has typically been performed with multi-locus sequence type (MLST) approaches ( 32 ). To test the quality of assembled genomes, we extracted the 7 genes from the E. coli and S. aureus MLST schemes ( 33 ) and compared the sequence type (ST) calls between finished and simulated genome assemblies. In both species, all called sequence types matched between complete and simulated genome assemblies. This demonstrates that high quality draft genome assemblies can often provide important sequence type information for comparison to previous or future studies. Comparison of real and simulated data . Although simulated draft genome assemblies provide comparable MLST and core genome information, they don‘t represent real data, which can be of variable quality. A comparison of contig numbers between E. coli genomes downloaded from GenBank and simulated assemblies generated in this study demonstrates this variability ( Figure 5 , panel A). The genome size is also highly variable in the real data ( Figure 5 , panel B), which could be due to either insufficient coverage or contamination with other genomes. If strict filtering on real genome sequence data is implemented, then much of this variation can and should be eliminated prior to comparative analyses. Download figure Open in new tab Figure 5: A comparison between all Escherichia coli genomes in Genbank (black) and simulated short read assemblies (red) in terms of (A) the number of total contigs, and (B) the summed genome assembly size. Discussion Short read sequencing technologies have been key in understanding the movement ( 34 ) and population structure ( 35 ) of bacterial species. Recent advances in DNA sequencing now allow for the push button assembly of bacterial genomes using long read sequencing approaches ( 36 ), which holds the promise of automated and complete genome assembly even for highly duplicated and repetitive genomes. However, due to cost limitations, many laboratories still rely on short read technologies for high-throughput SNP identification and genome assembly ensuring that short read applications will continue to be used for large-scale comparative genomics. While benefits to these approaches exist in the consensus calling of variants, limitations also exist due to the short nature of the read composition, which depending on the length of the read and length of the repeat, cannot span many repeat regions resulting in fragmented genome assemblies. Previous work has demonstrated the effects of different genome assembly algorithms on the recovery of a reference genome using short read technologies ( 27 , 37 , 38 ), and the GAGE-B study ( 27 ) evaluated assemblies of eight different bacterial species genomes with a number of different assemblers. However, less is known about how the composition of genomes from diverse species affects the ability to resolve the full genome with short read sequencing technology by keeping the assembly algorithm constant. In this study, we performed a comprehensive analysis of this issue through the assembly of greater than 5000 finished bacterial genomes. The results demonstrate that simulated short read assemblies recovered high percentages of most genomes; however, significant genome reduction was observed in some highly repetitive genomes, which has the ability to affect downstream comparative analyses. Comparative genomics studies include the identification of genomic features that are differentially conserved between genomes from isolates in the same or closely-related species. These comparisons are important for identifying gene differences that may be associated with diagnostics, virulence, or differential phenotypes ( 39 – 44 ). Artifacts generated from the assembly of short read data could potentially impact these sorts of comparisons. Our results indicate that coding region sequences identified in simulated draft genome assemblies were representative of the coding regions identified in complete genomes in most cases. Thus, draft assemblies can provide important information on genomic feature variation between strains, core and pan-genome comparisons, and isolate relationships based upon MLST genes extracted from draft assembles. The results of this study also demonstrate that draft genome quality in public repositories is variable and that quality control and filtering should be applied prior to comparative genomics studies. The results also indicate that genome reduction due to short read assembly can be a problem in downstream analyses for some genomes, although the impacts are variable, and perhaps predictable based on the repeat structure of a given genome. For large-scale comparative analyses, results must be interpreted with these limitations in mind. If missing genes are observed between groups of genomes, raw read mapping can be used to verify the gene presence or absence, although short read mapping may also suffer from some of the same limitations as short read genome assembly. Additionally, complete genomes representing species or clades of interest can provide a reference point for evaluating draft genome assemblies (e.g. provide information about repeat structure). This study indicates that draft genome assemblies generated from short read data often provide an acceptable representation of a bacterial genome for many comparative genomics applications. Acknowledgments This work was facilitated by the Monsoon High Performance Cluster (HPC) resource at Northern Arizona University. References 1. ↵ Sahl JW , Gillece JD , Schupp JM , Waddell VG , Driebe EM , Engelthaler DM , Keim P . 2013 . Evolution of a pathogen: a comparative genomics analysis identifies a genetic pathway to pathogenesis in Acinetobacter . PLoS ONE 8 : e54287 . OpenUrl CrossRef PubMed 2. ↵ Pearson T , Giffard P , Beckstrom-Sternberg S , Auerbach R , Hornstra H , Tuanyok A , Price EP , Glass MB , Leadem B , Beckstrom-Sternberg JS , Allan GJ , Foster JT , Wagner DM , Okinaka RT , Sim SH , Pearson O , Wu Z , Chang J , Kaul R , Hoffmaster AR , Brettin TS , Robison RA , Mayo M , Gee JE , Tan P , Currie BJ , Keim P . 2009 . Phylogeographic 11 reconstruction of a bacterial species with high levels of lateral gene 12 transfer . BMC Biol 7 : 78 . OpenUrl CrossRef PubMed 3. ↵ Chen PE , Shapiro BJ . 2015 . The advent of genome-wide association 14 studies for bacteria . Curr Opin Microbiol 25 : 17 - 24 . OpenUrl CrossRef PubMed 4. ↵ Sahl JW , Lemmer D , Travis J , Schupp JM , Gillece JD , Aziz M , Driebe EM , Drees KP , Hicks ND , Williamson CHD , Hepp CM , Smith DE , Roe C , Engelthaler DM , Wagner DM , Keim P . 2016 . NASP: an accurate, rapid method for the identification of SNPs in WGS datasets that supports flexible input and output formats . Microbial Genomics DOI: 10.1099/mgen.0.000074. . OpenUrl CrossRef 5. ↵ Tettelin H , Masignani V , Cieslewicz MJ , Donati C , Medini D , Ward NL , Angiuoli SV , Crabtree J , Jones AL , Durkin AS , DeBoy RT , Davidsen TM , Mora M , Scarselli M , Ros IMY , Peterson JD , Hauser CR , Sundaram JP , Nelson WC , Madupu R , Brinkac LM , Dodson RJ , Rosovitz MJ , Sullivan SA , Daugherty SC , Haft DH , Selengut J , Gwinn ML , Zhou LW , Zafar N , Khouri H , Radune D , Dimitrov G , Watkins K , O'Connor KJB , Smith S , Utterback TR , White O , Rubens CE , Grandi G , Madoff LC , Kasper DL , Telford JL , Wessels MR , Rappuoli R , Fraser CM . 2005 . Genome analysis of multiple pathogenic isolates of Streptococcus agalactiae: Implications for the microbial “pan-genome” . Proceedings of the National Academy of Sciences of the United States of America 102 : 13950 - 13955 . OpenUrl Abstract / FREE Full Text 6. ↵ Vernikos G , Medini D , Riley DR , Tettelin H . 2015 . Ten years of pan-genome analyses . Current Opinion in Microbiology 23 : 148 - 154 . OpenUrl CrossRef PubMed 7. ↵ Benedict MN , Henriksen JR , Metcalf WW , Whitaker RJ , Price ND . 2014 . ITEP: An integrated toolkit for exploration of microbial pan-genomes . Bmc Genomics 15 . 8. Chaudhari NM , Gupta VK , Dutta C . 2016 . BPGA- an ultra-fast pan-genome analysis pipeline . Scientific Reports 6 . 9. Contreras-Moreira B , Vinuesa P . 2013 . GET_HOMOLOGUES, a Versatile Software Package for Scalable and Robust Microbial Pangenome Analysis . Applied and Environmental Microbiology 79 : 7696 - 7701 . OpenUrl Abstract / FREE Full Text 10. Laing C , Buchanan C , Taboada EN , Zhang YX , Kropinski A , Villegas A , Thomas JE , Gannon VPJ . 2010 . Pan-genome sequence analysis using Panseq: an online tool for the rapid analysis of core and accessory genomic regions . Bmc Bioinformatics 11 . 11. Page AJ , Cummins CA , Hunt M , Wong VK , Reuter S , Holden MTG , Fookes M , Falush D , Keane JA , Parkhill J . 2015 . Roary: rapid large-scale prokaryote pan genome analysis . Bioinformatics 31 : 3691 - 3693 . OpenUrl CrossRef PubMed 12. ↵ Sahl JW , Caporaso JG , Rasko DA , Keim P . 2014 . The large-scale blast score ratio (LS-BSR) pipeline: a method to rapidly compare genetic content between bacterial genomes . PeerJ 2 : e332 . OpenUrl CrossRef 13. ↵ Zhao YB , Wu JY , Yang JH , Sun SX , Xiao JF , Yu J . 2012 . PGAP: pan-genomes analysis pipeline . Bioinformatics 28 : 416 - 418 . OpenUrl CrossRef PubMed Web of Science 14. ↵ Mukherjee S , Huntemann M , Ivanova N , Kyrpides NC , Pati A . 2015 . Large-scale contamination of microbial isolate genomes by Illumina PhiX control . Standards in Genomic Sciences 10 . 15. Parks DH , Imelfort M , Skennerton CT , Hugenholtz P , Tyson GW . 2015 . CheckM: assessing the quality of microbial genomes recovered from isolates, single cells, and metagenomes . Genome Research 25 : 1043 - 1055 . OpenUrl Abstract / FREE Full Text 16. ↵ Tennessen K , Andersen E , Clingenpeel S , Rinke C , Lundberg DS , Han J , Dangl JL , Ivanova N , Woyke T , Kyrpides N , Pati A . 2016 . ProDeGe: a computational protocol for fully automated decontamination of genomes . Isme Journal 10 : 269 - 272 . OpenUrl 17. ↵ Treangen TJ , Salzberg SL . 2011 . Repetitive DNA and next-generation sequencing: computational challenges and solutions . Nat Rev Genet 23 13 : 36 - 46 . OpenUrl CrossRef PubMed 18. ↵ Huang W , Li L , Myers JR , Marth GT . 2012 . ART: a next-generation sequencing read simulator . Bioinformatics 28 : 593 - 594 . OpenUrl CrossRef PubMed Web of Science 19. ↵ Bankevich A , Nurk S , Antipov D , Gurevich AA , Dvorkin M , Kulikov AS , Lesin VM , Nikolenko SI , Pham S , Prjibelski AD , Pyshkin AV , Sirotkin AV , Vyahhi N , Tesler G , Alekseyev MA , Pevzner PA . 2012 . SPAdes: a new genome assembly algorithm and its applications to single-cell sequencing . J Comput Biol 19 : 455 - 477 . OpenUrl CrossRef PubMed 20. ↵ Li H. 2013 . Aligning sequence reads, clone sequences and assembly contigs with BWA-MEM . arXivorg . 21. ↵ Quinlan AR , Hall IM . 2010 . BEDTools: a flexible suite of utilities for comparing genomic features . Bioinformatics 26 : 841 - 842 . OpenUrl CrossRef PubMed Web of Science 22. ↵ Delcher AL , Salzberg SL , Phillippy AM . 2003 . Using MUMmer to identify similar regions in large sequence sets . Curr Protoc Bioinformatics Chapter 10 :Unit 10 13 . OpenUrl 23. ↵ Rasko DA , Myers GS , Ravel J . 2005 . Visualization of comparative genomic analyses by BLAST score ratio . BMC Bioinformatics 6 : 2 . OpenUrl CrossRef PubMed 24. ↵ Hyatt D , Chen GL , Locascio PF , Land ML , Larimer FW , Hauser LJ . 2010 . Prodigal: prokaryotic gene recognition and translation initiation site identification . BMC bioinformatics 11 : 119 . OpenUrl CrossRef PubMed 25. ↵ Edgar RC. 2010 . Search and clustering orders of magnitude faster than BLAST . Bioinformatics 26 : 2460 - 2461 . OpenUrl CrossRef PubMed Web of Science 26. ↵ Kent WJ. 2002 . BLAT-the BLAST-like alignment tool . Genome Res 12 : 656 - 664 . OpenUrl Abstract / FREE Full Text 27. ↵ Magoc T , Pabinger S , Canzar S , Liu XY , Su Q , Puiu D , Tallon LJ , Salzberg SL . 2013 . GAGE-B: an evaluation of genome assemblers for bacterial organisms . Bioinformatics 29 : 1718 - 1725 . OpenUrl CrossRef PubMed Web of Science 28. ↵ Gurevich A , Saveliev V , Vyahhi N , Tesler G . 2013 . QUAST: quality assessment tool for genome assemblies . Bioinformatics 29 : 1072 - 1075 . OpenUrl CrossRef PubMed Web of Science 29. ↵ Akayama KN , Amashita AY , Urokawa KK , Orimoto TM , Gawa MO , Ukuhara MF , Rakami HU , Hnishi MO , Chiyama IU , Gura YO , Oka TO , Shima KO , Amura AT , Attori MH , Ayashi TH . 2008 . The Whole-genome Sequencing of the Obligate Intracellular Bacterium Orientia tsutsugamushi Revealed Massive Gene Amplification During Reductive Genome Evolution . DNA Research 15 : 185 - 199 . OpenUrl CrossRef PubMed Web of Science 30. ↵ Sahl JW , Vazquez AJ , Hall CM , Busch JD , Tuanyok A , Mayo M , Schupp JM , Lummis M , Pearson T , Shippy K , Colman RE , Allender CJ , Theobald V , Sarovich DS , Price EP , Hutcheson A , Korlach J , LiPuma JJ , Ladner J , Lovett S , Koroleva G , Palacios G , Limmathurotsakul D , Wuthiekanun V , Wongsuwan G , Currie BJ , Keim P , Wagner DM . 2016 . The effects of signal erosion and core genome reduction on the identification of diagnostic markers . MBio 7 : e00846 - 00816 . OpenUrl CrossRef PubMed 31. ↵ Jeong H , Pan J-G , Park S-H . 2016 . Contaminatin as a major factor in poor Illumina assembly of microbial isolate genomes . bioRxiv doi: http://dx.doi.org/10.1101/081885 . 32. ↵ Maiden MC , Bygraves JA , Feil E , Morelli G , Russell JE , Urwin R , Zhang Q , Zhou J , Zurth K , Caugant DA , Feavers IM , Achtman M , Spratt BG . 1998 . Multilocus sequence typing: a portable approach to the identification of clones within populations of pathogenic microorganisms . Proceedings of the National Academy of Sciences of the United States of America 95 : 3140 - 3145 . OpenUrl Abstract / FREE Full Text 33. ↵ Jolley KA , Maiden MC . 2010 . BIGSdb: Scalable analysis of bacterial genome variation at the population level . BMC bioinformatics 11 : 595 . OpenUrl CrossRef PubMed 34. ↵ Hendriksen RS , Price LB , Schupp JM , Gillece JD , Kaas RS , Engelthaler DM , Bortolaia V , Pearson T , Waters AE , Upadhyay BP , Shrestha SD , Adhikari S , Shakya G , Keim PS , Aarestrup FM . 2011 . Population genetics of Vibrio cholerae from Nepal in 2010: evidence on the origin of the Haitian outbreak . mBio 2 : e00157 - 00111 . OpenUrl CrossRef PubMed 35. ↵ Van Ert MN , Easterday WR , Huynh LY , Okinaka RT , Hugh-Jones ME , Ravel J , Zanecki SR , Pearson T , Simonson TS , U'Ren JM , Kachur SM , Leadem-Dougherty RR , Rhoton SD , Zinser G , Farlow J , Coker PR , Smith KL , Wang B , Kenefic LJ , Fraser-Liggett CM , Wagner DM , Keim P . 2007 . Global genetic population structure of Bacillus anthracis . PloS one 2 : e461 . OpenUrl CrossRef PubMed 36. ↵ Chin CS , Alexander DH , Marks P , Klammer AA , Drake J , Heiner C , Clum A , Copeland A , Huddleston J , Eichler EE , Turner SW , Korlach J . 2013 . Nonhybrid, finished microbial genome assemblies from long-read SMRT sequencing data . Nat Methods 10 : 563 - 569 . OpenUrl CrossRef PubMed Web of Science 37. ↵ Earl D , Bradnam K , St John J , Darling A , Lin D , Fass J , Yu HO , Buffalo V , Zerbino DR , Diekhans M , Nguyen N , Ariyaratne PN , Sung WK , Ning Z , Haimel M , Simpson JT , Fonseca NA , Birol I , Docking TR , Ho IY , Rokhsar DS , Chikhi R , Lavenier D , Chapuis G , Naquin D , Maillet N , Schatz MC , Kelley DR , Phillippy AM , Koren S , Yang SP , Wu W , Chou WC , Srivastava A , Shaw TI , Ruby JG , Skewes-Cox P , Betegon M , Dimon MT , Solovyev V , Seledtsov I , Kosarev P , Vorobyev D , Ramirez-Gonzalez R , Leggett R , MacLean D , Xia F , Luo R , Li Z , Xie Y , et al . 2011 . Assemblathon 1: a competitive assessment of de novo short read assembly methods . Genome Res 21 : 2224 - 2241 . OpenUrl Abstract / FREE Full Text 38. ↵ Salzberg SL , Phillippy AM , Zimin A , Puiu D , Magoc T , Koren S , Treangen TJ , Schatz MC , Delcher AL , Roberts M , Marcais G , Pop M , Yorke JA . 2012 . GAGE: A critical evaluation of genome assemblies and assembly algorithms (vol 22, pg 557, 2012) . Genome Research 22 : 1196 - 1196 . OpenUrl FREE Full Text 39. ↵ Sahl JW , Allender CJ , Colman RE , Califf KJ , Schupp JM , Currie BJ , Van Zandt KE , Gelhaus HC , Keim P , Tuanyok A . 2015 . Genomic Characterization of Burkholderia pseudomallei Isolates Selected for Medical Countermeasures Testing: Comparative Genomics Associated with Differential Virulence . Plos One 10 . 40. Sahl JW , Del Franco M , Pournaras S , Colman RE , Karah N , Dijkshoorn L , Zarrilli R . 2015 . Phylogenetic and genomic diversity in isolates from the globally distributed Acinetobacter baumannii ST25 lineage . Scientific Reports 5 . 41. Sahl JW , Morris CR , Emberger J , Fraser CM , Ochieng JB , Juma J , Fields B , Breiman RF , Gilmour M , Nataro JP , Rasko DA . 2015 . Defining the Phylogenomics of Shigella Species: a Pathway to Diagnostics . Journal of Clinical Microbiology 53 : 951 - 960 . OpenUrl Abstract / FREE Full Text 42. Sahl JW , Sistrunk JR , Fraser CM , Hine E , Baby N , Begum Y , Luo Q , Sheikh A , Qadri F , Fleckenstein JM , Rasko DA . 2015 . Examination of the Enterotoxigenic Escherichia coli Population Structure during Human Infection . MBio 6 : e00501 . OpenUrl CrossRef PubMed 43. Hazen TH , Donnenberg MS , Panchalingam S , Antonio M , Hossain A , Mandomando I , Ochieng JB , Ramamurthy T , Tamboura B , Qureshi S , Quadri F , Zaidi A , Kotloff KL , Levine MM , Barry EM , Kaper JB , Rasko DA , Nataro JP . 2016 . Genomic diversity of EPEC associated with clinical presentations of differing severity . Nat Microbiol 1 : 15014 . OpenUrl 44. ↵ Baig A , McNally A , Dunn S , Paszkiewicz KH , Corander J , Manning G . 2015 . Genetic import and phenotype specific alleles associated with hyper-invasion in Campylobacter jejuni . BMC Genomics 16 : 852 . OpenUrl CrossRef Back to top Previous Next Posted December 03, 2016. Download PDF Supplementary Material Email Thank you for your interest in spreading the word about bioRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following Bacterial genome reduction as a result of short read sequence assembly Message Subject (Your Name) has forwarded a page to you from bioRxiv Message Body (Your Name) thought you would like to see this page from the bioRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share Bacterial genome reduction as a result of short read sequence assembly Charles H.D. Williamson , Andrew Sanchez , Adam Vazquez , Joshua Gutman , Jason W. Sahl bioRxiv 091314; doi: https://doi.org/10.1101/091314 Share This Article: Copy Citation Tools Bacterial genome reduction as a result of short read sequence assembly Charles H.D. Williamson , Andrew Sanchez , Adam Vazquez , Joshua Gutman , Jason W. Sahl bioRxiv 091314; doi: https://doi.org/10.1101/091314 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Genomics Subject Areas All Articles Animal Behavior and Cognition (7830) Biochemistry (18308) Bioengineering (14486) Bioinformatics (43324) Biophysics (22074) Cancer Biology (19185) Cell Biology (26304) Clinical Trials (138) Developmental Biology (13701) Ecology (20507) Epidemiology (2067) Evolutionary Biology (24962) Genetics (15911) Genomics (23104) Immunology (18297) Microbiology (41529) Molecular Biology (17623) Neuroscience (91349) Paleontology (683) Pathology (2927) Pharmacology and Toxicology (4975) Physiology (7922) Plant Biology (15593) Scientific Communication and Education (2073) Synthetic Biology (4449) Systems Biology (10043) Zoology (2329) window.__CF$cv$params={r:'a2284e4c5bf54ec0',t:'MTc4NTI4ODA3Ng==',u:'019fab7632e97b22aef9bbf5efcbd655',ut:'QyGA84h7GBjyMXwTk_.dWyjveFGGSYu6DEwkPEd9HJM-1785288078-1.2.1.1-5kqIEUIa9NHjwyOr1XVCKsDXMkT..JwNiDUW_wm_6dH7vVWtAP3jUmYebY8GcOGMNzAY7IGzQVcQZGmho4cBmzK4MyuTu_To1N.eayZ4s4s',i:60};(function(){if(!document.body)return;var s=document.createElement('script');s.src='/cdn-cgi/challenge-platform/scripts/precursor/main.js';document.head.appendChild(s);})();
Text is read by the "Ask this paper" AI Q&A widget below.
Extraction quality varies by source — PMC NXML preserves structure
cleanly, OA-HTML may include some navigation residue, and OA-PDF can
have broken hyphenation. The publisher copy
(via DOI)
is the canonical version.