Full text
23,360 characters
· extracted from
preprint-html
· click to expand
SYNY: a pipeline to investigate and visualize collinearity between genomes | bioRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-M677548'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search New Results SYNY: a pipeline to investigate and visualize collinearity between genomes View ORCID Profile Alexander Thomas Julian , View ORCID Profile Jean-François Pombert doi: https://doi.org/10.1101/2024.05.09.593317 Alexander Thomas Julian 1 Department of Biology, Illinois Institute of Technology , Chicago, IL 60616, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Alexander Thomas Julian Jean-François Pombert 1 Department of Biology, Illinois Institute of Technology , Chicago, IL 60616, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Jean-François Pombert For correspondence: jpombert{at}iit.edu Abstract Full Text Info/History Metrics Data/Code Preview PDF Abstract Investigating collinearity between chromosomes is often used in comparative genomics to help identify gene orthologs, pinpoint genes that might have been overlooked as part of annotation processes and/or perform various evolutionary inferences. Collinear segments, also known as syntenic blocks, can be inferred from sequence alignments and/or from the identification of genes arrayed in the same order and relative orientations between investigated genomes. To help perform these analyses and assess their outcomes, we built a simple pipeline called SYNY (for synteny) that implements the two distinct approaches and produces different visualizations. The SYNY pipeline was built with ease of use in mind and runs on modest hardware. The pipeline is written in Perl and Python and is available on GitHub ( https://github.com/PombertLab/SYNY ) under the permissive MIT license. Introduction The identification of collinear segments (also known as syntenic blocks) between genomes is a common task in comparative genomics. Albeit traditionally used in genetics to refer to genes that are located on the same chromosomes but not necessarily in the same arrangement, the term synteny —from the Greek words syn and ταινια meaning on the same band —has been repurposed in genomics to refer to genes that are arrayed in the same order and relative orientations between genomes [ 1 ]. The positional information retrieved from collinearity inferences can be used for several purposes. For example, it can be used to help detect genes that might have been overlooked during genome annotation, it can be used to help identify which genes are real orthologs within a pool of similar paralogs retrieved from sequence homology searches, and it can be used to develop and test various evolutionary hypotheses such as reconstructing the structure and content of ancestral genomes [ 2 , 3 ]. In comparative genomics, collinearity assessments are usually performed by sequence alignments and/or by looking for sets of genes, i . e . clusters, that share the same order and relative orientations between genomes. Since alignment-based approaches do not rely on genome annotations, they can be performed on unannotated sequences, but because these approaches rely on the presence of a modicum of homology between the sequences being compared, they tend to struggle when confronted with high levels of divergence. Conversely, while approaches based on gene clusters require annotated sequences (and thus struggle with poorly annotated genomes), they tend to fare better with high levels of divergence. Hypervariable intergenic regions are not considered in gene cluster analyses and by further restricting the analyses to protein-coding genes, searches can be performed across larger evolutionary distances by leveraging the amino acid sequences of their products. Amino acid sequences are not affected by silent mutations nor by codon usage biases, they feature a larger-state space than nucleotide sequences, and their back mutation probabilities are lower than those from nucleotides [ 4 ]. Although several tools have been developed over the years to help infer synteny between genomes (see Liu et al. [ 5 ] for a review), no single tool met our comparative genomics needs. Namely, we needed a tool that could produce high quality circular and linear collinearity maps from both genome alignment and gene cluster inferences starting from NCBI GenBank Flat files. To rectify this and allow other research groups to perform similar analyses, we developed SYNY, a simple pipeline to investigate and visualize collinearity between genomes. Pipeline overview The SYNY pipeline was developed on Linux. Its dependencies can be installed automatically on most Linux distributions using setup_syny . pl . Collinearity inferences with SYNY involve three steps: (i) data parsing; (ii) collinearity inferences from genome alignments and/or gene clusters; and (iii) plotting of the results ( Fig. 1 ). All three steps can be performed using the run_syny . pl master script. Briefly, this script extracts genome/protein sequences and annotation data from NCBI GenBank (.gbff) input files. It identifies collinear sequences by performing round-robin pairwise genome alignments with minimap2 [ 6 ]. It locates collinear protein-coding gene clusters by identifying protein orthologs using bidirectional homology searches with DIAMOND [ 7 ], finding gene pairs arrayed identically between genomes (with and/or without gaps), and reconstructing clusters from overlapping gene pairs. It then plots the collinear segments inferred from these approaches as dotplots, chromosome maps (hereby referred to as barplots) and Circos [ 8 ] plots. It also summarises the corresponding metrics as heatmaps. All plots are generated in Portable Network Graphic (PNG) and Scalable Vector Graphics (Scalable Vector Graphics) formats. Download figure Open in new tab Figure 1. Fig. 1 . Overview of the SYNY pipeline. The pipeline contains three main steps: data parsing, collinear inferences, and data visualization. SYNY uses NCBI GenBank Flat files (.gbff) as input. Collinear inferences can be performed from genome alignments and/or from gene clusters. Collinear visualizations include chromosome maps (barplots), dotplots, and Circos plots. Collinear similarities are also summarized as heatmaps. Hardware and software requirements The SYNY pipeline was tested on Linux operating systems (Fedora 40, Ubuntu 22.04.4, and openSUSE Tumbleweed) running as virtual machines (VMs) inside the Microsoft Windows Subsystem for Linux. All tests were performed on a laptop equipped with an Intel Core i5-12500H processor and 12 Gb of RAM allocated to the VMs. The SYNY pipeline is built in Perl and Python. It requires the PerlIO::gzip Perl module and the matplotlib, seaborn, pandas and scipy Python libraries. Minimap2 [ 6 ] and DIAMOND [ 7 ] are required to perform pairwise genome alignments and sequence homology searches, respectively. Circos [ 8 ] is required to generate the corresponding plots. Case example – Comparing microsporidia genomes Microsporidia from the genus Encephalitozoon are human-infecting pathogens causing chronic diarrhea, bronchitis, conjunctivitis and/or encephalitis in afflicted patients [ 9 ]. With tiny yet complete nuclear genomes totaling less than 3 Mbp, these obligate intracellular pathogens constitute paradigms of genome streamlining in eukaryotes [ 10 ]. Their closest known relatives from the genus Ordospora , whose members infect crustaceans but not humans, also arbor diminutive genomes [ 11 ]. The Encephalitozoon and Ordospora genomes constitute a good example dataset for the following reasons. Their small sizes make for quick computations, they are highly collinear, and they exhibit a high level of sequence divergence, with average nucleotide identity (ANI) values within the Encephalitozoon and Ordospora genera of 74-81% and 83%, respectively [ 11 ], and between the two genera of about 68% (computed here with OrthoANIu v1.2 [ 12 ]). Altogether, the divergence thresholds within (> 15%) and between (> 30%) these genera allow us to demonstrate the limitations of alignment-based collinearity inferences. Using SYNY with default settings on a total of five genomes ( E. intestinalis, E. hellem, E. cuniculi , O. colligata, O. pajunii ) downloaded from NCBI (with Encephalitozoonidae.sh), we were able to infer and plot the collinearity between each genome in less than five minutes (264 seconds) on a laptop equipped with a mobile Intel i5-12500H central processing unit (CPU). This produced a total of 40 dotplots, 40 barplots, and 40 Circos plots: i . e . pairwise comparisons between queries and subjects are performed in both directions while self comparisons are skipped (5 genomes x [5 genomes – 1] * 2 directions = 40). When allowing three different gap thresholds (0, 1 and 5) for gene clusters inferences and requesting pairwise and concatenated Circos plots using both the normal and inverted karyotypes, the process ran for 13 minutes on the same laptop (from which 11 minutes were for Circos plotting), and produced a total of 80 dotplots, 80 barplots, and 168 Circos plots ([80 pairwise + 4 concatenated plots] * 2 karyotypes). Examples of collinear plots generated by SYNY from gene clusters and genome alignments are shown in Fig. 2 . As expected from the high level of sequence divergence inherent to this dataset, collinear segments inferred from genome alignments ( Fig. 2 ; right) were more fragmented than those reconstructed from gene clusters ( Fig. 2 ; left) in comparisons between species belonging to distinct genera. This artefact was caused by limitations of the genome alignment process, which resulted in as little as 19.1% of the bases mapped in collinear segments between Encephalitozoon and Ordospora species. In contrast, no less than 66.7% of the protein-coding genes between these species were found in clusters, resulting in much more contiguous collinear plots that more accurately represent the structural similarities between these genomes. When comparing genomes from within the same genera, both approaches produced similar results (not shown). Download figure Open in new tab Figure 2. Examples of collinear maps generated with SYNY. Plots shown are from gene cluster (left) and genome alignment (right) inferences between the Encephalitozoon hellem ATCC50604 and Ordospora colligata OC4 genomes. ( a ). Circos plots . The reference contigs (from E. hellem ) are labelled with roman numerals. The other contigs (from O. colligata ) are labelled with Arabic numerals. Ribbons are color-coded based on the reference contigs. Plots were produced with the label_size 80, no_ntbiases , and no_cticks options. ( b ). Chromosome maps (barplots) . Contigs from E. hellem are labelled with their NCBI accession numbers ( CP075147 to CP075157 ) and plotted to scale. Collinear regions with the O. colligata genome are highlighted by rectangular patches color-coded based on it contigs (NW_014575399 to NW_014575407). For brevity, only one of the identical legends between the left and right plots was kept. This legend was repositioned at the bottom of the panel to limit its width. ( c ). Heatmap summaries of proteins found in clusters (left) and of bases found aligned in collinear segments (right) between the Encephalitozoon and Ordospora genomes: Ec, E. cuniculi ; Eh, E. hellem ; Ei, E. intestinalis ; Oc, O. colligata ; Op, O. pajunii . Discussion When we initially designed SYNY, we set out to develop a simple tool that identifies the gene clusters that are shared between eukaryote genomes and then plots this information as useful figures. To do this, we began by identifying gene pairs arrayed identically between genomes, then reconstructed gene clusters based on these gene pairs. We restricted our searches to protein coding genes to enable searches across larger evolutionary time scales but also to bypass RNA coding genes that are often unannotated in eukaryote genome projects. However, because the annotations used as inputs were sometimes missing protein-coding genes, featured multi-exon proteins annotated as separate entities, and/or contained spurious open reading frames annotated as hypothetical proteins, gene clusters inferred by SYNY were often fragmented due to these issues. To account for these possibilities, we implemented the option to allow for gaps between genes when performing cluster-based collinearity inferences. Likewise, to handle instances where genome annotations are not available, for example when comparing a newly assembled genome to one or more reference(s), we implemented a parallel approach based on genome sequences that leverages the Minimap2 [ 6 ] pairwise alignment tool. One of our main goals with SYNY was to automate the production of collinearity plots that are both informative and suitable as is (or with as few modifications as possible) for publications purposes. To this end, we implemented some of the most common visualisation graphs in comparative genomics, including Circos [ 8 ] plots. Albeit SYNY runs relatively quickly on modest hardware, most of its runtime is dedicated to Circos plotting. Circos is not multithreaded, and while we parallelized its plotting instances across multiple threads (at the expense of using more random-access memory; RAM), generating the corresponding plots takes a substantial fraction of the total SYNY runtime. Because the number of pairwise comparisons to be performed can quickly balloon up with the number of genomes being compared, users may want to skip the Circos plots and use the barplots for preliminary visual investigations. The barplots are generated much faster (plotting 40 graphs from the case example took only 7 seconds for the barplots compared to 195 seconds for Circos) and run in a much smaller memory footprint. The second most time-consuming segment of the SYNY pipeline involves the pairwise alignment step with Minimap2. Aligning larger genomes will lengthen is runtime and, because Minimap2 does not work with queries with larger than 2,147,483,647 bases (as per its user manual), inferring collinearity between genomes using pairwise alignments is not possible for queries exceeding that threshold. Aligning larger genomes with Minimap2 will also require more RAM. Performing a pairwise alignment between two 150 Mbp genomes from Arabidopsis peaked at about 10 Gb of RAM (0.7 Gb operating system (OS) + 9.3 Gb Minimap2; result not shown) whereas running gene cluster inferences on the same two genomes never exceeded 2.5 Gb of RAM. Considering the above and that cluster-based inferences perform better than pairwise alignments methods when dealing with high levels of sequence divergence (as shown in Fig. 2 ), users may want to skip pairwise alignments when dealing with well annotated genomes. Author contributions JFP designed the pipeline; JFP and ATJ wrote the software and tested the pipeline; JFP and ATJ wrote and reviewed the manuscript. Funding This work was supported in part by the National Institute of Allergy and Infectious Diseases of the National Institutes of Health [grant number R15AI128627] to JFP. The content is solely the responsibility of the authors and does not necessarily represent the official views of the National Institutes of Health. Footnotes https://github.com/PombertLab/SYNY References 1. ↵ Myers P. Synteny: Inferring ancestral genomes . Nature Education . 2008 ; 1 : 47 . OpenUrl 2. ↵ Van De Peer Y Walden N , Schranz ME . Synteny Identifies Reliable Orthologs for Phylogenomics and Comparative Genomics of the Brassicaceae . Van De Peer Y , editor. Genome Biology and Evolution . 2023 ; 15 : evad034 . doi: 10.1093/gbe/evad034 OpenUrl CrossRef 3. ↵ Muffato M , Louis A , Nguyen NTT , Lucas J , Berthelot C , Roest Crollius H. Reconstruction of hundreds of reference ancestral genomes across the eukaryotic kingdom . Nat Ecol Evol . 2023 ; 7 : 355 – 366 . doi: 10.1038/s41559-022-01956-z OpenUrl CrossRef 4. ↵ Bogusz M , Whelan S. Phylogenetic Tree Estimation With and Without Alignment: New Distance Methods and Benchmarking . Syst Biol . 2016 ; syw074 . doi: 10.1093/sysbio/syw074 OpenUrl CrossRef PubMed 5. ↵ Liu D , Hunt M , Tsai IJ . Inferring synteny between genome assemblies: a systematic evaluation . BMC Bioinformatics . 2018 ; 19 : 26 . doi: 10.1186/s12859-018-2026-4 OpenUrl CrossRef 6. ↵ Alkan C Li H. New strategies to improve minimap2 alignment accuracy . Alkan C , editor. Bioinformatics . 2021 ; 37 : 4572 – 4574 . doi: 10.1093/bioinformatics/btab705 OpenUrl CrossRef PubMed 7. ↵ Buchfink B , Reuter K , Drost H-G. Sensitive protein alignments at tree-of-life scale using DIAMOND . Nat Methods . 2021 ; 18 : 366 – 368 . doi: 10.1038/s41592-021-01101-x OpenUrl CrossRef PubMed 8. ↵ Krzywinski M , Schein J , Birol I , Connors J , Gascoyne R , Horsman D , et al. Circos: An information aesthetic for comparative genomics . Genome Res . 2009 ; 19 : 1639 – 1645 . doi: 10.1101/gr.092759.109 OpenUrl Abstract / FREE Full Text 9. ↵ Wadi L , Reinke AW . Evolution of microsporidia: An extremely successful group of eukaryotic intracellular parasites . PLoS Pathog . 2020 ; 16 : e1008276 . doi: 10.1371/journal.ppat.1008276 OpenUrl CrossRef 10. ↵ Mascarenhas dos Santos AC , Julian AT , Liang P , Juárez O , Pombert J-F. Telomere-to-Telomere genome assemblies of human-infecting Encephalitozoon species . BMC Genomics . 2023 ; 24 : 237 . doi: 10.1186/s12864-023-09331-3 OpenUrl CrossRef 11. ↵ Albuquerque NRM , Haag KL , Fields PD , Cabalzar A , Ben-Ami F , Pombert J , et al. A new microsporidian parasite, Ordospora pajunii sp. nov (Ordosporidae), of Daphnia longispina highlights the value of genomic data for delineating species boundaries . J Eukaryotic Microbiology . 2022 ; 69 . doi: 10.1111/jeu.12902 OpenUrl CrossRef 12. ↵ Yoon S-H , Ha S , Lim J , Kwon S , Chun J. A large-scale evaluation of algorithms to calculate average nucleotide identity . Antonie van Leeuwenhoek . 2017 ; 110 : 1281 – 1286 . doi: 10.1007/s10482-017-0844-4 OpenUrl CrossRef PubMed View the discussion thread. Back to top Previous Next Posted May 13, 2024. Download PDF Data/Code Email Thank you for your interest in spreading the word about bioRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following SYNY: a pipeline to investigate and visualize collinearity between genomes Message Subject (Your Name) has forwarded a page to you from bioRxiv Message Body (Your Name) thought you would like to see this page from the bioRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share SYNY: a pipeline to investigate and visualize collinearity between genomes Alexander Thomas Julian , Jean-François Pombert bioRxiv 2024.05.09.593317; doi: https://doi.org/10.1101/2024.05.09.593317 Share This Article: Copy Citation Tools SYNY: a pipeline to investigate and visualize collinearity between genomes Alexander Thomas Julian , Jean-François Pombert bioRxiv 2024.05.09.593317; doi: https://doi.org/10.1101/2024.05.09.593317 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Bioinformatics Subject Areas All Articles Animal Behavior and Cognition (7649) Biochemistry (17738) Bioengineering (13925) Bioinformatics (42059) Biophysics (21496) Cancer Biology (18643) Cell Biology (25577) Clinical Trials (138) Developmental Biology (13406) Ecology (19946) Epidemiology (2067) Evolutionary Biology (24370) Genetics (15627) Genomics (22551) Immunology (17772) Microbiology (40497) Molecular Biology (17212) Neuroscience (88786) Paleontology (667) Pathology (2845) Pharmacology and Toxicology (4835) Physiology (7663) Plant Biology (15177) Scientific Communication and Education (2047) Synthetic Biology (4304) Systems Biology (9838) Zoology (2272)
Text is read by the "Ask this paper" AI Q&A widget below.
Extraction quality varies by source — PMC NXML preserves structure
cleanly, OA-HTML may include some navigation residue, and OA-PDF can
have broken hyphenation. The publisher copy
(via DOI)
is the canonical version.