Quantifying structural variants in chromosomes using landmark-based disparity

preprint OA: closed
📄 Open PDF Full text JSON View at publisher

Abstract

Chromosomal architecture has played a key role in the evolution of biodiversity. Detecting structural variants (SVs) on chromosomes has informed the study of speciation, sex determination, adaptation, and some of the earliest divergences in the tree of life. Here we present a computationally non-intensive approach, based on geometric morphometrics, that uses conserved DNA sequences as landmarks to quantify structural disparities of focal chromosomes across multiple species, individuals, or cell types. Based on two approaches, we show that this ‘geno-metric’ method can be applied at micro– and macroevolutionary scales to discover and diagnose SVs. Using human X-linked genes and ultraconserved elements as landmarks, we provide empirical demonstrations with amniote sex chromosomes, the Drosophila virilis group, and placental mammal genomes. Landmark-based structural disparity analysis effectively identifies chromosomal rearrangements and has parallels with traditional morphometrics regarding chromosome size, landmark orientation and landmark availability. Using simulations, we show that structural disparity inferred from ultraconserved elements is correlated with overall levels of chromosome evolution; an attribute which is consistent with observed disparity between and within mammalian orders. We found that the disparity patterns of SVs have significant phylogenetic signal, giving them broad importance for studying evolutionary biology. Structural disparity analyses are a valuable addition to the comparative genomic toolkit in that they offer an intuitive, rapid mechanism for detecting SVs associated with single copy genetic landmarks and the potential to reveal broader patterns of chromosome evolution related to expansions, contractions, rearrangements and phylogeny.
Full text 94,489 characters · extracted from preprint-html · click to expand
Quantifying structural variants in chromosomes using landmark-based disparity | bioRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-M677548'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search New Results Quantifying structural variants in chromosomes using landmark-based disparity View ORCID Profile Ashwini V. Mohan , Anjali Goswami , Jeffrey W. Streicher doi: https://doi.org/10.1101/2024.10.30.620815 Ashwini V. Mohan 1 Institut de Biologie, Université de Neuchâtel , Neuchâtel, Switzerland 2 Natural History Museum , London SW7 5BD, United Kingdom Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Ashwini V. Mohan For correspondence: ashwini.venkatanarayana{at}unine.ch j.streicher{at}nhm.ac.uk Anjali Goswami 2 Natural History Museum , London SW7 5BD, United Kingdom Find this author on Google Scholar Find this author on PubMed Search for this author on this site Jeffrey W. Streicher 2 Natural History Museum , London SW7 5BD, United Kingdom Find this author on Google Scholar Find this author on PubMed Search for this author on this site For correspondence: ashwini.venkatanarayana{at}unine.ch j.streicher{at}nhm.ac.uk Abstract Full Text Info/History Metrics Supplementary material Data/Code Preview PDF Abstract Chromosomal architecture has played a key role in the evolution of biodiversity. Detecting structural variants (SVs) on chromosomes has informed the study of speciation, sex determination, adaptation, and some of the earliest divergences in the tree of life. Here we present a computationally non-intensive approach, based on geometric morphometrics, that uses conserved DNA sequences as landmarks to quantify structural disparities of focal chromosomes across multiple species, individuals, or cell types. Based on two approaches, we show that this ‘geno-metric’ method can be applied at micro– and macroevolutionary scales to discover and diagnose SVs. Using human X-linked genes and ultraconserved elements as landmarks, we provide empirical demonstrations with amniote sex chromosomes, the Drosophila virilis group, and placental mammal genomes. Landmark-based structural disparity analysis effectively identifies chromosomal rearrangements and has parallels with traditional morphometrics regarding chromosome size, landmark orientation and landmark availability. Using simulations, we show that structural disparity inferred from ultraconserved elements is correlated with overall levels of chromosome evolution; an attribute which is consistent with observed disparity between and within mammalian orders. We found that the disparity patterns of SVs have significant phylogenetic signal, giving them broad importance for studying evolutionary biology. Structural disparity analyses are a valuable addition to the comparative genomic toolkit in that they offer an intuitive, rapid mechanism for detecting SVs associated with single copy genetic landmarks and the potential to reveal broader patterns of chromosome evolution related to expansions, contractions, rearrangements and phylogeny. Introduction We are now firmly in the age of genomics with fully sequenced genomes commonplace for most organismal groups. Because of this, multi-species genomic comparisons are increasingly used in evolutionary and ecological research. This is even the case for organisms with relatively large nuclear genomes [ 1 – 3 ]. Excitingly, the comparison of whole genome sequences has offered valuable evolutionary insights into a range of topics including gene orthology [ 4 ], transposable element evolution [ 5 ], and adaptation to novel environments [ 6 ]. One important area of comparative genomics research is the study of chromosome architecture [ 7 ]. Detecting structural variants (SVs) of chromosomes is highly relevant to the evolution of biodiversity and disease [ 8 ]. From supergenes [ 9 ] to sex chromosomes [ 10 ], evolutionary shifts in chromosome structure are known to have a substantial influence on organismal phenotype. Recent research utilizing forward in time simulations has also indicated the important role of SVs in sustaining genomic adaptation [ 11 ]. Traditional methods for detecting SVs include chromosome banding and fluorescent in situ hybridization [ 12 ]. Several modern methods using genomic sequencing have also been developed including aberrant read alignment patterns, split reads, coverage changes, alignment of long-read sequences, and strand-seq [ 13 – 14 ]. Amongst sequencing-based methods, synteny assessment [ 15 ] is increasingly used to compare chromosome structure across multiple species with visual interpretation achieved through complex network diagrams, e.g. [ 16 – 19 ]. Most of these high-throughput DNA sequencing approaches for detecting SVs depend on comparisons to a single reference sequence and/or whole genome alignment, which can be subject to reference-bias and time consuming, respectively [ 20 ]. To complement existing methods, here we present a method for detecting SVs across chromosomes from multiple species or individuals simultaneously that is reference-free, computationally non-intensive, and does not require whole genome alignment. Based on the principles of morphometrics, our method is applied to chromosomes from high-quality genome assemblies and facilitates the study of SVs in ordinated micro– and macroevolutionary contexts. Within the study of comparative morphology, structural variation in homologous structures across species is commonly termed ‘disparity’ [ 21 – 22 ]. Measures of morphological disparity are routinely used to explore the evolutionary dynamics of complex phenotypes, e.g. [ 23 – 24 ]. Structural disparity analysis of chromosomes is a disparity-based method that uses conserved landmark sequences to capture genomic variation across different species in a single analytical framework. Our ‘geno-metric’ method is based on a simple premise: chromosome structure is morphology (i.e., a shape, form, or structure) and there are well-validated methods for capturing morphological disparity across multiple species [ 23 – 25 ]. Furthermore, genomes are filled with homologous sequences (=landmarks) making them amenable for such analyses (e.g. [ 26 ]). To develop and scrutinize landmark-based structural disparity analyses, we analyzed three empirical datasets. First, we quantified SVs associated with sex chromosome evolution across amniotes using human X-linked BUSCO (Benchmarking Universal Single-Copy Orthologs) genes as landmarks which represents a scenario when the structure of a particular genomic/chromosomal feature is of interest [sensu 18). Second, we compared SVs related to a speciation-associated chromosome in Drosophila [ 27 ] using ultraconserved elements (UCEs; [ 28 ]) as landmarks which represents a scenario when chromosome-wide structure may be of interest. Third, we used genome sequences from placental mammals to develop a pipeline for discovering chromosomes that share homologous landmarks; a necessary step when chromosome homology is unclear or partial. We also used the placental mammal dataset to simulate SVs of chromosomes and determine the effects of indels and inversions on disparity measures. Finally, we tested all datasets for phylogenetic signals associated with disparity patterns of SVs. Results Pipeline development. We developed two approaches for conducting structural disparity analysis, [ 1 ] using generic feature format (gff3) annotation files as input and [ 2 ] a de novo method that allows users to landmark and analyze whole chromosome sequences from genome assemblies ( Fig. 1 ). The gff3-formatted pipeline allows users to extract genes from the chromosome of a single species in the gff3 file and then identify orthologs in the other species contained within the file. This approach is useful when chromosome homology is known, and focal landmarks are largely confined to single chromosomes. The de novo method begins by acquiring chromosome sequences from NCBI and then reference mapping landmarks to the assemblies using bwa [ 29 ]. We tested other approaches for reference mapping including minimap2 using default settings [ 30 ] but found bwa resulted in the most mapped landmarks per chromosome (Fig. S1). After orthologous landmarks are mapped, landmarked chromosomes are either prepared for downstream analysis (when chromosomal homology is known), or clustered based on the presence/absence of landmarks via metric multi-dimensional scaling (when chromosomal homology is not known). The latter option allows for the identification of chromosomes that share large numbers of landmarks, and thus likely large homologous genomic blocks. Following landmarking, both approaches have several steps in common including; removal of landmarks that are multi-copy or not shared across homologs, complementarity checks (5’-3’ versus 3’-5’), and exportation of landmark position information into a spreadsheet. The spreadsheet has rows representing species or individuals (one for each species or individual included in the input assemblies) and columns are landmark positions on a homologous chromosome set. The spreadsheet is then used as input for the disparity inference stage. Depending on user preference, absolute landmark positions can be used, or positions can be transformed using landmark bounding and/or chromosome size correction. Unless otherwise indicated, all steps in the pipeline occur using base functions in the R statistical package [ 31 ]. As with other disparity analyses, we use ordination techniques to reduce the dimensionality of variation in landmark position wherein the spreadsheet of landmark positions is analyzed using principal component analysis ( Fig. 1 ). Download figure Open in new tab Figure 1. Schematic illustration of the landmarking pipeline for structural disparity analysis of chromosomes. When using generic feature format (gff3) annotation files, landmark genes are selected from a reference followed by identification of orthologs from other species in the file and extraction of chromosome information and landmark positions. In the de novo approach, the first step is reference mapping landmarks to chromosome-level genome assemblies of different species, individuals, or cell types. If the homology of chromosomes is known, then the chromosomes are further prepared for disparity analysis. If homology is unknown, a presence-absence matrix is used to identify chromosomes with similar landmark compositions using a gap-based threshold for similarity. Chromosome sets are prepared for disparity analysis by extracting chromosome position information for each landmark. The resulting data matrix is used for multivariate Principal Component Analysis to visualize disparity across species. Structural disparity of human X-linked genes. Lovell et al. [ 18 ] examined 300 million years of vertebrate sex chromosome evolution using synteny-anchoring. We took their gff3-formatted annotation file and identified 53 human X-linked genes that could be used as landmarks on a set of single chromosomes for the focal taxa. We found that principal components (PCs) 1 and 2 explained 69.3% and 14.2% of structural variance, respectively. Using these first two axes, we found that major phylogenetic groups occupied distinct areas of geno-metric space (Fig. S2A). PC1 was significantly correlated with chromosome size (non-parametric Spearman’s test, rho = 0.70, p = 0.003), but not PC2 (rho = –0.27, p = 0.3043). Mean structural disparity was highest amongst those groups where the landmarks were found on sex chromosomes (placental mammals, marsupial mammals and squamate reptiles) compared to those where these markers are located on autosomal chromosomes (birds, monotreme mammals). When the data were analyzed using the reverse complement orientation of the chromosomes, we found that some chromosomes with clustered landmarks resulted in exaggerated PC scores for some taxa (e.g., chicken, Gallus gallus , Fig. S2B). We found that landmark bounding corrected for this methodological artifact regardless of the 5’ to 3’ orientation of the chromosomes ( Fig. 2A, S2C, S2D ). A size-corrected version of the analysis changed the patterns of disparity with birds being the most disparate group, followed closely by placental mammals (Fig. S2E). The size-corrected analysis was more robust to chromosome orientation with comparable results (Fig. S2F). Size-correction of the bounded version of the dataset generated indistinguishable results in the two directions of analysis (Fig. S2G, S2H). For comparison we generated a GENESPACE-style synteny link plot ( Fig. 2B ). Download figure Open in new tab Figure 2. Structural disparity analysis of 53 human X-linked BUSCO gene landmarks in select amniote vertebrates studied by Lovell et al. (2022). Disparity analysis using bounded, absolute landmark positions reveals larger disparity in those groups where landmarks are found on the X chromosomes (placental mammals and marsupial mammals) than in those groups where landmarks are found on autosomes or Z chromosomes (A). This disparity is largely explained by chromosome size in that some of these groups maintain high synteny despite also having high structural disparity (B). Size-corrected disparity analyses capture different aspects of variation in landmark placement revealing birds as especially disparate after chromosome size is accounted for (Fig. S2E, F, G, H). Structural disparity among closely related species of Drosophila . Poikela et al. [ 27 ] identified multiple chromosomal inversions that are associated with speciation patterns in the Drosophila virilis complex. We selected chromosome 5 for structural disparity analysis because there are known chromosomal inversions across the focal species [ 27 ]. We identified a set of 191 widely distributed dipteran UCEs [ 32 ] on chromosomes 5 for use as landmarks. Our structural disparity analysis revealed that three PC axes explained nearly 100% of the variance in landmark placement (PC1, 94%; PC2 5.2%; PC3, 0.07%). Patterns of disparity were highly polarized with PC1 separating D. flavomontana from the other species, PC2 separating D. montana from the other species, and PC3 scores separating D. americana + D. novamexicana from the two individuals of D. virilis ( Fig. 3 ). Those individuals with highly conserved, syntenic UCE landmark placements occurred closest in geno-metric space. Notably, this was the case for the two individuals of D. virilis ( Fig. 3A ) and close relatives D. americana and D. novamexicana ( Fig. 3B ; [ 33 ]). In contrast, separation corresponded to distinct SVs. PC1 scores were largely explained by two blocks of UCE landmarks with substantially different placements in D. flavomontana compared to the other taxa ( Fig. 3C ). PC2 scores largely explained by a pericentric chromosomal inversion in D. montana relative to the other taxa ( Fig. 3D ). Importantly, when we compare the landmark orientations on D. montana and D. flavomontana , we see evidence ( Fig. 3E ) of the two blocks of UCE landmarks (PC1) and the pericentric inversion (PC2). PC3 scores were largely explained by a paracentric inversion that D. virilis has relative to both D. americana ( Fig. 3F ) and D. novamexicana ( Fig. 3G ). None of the three most explanatory axes were significantly correlated with chromosome size (PC1, rho = 0.029, p = 1.000; PC2, rho = –0.371, p = 0.4972; PC3, rho = 0.600, p = 0.2417). Download figure Open in new tab Figure 3. Structural disparity analysis using raw, absolute landmark positions of 191 ultraconserved elements (UCEs) on chromosome 5 of the Drosophila virilis group, different colors correspond to different species in the complex. Synteny link maps around the main plot demonstrate what separation in geno-metric space corresponds to, in terms of landmark placement on chromosome 5. When samples are close to one another in space they are highly syntenic (A, B). Different axes correspond to discrete differences in landmark placement including two translocations (C), a pericentric inversion (D, E) and paracentric inversion (E, F). An ‘M’ was arbitrarily added by plotly (91) to values on component axes because they are less than one. Synteny link maps were generated using chromoMap [35). De novo homology discovery and structural disparity analysis using whole genome assemblies. Over long periods of evolutionary time, genome structure may change, substantially eroding the homology of chromosomes. Despite this, large genomic regions often remain evolutionarily intact on single chromosomes (e.g., [ 16 ]) among which structural disparity can be explored. We used the chromosome clustering module of our de novo pipeline to identify several chromosome sets containing homologous tetrapod UCEs [ 34 ] from 26 genome assemblies of five placental mammal orders (Artiodactyla, Carnivora, Perissodactyla, Primates, and Rodentia; Table 1 ). The results of the 10-axis chromosome clustering revealed that score thresholding could be used to identify eight sets of mutually exclusive chromosomes that (a) were highly similar in their UCE compositions and (b) contained only a single chromosome per species in the dataset ( Fig. 4A , Table 2 ). Chromosome similarity and threshold gaps decayed to after the second axis with not all sets containing 26 species and some instances of overlap with chromosomes belonging to other sets (e.g. Giraffa tippelskirchi chromosomes in sets 7 and 8, Fig 4A , Table 2 ). Axes 9 and 10 did not reveal any novel chromosome clusters associated with UCE landmark content ( Table 2 ). Viewing axes as bivariate plots made some clustering patterns clearer, especially in axis 4 ( Fig. 4B, C ). Given (a) the clear separation of chromosomes sets 1 and 2, and (b) that protein content was significantly more similar within these chromosome sets than between them (W = 6742, p < 0.0001, nonparametric Wilcoxon tests, Fig. S3) we used these chromosomes as input for a structural disparity analyses based on SVs of their UCE landmarks. Download figure Open in new tab Figure 4. Metric multi-dimensional scaling (MMDS) of 567 landmarked placental mammal chromosomes based on patterns of ultraconserved element presence and absence. These chromosomes represent the genomes of 26 species ( Table 1 ). MMDS axis thresholds are designated to establish ‘chromosome sets’ which contain a chromosome for most species included in the dataset. These chromosome sets likely contain mutually exclusive large homologous genomic blocks that are evolutionarily conserved and can be compared via structural disparity analysis. An initial assessment of 10 axes revealed eight sets of chromosomes that may be ideal for structural disparity analysis (A). Visualizing the clustering patterns in bivariate plots can assist with determining axis thresholds for defining the sets (B, C). View this table: View inline View popup Table 1. Chromosome sets identified from genome assemblies used to conduct landmark-based estimates of structural disparity. Those chromosomes appearing in bold text were used in the simulation/validation section of our study. CN = chromosome number (= reference number used in genomic assemblies), bp = base pairs, UCE = ultraconserved elements. Following removal of an outlier landmark, the chromosome sets had 185 and 81 landmarks, respectively. View this table: View inline View popup Download powerpoint Table 2. Axis thresholding based on metric multidimensional scaling (MMDS) of 567 placental mammal chromosomes from 26 species. Total number of species are unique chromosomes found to cluster on a given axis that are mutually exclusive when compared to clustering patterns of other axes. Threshold value is the smallest axis value identified in each group that is used to define the different clusters. Threshold gap value is the amount of space between the chromosome set and other chromosomes. An asterisk indicates two chromosomes sets where Macaca mulatta and/or Giraffa tippelskirchi chromosomes from other sets were within the threshold range. See Figure 4 for additional information. Landmarking of UCES on the two chromosome sets resulted in 186 (set 1) and 82 (set 2) landmarks, respectively. Using absolute landmark positions (Figs. S4A, S5A), we found that most of the variation in both chromosome sets was explained by PC1. In chromosome set 1, PC1 described 98.9% of the variation and in chromosome set 2, PC1 described 99.4% of the variation. We discovered that in both chromosome sets, PC1 was significantly correlated with chromosome size (set 1, rho = –0.65, p < 0.0001; set 2, rho = 0.76, p < 0.0001) when using absolute size. However, this was not the case for PC2 (set 1, rho = 0.19, p = 0.3308; set 2, rho = 0.37, p = 0.066). Bounding the landmark positions increased the amount of variation explained by PC1 (chromosome set 1, 99.2%, Fig. S4C; chromosome set 2, 99.9%, Fig. S5C), and produced unexpected patterns with several notable outliers across the first two PCs (e.g. Cricetulus griseus, Fig. S5C). Size-corrected analyses of chromosome set 1 resulted in each mammal order occupying a unique region of geno-metric across PCs 1 and 2 (Fig. S4E), but this was not the case in chromosome set 2 where orders largely overlapped (Fig. S5E). Size-correction using bounded landmarks resulted in less variation explained by PC1 for both chromosome sets (chromosome et 1, 89.2%, Fig. S4G; chromosome set 2, 91.6%, Fig. S5G), and similar outlier patterns that were observed in the raw bounded position analysis. Mapping of component loadings onto chromosome plots revealed a single landmark (UCE 6948) that was heavily loaded on PC1 and PC2 of both chromosome sets (Fig.S6). Removing this outlier landmark changed the inferred structural disparity patterns across the PCAs with mammalian orders mostly occupying distinctive regions of geno-metric space in both chromosome set 1 (Fig. S4B, D, F, H) and, apart from a single size-corrected analysis (Fig. S5F), chromosome set 2 (Fig. S5B, D, H). Following the removal of the outlier landmark, we found that PC1 explained most variation across the analyses. When using raw, absolute landmark positions PC1 explained 99.9% of the variance for both chromosome sets (Fig. S7A, S8A). Unlike the analyses including the outlier landmark, using bounded landmarks reduced the amount of variation explained on PC1 (95.3%, chromosome set 1, Fig. S7B; 98.6% chromosome set 2, Fig. S8B) compared to the raw positions. PC2 scores in the bounded analyses explained 2.0% (chromosome set 1) and 0.8% (chromosome set 2). Notably, the bounded landmark PC1 and PC2 separated all mammalian orders into distinct regions of geno-metric space. Size-correction via chromosome size resulted in different disparity patterns in both datasets with higher disparity within Artiodactyla in the chromosome set 1 analysis (Fig. S7E) and substantial overlap across orders in the chromosome set 2 analysis (Fig. S8E). The size-corrected, bounded analyses featured tighter clustering patterns within mammalian orders than the raw, size-corrected analyses in both chromosome sets (Figs. S7G, S8G). As observed in the X-linked BUSCO analyses, we found that using raw landmark positions on the reverse complemented version of the chromosome resulted in outlier taxa ( Macaca mulatta , Mus musculus , chromosome set 1, Fig. S7B; Equus asininus , E. caballus , Cricetulus griseus , Mus pahari , Peromyscus maniculatus , Rattus norvegicus , chromosome set 2, Fig. S8B). However, unlike the X-linked BUSCO analyses, these outliers were also observed in the size-corrected analyses (Fig. S7F, S8F). Reverse complemented versions of the raw and size-corrected bounded analyses were more consistent with their equivalent original analyses (Fig. S7D, E, S8D, E). As observed in the X-linked BUSCO analyses, the amount of variance explained by PCs 1 and 2 in the size-corrected, bounded analyses was identical for both directions of analysis. Based on our results, we selected the raw, bounded analysis with outlier landmark removed for further interpretation and analysis ( Fig. 5 ). Interestingly, although they explained a small proportion of the overall structural variation, PC3 (0.5%, chromosome set 1, 0.5% chromosome set 2) and PC4 (0.1% chromosome set 1, 0.09% chromosome set 2) still clustered mammalian orders in largely distinct regions of geno-metric space for both chromosome sets (Fig. S9). However, Capra aegragus was an outlier on PC3 of chromosome set 1 (Fig.S9A). Synteny analysis of landmarks using chromoMap [ 35 ] to visualize structural differences captured by UCE landmarks revealed that certain chromosomal regions are more conserved as indicated by low PC loading values of landmarks into PC1 (blue links, Fig. 6 ), compared to other regions where landmarks show higher differences in their placement (maroon links, Fig. 6 ). Unlike the Drosophila example where landmark inversions explained most structural disparity ( Fig. 3 ), insertions and deletion events between the UCE landmarks account for the structural disparity observed in the two placental mammal chromosomes sets ( Fig. 6 ). Download figure Open in new tab Figure 5. Ordination results from principal component (PC) analysis of UCE landmark positions in two sets of chromosomes from 26 species of placental mammals using bounded raw positions after the removal of an outlier landmark (see text). Representatives for five orders (Artiodactyla, Carnivora, Perissodactyla, Primates and Rodents) were included (Table1). The depicted structural variation is inferred from 185 (chromosome set 1, A) and 81 (chromosome set 2, B) landmarks, respectively. In both datasets, the first two PCs explained > 99% of the variation. Comparison of PCs 3 and 4 for both chromosome sets is provided in Figure S9. Download figure Open in new tab Figure 6. Select synteny link maps between Bos indicus and other placental mammals using bounded, absolute landmarking. Principal component loadings for various landmarks are indicated by the color of the links. The chromosome number (Chr) is indicated next to each species’ silhouette. The left and right columns are results from chromosome set 1 and set 2, respectively. Comparisons are (from top to bottom) between Bos indicus to Capra aegagrus (Artiodactyls), Bos indicus and Equus asinus (Perissodactyla), Bos indicus and Gorilla gorilla (Primates), Bos indicus and Mus musculus (Rodentia), and Bos indicus and the liger ( Panthera tigris + P. leo , Carnivora). Synteny link maps were generated using chromoMap [ 35 ]. Structural disparity increases through time. In the simulation study, > 99% of the variation in landmark placement was explained by PC1 with PC2 explaining 0.2% of the variation. As predicted, pairwise increase in disparity (ID) increased with the amount of simulated evolution confirming that, at least under neutral (random) conditions, disparity measures should correspond to relative amounts of genomic evolution ( Fig. 7A-D ). We compared the variance of log-transformed ID scores for indel versus inversion simulations using two-way ANOVAs. We found that on both axes, inversions caused a significantly higher increase in disparity than indels (PC1, F =2.88, p = 0.0251; PC2, F = 3.34, p = 0.0121; Fig. S10). Using empirical data, we found that inter-order disparity was significantly higher than intra-order disparity for both PC1 (chromosome set 1, W = 63997, p < 0.0001; chromosome set 2, W = 67457, p < 0.0001, Fig. 7E, F ) and PC2 (chromosome set 1, W = 67997, p < 0.0001; chromosome set 2, W = 61201, p < 0.0001, Fig. 7G, H ). We found that protein-coding content tended to be more similar within mammalian orders than between them, however, this was not a significant result (W = 499.5, p = 0.0722; nonparametric Wilcoxon tests; SI Appendix, Fig. S11). Download figure Open in new tab Figure 7. Log-transformed increase in disparity (ID) scores inferred from principal component (PC) analysis of ancestral and descendent insertion-deletion simulations for PC1 (A) and PC2 (B). Log-transformed ID scores inferred from inversion simulations for PC1 (C) and PC2 (D). Insertion-deletion simulation results are depicted in purple and inversion simulation results are depicted in blue. Observed empirical disparity scores for within and between mammalian orders for PC1 (E) and PC2 (F) of chromosome set 1 and PC1 (G) and PC2 (H) of chromosome set 2 are depicted in green. Phylogenetic signal of structural variants. We found that the structural disparity captured by the most explanatory axes in all datasets had significant phylogenetic signal in 58% of the test cases (14/24 tests, Table 3 ). This included some cases where PC3 (describing < 1% of the overall variation) had significant phylogenetic signal (e.g. X-linked BUSCO genes, placental mammal UCE chromosome sets). Values of K and λ were also high (indicative of putative phylogenetic signal) in several non-significant tests. Ancestral state estimations of disparity revealed an increase in structural disparity related with the evolution of an X chromosome in the ancestor of placental and marsupial mammals ( Fig. 8A ), the evolution of a pericentric inversion on chromosome 5 in the Drosophila virilis group ( Fig. 8B ), and phylogenetically associated patterns of structural disparity increase or decrease in placental mammals ( Fig. 8C, D ). Download figure Open in new tab Figure 8. Ancestral state estimation of structural variants (SV) using select principal component (PC) scores from disparity analyses conducted herein. Structural variants detected using disparity analysis often have phylogenetic signal ( Table 2 ) and can be used to quantitatively study the rate of SV evolution. Examples from this study include human X-linked BUSCO gene disparity of amniote vertebrates where structural disparity increases following the evolution of the X chromosome (PC1, A), the evolution of a speciation-associated paracentric inversion in chromosome 5 of the Drosophila virilis group that caused rearrangements in the position of UCE landmarks (PC3, B), and the identification of placental mammal orders with the least (chromosome set 1, PC1, C) and most (chromosome set 2, PC1, D) conserved UCE structure, respectively. Phylogenies were obtained from Kumar et al. [ 63 ]. View this table: View inline View popup Download powerpoint Table 3. Results of phylogenetic signal tests using K [ 64 ] and λ [ 65 ] statistics for three most explanatory principal components (PCs) of structural variant disparity from the empirical analyses conducted herein. P-values are based on 1,000 randomizations (K) and likelihood ratio tests (λ). Significant tests are indicated with bold text. Discussion Using a morphometrics-inspired approach we identified SVs associated with chromosome inversions, insertion-deletion events (=relative size change), and phylogeny. These qualities mirror the utility of morphological disparity analysis as applied to complex phenotypic structures. In a phenotypic context, disparity can be descriptive or serve as a proxy for phylogenetic affinities and ecology [ 22 ]. For example, disparity is widely used to understand adaptive radiation in the paleontological record [ 36 ]. Our simulation and empirical results suggest that, like classic morphological disparity, structural disparity of chromosomes increases through time ( Fig. 7 ). In a genomic context, we are optimistic that disparity can play a similar role in understanding how chromosome architecture varies among species and can be used for quantitative comparisons to study organismal evolution and adaptation. Below we discuss the results of our empirical case studies, suggest guidelines for best practice when conducting structural disparity analysis, and present several future directions for applications and methods development. Structural disparity and the evolution of chromosome architecture. Disparity analyses of SVs enable a broad understanding of how similar (or dissimilar) chromosome structures are across different species based on the distribution of conserved landmark sequences like BUSCO genes and UCEs. This analytical quality is particularly advantageous for studying large-scale shifts in chromosome organization across organisms relevant to the study of supergenes and homologous synteny blocks (sensu [ 16 ]). In our study we found that following the evolution of a key mammalian sex chromosome, structural disparity of 53 BUSCO genes increased compared to other amniotes ( Fig. 2A , Fig. 8A ). This disparity increase appears to be associated with an overall increase in chromosome size in those species where the landmark genes are located on an X chromosome. Due to the cessation of regular recombination, size increases and decreases in vertebrate sex chromosomes are well-documented [ 37 ]. When we scaled the positions by chromosome size, we observed that placental and marsupial mammals were no longer the most disparate group (Fig. S2E, G), further supporting chromosome size increase as the key driver of structural disparity among BUSCO genes in the mammalian X chromosome. Following size-correction, structural disparity analysis identified high disparity among the bird chromosomes analyzed, which from viewing the synteny plot ( Fig. 2B ) appears to be related to several inversions (within birds) associated with the landmarks. The ability for structural disparity analyses to capture SVs associated with inversions in closely related species or populations was also observed in our study of chromosome 5 from the Drosophila virilis group based on 191 UCE landmarks ( Fig. 3 ). In this simple example, we observed that the first three principal components all corresponded to distinctive changes previously linked to speciation [ 27 ]. Although not all axes had statistically significant phylogenetic signal ( Table 3 ), small sample sizes may have contributed to this outcome (see 38) as even those axes without significant signal (e.g. PC3) were consistent with phylogenetically associated changes ( Fig. 8B ). The de novo study of structural disparity using tetrapod UCEs with placental mammals revealed order-specific disparity patterns in PCs 1 and 2 of both chromosome sets ( Fig. 5 ). In addition to revealing clear phylogenetic patterns in UCE-related structural disparity, these analyses also revealed that disparity patterns are order-specific across different chromosomes ( Fig. 8C, D ). Specifically, using the chull function in the R package ‘sp’ [ 39 ] we found that convex hull values for each mammalian order on PCs 1 and 2 were perfectly correlated across chromosomes set 1 and set 2 (Pearson’s correlation coefficient = 0.996, p = 0.0003). This suggests that order-specific structural disparity associated with UCEs may be similar across different genomic regions. Given the importance of UCEs in development and gene regulation [ 40 ] these types of genome-wide patterns are important to explore as additional genomic data become available. Best practices for structural disparity analysis. Based on comparative analyses of the X-linked BUSCO genes from amniote vertebrates and UCEs from placental mammals, we can recommend a set of best practices for structural disparity that will allow users to avoid several analytical artifacts that we encountered. First, examining component loadings for each landmark is important for identifying outlier landmarks that may obscure more generalized signal of other landmarks. In our study of placental mammals, we found that a single outlier (UCE 6948) explained most variation in the first two PCs (Fig. S6). The presence of this UCE in both datasets demonstrates that it is not single-copy and justifies removing these types of spurious landmarks prior to disparity analysis. Second, using raw, absolute positions may lead to outlier taxa if the landmarks are clustered at one end of a large chromosome. This phenomenon was observed in both the X-linked BUSCO gene dataset and the placental mammals (Fig. S2B, S7B, S8B). In some cases, scaling the positions based on chromosome size may resolve this issue (Fig. S2F), whereas in others it does not (Fig. S7F, S8F). We found that the use of bound landmark positions resolves this analytical artifact associated with the directionality of the chromosome sequence. If size is an important parameter for understanding structural disparity, we recommend using bound, absolute positions (Fig. S2C, S2D, S7C, S7D, S8C, S8D). If patterns like inversions and landmark translocations are of primary interest, we recommend using size-corrected and bound landmark positions (Fig. S2G, S2H, S7G, S7H, S8G, S8H). Finally, if using the de novo approach and clustering chromosomes with MMDS to identify homologous landmark regions for analysis we recommend viewing various axes as bivariate plots ( Fig. 4B , 4C) where clustering may be more obvious than gap threshold sizes would suggest on the individual axes ( Table 2 ). A complementary tool for studying structural changes in chromosomes. Dimensionality reduction is already in wide use in comparative genomics (e.g., uniform manifold approximation and projection, [ 41 ]). Structural disparity analysis is therefore a natural companion method for assessing structural differences across the chromosomes of different species. It complements existing methods like WGA and synteny analysis by capturing a different dimension of how chromosomes vary across species. Synteny analysis is a powerful tool for identifying similarities in gene order and arrangement. While synteny analysis does capture spatial variation (unlike WGA), it is known to be very sensitive to parameter settings [ 42 ]. Furthermore, some methods of synteny inference are performed on post-WGA datasets meaning that structural information about the whole chromosomes has been removed a priori [ 15 ]. In contrast, landmark-based estimates of genome structure offer at least two qualities that syntenic analyses lack (Fig. S12). First, minor chromosomal rearrangements will be interpreted as breaks in synteny and ignored. In other words, inversions within syntenic regions not involving all the contiguous markers will break synteny. Second, large gap regions between contiguous orthologs can also break synteny (depending on parameter settings). The latter example highlights another important difference; disparity analyses will capture distance information between landmarks whereas synteny only provides landmark (ortholog) order information. Applications beyond diploid genomes and evolutionary biology. Apart from diploid organisms such as vertebrates and insects, the structural disparity pipeline could also be implemented on polyploid organisms when chromosome phasing has been performed. Utilizing the different phased genome assemblies, the pipeline can identify structural differences between haplotigs of the genome. In addition, the pipeline may be useful for measuring structural evolution of pangenomes without the need for a reference input genome. Instead, a class/category of genetic markers could be used as structural landmarks. As a visualization tool, structural disparity analysis may complement the routinely used pangenome graph building methods (e.g.[ 43 ]). Lastly, a potential use of structural disparity analysis relates to cancer genomics. The genomes of somatic cancer cells undergo rapid structural evolution and SVs in cancer cells are very specific to the type of cancer [ 44 ]. SVs in cancer cells have an impact on gene expression through modifying structural accessibility, DNA methylation (and more) in the cis-regulatory regions of the genome [ 44 ], therefore the ability to identify genomic rearrangement is valuable (especially when using long-read data). Avenues to expand the method towards genomic disparity. Structural disparity analyses could be developed for whole-genome comparisons. At present, the pipeline performs chromosome to chromosome comparisons, limiting comparisons across genomic regions that have undergone translocation, fission, or fusion events. With greater awareness of the frequency of these events, weighting parameters could be defined to measure global genomic disparity levels. While we limited our validation analyses to using BUSCOs and UCEs as landmarks and a handful of (largely homologous) chromosomes, landmarks could be any conserved genetic sequences which can be accurately mapped. Variation in the placement of conserved sequences can be used to study colocalization of transcribed genes (e.g. BUSCO [ 45 ]), cis-regulatory elements, enhancers, promoting regions, and more. The ethos of the approach is also scalable to three-dimensional genome structure once that information is more broadly available for many species (see [ 46 , 47 ]). The extensive literature on geometric morphometrics offers substantial guidance for how 3D comparisons may be performed in the future [ 25 ]. Landmark-based estimates of genomic disparity provide a rapid mechanism for comparisons of genome structure across species and populations, and we anticipate they will expedite discovery and study of novel structural patterns within comparative genomics. Methods Bioinformatics and identification of landmarks. The bioinformatics and computational analyses were performed on Crop Diversity HPC, described by Percival-Alwyn et al. [ 48 ]. The structural disparity pipelines are modular and can be modified for a variety of other uses. Three step-by-step tutorials along with custom R and shell scripts are available on GitHub ( https://github.com/nhm-herpetology/genomic-disparity/blob/main/README.md ). Structural disparity analysis can be conducted using a variety of genomic landmarks assuming they are single copy and orthologous (e.g. BUSCO genes [ 45 ], UCEs [ 32 , 34 ]). In this study, we used human X-linked BUSCO genes and UCEs as landmarks. We selected human X-linked genes because many of them are found in homologous blocks on a single chromosome across amniote vertebrate animals [ 18 ] and UCEs because they are highly conserved, lack repeats, and widely distributed in genomes [ 34 , 49 ]. UCEs have also been implicated as relevant to the structure of key developmental and transcriptomic pathways [ 49 , 50 ], making variation in their relative chromosomal placements of biological interest. Sensitivity analyses related to chromosome size, complementarity and landmark boundaries. Data transformations associated with object size are a well-studied component of morphometrics [ 51 ]. In our datasets where structural disparity was correlated with chromosome size, we conducted a series of analyses to establish best practice for structural disparity analysis. We analyzed these datasets in four ways using: (1) raw positions, (2) bounded, raw positions, (3) chromosome size-corrected positions, and (4) bounded, size-corrected positions. We repeated each analysis using the reverse complement version of the four datasets for a total of eight versions of the analysis. Boundary setting was achieved for each landmarked chromosome using a two-step procedure: (1) re-positioning all the landmarks relative to the first landmark and (2) using the position of the last landmark to establish the size of the genomic region being analyzed. Absolute size correction was performed with R functions that divided the total chromosome size by each landmark position. Bounded size correction was achieved by dividing the size of the bounded genomic region by each landmark position with R functions. BUSCO genes and amniote vertebrates. Using the BUSCO gene annotation file from Lovell et al. [ 18 ], we extracted all the genes annotated in the human ( Homo sapiens ) X chromosome. We utilized this list of x-linked genes and identified homologous chromosomes in other species based on the proportion of the human x-linked genes mapped on them. The other species included in the study were four placental mammals ( Mus musculus , Rhinolophus ferrumequinum , Tursiops truncatus , and Choloepus hoffmanni ), three marsupial mammals ( Sarcophilus harrisii , Monodelphis domestica , and Trichosurus vulpecula ), two monotreme mammals ( Tachyglossus aculeatus and Ornithorhynchus anatinus ), four birds ( Gallus gallus , Cygnus olor , Calypte anna , and Taeniopygia guttata ) and two squamate reptiles ( Thamnophis elegans and Lacerta agilis ). When there was more than a single chromosome that shared human x-linked genes (e.g. Tachyglossus, Ornithorhynchus , and Trichosurus ), the chromosome with the higher proportion of genes was used to maximize the number of landmarks available for structural disparity analysis. Lovell et al. (2022) also analyzed the budgie ( Melopsittacus undulatus ) genome, however, because our pipeline did not identify focal landmarks on a single chromosome, we did not include this species in the structural disparity analysis. Using the gff3 file approach ( Fig. 1 ), we identified human x-linked BUSCO genes that were homologous across all the species’ chromosomes. Detailed instructions for identifying and extracting these genes are available in Step 1 of the GitHub tutorial. Next, chromosomes are prepared for downstream analysis using a pipeline which requires the following R statistical software [ 31 ] packages: rehshape2 [ 52 ], matrixStats [ 53 ], tidyr [ 54 ], dplyr [ 55 ], and ggplot2 [ 56 ]). From 838 genes we identified 53 that were present in all focal species making them suitable for use as landmarks in a structural disparity analysis. Detailed instructions for identifying and extracting landmarks are available in Step 2 of the GitHub tutorial. We also conducted data matrix manipulation and principal component analysis in R, with code available in Step 3 of the GitHub tutorial. For size-corrections, we obtained chromosome sizes from NCBI. We tested for correlations between chromosome size and disparity using Spearman’s rho nonparametric tests in R. To create a synteny plot for comparative purposes, we used the Chromomap package in R [ 35 ]. A list of the 53 human X-linked BUSCO genes used as landmarks can be found in Appendix S1. Drosophila structural variants, speciation and UCEs. Based on the findings of Poikela et al. [ 27 ], we landmarked chromosome 5 for D. americana , D. flavomontana , D. montana , D. novamexicana , and two individuals of D. virilis . This chromosome has known inversions and translocations and may have contributed to speciation in the group. We downloaded chromosome 5 from the Drosophila virilis group either via NCBI Entrez Direct UNIX E-utilities (2) or directly from the literature (1). The following chromosomes sequence were used: D. americana (NCBI accession no: CM061086.1 ), D. flavomontana (1), D. montana (1), D. novamexicana (CM061080.1), and two individuals of D. virilis (CM017608.2 and CM061075.1). Detailed instructions for downloading chromosome sequences are available in Step 1 of the GitHub tutorial. For reference mapping of ultraconserved elements (UCEs), we utilized a published set of ultraconserved elements (UCEs) from dipterans [ 3 ]. We indexed the chromosome assemblies and reference mapped UCEs using default settings in BWA-MEM [ 29 ] and SAMtools [ 57 ]. Custom shell commands are used to loop the reference mapping and create a spreadsheet that contains presence and absence information for the UCEs on each chromosome 5 variant included in the analyses. Detailed instructions on this reference mapping approach are available in Step 2 of the associated GitHub tutorial. Next, chromosomes were prepared for downstream analysis using an R pipeline as described above. The pipeline includes three steps (i) remove mapped UCEs that the chromosomes do not share, (ii) complementarity checks to ensure all chromosomes have comparable orientations (5’ to 3’), and (iii) merging redundant UCE probes that target the same UCE locus. The pipeline resulted in the identification of 191 homologous UCE landmarks for chromosome 5 of the Drosophila virilis group. Detailed instructions for the preparation pipeline are available in Step 3 of the GitHub tutorial. In the final step of the disparity analysis, we conducted a principal component analysis (PCA) and visualized the results. The input file for the PCA is available on GitHub as ‘Drosophila_PCA.csv’. We used R base functions along with dplyr to format the data matrix and conduct the PCA. Visualization was conducted using ggplot2 [ 56 ] and cowplot [ 58 ]. Detailed instructions for this final step are available in Step 4 of the GitHub tutorial. De novo landmark discovery and placental mammal genomes. To apply our disparity pipeline to a group where homology among chromosomes was unknown, we used 26 nuclear genomes from five orders of placental mammals ( Table 1 ). Because placental mammal orders have diverged over millions of years [ 59 ] the homology of their chromosomes is fragmentary, making them a good candidate for exploring the extent to which disparity analysis can be applied to single genomic regions in divergent groups. We downloaded mammal genome assemblies using a custom script (chromosome_download.sh) that we developed to work with a standard input file (chromosome_download.config). Setting up a batch download for a subset of the mammal taxa (26 species) is described in detail in Step 1 of the GitHub tutorial. For reference mapping, we utilized a published set of tetrapod UCEs [ 32 ]. We indexed and reference-mapped UCEs on the 1,004 mammal chromosomes using default settings in BWA-MEM [ 29 ] and SAMtools [ 57 ]. Reference mapping was conducted using a custom shell script that loops through the chromosome sequences (landmark_mapping.sh). We tested the utility of minimap2 [ 30 ] for mapping conserved regions to the genome assemblies utilizing default settings on chromosome set 2 of the mammalian genomes. We then compared these results to those obtained using BWA MEM. We observed that BWA MEM mapped higher numbers of UCEs than minimap2 (Fig. S1). This difference in mapping quality could be explained by the short length of the UCE sequences used, at about 180 bp length [ 32 ] because BWA MEM can achieve higher accuracy with short-read mapping and produces less false negatives than minimap2 [ 30 ]. However, minimap2 may be preferable if larger sequences were used as landmarks, and that approach could then be easily combined with our post-mapping pipeline steps. In addition to providing a UCE presence-absence output file for downstream analysis in the pipeline, landmark_mapping.sh outputs the number of (UCE) landmarks mapped and length information for each chromosome. Detailed instructions for refence mapping are available in Step 2 of the GitHub tutorial. Interestingly, we found that the number of UCEs mapped onto each mammal chromosome was positively and significantly correlated with chromosome size (r = 0.75, p < 0.05; SI Appendix, Fig. S2). To identify sets of chromosomes that share homologous landmarks, we applied metric multidimensional scaling (MMDS) on the presence-absence matrix generated by landmark_mapping.sh. Using MMDS scores across 10 axes we identified mutually exclusive sets of chromosomes that consisted of only one chromosome from a species. These chromosome sets were identified by selecting a threshold value on the relevant axes of the MMDS plot. Detailed instructions for clustering chromosomes and thresholding are available in Step 3 of the GitHub tutorial. To determine if protein content was similar on chromosomes sets (implying that they share functional, protein-coding homologs in addition to UCE landmarks, we used protein annotations from the Ensembl database (release 110; [ 60 ]), to generate Jaccard similarity coefficients for comparing protein compositions on the two focal chromosome sets across a subset of placental species where annotations were available (N = 14). After extracting the protein-coding markers’ identities from the Ensembl annotation files of each species (for the specific chromosomes in set 1 and set 2 separately), we utilized a custom defined Jaccard’s similarity index which removed case sensitivity in the names of the markers (available on GitHub page as ‘Jaccard_proteins.R’). This index estimated pairwise similarities between each chromosome in set 1 and set 2 respectively, as well as for all chromosome pairs in both the sets. The latter was done to test if chromosomes within a set show higher similarity in protein-coding markers than chromosomes between the two sets. Following the identification of chromosome sets, we prepared them for PCA by (i) removing UCEs that did not occur on all chromosomes, (ii) checking the directional orientation (5’ to 3’), and (iii) merging redundant UCE probes. Unlike the highly conserved chromosomes from the Drosophila example, determining the directionality of mammal chromosomes was not always obvious. In other words, sometimes we observed mixed forward and reverse mappings of UCE landmarks, suggesting that only particular segments of the landmarked region were sequenced in the opposite direction of most taxa. This is likely explained by within-chromosome recombination and to account for it we used a majority rule criteria so that we only changed chromosome orientations if >50% of landmarks were mapped in the opposite direction of the reference chromosomes. Detailed instructions for the chromosome preparation pipeline are available in Step 4 of the GitHub tutorial. Finally, we conducted a PCA on each of the chromosome sets and visualized the results which are detailed in Step 5 of the GitHub tutorial. The input files for the PCA analysis are available on GitHub as ‘Set_1_mammals_PCA.csv’ and ‘Set_2_mammals_PCA.csv’, respectively. As with the Drosophila example, applying the disparity pipeline to placental mammal genomes required the following R packages: rehshape2, matrixStats, dplyr, ggplot2 and cowplot. Simulation and other tests for increased disparity through time. We simulated two common evolutionary shifts in chromosome structure that occur as species and populations diverge: insertion-deletion (indel) events and chromosomal inversions. We performed these simulations using chromosome set 1 from three species subsampled from the larger dataset, Capra aegagrus , Mus pahari and Neomonachus schauinslandi ( Table 1 ). We selected these taxa based on phylogenetic diversity (representing Artiodactyla, Carnivora and Rodentia) and being outliers in some analyses ( Capra aegragrus , Fig. S9). We simulated evolutionary events using SimuG [ 61 ]. We designed the experiment logarithmically, simulating 1, 10, 100, 1,000, and 10,000 random indels or inversions on the ancestral chromosome of the five focal species. We ran five replicated simulations for each of the five focal species. To assess the change in disparity between ancestral chromosomes and their simulated descendants, we used a simple pairwise metric of increase in disparity (ID): Where PC A is the principal component score of the ancestral chromosome, PC B is the principal component score of the descendent chromosome, and ID is calculated by taking the absolute difference of these values. As such, ID is an intuitive measurement of how much landmark placement changed during the simulations, where a value of 0 indicates no change and larger values correspond to increasingly disparate landmark placements. We predicted that if genomic disparity reflects the degree of evolutionary change of architecture between two genomes, there should be a positive correlation between ID and the number of simulated events. While analysis of individual PCs may be problematic when data have high effective dimensionality [ 62 ], here the dominance of PC1 and the overwhelming correlation with chromosome size on that primary axis, justifies separate analysis of these components. We compared the original chromosome (set 1) sequence from three species of mammals, referred to as the ‘ancestral chromosome’, to simulated ‘descendent chromosomes’ which we made by simulating random indel or inversion events (see main text for more detail). Following the disparity pipeline described above, we mapped UCEs onto the simulated chromosomes, extracted homologous landmarks and carried out a PCA. PCA also included ancestral chromosomes from each species. We calculated the disparity between the ancestral chromosomes and simulated descendent chromosomes using the equation described in the main manuscript. Disparity was calculated using the logarithm of the absolute difference between the PC values of the ancestral chromosome and each simulated chromosome. Furthermore, we tested for differences between the disparity measures of indel-simulated and inversion-simulated chromosomes using two-way ANOVAs implemented in R. To provide empirical context for the simulation results, we tested for the signature of disparity increase in our placental mammal datasets using nonparametric Wilcoxon tests in R. If UCE-inferred disparity increases with evolutionary time, we expected inter-order disparity to be significantly higher than intra-order disparity for the most explanatory PCs because there has been more evolutionary divergence between mammalian orders than within them. We also generated Jaccard similarity coefficients for comparing protein compositions between and within mammalian orders. Tests for phylogenetic signal and ancestral state estimation. Using the three most explanatory structural disparity PCs, we tested for phylogenetic signal in the (1) amniote vertebrate, (2) Drosophila virilis group, and (3) placental mammal (both chromosome sets 1 and 2) datasets. Phylogenies for each dataset were obtained from Kumar et al. [ 63 ]. We used two common metrics for phylogenetic signal, K [ 64 ] and λ [ 65 ]. We implemented test using the phytools package [ 66 ] in R. We assessed the significance of the tests using 1,000 randomizations (K) and likelihood ratio tests of the observed data to that estimated under Brownian motion (λ). To place structural disparity of genomic regions in a phylogenetic context, we estimated putative ancestral disparity values we used the ‘fastAnc’ function in phytools for select PCs in each dataset. Supplementary Information Download figure Open in new tab Figure S1. Number of ultraconserved elements (UCEs) mapped utilizing two different mapping software: minimap2 (blue) and bwa-mem (grey). Demonstrated using a subset of chromosome assemblies picked from the ‘chromosome set 2’ of the placental mammals UCE dataset discussed in the main text. Download figure Open in new tab Figure S2. Principal component (PC) sensitivity analyses of 53 X-linked BUSCO landmarks in amniote vertebrates taken from Lovell et al. (2022). The first row are PCAs conducted using raw, absolute positions for the original direction (A) and reverse complement (B), respectively. The second row are PCAs conducted using bounded, absolute positions for the original direction (C) and the reverse complement (D). The third row are PCAs conducted using size-correction with chromosome size for the original direction (E) and the reverse complement (F). The fourth row of plots are PCAs conducted using size-correction using bounded chromosome sizes and positions for the original direction (G) and reverse complement (H). The chicken ( Gallus gallus ) is highlighted in plot B to illustrate an analytical artifact associated with clustered landmarks on large chromosomes. Organism silhouettes downloaded from https://www.phylopic.org/ . Download figure Open in new tab Figure S3. Jaccard similarity coefficients for within and between protein-coding gene content in the two focal chromosome sets used in the study. Download figure Open in new tab Figure S4. Principal component analyses (PCAs) of ultraconserved element (UCE) landmarks in placental mammals for chromosome set 1 before and after the removal of an outlier landmark (UCE 6948). The first row of plots are PCAs conducted using raw, absolute positions for the original direction (A) and reverse complement (B), respectively. The second row of plots are PCAs conducted using bounded, absolute positions for the original direction (C) and the reverse complement (D). The third row of plots are PCAs conducted using size-correction with chromosome size for the original direction (E) and the reverse complement (F). The fourth row of plots are PCAs conducted using size-correction using bounded chromosome sizes and positions for the original direction (G) and reverse complement (H). The first column of plots are PCAs conducted with all 186 UCE landmarks and the second column of plots are PCAs based on 185 UCE landmarks (outlier removed). Download figure Open in new tab Figure S5. Principal component analyses (PCAs) of ultraconserved element (UCE) landmarks in placental mammals for chromosome set 2 before and after the removal of an outlier landmark (UCE 6948). The first row of plots are PCAs conducted using raw, absolute positions for the original direction (A) and reverse complement (B), respectively. The second row of plots are PCAs conducted using bounded, absolute positions for the original direction (C) and the reverse complement (D). The third row of plots are PCAs conducted using size-correction with chromosome size for the original direction (E) and the reverse complement (F). The fourth row of plots are PCAs conducted using size-correction using bounded chromosome sizes and positions for the original direction (G) and reverse complement (H). The first column of plots are PCAs conducted with all 82 UCE landmarks and the second column of plots are PCAs based on 81 UCE landmarks (outlier removed). Download figure Open in new tab Figure S6. ChromoMap plots visualizing principal component analysis (PCA) loadings of landmarks on PC1, PC2, and PC3 in the raw landmark PCA, bounded landmark PCA, and size corrected landmark PCA of chromosome set 1 (A) and set 2 (B) from the placental mammal visualized using links between Bos indicus and Equus asinus . Respective variances explained by each principal component are listed in the plot. A single landmark (UCE 6948) was found to be uniquely weighed on the various PCs identifying it as an outlier landmark. Download figure Open in new tab Figure S7. Principal component (PC) sensitivity analyses of outlier removed UCE landmarks in placental mammals chromosome set 1. The first row are PCAs conducted using raw, absolute positions for the original direction (A) and reverse complement (B), respectively. The second row are PCAs conducted using bounded, absolute positions for the original direction (C) and the reverse complement (D). The third row are PCAs conducted using size-correction with chromosome size for the original direction (E) and the reverse complement (F). The fourth row are PCAs conducted using size-correction using bounded chromosome sizes and positions for the original direction (G) and reverse complement (H). Download figure Open in new tab Figure S8. Principal component (PC) sensitivity analyses of outlier removed UCE landmarks in placental mammals chromosome set 2. The first row are PCAs conducted using raw, absolute positions for the original direction (A) and reverse complement (B), respectively. The second row are PCAs conducted using bounded, absolute positions for the original direction (C) and the reverse complement (D). The third row are PCAs conducted using size-correction with chromosome size for the original direction (E) and the reverse complement (F). The fourth row are PCAs conducted using size-correction using bounded chromosome sizes and positions for the original direction (G) and reverse complement (H). Download figure Open in new tab Figure S9. Ordination results from principal component (PC) analysis of UCE landmark positions in two sets of chromosomes from 26 species of placental mammals using bounded raw positions after the removal of an outlier landmark (see text). Representatives for five orders (Artiodactyla, Carnivora, Perissodactyla, Primates and Rodents) were included (Table1). The depicted structural variation is inferred from PCs 3 and 4 of 185 (chromosome set 1, A) and 81 (chromosome set 2, B) landmarks, respectively. In both datasets, the first two PCs explained > 99% of the variation (see Figure 5 , main text). While explaining < 1% of the overall variation, PCs 3 and 4 mostly separated mammalian orders in geno-metric space. Carpa aegragus was a notable outlier in chromosome set 1. Download figure Open in new tab Figure S10. Boxplots comparing increase in disparity (ID) between indel and inversion simulations for principal component (PC)1 (A) and PC2 (B), respectively. Note that inversions cause significantly greater ID than indels. Download figure Open in new tab Figure S11. Jaccard similarity coefficients for within and between protein-coding gene content in mammalian orders used in the study. Download figure Open in new tab Figure S12. Schematic diagram illustrating sensitivities of disparity analysis and synteny analysis to different landmark scenarios. Scenarios are described from top to bottom of the figure. First, both analytical approaches will report highly conserved landmark/ortholog order, indicating when two species have highly conserved genomic architecture. Second, both approaches will report large inversions among landmark positions. Third, depending on parameter settings syntenic analysis may ignore small landmark rearrangements whereas disparity analysis captures this variation. Finally, synteny analysis ignores spatial variation in landmark placement and ‘synteny’ can be broken if chromosomal gaps are large enough whereas disparity analysis captures this variation in landmark placement. Acknowledgments We thank members of the Goswami Lab and NHM Herpetology Group for many helpful comments on experimental design and methods development. The authors acknowledge ChatGPT for assistance with some R coding ( https://openai.com/index/chatgpt/ ). We thank Kate Thomas for an introduction to several R packages. Several delegates at the 2023 Society for Integrative and Comparative Biology (Austin, Texas, USA) and 2023 Evolution (Albuquerque, New Mexico, USA) and Young Investigators meeting 2024 (Bhopal, India) conferences provided valuable suggestions on earlier version of this work and putative applications for structural disparity analysis. We thank Jason Head and members of the Department of Zoology at the University of Cambridge for helpful suggestions related to size correction. We thank Christophe Liedtke for discussions on dimensionality reduction in genomics. We are indebted to Brant Faircloth and two anonymous reviewers for comments that greatly improved the quality of this manuscript. We are grateful to Iain Milne for bioinformatics support. The authors acknowledge Research Computing at the James Hutton Institute for providing computational resources and technical support for the “UK’s Crop Diversity Bioinformatics HPC” (BBSRC grants BB/S019669/1 and BB/X019683/1), use of which has contributed to the results reported within this paper. This study was part of a project funded by the European Commission through Marie Skłodowska-Curie Actions to A.V.M. (Grant Agreement ID: 101027832). This work also received support from the Genomics Research Theme at the Natural History Museum, London. Appendix S1. List of 53 BUSCO genes identified as landmarks for the human X-linked study of amniote vertebrates. eda, dlg3, snx12, gjb1, zmym3, nono, shroom4, amot, lrch2, pls3, plp1, gla, btk, drp2, sytl4, srpx2, tspan6, dach2, apool, rps6ka6, lpar4, magt1, fgf16, zdhhc15, uprt, nsdhl, vma21, cd99l2, mtmr1, mtm1, aff2, fmr1, slitrk4, atp11c, cd40lg, vgll1, htatsf1, gpc3, rap2c, frmd7, bcorl1, utp14a, smarca1, stag2, cul4b, lamp2, tmem255a Funder Information Declared European Union, https://ror.org/019w4f821 , 101027832 Footnotes The revised version has been prepared after two rounds of reviews and has significant improvements on clarity and example datasets used to demonstrate the proposed pipeline. https://github.com/nhm-herpetology/genomic-disparity/blob/main/README.md References 1. ↵ Genereux DP , Zody MC , Shen W , McGaugh SE , Stemple DO , Roberts AC , et al. 2020 . A comparative genomics multitool for scientific discovery and conservation . Nature 587 : 240 – 245 . doi: 10.1038/s41586-020-2874-7 OpenUrl CrossRef PubMed 2. ↵ Feng S , Stiller JW , Deng Y , Armstrong J , Burt DW , Kang H , et al. 2020 . Dense sampling of bird diversity increases power of comparative genomics . Nature 587 : 252 – 257 . doi: 10.1038/s41586-020-2873-8 OpenUrl CrossRef PubMed 3. ↵ Osmanski AB , Maples B , Blahnik J , Griesmann L , Willis WM , Schmitz RJ , et al. 2023 . Insights into mammalian TE diversity through the curation of 248 genome assemblies . Science 380 :eabn1430. doi: 10.1126/science.abn1430 OpenUrl CrossRef PubMed 4. ↵ Rubin GM , Yandell MD , Wortman JR , Clayton DF , Meeker S , Filipski AJ , et al. 2000 . Comparative genomics of the eukaryotes . Science 287 : 2204 – 2215 . doi: 10.1126/science.287.5461.2204 OpenUrl Abstract / FREE Full Text 5. ↵ Zhang X , Wessler SR . 2004 . Genome-wide comparative analysis of the transposable elements in the related species Arabidopsis thaliana and Brassica oleracea . Proc Natl Acad Sci U S A 101 : 5589 – 5594 . doi: 10.1073/pnas.0401243101 OpenUrl Abstract / FREE Full Text 6. ↵ Yuan Y , Liu X , Wang J , Sun Z , Smith AB , Peterson DG , et al. 2021 . Comparative genomics provides insights into the aquatic adaptations of mammals . Proc Natl Acad Sci U S A 118 : e2106080118 . doi: 10.1073/pnas.2106080118 OpenUrl Abstract / FREE Full Text 7. ↵ Lynch M , Walsh B . 2007 . The origins of genome architecture . Sunderland (MA): Sinauer Associates . 8. ↵ Krupina K , Goginashvili A , Cleveland DW . 2024 . Scrambling the genome in cancer: causes and consequences of complex chromosome rearrangements . Nat Rev Genet 25 : 196 – 210 . doi: 10.1038/s41576-023-00574-6 OpenUrl CrossRef PubMed 9. ↵ Hu W , Xie J , Song X , Chen S , Ling H , Li Z , et al. 2020 . Nondestructive 3D image analysis pipeline to extract rice grain traits using X-ray computed tomography . Plant Phenomics 2020 : 3414926 . doi: 10.34133/2020/3414926 OpenUrl CrossRef PubMed 10. ↵ Graves JAM . 2016 . Evolution of vertebrate sex chromosomes and dosage compensation . Nat Rev Genet 17 : 33 – 46 . doi: 10.1038/nrg.2015.2 OpenUrl CrossRef PubMed 11. ↵ Banse P , et al. Forward-in-time simulation of chromosomal rearrangements: The invisible backbone that sustains long-term adaptation . Mol Ecol [Internet]. n.d.;(n/a):[page-unspecified ]. 12. ↵ Chaves R , Adega F , Santos S , Guedes-Pinto H , Heslop Harrison JS . 2002 . [Title missing] . Cytogenet Cell Genet . 96 : 113 – 116 . doi: 10.1159/000063020 OpenUrl CrossRef PubMed Web of Science 13. ↵ Balachandran P , Beck CR . 2020 . Structural variant identification and characterization . Chromosome Res . 28 : 31 – 47 . doi: 10.1007/s10577-019-09623-z OpenUrl CrossRef PubMed 14. ↵ De Coster W , Van Broeckhoven C. 2019 . Newest methods for detecting structural variations . Trends Biotechnol . 37 : 973 – 982 . OpenUrl PubMed 15. ↵ Krasheninnikova K , et al. 2020 . halSynteny: a fast, easy-to-use conserved synteny block construction method for multiple whole-genome alignments . GigaScience . 9 : giaa047 . doi: 10.1093/gigascience/giaa047 OpenUrl CrossRef 16. ↵ Larkin DM , et al. 2009 . Breakpoint regions and homologous synteny blocks in chromosomes have different evolutionary histories . Genome Res . 19 : 770 – 777 . doi: 10.1101/gr.087407.108 OpenUrl Abstract / FREE Full Text 17. Chaney L , Sharp AR , Evans CR , Udall JA . 2016 . Genome mapping in plant comparative genomics . Trends Plant Sci . 21 : 770 – 780 . doi: 10.1016/j.tplants.2016.05.003 OpenUrl CrossRef PubMed 18. ↵ Lovell JT , et al. 2022 . GENESPACE: syntenic pan-genome annotations for eukaryotes . bioRxiv . 2022.03.09.483468 . doi: 10.1101/2022.03.09.483468 OpenUrl Abstract / FREE Full Text 19. ↵ Quigley S , Damas J , Larkin DM , Farré M . 2023 . syntenyPlotteR: a user-friendly R package to visualize genome synteny, ideal for both experienced and novice bioinformaticians . Bioinformatics Adv . 3 : vbad161 . doi: 10.1093/bioadv/vbad161 OpenUrl CrossRef 20. ↵ Armstrong J , Hickey G , Diekhans M , et al. 2020 . Progressive Cactus is a multiple-genome aligner for the thousand-genome era . Nature . 587 : 246 – 251 . doi: 10.1038/s41586-020-2871-y OpenUrl CrossRef PubMed 21. ↵ Foote M . 1997 . The evolution of morphological diversity . Annu Rev Ecol Evol Syst 28 : 129 – 152 . doi: 10.1146/annurev.ecolsys.28.1.129 philpapers.org+1onlinelibrary.wiley.com+1 OpenUrl CrossRef Web of Science 22. ↵ Guillerme T , Cooper N , Martin RD , Wills MA . 2020 . Disparities in the analysis of morphological disparity . Biol Lett 16 : 20200199 . doi: 10.1098/rsbl.2020.0199 OpenUrl CrossRef PubMed 23. ↵ Fabre AC , Botero-Castro F , Delsuc F , Douzery EJ , Goswami A . 2020 . Metamorphosis shapes cranial diversity and rate of evolution in salamanders . Nat Ecol Evol 4 : 1129 – 1140 . doi: 10.1038/s41559-020-1200-4 OpenUrl CrossRef PubMed 24. ↵ Briggs DEG , Fortey RA , Wills MA . 1992 . Morphological disparity in the Cambrian . Science 256 : 1670 – 1673 . doi: 10.1126/science.256.5064.1670 OpenUrl Abstract / FREE Full Text 25. ↵ Zelditch ML , Lundrigan BL , Garland T Jr . . 2004 . Developmental regulation of skull morphology . I. Ontogenetic dynamics of variance. Evol Dev 6 : 194 – 206 . doi: 10.1111/j.1525-142X.2004.04025.x OpenUrl CrossRef PubMed Web of Science 26. ↵ Christmas MJ , et al. 2023 . Evolutionary constraint and innovation across hundreds of placental mammals . Science 380 : eabn3943 . doi: 10.1126/science.abn3943 OpenUrl CrossRef PubMed 27. ↵ Poikela N , Laetsch DR , Hoikkala V , Lohse K , Kankare M . 2024 . Chromosomal inversions and the demography of speciation in Drosophila montana and Drosophila flavomontana . Genome Biol Evol 16 : evae024 . doi: 10.1093/gbe/evae024 OpenUrl CrossRef PubMed 28. ↵ Bejerano G , Pheasant M , Makunin I , Stephen S , Kent WJ , Mattick JS , et al. 2004 . Ultraconserved elements in the human genome . Science 304 : 1321 – 1325 . doi: 10.1126/science.1098119 OpenUrl Abstract / FREE Full Text 29. ↵ Li H , Durbin R . 2009 . Fast and accurate short read alignment with Burrows– Wheeler transform . Bioinformatics 25 : 1754 – 1760 . doi: 10.1093/bioinformatics/btp324 OpenUrl CrossRef PubMed Web of Science 30. ↵ Li H . 2018 . Minimap2: pairwise alignment for nucleic acid sequences . Bioinformatics 34 : 3094 – 3100 . doi: 10.1093/bioinformatics/bty191 OpenUrl CrossRef PubMed 31. ↵ R Core Team . 2021 . R: A language and environment for statistical computing. Vienna (Austria): R Foundation for Statistical Computing . Available from: https://www.R-project.org/faircloth-lab.orgbibsonomy.org+10ropensci.org+10scirp.org+10 32. ↵ Faircloth BC . 2017 . Identifying conserved genomic elements and designing universal bait sets to enrich them . Methods Ecol Evol 8 : 1103 – 1112 . doi: 10.1111/2041-210X.12754 academic.oup.com+9faircloth-lab.org+9pmc.ncbi.nlm.nih.gov+9 OpenUrl CrossRef 33. ↵ Yusuf LH , Tyukmaeva V , Hoikkala A , Ritchie MG . 2022 . Divergence and introgression among the virilis group of Drosophila . Evol Lett 6 : 537 – 551 . OpenUrl PubMed 34. ↵ Faircloth BC , Branstetter MG , White ND , Brady SG , Fisher BL , Joubert MF , et al. 2012 . Ultraconserved Elements Anchor Thousands of Genetic Markers Spanning Multiple Evolutionary Timescales . Syst Biol 61 : 717 – 726 . doi: 10.1093/sysbio/sys004 arxiv.org+6faircloth-lab.org+6pmc.ncbi.nlm.nih.gov+6 OpenUrl CrossRef PubMed 35. ↵ Anand L , Rodriguez Lopez CM . 2022 . ChromoMap: an R package for interactive visualization of multi omics data and annotation of chromosomes . BMC Bioinformatics 23 : 33 . doi: 10.1186/s12859-021-04516-6 OpenUrl CrossRef PubMed 36. ↵ Ruta M , Angielczyk KD , Fröbisch J , Benton MJ . 2013 . Decoupling of morphological disparity and taxic diversity during the adaptive radiation of anomodont therapsids . Proc R Soc B 280 : 20131071 . doi: 10.1098/rspb.2013.1071 OpenUrl CrossRef PubMed 37. ↵ Schartl M , Schmid M , Nanda I . 2016 . Dynamics of vertebrate sex chromosome evolution: from equal size to giants and dwarfs . Chromosoma 125 : 553 – 571 . doi: 10.1007/s00412-015-0569-y OpenUrl CrossRef PubMed 38. Münkemüller T , et al. 2012 . How to measure and test phylogenetic signal . Methods Ecol Evol 3 : 743 – 756 . doi: 10.1111/j.2041-210X.2012.00196.x OpenUrl CrossRef 39. ↵ Bivand R , Pebesma E , Gómez Rubio V . 2013 . Applied Spatial Data Analysis with R , 2nd ed. New York : Springer . 40. ↵ Cummins M , Watson C , Edwards RJ , Mattick JS . 2024 . The evolution of ultraconserved elements in vertebrates . Mol Biol Evol 41 : 1724 – 1734 . doi: 10.1093/molbev/msae146 OpenUrl CrossRef 41. ↵ Dorrity MW , Saunders LM , Queitsch C , Fields S , Trapnell C . 2020 . Dimensionality reduction by UMAP to visualize physical and genetic interactions . Nat Commun 11 : 1537 . doi: 10.1038/s41467-020-15351-4 doaj.org+9nature.com+9pubmed.ncbi.nlm.nih.gov+9 OpenUrl CrossRef PubMed 42. ↵ Liu D , Hunt M , Tsai IJ . 2018 . Inferring synteny between genome assemblies: a systematic evaluation . BMC Bioinformatics 19 : 26 . doi: 10.1186/s12859-018-2008-1 OpenUrl CrossRef PubMed 43. ↵ Garrison E , Sirén J , Novak AM , Hickey G , Eizenga JM , Paten B , et al. 2024 . Building pangenome graphs . Nat Methods 21 : 1 – 5 . doi: 10.1038/s41592-024-01450-1 OpenUrl CrossRef PubMed 44. ↵ Dubois F , Sidiropoulos N , Weischenfeldt J , Beroukhim R . 2022 . Structural variations in cancer and the 3D genome . Nat Rev Cancer 22 : 533 – 546 . doi: 10.1038/s41571-022-00698-0 OpenUrl CrossRef PubMed 45. ↵ Manni M , Berkeley MR , Seppey M , Simão FA , Zdobnov EM . 2021 . BUSCO Update: Novel and streamlined workflows along with broader and deeper phylogenetic coverage for scoring of eukaryotic, prokaryotic, and viral genomes . Mol Biol Evol 38 : 4647 – 4654 . doi: 10.1093/molbev/msab199 OpenUrl CrossRef PubMed 46. ↵ Oluwadare O , Highsmith M , Cheng J . 2019 . An overview of methods for reconstructing 3 D chromosome and genome structures from Hi C data . Biol Proced Online 21 : 7 . doi: 10.1186/s12575-019-0095-0 OpenUrl CrossRef PubMed 47. ↵ Lappala A , DeMaere MZ , Dansereau C , Villani TS , Lieberman Aiden E , Dekker J , et al. 2021 . Four dimensional chromosome reconstruction elucidates the spatiotemporal reorganization of the mammalian X chromosome . Proc Natl Acad Sci U S A 118 : e2107092118 . doi: 10.1073/pnas.2107092118 OpenUrl Abstract / FREE Full Text 48. ↵ Percival Alwyn L , Smith J , Patel K , Ahmed A , Zhang Y , Vickers M , et al. 2024 . UKCropDiversity HPC: A collaborative high performance computing resource approach for sustainable agriculture and biodiversity conservation . Plants People Planet . doi: 10.1002/ppp3.10607 OpenUrl CrossRef 49. ↵ Sandelin A , Klasson L , Medstrand P , Engström U , Fredman D , Lenhard B. 2004 . Arrays of ultraconserved non coding regions span the loci of key developmental genes in vertebrate genomes . BMC Genomics 5 : 99 . doi: 10.1186/1471-2164-5-99 OpenUrl CrossRef PubMed 50. ↵ Leclair NK , Anczuków O . 2022 . ‘ Poisoning’ of the transcriptome by ultraconserved elements . Nat Rev Mol Cell Biol 23 : 777 . doi: 10.1038/s41580-022-00527-1 europepmc.org+2pubmed.ncbi.nlm.nih.gov+2researchgate.net+2 OpenUrl CrossRef 51. ↵ Jungers WL , Falsetti AB , Wall CE . 1990 . Shape, relative size, and size-adjustments in morphometrics . J Hum Evol 19 : 333 – 348 . doi: 10.1016/0047-2484(90)90043-W OpenUrl CrossRef 52. ↵ Wickham H. 2007 . Reshaping data with the reshape package . J Stat Softw 21 : 1 – 20 . Available from: http://www.jstatsoft.org/v21/i12/ OpenUrl CrossRef PubMed 53. ↵ Müller P , Eddelbuettel D , Bates D. 2025 . matrixStats: Functions for Highly Optimized Computing of Matrix Statistics . Available from: https://cran.r-project.org/web/packages/matrixStats/matrixStats.pdf 54. ↵ Wickham H , Vaughan D , Girlich M . 2024 . tidyr: Tidy messy data. R package version 1.3.1 . Available from: https://github.com/tidyverse/tidyr , https://tidyr.tidyverse.org 55. ↵ Wickham H , François R , Henry L , Müller K , Vaughan D . 2023 . dplyr: A grammar of data manipulation. R package version 1.1.4 . Available from: https://github.com/tidyverse/dplyr 56. ↵ Wickham H . 2016 . ggplot2: Elegant graphics for data analysis . New York : Springer-Verlag . Available from: https://ggplot2.tidyverse.org 57. ↵ Danecek P , Bonfield JK , Liddle J , Marshall J , et al. 2021 . Twelve years of SAMtools and BCFtools . GigaScience 10 : giab008 . doi: 10.1093/gigascience/giab008 OpenUrl CrossRef PubMed 58. ↵ Wilke C. 2024 . cowplot: Streamlined plot theme and plot annotations for ‘ggplot2’. R package version . Available from: https://cran.r-project.org/package=cowplot 59. ↵ Foley NM , et al. 2023 . A genomic timescale for placental mammal evolution . Science 380 : 1239 – 1245 . doi: 10.1126/science.abl8189 OpenUrl CrossRef 60. ↵ Martin FJ , et al. 2023 . Ensembl 2023 . Nucleic Acids Res 51 : D933 – D941 . doi: 10.1093/nar/gkac1040 OpenUrl CrossRef PubMed 61. ↵ Yue JX , Liti G. 2019 . simuG: a general-purpose genome simulator . Bioinformatics 35 : 4442 – 4444 . doi: 10.1093/bioinformatics/btz366 OpenUrl CrossRef PubMed 62. ↵ Uyeda JC , Caetano DS , Pennell MW . 2015 . Comparative analysis of principal components can be misleading . Syst Biol 64 : 677 – 689 . doi: 10.1093/sysbio/syv019 OpenUrl CrossRef PubMed 63. ↵ Kumar S , Stecher G , Suleski M , Hedges SB . 2022 . TimeTree 5: an expanded resource for species divergence times . Mol Biol Evol 39 : msac174 . doi: 10.1093/molbev/msac174 OpenUrl CrossRef PubMed 64. ↵ Blomberg SP , Garland T Jr . , Ives AR . 2003 . Testing for phylogenetic signal in comparative data: behavioral traits are more labile . Evolution 57 : 717 – 745 . doi: 10.1111/j.0014-3820.2003.tb00285.x OpenUrl CrossRef PubMed Web of Science 65. ↵ Pagel M . 1999 . Inferring the historical patterns of biological evolution . Nature 401 : 877 – 884 . doi: 10.1038/44766 OpenUrl CrossRef PubMed Web of Science 66. ↵ Revell LJ . 2024 . phytools 2.0: an updated R ecosystem for phylogenetic comparative methods (and other things) . PeerJ 12 : e16505 . doi: 10.7717/peerj.16505 OpenUrl CrossRef PubMed 67. Canavez FC , Luche DD , Stothard P , Leite KR , Sousa-Canavez JM , Plastow G , et al. 2012 . Genome sequence and assembly of Bos indicus . J Hered 103 : 342 – 348 . doi: 10.1093/jhered/esr150 OpenUrl CrossRef PubMed 68. Rosen BD , Bickhart DM , Koren S , Schnabel RD , Hall R , Zimin A , et al. 2020 . De novo assembly of the cattle reference genome with single-molecule sequencing . Gigascience 9 : giaa021 . doi: 10.1093/gigascience/giaa021 OpenUrl CrossRef 69. Ananthasayanam S , Kothandaraman H , Nayee N , Saha S , Baghel DS , Gopalakrishnan K , et al. 2019 . First near complete haplotype phased genome assembly of River buffalo (Bubalus bubalis) . bioRxiv 618785 . doi: 10.1101/618785 OpenUrl Abstract / FREE Full Text 70. Dong Y , Zhang X , Xie M , Arefnezhad B , Wang Z , Wang W , et al. 2015 . Reference genome of wild goat (Capra aegagrus) and sequencing of goat breeds provide insight into genic basis of goat domestication . BMC Genomics 16 : 1 – 11 . doi: 10.1186/s12864-015-1925-6 OpenUrl CrossRef PubMed 71. Bickhart DM , Rosen BD , Koren S , Sayre BL , Hastie AR , Chan S , et al. 2017 . Single-molecule sequencing and chromatin conformation capture enable de novo reference assembly of the domestic goat genome . Nat Genet 49 : 643 – 650 . doi: 10.1038/ng.3802 OpenUrl CrossRef PubMed 72. Farré M , Li Q , Darolti I , Zhou Y , Damas J , Proskuryakova AA , et al. 2019 . An integrated chromosome-scale genome assembly of the Masai giraffe (Giraffa camelopardalis tippelskirchi) . GigaScience 8 : giz090 . doi: 10.1093/gigascience/giz090 OpenUrl CrossRef PubMed 73. Davenport KM , Bickhart DM , Worley K , Murali SC , Salavati M , Clark EL , et al. 2022 . An improved ovine reference genome assembly to facilitate in-depth functional annotation of the sheep genome . Gigascience 11 : giab096 . doi: 10.1093/gigascience/giab096 OpenUrl CrossRef 74. Bredemeyer KR , Harris AJ , Li G , Zhao L , Foley NM , Roelke-Parker M , et al. 2021 . Ultracontinuous single haplotype genome assemblies for the domestic cat (Felis catus) and Asian leopard cat (Prionailurus bengalensis) . J Hered 112 : 165 – 173 . doi: 10.1093/jhered/esaa047 OpenUrl CrossRef 75. Mohr DW , Naguib A , Weisenfeld NI , Kumar V , Shah P , Church DM , et al. 2017 . Improved de novo genome assembly: linked-read sequencing combined with optical mapping produce a high-quality mammalian genome at relatively low cost . bioRxiv 128348 . doi: 10.1101/128348 OpenUrl Abstract / FREE Full Text 76. Bredemeyer KR , Murphy WJ . 2021 . Unpublished Liger assemblies . Veterinary Integrative Biosciences , Texas A&M University, College Station, TX . Submitted 11 Jan 2021. 77. Wang G , Brändl B , Rohrandt C , Hong K , Pang A , Lee J , et al. 2021 . Chromosome-level genome assembly of the functionally extinct northern white rhinoceros (Ceratotherium simum cottoni) . bioRxiv . doi: 10.1101/2021.12.01.470802 OpenUrl Abstract / FREE Full Text 78. Wang C , Li H , Guo Y , Huang J , Sun Y , Min J , et al. 2020 . Donkey genomes provide new insights into domestication and selection for coat color . Nat Commun 11 : 6014 . doi: 10.1038/s41467-020-19702-6 OpenUrl CrossRef PubMed 79. Kalbfleisch TS , Rice ES , DePriest MS Jr . , Walenz BP , Hestand MS , Vermeesch JR , et al. 2018 . Improved reference genome for the domestic horse increases assembly contiguity and composition . Commun Biol 1 : 197 . doi: 10.1038/s42003-018-0209-z OpenUrl CrossRef PubMed 81. Gordon D , Murali SC , Ventura M , Mercuri L , Eichler EE . De novo PacBio assembly of a female gorilla (Kamilah) . Genome Sciences, University of Washington, Seattle, WA, USA. Unpublished . 81. Jayakumar V , Nishimura O , Kadota M , Hirose N , Sano H , Murakawa Y , et al. 2021 . Chromosomal-scale de novo genome assemblies of cynomolgus macaque and common marmoset . Sci Data 8 : 159 . doi: 10.1038/s41597-021-00911-0 OpenUrl CrossRef PubMed 82. Dutcher S , Fulton R , Lindsay T. The rhesus macaque (Macaca mulatta) genome assembly . The McDonnell Genome Institute, Washington University School of Medicine ; University of Washington , Seattle, WA . Unpublished. 83. Kronenberg ZN , Fiddes IT , Gordon D , Murali S , Cantsilieris S , Meyerson OS , et al. 2018 . High-resolution comparative analysis of great ape genomes . Science 360 : eaar6343 . doi: 10.1126/science.aar6343 OpenUrl Abstract / FREE Full Text 84. Batra SS , Levy-Sakin M , Robinson J , Guillory J , Durinck S , Vilgalys TP , et al. 2020 . Accurate assembly of the olive baboon (Papio anubis) genome using long-read and Hi-C data . Gigascience 9 : giaa134 . doi: 10.1093/gigascience/giaa134 OpenUrl CrossRef PubMed 85. Rupp O , MacDonald ML , Li S , Dhiman H , Polson S , Griep S , et al. 2018 . A reference genome of the Chinese hamster based on a hybrid assembly strategy . Biotechnol Bioeng 115 : 2087 – 2100 . doi: 10.1002/bit.2680986 OpenUrl CrossRef 86. Roller M , Navarro FCP , Fiddes I , Streeter I , Feig C , Martin-Galvez D , et al. Repeat associated mechanisms of genome evolution and function revealed by the Mus caroli and Mus pahari genomes . Available from: http://bura.brunel.ac.uk/handle/2438/16068 87. Church DM , Goodstadt L , Hillier LW , Zody MC , Goldstein S , She X , et al. 2009 . Lineage-specific biology revealed by a finished genome assembly of the mouse . PLoS Biol 7 : e1000112 . doi: 10.1371/journal.pbio.1000112 OpenUrl CrossRef PubMed 88. Lilue J , Doran AG , Fiddes IT , Abrudan M , Armstrong J , Bennett R , et al. 2018 . Sixteen diverse laboratory mouse reference genomes define strain-specific haplotypes and novel functional loci . Nat Genet 50 : 1574 – 1583 . doi: 10.1038/s41588-018-0223-8 OpenUrl CrossRef PubMed 89. Lassance JM , Hoekstra HE . Improved assembly of the deer mouse (Peromyscus maniculatus) genome . Organismic and Evolutionary Biology / Molecular and Cellular Biology, Harvard University / Howard Hughes Medical Institute , Cambridge, MA, USA . Unpublished. 90. Howe K , Dwinell M , Shimoyama M , Corton C , Betteridge E , Dove A , et al. 2021 . The genome sequence of the Norway rat, Rattus norvegicus Berkenhout 1769 . Wellcome Open Res 6 : 60 . doi: 10.12688/wellcomeopenres.17242.1 OpenUrl CrossRef PubMed 91. Sievert C. 2020 . Interactive web-based data visualization with R, plotly, and shiny . Chapman and Hall/CRC. ISBN 9781138331457. Available from: https://plotly-r.com Back to top Previous Next Posted August 18, 2025. Download PDF Supplementary Material Data/Code Email Thank you for your interest in spreading the word about bioRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following Quantifying structural variants in chromosomes using landmark-based disparity Message Subject (Your Name) has forwarded a page to you from bioRxiv Message Body (Your Name) thought you would like to see this page from the bioRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share Quantifying structural variants in chromosomes using landmark-based disparity Ashwini V. Mohan , Anjali Goswami , Jeffrey W. Streicher bioRxiv 2024.10.30.620815; doi: https://doi.org/10.1101/2024.10.30.620815 Share This Article: Copy Citation Tools Quantifying structural variants in chromosomes using landmark-based disparity Ashwini V. Mohan , Anjali Goswami , Jeffrey W. Streicher bioRxiv 2024.10.30.620815; doi: https://doi.org/10.1101/2024.10.30.620815 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Genomics Subject Areas All Articles Animal Behavior and Cognition (7967) Biochemistry (18614) Bioengineering (14761) Bioinformatics (44129) Biophysics (22450) Cancer Biology (19580) Cell Biology (26733) Clinical Trials (138) Developmental Biology (13898) Ecology (20880) Epidemiology (2067) Evolutionary Biology (25312) Genetics (16096) Genomics (23390) Immunology (18595) Microbiology (42208) Molecular Biology (17938) Neuroscience (92853) Paleontology (693) Pathology (2969) Pharmacology and Toxicology (5063) Physiology (8060) Plant Biology (15899) Scientific Communication and Education (2091) Synthetic Biology (4538) Systems Biology (10182) Zoology (2375) window.__CF$cv$params={r:'a36d9b2db9f21d20',t:'MTc4ODY5OTA5Nw==',u:'01a076c641bb7911b6dde16d7d82cda3',ut:'ehW0Al5GrNLj1GWJK.Y7bXRzeuobRVwhsFq_Vzzz_MU-1788699099-1.2.1.1-EMTGL8Em3R_siw9zTcfXaiPONy0P9mBM_FclpOQd.GE7exkmLM2qeIXNJjQh9Sns8Waq8JefUbtYMk8vTduSPsxTusESdHu3DRA4ihmdgPg',i:60};(function(){if(!document.body)return;var s=document.createElement('script');s.src='/cdn-cgi/challenge-platform/scripts/precursor/main.js';document.head.appendChild(s);})();

Text is read by the "Ask this paper" AI Q&A widget below. Extraction quality varies by source — PMC NXML preserves structure cleanly, OA-HTML may include some navigation residue, and OA-PDF can have broken hyphenation. The publisher copy (via DOI) is the canonical version.

My notes (saved in your browser only)

⚙ Ask this paper AI returns verbatim quotes from the full text · source: preprint-html ⓘ

Answers must be backed by verbatim quotes from this paper's full text. Hallucinated quotes are dropped automatically; if no verbatim passage answers the question, we say so. How this works

Citation neighborhood (no data yet)

We don't have any in-corpus citations linked to this paper yet. This is a recent paper (2024) — citers typically take a year or two to land, and the OpenAlex reference graph may still be filling in.

Source provenance

europepmc
last seen: 2026-05-19T01:45:01.086888+00:00
unpaywall
last seen: 2026-09-29T06:34:56.587159+00:00