Full text
39,265 characters
· extracted from
preprint-html
· click to expand
Inferring the history of gene copy number evolution | bioRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-M677548'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search New Results Inferring the history of gene copy number evolution View ORCID Profile Moritz Otto , View ORCID Profile Thomas Wiehe doi: https://doi.org/10.1101/2025.08.21.671444 Moritz Otto a University of Cologne, Institute for Genetics , Zülpicher Str. 47a, Cologne, 50674, Germany Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Moritz Otto For correspondence: moritz.otto{at}uni-koeln.de Thomas Wiehe a University of Cologne, Institute for Genetics , Zülpicher Str. 47a, Cologne, 50674, Germany Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Thomas Wiehe Abstract Full Text Info/History Metrics Supplementary material Data/Code Preview PDF Abstract Gene duplication plays a crucial role in the adaptive evolution and diversification of organisms by creating extra copies of genes that can evolve new functions while preserving the original. Duplicated genes can become fixed in populations or appear as copy number variants. However, inferring and dating these duplication events from present-day data is challenging, as gene copy count distributions could result from either a few ancient duplication events or many recent ones. Sequence based phylogenetic reconstruction, an often seen practice, does not include the history of individuals and hence may result in inconsistencies, which may lead to misinterpretations. Here, we introduce a novel model for inferring gene copy number evolution, which describes gene duplication and their evolution over time through a random walk on a coalescent duplication network. This approach is solely based on copy number counts and hence independent of the inconsistencies of sequence based inferences. Backward in time we implement structured coalescent simulations, where we re-interprete ‘structure’ as ‘genealogical distance’ based on copy number counts. We apply this model to the NB-ARC domain counts of NLR genes in A. thaliana to infer the number and times of duplication events that have led to the present day copy number distribution. Introduction Multicopy gene families play a fundamental role in the evolution and adaptation of organisms by providing genetic redundancy, which allows functional diversification. Gene duplication events give rise to additional copies of genes, which can either be retained as redundant sequences, undergo subfunctionalization by partitioning ancestral functions, or acquire entirely new roles through neofunctionalization [ 1 , 2 , 3 ]. The adaptive potential of gene duplication has been demonstrated across a wide range of taxa, including mammals, fish, and various plant species, and across a wide range of gene functions, including immune and stress response or digestion [ 4 , 5 , 6 , 7 , 8 , 9 , 10 ]. Recent studies have explored the evolution of gene copy numbers in plant immune receptor families, particularly nucleotide-binding leucine-rich repeat (NLR) genes, which play a crucial role in pathogen defense. Relying on the long-read pan-NLRome data of A. thaliana , conserved NB-ARC domains were used as an indirect measure of NLR gene copy number [ 11 , 12 ]. However, given the present-day copy number distribution, it remains unclear whether the observed gene copies originated from a few ancient duplication events or multiple recent ones (see Figure 1 ). Accurate estimation of the number and timing of events is required, for instance, to test if a recent ecological shift is synchronized with gene copy number alteration. Besides standard sequence based phylogenetic inferences several new tools were developed to identify and align gene copies [ 13 , 14 , 15 ]. Nevertheless, these sequence-based analyses of multicopy gene families require a well-defined theoretical null model to accurately interpret patterns of variation and divergence. Without such a framework, it becomes difficult to distinguish between signatures of neutral evolution, selection, or non-adaptive processes such as gene conversion. For instance, the presence of high sequence similarity among gene copies could be misinterpreted as evidence of recent duplication, when in fact it may result from homogenization via concerted evolution. Thus, theoretical modeling is essential to avoid biased or misleading conclusions in the study of complex gene families. Download figure Open in new tab Figure 1. Example of three different duplication histories resulting in the same copy number distribution. Left shows four recent, independent duplications (blue dots and lines), center shows three duplications and right shows one ancient and one recent duplication. The different duplication events shape the tree topology, even though the coalescent events (red) are identical. Here, we introduce a novel approach to model the history of multicopy gene families. Our approach starts from present day copy number counts and is twofold: First, we aim to estimate the most likely duplication rate that led to this present day configuration. This approach builds upon the work of [ 16 ], who utilized phase-type distributions in a random walk framework to estimate the joint site frequency spectrum in complex migration histories. We adapt this concept and introduce the Coalescent-Duplication Network (CDN) , see Figure 2 . Starting with one single individual with one copy, the lineage may either split into two separate individual lineages (offspring event, red arrows), or into two linked copies within the same individual (duplication event, blue arrows). Following this process a too high duplication rate may lead to a copy number configuration that is not compatible with the present day sample anymore. Vice versa, a random walk with a too low duplication rate consists of too many individual lineage splits and is also not compatible with the present day distribution (gray arrows). Conditioning on reaching the desired copy number in the sample we calculate a maximum likelihood estimator of the duplication rate. Download figure Open in new tab Figure 2. Example of a random walk on the Coalescence-Duplication-Network (CDN). Blue lines indicate duplication events, red lines split events. Duplication increases the copy number (dashed horizontal lines) by one, split events increase the number of individuals by one and may increase the number of copy numbers by one or more. Grey dashed lines indicate combined probabilities to reach a state, from which α end is no longer possible and the random walk gets out-of-bounce. The path in bold lines corresponds to the central genealogy in Figure 1 In a second step, using this estimator, we simulate the backward-in-time process of this sample back to its initial single individual with a single copy in a coalescent framework. More precisely, we re-interprete the term of ‘structure’ (which is often used to describe geographic distance between individuals) as ‘genealogical distance’ in the sense of copy number differences, as already established in [ 17 ]. Therefore, two individuals may only be offspring of the same ancestor, if their copy number coincides. This intuitive assumption is currently ignored in sequenced based phylogenetic reconstruction and hence might be error-prone in inferring the evolutionary history (see Figure 3 A , example of A. thaliana ). Especially when considering effects such as selection or gene conversion acting on sequences, those inconsistencies of gene trees and individual trees might be misleading in downstream analyses. In contrast, our approach starts from simply counting the copy numbers of a given gene family in different individuals and traces back the most likely path to their ancestor. There are two ways in which genealogical lineages can merge: by coalescent events (red circles in Figure 3 B ), where lineages of different individuals meet, and by ‘de-duplications’ (blue circles), where copy count is reduced by one when looking backward in time. Download figure Open in new tab Figure 3. A Maximum likelyhood neighbour joining tree of B3 NB-ARC domains in three accessions of A. thaliana , with the single copy of A. lyrata as outgroup (dashed line). Green shaded lines correspond to the lineages of the four copies of Alm-0 accession. Intuitively, the common ancestor of two copies of the same individual has to be a duplication event. However, since duplication events affect the same lineage, the subtrees of a duplication event have to include the same individuals. Green marks mark the events, where this is true, whereas the lightning highlights the inconsistency of this phylogenetic tree. B Example of a structured coalescent simulation of the same copy number sample (3,4,5). By definition, this tree does not contain any inconsistencies. Blue dots indicate duplication events, red dots coalescent events. Note, that a coalescent event may merge more than two lineages if the individuals have more than one copy. We implemented this structured coalescent framework to be compatible with the msprime and tskit packages [ 18 ]. Therefore, we are also able to superimpose the mutation process on the generated trees and to examine how duplication rates influence tree topology and, consequently, the estimation of key population genetic statistics such as Tajima’s D [ 19 ]. Finally, we applied our approach to a subset of 15 NLR-gene families of 10 A . thaliana accessions, with two families analyzed in more detail. We followed the workflow established by [ 11 ] to identify and align these gene copies. Using the CDN, we estimated the most likely duplication rate and, following the structured coalescent, we timed the duplication events and computed expected Tajima’s D under our model [ 19 ]. Our findings indicate that the most likely duplication history does not necessarily coincide with the most parsimonious one and that old duplications may lead to a biased Tajima’s D . Materials and Methods NB-ARC domains in A. thaliana We followed the workflow of [ 11 ] to analyze the pan-NLRome 2 [ 12 ]. Using the NB-ARC domains as proxies, we blast the annotated reference genes from the reference accession Col-0 (TAIR10.1 assembly 3 ) against the pan-NLRome dataset. We were able to reproduce the copy number counts from [ 11 ] and align the corresponding gene families. For B3 , which is assumed to be a young gene family, as it has only one copy in A. lyrata , we also built neighbour joining trees ( Figure 3 A and Figure 4 ). To test our model, we subsampled 10 accessions that showed a trustworthy alignment and represents the geographic distribution and the overall copy number distribution of two gene families ( B3 and RPS5 ), see Table S1 and Figure S1 . Furthermore, we used the copy number counts of further NLR-genes from [ 11 ], scaled down to 10 individuals, as a test set to evaluate the duplication rate estimation procedure, see Table 1 and Figure 5 . View this table: View inline View popup Download powerpoint Table 1: Duplication rate estimation for different distribution examples. Download figure Open in new tab Figure 4. A UPGMA tree of B3 sequences obtained from Moj-0, Gra-0 and A. lyrata as outgroup. This shows a tree, which is consistent with the expected history of the two individuals, and hence we can place duplication events (blue dots) and coalescent events (red dots). B Radial neighbour joining tree for the sequences of B3 , taken from 10 accessions, see Table S1 and Figure S1 . Each leaf corresponds to a copy of B3 . Grey bold lines show the genealogy of the four copies of Alm-0 . Their ancestor node is assumed to be a duplication event. Lightnings mark inconsistencies of these duplication events, if the subtree leafs do not coincide (here in all three duplication events). Download figure Open in new tab Figure 5. Correlation of duplication estimators d path , d negbinom (scaled with N ) with mean and standard deviation of copy number distributions from NLR genes, see Table 1 . Grey circle shows estimates for B3 and RPS5 . Note, that we do not aim to reproduce the data analysis done by [ 11 ] or [ 12 ] and do not proceed with an extensive sequence analysis but focus on the pure copy number counts as input data for our model. Duplication rate estimation We encode the copy number distribution of a gene by a vector α = ( α i ) i , where α i determines the number of individuals that have i copies. As an example, a sample of three individuals with (2,2,3) copies (as in Figure 1 ) is encoded as α = (0, 2, 1). Hence, thinking forward in time, we start with one single gene lineage in one individual, α = (1, 0, …). In each time step, a gene may duplicate with probability d and a lineage may split with probability 1 /N , where N denotes the population size. Let 1 [ i ] be a vector with zeroes everywhere, except for a single 1 at position i . Then This results in a transient random walk in an infinitely large state space. However, we aim to calculate the probability that this random walk hits a certain state, i.e. our present day configuration. Since we only allow an increase of copy number or number of individuals, the random walk may reach a state from which it is impossible to end up in the desired state, α end . We summarize all these nodes as a cemetery state, which we simply call α out , since one steps out of the network without return. An example of α end = (0, 2, 1) is shown in Figure 2 . Hence, the random walk either ends in α out or α end . Using the Markov property, we denote the probability to start from α and reach α end before hitting p out as p α , where and . Therefore, we can determine the probability to start in α start = (1, 0, …) and reach α end by solving the linear equation system, where for all α we find Therefore, given any α end we can construct the CDN and apply the transition probabilities according to d and N and solve the linear equation system for to determine the probability to generate the given copy number distribution, without hitting α out . This probability depends on d and N , and therefore we can use a maximum likelihood estimator to determine the duplication rate d that generates the copy number distribution. Thinking differently, one may model duplication events similar to mutation events on a given coalescent tree, as done in [ 20 ]. Hence, one can think of the duplication process as a pure Yule-birth process. Therefore, the number of duplication events that have happened since the MRCA follows a negative binomial distribution which simplifies to a geometric distribution when assuming the ancestral genotype to be a single copy gene. More precisely, given a fixed time T , the first duplication event occurs at rate d , the second at rate 2 d and so on. We define X 1 , X 2 , … as independent random variables with X i ~ Exp( i · d ) as the times between event i − 1 and i and as the cumulative time of n events. Then, the number of events in the interval T , which we define as Z T , follows and hence Z T ~ Geo( e −dT ). Therefore, with T being the time until the most recent common ancestor, we find Concluding, we find two different approaches to get a maximum likelihood estimator of the duplication rate, d path and d negbinom . Structured Coalescent When simulating backward in time coalescence of multicopy gene families it is crucial to ensure that two individuals only coalesce, if both have the same number of gene copies. If two individuals coalesce, all their copies are paired, which leads to a merging of multiple lineages. Therefore, the process can be seen as a simulation of a backward in time random walk in the graph of Figure 2 , using the previously estimated optimal duplication rate d . More precisely, starting with a copy number configuration given by α end we include the waiting time spent in each state until we reach the most recent common ancestor of all gene copies, α start = (1, 0, 0, …). We use the structure of tskit used in msprime [ 18 ] to implement a structured coalescent simulation in python. We trace the copy number distribution of a given sample and generate a random walk in the graph backward in time. Starting with α = (0, 1, 4, 3, 2) for B3 and (0, 0, 5, 5) for RPS5 , we ran 5.000 structured coalescent simulations for both copy number distributions. We count the number of duplication events in each random walk as well as the times of the duplication events measured relative to the TMRCA of individuals, to date the most likely duplication events. Furthermore, we uniformly place mutations on the tree branches according to their length with constant rate µ , as shown in Figure 3 , to determine the effect of the duplication process on the tree topology. Results and Discussion With the workflow established in [ 11 ] we were able to confirm their copy number counts for selected NLR genes. Two gene families, B3 and RPS5 , showed most reliable results. Furthermore, they both show same mean copy number of 4, but differ in their variance. Hence, their copy number counts were used as examples for further simulations. First, we focus on the results of the standard sequence based methods of phylogenetic inference. In Figure 4 A we find a well-fitting UPGMA gene tree for B3 in the accessions of Moj-0 and Gra-0. Gra-0 experienced one recent duplication event (blue dot), before coalescing with Moj-0 , and indeed, all the coalescent events of the three copies occurred at the same time (red dots). However, this consistency vanishes, when building the alignment and the neighbour-joining tree of all ten accessions. Considering the same example of Alm-0 as in Figure 3 , we find inconsistencies in this NJ tree. More precisely, if one traces back the history of two copies of the same individual, their common ancestor has to be a duplication event. Moreover, duplication events affect the same lineage and therefore the subtrees of the duplication event have to include the same individuals. Intuitively speaking, if all individuals of a sample have two copies and this duplication event was ancient, one would find two subtrees with all the individuals as leafs. However, in the example of the four copies of B3 in Alm-0 , none of the three duplication events satisfies this condition (marked as lightnings in Figure 4 B ). More strikingly, each of the four gene copies of Alm-0 is paired with a copy from a different accession. The chosen gene families B3 and RPS5 both have a mean copy number of ~ 4, but differ in their variance. For both distributions we calculated in the CDN for various d path N in (0,2). The widespread distribution of B3 led to an estimation of d path N ~ 0.469, around twice as high as the one of RPS5 ( Figure 6 ). This difference is biologically plausible, since a low-variance distribution may result from only a few older duplication events that have since stabilized, whereas a broader (high-variance) distribution suggests a greater number of more recent duplications contributing to ongoing diversification. Indeed, we obtain this result for the chosen copy number distributions of NLR genes of [ 11 ] (see Figure 5 and Table 1 ). When comparing d path with d negbinom we observe that both correlate with the mean and the standard deviation of the given copy number distribution. However, d path is more sensitive towards the variance, whereas d negbinom aligns better with the mean. Furthermore, d path tends to estimate a lower duplication rate and has a larger range of estimates (0.1 - 0.6) compared with d negbinom (0.4 - 0.6). This highlights the strength of the new estimator, as it includes the information of the variance of the distribution. Conceptually, a high mean copy number is not neccesarily an indicator of a high duplication rate, if the variance of the distribution is low. This distribution might rather be a result of a few, ancient duplications. The d negbinom does not cover these differences, as indicated in the example of B3 and RPS5 , which have the same mean value and result in the same d negbinom , despite their different variance (grey circle in Figure 5 ). Download figure Open in new tab Figure 6. Estimation of the duplication rate for two examples ( B3 and RPS5 ) of gene copy number distribution. Top shows copy number distribution. Second row shows for duplication rates dN ranging from 0 to 2, with vertical red line indicating maximum likelihood. Third row shows frequency table of duplication events for 5,000 backward in time structured coalescent simulations. Yellow bar highlights the most frequent one. Fourth row shows Tajima’s D given the number of duplications. Bottom boxplots show the time and the subtree size of the duplication events. Ancient duplication events (i.e. that occurred before the TMRCA) are highlighted with gray circle and the corresponding numbers. Given these duplication rate estimators, we aim to infer the number of duplication events leading to this distribution and to date those events. Therefore, we conducted 5,000 structured coalescent simulations for parameter sets α = (0, 1, 4, 3, 2) with dN = 0.469 ( B3 ) and α = (0, 0, 5, 5) with dN = 0.224 ( RPS5 ). For B3 , a total of 1,023 simulations (20%) resulted in exactly nine duplications, representing the most frequent and thus most likely scenario (see Figure 6 ). In the case of RPS5 , 1,355 simulations (27%) featured six such events. Recap that these results were obtained solely by the copy number counts, independent from sequence data. The results suggest that the observed copy number distributions are unlikely to arise from either a few ancient duplications or an accumulation of very recent ones alone. When focusing on the temporal distribution of the most likely number of events ( Figure 6 , yellow boxes), the inferred timings continuously span up to the most recent common ancestor (indicated by the red dashed line), which is consistent with a model that assumes constant and ongoing duplication pressure. For the subset of 1,023 B3 simulations yielding nine duplications, only 96 trajectories included an event predating the most recent common ancestor of the sample. The remaining runs indicate that the duplications happened relatively recently. This strongly supports a scenario in which the observed copy number variation arose from approximately nine independent and recent duplication events. In contrast, for RPS5 , a minimum of three duplications is required to reproduce the empirical copy number distribution. With dN = 0.24, only 161 of the 5,000 simulations met this minimum. The most frequently observed scenario comprises six duplications, where 431 out of 1,355 simulations involved at least one duplication predating the most recent common ancestor. This outcome is notable, given that the observed bimodal distribution – three and four copies among equal proportions of the sample – would intuitively suggest two ancestral duplications followed by a more recent one affecting only part of the population. However, under the assumption of a constant and uniformly acting duplication rate, our model instead favors a scenario involving six independent events distributed over time. Note, that the structured coalescent simulations strongly depend on the correct estimation of d , since the transition probabilities of the backward in time random walk on the graph of Figure 2 depend on the duplication rate. Hence, with a too high duplication rate one expects several duplication events to occur recently, i.e. starting at α end the random walk always follows the next blue path in Figure 2 . But then, no further ‘deduplication’ events are possible, which means that – even though we assume a constant duplication force – there will be no duplication event on the long branches in the coalescent tree. Vice versa, a too low duplication rate assumes that all duplication events have happened before the most recent common ancestor of the individuals, leading to long branches before the MRCA. Therefore, both a too high and a too low d estimate shape the topology of the coalescent tree and hence affect population statistics. With maximum likelihood estimated dN , we placed 500 mutations on the outcoming tree for each of the 5,000 simulations and calculated Tajima’s D (see Figure 6 ). We observe a small bias towards a negative D in B3 , which appears to be negligible. Therefore, with correct estimator and constant duplication rate, the tree topology appears to be similar to that of the neutral coalescent. Note, that this simulation scheme takes into account both coalescent of individuals and de-duplications of copies and results in a genealogical tree of gene copies. Caveats and Outlook Here we presented a novel approach to estimate the duplication rate of multicopy gene families based on their copy number counts. We implement the evolution as a random walk on the Coalescent Duplication Network , where the transitions from one state to another correspond to either duplication or coalescent events. However, we had to limit our estimation procedure on intermediate copy number counts and samples of size n = 10, as the graph network increases exponentially in size with additional copies and additional sampled individuals. Solving this scaling problem and subsequent analysis of the CDN needs to be addressed in the future. With larger sample sizes and larger copy numbers we also aim to include larger sequence analyses, including information of the population structure. Implementing selective pressure on gene copy number and, in the case of immune genes, diversifying selection acting on gene function as done in [ 21 , 22 ] may represent another prospective goal. Concluding, our results show the necessity of a duplication model when analyzing multicopy gene families and highlight the limitations of standard tree inference approaches and summary statistics when there is no clear distinction between paralogs and orthologs. Data Availability Statement All data and custom codes used for simulations and analysis can be found at https://github.com/Moritz-Otto/motto-randomwalk Funding This work has been funded by a grant from the German Research Foundation (DFG TRR341, subproject B5), to Laura Rose, HHU Düsseldorf, and to TW. Conflict of interest The authors declare no conflict of interest. Supplementary Material View this table: View inline View popup Download powerpoint Table S1: Subsample of 10 accessions that showed a trustworthy alignment and represents the geographic distribution and the overall copy number distribution of the two gene families. Download figure Open in new tab Figure S1: Subsample of accessions selected from [ 12 ] Funder Information Declared Deutsche Forschungsgemeinschaft, https://ror.org/018mejw64 , TRR 341 Footnotes https://github.com/Moritz-Otto/motto-randomwalk 1 http://ftp.tuebingen.mpg.de/ebio/alkeller/pan_NLRome/ 2 https://www.ncbi.nlm.nih.gov/datasets/genome/GCF_000001735.4/ References [1]. ↵ H. Innan , F. Kondrashov , The evolution of gene duplications: classifying and distinguishing between models , Nature Reviews Genetics 11 ( 2 ) ( 2010 ) 97 – 108 . doi: 10.1038/nrg2689 . URL http://dx.doi.org/10.1038/nrg2689 OpenUrl CrossRef PubMed Web of Science [2]. ↵ S. Magadum , U. Banerjee , P. Murugan , D. Gangapur , R. Ravikesavan , Gene duplication as a major force in evolution , Journal of Genetics 92 ( 1 ) ( 2013 ) 155 – 161 . doi: 10.1007/s12041-013-0212-8 . URL http://dx.doi.org/10.1007/s12041-013-0212-8 OpenUrl CrossRef PubMed Web of Science [3]. ↵ S. Ohno , Evolution by Gene Duplication , Springer Berlin Heidelberg , 1970 . doi: 10.1007/978-3-642-86659-3 . URL http://dx.doi.org/10.1007/978-3-642-86659-3 OpenUrl CrossRef [4]. ↵ Y. Schäfer , K. Palitzsch , M. Leptin , A. R. Whiteley , T. Wiehe , J. Suurväli , Copy number variation and population-specific immune genes in the model vertebrate zebrafish , eLife 13 ( Jun . 2024 ). doi: 10.7554/elife.98058 . URL http://dx.doi.org/10.7554/elife.98058 OpenUrl CrossRef [5]. ↵ K. Wei , S. Sharifova , X. Zhao , N. Sinha , H. Nakayama , A. Tellier , G. A. Silva-Arias , Evolution of gene networks underlying adaptation to drought stress in the wild tomato solanum chilense , Molecular Ecology 33 ( 21 ) ( Oct . 2024 ). doi: 10.1111/mec.17536 . URL http://dx.doi.org/10.1111/mec.17536 OpenUrl CrossRef [6]. ↵ J. K. von Dahlen , K. Schulz , J. Nicolai , L. E. Rose , Global expression patterns of r-genes in tomato and potato , Frontiers in Plant Science 14 ( Oct . 2023 ). doi: 10.3389/fpls.2023.1216795 . URL http://dx.doi.org/10.3389/fpls.2023.1216795 OpenUrl CrossRef [7]. ↵ M. Bouzid , F. He , G. Schmitz , R. E. Häusler , A. P. M. Weber , T. Mettler-Altmann , J. De Meaux , Arabidopsis species deploy distinct strategies to cope with drought stress , Annals of Botany 124 ( 1 ) ( 2019 ) 27 – 40 . doi: 10.1093/aob/mcy237 . URL http://dx.doi.org/10.1093/aob/mcy237 OpenUrl CrossRef [8]. ↵ G. A. Silva-Arias , E. Gagnon , S. Hembrom , A. Fastner , M. R. Khan , R. Stam , A. Tellier , Patterns of presence-absence variation of nlrs across populations of solanum chilense are clade-dependent and mainly shaped by past demographic history , New Phytologist 245 ( 4 ) ( 2024 ) 1718 – 1732 . doi: 10.1111/nph.20293 . URL http://dx.doi.org/10.1111/nph.20293 OpenUrl CrossRef PubMed [9]. ↵ M. Hanikenne , I. N. Talke , M. J. Haydon , C. Lanz , A. Nolte , P. Motte , J. Kroymann , D. Weigel , U. Krämer , Evolution of metal hyperaccumulation required cis-regulatory changes and triplication of hma4 , Nature 453 ( 7193 ) ( 2008 ) 391 – 395 . doi: 10.1038/nature06877 . URL http://dx.doi.org/10.1038/nature06877 OpenUrl CrossRef PubMed Web of Science [10]. ↵ G. H. Perry , N. J. Dominy , K. G. Claw , A. S. Lee , H. Fiegler , R. Redon , J. Werner , F. A. Villanea , J. L. Mountain , R. Misra , N. P. Carter , C. Lee , A. C. Stone , Diet and the evolution of human amylase gene copy number variation , Nature Genetics 39 ( 10 ) ( 2007 ) 1256 – 1260 . doi: 10.1038/ng2123 . URL http://dx.doi.org/10.1038/ng2123 OpenUrl CrossRef PubMed Web of Science [11]. ↵ R. R. Lee , E. Chae , Variation patterns of nlr clusters in arabidopsis thaliana genomes , Plant Communications 1 ( 4 ) ( 2020 ) 100089 . doi: 10.1016/j.xplc.2020.100089 . URL http://dx.doi.org/10.1016/j.xplc.2020.100089 OpenUrl CrossRef PubMed [12]. ↵ A.-L. Van de Weyer , F. Monteiro , O. J. Furzer , M. T. Nishimura , V. Cevik , K. Witek , J. D. Jones , J. L. Dangl , D. Weigel , F. Bemm , A species-wide inventory of nlr genes and alleles in arabidopsis thaliana , Cell 178 ( 5 ) ( 2019 ) 1260 – 1272.e14 . doi: 10.1016/j.cell.2019.07.038 . URL http://dx.doi.org/10.1016/j.cell.2019.07.038 OpenUrl CrossRef PubMed [13]. ↵ P. Karunarathne , Q. Zhou , K. Schliep , P. Milesi , A comprehensive framework for detecting copy number variants from single nucleotide polymorphism data: ‘rcnv’, a versatile r package for paralogue and cnv detection , Molecular Ecology Resources 23 ( 8 ) ( 2023 ) 1772 – 1789 . doi: 10.1111/1755-0998.13843 . URL http://dx.doi.org/10.1111/1755-0998.13843 OpenUrl CrossRef PubMed [14]. ↵ B. Tjeng , M. Arimond , H. B. Grindeland , A. D. Libera , A. Fulgione , Paramask, a new method to identify multicopy genomic regions, corrects major biases in whole-genome sequencing data , research Square, preprint server ( Jul . 2024 ). doi: 10.21203/rs.3.rs-4660628/v1 . URL http://dx.doi.org/10.21203/rs.3.rs-4660628/v1 OpenUrl CrossRef [15]. ↵ F.-D. Pajuste , M. Remm , Genetocn: an alignment-free method for gene copy number estimation directly from next-generation sequencing reads , Scientific Reports 13 ( 1 ) ( Oct . 2023 ). doi: 10.1038/s41598-023-44636-z . URL http://dx.doi.org/10.1038/s41598-023-44636-z OpenUrl CrossRef PubMed [16]. ↵ T. Røikjer , A. Hobolth , K. Munch , Graph-based algorithms for phase-type distributions , Statistics and Computing 32 ( 6 ) ( Nov . 2022 ). doi: 10.1007/s11222-022-10174-3 . URL http://dx.doi.org/10.1007/s11222-022-10174-3 OpenUrl CrossRef [17]. ↵ M. Otto , T. Wiehe , The structured coalescent in the context of gene copy number variation , Theoretical Population Biology 154 ( 2023 ) 67 – 78 . doi: 10.1016/j.tpb.2023.08.001 . URL http://dx.doi.org/10.1016/j.tpb.2023.08.001 OpenUrl CrossRef PubMed [18]. ↵ F. Baumdicker , G. Bisschop , D. Goldstein , G. Gower , A. P. Ragsdale , G. Tsambos , S. Zhu , B. Eldon , E. C. Ellerman , J. G. Galloway , A. L. Gladstein , G. Gorjanc , B. Guo , B. Jeffery , W. W. Kretzschumar , K. Lohse , M. Matschiner , D. Nelson , N. S. Pope , C. D. Quinto-Cortés , M. F. Rodrigues , K. Saunack , T. Sellinger , K. Thornton , H. van Kemenade , A. W. Wohns , Y. Wong , S. Gravel , A. D. Kern , J. Koskela , P. L. Ralph , J. Kelleher , Efficient ancestry and mutation simulation with msprime 1.0 , Genetics 220 ( 3 ) ( Dec . 2021 ). doi: 10.1093/genetics/iyab229 . URL http://dx.doi.org/10.1093/genetics/iyab229 OpenUrl CrossRef [19]. ↵ F. Tajima , Statistical method for testing the neutral mutation hypothesis by dna polymorphism ., Genetics 123 ( 3 ) ( 1989 ) 585 – 595 . doi: 10.1093/genetics/123.3.585 . URL http://dx.doi.org/10.1093/genetics/123.3.585 OpenUrl Abstract / FREE Full Text [20]. ↵ F. Freund , J. Wirtz , Y. Zheng , Y. Schäfer , T. Wiehe , Muller’s ratchet and gene duplication , Theoretical Population Biology 164 ( 2025 ) 12 – 22 . doi: 10.1016/j.tpb.2025.04.002 . URL http://dx.doi.org/10.1016/j.tpb.2025.04.002 OpenUrl CrossRef PubMed [21]. ↵ M. Otto , Y. Zheng , T. Wiehe , Recombination, selection, and the evolution of tandem gene arrays , Genetics 221 ( 3 ) ( Apr . 2022 ). doi: 10.1093/genetics/iyac052 . URL https://doi.org/10.1093/genetics/iyac052 OpenUrl CrossRef [22]. ↵ M. Otto , Y. Zheng , P. Grablowitz , T. Wiehe , Detecting adaptive changes in gene copy number distribution accompanying the human out-of-Africa expansion , Human Genome Variation 11 ( 1 ) ( Sep . 2024 ). doi: 10.1038/s41439-024-00293-w . URL http://dx.doi.org/10.1038/s41439-024-00293-w OpenUrl CrossRef View the discussion thread. Back to top Previous Next Posted August 22, 2025. Download PDF Supplementary Material Data/Code Email Thank you for your interest in spreading the word about bioRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following Inferring the history of gene copy number evolution Message Subject (Your Name) has forwarded a page to you from bioRxiv Message Body (Your Name) thought you would like to see this page from the bioRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share Inferring the history of gene copy number evolution Moritz Otto , Thomas Wiehe bioRxiv 2025.08.21.671444; doi: https://doi.org/10.1101/2025.08.21.671444 Share This Article: Copy Citation Tools Inferring the history of gene copy number evolution Moritz Otto , Thomas Wiehe bioRxiv 2025.08.21.671444; doi: https://doi.org/10.1101/2025.08.21.671444 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Evolutionary Biology Subject Areas All Articles Animal Behavior and Cognition (7629) Biochemistry (17660) Bioengineering (13881) Bioinformatics (41913) Biophysics (21436) Cancer Biology (18578) Cell Biology (25482) Clinical Trials (138) Developmental Biology (13372) Ecology (19889) Epidemiology (2067) Evolutionary Biology (24302) Genetics (15599) Genomics (22483) Immunology (17728) Microbiology (40365) Molecular Biology (17163) Neuroscience (88540) Paleontology (666) Pathology (2830) Pharmacology and Toxicology (4821) Physiology (7637) Plant Biology (15136) Scientific Communication and Education (2045) Synthetic Biology (4290) Systems Biology (9818) Zoology (2269)
Text is read by the "Ask this paper" AI Q&A widget below.
Extraction quality varies by source — PMC NXML preserves structure
cleanly, OA-HTML may include some navigation residue, and OA-PDF can
have broken hyphenation. The publisher copy
(via DOI)
is the canonical version.