Full text
62,069 characters
· extracted from
preprint-html
· click to expand
Resolving the source of branch length variation in the Y chromosome phylogeny | bioRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-M677548'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search New Results Resolving the source of branch length variation in the Y chromosome phylogeny Yaniv Swiel , Janet Kelso , Stéphane Peyrégne doi: https://doi.org/10.1101/2024.07.05.602100 Yaniv Swiel 1 Department of Evolutionary Genetics, Max Planck Institute for Evolutionary Anthropology , Leipzig, Germany Find this author on Google Scholar Find this author on PubMed Search for this author on this site For correspondence: yaniv_swiel{at}eva.mpg.de Janet Kelso 1 Department of Evolutionary Genetics, Max Planck Institute for Evolutionary Anthropology , Leipzig, Germany Find this author on Google Scholar Find this author on PubMed Search for this author on this site Stéphane Peyrégne 1 Department of Evolutionary Genetics, Max Planck Institute for Evolutionary Anthropology , Leipzig, Germany Find this author on Google Scholar Find this author on PubMed Search for this author on this site Abstract Full Text Info/History Metrics Preview PDF Abstract Genetic variation in the non-recombining part of the human Y chromosome has provided important insight into the paternal history of human populations. However, a significant and yet unexplained branch length variation of Y chromosome lineages has been observed, notably amongst those that are highly diverged from the human reference Y chromosome. Understanding the origin of this variation, which has previously been attributed to changes in generation time, mutation rate, or efficacy of selection, is important for accurately reconstructing human evolutionary and demographic history. Here, we analyze Y chromosomes from present-day and ancient modern humans, as well as Neandertals, and show that branch length variation amongst human Y chromosomes cannot solely be explained by differences in demographic or biological processes. Instead, reference bias results in mutations being missed on Y chromosomes that are highly diverged from the reference used for alignment. We show that masking fast-evolving, highly divergent regions of the human Y chromosome mitigates the effect of this bias and enables more accurate determination of branch lengths in the Y chromosome phylogeny. Finally, we show that this approach allows us to estimate the age of ancient samples from Y chromosome sequence data and provide updated TMRCA estimates using the portion of the Y chromosome where the effect of reference bias is minimised. Introduction Human Y chromosome lineages that are highly diverged from the reference Y chromosome appear to have accumulated fewer mutations than lineages that are more closely related to the reference, a non-African lineage of the R1b haplogroup. In particular, some human Y lineages in Africa, such as A00 [ 1 ], appear to have accumulated fewer mutations than other Y lineages since they last shared a common ancestor. In their phylogenetic analysis of a diverse selection of human Y chromosomes for worldwide populations, Wei et al. [ 2 ] observed a shortened branch for the A3 haplogroup represented by a single individual in their study. They attributed this shortened branch to an under-calling of variants due to the low-coverage sequencing (5x) of the A3 individual. Scozzari et al. [ 3 ] analysed Y chromosomes representing major haplogroups and noted shortened branches for the A haplogroups relative to the rest of the tree. They were not able to determine the cause of the shortened branches, but suggested that a reduction in the male effective population size coinciding with the out-of-Africa bottleneck allowed mildly deleterious mutations to accumulate on non-African Y chromosomes. Hallast et al. [ 4 ] tested whether the tissue from which the DNA is extracted has an effect on the branch length heterogeneity but could not detect any statistically significant differences between tissues. They therefore proposed that a change in the mean paternal age (resulting in the increase or decrease of the effective mutation rate) in a particular geographical region could result in associated haplogroups displaying similarly shortened or lengthened branches. Barbieri et al. [ 5 ] suggested a number of technical biases that could result in the undercalling of variants in A and B Y chromosome lineages. They proposed that capture bias could be introduced when using a Y chromosome capture array based on the reference Y chromosome and that the underrepresentation of A and B haplogroups in reference datasets could affect genotype calling and/or imputation. They also suggested that, in addition to these biases, population-based differences in average paternal age could contribute to the branch length variation across the Y chromosome phylogeny through the modification of the mutation rate. Although these studies relied on capture sequencing data (and low-coverage shotgun sequencing data in the case of Wei et al. [ 2 ]), Naidoo et al. [ 6 ] also noticed shortened branches among A haplogroups using high-coverage shotgun sequencing data, suggesting that these differences do not originate from capture bias. Understanding the cause of branch length variation in the Y chromosome phy-logeny is important to accurately reconstruct human evolutionary history. The A00 Y chromosome lineage has been used to estimate the time to the most recent common ancestor (TMRCA) of the Y chromosomes of all living humans [ 7 ] and studies of archaic human Y chromosomes [ 1 , 8 ] have used this TMRCA to estimate the TMRCA between modern and archaic human Y chromosomes. The branch length variation observed suggests that there may be differences in mutation rate among Y chromosome lineages. Therefore, incorrectly assuming a constant mutation rate over the Y chromosome phylogeny could result in biased TMRCA estimates that do not accurately reflect population split times. Moreover, variation in the Y chromosome mutation rate could bias methods that estimate the age of ancient and archaic humans by comparing the number of mutations accumulated on their Y chromosomes to those of present-day human Y chromosomes. Newly sequenced, high-quality Y chromosomes from present-day and ancient modern humans, as well as from Neandertals, provide an opportunity to re-evaluate the hypotheses that changes in mutation rate or generation time are the cause of the branch length variation in the Y chromosome phylogeny. These Y chromosomes also allow us to test the effect of reference bias on the mapping of short reads. Reference bias refers to the observation that sequencing reads carrying non-reference alleles are less likely to align successfully to the reference genome than those carrying reference alleles. The effect of reference bias is exacerbated for shorter read lengths and for reads carrying ancient DNA damage, which introduces more mismatches to the reference genome [ 9 , 10 ]. Y chromosomes that are divergent from the reference Y chromosome may be particularly affected by reference bias, making it a possible explanation for the branch shortening reported for some African Y chromosome lineages. The recent availability of high-quality Y chromosome assemblies — including a telomere-to-telomere Y chromosome assembly [ 11 ] and an almost fully contiguous assembly of a Y chromosome representing the basal A0b haplogroup [ 12 ] — allow us to test whether the reference used for alignment has any effect on branch length. Results Branch length variation among a diverse set of Y chromosomes To understand why there appear to be fewer mutations in Y chromosomes that are diverged from the reference used for alignment, we compiled a dataset comprising high-quality Y chromosomes from both present-day and ancient modern humans, as well as from Neandertals ( Table S1 ). While the dataset contains Y chromosomes representing all of the major haplogroups, our focus was on incorporating Y chromosomes that are significantly diverged from the human reference (hg19) Y chromosome. To determine whether there is branch length variation in our dataset, we compared the number of apparent mutations on each Y chromosome to the number in a present-day non-African Y chromosome (haplogroup O1b1a1a1a) since their common ancestor, thereafter referred to as relative branch length difference. If the mutation rate was constant across the Y chromosome phylogeny, we would expect no significant differences between the number of mutations that we detect on the Y chromosomes of present-day individuals. However, significantly fewer mutations are identified for those African Y chromosomes that are the most diverged from the reference, and this difference increases linearly with increasing divergence from the reference ( Figure 1 ). We find that ∼ 55 mutations are missed with each additional pairwise difference from the reference per 10,000 basepairs (bp) ( Figure S1 ). Download figure Open in new tab Fig. 1: Neighbour-joining tree of a Y chromosome phylogeny comprising two Nean-dertals [ 1 , 13 ], four ancient humans [ 14 – 17 ] and twenty eight present-day humans [ 18 , 19 ] with associated relative branch length differences (i.e. the difference in branch length normalised by the total number of sites used for the comparison) compared to a present-day non-African Y chromosome (O1b1a1a1a). The colours indicate the population of origin. The error bars represent 95% confidence intervals (CIs) computed by resampling branch lengths from a Poisson distribution as described in Petr et al. [ 1 ]. We expect to observe shorter branches for ancient individuals than for modern individuals because they have less time to accumulate mutations. This explains the shorter branches for Ust’Ishim and Yana1 ( Figure 1 ). However, we would expect that ancient individuals with similar radiocarbon dates should have branch lengths similar to one another. Instead, when we compare the Y chromosomes of ancient individuals with similar radiocarbon dates but with differing levels of divergence from the reference (i.e. Mezmaiskaya2 vs Ust’Ishim / Shum Laka vs Loschbour ), we find that the divergent branches are significantly shorter than the non-divergent branches. As shortened branches are specific to some African haplogroups (i.e. those most diverged from the reference) rather than to all Africans ( Figure 1 ), this branch length variation is unlikely to be the result of an accumulation of mutations on non-African Y chromosomes following the out-of-Africa bottleneck as proposed by Scozzari et al. [ 3 ]. Testing the hypothesis of mutation rate and generation time variation The significant branch length variation observed over the Y chromosome phylogeny suggests that the assumption of a constant Y chromosome mutation rate and constant generation time may be violated. Therefore, we next estimated the changes in mutation rate and generation time required to explain the observed differences in branch lengths. Previous studies [ 1 , 14 ] have estimated the Y chromosome mutation rate based on the difference in the number of mutations between an ancient human Y chromo-some (that of a 44.5 thousand year old Eurasian hunter-gatherer, Ust’Ishim ) [ 14 ] and those of present-day humans. Starting with the mutation rate of 7.34 × 10 − 10 mutations/bp/year (which was computed from present-day non-African Y chromosomes [ 1 ]), we conducted a series of pairwise branch length comparisons to estimate the relative changes in mutation rate that would result in the observed branch length differences. We used the Y chromosomes that are the most diverged from the reference and calibrated our estimates according to the known ages of the individuals. We find that multiple changes in mutation rate are needed to explain the data and that, in total, the mutation rate would need to have increased by 66% since the split between humans and Neandertals ( Figure 2a and Table S2 ). This contrasts with the finding, based on a comparison between the complete assemblies of human and chimpanzee Y chromosomes, that there has been an 11% decrease in the human Y chromosome mutation rate since the divergence between humans and chimpanzees ∼ 6 million years ago [ 20 ]. Mutation rate differences alone are, therefore, unlikely to explain the observations. Download figure Open in new tab Fig. 2: (a) Mutation rates (in units of mutations/bp/year) required to explain the branch lengths of the Y chromosomes that are highly diverged from the human reference. The dashed lines represent the branch shortening relative to the O1b1a1a1a branch. The colours indicate the branches of the tree and their corresponding mutation rate. (b) Percentage change in the mutation rate if the male generation time changed from the value on the x-axis to 32 years (the current average estimate in present-day populations [ 22 ]). The shaded area corresponds to the change in mutation rate if the current generation time is instead between 28 and 36 years. While the de novo mutation rate increases with paternal age (see Methods), a decrease in generation time (i.e. in the average paternal age) can result in an increase in the number of mutations because more generations occur over time. To determine if the branch length variation could be caused by changes in male generation time, we estimated what change in generation time could have lead to a 66% increase in mutation rate ( Figure 2b , see Methods). We relied on the relationship between paternal age and autosomal mutation rate as this relationship for the Y chromosome mutation rate is difficult to accurately estimate. This is due to the small size of the Y chromosome, which accumulates relatively few mutations per generation, and the limitations of short read sequencing when applied to the highly repetitive regions of the Y chromosome [ 21 ]. Despite this shortcoming, this analysis should give an indication of the magnitude of the change for the Y chromosome mutation rate. We find that no reasonable change in male generation time could account for the apparent significant increase in mutation rate since the split with Neandertals, as differences in generation time could only explain up to a 15% increase in mutation rate ( Figure 2b ). Further, we would expect generation time changes to affect all haplogroups in a population, rather than specific lineages as observed in the data. Therefore, while changes in paternal generation time may explain subtle differences in branch lengths in the Y chromosome phylogeny, they are unlikely to explain the full extent of the difference we observe. Exploring alternative hypotheses for the origin of branch length variation Since neither biological nor demographic factors fully explain the branch length variation in the human Y chromosome phylogeny, we attempted to determine the extent to which reference bias could cause us to undercount mutations in Y chromosomes that are diverged from the reference used for alignment. Mapping to the chimpanzee reference Y chromosome We first tested whether the use of an outgroup Y chromosome (that of the chimpanzee) as a reference can correct for the branch length variation in the human Y chromosome phylogeny. Although this strategy should result in a significant level of reference bias for all human Y chromosomes, we expect this bias to be the same across all lineages, as all are equally diverged from the chimpanzee Y chromosome reference. As we obtained many heterozygous genotype calls with the alignments to the chimpanzee reference (suggesting a high frequency of mapping errors) we concluded that human Y chromosomes are too diverged from the chimpanzee Y chromosome to allow us to accurately call genotypes ( Table S3 ). Mapping to different human Y chromosome assemblies Due to the highly repetitive nature of the human Y chromosome, more than half of the hg19 reference Y chromosome could not be assembled using short read sequencing methods. A fully contiguous telomere-to-telomere (T2T) Y chromosome assembly [ 11 ] (haplogroup J1a2b) allows us to test whether the use of a higher-quality reference influences the number of mutations we are able to detect by facilitating the alignment of more sequences in complex regions. A Y chromosome reference that is outside of the variation of most human haplogroups (haplogroup A0b [ 12 ]) provides an opportunity to test whether we can recover more mutations from the more basal lineages in the phylogeny. To study the effect of the reference used for alignment and determine whether reference bias affects mapping to the Y chromosome, we aligned whole-genome sequence data from the individuals in our dataset for whom we had unmapped reads to three different reference genomes: The hg19 reference with the original hg19 Y chromosome which is almost entirely derived from a lineage that represents the R1b haplogroup (one that is within the variation of present-day non-African Y chromosomes) The hg19 reference with a long-read assembled Y chromosome representing the A0b lineage [ 12 ] (one that is outside of the variation of most modern human Y chromosome lineages, with only the A00 lineage as a more basal outgroup) The hg19 reference with the long-read assembled T2T Y chromosome representing the J1a haplogroup [ 11 ] (which, like the hg19 Y chromosome, is within the variation of present-day non-African Y chromosomes) For each of the three reference genomes, we processed our dataset using the same pipeline, called genotypes, and compared the relative branch length differences with respect to a present-day non-African Y chromosome ( Figure 3a ). We find that the T2T reference has little effect on the branch length variation and we observe similar branch length differences for the two references that are within the variation of non-African Y chromosomes. This shows that mutations are not missed due to the quality of the reference assembly. By contrast, the A0b reference enables the detection of more mutations on the A0a1 lineage, but does not resolve the issue for the other A haplogroups that do not fall on the A0 lineage. We also miss mutations on the non-African lineages that are now highly diverged from the A0b reference ( Figure 3b ). Taken together, these results show that reference bias does affect the alignment of sequences from human Y chromosomes and that this bias will always be present for some lineages in the phylogeny depending on the reference used for alignment. Download figure Open in new tab Fig. 3: (a) Relative branch length differences compared to a present-day non-African lineage (O1b1a1a1a) using variants called from alignments to three different reference genomes (shown with different colours). The error bars correspond to 95% CIs computed by resampling branch lengths from a Poisson distribution. (b) Counts of private mutations for three present-day A lineages (A00, A0a1, and A1a, as indicated in each panel) since their last common ancestor with a present-day non-African lineage (O1b1a1a1a) using variants called from alignments to two different reference genomes (shown with different colours). The results with the alignments to the T2T reference genome are not shown, for simplicity, and are similar to those obtained with the hg19 reference. An approach to limit the effect of reference bias As the reference used for alignment cannot resolve the effect of reference bias for all Y chromosome haplogroups, we then investigated whether specific regions of the hg19 Y chromosome are correlated with a reduction in the ability to call private mutations in divergent Y chromosomes. We used sequence divergence between the hg19 Y chromosome assembly and the chimpanzee reference (panTro6) Y chromosome assembly to identify the faster-evolving regions of the human Y chromosome where the alignment of short read data may be most affected by reference bias. We used a sliding window approach to compute the sequence divergence for each uniquely mappable base of the hg19 Y chromosome based on the number of mismatches compared to the panTro6 Y chromosome. For three present-day haplogroups that are affected by reference bias (A00, A0a1 and B-M181), we computed the mean mutation count difference with respect to all of the haplogroups that do not show significant branch length differences (i.e. all non-African haplogroups along with the E haplogroups) for progressively increasing sequence divergence filters ( Figure 4 ). This allowed us to identify the sequence divergence cutoff that minimises the branch length difference between the lineages that are most diverged from the reference and those that are closer to the reference. As the non-recombining part of the Y chromosome is a single locus, only one phylogeny exists for the human Y chromosome. For this reason, removing regions of the Y chromosome does not introduce biases if we use the appropriate mutation rate in subsequent analyses. Download figure Open in new tab Fig. 4: Comparing the number of mutations on divergent Y chromosomes to non-divergent Y chromosomes for different maximum human-chimp sequence divergence filters (lower x-axis). The upper x-axis represents the average number of positions (in Mb) used for each comparison. The colours correspond to the three divergent Y chromosomes (A00, A0a1 and B-M181). The shaded area indicates the divergence filters that minimise the branch length variation while maximising the proportion of the Y chromosome available for further analysis. The error bars denote 95% CIs computed by resampling branch lengths from a Poisson distribution. There is a tradeoff, however, between filtering the regions of the Y chromosome where mapping is more difficult while simultaneously retaining a sufficient number of positions for further phylogenetic analysis, and we aimed to retain as many of the uniquely mappable positions of the human Y chromosome as possible. The branch length variation is minimised, and the number of positions maximised, using a human-chimp sequence divergence cutoff of 1.9% which retains around 2 Mb (megabases) of the uniquely mappable ∼ 4.6 Mb of the hg19 Y chromosome. We also looked at human-chimp sequence divergence for the complete nuclear genome to determine whether reference bias could have a similar impact on analyses of the autosomes or the X chromosome ( Table S4 ). The Y chromosome stands out for both the proportion of the chromosome that can be aligned and the divergence from the chimpanzee reference, suggesting that reference bias is largest on the Y chromosome. This probably results from the faster evolving nature of the Y chromosome compared to the autosomes [ 14 ]. We then analysed the regions of the Y chromosome that are removed by the human-chimp divergence filter. About half of the non-recombining part of the human Y chromosome consists of highly repetitive heterochromatin and the remaining euchro-matin comprises three sequence classes: the X-degenerate regions, the X-transposed regions, and the ampliconic regions, which primarily consist of palindromic repeats. We find that the majority of the uniquely mappable positions in the ampliconic and X-transposed regions are removed by the divergence filter, whereas more than half of the X-degenerate positions are retained ( Table 1 ). We investigated whether the divergence filter is required in the X-degenerate regions of the Y chromosome by using only these regions when testing for the presence of branch length variation. We still observe shortened branches for the Y chromosome lineages that are most diverged from the reference ( Figure S2 ) showing that reference bias also affects the X-degenerate regions and that the divergence filter is necessary for unbiased analyses of the human Y chromosome phylogeny. View this table: View inline View popup Download powerpoint Table 1: Uniquely mappable positions removed by the human-chimp divergence filter for each euchromatic sequence class of the Y chromosome. We hypothesized that regions masked by the divergence filter would show evidence for an increased rate of mapping errors. To assess this we identified heterozygous sites called by snpAD – a diploid genotype caller – on the haploid Y chromosome, and calculated the percentage change in the number of heterozygous calls for each individual after applying the human-chimp sequence divergence filter. We found that a large proportion (between 7 to 92%) of mapping errors occur in the divergent regions of the Y chromosome ( Table S5 ), indicating the increased difficulty of correctly aligning reads in quickly-evolving regions. Application to estimates of the modern human and the modern human-Neandertal Y chromosome TMRCAs After minimising the effect of reference bias in the Y chromosome phylogeny, we sought to recalibrate both the modern human Y chromosome TMRCA (using the A00 Y chromosome to represent the most basal modern human lineage) and the modern human-Neandertal Y chromosome TMRCA using the portion of the human Y chromosome with minimal reference bias. We used two different methods to determine the Y chromosome TMRCAs: 1) the method defined in [ 1 ] (with the lineages used shown in Figure 5 ) which computes the modern human-Neandertal TMRCA by scaling the modern human TMRCA based on modern human-Neandertal divergence and 2) Bayesian phylogenetic reconstruction as implemented in the software package BEAST2 (v2.6.7) [ 23 ]. Download figure Open in new tab Fig. 5: Tree depicting the lineages used to estimate the TMRCA of all modern humans and the TMRCA of modern and archaic humans. The lineage of the radiocarbon dated ancient human, Ust’Ishim , used to calculate the mutation rate, is also shown. We first calculated the Y chromosome mutation rate for those regions of the Y chromosome that pass the divergence filter ( Table 2 ). The mutation rate in these regions is slightly lower than the mutation rate computed over the entire uniquely mappable Y chromosome ( Table 2 ) and the rates reported elsewhere ( Table S6 ). This is expected as more divergent, faster-evolving regions of the Y chromosome are removed by the divergence filter. In all subsequent analyses, we used this recalibrated mutation rate to account for the effect of the divergence filter. View this table: View inline View popup Download powerpoint Table 2: Comparing the mutation rate estimates (in units of mutations/bp/year) for the uniquely mappable part of the Y chromosome and for the uniquely mappable, divergence-filtered Y chromosome (maximum 1.9% human-chimp sequence divergence). The method used to estimate the mutation rate (as described in the main text) is indicated. HPD - highest posterior density. We then computed the modern human and the modern human-Neandertal TMR-CAs, and compared the TMRCAs estimated with all of the uniquely mappable sites to those estimated with only sites that pass the divergence filter ( Figure 6 ). When the unfiltered Y chromosome is used, the choice of the modern human branch used for comparison has an effect on the TMRCA calculation with the difference in length between the A00 branch and the non-African branches resulting in inconsistent estimates. By applying the divergence filter, the effect of reference bias is minimised and the choice of modern human branch then has no effect on the TMRCA estimates. Download figure Open in new tab Fig. 6: TMRCA estimates between the Y chromosomes on the x-axis and 16 non-African Y chromosomes. Each dot represents the TMRCA with one non-African Y chromosome. The top plot shows the TMRCAs estimated by measuring the non-African branch, while the bottom plot shows the TMRCAs estimated by measuring the A00 branch. The dots in colour represent the TMRCA estimates based on the the filtered Y, while the dots in grey represent the TMRCAs estimated using the unfiltered Y chromosome. The vertical lines denote 95% CIs computed by resampling branch lengths from a Poisson distribution. The dashed horizontal lines represent the mean TMRCAs computed over all non-African Y chromosomes and the solid horizontal lines show overall 95% CIs. As we are able to calculate unbiased TMRCAs after masking divergent regions, we then used BEAST to reconstruct the Y chromosome phylogeny (see Methods) based on the sites that pass the divergence filter ( Figure 7 ). BEAST estimates similar TMRCAs to those estimated with the method described above, although with narrower confidence intervals. We estimated that the modern human TMRCA is 268 ka (95% HPD 238 to 301 ka) and the modern human-Neandertal TMRCA is 387 ka (95% HPD 344 to 432 ka). Our recalibrated TMRCAs are slightly older than previous estimates of 254 ka (95% CI 192 to 307 ka) for the modern human TMRCA [ 7 ] and 370 ka (95% CI 326 to 420 ka) for the modern human-Neandertal TMRCA [ 1 ], but within their confidence intervals. Download figure Open in new tab Fig. 7: Y chromosome phylogeny reconstructed with BEAST. The TMRCA estimates, as well as the estimated age of Chagyrskaya 2 , and their respective 95% HPD intervals are indicated on the tree. The ages of the other ancient samples were set to the estimated radiocarbon dates. The haplogroups from different populations are highlighted with different colours. The branches are to scale, in thousands of years (ka). We then tested whether we can use BEAST to estimate the age of a shotgun sequenced ancient individual from Shum Laka, Cameroon dated to ∼ 8 ka (7,970–7,800 calibrated years before present) who carried the A00 Y haplogroup using just the Y chromosome. For this analysis, we set a wide prior for the age of the ancient individual (-50,000 to 200,000) and did not include the present-day A00 individuals. Using the same parameters used for generating the full phylogeny above, we constructed one phylogeny with the filtered Y chromosome and one with the unfiltered Y chromosome ( Figure 8 ). We find that if we use the unfiltered Y chromosome to construct the phylogeny, we get an age-estimate for the ancient A00 individual that is about 30 thousand years older than the radiocarbon date due to the reference bias that leads to an underestimate of the number of mutations on his branch. If we use the filtered Y chromosome regions, we obtain an age estimate of 13 ka (95% CI 43 to 0) which is reasonably close to the radiocarbon date (albeit with wide confidence intervals) indicating that the phylogeny is unbiased. Download figure Open in new tab Fig. 8: Estimating the age of the radiocarbon dated A00 individual (radiocarbon date ∼ 8 ka) using molecular dating of the Y chromosome with BEAST. Phylogenies based on the filtered (a) and the unfiltered (b) Y chromosome are shown. The branches corresponding to 21 Y chromosomes included in the analysis ( Table S1 ) were collapsed. The TMRCA estimates and their 95% HPD intervals are indicated on the tree. The age of the A00 individual, along with its associated 95% HPD interval, is indicated in the branch label. We also re-estimated an age for the Chagyrskaya 2 Neandertal — previously dated to ∼ 64 ka (95% CI 48-83) [ 13 ]. Our estimate of 76 ka (95% CI 59-93) is consistent with a Neandertal occupation of Chagyrskaya Cave between 92 ka and 49 ka, and close to the genetic age of Chagyrskaya 8 ( ∼ 80 ka), another individual from the same cave, whose age was estimated from a high-coverage genome [ 24 ]. In summary, although the re-estimated TMRCAs are not significantly different from previous estimates, these analyses show that the divergence filtering is necessary to obtain Y chromosome-based age estimates for ancient individuals with Y chromosomes that are diverged from the human reference. Discussion We show here that variation in Y chromosome branch lengths is at least in part explained by biases introduced by alignment to a single Y reference genome representing a single haplogroup. Although we do not observe any significant haplogroup-specific variation after accounting for the effect of reference bias ( Figure S3 and Figure S4 ), a larger Y chromosome dataset with multiple samples representing specific lineages in populations of interest may allow for the detection of subtle branch length differences that result from changes in mutation rate or generation time. The Y chromosome is more affected than the autosomes by this reference bias due to its highly repetitive nature and extensive structural variation. We show that restricting analyses of human Y chromosomes to regions with a human-chimpanzee divergence of ≤ 1.9% allows accurate estimation of branch lengths and, therefore, unbi-ased estimates of the human Y chromosome phylogeny. However, doing so reduces the (already limited) amount of the Y chromosome that is accessible for analysis by more than half, leading to wider confidence intervals in both TMRCA and age estimates. For analyses of Y chromosomes in specific populations, it may be possible to recover more of the Y sequence by alignment to the closest high-quality Y chromosome assembly [ 12 ]. In contrast to using a single linear reference genome, approaches that incorporate Y chromosome haplotype diversity in the reference to which sequence reads are aligned, known as ‘pan-genomes’, have been shown to reduce the effect of reference bias when used as a reference for short-read mapping [ 25 , 26 ] and could potentially maximise the amount of the Y chromosome that is accessible for comparison. The repetitive nature of the Y chromosome, along with its extensive structural variation, makes it a particularly good target for a pangenome-based approach. A number of high-quality Y chromosome assemblies that could be used to construct a pangenome reference [ 12 ] are already available, although methods to incorporate ancient Y chromosome variation into pangenome graphs have yet to be explored and are an exciting direction for future research. We expect our divergence-filtering approach to be particularly useful for analyses of archaic human (i.e. Neandertal and Denisovan) Y chromosomes that are substantially diverged from the human reference, and for which the construction of pangenome graphs remains challenging. These Y chromosomes will be most affected by reference bias, so methods that rely on comparing the number of private mutations on archaic Y chromosomes to those of present-day human Y chromosomes will underestimate the archaic branch length if they do not account for this bias. Materials and Methods Dataset The dataset includes high-coverage Y chromosome data from present-day and ancient modern humans as well as from Neandertals ( Table S1 ). For the present day humans, a subset of individuals from the Human Genome Diversity Project (HGDP) [ 18 ] dataset were selected to represent the Y chromosome phylogeny with a particular focus toward individuals with Y chromosomes that are highly diverged from the human reference Y chromosome. We also included two individuals (carrying Y chromosomes representing the A haplogroup) from the high-coverage resequencing of the 1000 Genomes Project (1KGP) [ 19 ]. The dataset also contains Y chromosomes from two closely related individuals with the A00 haplogroup [ 7 ]. Due to the relatively low coverage of the sequence data for these two A00 Y chromosomes, we followed the approach of the original publication and merged the two A00 Y chromosomes into a single BAM file. To calibrate the Y chromosome phylogeny, we included high-coverage, shotgun-sequenced Y chromosome data from the following radiocarbon dated ancient modern humans: Ust’Ishim [ 14 ], Yana1 [ 17 ], Loschbour [ 16 ] and an individual from Shum Laka with the A00 haplogroup [ 15 ]. We also included two high-coverage Neandertal Y chromosomes that were obtained by enrichment-capture sequencing: Mezmaiskaya2 [ 1 ] and Chagyrskaya2 [ 13 ]. Data Processing Sequences were aligned using BWA-MEM (v0.7.17) [ 27 ] for present-day individuals and BWA-ALN (with ancient parameters — “-n 0.01 -o 2 -l 16500”) for ancient individuals. For the present-day HGDP and 1KGP individuals, adapter sequences were identified and flagged using MarkIlluminaAdapters from Picard tools (v2.18.29) before alignment. For the ancient samples, we used the merged and adapted trimmed sequences as published ( Table S1 ). The aligned BAM files for each individual were filtered with samtools [ 28 ] to retain reads at least 35 bp in length and with a minimum mapping quality of 25, and PCR duplicates were removed using bam-rmdup (v0.2, https://github.com/mpieva/biohazard-tools/ ). We also removed sequences with indels as we observed an increased frequency of genotyping errors around indels. We restricted our analysis to uniquely mappable regions of the Y chromosome so as to minimise errors caused by ambiguous alignment of short reads. To exclude non-unique regions of the hg19 Y chromosome, we used the map35_100 filter from [ 29 ] which retains only positions where all 35mers overlapping that position do not map to any other position in the genome allowing up to one mismatch. We also applied the capture_full filter from [ 1 ], which includes masks of tandem repeats identified by Tandem Repeat Finder [ 30 ]. Genotypes were called with snpAD [ 31 ] – a genotyper that accounts for ancient DNA damage – using bases with a minimum base quality of 30. For consistency, snpAD was used to call genotypes for both ancient and modern individuals, however, position-specific error rates were estimated independently for each sample. As snpAD is designed to call diploid genotypes, heterozygous genotype calls were converted to homozygous genotypes choosing the lowest homozygous posterior probability as calculated by snpAD. Finally, following [ 29 ], genotypes were filtered based on maximum GC-corrected coverage cutoffs for each individual (see Availability of data and materials) in order to remove potential duplications or regions with spurious alignments of microbial sequences. Methods to Estimate Branch Shortening Branch Length Difference To calculate the difference in branch length with respect to a reference individual, we counted the number of derived mutations on each branch (using the chimpanzee genome to infer the ancestral state) with the BranchShortening.get_mutations_vs_individual function from https://github.com/yanivsw/y_ chr_utils. The UCSC whole genome alignments between the human reference genome (hg19) and the chimpanzee reference genome (panTro6) were used to determine ancestral genotypes. Estimating the Human Y chromosome Mutation Rate A Y chromosome from an ancient human individual with an accurately estimated age can be used to determine the human Y chromosome mutation rate. The Ust’Ishim individual is estimated, based on radiocarbon dating, to have lived around 44 thousand years ago (45,930-42,904 calibrated years before present; [ 14 ]). We used the high-coverage genome sequence of Ust’Ishim to determine the number of derived mutations on the branch leading to Ust’Ishim , and compared this to an individual living today. Normalising the number of mutations accumulated per year by the total number of positions used for the comparison results in the Y chromosome mutation rate in units of mutations per position per year. Estimating the Mutation Rate on a Shortened Branch To estimate the mutation rate that would result in a shortened branch when comparing two Y chromosome lineages, we first calculate the expected branch length difference using the known ages of the two individuals and the expected mutation rate. We then calculate the proportion of missing mutations using the observed branch length difference and adjust the expected mutation rate, based on this proportion, to estimate the mutation rate on the shorter branch. Estimating TMRCAs of Modern and Archaic Human Y Chromosomes We used the method described in [ 1 ] to estimate the TMRCAs of modern and archaic Y chromosomes. Briefly, the method first estimates the modern human TMRCA by counting the differences between a present-day African Y chromosome and a present-day non-African Y chromosome and converting this count to time using a mutation rate. The archaic TMRCA is then computed by scaling the modern human TMRCA based on the fold increase in divergence between one of the present-day Y chromosomes and the archaic Y chromosome relative to the divergence between the two present-day Y chromosomes. Mapping to the Chimpanzee Reference Y Chromosome We replaced the hg19 Y chromosome with the chimpanzee reference (panTro6) Y chromosome (hg19_panTro6_Y) and aligned all sequence reads from two ancient genomes ( Ust’Ishim and Shum Laka ) to this reference. To conduct a fair comparison, we downsampled the aligned sequences of Ust’Ishim to the same coverage distribution and read length distribution of the aligned sequences of Shum Laka . We then called genotypes with the same pipeline that we use to call genotypes from the sequences aligned to the human hg19 reference. Estimating the Percentage Change in Mutation Rate in Response to a Change in Generation Time To estimate the relative change in mutation rate corresponding to a change in generation time, we used the relationship between paternal age ( a ) and the autosomal germline mutation rate ( µ , in units of male-transmitted mutations per generation) defined in [ 32 ] as To account for the fact that a decrease in the average paternal age can result in an increase in the number of mutations over time due to an increase in the number of generations, we define the relative change in mutation rate (if the paternal age changed from a initial to a current ) as Processing of Long-Read Assembled Y Chromosomes We replaced the Y chromosome of the hg19 reference genome with the A0b Y chromosome (hg19_A0b_Y) from [ 12 ] or with the T2T Y chromosome (hg19_T2T_Y) from [ 11 ]. We then generated new map35_100 filters and ran Tandem Repeat Finder [ 30 ] on both the hg19_A0b_Y genome and the hg19_T2T_Y genome to identify the uniquely mappable regions for each genome. To call genotypes, we used the same processing pipeline as the one used for hg19. To identify ancestral alleles, we used LASTZ [ 33 ] to align the chimpanzee reference genome (panTro6) to both the hg19_A0b_Y genome and the hg19_T2T_Y genome using the UCSC LASTZ alignment parameters for human-chimp alignments. The alignments were then chained with axtChain using the UCSC chaining parameters and the high-scoring ( > 2000) alignments were extracted from which we ‘called’ panTro6 genotypes. Human-chimp Sequence Divergence To compute human-chimp sequence divergence for each base of the hg19 Y chromosome, we extracted the Y chromosome alignments from the UCSC hg19-panTro6 pairwise alignment file (hg19.panTro6.synNet.maf.gz) and used the maf_to_pairwise_identity.py script from [ 34 ] to determine the alignment state (i.e. match, mismatch, insertion, deletion, no alignment) for each position in the hg19 Y chromosome. We then used a 400 bp sliding window centred on each position to compute human-chimp sequence divergence as where n match is the number of matching bases and n total is the total number of aligned bases in each window. Due to the fragmented nature of the uniquely mappable fraction of the human Y chromosome, we additionally required a minimum number of human-chimp aligned bases (200 bp) to compute the sequence divergence for a position. Bayesian Phylogenetic Analysis We used BEAST2 (v2.6.7) [ 23 ] to reconstruct the Y chromosome phylogeny and estimate the TMRCAs. As BEAST requires a FASTA file as input, the VcfHandler.convert_vcf_dataframe_to_fasta_file function from https://github . com/yanivsw/y_chr_utils was used to convert the Y chromosome VCF data to FASTA format. Only variable sites were used as input to the BEAST analysis in order to reduce the computational complexity of the phylogenetic reconstruction, although we defined the nucleotide composition of the invariant sites by editing the BEAST .xml file as was done in [ 7 ]. We used the BEAST2 bModelTest package [ 35 ] to estimate the most appropriate nucleotide substitution model for our data set. bModelTest averages over all substitution models while simultaneously reconstructing the phylogeny. To select the tree model and the clock model that best fit our dataset, we estimated the marginal likelihood for each model combination using a path sampling approach as implemented in the BEAST2 MODEL_SELECTION package ( Table S7 ). To generate the full phylogeny using the best fitting models (the strict clock model together with the Bayesian Skyline tree model), we performed four independent MCMC runs of 55 million iterations with a 10 million iteration pre-burnin stage, sampling every 5,000 steps. We then combined the resulting log files with logcom-biner using a 10% burnin for each run and generated a single best supported tree with TreeAnnotator. Declarations Ethics approval and consent to participate All sequencing data used in this work was previously published. Consent for publication Not applicable. Availability of data and materials All sequence data used are publicly available ( Table S1 ). The merged genotype files and filters used for the analysis are available from Zenodo (DOI:10.5281/zenodo.12635540) and all code is available from https://github.com/yanivsw/y_chr_reference_bias . Competing interests The authors declare that they have no competing interests. Funding This work was supported by funding from the Max Planck Society. Authors’ contributions All authors contributed to the study conception and design. Data collection, programming and data analysis were performed by Y.S. The first draft of the manuscript was written by Y.S and S.P. All authors revised and approved the final manuscript. Supplementary Materials Bayesian Phylogenetic Analysis The tip dates of modern samples were set to zero, while uniform priors were used to represent the ages of the ancient samples. For the radiocarbon dated ancient samples, the range of the prior was set as the 95% confidence interval based on the IntCal20 calibration curve and for Chagyrskaya2 (who cannot be radiocarbon dated, but has been genetically dated to 64,000 years ago [ 13 ]) we set a wide uniform prior ranging from 40,000 to 120,000. We set the initial Y chromosome mutation rate to 7.34 × 10 − 10 mutations/bp/year as calculated in [ 1 ]. In order to allow BEAST to effectively estimate the Y chromosome mutation rate, we set a wide uniform prior of 4 × 10 − 10 to 10 × 10 − 10 . The modern humans and the Neandertals were constrained as two different mono-phyletic groups in order to estimate their respective Y chromosome TMRCAs, and we estimated the modern human-Neandertal TMRCA by creating a third monophyletic group with only the chimpanzee as an outgroup. To select the tree model and the clock model that best fit our dataset, we estimated the marginal likelihood for each model combination using a path sampling approach as implemented in the BEAST2 MODEL_SELECTION package. We tested the following model combinations: Strict clock, constant population size Relaxed log-normal clock, constant population size Strict clock, Coalescent Bayesian Skyline Relaxed log-normal clock, Coalescent Bayesian Skyline For each model combination we used 50 path steps, an alpha parameter of 0.3 for the Beta distribution used to space out the steps, and a chain length of 10 million MCMC iterations for each step. The pre-burn-in stage was set at 5 million iterations and an 80% burn-in was used for each chain. The two best supported model combinations were those with the Coalescent Bayesian Skyline tree model with slightly higher support for the relaxed clock model ( Table S7 ). The support for the relaxed clock model over the strict clock model was not significant however (Bayes factor = 0.34), so we used the simpler model of the strict clock together with the Bayesian Skyline tree model to infer the various TMRCAs. The nucleotide substitution model with the highest posterior probability as estimated by BEAST using the bModelTest package is the unnamed 123421 model. The nucleotide substitution rates are summarised in Table S8 . Supplementary Figures Download figure Open in new tab Fig. S1: Branch length differences versus divergence from the reference for all present-day Y chromosomes. The black line represents the straight line that best fits the data with slope -54.66 mutations per pairwise differences per 10 kbp (kilobase pairs). Download figure Open in new tab Fig. S2: Relative branch length differences compared to a present-day non-African Y chromosome (O1b1a1a1a) based on the X-degenerate regions of the Y chromosome. Download figure Open in new tab Fig. S3: Relative branch length differences compared to a present-day non-African Y chromosome (O1b1a1a1a) for all present-day Y chromosomes after minimising the effect of reference bias. Download figure Open in new tab Fig. S4: Relative branch length differences compared to A00 for all present-day Y chromosomes after minimising the effect of reference bias. Supplementary Tables View this table: View inline View popup Download powerpoint Table S1: Dataset View this table: View inline View popup Download powerpoint Table S2: The mutation rates required to explain the observed differences in branch length. View this table: View inline View popup Download powerpoint Table S3: Results of mapping to the chimpanzee Y chromosome. View this table: View inline View popup Download powerpoint Table S4: Human-chimp sequence divergence across the human genome. View this table: View inline View popup Download powerpoint Table S5: Comparing the proportion of snpAD-called heterozygotes (likely mapping errors) for the uniquely mappable part of the Y chromosome and for the uniquely mappable, divergence-filtered Y chromosome (based on a human-chimp sequence divergence cutoff of 1.9%). View this table: View inline View popup Download powerpoint Table S6: Y chromosome mutation rates estimated in previous studies View this table: View inline View popup Download powerpoint Table S7: Marginal log-likelihood for the model combinations tested using the path sampling approach as implemented in BEAST2. View this table: View inline View popup Download powerpoint Table S8: Nucleotide substitution rates estimated by the BEAST bModelTest package. Acknowledgements We thank Kay Prüfer and Leonardo N. M. Iasi for helpful discussions and Mark Stoneking and Eleni Seferidou for both helpful discussions and feedback on the manuscript. References 1. ↵ Petr , M. et al. The Evolutionary History of Neanderthal and Denisovan Y Chromosomes . Science 369 , 1653 – 1656 ( 2020 ). OpenUrl Abstract / FREE Full Text 2. ↵ Wei , W. et al. A Calibrated Human Y-chromosomal Phylogeny Based on Resequencing . Genome Research 23 , 388 – 395 ( 2013 ). OpenUrl Abstract / FREE Full Text 3. ↵ Scozzari , R. et al. An Unbiased Resource of Novel SNP Markers Provides a New Chronology for the Human Y Chromosome and Reveals a Deep Phylogenetic Structure in Africa . Genome Research 24 , 535 – 544 ( 2014 ). OpenUrl Abstract / FREE Full Text 4. ↵ Hallast , P. et al. The Y-Chromosome Tree Bursts into Leaf: 13,000 High-Confidence SNPs Covering the Majority of Known Clades . Molecular Biology and Evolution 32 , 661 – 673 ( 2015 ). OpenUrl CrossRef PubMed 5. ↵ Barbieri , C. et al. Refining the Y Chromosome Phylogeny with Southern African Sequences . Human Genetics 135 , 541 – 553 ( 2016 ). OpenUrl CrossRef PubMed 6. ↵ Naidoo , T. et al. Y-Chromosome Variation in Southern African Khoe-San Popu-lations Based on Whole-Genome Sequences . Genome Biology and Evolution 12 , 1031 – 1039 ( 2020 ). OpenUrl CrossRef 7. ↵ Karmin , M. et al. A Recent Bottleneck of Y Chromosome Diversity Coincides with a Global Change in Culture . Genome Research 25 , 459 – 466 . pmid: 25770088 ( 2015 ). OpenUrl Abstract / FREE Full Text 8. ↵ Mendez , F. L. , Poznik , G. D. , Castellano , S. & Bustamante , C. D . The Divergence of Neandertal and Modern Human Y Chromosomes . The American Journal of Human Genetics 98 , 728 – 734 ( 2016 ). OpenUrl CrossRef PubMed 9. ↵ Günther , T. & Nettelblad , C . The Presence and Impact of Reference Bias on Population Genomic Studies of Prehistoric Human Populations . PLOS Genetics 15 , e1008302 ( 2019 ). OpenUrl 10. ↵ Peyrégne , S. et al. Nuclear DNA from Two Early Neandertals Reveals 80,000 Years of Genetic Continuity in Europe . Science Advances 5 , eaaw5873 ( 2019 ). 11. ↵ Rhie , A. et al. The Complete Sequence of a Human Y Chromosome . Nature ( 2023 ). 12. ↵ Hallast , P. et al. Assembly of 43 Human Y Chromosomes Reveals Extensive Complexity and Variation . Nature ( 2023 ). 13. ↵ Skov , L. et al. Genetic Insights into the Social Organization of Neanderthals . Nature 610 , 519 – 525 ( 2022 ). OpenUrl CrossRef 14. ↵ Fu , Q. et al. Genome Sequence of a 45,000-Year-Old Modern Human from Western Siberia . Nature 514 , 445 – 449 ( 2014 ). OpenUrl CrossRef PubMed Web of Science 15. ↵ Lipson , M. et al. Ancient West African Foragers in the Context of African Population History . Nature 577 , 665 – 670 ( 2020 ). OpenUrl CrossRef 16. ↵ Lazaridis , I. et al. Ancient Human Genomes Suggest Three Ancestral Populations for Present-Day Europeans . Nature 513 , 409 – 413 ( 2014 ). OpenUrl CrossRef PubMed Web of Science 17. ↵ Sikora , M. et al. The Population History of Northeastern Siberia since the Pleistocene . Nature 570 , 182 – 188 ( 2019 ). OpenUrl 18. ↵ Bergström , A. et al. Insights into Human Genetic Variation and Population History from 929 Diverse Genomes . Science 367 , eaay5012 ( 2020 ). 19. ↵ Byrska-Bishop , M. et al. High-Coverage Whole-Genome Sequencing of the Expanded 1000 Genomes Project Cohort Including 602 Trios . Cell 185 , 3426 – 3440 .e19 ( 2022 ). OpenUrl CrossRef PubMed 20. ↵ Makova , K. D. et al. The Complete Sequence and Comparative Analysis of Ape Sex Chromosomes . Nature 630 , 401 – 411 ( 2024 ). OpenUrl 21. ↵ Helgason , A. et al. The Y-chromosome Point Mutation Rate in Humans . Nature Genetics 47 , 453 – 457 ( 2015 ). OpenUrl CrossRef PubMed 22. ↵ Fenner , J. N . Cross-Cultural Estimation of the Human Generation Interval for Use in Genetics-Based Population Divergence Studies . American Journal of Physical Anthropology 128 , 415 – 423 ( 2005 ). OpenUrl CrossRef PubMed Web of Science 23. ↵ Bouckaert , R. et al. BEAST 2.5: An Advanced Software Platform for Bayesian Evolutionary Analysis . PLOS Computational Biology 15 , e1006650 ( 2019 ). OpenUrl 24. ↵ Mafessoni , F. et al. A High-Coverage Neandertal Genome from Chagyrskaya Cave . Proceedings of the National Academy of Sciences 117 , 15132 – 15136 ( 2020 ). OpenUrl Abstract / FREE Full Text 25. ↵ Liao , W.-W. et al. A Draft Human Pangenome Reference . Nature 617 , 312 – 324 ( 2023 ). OpenUrl CrossRef PubMed 26. ↵ Martiniano , R. , Garrison , E. , Jones , E. R. , Manica , A. & Durbin , R . Removing Reference Bias and Improving Indel Calling in Ancient DNA Data Analysis by Mapping to a Sequence Variation Graph . Genome Biology 21 , 250 ( 2020 ). OpenUrl CrossRef PubMed 27. ↵ Li , H. & Durbin , R . Fast and Accurate Long-Read Alignment with Burrows– Wheeler Transform . Bioinformatics 26 , 589 – 595 ( 2010 ). OpenUrl CrossRef PubMed Web of Science 28. ↵ Danecek , P. et al. Twelve Years of SAMtools and BCFtools . GigaScience 10 , giab008 ( 2021 ). 29. ↵ Prüfer , K. et al. The Complete Genome Sequence of a Neanderthal from the Altai Mountains . Nature 505 , 43 – 49 ( 2014 ). OpenUrl CrossRef GeoRef PubMed Web of Science 30. ↵ Benson , G . Tandem Repeats Finder: A Program to Analyze DNA Sequences . Nucleic Acids Research 27 , 573 – 580 ( 1999 ). OpenUrl CrossRef PubMed Web of Science 31. ↵ Prüfer , K . snpAD: An Ancient DNA Genotype Caller . Bioinformatics 34 , 4165 – 4171 ( 2018 ). OpenUrl 32. ↵ Jónsson , H. et al. Parental Influence on Human Germline de Novo Mutations in 1,548 Trios from Iceland . Nature 549 , 519 – 522 ( 2017 ). OpenUrl CrossRef PubMed 33. ↵ Harris , R. S. Improved Pairwise Alignment of Genomic DNA PhD thesis (United States – Pennsylvania). 84 pp. isbn: 9780549431701. 34. ↵ Cechova , M. et al. Dynamic Evolution of Great Ape Y Chromosomes . Proceedings of the National Academy of Sciences 117 , 26273 – 26280 ( 2020 ). OpenUrl Abstract / FREE Full Text 35. ↵ Bouckaert , R. R. & Drummond , A. J . bModelTest: Bayesian Phylogenetic Site Model Averaging and Model Comparison . BMC Evolutionary Biology 17 , 42 ( 2017 ). 36. Xue , Y. et al. Human Y Chromosome Base-Substitution Mutation Rate Measured by Direct Sequencing in a Deep-Rooting Pedigree . Current Biology 19 , 1453 – 1457 ( 2009 ). OpenUrl CrossRef PubMed Web of Science 37. Poznik , G. D. et al. Sequencing Y Chromosomes Resolves Discrepancy in Time to Common Ancestor of Males Versus Females . Science 341 , 562 – 565 ( 2013 ). OpenUrl Abstract / FREE Full Text View the discussion thread. Back to top Previous Next Posted July 06, 2024. Download PDF Email Thank you for your interest in spreading the word about bioRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following Resolving the source of branch length variation in the Y chromosome phylogeny Message Subject (Your Name) has forwarded a page to you from bioRxiv Message Body (Your Name) thought you would like to see this page from the bioRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share Resolving the source of branch length variation in the Y chromosome phylogeny Yaniv Swiel , Janet Kelso , Stéphane Peyrégne bioRxiv 2024.07.05.602100; doi: https://doi.org/10.1101/2024.07.05.602100 Share This Article: Copy Citation Tools Resolving the source of branch length variation in the Y chromosome phylogeny Yaniv Swiel , Janet Kelso , Stéphane Peyrégne bioRxiv 2024.07.05.602100; doi: https://doi.org/10.1101/2024.07.05.602100 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Genetics Subject Areas All Articles Animal Behavior and Cognition (7648) Biochemistry (17729) Bioengineering (13923) Bioinformatics (42053) Biophysics (21492) Cancer Biology (18638) Cell Biology (25569) Clinical Trials (138) Developmental Biology (13405) Ecology (19943) Epidemiology (2067) Evolutionary Biology (24368) Genetics (15625) Genomics (22550) Immunology (17764) Microbiology (40476) Molecular Biology (17208) Neuroscience (88768) Paleontology (667) Pathology (2843) Pharmacology and Toxicology (4834) Physiology (7660) Plant Biology (15175) Scientific Communication and Education (2047) Synthetic Biology (4304) Systems Biology (9836) Zoology (2272)
Text is read by the "Ask this paper" AI Q&A widget below.
Extraction quality varies by source — PMC NXML preserves structure
cleanly, OA-HTML may include some navigation residue, and OA-PDF can
have broken hyphenation. The publisher copy
(via DOI)
is the canonical version.