S4-Multi: enhancing polygenic score prediction in ancestrally diverse populations

preprint OA: closed CC-BY-4.0
📄 Open PDF Full text JSON View at publisher

Abstract

While polygenic scores (PGSs) have shown promise in advancing precision medicine by capturing the additive effects of common germline variants on inherited disease risk, they are presently limited by reduced performance outside of European-origin populations. We extend our previously developed Bayesian polygenic model (PGM) method, select and shrink with summary statistics (S4), to improve prediction accuracy in ancestrally diverse populations. We benchmark this multi-ancestry extension (S4-Multi) against alternative methods on both simulated and biobank data predicting type 2 diabetes, breast cancer, colorectal cancer, asthma, and stroke. In simulation tests, we find that S4-Multi achieves 169% improvement on average over its single ancestry S4 counterpart at prediction in non-European target populations. S4-Multi matches or exceeds top performing methods across the ancestry continuum. In biobank tests, we find that the top-performing PGM method varies considerably by target ancestry and phenotype, with S4-Multi achieving comparable performance to top multi-ancestry methods overall. However, S4-Multi does so while including between 9% and 77% fewer genetic variants relative to competing models, suggesting potential for robust performance in clinical settings with limited available genomic data.
Full text 56,441 characters · extracted from preprint-html · click to expand
S4-Multi: enhancing polygenic score prediction in ancestrally diverse populations | medRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-P4HH5NV'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search S4-Multi: enhancing polygenic score prediction in ancestrally diverse populations John Baierl , Jonathan P. Tyrer , Ping-Hung Lai , Simon A. Gayther , Yi-Wen Hsiao , View ORCID Profile Michelle Jones , View ORCID Profile Pei-Chen Peng , View ORCID Profile Paul D. P. Pharoah doi: https://doi.org/10.1101/2025.01.24.25321098 John Baierl 1 Department of Computational Biomedicine, Cedars-Sinai Medical Center , Los Angeles, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site Jonathan P. Tyrer 2 Department of Public Health and Primary Care, University of Cambridge , UK Find this author on Google Scholar Find this author on PubMed Search for this author on this site Ping-Hung Lai 1 Department of Computational Biomedicine, Cedars-Sinai Medical Center , Los Angeles, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site Simon A. Gayther 3 Department of Medicine, UT Health , San Antonio, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site Yi-Wen Hsiao 1 Department of Computational Biomedicine, Cedars-Sinai Medical Center , Los Angeles, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site Michelle Jones 4 Department of Biomedical Sciences, Cedars-Sinai Medical Center , Los Angeles, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Michelle Jones Pei-Chen Peng 1 Department of Computational Biomedicine, Cedars-Sinai Medical Center , Los Angeles, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Pei-Chen Peng For correspondence: pei-chen.peng{at}cshs.org Paul D. P. Pharoah 1 Department of Computational Biomedicine, Cedars-Sinai Medical Center , Los Angeles, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Paul D. P. Pharoah Abstract Full Text Info/History Metrics Data/Code Preview PDF Abstract While polygenic scores (PGSs) have shown promise in advancing precision medicine by capturing the additive effects of common germline variants on inherited disease risk, they are presently limited by reduced performance outside of European-origin populations. We extend our previously developed Bayesian polygenic model (PGM) method, select and shrink with summary statistics (S4), to improve prediction accuracy in ancestrally diverse populations. We benchmark this multi-ancestry extension (S4-Multi) against alternative methods on both simulated and biobank data predicting type 2 diabetes, breast cancer, colorectal cancer, asthma, and stroke. In simulation tests, we find that S4-Multi achieves 169% improvement on average over its single ancestry S4 counterpart at prediction in non-European target populations. S4-Multi matches or exceeds top performing methods across the ancestry continuum. In biobank tests, we find that the top-performing PGM method varies considerably by target ancestry and phenotype, with S4-Multi achieving comparable performance to top multi-ancestry methods overall. However, S4-Multi does so while including between 9% and 77% fewer genetic variants relative to competing models, suggesting potential for robust performance in clinical settings with limited available genomic data. Introduction The heritable component of most disease risk is polygenic in nature 1 , 2 , and polygenic scores (PGSs) have shown promise in refining genetic risk prediction and personalized medicine in recent years 3 – 6 . PGSs are able to capture the additive effects of common germline genetic variants on disease risks, including those that fail to reach genome-wide significance on their own, producing a single weighted sum for each individual that is more strongly associated with the phenotype of interest. PGSs have proven useful estimating the probabilistic susceptibility of individuals to a range of complex traits, showing validity in both research-based case-control studies and population-based cohort studies 7 – 9 . Genetic risk estimation holds the potential to aid in early disease detection, improve the effectiveness of screening programs, and inform preventative intervention choices both used alone and in conjunction with other lifestyle and clinical risk factors 10 – 12 . PGSs are currently incorporated in the comprehensive breast and ovarian cancer risk models implemented in the CanRisk Tool, providing guidance for health care professionals in the patient consultation process 13 , 14 . A PGS is computed by applying a polygenic model (PGM), which consists of a set of variants and their associated weights, to an individual’s genotypes at those variants. Multiple methods have been developed for PGM construction. These differ in their approaches for selecting variants for inclusion in the model and how those variants are weighted. Hard threshold approaches like linkage disequilibrium (LD) clumping and thresholding (CT) select SNPs based on a single p -value threshold. LD information is then accounted for by removing highly correlated SNPs, with weights corresponding to the SNP-specific effect sizes. Other methods employ regularization on a much larger set of variant effect sizes. These allow for multiple correlated SNPs at each locus to be included in the model by accounting for local LD structure between loci. Examples include LDpred2, polygenic risk score-continuous shrinkage (PRS-CS), and normal-mixture models 15 – 17 . These computationally intensive PGM methods have been shown to outperform simpler approaches across a range of settings. Despite the successes of PGMs, several barriers limit their present portability to clinical use. One of the most challenging obstacles is the reduced accuracy of PGMs in non-European (EUR)-origin populations 18 – 20 . A major reason for this is the continuing overrepresentation of European populations in genome-wide association studies (GWAS) to date 21 . According to the GWAS Diversity Monitor, as of February 2024, 95% of GWAS participants were of European ancestry, while only 3.7% and 0.2% of participants were of Asian and African ancestries, respectively 22 . Cross-ancestry PGM prediction performance has been shown to decrease as the genetic distance between training and validation populations grows, resulting in poor generalizability to non-EUR origin populations 23 . The underlying genetic architecture of disease can also vary by population, further amplifying these challenges 24 , 25 . PGS prediction in individuals with African (AFR) ancestries has proven particularly challenging due to the higher degree of genetic diversity in those populations 18 , 26 , 27 . As a result, current PGM methods risk exacerbating existing health disparities by systematically offering more accurate risk stratification for populations of European descent. The equitable clinical application of PGMs depends on the development of methods for generalizing across populations. Multi-ancestry PGM methods that integrate data from across the ancestry continuum and incorporate ancestry-specific local LD structure have shown improved predictive performance over those that only utilize GWAS data within a single group of ancestries, including models developed only on training data matching the target ancestries 28 – 30 . Several PGM methods have been developed in recent years to improve PGS performance in cross-ancestry prediction tasks. PRS-CSx and LDpred2 are two such examples that have demonstrated improved performance in cross-ancestry prediction tasks over common single-ancestry methods 16 , 31 . In this paper, we present a multi-ancestry extension of the “select and shrink with summary statistics” (S4) method, a Bayesian PGM that employs continuous shrinkage priors on effect sizes 32 . S4 extends the regularization properties of PRS-CSx by imposing additional penalization of rarer variants. This has previously been shown to achieve prediction accuracy matching or exceeding other computationally intensive PGM methods at single-ancestry prediction 33 . We compare the cross-ancestry predictive performance of the multi-ancestry S4 (S4-Multi) with other commonly used alternatives across a range of complex traits with different genetic architectures. This comparison uses both simulated genetic data developed for benchmarking 34 and large-scale biobank data from the UK Biobank (UKB), FinnGen, Biobank Japan (BBJ), All of Us (AoU), and Global Biobank (GBB). We replicated the simulation and testing structure used by Zhang et al 2023 34 , a recent and comprehensive comparison of twelve PGS methods on multi-ancestry prediction tasks, adding three S4-based methods for model comparison and benchmarking. Methods S4 Overview The single-ancestry S4 method was previously presented in Dareng et al 2022 32 . We review the main ideas here to highlight the extensions to the multi-ancestry case. Generally, a polygenic model comprises a set of variants associated with the phenotype together with their weights. The PGS for an individual is an application of a polygenic model to the genotype of the individual ( j ) and is the sum of the allele dosage at each variant ( i ) ( x ij ∈ {0,1,2}) weighted by its corresponding effect size ( w i ) over the i variants included in the model. So, the PGS for the j th individual ( PGS j ) is given by the linear function: The PGM weights correspond to the log odds ratios from a case-control study for a disease trait. For a quantitative trait, the weights correspond to the beta coefficients from a linear regression model. The central tasks in any PGM construction consist of: (1) selecting which SNPs to include in the model, and (2) adjusting their weights to impose shrinkage or other desired model properties. In the selection stage of S4 ( Fig. 1 , development stage 1), variants from a GWAS are ranked and selected based on their statistical significance and correlation with previously selected SNPs. S4 partitions SNPs into roughly independent genomic regions or blocks, performing SNP selection separately on each block to reduce computational cost. Top-ranked SNPs are iteratively added if their correlation with all other included SNPs ( r 2 ) is less than 0.85. New SNPs are put into groups with which they are best correlated, with a new group started when a newly added SNP has a correlation below 0.02 with other selected SNPs. The selection procedure terminates when the p -value/ r 2 ratio for candidate SNPs reaches a pre-determined threshold. Download figure Open in new tab Fig. 1: Comparison of pipelines for single and multi-ancestry S4 development, tuning, and validation S4 adjusts SNP weights ( Fig. 1 , development stage 2) in a similar manner to PRS-CSx, employing a Bayesian model construction that places a shared global-local continuous shrinkage prior on effect sizes to impose sparsity 31 . The SNP-specific shrinkage parameter, Ψ j , is assigned a gamma-gamma prior with two additional hyperparameters, α and β , that control the shrinkage of effect sizes around 0 and shrinkage of larger effect sizes, respectively. So, the full hierarchical prior specification is given by: with residual variance σ 2 , sample size N , and a global shrinkage parameter Φ shared across genetic markers. However, S4 extends the PRS-CSx algorithm by further adjusting the global shrinkage parameter Φ by dividing by the standard deviation of the summary statistic. Since effect size estimates for rarer variants will tend to carry a greater variance, this effectively imposes a further shrinkage penalty for rarer variants. Multiple polygenic models are then generated over a grid of values for α , β , and Φ * . Model tuning requires a second, independent data set to reduce over-fitting. Tuning involves evaluating the multiple polygenic models on individual genomic data. The best performing PGM is then selected for model validation and comparison with models developed using other methods. Multi-ancestry Extension The multi-ancestry implementation of S4 makes several adjustments to account for differences in population structure between the training, tuning, and target populations ( Fig. 1 ). At the initial stage, S4-Multi incorporates GWAS summary statistics from multiple ancestries to account for between-population genetic diversity. Ancestry-specific GWAS effect sizes are combined via a fixed-effects meta-analysis to produce SNP-specific weights for subsequent selection and effect-size shrinkage. Population-specific LD patterns are accounted for in the variant-selection stage by computing a weighted average of the correlations from each ancestry-specific GWAS. This is the used for SNP selection in the same manner as the single-ancestry S4 construction. Data and Results We evaluated the prediction accuracy of S4-Multi in two settings. First, we benchmarked S4-Multi against twelve other PGM constructions using simulated genomic data. Then we evaluated S4-Multi against other top-performing PGS methods for predicting a set of complex traits using available GWAS summary statistics and individual genomic data from UK Biobank, FinnGen, Biobank Japan, All of Us , and the Global Biobank Initiative. Simulated data testing We replicated the data simulation and testing structure used by Zhang et al 34 , a recent and comprehensive comparison of twelve PGM methods on multi-ancestry prediction tasks. We added three S4-based methods: (1) single-ancestry S4 generated on EUR-only GWAS data; (2) S4-Multi at a p -value cutoff of 0.02; and (3) S4-Multi at a p -value cutoff of 0.15. Varying the p -value threshold allowed for assessing the impact of including more or fewer SNPs in the PGM, comparing both against the single-ancestry S4. The twelve other PGM methods for benchmarking included two single-ancestry methods (CT 35 , LDpred2 16 ), two EUR-only methods (best EUR CT, best EUR LDpred2), three weighted PRS methods (weighted CT, weighted LDpred2, PolyPred-S+ 36 ), three Bayesian methods (XPASS 37 , PRS-CSx, PRS-CSx [all ancestries] 31 ), and two model-free superlearning-based methods (CT-SLEB, CT-SLEB [all ancestries]) 34 . All methods were run as recommended except for LDpred2, which was applied with minor adjustments that simplified implementation while improving performance. Rather than calculating separate PGSs for each ancestry and then combining results via regression, we estimated LD patterns by weighting SNPs according to their effect sizes in each ancestry before calculating their correlation. Simulated multi-ancestry genotype data sets were generated under a range of genetic architectures, mimicking the LD structure of Americans (AMR), Africans (AFR), East Asians (EAS), and South Asians (SAS) based on a reference panel from the 1000 Genomes Project 38 . Phenotypes were determined by randomly selecting causal SNPs across the genome, with casual SNP proportions set to either 0.01, 0.001, or 5 × 10 ’( to compare performance across polygenicity settings. Different heritability and selection patterns were also compared. Heritability distribution was set to either constant common SNP heritability or constant per-SNP heritability. Selection patterns were set to either strong, mild, or no negative selection pressure. See Table S1 and Zhang et al for additional details on the data simulation procedure 34 . Overall results from simulation studies are presented in Fig. 2 with model performance (prediction R 2 ) averaged over all tested genetic architectures, training sample sizes, and causal SNP proportions at each ancestry (60 total runs per ancestry). Results further stratified by causal SNP proportion and sample size are provided in Figs. S1 - S3 . Multi-ancestry methods consistently outperformed their single-ancestry counterparts, matching previous results 34 , with S4-Multi achieving 162% and 169% increases in accuracy over the EUR-only S4 implementation at p -value cutoffs of 0.02 and 0.15, respectively. Download figure Open in new tab Figure 2: simulation results, averaged over all training sample sizes While all PRS constructions showed lower performance in AFR target ancestries ( Figs. 2 - 3 ), multi-ancestry methods achieved a substantial improvement over single-ancestry and EUR-based constructions in AFR prediction. Moreover, this gap in accuracy decreased as the training sample size grew. At a training sample size of 15,000, prediction R 2 was 54% lower in AFR target sample than the next lowest accuracy ancestry (AFR: 0.036 vs. EAS: 0.078). While this difference dropped to a 36% decrease at a training sample size of 100,000 (AMR: 0.136 vs. AFR: 0.087), a substantial performance gap remained even at the largest tested sample size. Download figure Open in new tab Figure 3: Multi-ancestry methods, stratified by genetic architecture sparsity and ancestry Both tested p -value cutoffs (0.02 and 0.15) for S4-Multi performed comparably overall, with a mean accuracy increase of just 2.5% for the larger model averaged over all simulation runs ( ). This suggests that the specific threshold is generally not a make-or-break decision in implementing S4-Multi in most cases. The performance gain from the higher cutoff was greatest when applied to a highly polygenic setting ( Fig. S4 ), achieving a 10% increase in accuracy averaged over all such runs. In all other polygenicity settings, the performance difference was nearly negligible (< 1%). The maximum difference occurred under fixed common SNP heritability with mild negative selection and a training sample size of 15,000 and causal SNP proportion of 0.01, with the higher threshold model achieving a 16% increase in accuracy ( ). This was the most challenging prediction setting for S4-Multi in general—low degree of polygenicity and small sample size—so it is somewhat intuitive that this is where the additional SNPs in the larger model provided the most benefit. However, this modest overall change in accuracy over the genetic architectures and sample sizes tested gives some indication that S4-Multi retains relatively strong prediction accuracy as the number of available SNPs decreases. Table 1 records the number of simulation runs in which each method was the top performer over the 60 total runs in each target ancestry. These span all tested proportions of causal SNPs, genetic architectures, and sample sizes. S4-Multi and PRS-CSx consistently outperformed all other tested methods overall, with S4-Multi tending to perform comparably in AMR, EAS, and SAS ancestries, and PRS-CSx ranking consistently higher in AFR prediction. S4-Multi improved upon LDpred2 and CT-SLEB across EAS, SAS and EUR validation data irrespective of underlying genetic architecture and training sample size. While PRS-CSx achieved top performance in 122/240 = 51% of runs overall versus 99/240 = 41% for S4-Multi, this difference was primarily driven by prediction in AFR ancestries. View this table: View inline View popup Download powerpoint Table 1: The number of simulation runs (out of all 60 different sample constructions per ancestry) in which each PRS was the top performer is shown. All other tested models had zero runs in which they were the top performer. Fig. 3 highlights that the underlying genetic architecture had little impact on the relative accuracy of PGS conctructions at the smallest tested training sample size ( n = 15,000). At larger training sample sizes ( n = 80,000), S4-Multi showed modest improvement under mild negative selection pressure among AMR and EAS validation sets relative to other multi-ancestry methods (PRS-CSx, LDpred2, CT-SLEB). While there were marginal changes as causal SNP proportion and sample size varied, there was generally limited difference between methods. While the S4-Multi methods showed some improvement under mild negative selection pressure, genetic architecure had limited impact on relative accuracy of PRS constructions in general, with all methods performing slightly better under strong negative selection for a given sample size. The strongest overall trends were the superior performance of multi-ancestry methods over single-ancestry and EUR-only counterparts, and remaining performance gap for prediciton in AFR ancestries, particularly at lower traninng sample sizes. Biobank data testing We evaluated S4-Multi, LDpred2, and PRS-CSx on available biobank data. All models were trained, tuned, and validated on available data from UK Biobank (UKB), FinnGen, Biobank Japan (BBJ), All of Us (AoU), and the Global Biobank Initiative (GBB) to predict five complex traits: type 2 diabetes, breast cancer, colorectal cancer, asthma, and stroke (either ischemic or hemorrhagic). Sample sizes for each phenotype and cohort are available in Table S4 . During PRS-CSx model development, separate models were fit on each available training ancestry before regressing the results on the target ancestry to produce the final PGM for testing.Individuals were grouped based on genetically inferred ancestry computed from principal component (PC) analysis and classification algorithm within each biobank. Implementing the S4-Multi PGM requires three independent datasets for model development, tuning, and validation. Model development was performed on two settings: (1) using a consortium of large-scale GWAS summary statistics from UK Biobank, FinnGen, and Biobank Japan, spanning all available ancestries (AFR, EAS, EUR, and SAS), and (2) using EUR data only. Due to the limited sample size for colorectal cancer phenotype, model development data were available only for EAS and EUR ancestries. Model tuning was performed on available EUR genotype data, with UKB and FinnGen runs separated to control for differences in cohort population structure. Model validation was conducted on AFR, AMR, EAS, and EUR ancestries using data from available biobanks (see Table 3 ). Case and control counts for the validation sets of each phenotype and biobank source are available in the supplementary materials ( Table S3 ). Asthma and stroke development and testing was performed on a consortium of biobank data from the Global Biobank Initiative 39 . Full results for all runs on biobank data are presented in Table 2 . Model performance was measured with the log odds ratio per standard deviation of PGS (log(OR) per SD). χ 2 statistics from the corresponding likelihood ratio tests as well as results for models trainined only on EUR ancestry data are provided in the summplementary materials ( Tables S2 - S3 ). In subsequent discussion, we consider log(OR) per SD as the performance metric. Multi-ancestry methods trained on all available development ancestries showed the strongest performance in type 2 diabetes prediction across all target ancestries and PGM methods (mean = 0.570), with all results falling in the interval [0.254, 0.948]. Prediction in breast cancer (mean = 0.452, interval = [0.269, 0.623]), asthma (mean = 0.369, interval = [0.146, 0.530]), colorectal cancer (mean = 0.344, interval = [0.136, 0.551]), and stroke (mean = 0.090, interval = [-0.175, 0.204]) was more challenging with a substantial dropoff in overall performance for the latter. View this table: View inline View popup Table 2: Full-ancestry data used in model development consist of a consortium of available data from UK Biobank (UKB), FinnGen, and Biobank Japan (BBJ). The top performing PGM is highlighted for each combination of development and test set. Models for asthma and stroke were developed on full-ancestry data from Global Biobank (GBB). Results for models developed only on EUR data are available in the supplementary materials ( Table S2 ). Mirroring the simulation results, all methods tested showed improved performance when trained on all available ancestries versus only utilizing EUR training data ( Table S2 ). Multi-ancestry PGMs improved performance by 13% on average (in log(OR) / SD) relative to their single-ancestry counterparts across all methods, phenotypes, and ancestries. The overall performance gains for the multi-ancestry methods were greatest for type 2 diabetes and breast cancer prediction (0.085 and 0.032, respectively), with more modest increases in colorectal cancer (0.024). EUR-only models were not trained for asthma and stroke. Overall performance gains were largest in EAS (28% greater) and AFR (10% greater) prediction, though there were still modest improvements in AMR (5.6% greater) and EUR (3.4% greater) origin populations. Across all three phenotypes, we again found substantially reduced accuracy in AFR ancestries. In both breast cancer and type 2 diabetes prediction, the worst-performing EUR-based PGM in non-AFR ancestries outperformed the best-performing multi-ancestry PGM in AFR ancestries. This highlights the lingering challenges of PGS prediction in AFR-origin populations despite these improvements. The best-performing PGM method varied considerably both with target ancestry and phenotype ( Fig. 4 ). For example, while PRS-CSx was the strongest predictor of type 2 diabetes in EAS individuals, it was the least accurate of the three tested models for predicting breast cancer in the EAS target population. However, with some exceptions such as S4’s strong performance predicting breast cancer in EAS populations, the three multi-ancestry methods tended to achieve similar results in cases where all models were tested. We found similar results in cases when only S4 and LDpred2 were tested, apart from S4’s strong performance predicting type 2 diabetes in AMR-origin populations. Within each phenotype-ancestry-cohort combination, the worst-performing PGM was only 13% less accurate (in log(OR) per SD) than the top-performing model averaged over type 2 diabetes, breast cancer, colorectal cancer, and asthma tests. Performance in stroke prediction was more variable with a mean 53% decrease between the top-performing and worst-performing PGM. However, all methods suffered from considerably lower accuracy for stroke relative to other phenotypes. Among the non-stroke diseases tested, the maximum difference occurred in asthma prediction in AMR-origin individuals from the AoU cohort, where PRS-CSx was 62% less accurate than the top-preforming model (S4). Download figure Open in new tab Figure 4: The best-performing PGM method varied substantially with both phenotype and ancestry. Model performance was averaged over all biobank runs for a given phenotype-ancestry combination. Note that PRS-CSx was not fit on all phenotype-ancestry combinations. All tested models showed diminished accuracy predicting stroke relative to other phenotypes across all tested ancestries. Within each tested group of ancestries, stroke prediction was on average 84% less accurate than the next lowest phenotype. Notably, all three methods produced negative log(OR) per SD results in the tested AMR-origin population. This implies that both models produced lower scores on average in controls than in cases. We give these results particular consideration here. Using genetic principal components (PCs) computed from a set of LD-pruned ( r 2 < 0.1) autosomal SNPs (provided by All of Us ), we examined the correlation between the PGSs and the first four PCs. We find that the tested PGSs in AMR-origin individuals are weakly correlated with each of the first four PCs ( Table S5 , Fig. S5 ), and that cases and controls are distributed differently along those PCs. This suggest that differences in the population structure between cases and controls may confound the association between PGS and phenotype. Adjusting for the first 16 PCs in the logistic regression model returns positive log(OR) per SD estimates for tested PGSs, suggesting that population structure is a likely confounder here. Similar investigation into type 2 diabetes results also reveals correlations and distributional differences between AMR-origin cases and controls along genetic PCs, albeit to a lesser degree ( Fig. S6 ). However, since all PGM methods were more accurate in general at predicting type 2 diabetes, the tested models nonetheless produce a positive result (in log(OR) per SD) for AMR prediction. We present the results from the unadjusted models in Table 2 and Fig. 4 to maintain consistency across simulation test, biobank tests, and replicated analyses, though the potential for population structure confounding in highly admixed populations should be noted and likely warrants specific consideration for future work. Discussion In this paper, we presented S4-Multi, an extension of the S4 PGM that leverages data across multiple ancestries for improved cross-ancestry prediction. We compared its performance with a range of single- and multi-ancestry PGM methods for prediction across the ancestry continuum in both simulated and biobank data. We find that S4-Multi makes significant performance gains in cross-ancestry prediction compared to its single-ancestry implementation. Moreover, S4-Multi achieves state-of-the-art performance both in simulated data and large-scale biobank data across type 2 diabetes, breast cancer, and colorectal cancer prediction tasks. Simulation results indicate that S4’s retains strong performance at smaller sample sizes relative to alternative multi-ancestry PGS constructions. That S4-Multi achieved top or near-top performance in both the simulation benchmarking and real-world biobank prediction tests is encouraging for its robustness across a range of settings, phenotypes, and sample characteristics. We found limited differences in performance between the top multi-ancestry PGS methods in general, with the best-performing model being highly dependent on phenotype, target ancestry, and genetic architecture. This aligns with previous findings that no PGS method is uniformly superior in all contexts 34 . Our biobank tests lend credence to this in real world settings, with relative performance between PGS methods varying considerably between phenotypes. This suggests that disease-specific and ancestry-specific genetic architecture plays a role in differentiating PGS performance, matching prior work 40 . Evaluating multiple PGM constructions is likely prudent for clinical use to identify the best-performing model for specific prediction tasks and target populations. We see evidence that differences in underlying population structure within classified ancestries may confound the underlying association between the PGS and phenotype, as indicated by the discussion of AMR stroke prediction above. This is consistent with prior work finding PGSs for a range of traits to correlate with geographic population distribution, even in cases with relatively homogenous populations 41 . This trend being most pronounced in AMR target populations suggests that prediction in highly admixed populations likely warrants particular attention moving forward, as well as supporting the need to consider PGS performance along the continuum of genetic ancestries rather than simply in discrete ancestry groups 19 , 42 . A central limitation in model development and testing remains the smaller available sample sizes from AFR, AMR, and SAS origin populations in GWAS data. While projects aiming to diversify participation such as All of Us enabled model testing on observed data across the ancestry continuum, training and validation sample sizes remain limited for non-EUR origin populations. This was particularly challenging for lower-prevalence diseases like colorectal cancer, where sufficient training data was only available for training and development on EUR- and EAS-origin populations in our tests. Continuing to diversify GWAS participation is critical for multi-ancestry PGM development and evaluation moving forward. Given the current magnitude of the EUR overrepresentation in GWAS and limited available training data, strong performance at lower training sample sizes is likely to be a meaningful differentiator of PGMs for practical use. This is especially true for low-prevalence disease prediction, which exacerbates sample size issues for underrepresented populations in GWAS. Both simulation and biobank results suggest that while multi-ancestry methods make meaningful inroads for risk stratification in AFR target ancestries, this remains a challenging setting for PGS prediction. While increasing training sample size in simulation tests reduced this accuracy drop in AFR target ancestries to a degree, a meaningful performance gap remained at the largest tested training sample size ( n = 100,000). This suggests that while continuing to diversify GWAS participation will be able to attenuate this somewhat, further model refinement for multi-ancestry methods is likely necessary to match prediction accuracy in EUR-origin populations. A further benefit of S4 was its strong performance despite including fewer variants in the final model across all five tested phenotypes in biobank tests ( Fig. 5 , Table S6 ). Relative to the next sparsest PGM, S4 used from 9% fewer (asthma) to 77% fewer (colorectal cancer) variants. This is an encouraging indication that S4 retains good prediction accuracy with fewer available SNPs, though this warrants further investigation verifying the robustness of PGM methods in clinical settings where available genomic data are in the hundreds or thousands of variants are available rather than the millions considered here. Download figure Open in new tab Figure 5: S4 achieves state-of-the-art performance with fewer variants in final PGM, suggesting potential for robust performance in settings with fewer available variants In conclusion, we have presented a new multi-ancestry PGS construction generated from GWAS and genotype data from diverse populations. S4-Multi achieves strong performance across a range of phenotypes for improved cross-ancestry prediction. Data Availability All data used in this analysis is publicly available online. The simulated data is identical to that used in Zhang et al 2023 and is publicly available on GitHub (https://github.com/andrewhaoyu/multi_ethnic). Biobank tests utilized both summary statistics for model development and individual-level genomic data for testing. Individual-level genomic data used in this analysis is publicly accessible after registration with the following biobanks: All of Us (v7 release) (https://www.researchallofus.org/), UK Biobank (https://www.ukbiobank.ac.uk/), FinnGen (https://www.finngen.fi/en), Biobank Japan (https://biobankjp.org/en/), and the Global Biobank Meta-analysis Initiative (https://www.globalbiobankmeta.org/). GWAS summary statistics by phenotype are available here: breast cancer (http://bcac.ccge.medschl.cam.ac.uk), colorectal cancer (https://www.ebi.ac.uk/gwas/studies/GCST90255677, https://www.ebi.ac.uk/gwas/publications/36539618, https://www.ebi.ac.uk/gwas/studies/GCST90129505), type 2 diabetes (https://diagram-consortium.org/downloads.html, https://t2d.hugeamp.org), asthma (https://gbmi-sumstats.s3.amazonaws.com/Asthma_Bothsex_afr_inv_var_meta_GBMI_052021_nbbkgt1.txt.gz, https://gbmi-sumstats.s3.amazonaws.com/Asthma_Bothsex_amr_inv_var_meta_GBMI_052021_nbbkgt1.txt.gz, https://gbmi-sumstats.s3.amazonaws.com/Asthma_Bothsex_eas_inv_var_meta_GBMI_052021_nbbkgt1.txt.gz, https://gbmi-sumstats.s3.amazonaws.com/Asthma_Bothsex_eur_inv_var_meta_GBMI_052021_nbbkgt1.txt.gz, https://gbmi-sumstats.s3.amazonaws.com/Asthma_Bothsex_sas_inv_var_meta_GBMI_052021_nbbkgt1.txt.gz, https://gbmi-sumstats.s3.amazonaws.com/Asthma_Bothsex_leave_UKBB_inv_var_meta_GBMI_052021.txt.gz), stroke (https://gbmi-sumstats.s3.amazonaws.com/Stroke_Bothsex_afr_inv_var_meta_GBMI_052021_nbbkgt1.txt.gz, https://gbmi-sumstats.s3.amazonaws.com/Stroke_Bothsex_amr_inv_var_meta_GBMI_052021_nbbkgt1.txt.gz, https://gbmi-sumstats.s3.amazonaws.com/Stroke_Bothsex_amr_inv_var_meta_GBMI_052021_nbbkgt1.txt.gz, https://gbmi-sumstats.s3.amazonaws.com/Stroke_Bothsex_eur_inv_var_meta_GBMI_052021_nbbkgt1.txt.gz, https://gbmi-sumstats.s3.amazonaws.com/Stroke_Bothsex_leave_UKBB_inv_var_meta_GBMI_052021.txt.gz) https://github.com/andrewhaoyu/multi_ethnic https://www.researchallofus.org/ https://www.ukbiobank.ac.uk/ https://www.finngen.fi/en https://biobankjp.org/en/ https://www.globalbiobankmeta.org/ http://bcac.ccge.medschl.cam.ac.uk https://www.ebi.ac.uk/gwas/studies/GCST90255677 https://www.ebi.ac.uk/gwas/publications/36539618 https://www.ebi.ac.uk/gwas/studies/GCST90129505 Competing interests The authors declare no competing interests. Data availability The simulated data used in this analysis is identical to that used in Zhang et al 2023 34 and is publicly available on GitHub. Biobank tests utilized both summary statistics for model development and individual-level genomic data for testing. Sources for summary statistics are provided in the table below. Individual-level genomic data used in this analysis is publicly accessible after registration with the following biobanks: All of Us (v7 release) ( https://www.researchallofus.org/ ), UK Biobank ( https://www.ukbiobank.ac.uk/ ), FinnGen ( https://www.finngen.fi/en ), Biobank Japan ( https://biobankjp.org/en/ ), and the Global Biobank Meta-analysis Initiative ( https://www.globalbiobankmeta.org/ ). View this table: View inline View popup Supplementary Materials Download figure Open in new tab Figure S1: simulation results, stratified by target ancestries and causal SNP proportion Download figure Open in new tab Figure S2: Simulated data prediction accuracy stratified by ancestries and averaged over all runs at each sample size-ancestry combination Download figure Open in new tab Figure S3: Simulation results on only AFR-ancestries prediction, stratified by causal SNP proportion Download figure Open in new tab Figure S4: Comparing performance of different p-value cutoffs for S4-Multi across genetic architectures and sample sizes Download figure Open in new tab Figure S5: AMR stroke cases and controls differ in their distribution along PCs 1-4. In all four cases, kernel density estimates (B) show that more cases are found at PC levels that correlate with higher S4 PGS (A). LDpred2 produces similar results and is not plotted. Pearson correlation between the PGS and each principal component (cor) is shown in red. Download figure Open in new tab Figure S6: AMR type-2 diabetes (T2D) cases and controls also differ in their distributions along PCs 1-4 though to a lesser degree than stroke cases and controls. PCs 1-4 are weakly correlated with the S4 PGS (A). Similarly to stroke, population structure differences between cases and controls (B) push the PGSs in controls higher on average. Pearson correlation between the PGS and each principal component (cor) is shown in red. View this table: View inline View popup Download powerpoint Table S1: Breakdown of the 60 simulation runs per ancestry consisting of all 5 × 3 × 4 = 60 combinations of the three factors below View this table: View inline View popup Download powerpoint Table S2: Performance of models trained on EUR-only data in log(OR) per SD, following the same format as Table 3 in the main text. The top performing method in each row in highlighted in yellow. Note that an EUR-only version of PRS-CSx was not fit, nor were EUR-only models for asthma and stroke. View this table: View inline View popup Table S3: All indicated ancestry data sets used in model development consist of a consortium of available data from UK Biobank (UKB), FinnGen, and Biobank Japan (BBJ). View this table: View inline View popup Download powerpoint Table S4i: Case counts used for model validation in biobank tests by data source View this table: View inline View popup Download powerpoint Table S4ii: Case counts used for model validation in biobank tests stratified by predicted ancestry View this table: View inline View popup Download powerpoint Table S5: Both S4 and LDpred2 PGSs are weakly correlated (Pearson correlation between 0.20 and 0.40 in magnitude) with the first four principal components (PCs) in AMR target populations. In no other target ancestry is the correlation between PCs 1-5 and either PGS that exceeds 0.14 in magnitude. View this table: View inline View popup Download powerpoint Tables S6i-v: Number of variants in each PGM in biobank tests* Details on hyperparameter tuning and model fitting are available in the Supplementary Materials. Acknowledgements We thank All of Us participants for their contributions. We also thank the All of Us Research Program for making available the participant data examined in this study. Footnotes ↵ * Details on hyperparameter tuning and model fi4ng are available in the Supplementary Materials. References 1. ↵ Visscher , P. M. , Yengo , L. , Cox , N. J. & Wray , N. R . Discovery and implications of polygenicity of common diseases . Science 373 , 1468 – 1473 ( 2021 ). OpenUrl CrossRef PubMed 2. ↵ Zhang , Y. D. et al. Assessment of polygenic architecture and risk prediction based on common variants across fourteen cancers . Nat. Commun . 11 , 3353 ( 2020 ). OpenUrl CrossRef PubMed 3. ↵ Chatterjee , N. , Shi , J. & García-Closas , M . Developing and evaluating polygenic risk prediction models for stratified disease prevention . Nat. Rev. Genet . 17 , 392 – 406 ( 2016 ). OpenUrl CrossRef PubMed 4. Knowles , J. W. & Ashley , E. A . Cardiovascular disease: The rise of the genetic risk score . PLOS Med . 15 , e1002546 ( 2018 ). OpenUrl CrossRef PubMed 5. Khera , A. V. et al. Genome-wide polygenic scores for common diseases identify individuals with risk equivalent to monogenic mutations . Nat. Genet . 50 , 1219 – 1224 ( 2018 ). OpenUrl CrossRef PubMed 6. ↵ Baliakas , P. et al. Integrating a Polygenic Risk Score into a clinical setting would impact risk predictions in familial breast cancer . J. Med. Genet . 61 , 150 – 154 ( 2024 ). OpenUrl Abstract / FREE Full Text 7. ↵ Mavaddat , N. et al. Polygenic Risk Scores for Prediction of Breast Cancer and Breast Cancer Subtypes . Am. J. Hum. Genet . 104 , 21 – 34 ( 2019 ). OpenUrl CrossRef PubMed 8. Khera , A. V. et al. Polygenic Prediction of Weight and Obesity Trajectories from Birth to Adulthood . Cell 177 , 587 – 596.e9 ( 2019 ). OpenUrl CrossRef PubMed 9. ↵ Moll , M. et al. Chronic obstructive pulmonary disease and related phenotypes: polygenic risk scores in population-based and case-control cohorts . Lancet Respir. Med . 8 , 696 – 708 ( 2020 ). OpenUrl PubMed 10. ↵ Kullo , I. J. et al. Incorporating a Genetic Risk Score Into Coronary Heart Disease Risk Estimates: Effect on Low-Density Lipoprotein Cholesterol Levels (the MI-GENES Clinical Trial) . Circulation 133 , 1181 – 1188 ( 2016 ). OpenUrl Abstract / FREE Full Text 11. Paquette , M. et al. Polygenic risk score predicts prevalence of cardiovascular disease in patients with familial hypercholesterolemia . J. Clin. Lipidol . 11 , 725 – 732.e5 ( 2017 ). OpenUrl PubMed 12. ↵ Lakeman , I. M. M. et al. The predictive ability of the 313 variant–based polygenic risk score for contralateral breast cancer risk prediction in women of European ancestry with a heterozygous BRCA1 or BRCA2 pathogenic variant . Genet. Med . 23 , 1726 – 1737 ( 2021 ). OpenUrl CrossRef PubMed 13. ↵ Lee , A. et al. BOADICEA: a comprehensive breast cancer risk prediction model incorporating genetic and nongenetic risk factors . Genet. Med . 21 , 1708 – 1718 ( 2019 ). OpenUrl CrossRef PubMed 14. ↵ Carver , T. et al. CanRisk Tool—A Web Interface for the Prediction of Breast and Ovarian Cancer Risk and the Likelihood of Carrying Genetic Pathogenic Variants . Cancer Epidemiol. Biomarkers Prev . 30 , 469 – 473 ( 2021 ). OpenUrl Abstract / FREE Full Text 15. ↵ Ge , T. , Chen , C.-Y. , Ni , Y. , Feng , Y.-C. A. & Smoller , J. W . Polygenic prediction via Bayesian regression and continuous shrinkage priors . Nat. Commun . 10 , 1776 ( 2019 ). OpenUrl CrossRef PubMed 16. ↵ Privé , F. , Arbel , J. & Vilhjálmsson , B. J . LDpred2: better, faster, stronger . Bioinformatics 36 , 5424 – 5431 ( 2021 ). OpenUrl CrossRef PubMed 17. ↵ Zhou , X. , Carbonetto , P. & Stephens , M . Polygenic Modeling with Bayesian Sparse Linear Mixed Models . PLoS Genet . 9 , e1003264 ( 2013 ). OpenUrl CrossRef PubMed 18. ↵ Martin , A. R. et al. Human Demographic History Impacts Genetic Risk Prediction across Diverse Populations . Am. J. Hum. Genet . 100 , 635 – 649 ( 2017 ). OpenUrl CrossRef PubMed 19. ↵ Ding , Y. et al. Polygenic scoring accuracy varies across the genetic ancestry continuum . Nature 618 , 774 – 781 ( 2023 ). OpenUrl CrossRef PubMed 20. ↵ Du , Z. et al. Evaluating Polygenic Risk Scores for Breast Cancer in Women of African Ancestry . JNCI J. Natl. Cancer Inst . 113 , 1168 – 1176 ( 2021 ). OpenUrl CrossRef PubMed 21. ↵ Popejoy , A. B. & Fullerton , S. M . Genomics is failing on diversity . Nature 538 , 161 – 164 ( 2016 ). OpenUrl CrossRef PubMed 22. ↵ Mills , M. C. & Rahal , C . The GWAS Diversity Monitor tracks diversity by disease in real time . Nat. Genet . 52 , 242 – 243 ( 2020 ). OpenUrl CrossRef PubMed 23. ↵ Martin , A. R. et al. Clinical use of current polygenic risk scores may exacerbate health disparities . Nat. Genet . 51 , 584 – 591 ( 2019 ). OpenUrl CrossRef PubMed 24. ↵ McClellan , J. & King , M.-C . Genetic Heterogeneity in Human Disease . Cell 141 , 210 – 217 ( 2010 ). OpenUrl CrossRef PubMed Web of Science 25. ↵ Corona , E. et al. Analysis of the Genetic Basis of Disease in the Context of Worldwide Human Relationships and Migration . PLoS Genet . 9 , e1003447 ( 2013 ). OpenUrl CrossRef PubMed 26. ↵ Campbell , M. C. & Tishkoff , S. A . The Evolution of Human Genetic and Phenotypic Variation in Africa . Curr. Biol . 20 , R166 – R173 ( 2010 ). OpenUrl CrossRef PubMed Web of Science 27. ↵ Fatumo , S. et al. Polygenic risk scores for disease risk prediction in Africa: current challenges and future directions . Genome Med . 15 , 87 ( 2023 ). 28. ↵ Márquez-Luna , C. , Loh , P ., South Asian Type 2 Diabetes (SAT2D) Consortium, The SIGMA Type 2 Diabetes Consortium & Price, A. L. Multiethnic polygenic risk scores improve risk prediction in diverse populations . Genet. Epidemiol . 41 , 811 – 823 ( 2017 ). OpenUrl CrossRef PubMed 29. Coram , M. A. , Fang , H. , Candille , S. I. , Assimes , T. L. & Tang , H . Leveraging Multi-ethnic Evidence for Risk Assessment of Quantitative Traits in Minority Populations . Am. J. Hum. Genet . 101 , 638 ( 2017 ). OpenUrl CrossRef 30. ↵ Gunn , S. et al. Comparison of Methods for Building Polygenic Scores for Diverse Populations . Hum. Genet. Genomics Adv . 100355 ( 2024 ) doi: 10.1016/j.xhgg.2024.100355 . OpenUrl CrossRef 31. ↵ Ruan , Y. et al. Improving polygenic prediction in ancestrally diverse populations . Nat. Genet . 54 , 573 – 580 ( 2022 ). OpenUrl CrossRef PubMed 32. ↵ Dareng , E. O. et al. Polygenic risk modeling for prediction of epithelial ovarian cancer risk . Eur. J. Hum. Genet . 30 , 349 – 362 ( 2022 ). OpenUrl CrossRef PubMed 33. ↵ Tyrer , J. P. et al. Improving on polygenic scores across complex traits using select and shrink with summary statistics (S4) and LDpred2 . BMC Genomics 25 , 878 ( 2024 ). OpenUrl PubMed 34. ↵ Zhang , H. et al. A new method for multiancestry polygenic prediction improves performance across diverse populations . Nat. Genet . 55 , 1757 – 1768 ( 2023 ). OpenUrl CrossRef PubMed 35. ↵ The International Schizophrenia Consortium . Common polygenic variation contributes to risk of schizophrenia and bipolar disorder . Nature 460 , 748 – 752 ( 2009 ). OpenUrl CrossRef PubMed Web of Science 36. ↵ Weissbrod , O. et al. Leveraging fine-mapping and multipopulation training data to improve cross-population polygenic risk scores . Nat. Genet . 54 , 450 – 458 ( 2022 ). OpenUrl CrossRef PubMed 37. ↵ Cai , M. et al. A unified framework for cross-population trait prediction by leveraging the genetic correlation of polygenic traits . Am. J. Hum. Genet . 108 , 632 – 655 ( 2021 ). OpenUrl CrossRef PubMed 38. ↵ The 1000 Genomes Project Consortium et al. A global reference for human genetic variation . Nature 526 , 68 – 74 ( 2015 ). OpenUrl CrossRef PubMed 39. ↵ Zhou , W. et al. Global Biobank Meta-analysis Initiative: Powering genetic discovery across human disease . Cell Genomics 2 , 100192 ( 2022 ). OpenUrl PubMed 40. ↵ Wang , Y. et al. Polygenic prediction across populations is influenced by ancestry, genetic architecture, and methodology . Cell Genomics 3 , 100408 ( 2023 ). OpenUrl PubMed 41. ↵ Kerminen , S. et al. Geographic Variation and Bias in the Polygenic Scores of Complex Diseases and Traits in Finland . Am. J. Hum. Genet . 104 , 1169 – 1181 ( 2019 ). OpenUrl CrossRef PubMed 42. ↵ Lewis , A. C. F. et al. Getting genetic ancestry right for science and society . Science 376 , 250 – 252 ( 2022 ). OpenUrl CrossRef PubMed 43. Michailidou , K. et al. Association analysis identifies 65 new breast cancer risk loci . Nature 551 , 92 – 94 ( 2017 ). OpenUrl CrossRef PubMed 44. Mahajan , A. et al. Multi-ancestry genetic study of type 2 diabetes highlights the power of diverse populations for discovery and translation . Nat. Genet . 54 , 560 – 572 ( 2022 ). OpenUrl CrossRef PubMed View the discussion thread. Back to top Previous Next Posted January 27, 2025. Download PDF Data/Code Email Thank you for your interest in spreading the word about medRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following S4-Multi: enhancing polygenic score prediction in ancestrally diverse populations Message Subject (Your Name) has forwarded a page to you from medRxiv Message Body (Your Name) thought you would like to see this page from the medRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share S4-Multi: enhancing polygenic score prediction in ancestrally diverse populations John Baierl , Jonathan P. Tyrer , Ping-Hung Lai , Simon A. Gayther , Yi-Wen Hsiao , Michelle Jones , Pei-Chen Peng , Paul D. P. Pharoah medRxiv 2025.01.24.25321098; doi: https://doi.org/10.1101/2025.01.24.25321098 Share This Article: Copy Citation Tools S4-Multi: enhancing polygenic score prediction in ancestrally diverse populations John Baierl , Jonathan P. Tyrer , Ping-Hung Lai , Simon A. Gayther , Yi-Wen Hsiao , Michelle Jones , Pei-Chen Peng , Paul D. P. Pharoah medRxiv 2025.01.24.25321098; doi: https://doi.org/10.1101/2025.01.24.25321098 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Genetic and Genomic Medicine Subject Areas All Articles Addiction Medicine (568) Allergy and Immunology (863) Anesthesia (297) Cardiovascular Medicine (4421) Dentistry and Oral Medicine (443) Dermatology (382) Emergency Medicine (606) Endocrinology (including Diabetes Mellitus and Metabolic Disease) (1507) Epidemiology (15212) Forensic Medicine (30) Gastroenterology (1121) Genetic and Genomic Medicine (6581) Geriatric Medicine (667) Health Economics (996) Health Informatics (4520) Health Policy (1366) Health Systems and Quality Improvement (1611) Hematology (539) HIV/AIDS (1264) Infectious Diseases (except HIV/AIDS) (15906) Intensive Care and Critical Care Medicine (1103) Medical Education (620) Medical Ethics (144) Nephrology (667) Neurology (6580) Nursing (345) Nutrition (998) Obstetrics and Gynecology (1141) Occupational and Environmental Health (956) Oncology (3324) Ophthalmology (970) Orthopedics (369) Otolaryngology (420) Pain Medicine (435) Palliative Medicine (129) Pathology (663) Pediatrics (1689) Pharmacology and Therapeutics (691) Primary Care Research (710) Psychiatry and Clinical Psychology (5432) Public and Global Health (9212) Radiology and Imaging (2193) Rehabilitation Medicine and Physical Therapy (1368) Respiratory Medicine (1194) Rheumatology (593) Sexual and Reproductive Health (709) Sports Medicine (529) Surgery (709) Toxicology (99) Transplantation (288) Urology (265) (function(){function c(){var b=a.contentDocument||a.contentWindow.document;if(b){var d=b.createElement('script');d.innerHTML="window.__CF$cv$params={r:'9ff48b5208c106cf',t:'MTc3OTM3NjYwNw=='};var a=document.createElement('script');a.src='/cdn-cgi/challenge-platform/scripts/jsd/main.js';document.getElementsByTagName('head')[0].appendChild(a);";b.getElementsByTagName('head')[0].appendChild(d)}}if(document.body){var a=document.createElement('iframe');a.height=1;a.width=1;a.style.position='absolute';a.style.top=0;a.style.left=0;a.style.border='none';a.style.visibility='hidden';document.body.appendChild(a);if('loading'!==document.readyState)c();else if(window.addEventListener)document.addEventListener('DOMContentLoaded',c);else{var e=document.onreadystatechange||function(){};document.onreadystatechange=function(b){e(b);'loading'!==document.readyState&&(document.onreadystatechange=e,c())}}}})();

Text is read by the "Ask this paper" AI Q&A widget below. Extraction quality varies by source — PMC NXML preserves structure cleanly, OA-HTML may include some navigation residue, and OA-PDF can have broken hyphenation. The publisher copy (via DOI) is the canonical version.

My notes (saved in your browser only)

Ask this paper AI returns verbatim quotes from the full text · source: preprint-html

Answers must be backed by verbatim quotes from this paper's full text. Hallucinated quotes are dropped automatically; if no verbatim passage answers the question, we say so. How this works

Citation neighborhood (no data yet)

We don't have any in-corpus citations linked to this paper yet. This is a recent paper (2025) — citers typically take a year or two to land, and the OpenAlex reference graph may still be filling in.

Source provenance

europepmc
last seen: 2026-05-20T01:45:00.602351+00:00
unpaywall
last seen: 2026-05-26T02:00:01.498150+00:00
License: CC-BY-4.0