Full text
47,078 characters
· extracted from
preprint-html
· click to expand
The impact of ancestry on performance of type 1 diabetes genetic risk scores: high discrimination performance is maintained in African ancestry populations, but population specific thresholds may improve risk prediction | medRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-P4HH5NV'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search The impact of ancestry on performance of type 1 diabetes genetic risk scores: high discrimination performance is maintained in African ancestry populations, but population specific thresholds may improve risk prediction View ORCID Profile Steven Squires , Jean Claude Katte , Dana Dabelea , Catherine Pihoker , Jasmin Divers , View ORCID Profile Eugene Sobngwi , Moffat J. Nyirenda , Raymond J. Kreienkamp , Angela D Liese , Amy S Shah , Lawrence Dolan , Kristi Reynolds , Maria J. Redondo , William Hagopian , Segun Fatumo , Mesmin Y. Dehayem , View ORCID Profile Andrew Hattersley , Michael N. Weedon , Angus Jones , Richard A. Oram doi: https://doi.org/10.1101/2025.07.17.25330632 Steven Squires 1 Clinical and Biomedical Sciences, Faculty of Health and Life Sciences, University of Exeter PhD Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Steven Squires Jean Claude Katte 1 Clinical and Biomedical Sciences, Faculty of Health and Life Sciences, University of Exeter PhD Find this author on Google Scholar Find this author on PubMed Search for this author on this site Dana Dabelea 2 Lifecourse Epidemiology of Adiposity and Diabetes (LEAD) Center Colorado School of Public Health, University of Colorado Anschutz Medical Campus MD, PhD Find this author on Google Scholar Find this author on PubMed Search for this author on this site Catherine Pihoker 3 University of Washington , Seattle, WA, USA MD Find this author on Google Scholar Find this author on PubMed Search for this author on this site Jasmin Divers 4 Department of Foundations of Medicine, NYU Grossman Long Island School of Medicine PhD Find this author on Google Scholar Find this author on PubMed Search for this author on this site Eugene Sobngwi 5 Department of Internal Medicine and Specialities, Faculty of Medicine and Biomedical Sciences, University of Yaoundé 1 , Yaoundé, Cameroon PhD Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Eugene Sobngwi Moffat J. Nyirenda 6 Non-Communicable Diseases Theme, Medical Research Council/Uganda Virus Research Institute and London School of Hygiene and Tropical Medicine Uganda Research Unit , Entebbe, Uganda PhD Find this author on Google Scholar Find this author on PubMed Search for this author on this site Raymond J. Kreienkamp 7 Boston Children’s Hospital, Harvard Medical School , Boston, MA, USA MD, PhD Find this author on Google Scholar Find this author on PubMed Search for this author on this site Angela D Liese 8 Arnold School of Public Health PhD Find this author on Google Scholar Find this author on PubMed Search for this author on this site Amy S Shah 9 Cincinnati Children’s Hospital Medical Center and The University of Cincinnati , Cincinnati OH, USA MD MS Find this author on Google Scholar Find this author on PubMed Search for this author on this site Lawrence Dolan 9 Cincinnati Children’s Hospital Medical Center and The University of Cincinnati , Cincinnati OH, USA MD Find this author on Google Scholar Find this author on PubMed Search for this author on this site Kristi Reynolds 10 Department of Research & Evaluation , Kaiser Permanente Southern California, Pasadena, CA PhD Find this author on Google Scholar Find this author on PubMed Search for this author on this site Maria J. Redondo 11 Texas Children’s Hospital, Baylor College of Medicine , Houston, Texas, USA MD, PhD Find this author on Google Scholar Find this author on PubMed Search for this author on this site William Hagopian 12 Pacific Northwest Diabetes Research Institute, University of Washington PhD Find this author on Google Scholar Find this author on PubMed Search for this author on this site Segun Fatumo 13 Precision Healthcare University Research Institute (PHURI), Queen Mary University of London , London, UK PhD Find this author on Google Scholar Find this author on PubMed Search for this author on this site Mesmin Y. Dehayem 14 National Obesity Centre and The Endocrinology and Metabolic Diseases Unit, Yaoundé Central Hospital , Yaoundé, Cameroon PhD Find this author on Google Scholar Find this author on PubMed Search for this author on this site Andrew Hattersley 1 Clinical and Biomedical Sciences, Faculty of Health and Life Sciences, University of Exeter PhD Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Andrew Hattersley Michael N. Weedon 1 Clinical and Biomedical Sciences, Faculty of Health and Life Sciences, University of Exeter PhD Find this author on Google Scholar Find this author on PubMed Search for this author on this site Angus Jones 1 Clinical and Biomedical Sciences, Faculty of Health and Life Sciences, University of Exeter PhD Find this author on Google Scholar Find this author on PubMed Search for this author on this site Richard A. Oram 1 Clinical and Biomedical Sciences, Faculty of Health and Life Sciences, University of Exeter PhD Find this author on Google Scholar Find this author on PubMed Search for this author on this site For correspondence: r.oram{at}exeter.ac.uk Abstract Full Text Info/History Metrics Supplementary material Data/Code Preview PDF Abstract OBJECTIVE Genetic risk scores (GRSs) for type 1 diabetes (T1D) may assist T1D classification and prediction but are often developed from European populations. To improve health outcomes, it is important to understand the performance and utility of GRSs in diverse ancestry populations. RESEARCH DESIGN AND METHODS We assessed performance of three previously published T1D GRSs in differentiating people with and without Type 1 diabetes in African (with/without T1D=194/235), European (n=1109/125), and Hispanic (266/170) ancestry populations in the USA, and from Cameroon and Uganda (n=144/5001). The assessed GRSs were developed from European ancestry populations (GRS1, GRS2) and from an African ancestry population (AAGRS). RESULTS The discriminative power, as measured by the area under the receiver operating characteristic curve (AUC), for GRS2 and AAGRS were equivalent on the African ancestry populations, and both outperformed the GRS1: the AUCs produced by the GRS2, AAGRS and GRS1 on Uganda/Cameroon data were 0.882 (0.845-0.914), 0.874 (0.838-0.907) and 0.816 (0.772-0.857) respectively. GRS2 outperformed the AAGRS and GRS1 on Hispanic and European populations. The GRS2 distributions varied by population, with lower average scores for African populations. If the same GRS2 risk thresholds of 11.5 were set for European and African populations, the sensitivities were 0.91 and 0.53, respectively. CONCLUSIONS The GRS2 produced similar or improved discriminative power across the populations but the AAGRS matched performance on African ancestry participants with fewer single nucleotide polymorphisms. Varying GRS2 risk thresholds may be required for different populations due to the divergent distributions. INTRODUCTION Genetic risk scores (GRSs) summarise an individual’s known genetic variation about the risk of developing a disease into a single number. There is considerable interest in the use of GRSs for stratification of populations into risk classes to enable improved clinical outcomes ( 1 ), for example, through targeted monitoring of higher risk groups. GRSs can also be combined with other measures of risk to improve prediction or classification models ( 2 ), and can assist with study of diabetes aetiology ( 3 ). GRSs perform particularly well in type 1 diabetes (T1D) due to the high heritability of T1D, explained by a small number of variants in the human leukocyte antigen (HLA) region ( 4 , 5 ). GRSs are derived from genome wide association studies (GWASs), which find associations between individual single nucleotide polymorphisms (SNPs) and risk of a disease. These produce statistical significance of the SNPs alongside weights for use in a linear model. GWASs are predominantly performed in Europeans ( 6 ) with fewer studies in people of other ancestries ( 7 ). To reach SNP-wise statistical significance requires considerable quantities of genetic data so GRSs are either not produced, or may capture less genetic risk, in groups underrepresented in GWASs. An alternative to generation of GRSs for different populations is to utilise previously developed GRSs but the transferability of GRSs is often poor ( 8 ) or unknown. As GRSs become more useful for medicine ( 9 ) and research ( 10 ) we may see further increases in health disparities if the quality and validation of GRSs for underrepresented groups is not improved. T1D is less well studied and therefore the aetiology less well understood in people of African ancestry than in European populations ( 11 ). Recent analyses have shown genetic overlap in T1D between European and African ancestry populations ( 12 – 15 ). To date several GRSs have shown good discriminative ability in African-Americans ( 12 , 13 ), but there are no tests of these scores in populations in Africa. In this study we investigated how two European derived GRSs (GRS1 ( 16 ) and GRS2 ( 17 )), and one African GRS (AAGRS ( 13 )) performed on a dataset from the USA of people living with T1D and people without T1D which includes people of European, Hispanic, and African ancestry alongside a dataset with T1D and non-T1D from Cameroon and Uganda in sub-Saharan Africa. RESEARCH DESIGN AND METHODS We applied two T1D GRSs discovered on European populations (GRS1 and GRS2) and one T1D GRS discovered on an African ancestry population (AAGRS) on five datasets consisting of people with T1D and those without T1D. Three of these datasets contain participants with African ancestry with one group from the USA, one from Cameroon and one from Uganda. The other two groups are Europeans and Hispanics from the USA. We then investigated the performance of the three GRSs on these five datasets with the primary focus on the capacity of the GRSs to discriminate between T1D and non-T1D participants; we also examined effect of ancestry on risk thresholds; and how different ancestry populations in the same study would affect the apparent performance of the GRS. Furthermore, we investigated relationships between the GRSs, whether performance improved by combining them, and the importance of individual SNPs to the risk estimates. Research Cohorts USA Data from the USA came from The SEARCH for Diabetes in Youth study ( 18 ), an epidemiology study which includes people of African, European and Hispanic ancestries, labelled US-Africans, US-Europeans and US-Hispanics respectively, with predominantly T1D or type 2 diabetes (T2D) diagnosed before age 20. In this study we defined T1D as a clinical diagnosis alongside at least one autoantibody (of GAD, IA2 and Znt8) and non-T1D as those with a clinical diagnosis of T2D. We defined ancestry by clustering in the first two principal components of the genetic data. For US-Africans there were 194 participants with T1D and 235 without; for US-European there were 1109 participants with T1D and 125; and for US-Hispanics 266 with T1D and 170 without. These numbers for each ancestry and T1D status are additionally recorded in Table S1 of the supplementary material. Uganda and Cameroon The sub-Saharan ancestry T1D cohorts come from the Young Onset Diabetes in sub-Saharan Africa (YODA) study ( 3 ) (Clinical ID: NCT05013346 ) which includes participants from Cameroon and Uganda diagnosed with T1D before age 30. The T1D cases are defined as a clinical diagnosis of T1D alongside at least one positive autoantibody (of GAD, IA2 and ZNt8). Non-T1D data is from population data for Uganda ( 19 ) and Cameroon ( 20 ). We show results of these two datasets combined, with separate results in the supplementary material. The number of Uganda and Cameroon participants with T1D was 144 and those without was 5001. Details of these datasets are also in supplementary Table S1. 1000G To improve confidence in our conclusions we also used additional data from those without T1D from the 1000G project ( 21 ) which was an international study aiming to catalogue diverse genomes. We used African and European non-T1D samples from the 1000G to compare with the African and European ancestry cohorts; for comparison with the Hispanic population, we used the Americas superpopulations. Details of this data are in supplementary Table S1 (labelled as Non-T1D 1000G). Genotyping, quality control, and imputation The SEARCH genotyped data was produced on two chips: the Multi-Ethnic Global Array (MEGA; Illumina) and the Affymetric 500k imputation scaffold chip. We performed quality control, detailed in the supplementary material, on the data from the separate chips before merging them, removing any non-common SNPs and imputing using the TOPMed imputation panel ( 22 ) and the Minimac server ( 23 , 24 ). The YODA T1D data from Cameroon and Uganda was genotyped on a Global Screening Array (GSA; Illumina) chip. The data was quality controlled together and imputed using TOPMed. The Cameroon controls and Uganda population data were quality controlled separately and imputed to TOPMed. For the GRS2, some of the required SNPs could not be accurately imputed, primarily due to very low allele frequencies. The effect of these poorly imputed SNPs is discussed in the supplementary material. The overall effect is likely to be slightly poorer discriminative power but with no impact on the conclusions drawn. GRSs Two GRSs developed on European populations were calculated. The first ( 16 ) (denoted here as GRS1) includes 30 SNPs, with 28 contributing a linear weighted sum of the effect alleles with the weights produced from a GWAS. The final two SNPs are used in combination to tag the DR3/DR4-DQ8 haplotypes. The second European GRS (denoted as GRS2) ( 17 ) contains 67 SNPs of which 14 tag DR-DQ haplotypes, with specific DR-DQ interaction terms for 18 haplotype combinations. The other 53 SNPs are included in the linear weighted sum with 21 from other SNPs in the HLA and 32 from the non-HLA. This second European GRS has shown higher performance than the GRS1 and is believed to more comprehensively capture HLA and non-HLA risk. We additionally calculated a version of the GRS2 with a reduced number of SNPs (denoted as GRS47); with results available in the supplementary material. We also calculated a GRS (AAGRS) developed in people of African ancestry ( 13 ), which contains 7 SNPs with 5 in the HLA and 2 in the non-HLA. The AAGRS was shown to perform statistically significantly better than GRS1 in African ancestry populations ( 13 ). Statistical Methods We analysed the discriminative performance of the three GRSs at separation of the T1D from non-T1D participants in the African ancestry cohorts (SEARCH, Uganda and Cameroon) as well as the European and Hispanic ancestry participants from SEARCH by considering the receiver operating characteristic (ROC) curves and the area under the ROC curves (AUC). The 1000G data was utilised to assess similarities in distributions between the non-T1D data. We also explored how the sensitivity (true positive rate) and specificity (true negative rate) would alter by ancestry with choices of different thresholds to separate the populations into high and low risk groups. In addition, we explored how conclusions about overall GRS performance and the general separability of the data would change if we had different populations within the same dataset. The similarity of the GRSs on the datasets was investigated using correlation coefficients. We also examined if overlapping genetic information from the AAGRS and GRS2 could be combined for better performance. For the AAGRS we investigated how pairwise combinations altered performance. For the AAGRS and GRS2 we explored how the differences in average weighted scores of the SNPs contributed to the final differences between T1D and non-T1D, as well as adding all possible combinations of the 7-SNPs from AAGRS to GRS2. We also assessed which SNPs were important to the AAGRS and GRS2 discrimination performance. RESULTS Discriminative power of GRS1, GRS2 and AAGRS on the different populations We demonstrate the varying performance of the three GRSs on the datasets in Figure 1 . GRS2 and AAGRS showed similar performance in Africans and those of African ancestry; the AUCs for GRS2 and AAGRS on Uganda/Cameroon were 0.882 (95% CI 0.845-0.914) and 0.874 (95% CI 0.838-0.907), respectively; while the AUCs for GRS2 and AAGRS on the US-Africans were 0.839 (95% CI 0.799-0.877) and 0.838 (95% CI 0.799-0.874) respectively. The GRS1 had a statistically significantly reduced performance compared to GRS2 or AAGRS on the Uganda/Cameroon and US-Africans populations with AUCs of 0.816 (95% CI 0.772-0.857) and 0.796 (95% CI 0.751-0.837), respectively. In the US-European data, the GRS2 produced an AUC of 0.878 (95% CI 0.841-0.912), which significantly outperformed GRS1 and AAGRS with AUCs of 0.840 (95% CI 0.799-0.876) and 0.810 (95% CI 0.764-0.851), respectively. Similarly, for US-Hispanic data the GRS2 produced an AUC of 0.868 (95% CI 0.832-0.901), which was significantly higher than GRS1 or AAGRS, which had AUCs of 0.807 (95% CI 0.767-0.856) and 0.789 (0.742-0.834), respectively. In Table S2 of the supplementary material, we show the AUCs with uncertainty estimates for these results and for the separate Uganda and Cameroon data. In supplementary Figure S2, we show separate ROC curves for the Uganda and Cameroon populations. In supplementary Figure S3 we show ROC curves and AUCs for the GRS47 as well as plots of the GRS47 against GRS2; findings for GRS47 were similar to GRS2. We also used the 1000G data as non-T1D data showing similar discrimination performance by AUC (supplementary Table S3). Download figure Open in new tab Figure 1. The discriminative performance of the three GRSs (GRS1, GRS2, AAGRS) at separation of the T1D from non-T1D for the datasets. A, B, C, D: ROC curves for the three GRSs and the populations. Associated AUCs are shown in the legend. E: The AUCs with 95% uncertainties for the three GRSs and the populations Uganda and Cameroon combined (U.&C.), Uganda (U.), Cameroon (C.), US-Africans (Afr.), US-Europeans (Eur.) and US-Hispanics (Hisp.). Effect of ethnicity on GRS1, GRS2 and AAGRS distributions The GRS distributions for the three GRSs on the datasets are shown in Figure 2 . The African ancestry datasets had a lower distribution of scores for both non-T1D and T1D participants than the equivalents for the Europeans and Hispanics, which holds for all three GRSs. The Uganda and Cameroon data are shown separately in supplementary Figure S4. In supplementary Figure S5 we show distributions of two risk scores calculated by splitting the GRS2 into HLA (35 SNPs) and non-HLA (32 SNPs); most discrimination power came from the HLA SNPs but the change in GRS distribution was driven by both HLA and non-HLA SNPs. We also show the distributions of the GRSs on the 1000G data in Figure S6 of the supplementary material for superpopulations associated with African, Hispanic and European populations. Download figure Open in new tab Figure 2. GRS distributions for the three GRSs and the T1D and non-T1D populations. Sensitivity and specificity with different risk thresholds In Figure 3 (A, B) we show how, for GRS2, the sensitivity and specificity index vary with the choice of threshold defining those at lower and higher risk. The top left (A) and right (B) plots show the sensitivity and specificity respectively against GRS threshold for the datasets. The three African ancestry populations show similar patterns, which differ from the European or Hispanic, with generally lower sensitivity and higher specificity for the same threshold. Download figure Open in new tab Figure 3. The importance of risk thresholds for the GRS2 in the different populations. A, B) The sensitivity and specificity against risk threshold, respectively. The sensitivity/specificity is shown per population for each of the risk threshold values. The dashed vertical line shows the 11.5 threshold for generating the other plots. C and D) the densities of the non-T1D and T1D data at different GRS values. Samples to the left of the vertical dashed line would be defined as low risk (in blue columns) and to the right of the line as high risk (red columns). The number of non-T1D and T1D samples are shown as bars above (and marked with diagonal dashes) and below (marked with circles) the horizontal line, respectively. If the same threshold was chosen, there would be differences in both sensitivity and specificity for the populations. We demonstrate this effect of choosing one threshold (at 11.5) for the SEARCH African (bottom left plot, C) and European ancestry (bottom right plot, D). For the US-Africans there are small numbers of non-T1D being defined as high risk while for the US-Europeans there are more, i.e., the specificity of the US-Africans is higher than the US-Europeans. Conversely, for the US-Africans many of the US-Africans T1D samples are (falsely) defined as low risk while for the US-Europeans only few are. Equivalent plots for the other datasets are shown in the supplementary material (Figure S7). We also show confusion matrices in the supplementary material (Figure S8) demonstrating this same effect. The importance of correctly matching populations for cases and controls In Figure 4A we demonstrate the effect of assessing the quality of the GRS2 with different populations. We calculated the AUC produced if each of the five populations were assessed compared to each other for cases and controls. We then removed the expected AUC if the population cases/controls were correctly compared to itself; the expected AUC is the AUC between the cases and controls for the same population. Download figure Open in new tab Figure 4. AUCs for the GRS2 if comparing different populations. The heatmap (A, left plot) shows the differences in AUC from the expected (defined as the AUC between data from the same population) if comparing different populations. Negative values mean the AUC would appear lower and positive values the AUC would appear higher. The right plot (B) shows the AUCs for the AAGRS and GRS2 if the non-T1D controls combine populations (with under-sampling so the non-T1D population importance is equalised) for the Cameroon T1D. The black cross is where the non-T1D population is just Cameroon, the yellow circle is for all five populations included in the non-T1D populations, and the others are combinations of Cameroon with one other population. If the value is below 0 then the AUC between those pair of populations is lower than would be expected. For examples the largest gap is between the T1D Cameroon (C) and the US-Hispanic non-T1D which has a gap of -0.34. Conversely, the AUC would be overstated if the value is positive, so for the comparison of T1D US-Europeans with non-T1D US-Africans the apparent AUC of the comparison would be 0.13 higher than it should be. For all these plots, the pattern is the same with the comparisons between the US-Europeans or US-Hispanics with any of the three African populations yielding apparently higher discriminative power. Conversely the comparison between any of the three African populations with the US-Europeans or US-Hispanics would yield apparently worse performance. Another scenario is the comparison of a population with a set of mixed populations, in Figure 4B we show how this would alter the apparent quality between the GRS2 and AAGRS. We used the Cameroon T1D samples and calculated the AUC for the GRS2 and AAGRS if the non-T1D data was a combination of the Cameroon data with the other four datasets, both individually (the Cameroon non-T1D data and one other dataset) and for all datasets (the Cameroon data with all other four datasets). We under-sampled from the larger non-T1D datasets so we had the same weighting to the non-T1D data. The labels in the legend show which other non-T1D results are included with the Cameroon data. When the Cameroon non-T1D data is used on its own (black cross) the AAGRS and GRS2 show similar discrimination performance. When including the other datasets there is a change in apparent quality of the AAGRS and GRS2, with the GRS2 appearing to perform worse. Complementary plots for the other datasets are shown in supplementary Figure S9. Direct comparisons of GRSs and combining GRSs The direct relationships between the GRSs are shown in Supplementary Material with comparisons of the three GRSs for the five populations shown in Figure S10. The Pearson correlation coefficients for the combined T1D and non-T1D scores, are shown in Table S4; the correlations between GRSs tended to be higher in the SEARCH than the YODA data. We also considered the effect on the AUCs by combining together the final GRS2 and AAGRS. In supplementary Figure S11 we show all possible combinations of the weighted SNPs from the AAGRS with the GRS2 final score. The histogram shows the distribution of the scores, with the dashed, dotted and solid lines showing AUCs for the GRS2, AAGRS and final summed scores respectively. In the bottom right plot we show the individual AUCs, the sums ( Summed) and when we weight the two scores to optimise performance ( OptSummed ). For the African ancestry populations the AUC can be higher by combining the GRSs, or adding some of the AAGRS SNPs to the GRS2, but the changes are not statistically significant. For the Hispanic and European populations adding SNPs or scores from the AAGRS to the GRS2 tends to reduce the performance of the GRS2, although not statistically significantly. SNP importance for the AAGRS and GRS2 For GRS2, we show the relative importance of the SNPs for the separation of classes by plotting the allele frequencies of the T1D against non-T1D for the datasets (supplementary Figure S12). The risk increasing and protective alleles are red crosses and blue dots, respectively, and the marker size is proportional to the effect magnitude. We also show, in supplementary Figure S13, similar plots for the three African ancestry populations compared to one another to demonstrate the similarities in allele frequencies. In Supplementary Table S6, we show those SNPs which have either an average (across the five populations) difference of over 0.1 or individual population differences over 0.1, of which there are 11. For the AAGRS we consider the AUCs for the populations of individual, pairwise and all possible combinations of SNPs. The inclusion and removal of pairs of SNPs is shown in supplementary Figures S14. With just two of the 7-SNPs (rs2187668 and rs9273363) much of the discriminative power is still available. We also consider the effect of individual SNPs on the GRS2 performance by calculating the differences between the average of the weighted SNPs for the T1D and the non-T1D samples for each population. In general, the larger these differences, are the more important the SNPs are to the capacity of the GRS to separate the T1D from non-T1D (accepting effects of linkage disequilibrium which may make some of the SNPs partially redundant). In supplementary Table S5 we show the 11 most important SNPs from the GRS2 for separation of T1D from non-T1D. Correlations between AAGRS and GRS2 SNPs The GRS2 and AAGRS show a similar level of performance on the African ancestry datasets with different combinations of SNPs. In supplementary Figure S15 we show the correlations between the AAGRS SNPs and the previously mentioned 11 most important SNPs from the GRS2. There is substantial correlation between some of these SNPs. CONCLUSIONS On the three African ancestry populations the discriminative power between people with T1D and those without T1D of the GRS2 and AAGRS was higher than the GRS1; there was no statistically significant difference in the AUCs between GRS2 and AAGRS. This suggests that GRS2 and AAGRS are capturing similar risk information for the people of African ancestry which was not captured by GRS1. The GRS2 outperforms both the AAGRS and GRS1 on the Hispanic and European populations. The GRS2 either outperformed, or matched, the discriminative performance (defined by the AUC) of the GRS1 and AAGRS. Therefore, from this data, it could be used as an across-ancestry T1DGRS without loss of performance. A caveat is that GRS2 requires 67 SNP to be available rather than the 7 for the AAGRS and it is also a more complicated GRS with interaction terms between SNPs making it more difficult to accurately generate. The distributions of the GRSs varied considerably between the datasets; the three African populations generally have lower levels of all three GRSs compared to the European and Hispanic populations, in both participants with and without T1D. If the GRS2 was used as an across-ancestry GRS, then there are several consequences for its use. One is that the choice of a risk threshold would need to be considered carefully for different populations. If the same threshold were used for different populations the sensitivity and specificity would be substantially different; the African ancestry populations had higher specificity and lower sensitivity than the European or Hispanic populations at the same risk threshold. A second issue is that the judgement about the quality of the model, judged by AUC, would need to consider ancestry of T1D cases and non-T1D controls to avoid drawing false conclusions about the quality of the discrimination. Imbalance in ethnicity of cases and controls could lead to either under-or over-estimating the performance of the GRS. The correlation between the final scores of the three GRSs (on the same population) varies considerably, given the AAGRS and GRS2 produce similar discriminative performance on the African populations the correlations between these scores are modest. These differences in scores do not, however, result in any statistically significant improvement in AUC if the final scores are summed. For the three African populations the sum of the scores increases the AUC but not statistically significantly. The addition of the AAGRS scores to the GRS2 for the Europeans and Hispanics does not improve performance. Similarly, adding any combinations of the AAGRS weighted SNPs does not significantly improve the performance of the GRS2. Both these conclusions seem consistent – we would expect some modest improvement by adding the AAGRS as it would additionally emphasise African ancestry SNPs. However, the AAGRS and GRS2 both produce high AUCs and it is usually harder to improve an already effective classifier. The additional noise of extra SNPs to non-African ancestry data seems likely to slightly reduce the overall quality. Within the GRS2 there are a modest number of SNPs that are particularly important when considering the differences in average weighted values. Across the five populations there is reasonable agreement about the importance of many of the SNPs; while there are some SNPs that are more important for different populations. These European derived significant SNPs appear to capture substantial risk information on the African ancestry populations. For the AAGRS only a few SNPs (especially two HLA SNPs rs2187668 and rs9273363) produce most of the discriminative power, although all seven have some power on their own. The eleven most important GRS2 SNPs (defined by differences in mean scores between T1D and non-T1D) have substantial correlations with the AAGRS SNPs. There are complexities here with some differences between the populations but the overlap between the important GRS2 SNPs and SNPs from the AAGRS on the African population was substantial. Overall, a European ancestry GRS (GRS2) performed well on data from African, European and Hispanic ancestries. There is no clear evidence that the GRS2 is missing many variants when comparing to AAGRS which is based on African ancestry data. However, the distributions of the scores between ancestries can be very different. The consequences of this are that we need to be careful about combining ancestries as we may draw wrong conclusions about quality of models or about particular risk to individuals. In addition, to equalise measures such as sensitivity or specificity would require substantially different risk thresholds to be used. Data Availability Access to SEARCH data can be requested via https://www.searchfordiabetes.org/dspHome.cfm. Access to YODA data is restricted to associated research universities. Funding and Assistance This study was supported by a grant from the Randox Ltd. This study was supported by the National Institute for Health and Care Research Exeter Biomedical Research Centre. The views expressed are those of the author(s) and not necessarily those of the NIHR or the Department of Health and Social Care. M.J.R.’s work on this analysis was supported by NIH NIDDK grant R01 DK124395. The YODA study was funded by NIHR Global Health Award (17/63/131). Conflict of Interest The University of Exeter has a licensing and royalty agreement with Randox for a 10 SNP T1D GRS biochip. RAO and MNW report research funding from Randox. RAO has undertaken consulting for Janssen, Provention Bio, Novo Nordisk, and Sanofi, and advisory board membership for Sanofi. There are no other conflicts of interest to declare Prior Presentation Parts of this study were presented in abstract, poster and presentation form at the 9 th Meeting of the Study Group on Genetics of Diabetes in Exeter, United Kingdom 17 th -19 th April 2024. Acknowledgments Footnotes ↵ * Joint first authors References 1. ↵ Khera A V , Chaffin M , Aragam KG , Haas ME , Roselli C , Choi SH , et al. Genome-wide polygenic scores for common diseases identify individuals with risk equivalent to monogenic mutations . Nat Genet . 2018 ; 50 ( 9 ): 1219 – 24 . OpenUrl CrossRef PubMed 2. ↵ Ferrat LA , Vehik K , Sharp SA , Lernmark Å , Rewers MJ , She JX , et al. A combined risk score enhances prediction of type 1 diabetes among susceptible children . Nat Med . 2020 ; 26 ( 8 ): 1247 – 55 . OpenUrl CrossRef PubMed 3. ↵ Katte JC , Squires S , Dehayem M , Balungi PA , Padoa CJ , Sengupta D , et al. A Novel Non-Autoimmune Diabetes Subtype in Africa : Evidence from the Young-Onset Diabetes in Sub-Saharan Africa (Yoda) Study . 4. ↵ Cerolsaletti K , Hao W , Greenbaum CJ . Genetics Coming of Age in Type 1 Diabetes . Diabetes Care . 2019 ; 42 ( 2 ): 189 – 91 . OpenUrl FREE Full Text 5. ↵ Luckett AM , Weedon MN , Hawkes G , Leslie RD , Oram RA , Grant SFA . Utility of genetic risk scores in type 1 diabetes . Diabetologia . 2023 ; 66 ( 9 ): 1589 – 600 . OpenUrl CrossRef PubMed 6. ↵ Popejoy AB , Fullerton SM . Genomics is failing on diversity . Nature . 2016 ; 538 ( 7624 ): 161 – 4 . OpenUrl CrossRef PubMed 7. ↵ Sirugo G , Williams SM , Tishkoff SA . The missing diversity in human genetic studies . Cell . 2019 ; 177 ( 1 ): 26 – 31 . OpenUrl CrossRef PubMed 8. ↵ Duncan L , Shen H , Gelaye B , Meijsen J , Ressler K , Feldman M , et al. Analysis of polygenic risk score usage and performance in diverse human populations . Nat Commun . 2019 ; 10 ( 1 ): 3328 . OpenUrl CrossRef PubMed 9. ↵ Responsible use of polygenic risk scores in the clinic: potential benefits, risks and gaps . Nat Med . 2021 ; 27 ( 11 ): 1876 – 84 . OpenUrl CrossRef PubMed 10. ↵ Lewis CM , Vassos E. Polygenic risk scores: from research tools to clinical instruments . Genome Med . 2020 ; 12 ( 1 ): 1 – 11 . OpenUrl CrossRef 11. ↵ Katte JC , McDonald TJ , Sobngwi E , Jones AG . The phenotype of type 1 diabetes in sub-Saharan Africa . Front Public Health . 2023 ; 11 : 1014626 . OpenUrl PubMed 12. ↵ Oram RA , Sharp SA , Pihoker C , Ferrat L , Imperatore G , Williams A , et al. Utility of diabetes type–specific genetic risk scores for the classification of diabetes type among multiethnic youth . Diabetes Care . 2022 ; 45 ( 5 ): 1124 – 31 . OpenUrl PubMed 13. ↵ Onengut-Gumuscu S , Chen WM , Robertson CC , Bonnie JK , Farber E , Zhu Z , et al. Type 1 diabetes risk in African-ancestry participants and utility of an ancestry-specific genetic risk score . Diabetes Care . 2019 ; 42 ( 3 ): 406 – 15 . OpenUrl Abstract / FREE Full Text 14. Qu HQ , Qu J , Glessner J , Liu Y , Mentch F , Chang X , et al. Improved genetic risk scoring algorithm for type 1 diabetes prediction . Pediatr Diabetes . 2022 ; 23 ( 3 ): 320 – 3 . OpenUrl CrossRef PubMed 15. ↵ Balcha SA , Demisse AG , Mishra R , Vartak T , Cousminer DL , Hodge KM , et al. Type 1 diabetes in Africa: an immunogenetic study in the Amhara of North-West Ethiopia . Diabetologia . 2020 ; 63 : 2158 – 68 . OpenUrl PubMed 16. ↵ Oram RA , Patel K , Hill A , Shields B , McDonald TJ , Jones A , et al. A type 1 diabetes genetic risk score can aid discrimination between type 1 and type 2 diabetes in young adults . Diabetes Care . 2016 ; 39 ( 3 ): 337 – 44 . OpenUrl Abstract / FREE Full Text 17. ↵ Sharp SA , Rich SS , Wood AR , Jones SE , Beaumont RN , Harrison JW , et al. Development and standardization of an improved type 1 diabetes genetic risk score for use in newborn screening and incident diagnosis . Diabetes Care . 2019 ; 42 ( 2 ): 200 – 7 . OpenUrl Abstract / FREE Full Text 18. ↵ Dabelea D , Bell RA , D’Agostino Jr RB , Imperatore G , Johansen JM , Linder B , et al. Incidence of diabetes in youth in the United States . JAMA . 2007 ; 297 ( 24 ): 2716 – 24 . OpenUrl CrossRef PubMed Web of Science 19. ↵ Fatumo S , Mugisha J , Soremekun OS , Kalungi A , Mayanja R , Kintu C , et al. Uganda Genome Resource: a rich research database for genomic studies of communicable and non-communicable diseases in Africa . Cell Genomics . 2022 ; 2 ( 11 ). 20. ↵ Feutseu C , Kowo MP , Boli AO , Katte JC , Guewo-Fokeng M , Zemsi S , et al. Seroprevalence of hepatitis C virus infection in patients with type 2 diabetes mellitus is associated with increased age in sub-Saharan Africa: Results from a cross-sectional comparative analysis . Frontiers in Gastroenterology . 2023 ; 2 : 1063590 . OpenUrl 21. ↵ 1000 Genomes Project Consortium et al. A global reference for human genetic variation . Nature . 2015 ; 526 ( 7571 ): 68 – 74 . OpenUrl CrossRef PubMed 22. ↵ Taliun D , Harris DN , Kessler MD , Carlson J , Szpiech ZA , Torres R , et al. Sequencing of 53,831 diverse genomes from the NHLBI TOPMed Program . Nature . 2021 ; 590 ( 7845 ): 290 – 9 . OpenUrl CrossRef PubMed 23. ↵ Das S , Forer L , Schönherr S , Sidore C , Locke AE , Kwong A , et al. Next-generation genotype imputation service and methods . Nat Genet . 2016 ; 48 ( 10 ): 1284 – 7 . OpenUrl CrossRef PubMed 24. ↵ Fuchsberger C , Abecasis GR , Hinds DA . minimac2: faster genotype imputation . Bioinformatics . 2014 ; 31 ( 5 ): 782 – 4 . OpenUrl PubMed View the discussion thread. Back to top Previous Next Posted July 17, 2025. Download PDF Supplementary Material Data/Code Email Thank you for your interest in spreading the word about medRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following The impact of ancestry on performance of type 1 diabetes genetic risk scores: high discrimination performance is maintained in African ancestry populations, but population specific thresholds may improve risk prediction Message Subject (Your Name) has forwarded a page to you from medRxiv Message Body (Your Name) thought you would like to see this page from the medRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share The impact of ancestry on performance of type 1 diabetes genetic risk scores: high discrimination performance is maintained in African ancestry populations, but population specific thresholds may improve risk prediction Steven Squires , Jean Claude Katte , Dana Dabelea , Catherine Pihoker , Jasmin Divers , Eugene Sobngwi , Moffat J. Nyirenda , Raymond J. Kreienkamp , Angela D Liese , Amy S Shah , Lawrence Dolan , Kristi Reynolds , Maria J. Redondo , William Hagopian , Segun Fatumo , Mesmin Y. Dehayem , Andrew Hattersley , Michael N. Weedon , Angus Jones , Richard A. Oram medRxiv 2025.07.17.25330632; doi: https://doi.org/10.1101/2025.07.17.25330632 Share This Article: Copy Citation Tools The impact of ancestry on performance of type 1 diabetes genetic risk scores: high discrimination performance is maintained in African ancestry populations, but population specific thresholds may improve risk prediction Steven Squires , Jean Claude Katte , Dana Dabelea , Catherine Pihoker , Jasmin Divers , Eugene Sobngwi , Moffat J. Nyirenda , Raymond J. Kreienkamp , Angela D Liese , Amy S Shah , Lawrence Dolan , Kristi Reynolds , Maria J. Redondo , William Hagopian , Segun Fatumo , Mesmin Y. Dehayem , Andrew Hattersley , Michael N. Weedon , Angus Jones , Richard A. Oram medRxiv 2025.07.17.25330632; doi: https://doi.org/10.1101/2025.07.17.25330632 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Genetic and Genomic Medicine Subject Areas All Articles Addiction Medicine (568) Allergy and Immunology (863) Anesthesia (300) Cardiovascular Medicine (4436) Dentistry and Oral Medicine (444) Dermatology (382) Emergency Medicine (608) Endocrinology (including Diabetes Mellitus and Metabolic Disease) (1509) Epidemiology (15229) Forensic Medicine (30) Gastroenterology (1124) Genetic and Genomic Medicine (6600) Geriatric Medicine (668) Health Economics (997) Health Informatics (4538) Health Policy (1368) Health Systems and Quality Improvement (1613) Hematology (542) HIV/AIDS (1264) Infectious Diseases (except HIV/AIDS) (15916) Intensive Care and Critical Care Medicine (1103) Medical Education (623) Medical Ethics (146) Nephrology (667) Neurology (6599) Nursing (346) Nutrition (998) Obstetrics and Gynecology (1144) Occupational and Environmental Health (957) Oncology (3333) Ophthalmology (974) Orthopedics (369) Otolaryngology (420) Pain Medicine (436) Palliative Medicine (130) Pathology (663) Pediatrics (1693) Pharmacology and Therapeutics (691) Primary Care Research (711) Psychiatry and Clinical Psychology (5447) Public and Global Health (9232) Radiology and Imaging (2198) Rehabilitation Medicine and Physical Therapy (1370) Respiratory Medicine (1196) Rheumatology (593) Sexual and Reproductive Health (712) Sports Medicine (530) Surgery (712) Toxicology (99) Transplantation (289) Urology (265) (function(){function c(){var b=a.contentDocument||a.contentWindow.document;if(b){var d=b.createElement('script');d.innerHTML="window.__CF$cv$params={r:'a00fb356ceecc165',t:'MTc3OTY2MTM2MQ=='};var a=document.createElement('script');a.src='/cdn-cgi/challenge-platform/scripts/jsd/main.js';document.getElementsByTagName('head')[0].appendChild(a);";b.getElementsByTagName('head')[0].appendChild(d)}}if(document.body){var a=document.createElement('iframe');a.height=1;a.width=1;a.style.position='absolute';a.style.top=0;a.style.left=0;a.style.border='none';a.style.visibility='hidden';document.body.appendChild(a);if('loading'!==document.readyState)c();else if(window.addEventListener)document.addEventListener('DOMContentLoaded',c);else{var e=document.onreadystatechange||function(){};document.onreadystatechange=function(b){e(b);'loading'!==document.readyState&&(document.onreadystatechange=e,c())}}}})();
Text is read by the "Ask this paper" AI Q&A widget below.
Extraction quality varies by source — PMC NXML preserves structure
cleanly, OA-HTML may include some navigation residue, and OA-PDF can
have broken hyphenation. The publisher copy
(via DOI)
is the canonical version.