Estimating population structure using epigenome-wide methylation data

preprint OA: closed
📄 Open PDF Full text JSON View at publisher

Abstract

Introduction In epigenome-wide association analysis (EWAS), unaddressed population stratification often leads to inflation. We aimed to compute methylation population scores (MPSs) that predict genetic principal components (GPCs) using a feature selection and regression approach. Methods We used multi-ethnic methylation data (Illumina 450K/EPIC array) from unrelated MESA (n=929), CARDIA (n=1123), JHS (n=1365), ARIC (n=2338), and HCHS/SOL (n=1475) individuals, randomly assigning 85% of participants from each cohort to a training dataset and the remaining 15% to a test dataset. First, we estimated the associations of GPCs with each available CpG methylation site using linear regression within each cohort, adjusting for age, sex, smoking status, race/ethnic background (as a proxy for background information associated with lifestyle and other environmental exposures that may impact methylation), alcohol use status, body mass index, and cell type proportions. We meta-analyzed the associations across cohorts and selected CpG sites with association FDR-adjusted q-value <0.05. We next aggregated individual-level data across the cohort-specific training datasets, and applied two-stage weighted least squares Lasso regression, with the GPCs as the outcomes and the selected CpG sites as penalized predictors, adjusting for the aforementioned covariates. The developed MPSs are the weighted sum of selected CpG sites from the Lasso. To evaluate the developed MPSs, we constructed them in the test dataset, and compared them with GPCs, and with MPSs constructed based on a previously-published paper. Comparison was based on correlation analysis and data visualization. We demonstrate the use of the MPSs in EWAS. Results In the test dataset, the MPSs were highly correlated with GPCs, with correlation decreasing, though not monotonically, for later components. Specifically, MPS1 and GPC1 had R2= 0.99, while MPS7 and GPC7 had R2=0.27 (the lowest observed correlation). In data visualization, MPSs had similar patterns as GPCs in differentiating self-reported White, Black, and Hispanic/Latino groups, while outperforming MPC constructed using alternative published methods. MPSs showed comparable performance to GPCs in reducing some of the inflation in EWAS. Conclusions Methylation-based population scores provide a reliable estimate of population structure in the data and can complement GPCs when genetic data are absent. Unlike previous methods based on unsupervised methylation PCA, MPSs uses supervised learning with covariate adjustment to capture genetic structure across diverse populations. The weights for each GPCs derived in our study can be applied to generate MPSs in other studies.
Full text 50,470 characters · extracted from preprint-html · click to expand
Estimating population structure using epigenome-wide methylation data | medRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-P4HH5NV'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search Estimating population structure using epigenome-wide methylation data Ziqing Wang , Kent D Taylor , Jerome I Rotter , Stephen S Rich , Yinan Zheng , Lifang Hou , Xiuqing Guo , Jan Bressler , View ORCID Profile Laura M Raffield , Yongmei Liu , Robert Kaplan , Donald M Lloyd-Jones , Alanna C. Morrison , View ORCID Profile Myriam Fornage , View ORCID Profile Tamar Sofer the TOPMed Epigenetics working group doi: https://doi.org/10.1101/2025.09.01.25334865 Ziqing Wang 1 CardioVascular Institute, Beth Israel Deaconess Medical Center , Boston, MA, USA 2 Department of Medicine, Harvard Medical School , Boston, MA, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site For correspondence: zwang14{at}bidmc.harvard.edu Kent D Taylor 3 The Institute for Translational Genomics and Population Sciences, Department of Pediatrics, The Lundquist Institute for Biomedical Innovation at Harbor-UCLA Medical Center , Torrance, CA, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site Jerome I Rotter 3 The Institute for Translational Genomics and Population Sciences, Department of Pediatrics, The Lundquist Institute for Biomedical Innovation at Harbor-UCLA Medical Center , Torrance, CA, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site Stephen S Rich 4 Department of Public Health Genomics, University of Virginia School of Medicine , Charlottesville, VA, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site Yinan Zheng 5 Department of Preventive Medicine, Northwestern University Feinberg School of Medicine , Chicago, IL, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site Lifang Hou 5 Department of Preventive Medicine, Northwestern University Feinberg School of Medicine , Chicago, IL, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site Xiuqing Guo 3 The Institute for Translational Genomics and Population Sciences, Department of Pediatrics, The Lundquist Institute for Biomedical Innovation at Harbor-UCLA Medical Center , Torrance, CA, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site Jan Bressler 6 Human Genetics Center, Department of Epidemiology, School of Public Health, The University of Texas Health Science Center at Houston , Houston, TX, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site Laura M Raffield 7 Department of Genetics, University of North Carolina at Chapel Hill , Chapel Hill, NC, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Laura M Raffield Yongmei Liu 8 Department of Medicine, Divisions of Cardiology and Neurology, Duke University Medical Center , Durham, NC, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site Robert Kaplan 9 Department of Epidemiology and Population Health, Albert Einstein College of Medicine , Bronx, NY, USA 10 Division of Public Health Sciences, Fred Hutchinson Cancer Center , Seattle, WA, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site Donald M Lloyd-Jones 11 Department of Preventive Medicine, Boston University Chobanian & Avedisian School of Medicine , Boston, MA, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site Alanna C. Morrison 6 Human Genetics Center, Department of Epidemiology, School of Public Health, The University of Texas Health Science Center at Houston , Houston, TX, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site Myriam Fornage 6 Human Genetics Center, Department of Epidemiology, School of Public Health, The University of Texas Health Science Center at Houston , Houston, TX, USA 12 Brown Foundation Institute of Molecular Medicine, McGovern Medical School, University of Texas Health Science Center at Houston , Houston, TX, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Myriam Fornage Tamar Sofer 1 CardioVascular Institute, Beth Israel Deaconess Medical Center , Boston, MA, USA 2 Department of Medicine, Harvard Medical School , Boston, MA, USA 13 Division of Sleep Medicine and Circadian Disorders, Department of Medicine, Brigham and Women’s Hospital , Boston, MA, USA 14 Department of Biostatistics, Harvard T.H Chan School of Public Health , Boston, MA, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Tamar Sofer Abstract Full Text Info/History Metrics Supplementary material Data/Code Preview PDF Abstract Introduction In epigenome-wide association analysis (EWAS), unaddressed population stratification often leads to inflation. We aimed to compute methylation population scores (MPSs) that predict genetic principal components (GPCs) using a feature selection and regression approach. Methods We used multi-ethnic methylation data (Illumina 450K/EPIC array) from unrelated MESA (n=929), CARDIA (n=1123), JHS (n=1365), ARIC (n=2338), and HCHS/SOL (n=1475) individuals, randomly assigning 85% of participants from each cohort to a training dataset and the remaining 15% to a test dataset. First, we estimated the associations of GPCs with each available CpG methylation site using linear regression within each cohort, adjusting for age, sex, smoking status, race/ethnic background (as a proxy for background information associated with lifestyle and other environmental exposures that may impact methylation), alcohol use status, body mass index, and cell type proportions. We meta-analyzed the associations across cohorts and selected CpG sites with association FDR-adjusted q-value <0.05. We next aggregated individual-level data across the cohort-specific training datasets, and applied two-stage weighted least squares Lasso regression, with the GPCs as the outcomes and the selected CpG sites as penalized predictors, adjusting for the aforementioned covariates. The developed MPSs are the weighted sum of selected CpG sites from the Lasso. To evaluate the developed MPSs, we constructed them in the test dataset, and compared them with GPCs, and with MPSs constructed based on a previously-published paper. Comparison was based on correlation analysis and data visualization. We demonstrate the use of the MPSs in EWAS. Results In the test dataset, the MPSs were highly correlated with GPCs, with correlation decreasing, though not monotonically, for later components. Specifically, MPS1 and GPC1 had R2= 0.99, while MPS7 and GPC7 had R2=0.27 (the lowest observed correlation). In data visualization, MPSs had similar patterns as GPCs in differentiating self-reported White, Black, and Hispanic/Latino groups, while outperforming MPC constructed using alternative published methods. MPSs showed comparable performance to GPCs in reducing some of the inflation in EWAS. Conclusions Methylation-based population scores provide a reliable estimate of population structure in the data and can complement GPCs when genetic data are absent. Unlike previous methods based on unsupervised methylation PCA, MPSs uses supervised learning with covariate adjustment to capture genetic structure across diverse populations. The weights for each GPCs derived in our study can be applied to generate MPSs in other studies. Introduction Epigenome-wide association studies (EWAS) and genome wide association studies (GWAS) are powerful approaches to identify epigenetic and genetic associations, respectively, with complex traits and diseases. In both types of analyses, it is essential to address population stratification, a form of bias caused by linkage disequilibrium and allele frequency divergence between subpopulations, which arise from populations’ evolutionary and demographic histories 1 , 2 . Failure to account for population structure can lead to test statistic inflation and increased false positive findings, thus complicating the identification of true biological signals. The most common method to address population stratification is to include genetic principal components (GPCs), derived from principal components analysis (PCA) of genome-wide genetic data, as covariates in the association analysis. Consequently, the absence or incompleteness of genetic data often limits the sample size and thereby reduces the statistical power of EWAS association analyses. DNA methylation (DNAm) is an epigenetic modification that can influence phenotype and disease outcomes by regulating gene expression through transcription and chromosomal inactivation 3 , 4 . Notably, DNAm is shaped by both genetic factors and environmental exposures 5 , 6 , ranging from lifestyle and socioeconomic stressors 7 to infectious agents 8 , and is characterized by its long-term stability, with potential transgenerational impact 9 . That said, previous studies have attempted to compute methylation-based PCs by PCA on genome-wide DNA methylation data or using CpG sites highly correlated with single nucleotide polymorphisms (SNPs) to capture population structure 8 , 10 , 11 . For instance, Rahmani and colleagues proposed to construct methylation PC (MPC) by applying PCA on a set of genetically informative CpGs in close vicinity to SNPs selected by linear model 10 . These findings demonstrate that genetic ancestry information is also embedded in one’s DNAm profiles, offering a promising opportunity for addressing population structure in epigenome-wide association studies when genetic data are unavailable. One potential challenge with using PCs obtained by PCA of genome-wide methylation rather than genetic data, is that they may inadvertently capture technical or demographic variation, such as differences in cell type composition or age, since DNA methylation is influenced by a wide range of factors beyond genetic ancestry 11 , 12 . In this study, we develop methylation-based measures of GPCs, which we term “Methylation population scores (MPS)”. The MPSs are methylation scores that predict GPCs. We develop them using a feature selection approach informed by GPCs, while accounting for additional confounding variables, including age, sex, estimated cell type proportions, and environmental factors, including geographical locations and key lifestyles. We use methylation and genetic data from five multi-ethnic cohorts participating in the Trans-Omics in Precision Medicine (TOPMed) program, encompassing individuals self-reporting of White, Black, Chinese, and Hispanic backgrounds. We further compared our MPSs to MPC computed through PCA using reference list provided by Rahmani et al 10 . The significance of this study is two-fold, 1) develop MPSs based on available, continuous, genetic ancestry information (in the form of genetic PCs), meanwhile controlling for environmental and technical confounders through covariate adjustment; and 2) incorporate a broader range of population groups and a larger sample size compared to previous studies, which primarily focused on individuals of White and Black backgrounds. Methods Figure 1 provides an overview of the study. We used methylation data (Illumina 850K/EPIC array in primary analysis, limiting to CpG sites available on the Illumina HumanMethylation450 BeadChip in secondary analysis, from genetically unrelated, individuals from 5 studies representing multiple race/ethnicity backgrounds from the TOPMed program: Multi-Ethnic Study of Atherosclerosis (MESA), Coronary Artery Risk Development in Young Adults (CARDIA), Jackson Heart Study (JHS), Atherosclerosis Risk in Communities (ARIC) study, and the Hispanic Community Health Study/Study of Latinos (HCHS/SOL). Within each cohort, we randomly assigned 85% of participants to a training dataset and the remaining to a test dataset. We developed MPSs using CpG sites identified by feature selection methods trained on GPCs, and performed multiple analyses to evaluate their performance. Download figure Open in new tab Figure 1. Analysis plan of the study. ARIC: Atherosclerosis Risk in Communities study; CARDIA: Coronary Artery Risk Development in Young Adults; JHS: Jackson Heart Study; MESA: Multi-Ethnic Study of Atherosclerosis; HCHS/SOL: Hispanic Community Health Study/Study of Latinos The Trans-Omics for Precision Medicine (TOPMed) program The TOPMed program aims to improve diagnosis, treatment and prevention of heart, lung, blood and sleep disorders by elucidating genetic and phenotypic data over 130,000 participants from more than 80 participating studies 13 , available through dbGaP (Database of Genotypes and Phenotypes) 14 . To minimize batch bias across cohorts, sequencing centers and experiment years, TOPMed standardized laboratory methods to a single pipeline and performed variant and genotype calling jointly on all samples in a given TOPMed freeze 13 . For this study, we obtained TOPMed whole genome sequencing (WGS) data from the Freeze 10 release. Whole Genome Sequencing (WGS) DNA samples were extracted from whole blood samples collected from each participating cohort, before being processed for pair-ended 150 bp sequencing with a mean depth of 30x using Illumina HiSeq X Ten instruments at TOPMed sequencing centers 14 . As aforementioned, generated reads were first mapped to the human-genome build GRCh38 following a common pipeline across all sequencing centers. Variant discovery and genotyping were performed jointly across all available samples in freeze 10 using GotCloud pipeline 15 . Quality control included variant inclusion by minor allele frequency (MAF > 0.01), genotypes with a minimal depth of more than 10x, and sample-level checks for pedigree errors, self-reported and genetic sex discrepancies, and discordance with previous genotyping array data 14 . Genetic principal components (GPCs) and genetic relatedness were also centrally computed using the PC-AiR and PC-Relate algorithms 16 . Among our study participants, a set of unrelated individuals was defined as a maximal set of individuals where twice the kinship coefficient was lower than 0.0625*2 for all pairs of individuals, limiting relatedness to 3 rd degree. Epigenome-wide methylation quantification Methylation profiling for ARIC, CARDIA, and HCHS/SOL was performed at the University of Washington Northwest Genomics Center (NWGC), where DNA samples underwent normalization and bisulfite conversion using the EZ-96 DNA Methylation kit (#D5003; ZYMO Research). Methylation data for “NHLBI TOPMed: Multi-Ethnic Study of Atherosclerosis (MESA)” was performed at Keck Molecular Genomics Core Facility.” DNAm profiles were generated using the MethylationEPIC BeadChip and quality checked using Illumina GenomeStudio software (v2.0.3) to include samples with a genotyping call rate of 0.98 or higher. Samples passing quality control were normalized by normal-exponential deconvolution using out-of-band probes (Noob) background subtraction 17 using the R package Minfi 18 to obtain beta-values. We estimated cell type subpopulations from the processed methylation data for each cohort using reference-based Houseman’s method 19 . HCHS/SOL DNAm samples were also profiled using the Infinium MethylationEPIC array at the HCHS/SOL Data Coordinating Center (DCC) at the University of North Carolina at Chapel Hill, and examined for sex and SNPs mismatches, control and blind duplicate, followed by normalization using the R package SeSAMe 20 to mask 105,454 probes, dye bias correction, and Noob normalization 17 , 18 . The obtained beta-values were further corrected for type-2 probe bias using BMIQ method from the WateRmelon package 21 , and underwent ComBat batch correction 22 . Subsequently, cell type proportions were also estimated using Houseman’s method 19 . Development of MPSs: methylation predictors of genetic principal components We implemented a two-step feature selection method to identify the most representative CpG for each GPC in MPSs construction using training data. As shown in Figure 1 , the first step involved a preliminary selection of CpG sites for each GPC based on their associations. As meta-analysis is shown to be equivalent to aggregated data analysis 23 , we estimated CpG site associations with GPCs by linear regression within each cohort’s training dataset in order to reduce computational burden. Each linear model is adjusted for age, sex, smoking and alcohol use status (if applicable), BMI, cell type proportions, study center, and race/ethnicity (if applicable; as identification with race/ethnicity group is often associated with lifestyle and environmental exposure patterns that may impact methylation). Specifically, smoking and alcohol use were modeled as categorical variable (0= never, 1=former or current). We then meta-analyzed these associations across cohorts by inverse variance, fixed effects meta-analysis. We applied False-Discovery Rate (FDR) correction on the resulting p-values using the Benjamini-Hochberg procedure 24 and selected CpG sites with an FDR-adjusted q-value<0.05. We then aggregated the training datasets across all cohorts and applied weighted-Lasso regression to further refine the selection CpG sites and compute weights. Weighted-Lasso was implemented as follows. We first constructed an initial MP (MPS i ) for each GPC as weighted sums of the CpG sites identified by a first Lasso regression, adjusting for the same set of covariates (except for alcohol use due to partial availability) as were used in the initial regression analysis. Next, we added a weighting step because the variance of GPC varies by genetic ancestry due to patterns of allele frequency and linkage disequilibrium. Thus, each GPC was regressed on its corresponding, initial MPS i without covariate adjustment to compute individual-specific weights, calculated as normalized squared residuals such that their sum equaled the sample size, with extreme values above 90 th percentile truncated to the 90 th percentile for stability 25 ( Figure 1 ). These individual weights were implemented as observation weights in a second Lasso regression to improve the accuracy of MPSs by accounting for differences in prediction variation related to genetic diversity (where genetic structure impacts genetic variance 26 ). MPSs evaluation We evaluated the performance of the MPSss, computed as weighted sums of CpG sites identified in the second Lasso regression, by computing their correlations with GPCs, and comparing them via visualization to GPCs and methylation PCs constructed via PCA based on CpG sites provided by Rahmani and colleagues 10 and SNP adjacent CpG sites in the test dataset. Missing values in the methylation data were imputed using the mean across all samples before performing PCA. In particular, we performed visualization of population structure highlighting self-reported race/ethnicity groups, because race/ethnicity groups have shared patterns of genetic ancestry due to historical geographic migration patterns. We computed the variance explained by each MPSs for its corresponding GPC using linear regression adjusting for the same covariates aforementioned. Comparing alternative approaches to population stratification adjustment in epigenome-wide associations study of diabetes mellitus in HCHS/SOL We also conducted epigenome wide association analysis (EWAS) in HCHS/SOL participants with diabetes as the exposure, which was defined according to medical history and lab criteria defined by American Diabetes Association, as previously described 27 . These analyses compared models that adjusted for MPSs in the subset of individuals who have genetic data and therefore have GPCs (n=1475), MPSs in a larger dataset of individuals who have methylation data (and therefore MPSs) but not necessarily GPCs (n=2695), models adjusted for GPCs alone (n=1475), and unadjusted models. In addition, we identified significant associations (adjusted p-value using Bonferroni correction less than 0.05) from the EWAS results in HCHS/SOL and compared them to previously identified diabetes associated CpG sites to evaluate biological relevance 28 . To further evaluate generalizability, we constructed MPSs in HCHS/SOL participants without genetic data and assessed their ability to differentiate individuals of different Hispanic/Latino backgrounds. Results Descriptive statistics of demographic characteristics for the TOPMed cohorts, stratified by self-reported race/ethnicity, are summarized in Table 1 . Across all cohorts, women comprised a higher proportion of participants. The average age ranges from 40 years in CARDIA to 60 years in MESA. View this table: View inline View popup Download powerpoint Table 1. Demographic characteristics of study cohorts. Methylation principal components development We developed MPSs for the first 10 GPCs. The number of CpG sites selected for each GPC in step 1 (FDR <0.05) and by the weighted Lasso regression in step 2 are reported in Table S1, with GPC1 having the largest number of associated CpGs (n= 32172), and GPC7 the fewest (n= 44). Detailed lists of CpGs selected for MPSs construction, along with weights are provided in the Supplementary Data 1. The top four PCs accounted for the majority of selected CpGs, whereas PC7 was associated with only 25 CpGs (Table S1). Methylation principal components evaluation We constructed the MPSs in the aggregated TOPMed validation dataset (n=1090). Figure 2 visualizes the Pearson correlations between the 10 MPSs and GPCs. The MPSs were moderately to highly correlated with their corresponding GPCs. The strongest correlation was between the first MPSs and the first GPC (R 2 = 0.98), with the lowest correlation between MPS7 and GPC7 (R 2 = 0.28). Similar to the correlation pattern among GPCs, MPS2 through MPS4 showed relatively higher correlations compared to other components. Interestingly, MPS5 and MPS8 demonstrated moderate correlations with GPC2 through GPC4 as well as their corresponding MPS—a pattern not observed for GPC5 and GPC8. Notably, the top three MPSs explained more than 80% of variance in their corresponding GPC, whereas the variance explained by MPS7 and MPS8 dropped below 10% (Table S2). The MPSs constructed using the 450K array follow a similar pattern, albeit with slightly less variance explained (Table S3). Download figure Open in new tab Figure 2. Correlation analysis between the 10 methylation (MPSs) and genetic principal components (PC). Out of 4913 CpGs previously reported by Rahmani et al 10 . for construction of population stratification indicators, 1369 were available in our methylation data and used to construct the methylation principal components (MPC_Rahmani) in the aggregated test dataset across five TOPMed cohorts. We visualized the top three principal components of our MPSs, MPC_Rahmani, MPC based on SNP adjacent CpG sites, and GPCs using scatterplots to assess their performance in differentiating the four self-reported race/ethnicity groups in our test dataset ( Figure 3 ), which show group-level separation in GPCs space due to correlation of these groupings with genetic ancestry at the population level. We anticipated that methylation derived MPSs and MPCs showing similar patterns to GPCs. As shown in Figure 3A and 3D , MPSs derived from feature selection in this study exhibited clear separation among the four race/ethnicity groups, closely mirroring the clustering observed with GPCs, though the within-group dispersion appeared greater for MPSs. Additionally, the separation by MPSs between Chinese and Hispanic/Latino groups was less pronounced ( Figure 3A ). On the other hand, the top 3 MPC_Rahmani effectively distinguished Black, White and Hispanic/Latino participants, but not Chinese ( Figure 3B ). The clustering of the three race/ethnicity groups was more dispersed and less compact compared to MPSs ( Figure 3A-3B ). MPCs constructed using PCA on SNP adjacent CpG sites distinguishes Hispanic/Latino participants fairly well from other race/ethnicity groups, but Black and White participants are mixed together ( Figure 3C ). Moreover, a group of participants from all four race/ethnicity groups were separated out by MPCs ( Figure 3C ). Figure S1 illustrates the separation of self-reported Hispanic backgrounds in HCHS/SOL participants by MPSs, combining those in test data and without genetic data (n=689) (left), compared with the ones in the test dataset by GPCs (n=221) (right). While the self-reported Hispanic/Latino groups appear less distinct from each other in the MPSs space (left), it broadly replicates the structure observed with GPCs. In both cases, individuals of Mexican, Central American and South American backgrounds are more mixed together, while individuals of Dominican and Puerto Rican backgrounds were grouped closer to each other (Figure S1). According to the parallel coordinate plot, these backgrounds diverge more clearly at GPC3 and GPC8, as well as at MPS3 and MPS8 (Figure S2). However, Hispanic/Latino backgrounds exhibit minimal differentiation at MPS5 to MPS7 (Figure S2B). Download figure Open in new tab Figure 3. Scatter plots of GPCs, MPSs, and MPC_Rahmani, colored by race/ethnicity groups present in the data. A) Methylation population scores (MPSs) constructed in the current study; B) MPCs constructed via principal component analysis (PCA) using CpG sites from previously published paper by Rhamani et al.; C) Methylation PCs (MPC) constructed in test dataset through PCA on SNP adjacent CpGs; D) Genetic principal components (GPCs). Comparing population stratification adjustment in epigenome-wide association study of diabetes in HCHS/SOL When using participants with genetic data (n= 1475), i.e. using exactly the same sample size to compare population stratification adjustment approaches, all three models appear inflated in both QQ plots and Manhattan plots ( Figure 4a , Figure S2). The genomic inflation factors were 1.57 for no adjustment, 1.51 and 1.50 for adjusting for 5 GPCs and 5 MPSs, respectively, indicating moderate inflation across all three analyses, and some attenuation of inflation when adjusted for GPCs or MPSs. The number of statistically significant sites differed. Adjusting for 5 GPCs or 5 MPSs resulted in 153 and 157 significant CpGs, respectively, while no adjustment for population structure yielded 211 CpGs associated with diabetes (Figure S2). Figure 4B illustrates the results expanding the analysis to all available samples for no adjustment and adjusting for MPSs (n= 2695), with 2672 and 1749 significant CpGs, respectively. We also benchmarked the identified CpG sites against a previously validated set of 56 CpG sites associated with type 2 diabetes 28 . Of these sites, 15 have FDR p-value<0.05 (computed over the 56 sites) in the analysis that did not adjust for population structure and MPS-adjusted EWAS, and 14 for GPC-adjusted EWAS (Table S4). In the analysis using all available samples, 26 and 24 were identified for no-adjustment and MPS-adjusted EWAS, respectively (Table S5). Download figure Open in new tab Figure 4. QQplot of results from EWAS adjusting for 5 genetic principal components (5GPCs), 5 methylation population scores (5MPSs) and neither (None). A) using same sample size (n=1475); B) using optimized sample size (n=2695 for MPSs adjusted model (5MPSs) and unadjusted model (None); n=1475 for GPC adjusted model) Discussion We developed MPSs using DNAm data from multiple multi-ethnic cohorts via a two-step feature selection approach and evaluated their ability to capture population structure. The top three MPSs exhibited strong correlations with corresponding GPC and effectively separated self-reported race/ethnicity groups in the independent test dataset. On the other hand, MPCs derived from PCA using SNP adjacent CpG sites show inferior performance to MPSs in distinguishing across different race/ethnicity groups. The grouping of several individuals from distinct population backgrounds may be attributed to a certain degree of environmental effect captured by MPCs or the noise generated during the imputation step. While GPCs remain the gold standard to account for population structure in EWAS, our results indicate that MPSs offer comparable performance to GPCs and superior to unadjusted analyses, particularly when genetic data are unavailable. However, the moderate inflation factor suggests residual confounding that remains unaccounted for, likely attributable to unmeasured confounders or technical artifacts. This residual inflation underscores the importance of rigorous study design and analytical approaches to minimize confounding in EWAS 29 , 30 . The distinction of self-reported Hispanic backgrounds by MPSs among HCHS/SOL participants without genetic data further supports their consistency in capturing meaningful population structure. The closer distance between Mexican, Central American, and South American Hispanic/Latino groups may be caused by more substantial proportions of Amerindian genetic ancestry compared to other Hispanic/Latino groups 31 . In contrast, the genetic ancestry of Dominican and Puerto Rican individuals is less well captured by the available CpG sites in the current dataset. This limitation could result from the pre-processing/quality control step of the methylation data, where some CpG sites were removed/masked in order to reduce bias. Similar to the previously proposed methods to develop methylation-based variables that capture population structure by applying PCA over CpGs near SNPs 8 , 10 , 11 , our approach also relies on the assumption that some DNAm sites are associated with genetics 6 , 32 . Nevertheless, we select CpGs based on their associations with GPCs without assuming prior knowledge of ancestry-informative SNPs and flanking regions. Genetic PCs were treated as outcomes in our selection model to test the association between CpG sites and genetic PCs. While it is biologically reasonable to model CpG sites as outcomes, we retained genetic PCs as outcomes for consistency ( Figure 1 ) due to single-outcome requirement for flexible feature selection method. Note that it is appropriate to use GPCs as outcomes and CpG as exposures because covariates effects are “regressed out” of the CpGs based on the Frisch–Waugh–Lovell theorem, just like covariates would be “regressed out” of the GPCs if they are used as exposures while CpGs are outcomes 33 . This method is also less computational demanding than traditional PCA when applied to large cohort data. Such efficiency was gained by first filtering out DNAm methylation sites not associated with genetic PCs, which can also be achieved by pre-selecting those associated with genetic variants. To further account for environmental confounders and tissue heterogeneity, factors previously observed to influence the selection of methylation sites in SNP-based methods 11 , we incorporated these variables as covariates in our models. Nevertheless, the greater dispersion patterns of points within each group in the scatterplots ( Figure 3 ) and lower correlations between non-top MPSs and GPCs imply that these MPSs do not entirely captured the genetic structure in the data. It is possible that heterogeneity in methylation data attributable to cohort-specific differences in quality control and data processing limited inference and performance of MPSs. Another limitation of this study is the small number of Chinese participants (n= 69, Table 1 ), who primarily have East Asian genetic ancestry, whereas TOPMed individuals self-reporting other race/ethnicities typically have low levels of East Asian genetic ancestry (See Supplementary Figure 3 in Kurniansyah et al. 2023 34 ). This could partly explain their closer grouping towards Hispanics/Latinos participants, compared to when using GPCs ( Figure 3 ). This impacts the generalizability of our MPSs in distinguishing the Chinese population in other independent datasets. Both MPC_Rahmani derived from ancestry-informative SNPs and MPSs trained on GPCs effectively capture the prominent population structure. One limitation for the comparison with MPC created based on Rahmani et al. 10 is the limited overlap between their selected CpG and those available in our integrated dataset (1469 out of 4913 CpGs). Moreover, their reference CpGs were selected using 450k DNAm data from individuals of European ancestry, which may restrict generalizability to other populations and to DNAm data generated using different arrays. Future research could therefore focus on enhancing the reproducibility and transferability of DNAm-based population structure prediction across diverse platforms and ancestries, e.g. by creating imputation models for missing CpG sites. To facilitate the use of our computed MPSs for future studies, we have made the lists of CpGs selected for each GPC, along with the corresponding R code, publicly available at Supplementary Data 1 and on Zenodo repository. Conclusion Methylation-based scores predicting GPC, developed while integrating feature selection and adjustment for relevant confounders, provide a reliable estimate of population structure, showing strong concordance with GPCs, effective differentiation of racial and ethnic groups, as well as robust control of inflation in EWAS when genetic data are absent. With appropriate application, MPSs can complement GPCs to account for population structure in large cohorts when genetic data are unavailable for all or some individuals. Data Availability TOPMed freeze 10 WGS, methylation, and phenotype data are available by application to dbGaP according to the study specific accession: ARIC: phs001211, CARDIA: phs001612, JHS: phs000964, HCHS/SOL: phs001395. JHS methylation used in this manuscript are available via application to dbGaP, via accession phs000286. JHS methylation data can also be accessed through data use agreement to coordinating center (https://www.jacksonheartstudy.org/). HCHS/SOL methylation data used in this manuscript are available through application to the database of Genotypes and Phenotypes (dbGaP) accession phs000810, or via data use agreement with the HCHS/SOL Data Coordinating Center (DCC) at the University of North Carolina at Chapel Hill, see collaborators website: https://sites.cscc.unc.edu/hchs/. MPSs CpGs and weights will be provided at the Zenodo repository. Data Availability TOPMed freeze 10 WGS, methylation, and phenotype data are available by application to dbGaP according to the study specific accession: ARIC: phs001211, CARDIA: phs001612, JHS: phs000964, HCHS/SOL: phs001395. JHS methylation used in this manuscript are available via application to dbGaP, via accession phs000286. JHS methylation data can also be accessed through data use agreement to coordinating center (https://www.jacksonheartstudy.org/). HCHS/SOL methylation data used in this manuscript are available through application to the database of Genotypes and Phenotypes (dbGaP) accession phs000810, or via data use agreement with the HCHS/SOL Data Coordinating Center (DCC) at the University of North Carolina at Chapel Hill, see collaborators website: https://sites.cscc.unc.edu/hchs/. MPSs CpGs and weights will be provided at the Zenodo repository. Acknowledgements This work was supported by National Heart Lung and Blood Institute grant R01HL161012. Molecular data for the Trans-Omics in Precision Medicine (TOPMed) program was supported by the National Heart, Lung and Blood Institute (NHLBI). See the TOPMed Omics Support Table (Supplementary Note) for study specific omics support information. Core support including centralized genomic read mapping and genotype calling, along with variant quality metrics and filtering were provided by the TOPMed Informatics Research Center (3R01HL-117626-02S1; contract HHSN268201800002I). Core support including phenotype harmonization, data management, sample-identity QC, and general program coordination were provided by the TOPMed Data Coordinating Center (R01HL-120393; U01HL-120393; contract HHSN268201800001I). We gratefully acknowledge the studies and participants who provided biological samples and data for TOPMed. Study specific acknowledgements will be provided in the supplementary information. References 1. ↵ Peterson , R. E. et al. Genome-wide Association Studies in Ancestrally Diverse Populations: Opportunities, Methods, Pitfalls, and Recommendations . Cell 179 , 589 – 603 ( 2019 ). OpenUrl CrossRef PubMed 2. ↵ Jones , S. C. , Cardone , K. M. , Bradford , Y. , Tishkoff , S. A. & Ritchie , M. D. The Impact of Ancestry on Genome-Wide Association Studies . Pac Symp Biocomput 30 , 251 – 267 ( 2025 ). OpenUrl PubMed 3. ↵ Jin , B. , Li , Y. & Robertson , K. D. DNA Methylation: Superior or Subordinate in the Epigenetic Hierarchy? Genes Cancer 2 , 607 – 617 ( 2011 ). OpenUrl CrossRef PubMed 4. ↵ Yong , W.-S. Hsu , F.-M. & Chen , P.-Y. Profiling genome-wide DNA methylation . Epigenetics Chromatin 9 , 26 ( 2016 ). OpenUrl CrossRef PubMed 5. ↵ Jaenisch , R. & Bird , A. Epigenetic regulation of gene expression: how the genome integrates intrinsic and environmental signals . Nat Genet 33 , 245 – 254 ( 2003 ). OpenUrl CrossRef PubMed Web of Science 6. ↵ Min , J. L. et al. Genomic and phenotypic insights from an atlas of genetic effects on DNA methylation . Nat Genet 53 , 1311 – 1321 ( 2021 ). OpenUrl CrossRef PubMed 7. ↵ Champagne , F. A. Epigenetic influence of social experiences across the lifespan . Dev Psychobiol 52 , 299 – 311 ( 2010 ). OpenUrl CrossRef PubMed 8. ↵ Husquin , L. T. et al. Exploring the genetic basis of human population differences in DNA methylation and their causal impact on immune gene regulation . Genome Biol 19 , 222 ( 2018 ). OpenUrl CrossRef PubMed 9. ↵ Byun , H.-M. et al. Temporal Stability of Epigenetic Markers: Sequence Characteristics and Predictors of Short-Term DNA Methylation Variations . PLoS One 7 , e39220 ( 2012 ). OpenUrl CrossRef PubMed 10. ↵ Rahmani , E. et al. Genome-wide methylation data mirror ancestry information . Epigenetics Chromatin 10 , 1 ( 2017 ). OpenUrl CrossRef PubMed 11. ↵ Barfield , R. T. et al. Accounting for Population Stratification in DNA Methylation Studies . Genet Epidemiol 38 , 231 – 241 ( 2014 ). OpenUrl CrossRef PubMed 12. ↵ Koestler , D. C. et al. Blood-based profiles of DNA methylation predict the underlying distribution of cell types: a validation analysis . Epigenetics 8 , 816 – 26 ( 2013 ). OpenUrl CrossRef PubMed Web of Science 13. ↵ Taliun , D. et al. Sequencing of 53,831 diverse genomes from the NHLBI TOPMed Program . Nature 590 , 290 – 299 ( 2021 ). OpenUrl CrossRef PubMed 14. ↵ Mailman , M. D. et al. The NCBI dbGaP database of genotypes and phenotypes . Nat Genet 39 , 1181 – 1186 ( 2007 ). OpenUrl CrossRef PubMed Web of Science 15. ↵ Jun , G. , Wing , M. K. , Abecasis , G. R. & Kang , H. M. An efficient and scalable analysis framework for variant extraction and refinement from population-scale DNA sequence data . Genome Res 25 , 918 – 925 ( 2015 ). OpenUrl Abstract / FREE Full Text 16. ↵ Conomos , M. P. , Reiner , A. P. , Weir , B. S. & Thornton , T. A. Model-free Estimation of Recent Genetic Relatedness . The American Journal of Human Genetics 98 , 127 – 148 ( 2016 ). OpenUrl CrossRef PubMed 17. ↵ Triche , T. J. , Weisenberger , D. J. , Van Den Berg , D. , Laird , P. W. & Siegmund , K. D. Low-level processing of Illumina Infinium DNA Methylation BeadArrays . Nucleic Acids Res 41 , e90 – e90 ( 2013 ). OpenUrl CrossRef PubMed 18. ↵ Aryee , M. J. et al. Minfi: A flexible and comprehensive Bioconductor package for the analysis of Infinium DNA methylation microarrays . Bioinformatics 30 , 1363 – 1369 ( 2014 ). OpenUrl CrossRef PubMed Web of Science 19. ↵ Houseman , E. A. et al. DNA methylation arrays as surrogate measures of cell mixture distribution . BMC Bioinformatics 13 , 86 ( 2012 ). OpenUrl CrossRef PubMed 20. ↵ Ding , W. , Kaur , D. , Horvath , S. & Zhou , W. Comparative epigenome analysis using Infinium DNA methylation BeadChips . Brief Bioinform 24 , ( 2023 ). 21. ↵ Pidsley , R. et al. A data-driven approach to preprocessing Illumina 450K methylation array data . BMC Genomics 14 , 293 ( 2013 ). OpenUrl CrossRef PubMed 22. ↵ Leek , J. T. , Johnson , W. E. , Parker , H. S. , Jaffe , A. E. & Storey , J. D. The sva package for removing batch effects and other unwanted variation in high-throughput experiments . Bioinformatics 28 , 882 – 883 ( 2012 ). OpenUrl CrossRef PubMed Web of Science 23. ↵ Lin , D. Y. & Zeng , D. Meta-analysis of genome-wide association studies: no efficiency gain in using individual participant data . Genet Epidemiol 34 , 60 – 66 ( 2010 ). OpenUrl CrossRef PubMed Web of Science 24. ↵ Benjamini , Y. & Hochberg , Y. Controlling the False Discovery Rate: A Practical and Powerful Approach to Multiple Testing . Source: Journal of the Royal Statistical Society. Series B (Methodological) vol. 57 ( 1995 ). 25. ↵ Cole , S. R. & Hernán , M. A. Constructing inverse probability weights for marginal structural models . Am J Epidemiol 168 , 656 – 64 ( 2008 ). OpenUrl CrossRef PubMed Web of Science 26. ↵ Sofer , T. et al. Variant-specific inflation factors for assessing population stratification at the phenotypic variance level . Nat Commun 12 , 3506 ( 2021 ). OpenUrl PubMed 27. ↵ Wang , Z. et al. Methylation risk score of C-reactive protein associates sleep health with related health outcomes . Commun Biol 8 , 821 ( 2025 ). OpenUrl PubMed 28. ↵ Fraszczyk , E. et al. Epigenome-wide association study of incident type 2 diabetes: a meta-analysis of five prospective European cohorts . Diabetologia 65 , 763 – 776 ( 2022 ). OpenUrl PubMed 29. ↵ Campagna , M. P. et al. Epigenome-wide association studies: current knowledge, strategies and recommendations . Clin Epigenetics 13 , 214 ( 2021 ). OpenUrl CrossRef PubMed 30. ↵ Zheng , Y. et al. Design and methodology challenges of environment-wide association studies: A systematic review . Environ Res 183 , 109275 ( 2020 ). OpenUrl PubMed 31. ↵ Rao , H. et al. Advancements in genetic research by the Hispanic Community Health Study/Study of Latinos: A 10-year retrospective review . HGG Adv 6 , 100376 ( 2025 ). OpenUrl PubMed 32. ↵ Hannon , E. et al. Leveraging DNA-Methylation Quantitative-Trait Loci to Characterize the Relationship between Methylomic Variation, Gene Expression, and Complex Traits . The American Journal of Human Genetics 103 , 654 – 665 ( 2018 ). OpenUrl CrossRef PubMed 33. ↵ Lovell , M. C. A Simple Proof of the FWL Theorem . J Econ Educ 39 , 88 – 91 ( 2008 ). OpenUrl CrossRef 34. ↵ Kurniansyah , N. et al. Evaluating the use of blood pressure polygenic risk scores across race/ethnic background groups . Nat Commun 14 , 3202 ( 2023 ). OpenUrl CrossRef PubMed View the discussion thread. Back to top Previous Next Posted September 02, 2025. Download PDF Supplementary Material Data/Code Email Thank you for your interest in spreading the word about medRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following Estimating population structure using epigenome-wide methylation data Message Subject (Your Name) has forwarded a page to you from medRxiv Message Body (Your Name) thought you would like to see this page from the medRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share Estimating population structure using epigenome-wide methylation data Ziqing Wang , Kent D Taylor , Jerome I Rotter , Stephen S Rich , Yinan Zheng , Lifang Hou , Xiuqing Guo , Jan Bressler , Laura M Raffield , Yongmei Liu , Robert Kaplan , Donald M Lloyd-Jones , Alanna C. Morrison , Myriam Fornage , Tamar Sofer medRxiv 2025.09.01.25334865; doi: https://doi.org/10.1101/2025.09.01.25334865 Share This Article: Copy Citation Tools Estimating population structure using epigenome-wide methylation data Ziqing Wang , Kent D Taylor , Jerome I Rotter , Stephen S Rich , Yinan Zheng , Lifang Hou , Xiuqing Guo , Jan Bressler , Laura M Raffield , Yongmei Liu , Robert Kaplan , Donald M Lloyd-Jones , Alanna C. Morrison , Myriam Fornage , Tamar Sofer medRxiv 2025.09.01.25334865; doi: https://doi.org/10.1101/2025.09.01.25334865 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Epidemiology Subject Areas All Articles Addiction Medicine (568) Allergy and Immunology (863) Anesthesia (300) Cardiovascular Medicine (4435) Dentistry and Oral Medicine (444) Dermatology (382) Emergency Medicine (608) Endocrinology (including Diabetes Mellitus and Metabolic Disease) (1509) Epidemiology (15227) Forensic Medicine (30) Gastroenterology (1124) Genetic and Genomic Medicine (6597) Geriatric Medicine (668) Health Economics (997) Health Informatics (4534) Health Policy (1368) Health Systems and Quality Improvement (1613) Hematology (540) HIV/AIDS (1264) Infectious Diseases (except HIV/AIDS) (15916) Intensive Care and Critical Care Medicine (1103) Medical Education (623) Medical Ethics (146) Nephrology (667) Neurology (6599) Nursing (346) Nutrition (998) Obstetrics and Gynecology (1144) Occupational and Environmental Health (957) Oncology (3332) Ophthalmology (974) Orthopedics (369) Otolaryngology (420) Pain Medicine (436) Palliative Medicine (130) Pathology (663) Pediatrics (1693) Pharmacology and Therapeutics (691) Primary Care Research (711) Psychiatry and Clinical Psychology (5447) Public and Global Health (9230) Radiology and Imaging (2198) Rehabilitation Medicine and Physical Therapy (1370) Respiratory Medicine (1196) Rheumatology (593) Sexual and Reproductive Health (712) Sports Medicine (530) Surgery (712) Toxicology (99) Transplantation (289) Urology (265) (function(){function c(){var b=a.contentDocument||a.contentWindow.document;if(b){var d=b.createElement('script');d.innerHTML="window.__CF$cv$params={r:'a004672e3e2341e2',t:'MTc3OTU0Mjg5OA=='};var a=document.createElement('script');a.src='/cdn-cgi/challenge-platform/scripts/jsd/main.js';document.getElementsByTagName('head')[0].appendChild(a);";b.getElementsByTagName('head')[0].appendChild(d)}}if(document.body){var a=document.createElement('iframe');a.height=1;a.width=1;a.style.position='absolute';a.style.top=0;a.style.left=0;a.style.border='none';a.style.visibility='hidden';document.body.appendChild(a);if('loading'!==document.readyState)c();else if(window.addEventListener)document.addEventListener('DOMContentLoaded',c);else{var e=document.onreadystatechange||function(){};document.onreadystatechange=function(b){e(b);'loading'!==document.readyState&&(document.onreadystatechange=e,c())}}}})();

Text is read by the "Ask this paper" AI Q&A widget below. Extraction quality varies by source — PMC NXML preserves structure cleanly, OA-HTML may include some navigation residue, and OA-PDF can have broken hyphenation. The publisher copy (via DOI) is the canonical version.

My notes (saved in your browser only)

Ask this paper AI returns verbatim quotes from the full text · source: preprint-html

Answers must be backed by verbatim quotes from this paper's full text. Hallucinated quotes are dropped automatically; if no verbatim passage answers the question, we say so. How this works

Citation neighborhood (no data yet)

We don't have any in-corpus citations linked to this paper yet. This is a recent paper (2025) — citers typically take a year or two to land, and the OpenAlex reference graph may still be filling in.

Source provenance

europepmc
last seen: 2026-05-20T01:45:00.602351+00:00