Policy gradient-guided ensemble learning for enhanced polygenic risk prediction in ultra-high-dimensional genomics

preprint OA: closed
📄 Open PDF Full text JSON View at publisher

Abstract

Polygenic diseases challenge genetic risk prediction due to extreme dimensionality, low per-variant effect sizes, and non-additive interactions. Conventional marginal P -value-based methods potentially overlook subtle signals and complex dependencies, while inefficient random sampling in ensembles misses sparse signals. We introduce ELAG, an ensemble learning framework that advances feature bagging by reformulating variant selection as an approximate reinforcement learning problem. Leveraging policy gradients, a parameterized policy optimizes adaptive sampling—evolving beyond uniform strategies like random forests through importance-distribution guidance—to direct non-linear classifiers toward synergistic variants. This enables scalable navigation of millions of loci without pre-filtering, handling intricate architectures. In high-polygenicity, low-heritability simulations, ELAG boosted predictive accuracy (ΔAUROC = 0.0644). For neuro-immune diseases like rheumatoid arthritis, it enhanced AUROC from 0.6866 to 0.7354 and polygenic scores (e.g., Lassosum AUROC 0.7186 → 0.7543), outperforming mainstream methods. ELAG is robust to missing data, integrable with covariates, and yields interpretable variant sets enriched for pathways and networks. It can replace random sampling with learned guidance, advancing machine learning for ultra-high-dimensional data such as genetic risk prediction.
Full text 72,451 characters · extracted from preprint-html · click to expand
Policy gradient-guided ensemble learning for enhanced polygenic risk prediction in ultra-high-dimensional genomics | medRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-P4HH5NV'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search Policy gradient-guided ensemble learning for enhanced polygenic risk prediction in ultra-high-dimensional genomics View ORCID Profile Lihang Ye , Nan Lin , Ying Luo , Liubin Zhang , Meng Liu , Wenjie Peng , Xuegao Yu , Miaoxin Li doi: https://doi.org/10.1101/2025.09.23.25336425 Lihang Ye 1 Zhongshan School of Medicine, Sun Yat-sen University , Guangzhou, 510080, China 2 Key Laboratory of Tropical Disease Control (Sun Yat-sen University), Ministry of Education , Guangzhou, 510080, China Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Lihang Ye Nan Lin 1 Zhongshan School of Medicine, Sun Yat-sen University , Guangzhou, 510080, China 2 Key Laboratory of Tropical Disease Control (Sun Yat-sen University), Ministry of Education , Guangzhou, 510080, China Find this author on Google Scholar Find this author on PubMed Search for this author on this site Ying Luo 1 Zhongshan School of Medicine, Sun Yat-sen University , Guangzhou, 510080, China 2 Key Laboratory of Tropical Disease Control (Sun Yat-sen University), Ministry of Education , Guangzhou, 510080, China Find this author on Google Scholar Find this author on PubMed Search for this author on this site Liubin Zhang 1 Zhongshan School of Medicine, Sun Yat-sen University , Guangzhou, 510080, China 2 Key Laboratory of Tropical Disease Control (Sun Yat-sen University), Ministry of Education , Guangzhou, 510080, China Find this author on Google Scholar Find this author on PubMed Search for this author on this site Meng Liu 1 Zhongshan School of Medicine, Sun Yat-sen University , Guangzhou, 510080, China 2 Key Laboratory of Tropical Disease Control (Sun Yat-sen University), Ministry of Education , Guangzhou, 510080, China Find this author on Google Scholar Find this author on PubMed Search for this author on this site Wenjie Peng 1 Zhongshan School of Medicine, Sun Yat-sen University , Guangzhou, 510080, China 2 Key Laboratory of Tropical Disease Control (Sun Yat-sen University), Ministry of Education , Guangzhou, 510080, China Find this author on Google Scholar Find this author on PubMed Search for this author on this site Xuegao Yu 1 Zhongshan School of Medicine, Sun Yat-sen University , Guangzhou, 510080, China 2 Key Laboratory of Tropical Disease Control (Sun Yat-sen University), Ministry of Education , Guangzhou, 510080, China Find this author on Google Scholar Find this author on PubMed Search for this author on this site Miaoxin Li 1 Zhongshan School of Medicine, Sun Yat-sen University , Guangzhou, 510080, China 2 Key Laboratory of Tropical Disease Control (Sun Yat-sen University), Ministry of Education , Guangzhou, 510080, China Find this author on Google Scholar Find this author on PubMed Search for this author on this site For correspondence: limiaoxin{at}mail.sysu.edu.cn Abstract Full Text Info/History Metrics Supplementary material Data/Code Preview PDF Abstract Polygenic diseases challenge genetic risk prediction due to extreme dimensionality, low per-variant effect sizes, and non-additive interactions. Conventional marginal P -value-based methods potentially overlook subtle signals and complex dependencies, while inefficient random sampling in ensembles misses sparse signals. We introduce ELAG, an ensemble learning framework that advances feature bagging by reformulating variant selection as an approximate reinforcement learning problem. Leveraging policy gradients, a parameterized policy optimizes adaptive sampling—evolving beyond uniform strategies like random forests through importance-distribution guidance—to direct non-linear classifiers toward synergistic variants. This enables scalable navigation of millions of loci without pre-filtering, handling intricate architectures. In high-polygenicity, low-heritability simulations, ELAG boosted predictive accuracy (ΔAUROC = 0.0644). For neuro-immune diseases like rheumatoid arthritis, it enhanced AUROC from 0.6866 to 0.7354 and polygenic scores (e.g., Lassosum AUROC 0.7186 → 0.7543), outperforming mainstream methods. ELAG is robust to missing data, integrable with covariates, and yields interpretable variant sets enriched for pathways and networks. It can replace random sampling with learned guidance, advancing machine learning for ultra-high-dimensional data such as genetic risk prediction. Main Accurate prediction of complex polygenic diseases holds immense potential for preventive medicine, particularly for chronic conditions like Alzheimer’s disease, rheumatoid arthritis, and multiple sclerosis, where early intervention can fundamentally alter disease trajectories 1 - 6 . However, translating this potential into clinical reality is hindered by two fundamental and conflicting challenges inherent to genomic data. First, its extreme dimensionality—often millions of variants versus thousands of individuals—creates a severe p >> n problem that elevates the risk of overfitting 7 , 8 . Second, the genetic architecture of these diseases is profoundly complex, driven by myriad variants with small, non-additive effects that defy simple linear models 9 . The central task of genomic prediction is therefore to navigate the tension between aggressively reducing dimensionality and faithfully preserving this complex biological signal. To manage dimensionality, virtually all prediction pipelines, from polygenic risk scores (PRS) 10 - 12 to machine learning (ML) classifiers 13 - 18 , always rely on a preliminary feature filtering step. The de facto standard is the P -value thresholding approach (PTA) from genome-wide association studies (GWAS) 19 - 22 . While computationally tractable, PTA directly collides with the challenge of genetic complexity. By operating on marginal, additive statistics, it imposes a coarse, linear filter on a non-linear system, systematically overlooking variants whose importance emerges through interaction and is susceptible to selection bias (“winner’s curse”) 23 . Attempts to move beyond PTA have consistently failed to resolve the field’s central tension. Wrapper methods, such as ant-colony (ACO) 24 or genetic algorithms (GA) 25 , promise a more direct optimization of predictive performance but become computationally intractable at the genomic scale, forcing a reliance on the very PTA step they seek to replace. Integrated methods like decision-tree-based random forests 26 , 27 can capture non-linearities but face their scalability ceiling. In the ultra-high-dimensional genomic setting, their intrinsic random subspace sampling becomes inefficient, diluting the signal from causal variants to the point where an upstream PTA step is again required for feasible application 28 , 29 . This creates a pervasive methodological bottleneck: the field remains dependent on a filtering strategy that is fundamentally misaligned with the complex architectures we aim to model. Overcoming this bottleneck highlights the need for methodological advances. From the field of reinforcement learning, policy-gradient (PG) methods offer a powerful framework for optimizing complex decision-making strategies, from large language models 30 to experimental design in genomics 31 - 33 . Instead of static rules, PG learns a ‘policy’ that is iteratively refined to maximize a cumulative reward. Separately, in genomic machine learning, bootstrap aggregation (bagging) is a well-established ensemble technique for improving robustness and generalization in the face of high dimensionality 34 - 36 . However, a key challenge remains that standard bagging relies on random feature sampling or simple aggregation, which is inefficient for refined signal detection. Incorporating PG principles to optimize bagging represents a promising approach, potentially overcoming the bottleneck of marginal P -value filtering and enhancing feature selection, robustness, and generalization. Here, we introduce ELAG (Ensemble Learning with Adaptive Sampling Guided by Policy Gradient), a framework that operationalizes this concept. We demonstrate through extensive simulations that ELAG outperforms the baseline, particularly in challenging scenarios of high polygenicity and small sample sizes. We then validate ELAG on three neuro-immune disorders (Alzheimer’s disease, rheumatoid arthritis, and multiple sclerosis) in the UK Biobank (UKBB) 37 , showing that it consistently improves prediction performance over mainstream methods. We also demonstrate its robustness to genotype missingness and the ability to incorporate non-genetic covariates. Finally, ELAG produces biologically interpretable variant sets and enhances conventional PRS models, providing a versatile and clinically applicable tool for genomic risk prediction. Results Overview of ELAG ELAG is a machine learning framework designed to overcome the limitations of static feature selection in genomic risk prediction ( Figure 1 ). ELAG operationalizes a dynamic, learning-based approach by using PG optimization to guide adaptive sampling within an ensemble of bagged classifiers. This design enables the adaptive enrichment of informative variants from a vast pool, maintaining computational tractability while enhancing predictive power, particularly in challenging high-polygenicity and small-sample-size scenarios. Download figure Open in new tab Figure 1. Overview of the ELAG framework. The workflow of ELAG consists of four main stages. A . Data Preprocessing. The full cohort is split into training and test sets. After standard quality control (QC), the training data is divided into K subsets for cross-validation. Though ELAG is designed for genomic data initially, it also accepts covariates. Covariates can be concatenated with the variant subsets sampled by ELAG, and the combined feature vectors can be used for classifier training and evaluation. B . Sampling Strategy. Initially, a policy, initialized with information (e.g., GWAS P -values, CADD scores), defines a sampling probability for every variant. C . Policy Gradient Calculation. Multiple variant subsets are drawn from probability distribution sets. Ensemble classifiers are trained on each subset, and their aggregated predictive loss is calculated. Policy loss is computed to update the policy’s parameters, refining the sampling probabilities to favor variants improving model performance. The loop continues until training is completed. D . Testing and Application. The final trained ensembles are applied to the independent test set for disease risk prediction. The optimized variant subsets discovered by ELAG can also be extracted to enhance the performance of PRS methods. Thus, the framework is modular, allowing for flexible choices of input priors and downstream predictors. The core mechanism of ELAG replaces the P -value thresholding of conventional pipelines with a learnable, probabilistic sampling policy. This policy assigns a sampling probability to each variant, initialized using GWAS summary statistics and optionally augmented with other biological priors (e.g., CADD scores 38 ). During training within each bag, diverse variant subsets are sampled according to the current policy. These subsets are used to train base classifiers (e.g., XGBoost) capable of capturing complex, non-linear genotype-phenotype relationships. The predictive performance of the classifier ensemble serves as a reward signal, which is then used to update the policy parameters via a PG-inspired algorithm. This creates a synergistic loop: the policy is iteratively refined to favor variants that contribute to better prediction, while the bagging procedure ensures the reward signal is robust and promotes a generalizable solution. The final output of ELAG is a trained ensemble of models ready for individual risk prediction on independent test data. Furthermore, ELAG can function as a modular feature-selection engine. The prediction-optimized variant sets it identifies can be extracted and integrated into established downstream tools, such as conventional PRS methods, to significantly boost their performance. A detailed description of the methodology is provided in the Methods section. Simulation studies demonstrate ELAG’s superior predictive performance We performed extensive simulations to systematically benchmark ELAG against a standard PTA (as baseline) 39 . We designed scenarios to probe the method’s performance across key challenges in genetic prediction: varying heritability, sample size, and the degree of polygenicity (number of causal QTLs), while also assessing its compatibility with different classifiers and its ability to leverage external information ( Figure 2 ), based on a linear–nonlinear mixed model (LNMM). Unless specified otherwise, simulations used a balanced case-control design, pairwise in low linkage disequilibrium ( r 2 < 0.1), and XGBoost as the base classifier. More simulation settings are described in Methods. Download figure Open in new tab Figure 2. Results of simulations. A, B . Comparison of AUROC and MCC between the baseline method (dashed lines) and ELAG (solid lines) across varying sample sizes and heritability ( h 2 ) levels. ELAG’s advantage is most pronounced in low-power settings (small sample size, low h 2 ). C . Performance comparison (AUROC and AUPR) between ELAG combined with various machine learning classifiers and the baseline method, demonstrating ELAG’s model-agnostic versatility. D . Performance gains (AUROC and KS statistic) from integrating external priors. External GWAS and functional scores improve ELAG’s predictive accuracy, with robustness preserved under noisy annotations. E . Improvement in predictive performance (AUROC and MCC) of two standard PRS methods (PRSice, Lassosum) when using variant subsets selected by ELAG. F . Comparison of true causal QTLs identified by the baseline method versus the collective set identified across all classifiers in the ELAG ensemble. ELAG recovers a vastly larger number of true signals. G, H . Distribution and heatmap of pairwise Jaccard similarity for the QTL sets selected by classifiers in ELAG. The low similarity indicates that ELAG achieves its high QTL recovery rate by selecting diverse and complementary feature sets. ELAG excels in low-power and highly polygenic scenarios A primary challenge in genomics is detecting subtle genetic signals in cohorts of limited size. We first evaluated ELAG in such settings, simulating traits with 1,000 causal QTLs under varying sample sizes. As demonstrated across different heritability levels ( Figure 2A and 2B and Supplementary Table 1), ELAG’s performance advantage over the PTA baseline was most pronounced at smaller sample sizes. For a sample size of 1,000, ELAG delivered substantial gains across all heritability levels (e.g., ΔAUROC = 0.0588–0.0736). The improvement was particularly striking for the low heritability ( h 2 =0.2) trait, where ELAG achieved a 64.9% relative improvement in the Matthews Correlation Coefficient (MCC). The difference attenuated with larger sample sizes, with AUROC improvements reduced to 0.0237–0.0305 at the sample size of 10,000. View this table: View inline View popup Download powerpoint Table 1. Performance comparison of ELAG, PTA, ACO and GA across three diseases. Multi-metric classification performance under the best model configurations is reported, highlighting ELAG’s superior power and robustness across AD, RA and MS cohorts. The advantage of ELAG was also magnified in highly polygenic architectures, where individual variant effects are weaker (Supplementary Table 2). With 1,000 QTLs and a sample size of 1,000, ELAG improved AUROC by 0.064 over PTA. In contrast, for a less polygenic trait with only 100 QTLs, the gain was a modest 0.011. As expected, these results may highlight a fundamental limitation of PTA: its reliance on strong marginal effects causes it to fail when signals are subtle and distributed across many loci. ELAG’s adaptive sampling, however, may effectively enrich for these weaker, cumulative signals, conferring a distinct advantage in the very scenarios that are most challenging for genetic discovery. View this table: View inline View popup Table 2. Performance metrics for three diseases (AD, RA, MS) using feature-selected PRS (feature selection method + PRS) versus standard PRS. ELAG-integrated PRS models outperform both standalone PRS and other wrapper-based feature selection approaches across AD, RA and MS, demonstrating enhanced discriminative power and robustness in polygenic risk prediction. Framework versatility and integration of external priors We next confirmed that ELAG’s benefits are not specific to a single classifier. We tested its performance as an enhancement module for five different ML models, including AdaBoost, Naïve Bayes, CatBoost, and Gradient Boosting Decision Tree (GBDT). In a moderately powered setting ( N =3,000), ELAG consistently and substantially boosted the predictive performance of every classifier tested ( Figure 2C and Supplementary Table 3). For instance, it increased the AUROC of XGBoost from 0.6585 to 0.7082 and CatBoost from 0.642 to 0.691, potentially demonstrating its utility as a versatile, model-agnostic front-end. More details can be seen in Supplementary Note 1.1. View this table: View inline View popup Download powerpoint Table 3. Predictive performance of classifiers retrained using ELAG-aggregated top 1% variants. For each disease (AD, RA, MS), classifiers were retrained on the training set using only variants in the top 1% of aggregated XGBoost importance gains (zero-gain variants excluded) and evaluated on independent held-out test sets. Beyond its adaptability, ELAG can seamlessly integrate external knowledge to guide its sampling policy. To test this, we used a low-power scenario ( N =1,000, h 2 =0.2) where priors from a larger GWAS ( N =40,000) and informative functional annotations (CADD-like scores) were available. Incorporating GWAS statistics alone dramatically improved ELAG’s AUROC from 0.6324 to 0.7264 ( Figure 2D and Supplementary Table 4). The addition of informative functional scores provided a further boost to an AUROC of 0.7412. As a critical negative control, introducing shuffled scores reduced performance as expected (AUROC = 0.7184, KS = 0.40), yet ELAG still outperformed PTA supplemented with GWAS information (AUROC = 0.6592, KS = 0.32). These results demonstrate that ELAG may effectively harness external resources to further refine its search for causal variants to enhance predictive performance, while maintaining robustness in the presence of imperfect or misleading functional annotations. ELAG enhances standard polygenic risk score methods To assess its translational potential, we evaluated ELAG as a feature engineering front-end for two widely used PRS methods, PRSice 40 and Lassosum 41 . We used the variant subsets enriched by ELAG to construct ensemble PRS models. This strategy yielded dramatic performance improvements over the standard application of these tools ( Figure 2E and Supplementary Table 5). For PRSice, ELAG-selected variants increased the AUROC from 0.7640 to 0.8192 and the MCC from 0.4270 to 0.5454 in a local cohort of 500/500 cases/controls and an external GWAS sample of 20,000/20,000. Lassosum, which had a lower baseline, saw its MCC climb from 0.3746 to 0.5204. These results show that by replacing the coarse, PTA-based filtering inherent in standard PRS workflows with ELAG’s adaptive selection, we may unlock a more diverse and potent set of predictive variants, significantly enhancing the utility of these established tools. Diverse feature selection underlies ELAG’s performance gain Next, we investigated the mechanism behind ELAG’s superior performance. In a challenging simulation ( h 2 =0.2, 1,000 QTLs, N =1,000), we compared the variants selected by an optimized PTA pipeline against those selected across ELAG’s ensemble. The PTA using the top 100 variants identified a total of 60 true QTLs. In stark contrast, the 50 base classifiers with 100 feature-size sampling in ELAG collectively identified 845 distinct QTLs ( Figure 2F ). This vast improvement in coverage was achieved through profound diversity; the QTLs chosen by any two classifiers in the ensemble were almost entirely different, with a mean Jaccard similarity of just 0.0262 ( Figure 2G and 2H ). This suggests that while PTA may converge on a small set of variants with the strongest marginal signals, ELAG’s adaptive, ensemble-based search may encourage the exploration of a much wider range of weaker, complementary signals. This diversity in feature selection may be the key mechanism enabling ELAG to capture a more comprehensive picture of a trait’s complex genetic architecture. Ablation Study Finally, we conducted an ablation study ( h 2 =0.2, 1,000 QTLs) to analyze the contribution of different modules in ELAG. Specifically, we introduced a variant without the policy gradient (VWPG), which relies only on bagging and sampling, and compared it against both the simple P -value filtering baseline and the full ELAG model (Supplementary Table 6). With a sample size of N =1000, ELAG achieved an AUC-ROC of 0.6324, compared to 0.6008 for MWPG and 0.5680 for the baseline. Similarly, the KS statistic improved from 0.16 (baseline) and 0.20 (MWPG) to 0.28 with ELAG, while MCC increased from 0.1633 (baseline) to 0.2209 (MWPG) and 0.2693 (ELAG). As expected, at larger sample sizes, the superiority of ELAG declined. For instance, at NN =10000, ELAG reached an AUC-ROC of 0.7180 and an MCC of 0.3481, outperforming MWPG (0.7123/0.3220) and the baseline (0.6927/0.2954). These results demonstrate that while bagging and sampling already provide substantial gains over the baseline, incorporating the strategy gradient in ELAG further enhances stability and predictive performance, particularly in scenarios of limited sample sizes. ELAG enhances prediction and provides biological insights for complex human diseases To validate ELAG in real-world settings, we applied it to three complex neuro-immune disorders from the UK Biobank: Alzheimer’s disease (AD), rheumatoid arthritis (RA), and multiple sclerosis (MS). Following standard QC, the datasets contained 1,673,294 variants and 4,227 individuals for AD, 1,775,223 variants and 2,008 individuals for RA, and 1,712,010 variants and 3,258 individuals for MS. All of the datasets had an equal number of cases and controls. The Manhattan plots corresponding to the three diseases are shown as Supplementary Figure 1. We benchmarked a GWAS-informed 42 - 44 ELAG model (Supplementary Table 7 and Supplementary Note 1.2.1) against a P -value thresholding baseline and two wrapper methods (ACO 45 , GA 46 ), with all performance metrics reported on a held-out test set. ELAG consistently outperforms alternative feature selection methods Across all three diseases, ELAG demonstrated substantially and consistently superior predictive performance ( Figure 3A and 3B and Table 1 ). For RA, ELAG achieved an AUROC of 0.7354, an improvement over the range of other methods (0.6241–0.6866). For AD, ELAG’s AUROC of 0.7237 surpassed the 0.6699–0.6841 of competitors. Similarly, for MS, ELAG raised the AUROC to 0.7185 from a PTA baseline of 0.6556. These gains were mirrored across other key metrics. For instance, in AD, ELAG’s MCC of 0.351 represented a gain of up to 0.0968 over alternatives, while its F1-score reached 0.7034, driven by enhanced sensitivity in case detection. Download figure Open in new tab Figure 3. ELAG performance and biological interpretation on three complex diseases. A , B . ELAG consistently outperforms baseline methods (PTA, ACO, GA) in predicting Alzheimer’s disease (AD), rheumatoid arthritis (RA), and multiple sclerosis (MS), as shown by bar plots of AUROC, AUPR, and MCC ( A ) and summarized in radar plots ( B ). Red numbers denote the relative performance gain of ELAG compared to the lowest-performing method. C . Variant sets prioritized by ELAG boost the predictive performance of two standard PRS tools, Lassosum and PRSice, across all three diseases. D . Distribution of average gain values for the top-ranked variants in each disease and their relationship with –log 10 ( P ) values from GWAS, highlighting complementary signals. E . The predictive power (AUROC) of the top 1% of ELAG-selected variants (red line) is highly significant compared to the null distribution from 1,000 randomly selected variant sets of the same size. F . Top biological pathways enriched among genes associated with the highest-ranking ELAG variants, revealing disease-relevant mechanisms for AD, RA, and MS. G . A substantial fraction of top 1% ELAG-prioritized variants showed nonsignificant GWAS P -values, highlighting ELAG’s ability to capture signals that may be missed by conventional association analyses. To simulate clinically relevant scenarios with imperfect data, we evaluated ELAG’s robustness to increasing levels of individual genotype missingness (from 20% to 80%) in the test set. The framework showed remarkable resilience. Even at 20% missingness, the drop in AUROC was minimal (AD: -0.0068; RA: -0.0008; MS: -0.0015). Critically, even when faced with 80% missing data, ELAG still retained over 95% of its original AUROC across all three diseases (AD: 0.6983; RA: 0.7069; MS: 0.6879). This stability underscores ELAG’s potential for deployment in real-world clinical applications where genomic data may be incomplete (Supplementary Table 8 and Supplementary Note 1.2.2). ELAG variant sets boost the performance of standard polygenic risk scores We next evaluated ELAG’s capacity to function as a modular feature selection engine for two PRS methods, PRSice and Lassosum. By using the variant subsets identified by ELAG as input for these tools, we observed significant performance gains compared to their standard application across all tested diseases ( Figure 3C and Table 2 ). Specifically, the integration of ELAG with Lassosum for AD increased the AUROC from a baseline of 0.6514 to 0.7207, approaching the performance of the ELAG-only model (0.7237). Similar improvements were observed for MS), where the ELAG-Lassosum model improved the AUROC from 0.6579 to 0.6925 (ELAG-only: 0.7185). Notably, for RA, the synergistic effect was even more pronounced: the ELAG-Lassosum model achieved an AUROC of 0.7543, which not only surpassed the Lassosum baseline (0.7186) but also exceeded the performance of the ELAG-only model (0.7354). In contrast, variant sets generated by alternative algorithms (ACO and GA) provided negligible or even detrimental effects on PRS performance. These results suggest that ELAG’s adaptive, performance-driven selection strategy can identify a more potent and generalizable set of variants, thereby enhancing the predictive power of established genomic tools. ELAG identifies compact, predictive, and potentially biologically relevant variant sets To interpret the models, we analyzed the variants prioritized by the XGBoost ensemble for each disease, ranking them by their average feature importance (‘gain’) 47 , 48 . We found that a remarkably compact subset of these variants could recapitulate the full model’s performance. For AD, the top 790 variants (top 1%) achieved an AUROC of 0.6683, recovering 92.3% of the full model’s performance. For RA and MS, the top 1% of variants recovered over 96% and 95% of the full model’s AUROC, respectively ( Table 3 ). This high performance was statistically significant, vastly exceeding that of 1,000 randomly selected variant sets of equivalent size ( P < 10 −3 for all diseases, Figure 3E ). Crucially, these predictive variant sets were enriched for disease-relevant biology ( Figure 3F ) using g:Profiler 49 . For AD, prioritized genes converged on pathways including ‘MHC protein complex binding’ and the ‘adaptive immune system,’ both implicated in AD-related neuroinflammation 50 , 51 (more details can be seen in Supplementary Table 9). For RA, we observed strong enrichment for ‘antigen processing and presentation via MHC class II’ and ‘cytokine-mediated signaling’—vital pillars of RA pathogenesis 52 , 53 (more details can be seen in Supplementary Table 10). Similarly, for MS, top variants were linked to ‘MHC class II receptor activity’ and ‘T-cell activation,’ aligning with its autoimmune etiology 54 , 55 (more details can be seen in Supplementary Table 11). For full description can be seen in Supplementary Note 1.2.3. A key finding was ELAG’s ability to prioritize variants with modest GWAS P -values that would be missed by standard filtering ( Figure 3G ). Within the top 1% of prioritized variants, we identified 702 variants in AD, 591 in RA, and 7 in MS, spanning multiple genes and intergenic regions—all of which may not reach genome-wide significance in GWAS. Representative examples include: rs9272320 ( HLA-DQA1 ; ELAG rank=10 vs. GWAS P =0.0189, rank=35,512), implicated in AD neuroimmunity 56 ; rs3130571 ( PSORS1C1 ; ELAG rank=27 vs. P =0.00340, rank=9,539) in RA 57 ; and rs9274440 ( HLA-DQB1 ; ELAG rank=24 vs. P =0.00220, rank=6,449) in MS 58 , 59 . These results may suggest that ELAG may effectively uncover potential variants whose contributions arise from non-additive or complex effects not captured by marginal statistics, though the true functional roles of these variants require further validation. By providing a principled way to discover such hidden signals, ELAG not only improves prediction but may also offer a powerful engine for generating novel, testable biological hypotheses. Full variant lists are provided in Supplementary Tables 12-14. Incorporation of non-genetic covariates To further showcase ELAG’s versatility and robust capability, we investigated its capacity to integrate diverse non-genetic covariates with adaptively selected genetic variant subsets. Utilizing Alzheimer’s disease (AD) as a compelling case study, where age is a well-established and critical determinant, we observed that genetic variants alone yielded an AUROC of 0.7237. Strikingly, upon the incorporation of age at recruitment (a practical surrogate for the generally unrecorded exact onset age) and sex, ELAG’s predictive performance significantly improved, achieving an AUROC of 0.8658 (Supplementary Figure 2, Supplementary Table 15, and Supplementary Note 1.2.4). These findings underscore ELAG’s exceptional ability to effectively synthesize heterogeneous risk factors, leading to enhanced prediction accuracy and broadening its potential applications in comprehensive disease risk modeling. Discussion In this work, we introduced ELAG, a machine learning framework that reframes genomic feature selection from a static filtering problem into a dynamic, adaptive learning task. The central innovation of ELAG is its PG-driven sampling strategy, which learns to select variants by directly optimizing for predictive performance. This approach breaks the reliance on coarse, static P -value thresholds that constrain conventional methods. By creating a closed feedback loop where an ensemble of classifiers guides the sampling policy, ELAG can effectively explore the genomic space to uncover a richer, more potent set of disease-associated variants, including those with subtle, non-additive, or context-dependent effects that are systematically missed by standard pipelines. The power of this adaptive framework was borne out in our extensive evaluations. In simulations designed to mimic the challenging aspects of genomic data—low heritability 60 , small sample sizes 61 , and high polygenicity 62 —ELAG consistently and substantially outperformed the PTA baseline, with average gains of ΔAUROC = 0.0644 and ΔMCC = 0.106. Analysis of feature diversity further revealed low Jaccard similarity ( J < 0.1) among ensemble members, indicating that ELAG explores complementary feature subspaces and thereby captures signals beyond those prioritized by marginal P -value filtering. This simulated advantage translated directly to real-world performance on three complex neuro-immune disorders (AD, RA, and MS). ELAG substantially improved standalone prediction performance (e.g., AUROC for RA increased from 0.6866 to 0.7354) and further enhanced established PRS methods (e.g., Lassosum AUROC for RA from 0.7186 to 0.7543). These consistent gains validate ELAG’s core premise: that a performance-guided, adaptive search may enrich biologically relevant, sub-threshold signals beyond the reach of conventional thresholding. Beyond its predictive performance, ELAG is designed for real-world clinical utility. A key requirement for real-world deployment is the ability to generate reliable predictions under imperfect data conditions and heterogeneous sources of information. In our analyses, ELAG demonstrated strong robustness to high levels of genotype missingness, a common challenge in clinical datasets, indicating its capacity to deliver stable risk estimates even when genomic data are incomplete. Moreover, its extensible framework enables the seamless incorporation of clinically relevant covariates (e.g., age and sex) alongside adaptively selected genetic variants, thereby enhancing predictive performance potentially and capturing disease risk more comprehensively. Taken together, these properties suggest that ELAG is well-suited for translational applications where both data quality and multimodal integration are critical for precision risk prediction. Furthermore, ELAG serves as not just a predictive tool but also a potential engine for biological discovery. The compact, high-performance variant sets it identifies are enriched for disease-relevant biological pathways, suggesting that the model may capture biologically meaningful signals. Critically, some of these predictively important variants lack genome-wide significance in standard GWAS, underscoring ELAG’s ability to computationally recover candidate causal variants that may be obscured by stringent multiple-testing corrections in association studies. By prioritizing variants based on their contribution to predictive performance rather than solely on marginal statistical effects, ELAG may provide a principled avenue for generating testable hypotheses about the complex genetic architecture of disease. Our study has several limitations that suggest future directions. First, while ELAG highlights statistically and biologically plausible variants, their causal roles require experimental validation. Second, its ensemble design, though key to robustness, is more computationally demanding than single-model methods; algorithmic and parallelization advances could improve scalability. Third, this work is an initial attempt to apply PG ideas to genomic feature selection, with scope for refining the policy-learning strategy to enhance efficiency and interpretability. Finally, although we incorporated GWAS and functional annotation priors, the framework is readily extensible to multi-modal data, which may further improve predictive performance and biological insight. In conclusion, ELAG establishes a powerful new paradigm for polygenic risk prediction by synergistically integrating ensemble learning with PG optimization. By moving beyond static filtering, it mitigates a core bottleneck in genomic analysis. Its demonstrated success in both challenging simulations and real-world disease cohorts paves the way for a new class of intelligent, adaptive models at the intersection of machine learning and precision medicine. Methods The ELAG framework Overall architecture ELAG is a machine learning framework designed for polygenic risk prediction from high-dimensional genomic data. The workflow begins with genotype data that has undergone standard quality control to preserve a large variant pool. ELAG’s core is a PG-inspired algorithm that dynamically samples variant subsets, which are then used to train an ensemble of base classifiers. The collective performance of this ensemble provides a reward signal that iteratively refines the sampling policy. The final output is a trained ensemble model for predicting binary disease status (case/control). The training process is embedded within a five-fold cross-validation scheme to ensure robust evaluation. In each fold, the data is partitioned into an 80% training set and a 20% validation set. Within each fold, the framework executes multiple rounds of policy-guided sampling and model training. Specifically, for each round, a subset of variants is sampled based on the current policy, and a new base classifier is trained on this subset. The predictions from all trained classifiers in the ensemble are averaged, and the resulting binary cross-entropy loss on the validation set is used to compute the policy gradient. This gradient informs the update of the policy parameters, creating a closed loop where predictive performance directly guides feature selection. The final model selected for testing is the ensemble that achieves the highest AUROC on the validation data across all training rounds. Novel policy gradient-inspired algorithm As illustrated in Figure 1 , ELAG assigns each variant a sampling probability pro i based on Equation (1) , which reflects its combined importance derived from external information. where |α v | acts as an adaptive temperature value with v corresponding to the cross-validation fold, which modulates the sharpness of the sampling distribution. It is dynamically updated according to classification performance, thereby adjusting the selection bias over variants. p i represents the P -value of variant i and s i denotes the corresponding P -value for variant i from external GWAS summary statistics information. The function g ( p i , s i ) represents a trainable combination of p i and s i , parameterized by a learnable weight β : where β serves as a tunable scalar that balances the contribution of external GWAS-derived signals (e.g., P -values) and annotation scores (e.g., CADD). The sigmoid function constrains β to the interval (0, 1), ensuring a balance between data sources. When annotation scores are available, C i represents the corresponding annotation score for that variant i. f w ( C i ) is a weighting function based on the input C i , the annotation score for that variant i ; its explicit form is given by For backpropagation,we establish the following gradient equation (4) to update the parameters in the framework through loss. where the gradient is computed as , where x j represent a learnable parameter. ϰ denotes the set of variants selected in the current iteration, ℒ is the loss value, λ is a tunable hyperparameter, and n is the number of learnable parameters. μ pro denotes the average probabilities assigned to all variants. The gradient is constructed such that subsets of variants whose combined sampling probability ∑ j ∈ϰ pro i exceeds the global average |ϰ| · μ pro are preferentially reinforced in proportion to their impact on the binary cross-entropy loss ℒ. This design helps to mitigate fluctuations caused by varying batch sizes and ensures stability during optimization. L 2 regularization term imposes weight decay to improve generalization. Furthermore, by operating on the raw sum of probabilities ∑ j ∈ϰ pro i rather than the log-sum ∑ j ∈ϰ l o g( pro i ), this formulation avoids numerical instability caused by extremely small probability values. When the probability vector is derived from Equation (1) over more than 1,000,000 variants, the resulting probabilities can become exceedingly small. Applying a logarithm to such values would produce large negative numbers, leading to unstable gradients and potential exploding loss values. By using the raw sum, we maintain both numerical stability and a meaningful signal for guiding the adaptive selection of informative variant subsets in high-dimensional settings. Thus, the parameters can be updated by gradient descent, as follows: Equation (5) defines the update rule that adjusts the learnable parameters from iteration t to t +1 based on the computed gradient and the learning rate η . Incorporation of non-genetic covariates ELAG is designed primarily for genomic data but readily accommodates non-genetic covariates (e.g., clinical phenotypes) when they are relevant to disease. Inclusion of non-genetic covariates is optional and not the primary focus of this study. For the real-world analyses, we concatenated covariates—specifically age at recruitment and sex—with the variant subsets sampled by ELAG; the combined feature vectors were then used jointly for classifier training and evaluation. For sample i , let sex i denote sex (encoded as 0 = female, 1 = male), age i denote age at recruitment, and G i 1 , …, G in denote the n variant features collected for that sample. Each ELAG-sampled variant subset was concatenated with the covariates as and these combined vectors were used for classifier training and subsequent evaluation. Model training and testing The genomic dataset was randomly divided into a training set (90%) and an independent testing set (10%). A 1:1 case-to-control ratio was maintained in all datasets to ensure balanced class representation during model training and evaluation. We implemented five machine learning classifiers: AdaBoost, Naive Bayes, CatBoost, GBDT, and XGBoost. Hyperparameter tuning was conducted using grid search based on the scikit-learn (v1.3.0) framework. The models yielding the highest AUROC on the validation sets were selected for final evaluation. Our proposed policy gradient–inspired optimization algorithm was implemented in PyTorch (v2.0.1). The learning rate was set to 0.1, with an L 2 regularization coefficient ( λ ) of 0.01. The models were trained for 20 iterations to ensure performance stability. For both ACO 45 and GA 46 , we wrapped XGBoost–based AUROC on a validation split as the fitness function and used grid search to identify optimal hyperparameters (population size, generations, crossover/mutation rates for GA; number of ants, iterations, pheromone evaporation, α/β weights for ACO). Due to the exponential growth of the combinatorial search space and resulting computational constraints, we first filtered variants by GWAS P < 0.01, reducing the candidate pool to 19,231 variants for AD, 21,101 for RA, and 21,320 for MS before applying these wrapper methods. Polygenic Risk Score Methods In this study, we applied two widely used PRS approaches—PRSice 40 and Lassosum 41 —as our genetic risk predictors. To integrate these with ELAG, we used ELAG as a feature selector. In our project, ELAG was trained using cross-validation, and in each fold, multiple bagged classifiers were generated, each selecting its own subset of variants. For example, with 5-fold cross-validation and 10 classifiers per fold, this resulted in 50 distinct feature sets across all folds. We then passed each of these ELAG-selected variant sets into the PRS pipelines. For each PRS method, scores were computed on the test set separately for each feature set. The resulting 50 PRS scores were then averaged to produce a single ELAG–enhanced PRS prediction for each individual (Supplementary Figure 3). As a baseline, we also computed PRS using the full genome-wide variant set, without ELAG-based feature selection, to assess the added predictive value of integrating ELAG with standard PRS methods. Model interpretation In order to evaluate the contribution of each SNP feature to the predictive performance of the XGBoost models, we adopted the gain metric 47 , 48 , which quantifies the average improvement of the objective function when a feature is used for a split. Specifically, for a given split, the gain is defined as: where G L and G R denote the sums of the first-order gradients in the left and right child nodes, respectively, while H L and H R denote the corresponding sums of the second-order gradients. The term λ g and is the L2 regularization coefficient, and γ is the minimum split loss parameter that penalizes overly complex trees. Intuitively, the gain measures the reduction in the loss function after a split compared with the parent node, adjusted by regularization. The overall feature importance of each SNP was obtained by aggregating the gains across all occurrences of that feature in all trees and reporting their average values, following the default definition of importance_type=‘gain’ in XGBoost. After training an ensemble of classifiers, we ranked all variants by their gain and discarded those with zero gain, retaining only the top 1% of variants. We then assessed the predictive power of this subset by retraining the classifiers on the training data using only these top variants and evaluating performance on the held-out test set. As a null benchmark, we performed the same procedure 1,000 times with randomly sampled variants of equal cardinality, training and testing under identical conditions. Empirical P -values were estimated via Monte Carlo 63 , defined as where m is the number of random replicates whose background AUROC met or exceeded the observed AUROC. For the three diseases studied, the total number of variants involved was 790 for AD, 888 for RA, and 82 for MS. To investigate the biological functions associated with the genes linked to these variants, we performed functional enrichment analysis using g:Profiler 49 . Analyses included Gene Ontology (GO) 64 , Human Protein Atlas (HPA) 65 , and Reactome (REAC) 66 pathways. Furthermore, we also quantified how many of our top-ranking variants may be missed by a standard GWAS threshold. Specifically, we first took the top 1% of variants by XGBoost gain importance and, for each of those variants, retrieved its GWAS P -value. We then sorted these P -values in ascending order to assign each variant a rank. Next, we compared the rank of each variant against the optimal feature set size identified by our grid search (the number of variants that maximized classifier performance). Any variant whose P -value rank exceeded that optimal set size—meaning it would not have been selected by a hard cutoff at that rank— was deemed potentially overlooked by conventional association analysis. Jaccard similarity To quantify the overlap between different sets of selected QTLs, we employed the Jaccard similarity coefficient. For any two sets A and B , the Jaccard index is defined as where | A ∩ B | is the number of elements common to both sets and | A ∪ B | is the total number of unique elements across both sets. A higher Jaccard index indicates greater agreement between two selection methods or replicates, whereas a lower value reflects more divergent QTL choices. Simulation data Genotypes for 2,596 variants on chromosome 1 were simulated based on allele frequencies and linkage disequilibrium (LD) coefficients from the European (EUR) panel of the 1000 Genomes Project 67 . Genotypes were encoded as 0, 1, or 2, representing the number of alternative alleles. Only variants with a minor allele frequency (MAF) above 5% and pairwise LD r 2 <0.1 were retained to ensure low redundancy. Given the genotypes, subjects’ and phenotypes were simulated using a liability-threshold model 68 , 69 . The relationship between multiple variant genotypes and liability are present in the following equation (9) , a linear–nonlinear mixed model (LNMM). where the effect sizes γ and β were drawn from a uniform distribution between 0 and 1, with X denoting the genotype of a given variant. We set k 1 and k 2 , corresponding to the numbers of linear and nonlinear QTLs, respectively, where k 1 = k 2 =500. The environmental component ε followed a normal distribution N (0, σ 2 ). For each individual, the total genetic liability was computed as the sum of the effects of alternative alleles at causal QTLs, combined with a random environmental contribution. The variance of environmental noise was calibrated to achieve a targeted heritability h 2 (0.2, 0.3, 0.4). A synthetic population of 3 million individuals was generated. Individuals in the top 1% of the liability distribution were labeled as cases, simulating a disease prevalence of 1%; the rest were treated as controls. From this population, case/control cohorts of sizes 500, 1,500, and 5,000 were randomly sampled without replacement for downstream analysis. Functional annotation scores in simulations were generated under two conditions: (1) informative annotations, where true QTLs received scores drawn from N ( μ = 0.75, σ 2 = 0.1) and non-QTLs from N ( μ = 0.25, σ 2 = 0.1); and (2) shuffled annotations, in which those same scores were randomly permuted across variants. All simulated scores were truncated to the [0, 1] interval. Real data: UK Biobank All genomic data and phenotypes in the real datasets are from the UK Biobank with the ICD-10 diagnoses. All individuals included in the study were born in the UK. The AD, RA, and MS sub-datasets have sample sizes of 4,227, 2,008, and 3,258, respectively. The case-to-control ratio is maintained at 1:1 in each of the three datasets. Variants were preprocessed and analyzed by PLINK 2.0 70 and KGGA ( https://pmglab.top/kgga/ ) 71 based on the following criteria: (1) subject call rate ≥ 98%; (2) variant call rate ≥ 98%; (3) Hardy-Weinberg equilibrium P -value ≥ 1 × 10 −5 ; (4) MAF ≥ 5%; (5) LD pruning threshold pp 2 < 0.9. The model focused on autosomal chromosomes. After applying standard quality control, the number of variants per sample is 1,673,294 for AD, 1,775,223 for RA, and 1,712,010 for MS. Computational implementation The computational experiments were carried out on a dedicated desktop workstation configured as follows: 32 GB of dual-channel GLOWAY DDR5 RAM clocked at 6000 MHz; an Intel ® Core™ i7-13700K CPU (3.4 GHz base, up to 5.4 GHz turbo); a 2 TB PCIe 4.0 NVMe SSD (aigo P7000Z series); and an NVIDIA GeForce RTX 4090 GPU with 24 GB of GDDR6X memory. As an illustrative example in the Alzheimer’s disease case study, a single CPU core dynamically sampled 5,000 variants per classifier (5-fold cross-validation repeated 10 times, totaling 50 independent models) from a pool of 1,673,294 variants. Under these conditions, each training epoch required approximately 5.9 minutes of wall-clock time and incurred an average additional memory overhead of 593.97 MB. Data availability In this study, partial genotype data from the UK Biobank ( https://www.ukbiobank.ac.uk/ ) was accessed through a collaboration with application no.86920. Whole Exome Sequencing (WES) data in VCF.GZ format was accessed under Field ID 23157 and imputed genotypes in PGEN format were accessed under Field ID 22828. Data are available for bona fide researchers upon application to the UK Biobank, and the dataset of the 1000 Genomes Project is shown at https://www.internationalgenome.org/ , which does not require access rights. Code availability The source code of ELAG is publicly available at https://github.com/LHDLHUB/ELAG . Data Availability All data produced in the present study are available upon reasonable request to the authors. https://github.com/LHDLHUB/ELAG Supplementary information Supplementary Information Supplementary Note, Figs. 1 - 3 , and Tables 1 -8, 15. Supplementary Tables Supplementary Tables 9-14. Reference 1. ↵ Nandi , A. et al. Global and regional projections of the economic burden of Alzheimer’s disease and related dementias from 2019 to 2050: A value of statistical life approach . eClinicalMedicine 51 ( 2022 ). 2. ↵ Lau , C.S. Burden of rheumatoid arthritis and forecasted prevalence to 2050 . The Lancet Rheumatology 5 , e567 – e568 ( 2023 ). OpenUrl 3. Portaccio , E. et al. Multiple sclerosis: emerging epidemiological trends and redefining the clinical course . The Lancet Regional Health – Europe 44 ( 2024 ). 4. Lampit , A. & Finke , C. Combining cognitive interventions in multiple sclerosis . The Lancet Neurology 22 , 875 – 876 ( 2023 ). OpenUrl PubMed 5. Krijbolder , D.I. et al. Intervention with methotrexate in patients with arthralgia at risk of rheumatoid arthritis to reduce the development of persistent arthritis and its disease burden (TREAT EARLIER): a randomised, double-blind, placebocontrolled, proof-of-concept trial . The Lancet 400 , 283 – 294 ( 2022 ). OpenUrl 6. ↵ Niotis , K. , Saperia , C. , Saif , N. , Carlton , C. & Isaacson , R.S. Alzheimer’s disease risk reduction in clinical practice: a priority in the emerging field of preventive neurology . Nature Mental Health 2 , 25 – 40 ( 2024 ). OpenUrl 7. ↵ Clarke , R. et al. The properties of high-dimensional data spaces: implications for exploring gene and protein expression data . Nature Reviews Cancer 8 , 37 – 49 ( 2008 ). OpenUrl CrossRef PubMed Web of Science 8. ↵ Kachuri , L. et al. Principles and methods for transferring polygenic risk scores across global populations . Nature Reviews Genetics 25 , 8 – 25 ( 2024 ). OpenUrl CrossRef PubMed 9. ↵ van Rheenen , W. , Peyrot , W.J. , Schork , A.J. , Lee , S.H. & Wray , N.R. Genetic correlations of polygenic disease traits: from theory to practice . Nature Reviews Genetics 20 , 567 – 581 ( 2019 ). OpenUrl PubMed 10. ↵ Torkamani , A. , Wineinger , N.E. & Topol , E.J. The personal and clinical utility of polygenic risk scores . Nature Reviews Genetics 19 , 581 – 590 ( 2018 ). OpenUrl CrossRef PubMed 11. Adeyemo , A. et al. Responsible use of polygenic risk scores in the clinic: potential benefits, risks and gaps . Nature Medicine 27 , 1876 – 1884 ( 2021 ). OpenUrl CrossRef PubMed 12. ↵ Wand , H. et al. Improving reporting standards for polygenic scores in risk prediction studies . Nature 591 , 211 – 219 ( 2021 ). OpenUrl CrossRef PubMed 13. ↵ Ye , L. et al. Ge-SAND: an explainable deep learning-driven framework for disease risk prediction by uncovering complex genetic interactions in parallel . BMC Genomics 26 , 432 ( 2025 ). OpenUrl PubMed 14. Zhang , Z. , Ye , L. , Zheng , L. & Luo , Y. A Novel Solution to the Time-Varying Lyapunov Equation: The Integral Dynamic Learning Network . IEEE Transactions on Systems, Man, and Cybernetics: Systems 53 , 6731 – 6743 ( 2023 ). OpenUrl 15. Zhang , Z. , Ye , L. , Chen , B. & Luo , Y. An anti-interference dynamic integral neural network for solving the time-varying linear matrix equation with periodic noises . Neurocomputing 534 , 29 – 44 ( 2023 ). OpenUrl 16. Thomas , M. et al. Genome-wide Modeling of Polygenic Risk Score in Colorectal Cancer Risk . The American Journal of Human Genetics 107 , 432 – 444 ( 2020 ). OpenUrl CrossRef PubMed 17. Bracher-Smith , M. , Crawford , K. & Escott-Price , V. Machine learning for genetic prediction of psychiatric disorders: a systematic review . Molecular Psychiatry 26 , 70 – 79 ( 2021 ). OpenUrl PubMed 18. ↵ Hahn , S.-J. , Kim , S. , Choi , Y.S. , Lee , J. & Kang , J. Prediction of type 2 diabetes using genome-wide polygenic risk score and metabolic profiles: A machine learning analysis of population-based 10-year prospective cohort study . eBioMedicine 86 ( 2022 ). 19. ↵ Zhang , Y.D. et al. Assessment of polygenic architecture and risk prediction based on common variants across fourteen cancers . Nature Communications 11 , 3353 ( 2020 ). OpenUrl PubMed 20. Shen , X. et al. A phenome-wide association and Mendelian Randomisation study of polygenic risk for depression in UK Biobank . Nature Communications 11 , 2301 ( 2020 ). OpenUrl PubMed 21. Ding , Y. et al. Large uncertainty in individual polygenic risk score estimation impacts PRS-based risk stratification . Nature Genetics 54 , 30 – 39 ( 2022 ). OpenUrl CrossRef PubMed 22. ↵ Fadista , J. , Manning , A.K. , Florez , J.C. & Groop , L. The (in)famous GWAS P-value threshold revisited and updated for low-frequency variants . European Journal of Human Genetics 24 , 1202 – 1205 ( 2016 ). OpenUrl CrossRef PubMed 23. ↵ Xiao , R. & Boehnke , M. Quantifying and correcting for the winner’s curse in genetic association studies . Genetic Epidemiology 33 , 453 – 462 ( 2009 ). OpenUrl CrossRef PubMed Web of Science 24. ↵ Mullen , R.J. , Monekosso , D. , Barman , S. & Remagnino , P. A review of ant algorithms . Expert Systems with Applications 36 , 9608 – 9617 ( 2009 ). OpenUrl 25. ↵ Leardi , R. , Boggia , R. & Terrile , M. Genetic algorithms as a strategy for feature selection . Journal of Chemometrics 6 , 267 – 281 ( 1992 ). OpenUrl 26. ↵ Helmy , M. , Eldaydamony , E. , Mekky , N. , Elmogy , M. & Soliman , H. Predicting Parkinson disease related genes based on PyFeat and gradient boosted decision tree . Scientific Reports 12 , 10004 ( 2022 ). OpenUrl PubMed 27. ↵ Silva , P.P. et al. A machine learning-based SNP-set analysis approach for identifying disease-associated susceptibility loci . Scientific Reports 12 , 15817 ( 2022 ). OpenUrl PubMed 28. ↵ Hu , J. & Szymczak , S. A review on longitudinal data analysis with random forest . Briefings in Bioinformatics 24 , bbad002 ( 2023 ). OpenUrl CrossRef PubMed 29. ↵ Ghosh , D. & Cabrera , J. Enriched Random Forest for High Dimensional Genomic Data . IEEE/ACM Transactions on Computational Biology and Bioinformatics 19 , 2817 – 2828 ( 2022 ). OpenUrl 30. ↵ Cao , Y. et al. Survey on Large Language Model-Enhanced Reinforcement Learning: Concept, Taxonomy, and Methods . IEEE Transactions on Neural Networks and Learning Systems 36 , 9737 – 9757 ( 2025 ). OpenUrl 31. ↵ Zhang , Z. , Park , C.Y. , Theesfeld , C.L. & Troyanskaya , O.G. An automated framework for efficiently designing deep convolutional neural networks in genomics . Nature Machine Intelligence 3 , 392 – 400 ( 2021 ). OpenUrl 32. Korshunova , M. et al. Generative and reinforcement learning approaches for the automated de novo design of bioactive compounds . Communications Chemistry 5 , 129 ( 2022 ). OpenUrl PubMed 33. ↵ Lin , T. et al. A dosing strategy model of deep deterministic policy gradient algorithm for sepsis patients . BMC Medical Informatics and Decision Making 23 , 81 ( 2023 ). OpenUrl 34. ↵ Mi , X. , Zou , B. , Zou , F. & Hu , J. Permutation-based identification of important biomarkers for complex diseases via machine learning models . Nature Communications 12 , 3008 ( 2021 ). OpenUrl PubMed 35. Dincer , T.U. & Ernst , J. ChromActivity: integrative epigenomic and functional characterization assay based annotation of regulatory activity across diverse human cell types . Genome Biology 26 , 123 ( 2025 ). OpenUrl PubMed 36. ↵ Breiman , L. Bagging predictors . Machine Learning 24 , 123 – 140 ( 1996 ). OpenUrl 37. ↵ Bycroft , C. et al. The UK Biobank resource with deep phenotyping and genomic data . Nature 562 , 203 – 209 ( 2018 ). OpenUrl CrossRef PubMed 38. ↵ Schubach , M. , Maass , T. , Nazaretyan , L. , Röner , S. & Kircher , M. CADD v1.7: using protein language models, regulatory CNNs and other nucleotide-level scores to improve genome-wide variant predictions . Nucleic Acids Research 52 , D1143 – D1154 ( 2024 ). OpenUrl CrossRef PubMed 39. ↵ Choi , S.W. , Mak , T.S.-H. & O’Reilly , P.F. Tutorial: a guide to performing polygenic risk score analyses . Nature Protocols 15 , 2759 – 2772 ( 2020 ). OpenUrl PubMed 40. ↵ Euesden , J. , Lewis , C.M. & O’Reilly , P.F. PRSice: Polygenic Risk Score software . Bioinformatics 31 , 1466 – 1468 ( 2015 ). OpenUrl CrossRef PubMed 41. ↵ Mak , T.S.H. , Porsch , R.M. , Choi , S.W. , Zhou , X. & Sham , P.C. Polygenic scores via penalized regression on summary statistics . Genetic Epidemiology 41 , 469 – 480 ( 2017 ). OpenUrl CrossRef PubMed 42. ↵ Kunkle , B.W. et al. Genetic meta-analysis of diagnosed Alzheimer’s disease identifies new risk loci and implicates Aβ, tau, immunity and lipid processing . Nature Genetics 51 , 414 – 430 ( 2019 ). OpenUrl CrossRef PubMed 43. Okada , Y. et al. Genetics of rheumatoid arthritis contributes to biology and drug discovery . Nature 506 , 376 – 381 ( 2014 ). OpenUrl CrossRef PubMed Web of Science 44. ↵ International Multiple Sclerosis Genetics, C . et al. Multiple sclerosis genomic map implicates peripheral immune cells and microglia in susceptibility . Science 365 , eaav7188 ( 2019 ). OpenUrl Abstract / FREE Full Text 45. ↵ Dorigo , M. , Birattari , M. & Stutzle , T. Ant colony optimization . IEEE Computational Intelligence Magazine 1 , 28 – 39 ( 2006 ). OpenUrl CrossRef 46. ↵ Forrest , S. Genetic algorithms . ACM Comput. Surv . 28 , 77 – 80 ( 1996 ). OpenUrl 47. ↵ Li , B. et al. Genomic Prediction of Breeding Values Using a Subset of SNPs Identified by Three Machine Learning Methods . Frontiers in Genetics Volume 9 - 2018( 2018 ). 48. ↵ Gill , M. et al. Machine learning models outperform deep learning models, provide interpretation and facilitate feature selection for soybean trait prediction . BMC Plant Biology 22 , 180 ( 2022 ). OpenUrl PubMed 49. ↵ Kolberg , L. et al. g:Profiler—interoperable web service for functional enrichment analysis and gene identifier mapping (2023 update) . Nucleic Acids Research 51 , W207 – W212 ( 2023 ). OpenUrl CrossRef PubMed 50. ↵ Zalocusky , K.A. et al. Neuronal ApoE upregulates MHC-I expression to drive selective neurodegeneration in Alzheimer’s disease . Nature Neuroscience 24 , 786 – 798 ( 2021 ). OpenUrl CrossRef PubMed 51. ↵ DeMaio , A. , Mehrotra , S. , Sambamurti , K. & Husain , S. The role of the adaptive immune system and T cell dysfunction in neurodegenerative diseases . Journal of Neuroinflammation 19 , 251 ( 2022 ). OpenUrl PubMed 52. ↵ Curran , A.M. et al. Citrullination modulates antigen processing and presentation by revealing cryptic epitopes in rheumatoid arthritis . Nature Communications 14 , 1061 ( 2023 ). OpenUrl PubMed 53. ↵ McInnes , I.B. , Buckley , C.D. & Isaacs , J.D. Cytokines in rheumatoid arthritis — shaping the immunological landscape . Nature Reviews Rheumatology 12 , 63 – 68 ( 2016 ). OpenUrl PubMed 54. ↵ Jones , E.Y. , Fugger , L. , Strominger , J.L. & Siebold , C. MHC class II proteins and disease: a structural perspective . Nature Reviews Immunology 6 , 271 – 282 ( 2006 ). OpenUrl CrossRef PubMed Web of Science 55. ↵ Lückel , C. et al. IL-17+ CD8+ T cell suppression by dimethyl fumarate associates with clinical response in multiple sclerosis . Nature Communications 10 , 5722 ( 2019 ). OpenUrl PubMed 56. ↵ Zhang , X. et al. Regulation of the late onset Alzheimer’s disease associated HLA-DQA1/DRB1 expression . American Journal of Alzheimer’s Disease & Other Dementias® 37 , 15333175221085066 ( 2022 ). OpenUrl 57. ↵ Sun , H. , Xia , Y. , Wang , L. , Wang , Y. & Chang , X. PSORS1C1 may be involved in rheumatoid arthritis . Immunology Letters 153 , 9 – 14 ( 2013 ). OpenUrl PubMed 58. ↵ Spurkland , A. , Skjold Rønningen , K. , Vandvik , B. , Thorsby , E. & Vartdal , F. HLA-DQA1 and HLA-DQB1 genes may jointly determine susceptibility to develop multiple sclerosis . Human Immunology 30 , 69 – 75 ( 1991 ). OpenUrl CrossRef PubMed Web of Science 59. ↵ Moutsianas , L. et al. Class II HLA interactions modulate genetic risk for multiple sclerosis . Nature Genetics 47 , 1107 – 1113 ( 2015 ). OpenUrl CrossRef PubMed 60. ↵ Manolio , T.A. et al. Finding the missing heritability of complex diseases . Nature 461 , 747 – 753 ( 2009 ). OpenUrl CrossRef PubMed Web of Science 61. ↵ Cardon , L.R. & Bell , J.I. Association study designs for complex diseases . Nature Reviews Genetics 2 , 91 – 99 ( 2001 ). OpenUrl CrossRef PubMed Web of Science 62. ↵ Visscher , P.M. , Yengo , L. , Cox , N.J. & Wray , N.R. Discovery and implications of polygenicity of common diseases . Science 373 , 1468 – 1473 ( 2021 ). OpenUrl CrossRef PubMed 63. ↵ North , B.V. , Curtis , D. & Sham , P.C. A Note on the Calculation of Empirical P Values from Monte Carlo Procedures . The American Journal of Human Genetics 71 , 439 – 441 ( 2002 ). OpenUrl CrossRef PubMed Web of Science 64. ↵ Ashburner , M. et al. Gene Ontology: tool for the unification of biology . Nature Genetics 25 , 25 – 29 ( 2000 ). OpenUrl CrossRef PubMed Web of Science 65. ↵ Pontén , F. , Jirström , K. & Uhlen , M. The Human Protein Atlas—a tool for pathology . The Journal of Pathology 216 , 387 – 393 ( 2008 ). OpenUrl CrossRef PubMed Web of Science 66. ↵ Gillespie , M. et al. The reactome pathway knowledgebase 2022 . Nucleic Acids Research 50 , D687 – D692 ( 2022 ). OpenUrl CrossRef PubMed 67. ↵ Siva , N. 1000 Genomes project . Nature Biotechnology 26 , 256 – 256 ( 2008 ). OpenUrl CrossRef PubMed Web of Science 68. ↵ Davies , G. et al. Genome-wide association studies establish that human intelligence is highly heritable and polygenic . Molecular psychiatry 16 , 996 – 1005 ( 2011 ). OpenUrl CrossRef PubMed Web of Science 69. ↵ Miao , L. , Jiang , L. , Tang , B. , Sham , P.C. & Li , M. Dissecting the high-resolution genetic architecture of complex phenotypes by accurately estimating gene-based conditional heritability . The American Journal of Human Genetics 110 , 1534 – 1548 ( 2023 ). OpenUrl CrossRef PubMed 70. ↵ Chang , C.C. et al. Second-generation PLINK: rising to the challenge of larger and richer datasets . GigaScience 4 , s13742-015-0047-8 ( 2015 ). 71. ↵ Zhang , L. et al. GBC: a parallel toolkit based on highly addressable byte-encoding blocks for extremely large-scale genotypes of species . Genome Biology 24 , 76 ( 2023 ). OpenUrl PubMed View the discussion thread. Back to top Previous Next Posted September 25, 2025. Download PDF Supplementary Material Data/Code Email Thank you for your interest in spreading the word about medRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following Policy gradient-guided ensemble learning for enhanced polygenic risk prediction in ultra-high-dimensional genomics Message Subject (Your Name) has forwarded a page to you from medRxiv Message Body (Your Name) thought you would like to see this page from the medRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share Policy gradient-guided ensemble learning for enhanced polygenic risk prediction in ultra-high-dimensional genomics Lihang Ye , Nan Lin , Ying Luo , Liubin Zhang , Meng Liu , Wenjie Peng , Xuegao Yu , Miaoxin Li medRxiv 2025.09.23.25336425; doi: https://doi.org/10.1101/2025.09.23.25336425 Share This Article: Copy Citation Tools Policy gradient-guided ensemble learning for enhanced polygenic risk prediction in ultra-high-dimensional genomics Lihang Ye , Nan Lin , Ying Luo , Liubin Zhang , Meng Liu , Wenjie Peng , Xuegao Yu , Miaoxin Li medRxiv 2025.09.23.25336425; doi: https://doi.org/10.1101/2025.09.23.25336425 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Genetic and Genomic Medicine Subject Areas All Articles Addiction Medicine (570) Allergy and Immunology (863) Anesthesia (301) Cardiovascular Medicine (4442) Dentistry and Oral Medicine (444) Dermatology (383) Emergency Medicine (609) Endocrinology (including Diabetes Mellitus and Metabolic Disease) (1511) Epidemiology (15232) Forensic Medicine (30) Gastroenterology (1126) Genetic and Genomic Medicine (6610) Geriatric Medicine (669) Health Economics (998) Health Informatics (4542) Health Policy (1370) Health Systems and Quality Improvement (1613) Hematology (543) HIV/AIDS (1266) Infectious Diseases (except HIV/AIDS) (15924) Intensive Care and Critical Care Medicine (1104) Medical Education (623) Medical Ethics (147) Nephrology (668) Neurology (6609) Nursing (346) Nutrition (999) Obstetrics and Gynecology (1146) Occupational and Environmental Health (957) Oncology (3338) Ophthalmology (974) Orthopedics (369) Otolaryngology (420) Pain Medicine (436) Palliative Medicine (130) Pathology (665) Pediatrics (1693) Pharmacology and Therapeutics (692) Primary Care Research (712) Psychiatry and Clinical Psychology (5450) Public and Global Health (9240) Radiology and Imaging (2203) Rehabilitation Medicine and Physical Therapy (1370) Respiratory Medicine (1197) Rheumatology (596) Sexual and Reproductive Health (714) Sports Medicine (530) Surgery (712) Toxicology (99) Transplantation (289) Urology (265) (function(){function c(){var b=a.contentDocument||a.contentWindow.document;if(b){var d=b.createElement('script');d.innerHTML="window.__CF$cv$params={r:'a020dce3e9b1a09e',t:'MTc3OTg0MTMyMQ=='};var a=document.createElement('script');a.src='/cdn-cgi/challenge-platform/scripts/jsd/main.js';document.getElementsByTagName('head')[0].appendChild(a);";b.getElementsByTagName('head')[0].appendChild(d)}}if(document.body){var a=document.createElement('iframe');a.height=1;a.width=1;a.style.position='absolute';a.style.top=0;a.style.left=0;a.style.border='none';a.style.visibility='hidden';document.body.appendChild(a);if('loading'!==document.readyState)c();else if(window.addEventListener)document.addEventListener('DOMContentLoaded',c);else{var e=document.onreadystatechange||function(){};document.onreadystatechange=function(b){e(b);'loading'!==document.readyState&&(document.onreadystatechange=e,c())}}}})();

Text is read by the "Ask this paper" AI Q&A widget below. Extraction quality varies by source — PMC NXML preserves structure cleanly, OA-HTML may include some navigation residue, and OA-PDF can have broken hyphenation. The publisher copy (via DOI) is the canonical version.

My notes (saved in your browser only)

Ask this paper AI returns verbatim quotes from the full text · source: preprint-html

Answers must be backed by verbatim quotes from this paper's full text. Hallucinated quotes are dropped automatically; if no verbatim passage answers the question, we say so. How this works

Citation neighborhood (no data yet)

We don't have any in-corpus citations linked to this paper yet. This is a recent paper (2025) — citers typically take a year or two to land, and the OpenAlex reference graph may still be filling in.

Source provenance

europepmc
last seen: 2026-05-20T01:45:00.602351+00:00