GBoost-CTL: A novel method in multi-tissue transcriptome-wide associations studies in cross-tissue learner incorporating GWAS information

preprint OA: closed
📄 Open PDF Full text JSON View at publisher

Abstract

Genome-wide association studies (GWAS) have uncovered numerous genetic variants linked to complex human diseases, yet linking these variants to transcripts and tissues that drive pathology remains difficult. Multi-tissue transcriptome-wide association studies (TWAS) offer a powerful bridge, but existing analytical methods have some limitations, either by discarding important signals by separately analyzing and then aggregating results across tissues, implying imputation models in individual tissues, or fusing them with weights that ignore how much GWAS signal each tissue actually carries. Therefore, most of the existing methods do not work uniformly across different GWAS cohorts. Here, we propose GBoost-CTL - a GWAS-boosted cross-tissue learner that can overcome those aforementioned limitations. The method starts with any collection of single-tissue learners (STLs), allowing investigators to choose the most suitable imputation engine for each tissue. It then (i) allocates weights according to each STL’s out-of-sample predictive accuracy and (ii) refines those weights incorporating the GWAS-derived information, so that informative tissues are automatically up-weighted while uninformative tissues are down-weighted. This dual weighting strategy lets GBoost CTL adapt to fully shared, partially shared, or highly tissue-specific regulatory architectures while preserving nominal type I error control and delivering substantially higher power than existing linear or covariance-based methods. Through extensive simulation, we have found that this dual weighting strategy lets GBoost-CTL adapt to fully shared, partially shared, or highly tissue-specific regulatory architectures while preserving nominal type I error control and delivering substantially higher power than existing linear or covariance-based methods. When applied to real data, GBoost-CTL consistently outperformed some existing multi-tissue TWAS methods (e.g., TWAS-CTL, UTMOST and PrediXcan) by identifying a greater number of disease-associated genes with more stringent p-values. Given its modular design, computational scalability, and demonstrable gains in discovery power, we believe that GBoost-CTL offers a practical tool for the analysis of multi-tissue TWAS.
Full text 79,863 characters · extracted from preprint-html · click to expand
GBoost-CTL: A novel method in multi-tissue transcriptome-wide associations studies in cross-tissue learner incorporating GWAS information | bioRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-M677548'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search New Results GBoost-CTL: A novel method in multi-tissue transcriptome-wide associations studies in cross-tissue learner incorporating GWAS information View ORCID Profile Md Mutasim Billah , View ORCID Profile Hairong Wei , View ORCID Profile Fengzhu Sun , View ORCID Profile Kui Zhang doi: https://doi.org/10.1101/2025.11.22.689782 Md Mutasim Billah 1 Department of Mathematics, Statistics, and Computer Science, Macalester College , Saint Paul, MN, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Md Mutasim Billah For correspondence: mutasimsbidu{at}gmail.com Hairong Wei 2 Department of Mathematical Sciences, Michigan Technological University , Houghton, MI, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Hairong Wei Fengzhu Sun 3 Department of Quantitative and Computational Biology, University of Southern California , Los Angeles, CA, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Fengzhu Sun Kui Zhang 2 Department of Mathematical Sciences, Michigan Technological University , Houghton, MI, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Kui Zhang Abstract Full Text Info/History Metrics Data/Code Preview PDF Abstract Genome-wide association studies (GWAS) have uncovered numerous genetic variants linked to complex human diseases, yet linking these variants to transcripts and tissues that drive pathology remains difficult. Multi-tissue transcriptome-wide association studies (TWAS) offer a powerful bridge, but existing analytical methods have some limitations, either by discarding important signals by separately analyzing and then aggregating results across tissues, implying imputation models in individual tissues, or fusing them with weights that ignore how much GWAS signal each tissue actually carries. Therefore, most of the existing methods do not work uniformly across different GWAS cohorts. Here, we propose GBoost-CTL - a GWAS-boosted cross-tissue learner that can overcome those aforementioned limitations. The method starts with any collection of single-tissue learners (STLs), allowing investigators to choose the most suitable imputation engine for each tissue. It then (i) allocates weights according to each STL’s out-of-sample predictive accuracy and (ii) refines those weights incorporating the GWAS-derived information, so that informative tissues are automatically up-weighted while uninformative tissues are down-weighted. This dual weighting strategy lets GBoost CTL adapt to fully shared, partially shared, or highly tissue-specific regulatory architectures while preserving nominal type I error control and delivering substantially higher power than existing linear or covariance-based methods. Through extensive simulation, we have found that this dual weighting strategy lets GBoost-CTL adapt to fully shared, partially shared, or highly tissue-specific regulatory architectures while preserving nominal type I error control and delivering substantially higher power than existing linear or covariance-based methods. When applied to real data, GBoost-CTL consistently outperformed some existing multi-tissue TWAS methods (e.g., TWAS-CTL, UTMOST and PrediXcan) by identifying a greater number of disease-associated genes with more stringent p-values. Given its modular design, computational scalability, and demonstrable gains in discovery power, we believe that GBoost-CTL offers a practical tool for the analysis of multi-tissue TWAS. INTRODUCTION Genome-wide association studies (GWAS) have mapped thousands of genetic loci that affect complex human phenotypes ( Tam et al., 2019 ; Visscher et al., 2017 ; Vujkovic et al., 2020 ), but the mechanistic path that links DNA sequence to disease remains only partially resolved. A substantial share of risk alleles act by modulating gene expression - either by altering coding sequence or through cis- and trans-acting eQTLs that tune transcriptional output ( Albert & Kruglyak, 2015 ; Lappalainen et al., 2013 ; Zhang et al., 2015 ). Profiling gene expression at the scale needed to uncover these modest effects is often cost-prohibitive ( Spencer et al., 2009 ), prompting the development of transcriptome-wide association studies (TWAS) ( Gamazon et al., 2015 ; Gusev et al., 2018 ; Zhu et al., 2016 ). TWAS leverages a reference panel such as GTEx ( Consortium et al., 2020 ): cis-SNP weights trained in the panel are used to impute expression in the GWAS cohort, and imputed expression is then tested for association with the trait. This paradigm has yielded novel biological insights across a wide array of diseases ( Barbeira et al., 2018 ; Mancuso et al., 2017 ). As multi-tissue reference resources have matured ( Saha et al., 2017 ; Yang et al., 2017 ), it has become clear that modelling tissues in isolation squanders power and can even generate artefactual signals ( Liu et al., 2017 ; Mohammadi Pejman 5 6 Park YoSon 11 Parsana Princy 12 Segrè Ayellet V. 1 Strober Benjamin J. 9 Zappala Zachary 7 8 et al., 2017; Wainberg et al., 2017 ). Multi-tissue approaches have enhanced the statistical power of eQTL discovery ( Flutre et al., 2013 ; Li et al., 2018 ; Sul et al., 2013 ), indicating that a cross-tissue framework may similarly improve TWAS. UTMOST addressed part of this problem by jointly estimating weights across tissues with a sparse-group LASSO penalty ( Hu et al., 2019 ), but it still trains each tissue independently and therefore fails to capture graded similarities among related tissues ( Zhou et al., 2020 ). To embed homogeneity and heterogeneity directly into model building, we recently proposed TWAS-CTL, a two-stage framework that (i) trains a single-tissue learner (STL) per tissue using any predictive algorithm and (ii) fuses those predictions through an empirically derived utility weight that rewards STLs whose expression profiles generalize across tissues. Extensive simulations and an application to the GOKIND type I diabetes cohort showed that TWAS-CTL controls type I error at nominal levels and outperforms UTMOST and PrediXcan in discovery yield ( Billah et al., 2025 ). Yet classical TWAS-CTL leaves one crucial source of information untapped. It evaluates single tissue learners (STLs) purely on cross-tissue predictive accuracy, ignoring whether the imputed expression carries GWAS signal. Here we introduce GWAS Boosted Cross-Tissue Learner (GBoost-CTL), an augmented TWAS-CTL pipeline that incorporates both advances. For each tissue we incorporate GWAS information into the existing TWAS-CTL framework. By integrating this GWAS-level knowledge into the CTL weighting, GBoost-CTL explicitly couples imputation fidelity with downstream association potential, an idea that recent theories suggest can enhance TWAS power without inflating false positives ( Bhattacharya et al., 2021 ; Lin et al., 2022 ). Through comprehensive simulations spanning multiple sample sizes, replication depths and tissue-effect scenarios we demonstrated that GBoost-CTL retained strict type I error control while delivering markedly higher power than UTMOST - particularly when heterogeneity was present in the setup. Applied to the GOKIND cohort, GBoost-CTL uncovered a richer, biologically coherent set of type-1-diabetes loci compared to TWAS-CTL, UTMOST, and PrediXcan, all at more conservative p-values than those returned by earlier methods. In short, GBoost-CTL unites cross-tissue sharing, GWAS-informed weighting and modern machine-learning prediction in a single, modular framework. The result is a next-generation multi-tissue TWAS tool that scales to the complexity of human gene regulation and, promises to sharpen our view of the molecular mechanisms that drive complex disease. MATERIAL AND METHODS Data Acquisition and Quality Control In this project, we utilized the GTEx (Genotype-Tissue Expression) Project version 8 (v8) dataset as the reference panel for gene expression imputation. The open-access RNA-seq data includes expression profiles from 49 human tissues across 838 donors, the majority of whom were European American (85.3%), followed by African American (12.3%), Asian American (1.4%), and Hispanic or Latino (1.9%) individuals. The cohort comprised 557 males (66.4%) and 281 females (33.5%) ( Consortium et al., 2020 ). While previous cross-tissue TWAS implementations such as UTMOST employed RPKM (v6) or TPM (v7) normalized expression data, these methods are primarily suitable for within-sample normalization and may bias downstream analyses in cross-tissue models. Between-sample normalization methods, particularly count adjustment using upper quartile factors (CUF) and trimmed mean of M values (CTF), offer greater consistency across tissues and samples ( Johnson & Krishnan, 2022 ). Accordingly, we downloaded raw gene read counts for each tissue from GTEx v8 and applied CUF-based normalization to enable robust cross-tissue modeling. To correct for potential confounding effects, we again refined the normalized expression data for sex, sequencing platform, the top three genotype principal components (PCs), and the top PEER (Probabilistic Estimation of Expression Residuals) factors. Access to the protected GTEx genotype dataset was obtained via dbGaP under study accession phs000424.v8.p2 ( https://www.ncbi.nlm.nih.gov/projects/gap/cgi-bin/study.cgi?study_id=phs000424.v8.p2 ), following submission by Dr. Kui Zhang. For genotype quality control (QC), we employed PLINK v1.9 (( Purcell et al., 2007 ); https://www.cog-genomics.org/plink/ ). SNPs with a missing call rate >10% were excluded ( Marees et al., 2018 ). Additionally, variants failing the Hardy-Weinberg equilibrium (Kalogeropoulou et al.) test (p-value 10% missing genotype data were also discarded ( Zhou et al., 2020 ). Individual heterozygosity rates were used to detect potential contamination or inbreeding. We excluded five samples with the highest and five with the lowest heterozygosity, following the recommendation to remove outliers exceeding ±3 standard deviations from the mean ( Marees et al., 2018 ). This step was executed using PLINK and Git Bash ( https://git-scm.com/downloads ). To reduce redundancy due to linkage disequilibrium (LD), we performed LD pruning using an r 2 threshold of 0.9, retaining relatively independent variants ( Zhou et al., 2020 ). Moreover, SNPs with minor allele frequency (MAF) below 0.05 were filtered out to prevent spurious associations, as low-frequency variants are more susceptible to sampling noise and less likely to exhibit reliable trait associations in modest-sized cohorts ( Zhou et al., 2020 ). After applying all QC filters - missingness, HWE, LD pruning, heterozygosity outlier removal, and MAF thresholding - the final genotype dataset included 2,645,120 SNPs across 828 individuals and was used for downstream multi-tissue TWAS analyses. TWAS Figure 1 illustrates the contemporary TWAS pipeline into a two-tier framework that traces a path from local genetic variation to disease risk. The process begins with a reference panel - the GTEx resource is a typical example - where matched genotype data ( M SNPs in n donors) and RNA-seq measurements are available for dozens of tissues. Within each tissue, we train a predictive model that uses the surrounding cis-SNPs to predict gene expression levels; the resulting weight vectors capture that tissue’s regulatory architecture. Those weights are then transferred to a generally far larger GWAS cohort ( M SNPs, n 1 participants). Applying them to every individual generates tissue-specific vectors of imputed expression even though no transcriptomic data were collected. Because the same SNP predictors are reused across organs, the regulatory patterns learned in GTEx are faithfully projected onto the disease sample. In the final step, the GWAS genotypes and their corresponding predicted transcriptomes are analyzed jointly: the phenotype is regressed on each gene’s imputed expression (in one or many tissues), yielding a set of gene-trait association statistics. Taken together, the three stages-learning cis effects in a reference panel, broadcasting those effects into a GWAS panel, and testing the resulting expression–trait links - constitute the core of multi-tissue TWAS, offering a principled route to identify both the genes and the tissues most likely to mediate complex-disease risk. Download figure Open in new tab Figure 1. An overview of Transcriptome-wide Association Studies (TWAS). G-Boost-CTL Figure 2 (created with https://app.diagrams.net/ ) outlines the workflow of our proposed G-Boosted Cross-Tissue Learner (G-Boost-CTL) strategy. Starting with any target gene, we collect genotype data and gene expression measurements of different tissues from the reference panel (e.g. GTEx). In phase I, we fit a separate Single-Tissue Learner (STL) for each tissue - these learners can be penalized regression methods, tree-based regressors, deep networks, or any other methods capable of mapping cis-SNPs to expression ( Billah et al., 2025 ). Each STL yields a tissue-specific vector of predicted expression values. Phase II, we assign reliability scores to every tissue–model pair. A bespoke weighting function up-weights predictions that generalize well and down-weights noisy or weakly predictive signals ( Billah et al., 2025 ; Patil & Parmigiani, 2018 ). To harmonize the model with the downstream GWAS, in phase III, we refine the raw STL weights by the normalized variance of the predicted expression in the GWAS genotypes, so that tissues explaining more cross-individual variation receive the reasonable contribution in the adjusted weighing scheme. In the final phase (phase IV), G-Boost-CTL forms a composite, cross-tissue expression estimate by summing the STL outputs according to their adjusted weights, thereby capturing both shared and tissue-unique regulatory signals, yielding more accurate, flexible, and robust expression imputation than relying on a single tissue model or cross-tissue learner approach as mentioned in TWAS-CTL ( Billah et al., 2025 ). Download figure Open in new tab Figure 2. Architecture of GBoost-CTL (GWAS boosted cross-tissue learner). For each gene, we start with genotype - expression pairs from a reference panel such as GTEx and fit an independent single-tissue learner (STL) in every tissue j ( j = 1,2,…,. P ) with learner type l ( l = 1,2,…, L ): where G R is the n × M genotype matrix in the reference panel and is the vector of SNP weights estimated by learner l for tissue j . Applying those weights to the GWAS genotypes G w ( n 1 × M ) yields the tissue-specific, imputed expression: To construct a cross-tissue prediction, we aggregate the STL outputs with data-driven weights: Here is a GWAS boosted CTL weight derived from equation (2.2.9) and the corresponding tissue-specific prediction is obtained from equation (2.2.2). To obtain , first, cross-tissue predictive performance is quantified by a utility function: , where includes the predictions obtained using model , trained in tissue j′ , on the prediction profiles in tissue j , using learner l . For each learner, we form a P × P matrix: so that lower values correspond to better cross-tissue performance; each row of X l therefore summarizes how well the predictor trained in tissue transports to every other tissue. For learner l we reduce row j of X l to a single score by averaging the off-diagonal entries (the within-tissue resubstituting error is excluded). For each learner l , we derive its weights by first summarizing the row j of X by the mean of off-diagonal entries (omitting the resubstituting error ): Other summaries - e.g., the root-mean of off-diagonals - can be used in place of the simple mean. The CTL weights then can be found as, In this approach, the worst-performing STL receives a weight of zero, the losses of all other STLs are rescaled with respect to that baseline, and the adjusted values are subsequently normalized to obtain the final weights. GBoost-CTL has the flexibility to incorporate any GWAS information as a tissue-specific regulatory weight into its framework. In this work we employ the normalized variance of the imputed expression across the GWAS cohort to quantify how much genetically driven signal each tissue contributes. Let, denote the matrix of predicted expression values obtained from learner l , where n 1 is the number of GWAS individuals and the j -th column corresponds to tissue j . For tissue j we compute the GWAS variance: Collecting these variances into the vector , we normalize them to obtain a GWAS regulatory-weight vector: The elements of form a probability simplex and reflect each tissue’s relative contribution to genetically regulated expression variability in the GWAS cohort. Finally, we adjust the CTL weights obtained in Equation (2.2.6) by these regulatory factors: The vector simultaneously captures (i) predictive fidelity of each STL and the degree to which the corresponding tissue’s expression varies in the GWAS sample. Empirically, this dual weighting yields a more informative cross-tissue ensemble, particularly when regulatory influence varies markedly across tissues. As the utility function mentioned in Equation (2.2.4), For every learner and tissue pair ( j , j ′) with j ≠ j ′ , we quantify out-of-sample performance by the coefficient of determination: where z j,i is the observed expression of individual i in tissue j , is the value predicted by the model trained in tissue j′ , and is the sample mean of tissue j . This choice rewards models that explain a larger fraction of the target-tissue variance. For the row-wise summary for weighting, for each learner l we condense the off-diagonal entries of the j -th row of X l by an arithmetic mean, as mentioned in equation (2.2.5), as because higher R 2 values denote better performance, no square-root or sign adjustment is necessary; the subsequent normalization steps (Equations 2.2.6 - 2.2.9) converts these scores into the final G-Boost-CTL weights. STLs: Modeling Cross-Tissue Expression Imputation To estimate the effect size in Equation (2.2.1), any predictive modeling algorithm can serve as a candidate learner. In this study, we employ a diverse ensemble of four modeling strategies to construct our GBoost-CTL framework. Specifically, we incorporate penalized linear models - Ridge Regression ( Hoerl & Kennard, 1970 ), LASSO (Least Absolute Shrinkage and Selection Operator; ( Tibshirani, 1996 )), and Elastic Net ( Zou & Hastie, 2005 ) - which are widely used in high-dimensional settings for their regularization properties. Additionally, consistent with prior work in the field ( Gamazon et al., 2015 ), we integrate Random Forest ( Breiman, 2001 ), a nonparametric ensemble learning technique known for capturing non-linear interactions. Model training and hyperparameter tuning for each learner were conducted using five-fold cross-validation to ensure robust estimation and avoid overfitting. Gene-level Association Testing For each gene g, we assess its relationship with the phenotype by testing the null hypothesis H 0 :β g =0 against the alternative H 1 : β g ≠ 0. The phenotype for every GWAS individual is described by the linear model where, y is the trait vector, β g is the total effect, is the cross-tissue combined imputed gene expression, and . Simulation Architecture To benchmark the proposed GBoost-CTL framework we drew on the Genetics of Kidneys in Diabetes (GOKIND) study (dbGaP accession phs000018.v2.p1), using both its phenotype data for real-world association tests and its genotype panel as the basis for our simulations. Access to the protected dataset was obtained via dbGaP ( https://www.ncbi.nlm.nih.gov/projects/gap/cgi-bin/study.cgi?study_id=phs000018.v2.p1 ), and all marker coordinates were updated from their original GRCh36 positions to the current GRCh38 assembly with the UCSC LiftOver utility ( https://genome.ucsc.edu/cgi_bin/hgLiftOver ), ensuring compatibility with modern annotation resources. Genotypes then underwent the same quality-control pipeline applied to our GTEx reference panel - filtering on call-rate, Hardy–Weinberg equilibrium, heterozygosity outliers and minor-allele frequency. SNPs were mapped to genes through the Bioconductor biomaRt interface ( Durinck et al., 2009 ) using GRCh37 identifiers, adopting a ±100 kb cis window motivated by GTEx v8, where more than 70 % of significant cis-eQTLs fall within this range ( Consortium et al., 2020 ). Our simulation strategy largely mirrors the design of UTMOST ( Hu et al., 2019 ) but advances its ad-hoc tissue choices with a biologically motivated setup trio chosen to span both within-organ similarity and cross-organ diversity as mentioned in ( Billah et al., 2025 ) (Nucleus Accumbens (n = 198), Brain Cortex (n = 204), and Esophagus Mucosa (n = 492)). For any specific gene, the predicted expression levels in each tissue is generated by one of several single-tissue learners - Lasso, Ridge, Random Forest, Elastic-Net, and UTMOST - thereby allowing the downstream cross-tissue learners to be assessed under a variety of expression-prediction architectures. Phenotypes were then simulated as: with Setting all yields null data for assessing type-I error; power was examined under three regulatory regimes: (i) a homogeneous scenario ( u 1 = 1, u 2 = u 3 = 0); moderate heterogeneity ( u 1 = u 2 = 1, u 3 = 0)and (iii) full heterogeneity . Genotype data from the GOKIND cohort were served as G w (equation 2.2.2) when constructing imputed expression, ensuring that linkage structure and allele frequencies match those encountered in the association analysis. Each configuration was evaluated at sample sizes n = 500 and n = 750, with 1000, 5000, and 10000 Monte-Carlo replications per gene. Cross-tissue association statistics were computed for both G-Boost-CTL and the UTMOST pipeline, and power was defined as the proportion of replications achieving nominal significance at α = 0.05. This large-scale design allows a comprehensive comparison of error control and sensitivity across a spectrum of tissue-effect patterns and predictive-model pairings. RESULTS Simulation Studies: Empirical Type I Error Rate We compared the simulation outcomes of our proposed method against UTMOST, following its implementation as detailed by ( Hu et al., 2019 ; Zhou et al., 2020 ). Additionally, we incorporated PrediXcan ( Gamazon et al., 2015 ) to impute enabling a direct comparison with UTMOST. For this, we applied the Generalized Berk-Jones (GBJ) test ( Sun & Lin, 2017 ) to integrate the imputed gene expressions across multiple tissues, consistent with the UTMOST framework. Table 1 summarizes the empirical type I error rates obtained from G-Boost-CTL ensemble (averaged over twelve component STLs) and from UTMOST across three cohort sizes (N = 500, 750, 1000) and three Monte-Carlo replication depths (M = 1000, 5000, 10000). In every scenario both procedures remained close to the nominal 5 % level, confirming that neither approach is prone to systematic inflation. At the smallest sample size (N = 500) UTMOST displayed a noticeable upward bias when only 1,000 replicates were generated (6.2 %), whereas G-Boost-CTL tracked the target rate more closely (5.3 %). Increasing the number of replicates immediately narrowed this gap; by M = 5,000 the ensemble became slightly conservative (4.97 %), and by M = 10,000 the two methods were essentially indistinguishable. For the intermediate cohort (N = 750) the pattern reversed at low replication: G-Boost-CTL edged above the benchmark (5.35 % versus 4.80 %), yet the difference vanished once M reached 5,000, where both procedures settled just under 5 %. At the largest sample (N = 1,000) UTMOST achieved the minimum error with 1000 replicates, but the ensemble converged rapidly; by M = 10,000 G-Boost-CTL produced the lowest empirical rate (4.78 %), while UTMOST remained marginally higher (4.85 %). Taken together, these results indicate that the GWAS-boosted ensemble offers dependable control of false positives across a wide range of study sizes. Any mild inflation observed at the smallest replication depth disappears as soon as the Monte-Carlo sample is enlarged, and in the high-replication setting the ensemble is either comparable to or more conservative than the UTMOST benchmark. The stability of the error rate across ten CTLs suggests that the weighting scheme effectively shields the global test from individual learner misspecification, reinforcing the robustness of G-Boost-CTL for large-scale multi-tissue TWAS inference. View this table: View inline View popup Download powerpoint Table 1. Type I Error Comparison: Average type I error rate of GBoost-CTL across 12 CTLs with UTMOST. Simulation Studies: Statistical Power To evaluate the statistical power of our proposed G-Boost-CTL against the widely used UTMOST method, we considered three distinct cross-tissue regulatory scenarios: homogeneity, moderate heterogeneity, and full heterogeneity. For each scenario (denoted as U=1, 2, 3 respectively), we tested twelve GBoost-CTL combinations, each incorporating GWAS regulatory weights derived from a designated learner (Lasso, Ridge, Elastic Net, or Random Forest), paired with a secondary learner for association analysis. Power was estimated across 10,000 replicates using a sample size of 750. Table 2 and Figure 3 depicts the power comparison of different GBoost-CTL with UTMOST when the predicted gene expression was imputed using PrediXcan. When the trait was predominantly driven by a single tissue (homogeneity), methods that incorporated Random Forest (RF) for either GWAS weighting or downstream association consistently outperformed all others. Notably, the Lasso+RF (0.923), Elastic Net+Ridge (0.837), and Elastic Net+RF (0.719) CTLs achieved the highest power. These combinations exhibited an order of magnitude improvement over UTMOST (0.148), reaffirming the value of nonlinear modeling in settings with strong single-tissue influence. Even linear models like Lasso+Ridge (Ri) (0.648) and Elastic Net+Ridge (0.837) demonstrated marked improvement when paired with RF weighting. In contrast, the baseline UTMOST struggled to detect associations, indicating its limitations under pure homogeneity. In the intermediate regime of moderate homogeneity, where two of the three tissues contributed to the phenotype, nearly all GBoost-CTL combinations reached near-perfect sensitivity. Six methods achieved power ≥ 0.99, including Lasso+RF (RF), Elastic Net+Ridge, and Elastic Net+RF. Interestingly, both Ridge+RF variants (0.821, 0.927) performed exceptionally well, emphasizing the synergistic effect of combining nonlinear GWAS-informed weights with stable linear associations. UTMOST, by comparison, delivered poor performance (0.086), suggesting an inability to fully integrate partially shared signals. These results aligned with the observations from Project 1, where CTLs that could flexibly modulate tissue contribution had greater success across moderate heterogeneity. When expression-trait influence was equally distributed across all tissues, performance declines for all methods, as expected. However, GBoost-CTL models maintained a consistent lead over UTMOST. The Elastic Net+Ridge (0.370), Elastic Net+RF (0.357), and Lasso+RF (0.451) combinations achieved the highest power, nearly tripling that of UTMOST (0.102). While no method reached high absolute sensitivity in this setting, G-Boost-CTLs still captured the distributed signal more effectively. Linear–linear combinations (e.g., Lasso+Ridge (L), 0.112) underperformed, underscoring the need for flexible learners in highly heterogeneous contexts. View this table: View inline View popup Download powerpoint Table 2. Power Comparison of different GBoost-CTLs with UTMOST in different U settings (sample size 750, replication 10000)–when PrediXcan was used to impute . The highest power across different heterogeneity levels is bold. Download figure Open in new tab Figure 3. Statistical power comparison across varying levels of heterogeneity when Elastic Net is used for imputation in GBoost-CTL and UTMOST frameworks. Each panel corresponds to a different heterogeneity level (Homogeneity, Moderate Homogeneity, and Heterogeneity). Within each panel, the x-axis indicates which single-tissue learner (STL) was used to integrate GWAS information into the G-Boost-CTLs. Table 3 and Figure 4 illustrate the power comparison between various GBoost-CTL methods and UTMOST, based on gene expression imputed using Ridge Regression. In settings where a single tissue dominated the genetic effect, GBoost-CTL methods - especially those leveraging Ridge as the GWAS integrator - exhibited exceptional power. The highest power was achieved by combinations such as L + RF (L) and L + RF (R), both yielding 0.995, followed closely by E + Ri (E) at 0.974. These CTLs consistently outperformed UTMOST, which recorded a power of only 0.149. Interestingly, the Lasso+Ridge and ElasticNet+Ridge configurations performed robustly regardless of which learner was used to encode the GWAS feature, underscoring their strength in homogeneous settings. Under moderate heterogeneity (U=2), all G-Boost-CTLs maintained strong or near-perfect power, especially combinations with Ridge or Elastic Net in either position. For instance, L + Ri (R), L + RF (RF), and both E + RF variants all achieved power between 0.994 and 1.000. UTMOST also performed well here (0.398) but still fell short of the best G-Boost-CTLs. The E + Ri (Ri) configuration had slightly lower power (0.030), but this was a rare exception. When the genetic effect was uniformly spread across all three tissues, power declined overall, but GBoost-CTLs still held notable advantage. L + Ri (R) and L + RF (L) reached 0.770 and 0.693, respectively, outperforming UTMOST (0.178) and other less effective CTLs such as E + Ri (Ri) (0.048). Notably, Ridge-based learners integrated with Random Forest or Elastic Net consistently provided higher power under this complex, heterogeneous setting. The figure and table clearly indicate that the learner chosen to encode GWAS information (X) played a crucial role in the model’s effectiveness. For example: L + RF (L) outperformed L + RF (RF) in most scenarios, suggesting that Lasso captures GWAS-informed variability better in this pairing. Similarly, E + RF (E) consistently achieved higher power than E + RF (RF). This pattern reflected a broader trend where penalized regression learners (Lasso, ElasticNet) tended to provide more stable integration of GWAS information than tree-based learners when placed in the (X) position. View this table: View inline View popup Download powerpoint Table 3. Power Comparison of different GBoost-CTLs with UTMOST in different U settings (sample size 750, replication 10000)–when Ridge Regression was used to impute . The highest power across different heterogeneity levels is bold. Download figure Open in new tab Figure 4. Statistical power comparison across varying levels of heterogeneity when Ridge regression is used for imputation in GBoost-CTL and UTMOST frameworks. Each panel corresponds to a different heterogeneity level (Homogeneity, Moderate Homogeneity, and Heterogeneity). Within each panel, the x-axis indicates which single-tissue learner (STL) was used to integrate GWAS information into the G-Boost-CTLs. Table 4 and Figure 5 depicts the power comparison between various G-Boost-CTL methods and UTMOST, based on gene expression imputation using Lasso. Under homogeneity, the best-performing GBoost CTL was L + Ri (R) with a power of 0.461, followed by L + RF (L) at 0.318 and E + Ri (E) at 0.377. UTMOST yielded a comparable power of 0.293, outperforming several CTL variants. However, most GBoost CTLs, particularly those integrating Lasso as GWAS feature (X), underperformed under homogeneity, likely due to Lasso’s sparsity-induced instability in low-variability environments. GBoost CTLs clearly outshined UTMOST in moderate homogeneity. L + Ri (R) attained near-perfect power (0.998), and several other configurations - like E+RF (E) (0.958), E + RF (RF) (0.924), and L + RF (RF) (0.672) - also demonstrated strong performance. In contrast, UTMOST recorded only 0.022, revealing a sharp drop in power under this heterogeneity level. These findings show that GBoost CTLs effectively capture partially shared tissue signals, especially when Ridge or Elastic Net is used as the boosting learner. Power dropped across the board under the configuration of heterogeneity, as expected due to the uniform spread of genetic signal across all tissues. However, GBoost CTLs like E + RF (E) (0.249) and E + RF (RF) (0.210) still outperformed UTMOST (0.107) and other combinations. This suggests that elastic net-boosted CTLs may have better resilience in complex genetic architectures compared to traditional approaches. The Table 4 and Figure 5 also underscore the importance of choosing the right learner for incorporating GWAS variability. As for example, L+RF (L) consistently outperformed L+RF (RF) across all U settings, particularly under moderate heterogeneity (0.540 vs. 0.639), indicating that Lasso better captured regulatory influence in this pairing. E+RF (E) outperformed its RF-integrated counterpart across all U settings, especially under U=2 and U=3, reaffirming the advantage of using regularized learners like Elastic Net for GWAS weight encoding. Download figure Open in new tab Figure 5. Statistical power comparison across varying levels of heterogeneity when Lasso is used for imputation in GBoost-CTL and UTMOST frameworks. Each panel corresponds to a different heterogeneity level (Homogeneity, Moderate Homogeneity, and Heterogeneity). Within each panel, the x-axis indicates which single-tissue learner (STL) was used to integrate GWAS information into the G-Boost-CTLs. View this table: View inline View popup Download powerpoint Table 4. Power Comparison of different GBoost-CTLs with UTMOST in different U settings (sample size 750, replication 10000)–when Lasso was used to impute . The highest power across different heterogeneity levels is bold. Table 5 and Figure 6 illustrates the power comparison between various GBoost-CTL methods and UTMOST, based on gene expression imputation using UTMOST. In the homogeneous setting (U = 1), UTMOST exhibited the highest power (0.984), outperforming all G-Boost-CTLs, which remained below 0.19 - highlighting UTMOST’s strength in single-tissue dominant scenarios. For moderate heterogeneity (U = 2), while UTMOST achieved perfect power (1.000), Elastic Net–boosted CTLs like L + E (E) and E + RF (RF) performed competitively, reaching powers above 0.90. Under high heterogeneity (U = 3), UTMOST again dominated (0.998), but L + E (E) and E + RF (RF) remained the best-performing GBoost-CTLs (0.254 and 0.228). Across all U configurations, Elastic Net consistently emerged as the most effective boosting learner, outperforming Ridge- and Lasso-based variants. While UTMOST remains the overall top performer when used end-to-end, G-Boost-CTLs - especially those enhanced with Elastic Net-driven GWAS weights - provide competitive alternatives in heterogeneous environments, suggesting potential in hybrid frameworks that leverage UTMOST’s imputation with adaptive, interpretable weighting. Download figure Open in new tab Figure 6. Statistical power comparison across varying levels of heterogeneity when UTMOST is used for imputation in GBoost-CTL and UTMOST frameworks. Each panel corresponds to a different heterogeneity level (Homogeneity, Moderate Homogeneity, and Heterogeneity). Within each panel, the x-axis indicates which single-tissue learner (STL) was used to integrate GWAS information into the G-Boost-CTLs. View this table: View inline View popup Download powerpoint Table 5. Power Comparison of different GBoost-CTLs with UTMOST in different U settings (sample size 750, replication 10000)–when UTMOST was used to impute . The highest power across different heterogeneity levels is bold. Finally, Table 2 . 6 and Figure 2 . 6 show the power comparison when the imputation was done with Random Forest. In this setting, where one tissue dominated the genetic influence on the phenotype, several GBoost-CTLs clearly outperformed UTMOST. The top-performing method was E+RF (E) with a power of 0.921, followed by L + Ri (R) at 0.786, and Ri + RF (Ri) at 0.762. UTMOST trailed at 0.106. These results reinforce the effectiveness of ensemble CTLs in prioritizing tissue-specific signals even when only one tissue contributed. Moderate homogeneity is the most favorable scenario for G-Boost-CTLs. Many methods, including Ri + RF (RF), E + RF (E), and E + RF (RF), achieved perfect or near-perfect power (1.000). Even combinations like L + RF (L) and L + RF (RF) scored 0.999, dramatically outperforming UTMOST (0.999) only marginally. However, UTMOST performed well here, reflecting that its design handles moderately heterogeneous tissue influences effectively. Still, G-Boost-CTLs delivered competitive or superior results by leveraging additional GWAS regulatory information. In heterogeneous situation, G-Boost-CTLs utilizing RF-based boosting, such as E + RF (RF) and Ri + RF (RF), maintained perfect power (1.000), while E + RF (E) achieved 0.998. UTMOST registered 0.579, a notable drop compared to the top-performing CTLs. These results indicate that G-Boost-CTLs are capable of maintaining high power even when the signal was equally distributed across diverse tissues—a setting typically challenging for traditional TWAS methods. A noteworthy observation is that RF as a GWAS weight learner (in combinations like L + RF (RF), Ri + RF (RF), and E + RF (RF)) generally yielded higher or comparable power compared to using Lasso, Ridge, or Elastic Net as the boosting learner. For instances, E + RF (RF) showed high power across all U settings: 0.439 (U=1), 1.000 (U=2), and 1.000 (U=3). Comparatively, L+RF (L) underperformed with the power of 0.612, 0.999, and 0.398 across the same settings. This suggests that Random Forest, being a non-linear and variance-aware model, is particularly effective in capturing the regulatory effects in GWAS-boosted frameworks. View this table: View inline View popup Download powerpoint Table 6. Power Comparison of different GBoost-CTLs with UTMOST in different U settings (sample size 750, replication 10000)–when Random Forest was used to impute . The highest power across different heterogeneity levels is bold. Application to Real Data To illustrate the effectiveness of our method, we applied it to the GOKIND dataset, which includes genotype and phenotype data relevant to type 1 diabetes. Following quality control, the dataset comprised 8,859 genes and 62,701 SNPs across 1,108 individuals. While the Bonferroni correction yields a genome-wide threshold of approximately , it is known to be overly conservative in multi-tissue TWAS due to its assumption of test independence. Therefore, following prior studies ( Billah et al., 2025 ; Fischer et al., 2018 ), we adopted a more lenient significance level of 3 ×10 −4 to strike a balance between discovery and false-positive control, particularly given the modest sample size of the GOKIND cohort relative to large-scale GWAS. Table 7 presents the gene-level findings from our G-Boost-CTL analysis. Trait annotations were drawn from the GWASL Catalog ( https://www.ebi.ac.uk/gwas/home ), and the “Description” field briefly summarizes previously reported associations, thereby placing each hit in a biological context. At the stringent Bonferroni threshold, GBoost-CTL pinpointed two genes: (i) FCHSD2 (11:72836745-73142318, p-value: 4.09 × 10 −7 ), a locus linked to Systemic lupus erythematosus ( Su et al., 2023 ), Crohn’s disease ( Chranioti et al., 2022 ), Psoriasis ( Abramczyk et al., 2020 ), HbA1c levels ( Patiño-Fernández et al., 2009 ) - highlighting its association with type-1 diabetes ( Billah et al., 2025 ); (ii) BFSP2 (3:133400056-133475222, p-value: 3.17 × 10 −6 ), with its link with the traits like diffuse plaque measurement ( Orchard et al., 2006 ), Serum Alanine Aminotransferase measurement ( West et al., 2006 ), Platelet count ( Schneider, 2009 ), Hemoglobin A1c (HbA1c) testing ( Eyth & Naik, 2023 ), and Total Cholesterol measurement ( Candler et al., 2017 ) - associated with type I diabetes ( Billah et al., 2025 ). View this table: View inline View popup Table 7. Genes significantly associated by GBoost-CTL at a p-value threshold of 3 × 10 −4 . Trait information in the “Description” column was sourced from the GWAS Catalog and reflects phenotypes previously linked to these genes. Citations indicate prior studies connecting the traits to type 1 diabetes. Genes with Bonferroni corrected threshold setting are bold and novel genes are bold and italic. The p-values of these genes identified using GBoost-CTL are notably smaller than those reported under TWAS-CTL. For instance, the gene FCHSD2 exhibited a p-value of 2.81 × 10 −6 under TWAS-CTL, whereas its significance under GBoost-CTL was even stronger. Similarly, BFSP2 had a p-value of 8.72 × 10 −6 with TWAS-CTL, but was detected with greater confidence using G-Boost-CTL. Importantly, the previously highlighted novel gene - RP11-550H2, a long intergenic non-protein coding RNA (chr1:76,507,376-76,531,913) - was also identified by both methods. However, GBoost-CTL again yielded a smaller p-value of 4.09 × 10 −5 , compared to 2.90 × 10 −4 under TWAS-CTL. This consistent improvement in statistical strength highlights the enhanced sensitivity of GBoost-CTL. Beyond these replicated loci, the GBoost-CTL also identified four previously unreported Long Intergenic Non-Protein Coding RNAs to be associated with type I diabetes: (i) RP11-202H2 (chr12:86,993,355-87,232,775, p-value: 3.20 × 10 −5 ), (ii) RP11-770E5 (chr8:48,658,981-48,696,887, p-value: 1.03 × 10 −4 ), (iii) RP11-403H13 (chr9:6,902,670-6,978,859, p-value: 1.27 × 10 −4 ), and (iv) ISX-AS1 (chr822:34,922,609-35,072,889, p-value: 2.33 × 10 −4 ), and suggesting possible new regulatory elements worthy of functional follow-up and validation in independent cohorts. Furthermore, in comparison to TWAS-CTL - which identified five genes at the X × 10 −5 significance threshold - GBoost-CTL detected nine genes at the same level. At the study-wide suggested threshold of 3 × 10 −4 , G-Boost-CTL identified a total of 23 genes, outperforming TWAS-CTL, which discovered 15 genes; with an overlap of nine genes in between these methods ( Figure 8 ). Notably, GBoost-CTL consistently uncovered a broader spectrum of associations even at stringent significance levels. In contrast, UTMOST identified only two genes - MAST2 (p-value: 2 × 10 −5 ) and LINC00575 - while PrediXcan, combined with the GBJ test, identified one gene (KLHL8) with a p-value of 9.37 ×0 10 −5 . Interestingly, GBoost-CTL also identifies KLHL8, however, with a more stringent p-value of 4.09 × 10 −5 . When the p-value threshold was relaxed to 1 × 10 −3 , GBoost-CTL identifies 50 genes, whereas TWAS-CTL, UTMOST, and PrediXcan only detected 35, 4, and 8 genes, respectively. These comparisons underscore the enhanced sensitivity and broader discovery potential of GBoost-CTL, which is capable of identifying a larger set of biologically meaningful loci without compromising statistical rigor. Download figure Open in new tab Figure 7. Statistical power comparison across varying levels of heterogeneity when Random Forest is used for imputation in GBoost-CTL and UTMOST frameworks. Each panel corresponds to a different heterogeneity level (Homogeneity, Moderate Homogeneity, and Heterogeneity). Within each panel, the x-axis indicates which single-tissue learner (STL) was used to integrate GWAS information into the G-Boost-CTLs. Download figure Open in new tab Figure 8. Comparison of TWAS-CTL and GBoost-CTL in identifying genes associated with TypeL1LDiabetes. DISCUSSION This study introduces a novel mutli-tisue ensemble method in TWAS named G-Boost-CTL - which couples flexible single-tissue imputers with data-driven, GWAS-informed weighting - can outperform the prevailing linear and covariance-based frameworks used in transcriptome-wide association studies. Moreover, our framework can eclipse the performance of TWAS-CTL - the cross-tissue learner proposed by ( Billah et al., 2025 ) - by replacing its “penalize-the-worst” utility weighting with a GWAS-informed scheme that elevates, rather than merely rescales, the most biologically informative tissues. Across an extensive grid of simulations and a real-data application to the GOKIND type 1 diabetes cohort, the superiority of GBoost-CTL is established. GBoost-CTL maintained nominal type I error while delivering substantially higher power than UTMOST in almost every heterogeneity regime, when gene expression was imputed in statistical power simulation with any of the four learners we evaluated (Random Forest, PrediXcan, Ridge, or Lasso). This advantage was most striking under moderate tissue heterogeneity - the setting that best mirrors GTEx observations of shared-yet-unequal cis regulation - but it persisted, albeit at a smaller margin, when effects were either fully homogeneous or broadly dispersed. Two ingredients proved critical. First, the use of weighting function that prioritizes tissues with higher contribution can play a crucial role in identifying the intrinsic relationship between the eQTL effect on the genetic variants across the tissues–shared in withing and between human organs. Second, GWAS-derived information weights allowed GBoost-CTL to adapt to whichever tissue carried the informative signal in the target cohort, something UTMOST’s fixed covariance structure or TWAS-CTL’s weighting function cannot do. Notably, GBoost-CTLs boosted with Elastic Net or Random-Forest weights were the most stable across all scenarios, echoing theoretical work showing that mixed linear–non-linear ensembles can approximate a wide class of genetic architectures. The simulation patterns translated directly to the empirical setting. Using a study-wide threshold of 3 × 10 −4 - a compromise that balances discovery with control of false positives in moderately sized TWAS cohorts - GBoost-CTL identified 23 genes, versus 15 for TWAS-CTL, 2 for UTMOST, and a single locus for PrediXcan. Most of the loci recovered by the legacy methods were recaptured with more stringent p-values in most cases under G-Boost-CTL, and the framework also unearthed several long intergenic non-coding RNAs that have not yet appeared in the Type 1 diabetes literature. Like all statistical frameworks, multi-tissue TWAS methodologies are not without their limitations. The proposed GBoost-CTL, while methodologically innovative, also introduces certain computational challenges - particularly in the context of training multiple learner combinations across 49 tissues. This level of complexity can be computationally demanding and may not scale efficiently without further optimization. Future methodological improvements could include tissue dimensionality reduction strategies or shared representation learning to compress information across similar tissues while retaining biological signal. In addition to this, in this current work we have used GWAS information directly from the raw genotype data, however, in most cases, only GWAS summary statistics data is available. In the future work, we plan to incorporate G-Boost-CTL in GWAS summary level data. Moreover, the promise of aggregation of linear and non-linear ensembles shows promising results and hence our future work will focus on exploring the application of advanced machine learning (e.g. gradient-boosted tree) and deep learning methods in the essence of multi-tissue TWAS framework. Furthermore, while the GOKIND dataset provides a valuable testbed for methodological development, its sample size is relatively small compared to modern biobank-scale resources. Moving forward, the application of GBoost-CTL to large-scale, ancestrally diverse cohorts such as the UK Biobank will be essential. These datasets offer the statistical power and heterogeneity necessary to validate novel associations and build high-resolution, tissue-specific regulatory maps, ultimately enhancing the biological interpretability and generalizability of multi-tissue TWAS findings. Nonetheless, GBoost-CTL unites the predictive strength of modern machine-learning imputers with a principled, GWAS-informed weighting scheme, delivering a demonstrable uplift in gene-trait discovery over TWAS-CTL, UTMOST, and PrediXcan without sacrificing error control. Taken the extensive simulation and real-data analysis into consideration, GBoost-CTL reinforces that: (i) Adaptive weighting matters: Borrowing information from GWAS to modulate tissue weights complements the genetic-covariance strategy of UTMOST and reduces the risk of down-weighting the causal tissue when effects are unbalanced, (ii) Non-linearity is beneficial: Tree-boosted methods (e.g. Bootstrap aggregation like Random Forest) and Elastic-Net-weighted ensembles capture interaction and dominance effects that linear models systematically may underfit, and (iii) Statistical integrity is preserved: Despite the added flexibility, type I error remained at the nominal level across all sample sizes and replication depths, addressing long-standing concerns that complex learners might over-fit reference panels. As multi-tissue resources enlarge and computational barriers fall, such hybrid strategies like G-Boost-CTL are poised to become the new standard for elucidating the regulatory landscape of complex disease genetics. Footnotes https://gtexportal.org/home/ https://www.ncbi.nlm.nih.gov/projects/gap/cgi-bin/study.cgi?study_id=phs000424.v8.p2 https://www.ncbi.nlm.nih.gov/projects/gap/cgi-bin/study.cgi?study_id=phs000018.v2.p1 References ↵ Abramczyk , R. , Queller , J. N. , Rachfal , A. W. , & Schwartz , S. S. ( 2020 ). Diabetes and psoriasis: different sides of the same prism . Diabetes, Metabolic Syndrome and Obesity , 3571 – 3577 . ↵ Albert , F. W. , & Kruglyak , L. ( 2015 ). The role of regulatory variation in complex traits and disease . Nature Reviews Genetics , 16 ( 4 ), 197 – 212 . OpenUrl CrossRef PubMed Arneth , B. , Arneth , R. , & Shams , M. ( 2019 ). Metabolomics of type 1 and type 2 diabetes . International journal of molecular sciences , 20 ( 10 ), 2467 . OpenUrl PubMed Baghersalimi , A. , Koohmanaee , S. , Darbandi , B. , Farzamfard , V. , Rad , A. H. , Zare , R. , Tabrizi , M. , & Dalili , S. ( 2019 ). Platelet indices alterations in children with type 1 diabetes mellitus . Journal of Pediatric Hematology/Oncology , 41 ( 4 ), e227 – e232 . OpenUrl ↵ Barbeira , A. N. , Dickinson , S. P. , Bonazzola , R. , Zheng , J. , Wheeler , H. E. , Torres , J. M. , Torstenson , E. S. , Shah , K. P. , Garcia , T. , Edwards , T. L. , Stahl , E. A. , Huckins , L. M. , Aguet , F. , Ardlie , K. G. , Cummings , B. B. , Gelfand , E. T. , Getz , G. , Hadley , K. , Handsaker , R. E. , … Visualization—Ucsc Genomics Institute, U. o. C. S. C . ( 2018 ). Exploring the phenotypic consequences of tissue specific gene expression variation inferred from GWAS summary statistics . Nature Communications , 9 ( 1 ), 1825 . doi: 10.1038/s41467-018-03621-1 OpenUrl CrossRef PubMed ↵ Bhattacharya , A. , Li , Y. , & Love , M. I. ( 2021 ). MOSTWAS: multi-omic strategies for transcriptome-wide association studies . PLoS genetics , 17 ( 3 ), e1009398 . OpenUrl Bielka , W. , Przezak , A. , Molęda , P. , Pius-Sadowska , E. , & Machaliński , B. ( 2024 ). Double diabetes—when type 1 diabetes meets type 2 diabetes: definition, pathogenesis and recognition . Cardiovascular diabetology , 23 ( 1 ), 62 . OpenUrl PubMed ↵ Billah , M. M. , Wei , H. , Sun , F. , & Zhang , K. ( 2025 ). TWAS-CTL: A robust and efficient method for multi-tissue transcriptome-wide association studies using cross-tissue learners . bioRxiv . doi: 10.1101/2025.10.25.684514 OpenUrl Abstract / FREE Full Text Bjornstad , P. , Nguyen , N. , Reinick , C. , Maahs , D. M. , Bishop , F. K. , Clements , S. A. , Snell-Bergeon , J. K. , Lieberman , R. , Pyle , L. , & Daniels , S. R. ( 2015 ). Association of apolipoprotein B, LDL-C and vascular stiffness in adolescents with type 1 diabetes . Acta diabetologica , 52 , 611 – 619 . OpenUrl PubMed Bjornstad , P. , Snell-Bergeon , J. K. , Nadeau , K. J. , & Maahs , D. M. ( 2015 ). Insulin sensitivity and complications in type 1 diabetes: new insights . World journal of diabetes , 6 ( 1 ), 8 . OpenUrl PubMed Boumarah , D. N. , AlSinan , A. S. , AlMaher , E. M. , Mashhour , M. , & AlDuhileb , M. ( 2022 ). Diabetic mastopathy: A rare clinicopathologic entity with considerable autoimmune potential . International Journal of Surgery Case Reports , 95 , 107151 . OpenUrl PubMed Bradshaw , E. M. , Raddassi , K. , Elyaman , W. , Orban , T. , Gottlieb , P. A. , Kent , S. C. , & Hafler , D. A. ( 2009 ). Monocytes from patients with type 1 diabetes spontaneously secrete proinflammatory cytokines inducing Th17 cells . The Journal of Immunology , 183 ( 7 ), 4432 – 4439 . OpenUrl PubMed ↵ Breiman , L. ( 2001 ). Random forests . Machine learning , 45 , 5 – 32 . OpenUrl ↵ Candler , T. , Mahmoud , O. , Edge , J. , & Hamilton Shield , J. ( 2017 ). Hypercholesterolaemia screening in Type 1 diabetes: a difference of opinion . Diabetic Medicine , 34 ( 7 ), 983 – 986 . OpenUrl PubMed ↵ Chranioti , I. , Vartzelis , G. , Maritsi , D. , & Tsolia , M. ( 2022 ). A co-diagnosis of Crohn disease and autoimmune diabetes in an adolescent patient . JPGN reports , 3 ( 4 ), e265 . OpenUrl ↵ Consortium , T. G. , Aguet , F. , Anand , S. , Ardlie , K. G. , Gabriel , S. , Getz , G. A. , Graubert , A. , Hadley , K. , Handsaker , R. E. , Huang , K. H. , Kashin , S. , Li , X. , MacArthur , D. G. , Meier , S. R. , Nedzel , J. L. , Nguyen , D. T. , Segrè , A. V. , Todres , E. , Balliu , B. , … Volpi , S. ( 2020 ). The GTEx Consortium atlas of genetic regulatory effects across human tissues . Science , 369 ( 6509 ), 1318 – 1330 . doi: 10.1126/science.aaz1776 OpenUrl Abstract / FREE Full Text De Jong , H. , Koh , G. , Bulder , I. , Stephan , F. , Wiersinga , W. , & Zeerleder , S. ( 2015 ). DiabetesLindependent increase of factor VII activating protease activation in patients with Gram negative sepsis (melioidosis) . Journal of Thrombosis and Haemostasis , 13 ( 1 ), 41 – 46 . OpenUrl Dejan , D. , Jelena , A. , Goran , R. , & Ljiljana , A. ( 2023 ). Platelet indices in children with type 1 diabetes mellitus: a simple glucoregulation monitoring tool . African Health Sciences , 23 ( 4 ), 333 – 338 . OpenUrl PubMed ↵ Durinck , S. , Spellman , P. T. , Birney , E. , & Huber , W. ( 2009 ). Mapping identifiers for the integration of genomic datasets with the R/Bioconductor package biomaRt . Nature protocols , 4 ( 8 ), 1184 – 1191 . OpenUrl PubMed ↵ Eyth , E. , & Naik , R. ( 2023 ). Hemoglobin A1c . In StatPearls [Internet] . StatPearls Publishing . Filip , P. , Canna , A. , Moheet , A. , Bednarik , P. , Grohn , H. , Li , X. , Kumar , A. F. , Olawsky , E. , Eberly , L. E. , & Seaquist , E. R. ( 2020 ). Structural alterations in deep brain structures in type 1 diabetes . Diabetes , 69 ( 11 ), 2458 – 2466 . OpenUrl Abstract / FREE Full Text ↵ Fischer , S. T. , Jiang , Y. , Broadaway , K. A. , Conneely , K. N. , & Epstein , M. P. ( 2018 ). Powerful and robust crossLphenotype association test for caseLparent trios . Genetic epidemiology , 42 ( 5 ), 447 – 458 . OpenUrl PubMed ↵ Flutre , T. , Wen , X. , Pritchard , J. , & Stephens , M. ( 2013 ). A statistical framework for joint eQTL analysis in multiple tissues . PLoS genetics , 9 ( 5 ), e1003486 . OpenUrl PubMed Forga , L. , López-Andrés , N. , Tamayo , I. , Fernández-Celis , A. , García-Mouriz , M. , & Goñi , M. J. ( 2022 ). Relationship between soluble protein ST2 (sST2) levels and microvascular complications in a cohort of patients with type 1 diabetes . Endocrinología, Diabetes y Nutrición (English ed .), 69 ( 5 ), 322 – 330 . OpenUrl ↵ Gamazon , E. R. , Wheeler , H. E. , Shah , K. P. , Mozaffari , S. V. , Aquino-Michaels , K. , Carroll , R. J. , Eyler , A. E. , Denny , J. C. , Consortium , G. , & Nicolae , D. L. ( 2015 ). A gene-based association method for mapping traits using reference transcriptome data . Nature Genetics , 47 ( 9 ), 1091 – 1098 . OpenUrl CrossRef PubMed Ganesan , S. , Tantone , R. P. , Komatsu , D. E. , & Hurst , L. C. ( 2023 ). The prevalence of Dupuytren’s disease in patients with diabetes mellitus . Communications Medicine , 3 ( 1 ), 96 . OpenUrl PubMed Getawa , S. , & Adane , T. ( 2022 ). Hematological abnormalities among adults with type 1 diabetes mellitus at the University of Gondar Comprehensive Specialized Hospital . SAGE open medicine , 10 , 20503121221094212. Gomes , K. B. ( 2017 ). IL-6 and type 1 diabetes mellitus: T cell responses and increase in IL-6 receptor surface expression . Annals of translational medicine , 5 ( 1 ), 16 . OpenUrl PubMed Group , D. E. R. ( 2011 ). Intensive diabetes therapy and glomerular filtration rate in type 1 diabetes . New England Journal of Medicine , 365 ( 25 ), 2366 – 2376 . OpenUrl CrossRef PubMed Web of Science ↵ Gusev , A. , Mancuso , N. , Won , H. , Kousi , M. , Finucane , H. K. , Reshef , Y. , Song , L. , Safi , A. , Schizophrenia Working Group of the Psychiatric Genomics, C ., McCarroll , S. , Neale , B. M. , Ophoff , R. A. , O’Donovan , M. C. , Crawford , G. E. , Geschwind , D. H. , Katsanis , N. , Sullivan , P. F. , Pasaniuc , B. , & Price , A. L. ( 2018 ). Transcriptome-wide association study of schizophrenia and chromatin activity yields mechanistic disease insights . Nat Genet , 50 ( 4 ), 538 – 548 . doi: 10.1038/s41588-018-0092-1 OpenUrl CrossRef PubMed Hobday , A. L. , Parmar , M. S. , & Hobday , A. ( 2021 ). The link between diabetes mellitus and tau hyperphosphorylation: implications for risk of Alzheimer’s disease . Cureus , 13 ( 9 ). ↵ Hoerl , A. E. , & Kennard , R. W. ( 1970 ). Ridge regression: applications to nonorthogonal problems . Technometrics , 12 ( 1 ), 69 – 82 . OpenUrl CrossRef Web of Science ↵ Hosking , L. , Lumsden , S. , Lewis , K. , Yeo , A. , McCarthy , L. , Bansal , A. , Riley , J. , Purvis , I. , & Xu , C.-F. ( 2004 ). Detection of genotyping errors by Hardy–Weinberg equilibrium testing . European Journal of Human Genetics , 12 ( 5 ), 395 – 399 . OpenUrl CrossRef PubMed Web of Science ↵ Hu , Y. , Li , M. , Lu , Q. , Weng , H. , Wang , J. , Zekavat , S. M. , Yu , Z. , Li , B. , Gu , J. , Muchnik , S. , Shi , Y. , Kunkle , B. W. , Mukherjee , S. , Natarajan , P. , Naj , A. , Kuzma , A. , Zhao , Y. , Crane , P. K. , Lu , H. , … Alzheimer’s Disease Genetics, C . ( 2019 ). A statistical framework for cross-tissue transcriptome-wide association analysis . Nature Genetics , 51 ( 3 ), 568 – 576 . doi: 10.1038/s41588-019-0345-7 OpenUrl CrossRef PubMed Huang , J. , Xiao , Y. , Xu , A. , & Zhou , Z. ( 2016 ). Neutrophils in type 1 diabetes . Journal of diabetes investigation , 7 ( 5 ), 652 – 663 . OpenUrl PubMed Jarvisalo , M. J. , Putto-Laurila , A. , Jartti , L. , LehtimaLki , T. , Solakivi , T. , Ronnemaa , T. , & Raitakari , O. T. ( 2002 ). Carotid artery intima-media thickness in children with type 1 diabetes . Diabetes , 51 ( 2 ), 493 – 498 . OpenUrl Abstract / FREE Full Text ↵ Johnson , K. A. , & Krishnan , A. ( 2022 ). Robust normalization and transformation techniques for constructing gene coexpression networks from RNA-seq data . Genome biology , 23 , 1 – 26 . OpenUrl CrossRef PubMed Kalogeropoulou , D. , LaFave , L. , Schweim , K. , Gannon , M. C. , & Nuttall , F. Q. ( 2009 ). Lysine ingestion markedly attenuates the glucose response to ingested glucose without a change in insulin response . The American journal of clinical nutrition , 90 ( 2 ), 314 – 320 . OpenUrl Abstract / FREE Full Text Katra , B. , Fedak , D. , Matejko , B. , Małecki , M. T. , & Wędrychowicz , A. ( 2021 ). The enteroendocrine-osseous axis in patients with long-term type 1 diabetes mellitus . Bone , 153 , 116105 . OpenUrl PubMed Kothari , V. , Ho , T. W. , Cabodevilla , A. G. , He , Y. , Kramer , F. , Shimizu-Albergine , M. , Kanter , J. E. , Snell-Bergeon , J. , Fisher , E. A. , & Shao , B. ( 2024 ). Imbalance of APOB lipoproteins and large HDL in type 1 diabetes drives atherosclerosis . Circulation research , 135 ( 2 ), 335 – 349 . OpenUrl CrossRef PubMed Kouchaksaraei , Y. M. , Ghazalian , F. , Abediankenari , S. , Ebrahim , K. , & Abednatanzi , H. ( 2022 ). Determination of CRP blood level in type 1 diabetic patients and the effect of aerobic and resistance training on the level of this biomarker . Caspian Journal of Internal Medicine , 13 ( 1 ), 38 . OpenUrl PubMed Kyläheiko , I. , Tarkkonen , A. , Martola , J. , Paajanen , T. , Virkkala , J. , Groop , P.-H. , Thorn , L. M. , Putaala , J. , Gordin , D. , & Jokinen , H. ( 2024 ). Cerebral microbleeds are associated with slowed processing speed in middle-aged adults with type 1 diabetes . Cerebral Circulation-Cognition and Behavior , 6 , 100276 . OpenUrl Lacy , M. E. , Gilsanz , P. , Karter , A. J. , Quesenberry , C. P. , Pletcher , M. J. , & Whitmer , R. A. ( 2018 ). Long-term glycemic control and dementia risk in type 1 diabetes . Diabetes Care , 41 ( 11 ), 2339 – 2345 . OpenUrl Abstract / FREE Full Text ↵ Lappalainen , T. , Sammeth , M. , Friedländer , M. R. , ‘t Hoen , P. A. , Monlong , J. , Rivas , M. A. , Gonzalez-Porta , M. , Kurbatova , N. , Griebel , T. , & Ferreira , P. G. ( 2013 ). Transcriptome and genome sequencing uncovers functional variation in humans . Nature , 501 ( 7468 ), 506 – 511 . OpenUrl CrossRef PubMed Web of Science ↵ Li , G. , Jima , D. , Wright , F. A. , & Nobel , A. B. ( 2018 ). HT-eQTL: integrative expression quantitative trait loci analysis in a large number of human tissues . BMC bioinformatics , 19 , 1 – 11 . OpenUrl CrossRef PubMed ↵ Lin , M. , Qiao , P. , Matschi , S. , Vasquez , M. , Ramstein , G. P. , Bourgault , R. , Mohammadi , M. , Scanlon , M. J. , Molina , I. , & Smith , L. G. ( 2022 ). Integrating GWAS and TWAS to elucidate the genetic architecture of maize leaf cuticular conductance . Plant Physiology , 189 ( 4 ), 2144 – 2158 . OpenUrl CrossRef PubMed ↵ Liu , X. , Finucane , H. K. , Gusev , A. , Bhatia , G. , Gazal , S. , O’Connor , L. , Bulik-Sullivan , B. , Wright , F. A. , Sullivan , P. F. , & Neale , B. M. ( 2017 ). Functional architectures of local and distal regulation of gene expression in multiple human tissues . The American journal of human genetics , 100 ( 4 ), 605 – 616 . OpenUrl CrossRef PubMed Liu , Z. , Jeppesen , P. B. , Gregersen , S. , Larsen , L. B. , & Hermansen , K. ( 2016 ). Chronic exposure to proline causes aminoacidotoxicity and impaired beta-cell function: studies in vitro . The review of diabetic studies: RDS , 13 ( 1 ), 66 . OpenUrl PubMed Loxton , P. , Narayan , K. , Munns , C. F. , & Craig , M. E. ( 2021 ). Bone mineral density and type 1 diabetes in children and adolescents: a meta-analysis . Diabetes care , 44 ( 8 ), 1898 – 1905 . OpenUrl Abstract / FREE Full Text ↵ Mancuso , N. , Shi , H. , Goddard , P. , Kichaev , G. , Gusev , A. , & Pasaniuc , B. ( 2017 ). Integrating gene expression with summary association statistics to identify genes associated with 30 complex traits . The American journal of human genetics , 100 ( 3 ), 473 – 487 . OpenUrl CrossRef PubMed ↵ Marees , A. T. , De Kluiver , H. , Stringer , S. , Vorspan , F. , Curis , E. , MarieLClaire , C. , & Derks , E. M. ( 2018 ). A tutorial on conducting genomeLwide association studies: Quality control and statistical analysis . International journal of methods in psychiatric research , 27 ( 2 ), e1608 . OpenUrl CrossRef PubMed Menart-Houtermans , B. , Rütter , R. , Nowotny , B. , Rosenbauer , J. , Koliaki , C. , Kahl , S. , Simon , M.-C. , Szendroedi , J. , Schloot , N. C. , & Roden , M. ( 2014 ). Leukocyte profiles differ between type 1 and type 2 diabetes and are associated with metabolic phenotypes: results from the German Diabetes Study (GDS) . Diabetes care , 37 ( 8 ), 2326 – 2333 . OpenUrl Abstract / FREE Full Text Mohammadi Pejman 5 6 Park YoSon 11 Parsana Princy 12 Segrè Ayellet V. 1 Strober Benjamin J. 9 Zappala Zachary 7 8, G. C. L. a. A. F. B. A. A. C. S. E. D. J. R. H. Y. J. B., P. 19 Volpi Simona 19, N. p. m. A. A. G. P. K. S. L. A. R. L. N. C. M. H. M. R. A. S. J., 16, P. S. L. B. M. E. B. P. A., & 137, N. C. F. N. C. R . ( 2017 ). Genetic effects on gene expression across human tissues . Nature , 550 ( 7675 ), 204 – 213 . OpenUrl CrossRef PubMed Web of Science Nelson , J. ( 2021 ). The Effects of Sodium Bicarbonate on Type One Diabetes Development in Two Mouse Models . ↵ Orchard , T. J. , Costacou , T. , Kretowski , A. , & Nesto , R. W. ( 2006 ). Type 1 diabetes and coronary artery disease . Diabetes care , 29 ( 11 ), 2528 – 2538 . OpenUrl FREE Full Text Pan , X. , Kaminga , A. C. , Kinra , S. , Wen , S. W. , Liu , H. , Tan , X. , & Liu , A. ( 2022 ). Chemokines in type 1 diabetes mellitus . Frontiers in immunology , 12 , 690082 . OpenUrl PubMed ↵ Patil , P. , & Parmigiani , G. ( 2018 ). Training replicable predictors in multiple studies . Proceedings of the National Academy of Sciences , 115 ( 11 ), 2578 – 2583 . OpenUrl Abstract / FREE Full Text ↵ Patiño-Fernández , A. M. , Eidson , M. , Sanchez , J. , & Delamater , A. M. ( 2009 ). What do youth with type 1 diabetes know about the HbA1c test? Children’s health care , 38 ( 2 ), 157 – 167 . OpenUrl Pollakova , D. , Tubili , C. , Di Folco , U. , De Giuseppe , R. , Battino , M. , & Giampieri , F. ( 2023 ). Muscular involvement in long-term type 1 diabetes: Does it represent an underestimated complication? Nutrition , 112 , 112060 . OpenUrl PubMed ↵ Purcell , S. , Neale , B. , Todd-Brown , K. , Thomas , L. , Ferreira , M. A. , Bender , D. , Maller , J. , Sklar , P. , De Bakker , P. I. , & Daly , M. J. ( 2007 ). PLINK: a tool set for whole-genome association and population-based linkage analyses . The American journal of human genetics , 81 ( 3 ), 559 – 575 . OpenUrl CrossRef PubMed Rusak , E. , Rotarska-Mizera , A. , Adamczyk , P. , Mazur , B. , Polanska , J. , & Chobot , A. ( 2018 ). Markers of anemia in children with type 1 diabetes . Journal of diabetes research , 2018 ( 1 ), 5184354 . OpenUrl PubMed ↵ Saha , A. , Kim , Y. , Gewirtz , A. D. , Jo , B. , Gao , C. , McDowell , I. C. , Engelhardt , B. E. , Battle , A. , Aguet , F. , & Ardlie , K. G. ( 2017 ). Co-expression networks reveal the tissue-specific regulation of transcription and splicing . Genome research , 27 ( 11 ), 1843 – 1858 . OpenUrl Abstract / FREE Full Text ↵ Schneider , D. J. ( 2009 ). Factors contributing to increased platelet reactivity in people with diabetes . Diabetes Care , 32 ( 4 ), 525 – 527 . OpenUrl FREE Full Text Schofield , J. , Ho , J. , & Soran , H. ( 2019 ). Cardiovascular risk in type 1 diabetes mellitus . Diabetes Therapy , 10 ( 3 ), 773 – 789 . OpenUrl PubMed Semova , I. , Levenson , A. E. , Krawczyk , J. , Bullock , K. , Williams , K. A. , Wadwa , R. P. , Shah , A. S. , Khoury , P. R. , Kimball , T. R. , & Urbina , E. M. ( 2019 ). Type 1 diabetes is associated with an increase in cholesterol absorption markers but a decrease in cholesterol synthesis markers in a young adult population . Journal of clinical lipidology , 13 ( 6 ), 940 – 946 . OpenUrl PubMed Shah , V. N. , Carpenter , R. D. , Ferguson , V. L. , & Schwartz , A. V. ( 2018 ). Bone health in type 1 diabetes . Current Opinion in Endocrinology, Diabetes and Obesity , 25 ( 4 ), 231 – 236 . OpenUrl Sharma , A. , Purohit , S. , Sharma , S. , Bai , S. , Zhi , W. , Ponny , S. R. , Hopkins , D. , Steed , L. , Bode , B. , & Anderson , S. W. ( 2016 ). IGF-binding proteins in type-1 diabetes are more severely altered in the presence of complications . Frontiers in endocrinology , 7 , 2 . OpenUrl PubMed ↵ Spencer , C. C. , Su , Z. , Donnelly , P. , & Marchini , J. ( 2009 ). Designing genome-wide association studies: sample size, power, imputation, and the choice of genotyping chip . PLoS genetics , 5 ( 5 ), e1000477 . OpenUrl PubMed ↵ Su , K. , Jia , Z. , Wu , Y. , Sun , Y. , Gao , Q. , Jiang , Z. , & Jiang , J. ( 2023 ). A network causal relationship between type-1 diabetes mellitus, 25-hydroxyvitamin D level and systemic lupus erythematosus: Mendelian randomization study . PloS One , 18 ( 5 ), e0285915 . OpenUrl PubMed ↵ Sul , J. H. , Han , B. , Ye , C. , Choi , T. , & Eskin , E. ( 2013 ). Effectively identifying eQTLs from multiple tissues by combining mixed model and meta-analytic approaches . PLoS genetics , 9 ( 6 ), e1003491 . OpenUrl PubMed ↵ Sun , R. , & Lin , X. ( 2017 ). Set-based tests for genetic association using the generalized Berk-Jones statistic . arXiv preprint arXiv: 1710.02469 . Suvitaival , T. ( 2020 ). Lipidomic abnormalities during the pathogenesis of type 1 diabetes: a quantitative review . Current Diabetes Reports , 20 , 1 – 9 . OpenUrl CrossRef PubMed ↵ Tam , V. , Patel , N. , Turcotte , M. , Bossé , Y. , Paré , G. , & Meyre , D. ( 2019 ). Benefits and limitations of genome-wide association studies . Nature Reviews Genetics , 20 ( 8 ), 467 – 484 . OpenUrl CrossRef PubMed Thorn , L. M. , Forsblom , C. , Fagerudd , J. , Thomas , M. C. , Pettersson-Fernholm , K. , Saraheimo , M. , WADen , J. , Ronnback , M. , RosengaLrd-BaLrlund , M. , & BjoLrkesten , C.-G. r. a. ( 2005 ). Metabolic syndrome in type 1 diabetes: association with diabetic nephropathy and glycemic control (the FinnDiane study) . Diabetes care , 28 ( 8 ), 2019 – 2024 . OpenUrl Abstract / FREE Full Text Tian , Z. , Mclaughlin , J. , Verma , A. , Chinoy , H. , & Heald , A. H. ( 2021 ). The relationship between rheumatoid arthritis and diabetes mellitus: a systematic review and meta-analysis . Cardiovascular endocrinology & metabolism , 10 ( 2 ), 125 – 131 . OpenUrl PubMed ↵ Tibshirani , R. ( 1996 ). Regression shrinkage and selection via the lasso . Journal of the Royal Statistical Society Series B: Statistical Methodology , 58 ( 1 ), 267 – 288 . OpenUrl CrossRef PubMed Web of Science ↵ Visscher , P. M. , Wray , N. R. , Zhang , Q. , Sklar , P. , McCarthy , M. I. , Brown , M. A. , & Yang , J. ( 2017 ). 10 years of GWAS discovery: biology, function, and translation . The American journal of human genetics , 101 ( 1 ), 5 – 22 . OpenUrl CrossRef PubMed ↵ Vujkovic , M. , Keaton , J. M. , Lynch , J. A. , Miller , D. R. , Zhou , J. , Tcheandjieu , C. , Huffman , J. E. , Assimes , T. L. , Lorenz , K. , & Zhu , X. ( 2020 ). Discovery of 318 new risk loci for type 2 diabetes and related vascular outcomes among 1.4 million participants in a multi-ancestry meta-analysis . Nature Genetics , 52 ( 7 ), 680 – 691 . OpenUrl CrossRef PubMed ↵ Wainberg , M. , Sinnott-Armstrong , N. , Knowles , D. A. , Golan , D. , David , R. , Ruusalepp , A. , Quertermous , T. , Hao , K. , Björkegren , J. L. , & Rivas , M. A. ( 2017 ). Vulnerabilities of transcriptome-wide association studies . bioRxiv , 206961 . Wang , Q. , Long , M. , Qu , H. , Shen , R. , Zhang , R. , Xu , J. , Xiong , X. , Wang , H. , & Zheng , H. ( 2018 ). DPPL4 inhibitors as treatments for type 1 diabetes mellitus: a systematic review and metaLanalysis . Journal of diabetes research , 2018 ( 1 ), 5308582 . OpenUrl PubMed Wang , Y. , Yang , P. , Yan , Z. , Liu , Z. , Ma , Q. , Zhang , Z. , Wang , Y. , & Su , Y. ( 2021 ). The relationship between erythrocytes and diabetes mellitus . Journal of diabetes research , 2021 ( 1 ), 6656062 . OpenUrl PubMed Wen , Q. , Wu , F. , Yang , J. , Wu , J. , Zhang , X. , He , M. , Wu , T. , & Cheng , L. ( 2015 ). A singlenucleotide polymorphism in the interleukin 1 receptorLassociated protein gene is associated with impaired glucose regulation and type 2 diabetes in a caseLcontrolled study . Biomedical Reports , 3 ( 4 ), 549 – 553 . OpenUrl PubMed ↵ West , J. , Brousil , J. , Gazis , A. , Jackson , L. , Mansell , P. , Bennett , A. , & Aithal , G. ( 2006 ). Elevated serum alanine transaminase in patients with type 1 or type 2 diabetes mellitus . Journal of the Association of Physicians , 99 ( 12 ), 871 – 876 . OpenUrl ↵ Yang , F. , Wang , J. , Pierce , B. L. , Chen , L. S. , Aguet , F. , Ardlie , K. G. , Cummings , B. B. , Gelfand , E. T. , Getz , G. , & Hadley , K. ( 2017 ). Identifying cis-mediators for trans-eQTLs across many human tissues using genomic mediation analysis . Genome research , 27 ( 11 ), 1859 – 1871 . OpenUrl Abstract / FREE Full Text ↵ Zhang , X. , Joehanes , R. , Chen , B. H. , Huan , T. , Ying , S. , Munson , P. J. , Johnson , A. D. , Levy , D. , & O’Donnell , C. J. ( 2015 ). Identification of common genetic variants controlling transcript isoform variation in human whole blood . Nature Genetics , 47 ( 4 ), 345 – 352 . OpenUrl CrossRef PubMed Zheng , P. , Li , Z. , & Zhou , Z. ( 2018 ). Gut microbiome in type 1 diabetes: A comprehensive review . Diabetes/metabolism research and reviews , 34 ( 7 ), e3043 . OpenUrl CrossRef PubMed ↵ Zhou , D. , Jiang , Y. , Zhong , X. , Cox , N. J. , Liu , C. , & Gamazon , E. R. ( 2020 ). A unified framework for joint-tissue transcriptome-wide association and Mendelian randomization analysis . Nature genetics , 52 ( 11 ), 1239 – 1246 . OpenUrl CrossRef PubMed ↵ Zhu , Z. , Zhang , F. , Hu , H. , Bakshi , A. , Robinson , M. R. , Powell , J. E. , Montgomery , G. W. , Goddard , M. E. , Wray , N. R. , & Visscher , P. M. ( 2016 ). Integration of summary data from GWAS and eQTL studies predicts complex trait gene targets . Nature Genetics , 48 ( 5 ), 481 – 487 . OpenUrl CrossRef PubMed ↵ Zou , H. , & Hastie , T. ( 2005 ). Regularization and variable selection via the elastic net . Journal of the Royal Statistical Society Series B: Statistical Methodology , 67 ( 2 ), 301 – 320 . OpenUrl CrossRef Web of Science View the discussion thread. Back to top Previous Next Posted November 22, 2025. Download PDF Data/Code Email Thank you for your interest in spreading the word about bioRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following GBoost-CTL: A novel method in multi-tissue transcriptome-wide associations studies in cross-tissue learner incorporating GWAS information Message Subject (Your Name) has forwarded a page to you from bioRxiv Message Body (Your Name) thought you would like to see this page from the bioRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share GBoost-CTL: A novel method in multi-tissue transcriptome-wide associations studies in cross-tissue learner incorporating GWAS information Md Mutasim Billah , Hairong Wei , Fengzhu Sun , Kui Zhang bioRxiv 2025.11.22.689782; doi: https://doi.org/10.1101/2025.11.22.689782 Share This Article: Copy Citation Tools GBoost-CTL: A novel method in multi-tissue transcriptome-wide associations studies in cross-tissue learner incorporating GWAS information Md Mutasim Billah , Hairong Wei , Fengzhu Sun , Kui Zhang bioRxiv 2025.11.22.689782; doi: https://doi.org/10.1101/2025.11.22.689782 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Bioinformatics Subject Areas All Articles Animal Behavior and Cognition (7642) Biochemistry (17715) Bioengineering (13907) Bioinformatics (42003) Biophysics (21470) Cancer Biology (18624) Cell Biology (25533) Clinical Trials (138) Developmental Biology (13390) Ecology (19935) Epidemiology (2067) Evolutionary Biology (24356) Genetics (15617) Genomics (22529) Immunology (17753) Microbiology (40432) Molecular Biology (17200) Neuroscience (88681) Paleontology (667) Pathology (2840) Pharmacology and Toxicology (4828) Physiology (7653) Plant Biology (15161) Scientific Communication and Education (2046) Synthetic Biology (4304) Systems Biology (9826) Zoology (2271)

Text is read by the "Ask this paper" AI Q&A widget below. Extraction quality varies by source — PMC NXML preserves structure cleanly, OA-HTML may include some navigation residue, and OA-PDF can have broken hyphenation. The publisher copy (via DOI) is the canonical version.

My notes (saved in your browser only)

Ask this paper AI returns verbatim quotes from the full text · source: preprint-html

Answers must be backed by verbatim quotes from this paper's full text. Hallucinated quotes are dropped automatically; if no verbatim passage answers the question, we say so. How this works

Citation neighborhood (no data yet)

We don't have any in-corpus citations linked to this paper yet. This is a recent paper (2025) — citers typically take a year or two to land, and the OpenAlex reference graph may still be filling in.

Source provenance

europepmc
last seen: 2026-05-20T01:45:00.602351+00:00