Full text
57,727 characters
· extracted from
preprint-html
· click to expand
Trans-ancestry Mendelian Randomization Discovers Novel Causal Genes for Autoimmune Disease Traits | bioRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-M677548'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search New Results Trans-ancestry Mendelian Randomization Discovers Novel Causal Genes for Autoimmune Disease Traits Siyuan Chen , Chen Wang , Kent D. Taylor , Jerome I. Rotter , Xiuqing Guo , Stephen S. Rich , Ani Manichaikul , Dajiang Liu doi: https://doi.org/10.1101/2025.04.17.649410 Siyuan Chen 1 Department of Public Health Sciences, College of Medicine, Penn State University , Hershey, PA, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site Chen Wang 1 Department of Public Health Sciences, College of Medicine, Penn State University , Hershey, PA, USA 2 Bioinformatics and Genomics Graduate Program, College of Medicine, Penn State University , Hershey, PA, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site Kent D. Taylor 3 The Institute for Translational Genomics and Population Sciences, Department of Pediatrics, The Lundquist Institute for Biomedical Innovation at Harbor-UCLA Medical Center , Torrance, CA USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site Jerome I. Rotter 3 The Institute for Translational Genomics and Population Sciences, Department of Pediatrics, The Lundquist Institute for Biomedical Innovation at Harbor-UCLA Medical Center , Torrance, CA USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site Xiuqing Guo 3 The Institute for Translational Genomics and Population Sciences, Department of Pediatrics, The Lundquist Institute for Biomedical Innovation at Harbor-UCLA Medical Center , Torrance, CA USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site Stephen S. Rich 4 Department of Genome Sciences, University of Virginia , Charlottesville, VA USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site Ani Manichaikul 4 Department of Genome Sciences, University of Virginia , Charlottesville, VA USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site Dajiang Liu 1 Department of Public Health Sciences, College of Medicine, Penn State University , Hershey, PA, USA 2 Bioinformatics and Genomics Graduate Program, College of Medicine, Penn State University , Hershey, PA, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site For correspondence: dajiang.liu{at}psu.edu Abstract Full Text Info/History Metrics Preview PDF Abstract Mendelian randomization (MR) using summary statistics from genome-wide association studies (GWAS) has become a powerful tool for dissecting the causal relationships between exposure and outcomes. As GWAS begins to incorporate samples of diverse ancestries, the optimal strategy to perform MR analysis integrating these GWAS results remains to be determined. To fill in this gap, we proposed a robust method that aggregates the genetic variant–risk factor association summary statistics of multiple ancestries by modelling the genetic effects of the exposure as a function of the principal components of genome-wide allele frequency. This allows borrowing information from different ancestry groups and accounting for potential between-ancestry heterogeneities. The method is general and can be used with different MR methods. We demonstrated that our method improves accuracy and power compared to the inverse-variance weighting method based on only ancestry-stratified samples, with additional benefits in correcting winner’s curse bias. We also illustrated the flexibility and potential of discovering novel causal genes for autoimmune disease traits of our method in practice. Introduction Mendelian Randomization (MR) has been a popular tool for inferring causal relationships between factors of interest (exposure) and outcomes in observational studies. With genetic variants (e.g., SNPs) as instrumental variables (IVs), the inference of causal effects can be made for the exposures that are heritable, using GWAS summary statistics as input. In two-sample MR, the variant-exposure and variant- outcome associations are derived from two non-overlapping samples, and ideally, they should be from the same underlying population[ 1 ]. When multi-ancestry GWAS summary statistics are available, standard inverse variance weighted (IVW) MR analysis would need to be conducted in each ancestry separately, which directly leads to a reduction in effective sample size, and hence possibly depletion in performance. The genetic effect may differ between ancestries. More than simply pooling multiple ancestry cohorts together using fixed-effect meta-analysis may be underpowered. Previously, we developed methods that model the genetic effects of each cohort as a function of the principal components of genome-wide allele frequencies [ 2 ]. This adjustment allows us to accommodate the heterogeneities of genetic effect sizes between ancestries and borrow strength across ancestries to improve the precision of the genetic effect estimates. Extending the framework to improve multi-ancestry MR would be necessary to improve the power of MR methods. Furthermore, a two-sample setting raises concern for the winner’s curse bias, which occurs when the discovery GWAS for IV selection is also used for association estimation[ 3 ]. It is well known that genetic effects for variants selected as IVs tend to be inflated in the original discovery data set, especially when the power is low for genome-wide association studies. This may induce downward biases in the IVW estimator, and hence, a three-sample design is recommended, but it is not always feasible. Re-estimating the IV-exposure association using multi-ancestry summary statistics partially corrects for winner’s curse bias. It can be extended by incorporating the conditional likelihood method by Ghosh et al. [ 4 ] to include the conditional probability of the IV selection event within only matched global ancestry. Here, we present our method, transMR, which aggregates information from multi-ancestry GWAS summary statistics for MR analysis. Compared to the IVW method, our model allows us to obtain a more plausible estimation of variant-exposure association from available summary statistics for the target population, i.e., of which the outcome cohort is. This is helpful especially when the sample size of the target population is small, as we will demonstrate in our simulation. In addition, our method brings adjustment on only associations and choice of instruments, which can be combined with other popular MR methods which focus on tackling issues like pleiotropic effects, such as GSMR[ 5 ]. Furthermore, we implement transMR in large-scale multi-ancestry studies that combine expression quantitative trait loci and GWAS. Our results identify more plausible causal traits in cohorts of relatively small sample sizes, as shown in our data application in causal gene discovery for autoimmune diseases. Results Method Overview In a univariate MR analysis, our goal is to assess whether the exposure of interest has a causal effect on the outcome by utilizing genetic instrumental variables. Genetic variants serving as instruments must satisfy the three core IV assumptions, where the IVs should be associated with the exposure, uncorrelated with the confounders, and the IVs should only be associated with the outcome via their effects on the exposure (see Methods and Materials). The exchangeability assumption requires that instruments should not be confounded by confounders of the exposure-outcome relationship. At the same time, it is known that the distribution of SNPs and genetic effects can differ by ancestry structure. When only summary statistics are available, in a two-sample setting, the GWAS summary statistics should come from the same underlying population, otherwise the values of true IV-exposure/outcome effect may not be identical. Finally, the first-order inverse-variance weights are used to calculate the IVW estimate, and this assumes that the standard error of the estimated IV-exposure association is zero (the NOME assumption[ 6 ]). We aim to find a better strategy for utilizing the GWAS results from multi-ancestry studies. Our method, transMR, uses a weighted meta-regression to model the genetic variant-exposure association (‘s) with principal components of the allele frequency (AF) matrix, which serves as quantitative proxies of ancestry. Estimation from each ancestry is weighted by the inverse of its variance in the regression. To avoid overfitting, we also included a ridge penalty on the regression parameters for the PCs as ancestry proxies. Such a model can be extended to further correct for winner’s curse bias brought by the two- sample design by replacing the likelihood contribution of estimation of β from the target population into a conditional likelihood that is conditional on the genetic effect. To avoid overfitting, we also included a ridge penalty on the regression parameters for the PCs as ancestry proxies. Such a model can be extended to further correct for winner’s curse bias brought by the two-sample design by replacing the likelihood contribution of estimation of β from the target population into a conditional likelihood that is conditional on the genetic effect being significant according to a pre-specified significance threshold (Details in Methods and Materials). The winner’s curse correction is performed for each SNP separately. Based on the proposed model, one can obtain an adjusted estimation of β in each population (possibly of different ancestries) on which the genetic variant-outcome association (Γ) was estimated. The improved genetic effect estimates can be used with any MR method. Furthermore, when the target population for inference has a relatively small sample size (which is likely the case for a non-European population) and more significant uncertainty in estimated association arises, the violation of the NOME assumption is unignorable, transMR can borrow information from other population of a decent sample size to reduce such uncertainty and make a better estimation for the causal effect while still residing in the IVW framework. Trans-ancestry MR increases power for underrepresented population We refer to the method without conditional likelihood transMR, while the one with correction for winner’s curse bias as transMR-c and that uses BIC to select the ridge penalty parameter as transMR- cBIC. We evaluate the performance of transMR methods through numerical simulations. In detail, we investigated the accuracy and precision of estimating the true causal effect θ = 0.5 with transMR and the original IVW method when all IVs are relatively strong and valid. We also considered two study designs. In the two-sample design, summary statistics for the exposure and outcome are obtained from two sets of independent samples, while in the three-sample design, another separate set of samples is used to select genetic variants with significant p-values as IV. Note that the three-sample design serves as a gold standard in our simulation. We also compared the power of proposed methods by setting true causal effect θ = 0.1 and the following extra two scenarios: (1) most of the IVs are weak; (2) the target population is underrepresented. The heterogeneity in genetic effects on the exposure is controlled by a scale parameter φ ∈ {0,0.5,1,1.5}. We start by generating individual-level data of multi-ancestry samples, running ancestry-stratified single-trait GWAS analysis, and then following a standard MR analysis workflow based only on summary-level data. We select IVs with a fixed p-value threshold and finally make inference on θ with candidate methods respectively. We ran about 2000 iterations for each scenario for accuracy evaluation. In each case of SNP-exposure effect, along with no pleiotropy but unmeasured confounding, our transMR method proves to have a smaller bias, comparable variance and, in general, smaller MSE. The transMR-c and transMR-cBIC constantly have larger variances than other methods, but this also leads to higher coverage probability. Table (1) gives a detailed statistics table. View this table: View inline View popup Download powerpoint Table 1. Average bias, mean squared errors (MSE) and coverage probability (CP) for transMR and IVW methods. We also investigated the power and type-I errors of the proposed methods with around 5000 iterations. All methods have controlled type-I error, and there is a slight improvement in power from transMR and transMR-c when (1) no requirement of IV is violated; (2) most SNP effects are relatively small. We want to point out that when the target population is of a small sample size, transMR remarkably improved in power (see Table (2) ). In detail, IVW estimators are biased towards 0, although they still have a smaller variance. Notably, transMR methods outperformed the three-sample design with a relatively small population for discovery GWAS, suggesting that even though a separate cohort is available for GWAS and IV selection for underrepresented population, the performance is not as good as directly borrowing information from cohorts of a larger sample size but of different ancestry. View this table: View inline View popup Download powerpoint Table 2. Power of transMR methods and IVW methods with underrepresented population. β j denotes the true genetic effects on the exposure. Trans-ancestry MR discovers novel causal genes for autoimmune disease traits in East Asian population Autoimmune diseases (AiDs), e.g., systemic lupus erythematosus, have a strong genetic influence[ 7 ]. Genome-wide association studies have identified hundreds of loci associated with several AiDs, but actual causal genes remain unclear, especially for the East Asian population, due to limited data availability. We first conducted expression quantitative trait loci (eQTL) analysis on the gene expression of human peripheral blood mononuclear cells (PBMC) from the Multi-Ethnic Study of Atherosclerosis (MESA)[ 8 ], which is a multi-omics pilot study of the NHLBI Trans-Omics for Precision Medicine (TOPMed) consortium. It comprises gene expressions obtained from whole genome sequencing, as well as transcripts per million (TPM) values derived from RNA-Seq. The samples originate from four distinct populations: African American (AFA, n = 381), East Asian (ASN, n = 106), European (EUR, n = 538), and Hispanic/Latino (HIS, n = 292). Besides, our study used GWAS summary statistics from samples of East Asian ancestry as our target cohort for Systemic lupus erythematosus (SLE) and five genetically correlated autoimmune diseases (Sjogren’s syndrome, type 1 diabetes, ulcerative colitis, Hashimoto’s disease and Graves’ disease) in [ 9 ]. Because our exposure of interest is gene expression, we compare our method transMR to the popular MR framework that integrates gene expression study and GWAS, GSMR[ 5 ]. In detail, for cis-SNPs after LD- pruning (r 2 < 0.3), GSMR exclude possible pleiotropic IVs by the HEIDI-Outlier test. While GSMR estimates the causal effect with IVW estimators based on remaining SNPs, we extend this framework with transMR, which models trans-ancestry eQTL effects and use them to estimate and refine the effects in the target cohort, then conduct HEIDI-Outlier test to remove pleiotropic IVs. Also, we extended a robust pleiotropy method for multivariable Mendelian randomization, MVMR-robost[ 10 ], in order to explore possible patterns of genes regulating each other’s expression levels. Our method transMR managed to identify more significant causal genes to the autoimmune diseases we considered comparing to the traditional IVW method, and the numbers of significant genes after the Benjamini-Hochberg correction are shown in Table (3) ; Most of these genes are novel findings for East Asian population (see Supplemental Materials for the full list of genes). View this table: View inline View popup Download powerpoint Table 3. The numbers of significant causal genes for SLE after the Benjamini-Hochberg correction from IVW and transMR. The novel genes from transMR methods are defined as causal genes that are not identified by IVW. P-values are ACAT aggregated and BH adjusted (See Materials and Methods for details). First, transMR identified more causal genes from the HLA region. A strong association between the HLA region and autoimmune disease has been recognized for decades[ 11 ]. TransMR was able to pick up the well-known genetic effects such as HLA-DQA1 on Type I diabetes (T1D) and HLA-B on Grave’s disease (GD)[ 12 ]. Existing evidence from studies on the East Asian population supports the causal effects of genes such as HLA-C for Ulcerative colitis (UC)[ 13 ] and MICA for Hashimoto’s disease (HD)[ 14 ]. While previous works have limited their recognition of associations to HLA classes I and II, we noticed that transMR marked some causal genes from HLA class III, such as HLA-F on SLE, HLA-H on HD, and C4B on GD. Besides, several genes were identified as causal for more than one autoimmune disease. Krüppel-like factor 2 ( KLF2 ) is involved in regulatory pathways of regulatory T cells and apoptotic cell clearance, which are related to autoimmunity prevention[ 15 ]. KLF2 had a significant causal effect on GD and SLE, and the latter relationship agrees with a GWA study on the Northern Chinese population[ 16 ]. Other causal genes in common are MORN3 (T1D and HD) and RPL23AP1 (UC and HD). Finally, for SLE specifically, transMR managed to make a causal conclusion for genes that were previously reported to be associated with the disease besides the KLF2 mentioned before, naming MBP [ 17 ], STK19 [ 18 ], and NEAT1 [ 19 ]. Genes such as PSTPIP1 and CLEC12A possibly are on the pathological pathways of SLE, as the former was reported to control immune synapse stability in human T cells[ 20 ], and the latter was reported to be downregulated by SLE immune complexes in normal human peripheral blood mononuclear cells[ 21 ]. Trans-ancestry extended MVMR analysis was able to identify genes PBX2, TNXB and HLA-DRB1 with direct causal effects on the disease trait, compared to the default MVMR analysis. PBX2 was reported to be aberrantly methylated in T cells, B cells, and/or monocytes in SLE patients[ 22 ]. TNXB was reported to be a candidate gene susceptible to SLE in the Japanese population[ 23 ]. HLA-DRB1 , belonging to HLA Class II, has multiple detected risk alleles for SLE in the East Asian population[ 24 ] [ 25 ] [ 26 ]. We conducted the same set of analyses for the population of European ancestry based on the summary statistics for SLE. TransMR managed to identify a comparable number of causal genes to IVW (188 vs 192) for all genes available for MR after LD pruning and HEIDI-Outlier test. Then we identify potentially East Asian-specific causal genes including PSTPIP1 and MBP (see Supplemental Materials for the full list of genes). Although GWA studies have reported risk loci and ancestry-specific risk loci for SLE, limited studies tested for transcriptome-wide association to make inferences on the gene expression level for the East Asian population (see [ 27 ]). Our findings shed new insights into genes of risk for SLE. Finally, we conducted computational drug repurposing analysis using the Connectivity Map (CMap) for Systemic Lupus Erythematosus (SLE) and Graves’ disease with significant causal genes and identified several therapeutic targets. For SLE, current treatments and medications include anti-inflammatory and immunosuppressive drugs such as antimalarial drugs and glucocorticoids[ 28 ]. Anti-inflammatory drugs are beneficial in treating autoimmune diseases like Systemic Lupus Erythematosus (SLE) because they help mitigate the overactive immune response characteristic of these diseases. SLE involves chronic inflammation driven by autoantibodies targeting various cellular components, leading to systemic tissue damage, particularly in organs like the kidneys, skin, and joints. Anti-inflammatory drugs, such as glucocorticoids, reduce cytokine production, diminish immune cell recruitment to inflammation sites, and lower overall immune system activity, thereby reducing tissue damage and alleviating symptoms. Antimalarial drugs, like hydroxychloroquine and, as identified, Artesunate, offer specific benefits for autoimmune conditions due to their immunomodulatory effects beyond their traditional use. They inhibit certain pathways, such as Toll-like receptor (TLR) signaling, which plays a role in immune activation and the production of inflammatory cytokines. By interfering with these pathways, antimalarials help regulate the immune system, reducing inflammation and autoimmune activity. Further, inhibitors like IKK-2 target critical nodes in inflammatory pathways, such as the NF-κB pathway, which is pivotal for immune response regulation. NF-κB activation leads to the transcription of various pro-inflammatory genes; thus, blocking IKK-2 can dampen the downstream inflammatory response. This mechanism is especially relevant in SLE, where persistent NF-κB activity contributes to ongoing immune activation. Similarly, kinase inhibitors (e.g., H-7, Tyrphostin-AG-835), including MEK1/2 inhibitors, offer therapeutic potential by intervening in specific signaling pathways essential for immune cell activation, thereby providing a targeted approach to modulate the immune response without broadly suppressing the immune system. Besides, Janus kinase inhibitors are of interest for treating SLE[ 37 ] and we identified BRD-K06817181. For Graves’s disease, immunosuppressors such as Prednisolone, Tofacitinib, and Triptolide were identified. Discussion We introduced transMR, a flexible method for inferring causal effects between pairs of traits based on summary statistics from trans-ancestry GWAS. TransMR uses a meta-regression model for SNP-exposure effects that enables aggregating summary statistics from ancestry-stratified GWAS on the same phenotype as more and more multi-ancestry genotype databases occur, which also enables winner’s curse bias correction. We showed by both simulation and real data analysis causal effect estimates from our proposed framework have smaller MSE, better power and controlled type-I error, which are especially crucial when the sample sizes of non-European cohorts are small. By further extending our framework to some existing MR methods based on summary level data (GSMR and MVMR-robust), more robust MR analysis can be conducted for the existence of pleiotropy. As our method is general, transMR can be further extended to accommodate other MR methods, e.g., MRPRESSO[ 29 ] and MRAID[ 30 ] for robust MR analysis under other assumptions, such as the InSIDE assumption[ 28 ]. Several things need to be noted when applying transMR methods to simulated or real data: (1) we used ridge to avoid overfitting as a technical consideration, while the performance was not sensitive towards the choice of λ in our simulation; (2) We need to emphasize that although it has been shown that integrating multi-ancestry information improves the power, it is still important to exclude other confounders if possible when modelling genetic effects across ancestries, since PCs of allele frequency matrix capture population stratification only. The ideal scenario to apply transMR is when summary statistics of association from large-scale, trans-ancestry GWAS are available; (3) while transMR is easier to implement with existing R function glmnet , transMR-c methods require customized likelihood function and extra optimization steps, which may make the method less robust, as shown in our simulation. We also would like to mention that we show through simulation (Supplemental Materials) that recruiting IVs from other ancestry doesn’t seem to have a consistent improvement in the estimation or the power a lot compared to using IVs from identical ancestry only in transMR. The idea of looking for possible IVs from ancestries originated from the spirit of borrowing as much useful information as possible, although it was not reported that the number of IVs influences the performance of IVW estimators. Our work took the common statistical approach of selecting genetic variants, that is, to include all variants associated with the exposure at a fixed level of statistical significance (when not considering pleiotropy), hence there will always be a trade-off between re-picking up a missed IV in the target ancestry and selecting a relatively weak IV which causes bias. We illustrate that transMR shines in discovering causal genes for autoimmune disease traits for the underrepresented East Asian population in our application since transMR identified more novel causal genes within the HLA region, which serve as positive controls, and outside the HLA region but are reported to function in immune initiation and response. On the other hand, when the target population has a large enough sample size, transMR may suffer from missing data or non-informative external data, hence would not notably improve the power, as proved when we implemented transMR in European samples. Methods and Materials Two-sample Mendelian Randomization A classic two-sample Mendelian Randomization analysis framework is described as follows. Two-sample MR considers two separate studies in the setting of GWAS: the GWAS that measures both the exposure and the genotype data on n 1 samples; and the GWAS that measures both the outcome variable of interest and the genotype data on n 2 samples, without sample overlapping. If the selection of genetic instruments is done on another GWAS for exposure, where the cohort has no overlapping individuals with the exposure and disease cohorts, we call this a three-sample design. Denote the genotype matrix for the first study as Z and the n 1 -vector of exposure as x . Denote the genotype matrix for the second study as and the n 2 -vector of the outcome as y . Assume the following linear structural equation model[ 31 ]. In the first step, for the cohort that measures the genetic instrument and the exposure, we model the association as: We then impute the exposure based on the effect estimates from step 1, estimate the expected exposure values and test for the association between the estimated exposure and the outcome, i.e., where is unobserved for the second sample, and and ϵ y denote the random error term for each equation respectively. The goal of MR methods is to make inferences on the causal effect θ, which relies on each of the selected IVs satisfying the IV assumptions mentioned before. Depending on data availability, one can use two-stage least squares (2SLS) methods for individual-level data or inverse- variance weighting (IVW) methods for summary-level data to make inferences on θ. If all IVs are perfectly uncorrelated, the estimate for θ from these two methods are similarly efficient[ 32 ]. Our method focuses on summarized data because GWAS summary statistics are more accessible for studies involving samples of diverse ancestries. In summary-level data, for one genetic variant only and corresponding allele frequency for x j are given. Denote the estimated association and corresponding standard error of j th genetic variant and Y in GWAS summary statistics as and . To serve as a valid instrument, the genetic variant needs to satisfy the following 3 core assumptions: Relevance: The IV is associated with exposure. Exchangeability: The IV is not associated with confounders of exposure-outcome association. Exclusive: The IV is only associated with the outcome through exposure. The Wald ratio estimator for the causal effect θ from the j th instrument is with first-order approximate standard error If all instruments are independent, valid instruments, we can obtain a consistent estimator for θ by the usual inverse-variance weighted (IVW) regression: and its standard error is: Trans-ancestry two-sample Mendelian Randomization To accommodate the situation when multi-ancestry exposure cohorts are available, we assume K cohorts are available and index different ancestries by k, k = 1, …, K , and model each β j with a fixed effect of the ancestry captured by the allele frequency (AF) principal components. Denote the ( K + 1) × J allele frequency matrix of genetic variants shared in all cohorts, along with AF for the same set of variants from the reference panel (e.g., 1000 Genomes Project) as F and its principal components (PCs) as W 1 , …, W K , assuming J ≥ ( K + 1). For the j th instrument, we model its effect on exposure as which is a weighted linear regression model with larger weights assigned to variant-exposure GWAS cohorts with smaller se . Usually, this weight is set to . Each instrument has an individual model, and these models are not related to each other. For instruments unaffected by ancestry, coefficient estimates are expected to remain close to zero. To prevent overfitting in these estimates, we introduce a ridge penalty with a small regularization parameter λ ∈ {0.01,0.5,1} penalizing for large values of η j . The ridge penalty modifies the standard loss function by adding a regularization term, expressed as follows: The winner’s curse bias correction is done by replacing the contribution from the target population, say cohort k, with a log-conditional probability in the ridge penalized log-likelihood for (5): while which accounts for the bias brought by the ridge penalty. Then we make a prediction of β j,K +1 based on the previous model with W K +1 , and denote the predicted values as . The estimator, standard error for θ j and furthermore, θ can be obtained by simply replacing with when using equation (1), (2) and (3). We also consider a 3-sample design as a gold standard to compare with TransMR. Specifically, if another independent exposure sample is available, we can also conduct a 3-sample design, where we use one exposure sample to select the instrument variable and another sample to estimate the genetic effect on the exposure, which can overcome the inflation caused by the winner’s curse. Yet, the availability of an independent GWAS on the exposure is questionable, particularly for non-European populations. So, the consideration of a 3-sample design is theoretical, which can be used as an upper bound for power. Extension of GSMR and MVMR Now we show how to extend existing popular MR methods, in particular, GSMR and MVMR based on our method, so they can be implemented in our data application of identifying causal genes for autoimmune disease traits from samples of diverse ancestries. Both SMR[ 33 ] and GSMR have been applied to perform integrative analysis of gene expression study and GWAS. Specifically, SMR selects one SNP that has the smallest p-value of the marginal association test in the cis-region of the gene to serve as the IV. Afterwards, SMR uses the standard MR ratio method to estimate the causal effect. Different from SMR, which uses only one IV, GSMR selects multiple independent SNPs in the cis-region to serve as IVs. In particular, after LD-pruning, GSMR tests for and rules out possible pleiotropic variants through the HEIDI-Outlier test, estimates the causal effect of each IV in turn using the standard MR ratio method, and combines these causal effect estimates with the standard IVW approach. Next, we illustrate how to conduct the HEIDI-Outlier test after obtaining . Denote the weighting matrix for model (5) as M . The variance of estimation for η j from the regular weighted linear regression forms as where Σ = Var(ϵ). The ridge estimator for η j , can be represented as a linear transformation of the weighted least square estimator: and we have The basic idea of HEIDI-Outlier is to test where there is a significant difference between the estimation from any instrument, , and estimation from a target SNP that shows a strong association with the exposure, . The difference has a variance . Use the χ 2 -statistic and remove the SNPs with p-value < 0.01. Multivariable MR (MVMR) is an extension to MR that uses genetic variants associated with multiple, potentially related exposures to estimate the effect of each exposure on a single outcome. This allows for mediation analysis and therefore can be used to estimate gene regulatory effects[ 34 ]. Based on summary- level data only, this is done by fitting the following weighted regression model: While univariable MR estimates the total effect of the exposure on the outcome, MVMR estimates the marginal effect of the exposure on the outcome conditional on the mediators. The difference between these estimates (i.e., “difference in effects”) will then give the mediated effects. In order to overcome pleiotropic effects, MVMR-Robust described in [ 10 ] uses robust regression methods MM-estimation. In a multi-ancestry setting, we simply plug in the corresponding (and variance estimated by equation (6) into (7). Simulation Studies We generated individual-level data of independent SNPs ( Z ), generic continuous exposure ( Σ ), and outcome ( Y ) for six cohorts (5 with exposure measured, 1 with outcome) of different ancestry of European (EUR), African (AFR), East Asian (ASN), Hispanic (HIS), Brazilian (BRZ), and European (EUR), with independent samples. For the scenario of weak instruments, we set the sample size to 15k to ensure basic power. For the scenario of the underrepresented population, we set the cohorts as (AFR, EUR, ASN, ASN, AFR, AFR), and sample sizes for EUR ancestry as 20000, while ASN and AFR are of 10% and 5%. Otherwise, all cohorts have the same sample size equal to 5000. In each exposure cohort, the variables are generated according to the following structural equations: where β j ’s ∼ N 0.1, 0.05 2 ). For the scenario of weak instruments, we set β j ∼ N (0.08,0.05 2 ) and β j ∼(0.05,0.05 2 ). We use the reference population-stratified allele frequencies reported in [ 35 ] to generate genotypes. Out of 3076 SNPs, we randomly selected 25 as the pool of causal variants for N and 20 out of the pool for each ancestry. SNPs N = ( N 1 , …, N 20 ) denotes the SNP counts (0,1,2) for each causal variant and were generated according to the reference AF, and is the column-wise standardized genotype matrix. Confounder, N , is generated by a normal distribution with fixed pre-specified mean and variance (−0.5,4). The random error terms ϵ x , ϵ y follow the standard normal distribution. The heterogeneity in β’s is generated by ϵ k ∼ N (0, (0.02φ k ) 2 ), which denotes the scale parameter of k th ancestry among EUR, AFR, ASN, HIS and BRZ and is equal to 0, φ, 2φ,0.5φ,0.5φ respectively. We ran a single-trait association test on each cohort using the package GENESIS[ 36 ] and select SNPs with p-values smaller than 1 × 10 −3 which is equivalent to IV strength defined by the approximated F-statistic For multi-ancestry exposure cohorts, we select IVs for the ancestry that matches the outcome study and use them with TransMR for analysis. For example, if the outcome cohort is East Asian, we will use the exposure cohorts with East Asian ancestry to select our IVs. Application We conducted eQTL analysis on the MESA RNASeq data obtained from [ 37 ] with the package MatrixEQTL[ 38 ]. For the genotype data excluding chromosome X, we performed quality control (QC) that removed SNPs that are INDELs, multiallelic, and ambiguous-strand (A/T, T/A, C/G, G/C) and removed the remaining variants with minor allele frequencies (MAFs) < 0.01 and Hardy-Weinberg Equilibrium (HWE) p < 1 × 10 −6 . We filtered for significant cis-eQTLs with p-value < 1e-3 in the ASN population, conducted LD-pruning, and extracted the same set of eQTLs from other ancestries. After filtering for SNPs available in GWAS summary statistics, significant eQTLs and reference panel, and harmonization of alleles, we perform eQTL analysis for more than 7,000 genes. For ridge regression on β jk , we use the first 2 PCs of the AF matrix since there are only four exposure cohorts. To ensure the power after the meta-regression model of eQTL effects, we use 0 PC (i.e., the fixed-effect model), 1 PC, and 2 PCs in the meta-regression step, and based on predicted eQTLs effects, we obtain estimations for causal effect respectively. Then we use the Cauchy combination test [ 39 ] to combine individual p-values from 3 pairs of estimations. We applied Benjamini-Hochberg correction on the p- values for each set of results. Supplemental Materials for “Trans-ancestry Mendelian Randomization Discovers Novel Causal Genes for Autoimmune Disease Traits” View this table: View inline View popup Download powerpoint Table S1. Top 5 novel significant causal genes and number of significant causal genes inferred by transMR methods for autoimmune diseases of Systemic lupus erythematosus, Sjogren’s syndrome, type 1 diabetes, ulcerative colitis, Hashimoto’s disease and Graves’ disease. View this table: View inline View popup Table S2. Full list of novel causal genes inferred by transMR methods for autoimmune diseases of Systemic lupus erythematosus, Sjogren’s syndrome, type 1 diabetes, ulcerative colitis, Hashimoto’s disease and Graves’ disease. Download figure Open in new tab Download figure Open in new tab Figure S3. Multivariate Mendelian Randomization results for causal genes of Systemic lupus erythematosus study 1 and 2. PBX2 and TNXB showed significant direct effect from SLE study 1 (up left), while HLA-DRB1 showed significant direct effect from SLE study 2 (lower left), and the effects of PBX2 and HLA-DRB1 are borderline non-significant when using default MVMR-Robust method (up right, lower right). View this table: View inline View popup Download powerpoint Table S4. TransMR inferred causal genes for SLE in European ancestry population. Significant genes are evaluated on genes that are valid for both inference methods. Download figure Open in new tab Figure S5. Recruiting IVs from other ancestry (labelled as “transMR-global IV”) doesn’t seem to have a consistent improvement in the estimation or the power a lot compared to using IVs from identical ancestry only in transMR. Left: bias from transMR, IVW and transMR-global IV estimators under different levels of heterogeneity in SNP effects in exposure, which is controlled by N ; Right: Power of transMR, IVW and transMR-global IV when the true causal effect = 0.1 with 10k samples in each cohort, and for senarios (A) the heterogeity in SNP effects is small; (B) all the IVs are weak. Acknowledgement MESA-TOPMed/MESA Study Acknowledgement MESA and the MESA SHARe project are conducted and supported by the National Heart, Lung, and Blood Institute (NHLBI) in collaboration with MESA investigators. Support for MESA is provided by contracts HHSN268201500003I, N01-HC-95159, N01-HC-95160, N01-HC-95161, N01-HC-95162, N01-HC- 95163, N01-HC-95164, N01-HC-95165, N01-HC-95166, N01-HC-95167, N01-HC-95168, N01-HC-95169, UL1-TR-000040, UL1-TR-001079, UL1-TR-001420. MESA Family is conducted and supported by the National Heart, Lung, and Blood Institute (NHLBI) in collaboration with MESA investigators. Support is provided by grants and contracts R01HL071051, R01HL071205, R01HL071250, R01HL071251, R01HL071258, R01HL071259, and by the National Center for Research Resources, Grant UL1RR033176. This study was supported in part by the National Center for Advancing Translational Sciences, CTSI grant UL1TR001881, and the National Institute of Diabetes and Digestive and Kidney Disease Diabetes Research Center (DRC) grant DK063491 to the Southern California Diabetes Endocrinology Research Center. The TOPMed MESA Multi-Omics project was conducted by the University of Washington and LABioMed (HHSN2682015000031/HHSN26800004). Molecular data for the Trans-Omics in Precision Medicine (TOPMed) program was supported by the National Heart, Lung and Blood Institute (NHLBI). RNA-seq for the NHLBI TOPMed: Multi-Ethnic Study of Atherosclerosis (MESA)” (phs001416.v1.p1) was performed at the Northwest Genomics Center (HHSN268201600032I) and at the Broad Institute Genomics Platform (HHSN268201600034I). Core support including centralized genomic read mapping and genotype calling, along with variant quality metrics and filtering were provided by the TOPMed Informatics Research Center (3R01HL-117626-02S1; contract HHSN268201800002I). Core support including phenotype harmonization, data management, sample-identity QC, and general program coordination were provided by the TOPMed Data Coordinating Center (R01HL-120393; U01HL-120393; contract HHSN268201800001I). We gratefully acknowledge the studies and participants who provided biological samples and data for TOPMed. Whole genome sequencing (WGS) for the Trans-Omics in Precision Medicine (TOPMed) program was supported by the National Heart, Lung and Blood Institute (NHLBI). WGS for “NHLBI TOPMed: Multi- Ethnic Study of Atherosclerosis (MESA)” (phs001416.v3.p1) was performed at the Broad Institute of MIT and Harvard (3U54HG003067-13S1). Centralized read mapping and genotype calling, along with variant quality metrics and filtering were provided by the TOPMed Informatics Research Center (3R01HL- 117626-02S1). Phenotype harmonization, data management, sample-identity QC, and general study coordination, were provided by the TOPMed Data Coordinating Center (3R01HL-120393-02S1), and TOPMed MESA Multi-Omics (HHSN2682015000031/HSN26800004). The MESA projects are conducted and supported by the National Heart, Lung, and Blood Institute (NHLBI) in collaboration with MESA investigators. Support for the Multi-Ethnic Study of Atherosclerosis (MESA) projects are conducted and supported by the National Heart, Lung, and Blood Institute (NHLBI) in collaboration with MESA investigators. Support for MESA is provided by contracts 75N92020D00001, HHSN268201500003I, N01- HC-95159, 75N92020D00005, N01-HC-95160, 75N92020D00002, N01-HC-95161, 75N92020D00003, N01- HC-95162, 75N92020D00006, N01-HC-95163, 75N92020D00004, N01-HC-95164, 75N92020D00007, N01- HC-95165, N01-HC-95166, N01-HC-95167, N01-HC-95168, N01-HC-95169, UL1-TR-000040, UL1-TR- 001079, UL1-TR-001420, UL1TR001881, DK063491, and R01HL105756. The authors thank the other investigators, the staff, and the participants of the MESA study for their valuable contributions. A full list of participating MESA investigators and institutes can be found at http://www.mesa-nhlbi.org Reference [1]. ↵ D. A. Lawlor , “ Commentary: Two-sample Mendelian randomization: opportunities and challenges ,” Int. J. Epidemiol ., vol. 45 , no. 3 , pp. 908 – 915 , Jun . 2016 , doi: 10.1093/ije/dyw127 . OpenUrl CrossRef PubMed [2]. ↵ R. Mägi et al. , “ Trans-ethnic meta-regression of genome-wide association studies accounting for ancestry increases power for discovery and improves fine-mapping resolution ,” Hum. Mol. Genet ., vol. 26 , no. 18 , pp. 3639 – 3650 , Jul . 2017 , doi: 10.1093/hmg/ddx280 . OpenUrl CrossRef PubMed [3]. ↵ T. Jiang , D. Gill , A. S. Butterworth , and S. Burgess , “ An empirical investigation into the impact of winner’s curse on estimates from Mendelian randomization ,” Int. J. Epidemiol ., vol. 52 , no. 4 , pp. 1209 – 1219 , Aug . 2023 , doi: 10.1093/ije/dyac233 . OpenUrl CrossRef PubMed [4]. ↵ A. Ghosh , F. Zou , and F. A. Wright , “ Estimating Odds Ratios in Genome Scans: An Approximate Conditional Likelihood Approach ,” Am. J. Hum. Genet ., vol. 82 , no. 5 , pp. 1064 – 1074 , May 2008 , doi: 10.1016/j.ajhg.2008.03.002 . OpenUrl CrossRef PubMed Web of Science [5]. ↵ Z. Zhu et al. , “Causal associations between risk factors and common diseases inferred from GWAS summary data,” Jul . 2017 , doi: 10.1101/168674 . OpenUrl Abstract / FREE Full Text [6]. ↵ J. Bowden , F. D. G. M C. Minelli , G. D. Smith , N. A. Sheehan , and J. R. Thompson , “ Assessing the suitability of summary data for two-sample Mendelian randomization analyses using MR-Egger regression: the role of the I2 statistic ,” Int. J. Epidemiol ., 2016 , doi: 10.1093/ije/dyw220 . OpenUrl CrossRef PubMed [7]. ↵ P. K. Gregersen and L. M. Olsson , “ Recent Advances in the Genetics of Autoimmune Disease ,” Annu. Rev. Immunol ., 2009 , doi: 10.1146/annurev.immunol.021908.132653 . OpenUrl CrossRef PubMed Web of Science [8]. ↵ D. E. Bild , “ Multi-Ethnic Study of Atherosclerosis: Objectives and Design ,” Am. J. Epidemiol ., vol. 156 , no. 9 , pp. 871 – 881 , Nov . 2002 , doi: 10.1093/aje/kwf113 . OpenUrl CrossRef PubMed Web of Science [9]. ↵ C. Khunsriraksakul et al. , “ Multi-ancestry and multi-trait genome-wide association meta-analyses inform clinical risk prediction for systemic lupus erythematosus ,” Nat. Commun ., vol. 14 , no. 1 , Feb . 2023 , doi: 10.1038/s41467-023-36306-5 . OpenUrl CrossRef [10]. ↵ A. J. Grant and S. Burgess , “ Pleiotropy robust methods for multivariable Mendelian randomization ,” Stat. Med ., 2021 , doi: 10.1002/sim.9156 . OpenUrl CrossRef PubMed [11]. ↵ M. Simmonds and S. Gough , “ The HLA Region and Autoimmune Disease: Associations and Mechanisms of Action ,” Curr. Genomics , vol. 8 , no. 7 , pp. 453 – 465 , Nov . 2007 , doi: 10.2174/138920207783591690 . OpenUrl CrossRef PubMed Web of Science [12]. ↵ L. Wang , F. Wang , and M. E. Gershwin , “ Human autoimmune diseases: a comprehensive update ,” J. Intern. Med ., vol. 278 , no. 4 , pp. 369 – 395 , Oct . 2015 , doi: 10.1111/joim.12395 . OpenUrl CrossRef PubMed [13]. ↵ H. Saito et al. , “ Association between KIR-HLA combination and ulcerative colitis and Crohn’s disease in a Japanese population ,” PLOS ONE , vol. 13 , no. 4 , p. e0195778 , Apr . 2018 , doi: 10.1371/journal.pone.0195778 . OpenUrl CrossRef PubMed [14]. ↵ W. K. Cho et al. , “ Association of MICA Alleles with Autoimmune Thyroid Disease in Korean Children ,” Int. J. Endocrinol ., vol. 2012 , pp. 1 – 7 , 2012 , doi: 10.1155/2012/235680 . OpenUrl CrossRef PubMed [15]. ↵ J. Wittner and W. Schuh , “ Krüppel-like Factor 2 (KLF2) in Immune Cell Migration ,” Vaccines , vol. 9 , no. 10 , p. 1171 , Oct . 2021 , doi: 10.3390/vaccines9101171 . OpenUrl CrossRef PubMed [16]. ↵ Q. Song et al. , “ Genome-wide association study on Northern Chinese identifies KLF2, DOT1L and STAB2 associated with systemic lupus erythematosus ,” Rheumatology , vol. 60 , no. 9 , pp. 4407 – 4417 , Sep . 2021 , doi: 10.1093/rheumatology/keab016 . OpenUrl CrossRef PubMed [17]. ↵ E. Davies , “ Mannose-binding protein gene polymorphism in South African systemic lupus erythematosus ,” Rheumatology , 1998 , doi: 10.1093/rheumatology/37.4.465 . OpenUrl CrossRef PubMed [18]. ↵ B. Rhodes and T. J. Vyse , “ The genetics of SLE: an update in the light of genome-wide association studies ,” Rheumatology , vol. 47 , no. 11 , pp. 1603 – 1611 , Aug . 2008 , doi: 10.1093/rheumatology/ken247 . OpenUrl CrossRef PubMed Web of Science [19]. ↵ H. Wu et al. , “ LncRNA Expression Profiles in Systemic Lupus Erythematosus and Rheumatoid Arthritis: Emerging Biomarkers and Therapeutic Targets ,” Front. Immunol ., vol. 12 , p. 792884 , Dec . 2021 , doi: 10.3389/fimmu.2021.792884 . OpenUrl CrossRef [20]. ↵ W. J. M. Janssen et al. , “ Proline-serine-threonine phosphatase interacting protein 1 (PSTPIP1) controls immune synapse stability in human T cells ,” J. Allergy Clin. Immunol ., vol. 142 , no. 6 , pp. 1947 – 1955 , Dec . 2018 , doi: 10.1016/j.jaci.2018.01.030 . OpenUrl CrossRef [21]. ↵ D. M. Santer , A. E. Wiedeman , T. H. Teal , P. Ghosh , and K. B. Elkon , “ Plasmacytoid Dendritic Cells and C1q Differentially Regulate Inflammatory Gene Induction by Lupus Immune Complexes ,” J. Immunol ., vol. 188 , no. 2 , pp. 902 – 915 , Jan . 2012 , doi: 10.4049/jimmunol.1102797 . OpenUrl Abstract / FREE Full Text [22]. ↵ J. S. Hui-Yuen et al. , “ Chromatin landscapes and genetic risk in systemic lupus ,” Arthritis Res. Ther ., vol. 18 , no. 1 , p. 281 , Dec . 2016 , doi: 10.1186/s13075-016-1169-9 . OpenUrl CrossRef PubMed [23]. ↵ Y. Kamatani et al. , “ Identification of a significant association of a single nucleotide polymorphism in TNXB with systemic lupus erythematosus in a Japanese population ,” J. Hum. Genet ., 2007 , doi: 10.1007/s10038-007-0219-1 . OpenUrl CrossRef PubMed Web of Science [24]. ↵ X. Wang et al. , “ Association between HLA-B and HLA-DRB1 polymorphisms and systemic lupus erythematosus in Han population in China ,” Rheumatol. Autoimmun ., vol. 2 , no. 1 , pp. 31 – 39 , Mar . 2022 , doi: 10.1002/rai2.12023 . OpenUrl CrossRef [25]. ↵ K. Shimane et al. , “ An association analysis of HLA-DRB1 with systemic lupus erythematosus and rheumatoid arthritis in a Japanese population: effects of *09:01 allele on disease phenotypes ,” Rheumatology , vol. 52 , no. 7 , pp. 1172 – 1182 , Jul . 2013 , doi: 10.1093/rheumatology/kes427 . OpenUrl CrossRef PubMed [26]. ↵ H. J.-L, S. C.-K, L. A, L. T.-D , W. Chou , and K. M.-L, “ HLA-DRB1 antigens in Taiwanese patients with juvenile-onset systemic lupus erythematosus ,” Rheumatol. Int ., 2001 , doi: 10.1007/s00296-001-0139-x . OpenUrl CrossRef PubMed Web of Science [27]. ↵ X. Yin et al. , “ Biological insights into systemic lupus erythematosus through an immune cell-specific transcriptome-wide association study ,” Ann. Rheum. Dis ., vol. 81 , no. 9 , pp. 1273 – 1280 , 2022 , doi: 10.1136/annrheumdis-2022-222345 . OpenUrl Abstract / FREE Full Text [28]. ↵ C. H. Siegel and L. R. Sammaritano , “ Systemic Lupus Erythematosus: A Review ,” JAMA , vol. 331 , no. 17 , p. 1480 , May 2024 , doi: 10.1001/jama.2024.2315 . OpenUrl CrossRef PubMed [29]. ↵ M. Verbanck , C.-Y. Chen , B. Neale , and R. Do , “ Detection of widespread horizontal pleiotropy in causal relationships inferred from Mendelian randomization between complex traits and diseases ,” Nat. Genet ., 2018 , doi: 10.1038/s41588-018-0099-7 . OpenUrl CrossRef PubMed [30]. ↵ Z. Yuan , L. Liu , P. Guo , R. Yan , F. Xue , and X. Zhou , “ Likelihood-based Mendelian randomization analysis with automated instrument selection and horizontal pleiotropic modeling ,” Sci. Adv ., 2022 , doi: 10.1126/sciadv.abl5744 . OpenUrl CrossRef [31]. ↵ S. Burgess and S. G. Thompson , Mendelian Randomization . CRC Press , 2021 . [32]. ↵ S. Burgess , A. Butterworth , and S. G. Thompson , “ Mendelian Randomization Analysis With Multiple Genetic Variants Using Summarized Data ,” Genet. Epidemiol ., vol. 37 , no. 7 , pp. 658 – 665 , Sep . 2013 , doi: 10.1002/gepi.21758 . OpenUrl CrossRef PubMed [33]. ↵ Z. Zhu et al. , “ Integration of summary data from GWAS and eQTL studies predicts complex trait gene targets ,” Nat. Genet ., vol. 48 , no. 5 , pp. 481 – 487 , Mar . 2016 , doi: 10.1038/ng.3538 . OpenUrl CrossRef PubMed [34]. ↵ E. Sanderson , “ Multivariable Mendelian Randomization and Mediation ,” Cold Spring Harb. Perspect. Med ., 2020 , doi: 10.1101/cshperspect.a038984 . OpenUrl Abstract / FREE Full Text [35]. ↵ Y. J. Sung et al. , “ A Large-Scale Multi-ancestry Genome-wide Study Accounting for Smoking Behavior Identifies Multiple Significant Loci for Blood Pressure ,” Am. J. Hum. Genet ., vol. 102 , no. 3 , pp. 375 – 400 , 2018 , doi: 10.1016/j.ajhg.2018.01.015 . OpenUrl CrossRef PubMed [36]. ↵ S. M. Gogarten et al. , “ Genetic association testing using the GENESIS R/Bioconductor package ,” Bioinformatics , 2019 , doi: 10.1093/bioinformatics/btz567 . OpenUrl CrossRef PubMed [37]. ↵ D. S. Araujo et al. , “ Multivariate adaptive shrinkage improves cross-population transcriptome prediction and association studies in underrepresented populations ,” Hum. Genet. Genomics Adv ., 2023 , doi: 10.1016/j.xhgg.2023.100216 . OpenUrl CrossRef PubMed [38]. ↵ A. A. Shabalin , “ Matrix eQTL: ultra fast eQTL analysis via large matrix operations ,” Bioinformatics , 2012 , doi: 10.1093/bioinformatics/bts163 . OpenUrl CrossRef PubMed Web of Science [39]. ↵ Y. Liu and J. Xie , “ Cauchy Combination Test: A Powerful Test With Analytic p-Value Calculation Under Arbitrary Dependency Structures ,” J. Am. Stat. Assoc ., vol. 115 , no. 529 , pp. 393 – 402 , Apr . 2019 , doi: 10.1080/01621459.2018.1554485 . OpenUrl CrossRef PubMed View the discussion thread. Back to top Previous Next Posted April 23, 2025. Download PDF Email Thank you for your interest in spreading the word about bioRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following Trans-ancestry Mendelian Randomization Discovers Novel Causal Genes for Autoimmune Disease Traits Message Subject (Your Name) has forwarded a page to you from bioRxiv Message Body (Your Name) thought you would like to see this page from the bioRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share Trans-ancestry Mendelian Randomization Discovers Novel Causal Genes for Autoimmune Disease Traits Siyuan Chen , Chen Wang , Kent D. Taylor , Jerome I. Rotter , Xiuqing Guo , Stephen S. Rich , Ani Manichaikul , Dajiang Liu bioRxiv 2025.04.17.649410; doi: https://doi.org/10.1101/2025.04.17.649410 Share This Article: Copy Citation Tools Trans-ancestry Mendelian Randomization Discovers Novel Causal Genes for Autoimmune Disease Traits Siyuan Chen , Chen Wang , Kent D. Taylor , Jerome I. Rotter , Xiuqing Guo , Stephen S. Rich , Ani Manichaikul , Dajiang Liu bioRxiv 2025.04.17.649410; doi: https://doi.org/10.1101/2025.04.17.649410 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Genetics Subject Areas All Articles Animal Behavior and Cognition (7629) Biochemistry (17660) Bioengineering (13881) Bioinformatics (41909) Biophysics (21436) Cancer Biology (18576) Cell Biology (25479) Clinical Trials (138) Developmental Biology (13367) Ecology (19887) Epidemiology (2067) Evolutionary Biology (24302) Genetics (15598) Genomics (22482) Immunology (17726) Microbiology (40359) Molecular Biology (17162) Neuroscience (88532) Paleontology (666) Pathology (2830) Pharmacology and Toxicology (4821) Physiology (7636) Plant Biology (15129) Scientific Communication and Education (2044) Synthetic Biology (4290) Systems Biology (9817) Zoology (2269)
Text is read by the "Ask this paper" AI Q&A widget below.
Extraction quality varies by source — PMC NXML preserves structure
cleanly, OA-HTML may include some navigation residue, and OA-PDF can
have broken hyphenation. The publisher copy
(via DOI)
is the canonical version.