Full text
41,925 characters
· extracted from
preprint-html
· click to expand
Impact of control selection strategies on GWAS results: a study of prostate cancer in the UK Biobank | medRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-P4HH5NV'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search Impact of control selection strategies on GWAS results: a study of prostate cancer in the UK Biobank View ORCID Profile Jingzhan Lu , View ORCID Profile Johan H. Thygesen , View ORCID Profile Robin N. Beaumont , View ORCID Profile Michael N. Weedon , View ORCID Profile Harry D. Green doi: https://doi.org/10.1101/2025.10.08.25337574 Jingzhan Lu 1 Department of Clinical and Biomedical Sciences, University of Exeter , EX1 2LU, Exeter, UK 2 Institute of Health Informatics, University College London , NW1 2DA, London, UK Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Jingzhan Lu Johan H. Thygesen 2 Institute of Health Informatics, University College London , NW1 2DA, London, UK Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Johan H. Thygesen Robin N. Beaumont 1 Department of Clinical and Biomedical Sciences, University of Exeter , EX1 2LU, Exeter, UK Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Robin N. Beaumont Michael N. Weedon 1 Department of Clinical and Biomedical Sciences, University of Exeter , EX1 2LU, Exeter, UK Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Michael N. Weedon Harry D. Green 1 Department of Clinical and Biomedical Sciences, University of Exeter , EX1 2LU, Exeter, UK Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Harry D. Green For correspondence: H.D.Green{at}exeter.ac.uk Abstract Full Text Info/History Metrics Supplementary material Data/Code Preview PDF Abstract As GWAS studies move from array-based genotyping to whole exome and genome sequencing, there is a significant increase in cost. Applying an appropriate technique for the selection of which controls to include, in large studies where more potential controls are available than needed for the study, may be a useful technique for minimising resource intensity while maintaining statistical power. We evaluated three control selection strategies in prostate cancer GWAS using 15,250 UK Biobank cases: (a) all controls, (b) matched controls, and (c) random selection. Both (b) and (c) achieved comparable power in detecting significant loci relative to (a), but matched controls (b) showed greater consistency in identifying leading SNPs. However, using (b) matched controls reduced discovery power by ∼30% compared with (a) all controls, highlighting a trade-off. Matching controls (1:4 ratio) offers a cost-effective approach for targeted SNP analysis across phenotypes but may miss novel associations. Availability and Implementation R code for implementing matching and random control selection is provided and available on GitHub ( https://github.com/Jingzhan-Lu/GWAS-Control-Selection ). 1. Introduction 1.1 Genome-wide association studies Standard disease GWAS studies are typically case-control studies aimed to detect genetic variation, known as single nucleotide polymorphisms (SNPs), which associate with the disease of interest. Significant associations between SNP and disease can help understand aetiology and potentially contribute to risk stratification tools ( Wu et al., 2010 ). Also, within cancer research, GWAS has been used successfully to identify over 200 genetic risk loci associated with susceptibility to prostate cancer to date ( Conti et al., 2021 ; Qian et al., 2023 ). 1.2 Control selection further complicated in large longitudinal study Chatterjee et al., (2016) emphasized that the validity of case-control studies relies heavily on the appropriate selection of controls. Proper control selection reduces the risk of selection bias and enhances the credibility of study findings. This issue is particularly salient in genome-wide association studies (GWAS) using biobank-scale datasets, where vast numbers of potential controls are available, making the strategy for selecting appropriate controls a critical study design consideration. As digital health infrastructure expands and increasing volumes of electronic health records (EHRs) and biobank-based data become accessible, the challenge of selecting appropriate controls becomes more complex. Prostate cancer is a highly heritable disease, with heritability estimates as high as 58% (95% CI, 51%–63%) according to twin studies ( Mucci et al., 2016 ). In this study, we will use prostate cancer as an example phenotype because there is a large genetic component to prostate cancer. In the UK Biobank, once prostate cancer cases are defined, virtually any other individual may qualify as a “healthy” control. Therefore, instead of including all eligible controls, we limited our selection to a maximum 1:4 ratio ( Hong & Park, 2012 ), aiming to balance power, reduce bias, and optimise computational efficiency. This study investigates how different strategies for control selection impact GWAS outcomes, with an emphasis on sufficient statistical power and cost-benefit of time and computational resources. In large-scale cohorts, it is common practice to include all available non-case individuals as controls in GWAS. This issue is especially critical in population-based studies where subtle ancestral differences may remain even after statistical adjustments, leading to both false-positive and false-negative associations ( Armstrong et al., 2014 ; Uffelmann et al., 2021 ). Failure to adequately address this issue remains a key limitation in using large biobank or EHR data where extensive inclusion of all controls without careful matching may compromise the accuracy of GWAS results. 1.3 Cost and scalability challenges in case–control GWAS Traditionally, GWAS studies have been performed on SNP array data imputed to ∼10 million variants, whole genome sequencing (WGS) and whole exome sequencing data (WES) are now more readily available in large cohort datasets such as the UK Biobank ( Bycroft et al., 2018 ) and All of Us (“The ‘All of Us’ Research Program,” 2019). While WGS data increases the potential to find novel rare variants, the increased scale of the data causes an increase in computational intensity. Most research in this field is now performed on a cloud-based research analysis platform (RAP) in which researchers pay for the computing resources used, so computational costs also incur a financial cost. More critically, when conducting new case–control studies, it is not sufficient to sequence only the cases; appropriately matched controls must also be identified and sequenced, which often represents one of the most expensive components of the study. For example, even within large biobank resources, identifying around 3,000 prostate cancer cases for whole-genome sequencing also requires sequencing an equivalent or larger number of unaffected male controls. Selecting an appropriate number of individuals and defining a suitable health control cohort can decrease sequencing costs and logistical complexity. Against this backdrop, case-control selection could potentially reduce cost while minimising power loss, especially for studies planning to run multiple GWAS across many phenotypes. 1.4 Matching Strategy for Control Selection Matching is a commonly employed strategy in case-control studies to ensure comparability between cases and controls by balancing important covariates, thereby minimising the influence of confounders on case-control study results. Several matching techniques exist, including nearest-neighbour matching, optimal matching, and ratio-based matching. In a comprehensive evaluation using Monte Carlo simulations, Austin (2014) compared 12 propensity score matching algorithms, such as optimal matching, greedy nearest-neighbour matching without replacement, and calliper matching, which reported that nearest-neighbour matching consistently demonstrated superior performance. Following these findings, this study, applied a propensity score–based nearest-neighbour matching to select controls with characteristics similar to cases. 2 Methods 2.1 UKBB genotyping quality control (QC) and DNAnexus The UK Biobank (UKBB) genotyping data includes 488,377 participants. All analyses were performed within the secure UKBB Research Analysis Platform (hosted on DNAnexus). The quality control (QC) procedures were based on the publicly released metrics and recommendations provided by the UKBB research team. Samples with a mismatch between self-reported sex and genetically inferred sex were excluded, as were individuals of non-European (non-CEU) ancestry. In addition, 968 outliers based on heterozygosity or missing genotype rate were removed, and samples with sex chromosome aneuploidy were excluded. Variant-level quality control excluded genetic variants with an imputation INFO score <0.7. Minor allele frequencies (MAF) were recalculated within the prostate cancer phenotype subset, and variants with MAF < 1% were excluded. 2.2 UKBB phenotypes and covariates definition Prostate Cancer cases (N = 15,250) were identified in the UK Biobank using cancer registry records based on the ICD-10 code C61, any participant with this diagnostic code was considered a case. Female participants inferred by genetics were excluded from the analysis. A total of 208,128 male participants without a diagnosis of prostate cancer were available as potential controls. A total of 13 covariates (recruitment age, assessment centre, genotyping chip and first 10 principal components) were used in the GWAS analysis and matching process. 2.3 GWAS power calculations A sample size with sufficient statistical power is critical to the success of genetic association studies to detect causal genes of human complex diseases. We performed a power calculation for a case-control GWAS with 15,250 prostate cancer cases and 4 times the number of controls (1:4 case-control ratio). Power calculations were conducted using the developed Genetic Power Calculator framework on additive genetic models ( Green, 2025 ), the key parameters included varying minor allele frequencies (MAF) from rare to common variants, and odds ratios (OR) of 1.5, 2, and 2.5, reflecting small to moderate genetic effect sizes. This power calculation reflects the theoretical baseline under random control selection, against which the observed gains from random controls can be evaluated. 2.4 Three Control Selection strategies and GWAS analysis In this study, we conducted GWAS for prostate cancer within the UKBB using three different control selection strategies to evaluate how the different control definitions affect GWAS results. All controls, included all male participants from the whole population without prostate cancer (N = 208,128); Match controls, selected controls matched to cases based on 13 covariates; and Random controls, included a random selection of male controls. For both strategy match and random, a 1:4 case-to-control ratio was used (n = 61,000). For each strategy, prostate cancer GWAS was performed independently using REGENIE software. To assess the impact of control selection on GWAS results, we compared our findings against the largest-to-date trans-ancestry meta-analysis of prostate cancer ( Conti et al., 2021 ), which included 107,247 cases and 127,006 controls. We used the 269 genome-wide significant SNPs reported in Conti et al. (2021) as a benchmark, matching them to corresponding SNPs in our GWAS results to evaluate the consistency and effect size of our GWAS results. 2.5 Nearest Neighbour Matching and Random for Control Selection To minimize confounding and ensure balance in key covariates between prostate cancer cases and controls, we applied nearest-neighbour propensity score matching using the MatchIt R package (v4.5.2). The nearest-neighbour algorithm was employed without replacement to pair each case with four control individuals having the most similar estimated propensity scores. Matched between cases and controls was assessed using empirical quantile-quantile (eQQ) plots to evaluate covariate balance. Post-matching, matched individuals were retained for downstream GWAS. For Random, we randomly selected four controls for each case from the eligible control population. To ensure reproducibility, a random subset of 61,000 male participants was selected from the full pool of 208,128 eligible controls using a fixed random seed (set.seed = 42). 2.6 Locus Signal Analysis To identify independent association signals from GWAS summary statistics, we implemented a custom locus-filtering script written in Perl and LocusZoom ( Pruim et al., 2010 ). The script processes one or more GWAS result files and extracts lead variants based on statistical significance, variant quality, and genomic proximity. For variants passing the filtered criteria, the script then performs locus-level clustering using a distance-based approach. Specifically, variants within ±500 kb of each other on the same chromosome were grouped as a single locus, and the variant with the most significant P-value (i.e., highest – log10P) within each group was selected as the lead SNP. This procedure enables efficient identification of genome-wide significant loci while accounting for local linkage disequilibrium and variant quality. 3 Results We characterised the demographic and covariates of the GWAS cohorts to evaluate their comparability ( Table 1 ). The full set of cases (N = 15,250) was contrasted with the three control strategies: all controls (N = 223,378), matched controls (N = 76,250), and random controls (N = 76,250). View this table: View inline View popup Download powerpoint Table 1. Baseline characteristics of UK Biobank participants included in prostate cancer case and control cohorts under different GWAS control selection strategies. We compared the demographic and covariates characteristics of all cases with the three GWAS cohorts (All, Match, Random). Among them, the match control selection showed the closest similarity to the case group, especially in age distribution (e.g., 72.1% aged 60–69 in both groups), sequencing centre proportion, and genetic principal components. These results indicate that match control selection method best preserved covariate balance relative to the case group. 3.1 GWAS power calculation and cost analyiss To ensure sufficient statistical power for detecting genetic associations, we performed power calculations in Figure 1 . for a prostate cancer GWAS using 15,250 cases and a 1:4 case-control ratio. Under an additive model, the study reaches over 80% power to detect variants with an odds ratio (OR) of 1.5 when the minor allele frequency (MAF) exceeds 0.007. For variants with ORs of 2 or higher, the study achieves nearly 100% power even at lower MAFs (as low as 0.003). These results demonstrate that a 1:4 case-control ratio using random controls provides sufficient GWAS power and can be generalised to other phenotypes requiring large-scale genetic association studies. Download figure Open in new tab Figure 1. Power estimation for prostate cancer under a 1:4 case-control design. Three panels show additive model power curves for odds ratios (OR) of 1.5, 2, and 2.5. The x-axis represents minor allele frequency (MAF) and the y-axis shows statistical power. Using a case-control selection strategy reduced the total sample size from 200,000 to 60,000, theoretically offering a 70% reduction in computing resource usage, or 91% reduction for algorithms are of O(N^2) complexity. Sample size reduction and thus resources saved will depend heavily on the prevalence of the phenotype and case/control ratio used. This offers both a cost-saving opportunity and an environmental benefit from reduction in high performance computing usage. 3.2 Manhattan Plots of GWAS results In Figure 3 , the Manhattan plot revealed genome-wide significant loci peaks are consistent with previously reported prostate cancer susceptibility regions. These consistent patterns further support the validity and robustness of our GWAS findings. However, differences are noted in the upper tail of the p-value distribution, where the maximum observed –log10(p) values vary: 82.3* 10 -8 or the All-controls, 72.5 * 10 -8 for the Matching, and 66.1* 10 -8 for the Random. Download figure Open in new tab Download figure Open in new tab Figure 2. Manhattan plots of GWAS results under different control strategies. Overall signal patterns are consistent, while the top – log10(p) peaks differ at 82.3, 72.5, and 66.1. Download figure Open in new tab Figure 3. Venn diagram of significant loci and leading SNPs across GWAS from All, Match and Random cohorts 3.3 GWAS Significant and suggestive SNPs We compared the number and concordance of genome-wide significant (P < 5 × 10−8) and suggestive (P < 5 × 10−5) SNPs identified under three strategies: All, Match, and Random ( Table 2 ). The total number of tested SNPs varied slightly across strategies due to control number differences, with All-controls analysing over 13.5 million SNPs, compared to ∼10.9 million in Match and Random. View this table: View inline View popup Download powerpoint Table 2. GWAS results of significant and suggestive SNPs statistics across three control strategies. While All and Match yielded comparable capture rates for both significant and suggestive SNPs, Random exhibited a noticeably lower yield in both categories. Specifically, All identified 4,572 significant SNPs (3.4×10 -4 ) % and 11,329 suggestive SNPs (8.4×10 -4 ) %; Match identified 3,572 significant (3.3×10 -4 ) % and 9,233 suggestive (8.5×10 -4 ) % SNPs; and Random yielded 3,064 significant (2.8×10 -4 ) % and 8,503 suggestive (7.8×10 -4 ) % SNPs. 3.4 GWAS locus signal results The Table 3 displays genome-wide significant association signals (–log10P > genome-wide threshold) detected across three control selection strategies (All, Matched, Random). Shown are loci restricted to chromosomes 8, 10, 11, and 17, which had the most significant findings across the three analyses. Each row represents the lead SNP per locus with –log10P values and allelic information. View this table: View inline View popup Table 3. GWAS significant loci with leading SNPs and P values on chromosomes 8, 10, 11, and 17 The divergence of both Match and Random strategies from the All-controls analysis was notable, with the largest discrepancy observed for random. Among the 60 loci identified using All-controls, approximately 30% leading SNPs were not replicated in the Match group and more than 50% leading SNPs were absent from the Random group. Furthermore, many loci showed attenuated test statistics or inconsistent effect directions in match and especially in random. For instance, lead SNPs at 8q24.21 (chr8:127516190; P = 8.2×10−8 3 ) and 11q13.3 (chr11: 69236512; P = 5.4×10− 54 ) remained significant in both All and Match, but the corresponding signals weakened or dropped below significance in Random, suggesting reduced sensitivity under random selection. Notably, despite Match and Random detecting a similar number of significant loci, Random often identified different leading SNPs compared to All-controls, indicating frequent lead SNP swapping under random selection. A total of 60, 43, and 43 genome-wide significant loci were identified in GWAS using All, matched, and random, respectively. In Figure 3 , shows 38 loci were significant across all three analyses. The overlap between All and Match, as well as between All and Random, was 42 loci in each case, indicating that both matched and random control designs achieved comparable sensitivity in detecting associated regions. However, analysis of leading SNPs revealed greater consistency between All and Match (30 shared SNPs), compared to All and Random (24 shared SNPs), suggesting that the matched design provided more precise localization of association signals. The smaller overlap between Match and Random compared to other pairings (17 shared SNPs and 38 loci, all of which were also shared with All) highlights divergent signal prioritization, implying that random controls may often select alternate tag SNPs within the same loci. Among the unique leading SNPs to the match GWAS (n = 13), 11 (84.62%) were in high linkage disequilibrium (LD; r 2 > 0.8) with SNPs identified in all GWAS. In contrast, only 11 out of 19 (57.89%) unique leading SNPs from the random GWAS showed such high LD with the all-controls SNPs, indicating improved signal consistency under the matched design. The detailed information of loci and leading SNPs provided in the supplementary table. 3.5 GWAS results show diff strategies OR similar with Conti’s results The scatter plot of Figure 4 illustrates the correlation between β estimates from Conti et al., (2021 )’s study (x-axis) and the GWAS results under the “Matched controls” strategy (y-axis). Each point represents a genetic variant (MAF >1%). A linear regression line is overlaid, indicating strong concordance in effect size estimates between the two analysis. Similar linear relationships were observed when using All controls and Random selection strategies. Therefore, only one representative scatter plot (Matched controls vs. Conti) is shown. The adjusted R 2 of 0.82 for the three linear regression models suggests that approximately 82% of the variance in GWAS effect sizes can be explained by Conti’s results, reflecting high consistency and supporting the notion that around 82% of observed GWAS signals in the real world are reproducible across different control strategies. Download figure Open in new tab Figure 4. Concordance of 269 SNPs effect size estimates between Conti’s study and GWAS with Matched Controls. 4 Discussion 4.1 Principal Findings This study evaluated how control selection strategies influence GWAS outcomes. We compared three distinct approaches: All, Matched, and Random. While all strategies revealed broadly similar global association patterns, reflected in the overall alignment of Manhattan plots and identification of key prostate cancer loci, the depth and consistency of association signals varied notably across designs. A similar percentage of genome-wide significant SNPs were identified between all controls 3.4×10 -4 (0.034%) and matched controls 3.3×10 -4 (0.033%), while using random controls identified a significantly lower percentage of SNPs 2.8×10 -4 (0.028 %). While both Matching and Random selection resulted in a ∼30% reduction in genome-wide significant loci compared with All-controls, Matching offered greater consistency with All-controls in terms of leading SNP identification and direction of β estimation. 4.2 Interpretation Our findings demonstrate empirically the trade-off between computational intensity and statistical power. The increased computational requirements of modern GWAS studies (especially when WGS data is used) incurs an increased computational cost. In our study, the GWAS with All-controls was performed with a total sample size of 200,000 male UK Biobank participants, while Random and Matching was performed with 60,000. Optimal case-control ratios have long been studied in the context of genetic association studies. While increasing the number of controls can enhance statistical power, especially when case numbers are limited, the benefit of adding more controls diminishes beyond a certain point. Empirical and theoretical evaluations suggest that power gains plateau around a 1:4 to 1:5 case-control ratio, after which additional controls yield minimal improvement relative to cost and computational burden ( Lakens, 2022 ; Ritchie et al., 2015 ). In large-scale studies, controlling for unbalanced case-control ratios and accounting for relatedness among samples becomes computationally intensive. Moreover, the risk of population stratification and residual confounding tends to increase with larger sample sizes, as subtle structure in the data can lead to spurious associations ( Turner et al., 2011 ; Uffelmann et al., 2021 ). This underscores the importance of careful control selection. Matching cases and controls on important covariates such as age, ancestry, and study centre can effectively reduce confounding due to unaccounted population structure and environmentally driven allele frequency differences ( Yang et al., 2011 ; Zhou et al., 2020 ). The utility of case-control matching becomes even more evident in studies of rare diseases. For traits with low prevalence, the ratio of cases to available controls can be extremely imbalanced, often 1:1000 or higher. While using all available controls may seem statistically appealing, such extreme imbalance increases computational burden, especially in models testing rare variants or modest effect sizes ( Lee et al., 2014 ). 4.3 Strengths and Weaknesses The matching on case-control selection is a plug-in approach to GWAS, which is easy to implement and combine with any pipeline or code module. Demographic comparisons show that matched controls more closely resemble cases in age and genetic background, improving covariate balance and reducing confounding. This is especially beneficial in large-scale whole genome (WGS) or exome sequencing (WES) studies, where computational and financial costs increase with sample size. Although both the Match and Random groups are subsets of the overall All-controls group, they represent distinct sampling strategies. Consequently, the greatest divergence in association results is observed between the Match and Random designs. This divergence suggests that case-control matching more effectively controls for confounding, potentially yielding more reliable SNP-level associations compared to random selection. However, this improvement in internal validity comes at a cost. matching reduces the effective sample size by discarding unmatched controls, but it does not eliminate bias, which may lead to a loss of statistical power compared to using the full set of available controls. This study has several limitations. First, prostate cancer cases were defined using ICD-10 codes in the UK Biobank, which lack histopathological detail and staging information. This may lead to phenotype misclassification and reduce the precision of genetic association estimates. Second, analyses were limited to individuals of European ancestry, which improves internal validity but restricts the generalisability of findings to other populations. Third, the study focused exclusively on prostate cancer as a representative phenotype. Although this enabled detailed comparison of control selection strategies, future studies should include a wider range of traits, particularly rare diseases and phenotypes with different genetic architectures. 4.4 Conclusion From a practical perspective, we recommend case-control matching GWAS as an efficient strategy for rapid association testing across multiple phenotypes. This approach is particularly useful when the goal is to verify known loci or replicate associations in costly sequencing contexts, rather than to discover novel variants. Looking forward, we plan to extend these evaluations to WGS and WES data and to perform external validation in deeply phenotype cohorts such as the All of Us Research Program. This will allow us to assess generalisability across diverse ancestries and health systems. Data Availability Availability and Implementation: R code for implementing matching and random control selection is provided and available on GitHub. https://github.com/Jingzhan-Lu/GWAS-Control-Selection Acknowledgements This study was supported by the National Institute for Health and Care Research Exeter Biomedical Research Centre (NIHR Exeter BRC) and and the University of Exeter Medical School. This research has been conducted using data from UK Biobank ( http://www.ukbiobank.ac.uk/ ), a major biomedical database, and was made possible through access to the data and findings generated by the UK Biobank project number 103356. References ↵ Armstrong , D. L. , Zidovetzki , R. , Alarcón-Riquelme , M. E. , Tsao , B. P. , Criswell , L. A. , Kimberly , R. P. , Harley , J. B. , Sivils , K. L. , Vyse , T. J. , Gaffney , P. M. , Langefeld , C. D. , & Jacob , C. O. ( 2014 ). GWAS identifies novel SLE susceptibility genes and explains the association of the HLA region . Genes and Immunity , 15 ( 6 ), 347 – 354 . doi: 10.1038/GENE.2014.23;TECHMETA OpenUrl CrossRef PubMed ↵ Austin , P. C. ( 2014 ). A comparison of 12 algorithms for matching on the propensity score . Statistics in Medicine , 33 ( 6 ), 1057 – 1069 . doi: 10.1002/SIM.6004 OpenUrl CrossRef PubMed ↵ Bycroft , C. , Freeman , C. , Petkova , D. , Band , G. , Elliott , L. T. , Sharp , K. , Motyer , A. , Vukcevic , D. , Delaneau , O. , O’Connell , J. , Cortes , A. , Welsh , S. , Young , A. , Effingham , M. , McVean , G. , Leslie , S. , Allen , N. , Donnelly , P. , & Marchini , J. ( 2018 ). The UK Biobank resource with deep phenotyping and genomic data . Nature 2018 562:7726 , 562 ( 7726 ), 203 – 209 . doi: 10.1038/s41586-018-0579-z OpenUrl CrossRef PubMed ↵ Chatterjee , N. , Shi , J. , & García-Closas , M. ( 2016 ). Developing and evaluating polygenic risk prediction models for stratified disease prevention . Nature Reviews Genetics , 17 ( 7 ), 392 – 406 . doi: 10.1038/NRG.2016.27;SUBJMETA OpenUrl CrossRef PubMed ↵ Conti , D. V. , Darst , B. F. , Moss , L. C. , Saunders , E. J. , Sheng , X. , Chou , A. , Schumacher , F. R. , Olama, A. A. Al , Benlloch , S. , Dadaev , T. , Brook , M. N. , Sahimi , A. , Hoffmann , T. J. , Takahashi , A. , Matsuda , K. , Momozawa , Y. , Fujita , M. , Muir , K. , Lophatananon , A. , … Haiman , C. A. ( 2021 ). Trans-ancestry genome-wide association meta-analysis of prostate cancer identifies new susceptibility loci and informs genetic risk prediction . Nature Genetics , 53 ( 1 ), 65 – 75 . doi: 10.1038/S41588-020-00748-0 , OpenUrl CrossRef PubMed ↵ Green , H. ( 2025 ). Harry’s GWAS Power Calculator . Https://Pgv6nn-Harry-Green.Shinyapps.Io/Gwaspower/ . ↵ Hong , E. P. , & Park , J. W. ( 2012 ). Sample Size and Statistical Power Calculation in Genetic Association Studies . Genomics & Informatics , 10 ( 2 ), 117 – 122 . doi: 10.5808/GI.2012.10.2.117 OpenUrl CrossRef PubMed ↵ Lakens , D. ( 2022 ). Sample Size Justification . Collabra: Psychology , 8 ( 1 ). doi: 10.1525/COLLABRA.33267 OpenUrl CrossRef ↵ Lee , S. , Abecasis , G. R. , Boehnke , M. , & Lin , X. ( 2014 ). Rare-Variant Association Analysis: Study Designs and Statistical Tests . American Journal of Human Genetics , 95 ( 1 ), 5 . doi: 10.1016/J.AJHG.2014.06.009 OpenUrl CrossRef PubMed ↵ Mucci , L. A. , Hjelmborg , J. B. , Harris , J. R. , Czene , K. , Havelick , D. J. , Scheike , T. , Graff , R. E. , Holst , K. , Möller , S. , Unger , R. H. , McIntosh , C. , Nuttall , E. , Brandt , I. , Penney , K. L. , Hartman , M. , Kraft , P. , Parmigiani , G. , Christensen , K. , Koskenvuo , M. , … Halekoh , U. ( 2016 ). Familial Risk and Heritability of Cancer Among Twins in Nordic Countries . JAMA , 315 ( 1 ), 68 . doi: 10.1001/JAMA.2015.17703 OpenUrl CrossRef PubMed ↵ Pruim , R. J. , Welch , R. P. , Sanna , S. , Teslovich , T. M. , Chines , P. S. , Gliedt , T. P. , Boehnke , M. , Abecasis , G. R. , Willer , C. J. , & Frishman , D. ( 2010 ). LocusZoom: regional visualization of genome-wide association scan results . Bioinformatics , 26 ( 18 ), 2336 . doi: 10.1093/BIOINFORMATICS/BTQ419 OpenUrl CrossRef PubMed Web of Science ↵ Qian , Y. , Wang , J. , Wang , B. , Wang , W. , Li , P. , Zhao , Z. , Jiang , Y. , Ren , H. , Huang , D. , Yang , Y. , Zhao , Z. , Zhang , L. , Shi , J. , Li , M. J. , & Lu , W. ( 2023 ). Systematic fine-mapping and functional studies of prostate cancer risk variants . IScience , 26 ( 4 ), 106497 . doi: 10.1016/J.ISCI.2023.106497 OpenUrl CrossRef PubMed ↵ Ritchie , M. D. , Holzinger , E. R. , Li , R. , Pendergrass , S. A. , & Kim , D. ( 2015 ). Methods of integrating data to uncover genotype-phenotype interactions . Nature Reviews Genetics , 16 ( 2 ), 85 – 97 . doi: 10.1038/NRG3868;SUBJMETA OpenUrl CrossRef PubMed The “All of Us” Research Program . ( 2019 ). New England Journal of Medicine , 381 ( 7 ), 668 – 676 . doi: 10.1056/NEJMSR1809937/SUPPL_FILE/NEJMSR1809937_DISCLOSURES.PDF OpenUrl CrossRef PubMed ↵ Turner , S. , Armstrong , L. L. , Bradford , Y. , Carlsony , C. S. , Crawford , D. C. , Crenshaw , A. T. , de Andrade , M. , Doheny , K. F. , Haines , J. L. , Hayes , G. , Jarvik , G. , Jiang , L. , Kullo , I. J. , Li , R. , Ling , H. , Manolio , T. A. , Matsumoto , M. M. , McCarty , C. A. , McDavid , A. N. , … Ritchie , M. D. ( 2011 ). Quality Control Procedures for Genome Wide Association Studies. Current Protocols in Human Genetics / Editorial Board, Jonathan L . Haines … [et Al.], CHAPTER(SUPPL.68), Unit1.19 . doi: 10.1002/0471142905.HG0119S68 OpenUrl CrossRef ↵ Uffelmann , E. , Huang , Q. Q. , Munung , N. S. , de Vries , J. , Okada , Y. , Martin , A. R. , Martin , H. C. , Lappalainen , T. , & Posthuma , D. ( 2021 ). Genome-wide association studies . Nature Reviews Methods Primers , 1 ( 1 ), 1 – 21 . doi: 10.1038/S43586-021-00056-9;SUBJMETA OpenUrl CrossRef ↵ Wu , M. C. , Kraft , P. , Epstein , M. P. , Taylor , D. M. , Chanock , S. J. , Hunter , D. J. , & Lin , X. ( 2010 ). Powerful SNP-Set Analysis for Case-Control Genome-wide Association Studies . The American Journal of Human Genetics , 86 ( 6 ), 929 – 942 . doi: 10.1016/j.ajhg.2010.05.002 OpenUrl CrossRef PubMed Web of Science ↵ Yang , J. , Weedon , M. N. , Purcell , S. , Lettre , G. , Estrada , K. , Willer , C. J. , Smith , A. V. , Ingelsson , E. , O’Connell , J. R. , Mangino , M. , Mägi , R. , Madden , P. A. , Heath , A. C. , Nyholt , D. R. , Martin , N. G. , Montgomery , G. W. , Frayling , T. M. , Hirschhorn , J. N. , McCarthy , M. I. , … Visscher , P. M. ( 2011 ). Genomic inflation factors under polygenic inheritance . European Journal of Human Genetics , 19 ( 7 ), 807 – 812 . doi: 10.1038/EJHG.2011.39;SUBJMETA OpenUrl CrossRef PubMed ↵ Zhou , H. , Sealock , J. M. , Sanchez-Roige , S. , Clarke , T. K. , Levey , D. F. , Cheng , Z. , Li , B. , Polimanti , R. , Kember , R. L. , Smith , R. V. , Thygesen , J. H. , Morgan , M. Y. , Atkinson , S. R. , Thursz , M. R. , Nyegaard , M. , Mattheisen , M. , Børglum , A. D. , Johnson , E. C. , Justice , A. C. , … Gelernter , J. ( 2020 ). Genome-wide meta-analysis of problematic alcohol use in 435,563 individuals yields insights into biology and relationships with other traits . Nature Neuroscience , 23 ( 7 ), 809 – 818 . doi: 10.1038/S41593-020-0643-5;TECHMETA OpenUrl CrossRef PubMed View the discussion thread. Back to top Previous Next Posted October 09, 2025. Download PDF Supplementary Material Data/Code Email Thank you for your interest in spreading the word about medRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following Impact of control selection strategies on GWAS results: a study of prostate cancer in the UK Biobank Message Subject (Your Name) has forwarded a page to you from medRxiv Message Body (Your Name) thought you would like to see this page from the medRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share Impact of control selection strategies on GWAS results: a study of prostate cancer in the UK Biobank Jingzhan Lu , Johan H. Thygesen , Robin N. Beaumont , Michael N. Weedon , Harry D. Green medRxiv 2025.10.08.25337574; doi: https://doi.org/10.1101/2025.10.08.25337574 Share This Article: Copy Citation Tools Impact of control selection strategies on GWAS results: a study of prostate cancer in the UK Biobank Jingzhan Lu , Johan H. Thygesen , Robin N. Beaumont , Michael N. Weedon , Harry D. Green medRxiv 2025.10.08.25337574; doi: https://doi.org/10.1101/2025.10.08.25337574 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Genetic and Genomic Medicine Subject Areas All Articles Addiction Medicine (568) Allergy and Immunology (863) Anesthesia (300) Cardiovascular Medicine (4436) Dentistry and Oral Medicine (444) Dermatology (382) Emergency Medicine (608) Endocrinology (including Diabetes Mellitus and Metabolic Disease) (1509) Epidemiology (15229) Forensic Medicine (30) Gastroenterology (1124) Genetic and Genomic Medicine (6600) Geriatric Medicine (668) Health Economics (997) Health Informatics (4538) Health Policy (1368) Health Systems and Quality Improvement (1613) Hematology (542) HIV/AIDS (1264) Infectious Diseases (except HIV/AIDS) (15916) Intensive Care and Critical Care Medicine (1103) Medical Education (623) Medical Ethics (146) Nephrology (667) Neurology (6599) Nursing (346) Nutrition (998) Obstetrics and Gynecology (1144) Occupational and Environmental Health (957) Oncology (3333) Ophthalmology (974) Orthopedics (369) Otolaryngology (420) Pain Medicine (436) Palliative Medicine (130) Pathology (663) Pediatrics (1693) Pharmacology and Therapeutics (691) Primary Care Research (711) Psychiatry and Clinical Psychology (5447) Public and Global Health (9232) Radiology and Imaging (2198) Rehabilitation Medicine and Physical Therapy (1370) Respiratory Medicine (1196) Rheumatology (593) Sexual and Reproductive Health (712) Sports Medicine (530) Surgery (712) Toxicology (99) Transplantation (289) Urology (265) (function(){function c(){var b=a.contentDocument||a.contentWindow.document;if(b){var d=b.createElement('script');d.innerHTML="window.__CF$cv$params={r:'a01011c8e901c13d',t:'MTc3OTY2NTIzMA=='};var a=document.createElement('script');a.src='/cdn-cgi/challenge-platform/scripts/jsd/main.js';document.getElementsByTagName('head')[0].appendChild(a);";b.getElementsByTagName('head')[0].appendChild(d)}}if(document.body){var a=document.createElement('iframe');a.height=1;a.width=1;a.style.position='absolute';a.style.top=0;a.style.left=0;a.style.border='none';a.style.visibility='hidden';document.body.appendChild(a);if('loading'!==document.readyState)c();else if(window.addEventListener)document.addEventListener('DOMContentLoaded',c);else{var e=document.onreadystatechange||function(){};document.onreadystatechange=function(b){e(b);'loading'!==document.readyState&&(document.onreadystatechange=e,c())}}}})();
Text is read by the "Ask this paper" AI Q&A widget below.
Extraction quality varies by source — PMC NXML preserves structure
cleanly, OA-HTML may include some navigation residue, and OA-PDF can
have broken hyphenation. The publisher copy
(via DOI)
is the canonical version.