Integrative Harmonization of Phenotypic and Genomic Data Improves Bone Mineral Density Prediction in Multi-Study Osteoporosis Research

preprint OA: closed
📄 Open PDF Full text JSON View at publisher

Abstract

Purpose Harmonizing osteoporosis-related data across multiple datasets is essential for improving the accuracy and generalizability of bone mineral density (BMD) assessments. This study developed a harmonization framework to standardize phenotypic and genomic variables across three major U.S. osteoporosis datasets: GDBF, GWAS, and NHANES. Methods We standardized key phenotypic variables (BMD, body mass index (BMI), age, sex, and race/ethnicity) using cohort-specific data dictionaries and applied multiple imputations by chained equations (MICE) to manage missing data. Genomic data were harmonized using principal component analysis (PCA)-based batch effect corrections. Residual regression methods were applied to standardize BMD values. The effectiveness of harmonization on BMD prediction was evaluated using generalized estimating equations (GEE) and mixed-effects models. Results Post-harmonization, inter-study variability in BMI was significantly reduced (Ω² = 0.0028), and BMD associations with covariates remained consistent across datasets. Harmonized models showed improved predictive performance, with explained variance in BMD increasing (R² = 0.14). PCA confirmed the effective alignment of genetic data, reducing batch effects and improving cross-study compatibility. Conclusion This study demonstrates the feasibility and effectiveness of harmonizing phenotypic and genomic data for osteoporosis research. The harmonization framework enhances BMD prediction accuracy, supports more inclusive osteoporosis risk assessment, and improves the integration of multi-cohort datasets for future research. These findings highlight the potential of data harmonization in advancing precision medicine for osteoporosis prevention and management. Mini-Abstract Harmonizing osteoporosis datasets improves BMD prediction accuracy and enhances risk assessment. This study developed a harmonization framework integrating phenotypic and genomic data across three major datasets. After harmonization, predictive model performance improved, enabling better osteoporosis risk stratification and advancing precision medicine for fracture prevention.
Full text 44,508 characters · extracted from preprint-html · click to expand
Integrative Harmonization of Phenotypic and Genomic Data Improves Bone Mineral Density Prediction in Multi-Study Osteoporosis Research | medRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-P4HH5NV'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search Integrative Harmonization of Phenotypic and Genomic Data Improves Bone Mineral Density Prediction in Multi-Study Osteoporosis Research View ORCID Profile Anqi Liu , Jianing Liu , Lang Wu , View ORCID Profile Qing Wu doi: https://doi.org/10.1101/2025.05.12.25327471 Anqi Liu 1 Department of Biomedical Informatics, College of Medicine, The Ohio State University MS Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Anqi Liu Jianing Liu 1 Department of Biomedical Informatics, College of Medicine, The Ohio State University Ph. D Find this author on Google Scholar Find this author on PubMed Search for this author on this site Lang Wu 2 Pacific Center for Genome Research University of Hawai’i at Mānoa Honolulu HI USA 3 Population Sciences in the Pacific Program, University of Hawai’i Cancer Center, University of Hawai’i at Mānoa , Honolulu, HI, USA Ph. D Find this author on Google Scholar Find this author on PubMed Search for this author on this site Qing Wu 1 Department of Biomedical Informatics, College of Medicine, The Ohio State University MD Sc.D Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Qing Wu For correspondence: Qing.Wu{at}osumc.edu Abstract Full Text Info/History Metrics Supplementary material Preview PDF Abstract Purpose Harmonizing osteoporosis-related data across multiple datasets is essential for improving the accuracy and generalizability of bone mineral density (BMD) assessments. This study developed a harmonization framework to standardize phenotypic and genomic variables across three major U.S. osteoporosis datasets: GDBF, GWAS, and NHANES. Methods We standardized key phenotypic variables (BMD, body mass index (BMI), age, sex, and race/ethnicity) using cohort-specific data dictionaries and applied multiple imputations by chained equations (MICE) to manage missing data. Genomic data were harmonized using principal component analysis (PCA)-based batch effect corrections. Residual regression methods were applied to standardize BMD values. The effectiveness of harmonization on BMD prediction was evaluated using generalized estimating equations (GEE) and mixed-effects models. Results Post-harmonization, inter-study variability in BMI was significantly reduced (Ω² = 0.0028), and BMD associations with covariates remained consistent across datasets. Harmonized models showed improved predictive performance, with explained variance in BMD increasing (R² = 0.14). PCA confirmed the effective alignment of genetic data, reducing batch effects and improving cross-study compatibility. Conclusion This study demonstrates the feasibility and effectiveness of harmonizing phenotypic and genomic data for osteoporosis research. The harmonization framework enhances BMD prediction accuracy, supports more inclusive osteoporosis risk assessment, and improves the integration of multi-cohort datasets for future research. These findings highlight the potential of data harmonization in advancing precision medicine for osteoporosis prevention and management. Mini-Abstract Harmonizing osteoporosis datasets improves BMD prediction accuracy and enhances risk assessment. This study developed a harmonization framework integrating phenotypic and genomic data across three major datasets. After harmonization, predictive model performance improved, enabling better osteoporosis risk stratification and advancing precision medicine for fracture prevention. Introduction Osteoporosis, a chronic condition characterized by reduced bone mineral density (BMD) and increased fracture risk, affects approximately 18.3% of the global population, with 1 in 3 women and 1 in 5 men over 50 predicted to experience an osteoporotic fracture[ 1 , 2 ]. Despite its prevalence and significant burden on public health, osteoporosis often remains underdiagnosed and undertreated, contributing to avoidable morbidity and mortality globally [ 3 , 4 ]. Genetic studies, particularly genome-wide association studies (GWAS), have identified numerous single-nucleotide polymorphisms (SNPs) associated with BMD and fracture risk, shedding light on the genetic basis of osteoporosis[ 5 – 7 ]. However, existing research is limited by challenges in harmonizing genomic and phenotypic data across diverse studies[ 8 , 9 ]. Variability in data collection methods, demographic representation, and the lack of standardized protocols for defining key variables such as BMD and body mass index (BMI) impede the ability to perform robust, large-scale analyses[ 10 ]. For example, inconsistencies in SNP imputation, differences in measurement techniques, and underrepresentation of minority populations have limited the generalizability of findings to non-European ancestry groups[ 11 – 14 ]. Harmonizing phenotypic and genomic data is critical to overcoming these barriers and advancing osteoporosis research. By aligning variable definitions, standardizing measurement units, and addressing demographic imbalances, harmonized datasets enable more comprehensive analyses of gene-environment interactions, improve statistical power, and foster equitable research outcomes.[ 15 – 18 ] However, Previous harmonization efforts often lacked comprehensive approaches to simultaneously standardizing both phenotypic and genomic variables, limiting their generalizability and practical applicability. Significant obstacles remain, including biases introduced by missing data, measurement instrument incompatibility, and varying inclusion criteria across datasets.[ 19 , 20 ]. This study aimed to address these challenges by implementing a harmonization framework across multiple major U.S. osteoporosis datasets. The framework focuses on: Standardizing phenotypic variables such as BMD, BMI, and demographic factors across datasets. Applying robust statistical techniques, including mixed-effects models and generalized estimating equations (GEE), to account for inter-study variability. Harmonizing genomic data through stringent quality control and imputation protocols to enable integration across datasets. By providing a harmonized dataset that integrates both phenotypic and genomic data, we aim to establish a foundation for harmonized large-scale investigations into the phenotypic and genetic determinants of osteoporosis. The insights derived from this harmonization framework have the potential to enhance the understanding of BMD variation across diverse populations, advancing the field toward more inclusive and effective strategies for osteoporosis prevention and management. Methods Section 2.1 Study Design and Data Sources This study harmonized data from three major U.S. osteoporosis studies/datasets: the Genetic Determinants of Bone Fragility (GDBF) study, the Genome-Wide Association Study (GWAS) for Osteoporosis, and the National Health and Nutrition Examination Survey (NHANES). These datasets were selected because they include key bone health measures, such as BMD, and possess comprehensive phenotypic variables and genomic data for osteoporosis research, enabling a multi-dataset evaluation of phenotype harmonization. The GDBF study is a cross-sectional investigation assessing genetic factors influencing peak BMD in premenopausal women[ 21 ]. The GWAS study is a longitudinal cohort designed to explore genome-wide associations with osteoporosis, emphasizing genetic markers relevant to bone health[ 22 ]. NHANES is a nationally representative, cross-sectional survey conducted by the National Center for Health Statistics, providing data on BMD and other health parameters[ 23 ]. This study specifically utilized data from the 2009–2010, 2013–2014, and 2017– 2020 cycles[ 24 ]. The overall harmonization process, including both phenotypic and genomic data, is summarized in Figure 1 , which illustrates the multi-step approach used in this study, including multi-study phenotype harmonization and BMD-specific harmonization. Download figure Open in new tab Figure 1. Harmonization Pipeline for Phenotypic and Genomic Data. Phenotypic harmonization involved data cleaning (handling missing values and outlier detection), followed by standardization and statistical analyses (Kruskal-Wallis test and residual regression) to ensure consistency across cohorts. Genomic harmonization of GWAS and GDBF datasets included rigorous quality control, imputation, and batch-effect correction, facilitating accurate integration and analysis. Data were obtained from both controlled-access and publicly available repositories. The GDBF and GWAS datasets were accessed through the Database of Genotypes and Phenotypes (dbGaP). (dbGaP Study Accession: phs000390.v1.p1, phs000138.v2.p1). The NHANES was obtained from the National Center for Health Statistics (NCHS). Ethical approval was obtained from the relevant institutional review board (IRB #2025E0335). Phenotypic Harmonization In this study, bone mineral density (BMD) specifically refers to total hip BMD measured by dual-energy X-ray absorptiometry (DXA) and expressed in grams per square centimeter (g/cm²). BMD values in the GDBF cohort were pre-adjusted in dbGaP using residuals derived from linear regression models adjusting for baseline age and weight. To align methodologically with the GDBF adjustments, we applied the same residual regression approach separately within the GWAS cohort, replacing original BMD values with residuals computed from models adjusting for baseline age and weight. NHANES lacked hip BMD measurements and were therefore included solely for demographic comparisons. Other key phenotypic variables—including BMI, age, sex, and race/ethnicity—were standardized using a structured harmonization framework. Variable definitions were consistently aligned across datasets through cohort-specific data dictionaries. BMI values were uniformly recalculated from height and weight measurements. Additionally, racial and ethnic categories were consolidated into four standardized groups: Black, Hispanic, White, and Other. Handling of Missing Data and Imputation Multiple imputations by chained equations (MICE) were applied separately within each cohort, assuming missing-at-random (MAR)[ 25 , 26 ]. The imputation model included age, sex, BMI, race, and BMD to ensure unbiased estimates. Convergence diagnostics were performed, and post-imputation validation confirmed that the imputed values were closely aligned with observed data distributions. 2.3. Genomic Harmonization Genomic data from GWAS and GDBF studies underwent rigorous quality control and harmonization. Genotyping and Imputation Genotyping was conducted using Illumina and Affymetrix arrays[ 27 ], followed by genotype imputation via the Haplotype Reference Consortium (HRC) panel using the Michigan Imputation Server[ 28 ]. Quality control measures included the removal of SNPs with a minor allele frequency (MAF) <0.01, call rate <99%, or Hardy-Weinberg equilibrium (HWE) p-value < 1e-5. Population Structure and Batch Effect Correction Principal component analysis (PCA) was used to evaluate population structure and detect batch effects[ 29 , 30 ]. Scatter plots of PC1 vs. PC2 confirmed genetic similarity between the GWAS and GDBF datasets. Study-specific genomic batch effects were assessed and corrected, ensuring consistency across datasets. Quality control, PCA, and batch-effect corrections were performed using PLINK 2.0 and R statistical software (version 4.3.2), ensuring transparency and reproducibility. 2.4 Statistical Analysis A combination of statistical approaches was applied to evaluate BMD associations and harmonization effectiveness. Harmonization Effectiveness Assessment Harmonization effectiveness was assessed by comparing the pre- and post-harmonization distributions of key phenotypic variables using Analysis of Variance (ANOVA) and Kruskal-Wallis tests. Residual regression analyses were performed to reduce inter-study variability, adjusting for age, sex, and race. Effect sizes (Omega squared, Ω²) were calculated to quantify the proportion of variance explained by study differences after harmonization[ 31 ]. BMD Association Analysis BMD was modeled as a function of age, sex, BMI, and race using Generalized Estimating Equations (GEE)[ 32 ] for harmonized data, which accounted for the within-study correlation. Mixed-effects models were implemented[ 33 ], incorporating the study cohort as a random effect to capture inter-data variability. Additionally, separate linear regression models were applied to pre-harmonized GWAS and GDBF datasets to compare association estimates before and after harmonization. Model Fit and Validation Model performance was evaluated by comparing estimates across different statistical approaches. Adjusted R² (for linear regression), Pseudo R² (for mixed-effects models), and Quasi R² (for generalized estimating equations) were used to assess predictive accuracy, providing insights into model selection and comparative effectiveness. 2.5 Data Availability The harmonized datasets and statistical analysis code are available upon reasonable request, subject to data use agreements (Accession: phs000138.v2. p1, phs000390.v1. p1). NHANES data used in this study are publicly available from the National Center for Health Statistics (NCHS) website ( https://www.cdc.gov/nchs/nhanes/index.html ). Results 3.1 Dataset Characteristics The final harmonized dataset included 5,663 participants from three major datasets: GDBF (n=1,493), GWAS (n=1,617), and NHANES (n=2,553). Table 1 presents the baseline characteristics before and after harmonization. Before harmonization, significant differences were observed across the datasets in age, BMI, sex, and racial composition (ANOVA, p <0.01). The mean BMI varied across datasets, with GDBF participants having a lower mean BMI (26.2 kg/m²) compared with NHANES (28.2 kg/m²) and GWAS (26.8 kg/m²). Similarly, the racial composition was notably different, with NHANES containing a more diverse population (39.5% White, 29.7% Hispanic, 20.6% Black, and 10.1% Other), whereas GWAS and GDBF were predominantly White (100% and 99.6%, respectively). The GDBF data consisted of younger participants (mean age: 32.7 years), whereas GWAS and NHANES participants had higher mean ages of 48.1 and 48.8 years, respectively. View this table: View inline View popup Table 1. Participant Characteristics of the NHANES, GWAS, GDBF datasets and post-harmonization dataset 3.2 Missing Data Patterns and Imputation Outcomes The extent of missing data varies across datasets, with GDBF exhibiting the highest proportion of missing BMD and BMI values. We found that up to 25.5% of BMD variables and 0.67% of BMI variables were missing in the GDBF study. In the NHANES study, 0.27% of BMI variables were missing. The GWAS study exhibited relatively complete phenotypic data, although additional quality control was required for some genomic markers due to low call rates. Multiple imputations via the MICE approach were applied separately within each dataset. Post-imputation diagnostics confirmed that the imputed values closely matched the observed data distributions. Density plots comparing pre- and post-imputation distributions validated the imputation model, ensuring bias introduced by missing data to be minimized ( SFigure 2 ). Download figure Open in new tab Figure 2. Comparison of Pre- and Post-Harmonization Distributions of BMI Across Datasets. (A) GDBF study, (B) GWAS study, and (C) NHANES cohort before harmonization; (D) Harmonized data across all cohorts, illustrating improved alignment. 3.3 Assessment of Harmonization Effectiveness The differences in BMI across the GDBF, NHANES, and GWAS datasets were statistically significant, as indicated by the Kruskal-Wallis test ( p <0.05 η² = 0.032), highlighting notable heterogeneities across study populations. Figure 2 compares the pre- and post-harmonization distributions of BMI across the three datasets. Before harmonization, the BMI distribution showed notable differences, with the GDBF cohort having a more skewed distribution towards lower BMI values compared with the NHANES and GWAS datasets. Post-harmonization, the BMI distributions across the three datasets became more aligned, reflecting the effectiveness of the harmonization process. Post-harmonization adjustments for age, sex, and race significantly reduced inter-cohort variability, particularly in BMI distributions, as illustrated in Figure 3 . The residual regression analyses confirmed that BMI differences across datasets decreased after harmonization. The calculated Omega squared (Ω² = 0.0028) indicated a meaningful reduction in the variance attributable to study differences, further supporting the success of the harmonization process. The overall BMI distribution became more aligned across datasets, demonstrating the efficacy of the harmonization process. The final datasets were well-balanced, enabling robust downstream analyses. The racial composition across participants after harmonization demonstrates a more balanced distribution across racial groups ( SFigure 1 ). Download figure Open in new tab Figure 3. BMI Distributions Before and After Harmonization Adjustments. A) Boxplots illustrating BMI values across datasets before adjustment. (B) Boxplots of BMI residuals after adjustment for age, sex, and race. (C) Density plots of BMI values before adjustment, stratified by dataset. (D) Density plots of BMI residuals after adjustment for age, sex, and race, showing improved alignment across datasets. STable 1 provides further evidence of the success of the harmonization process, highlighting effect sizes and 95% confidence intervals (CI) for age and BMI across the GWAS, GDBF, and NHANES datasets before and after phenotype harmonization. Post-harmonization values were more consistent across datasets, demonstrating that cohort differences were minimized while retaining essential phenotypic information. 3.4 BMD Association Analysis and Model Performance The associations between BMD and key covariates were evaluated using Generalized Estimating Equations (GEE) and Mixed-Effects Models. As detailed in STable 2 , race and sex remained significant predictors for the GEE model. After harmonization, the adjusted R² values increased from −0.07% (GDBF) and 0.002% (GWAS) to 0.14% (harmonized dataset), reflecting improved model performance and reduced inter-study variability ( Figure 4 ). Although this increase was relatively modest, it is important for enabling robust comparative analyses and future pooled studies. The consistency of model estimates across different statistical approaches further confirms that the harmonization process effectively preserved essential phenotype-genotype relationships without introducing bias. Download figure Open in new tab Figure 4. Comparison of Model Estimates and Residual Distributions for BMD Prediction. (A) Coefficient estimates with 95% confidence intervals for generalized estimating equations (GEE, red) and mixed-effects models (blue), assessing predictors (sex, race, BMI, and age) of BMD. (B) Density plots of residuals from pre- and post-harmonization models, showing consistent variance explained (R² values) by harmonized GEE and mixed-effects models across datasets 3.5 Population Structure and Genomic Harmonization Outcomes Principal component analysis (PCA) was conducted to assess genomic harmonization between the GWAS and GDBF datasets. These analyses confirmed that no systematic biases remained after harmonization. The scatter plot of PC1 and PC2 ( Figure 5 ) illustrates substantial overlaps between the two datasets, confirming that the genomic harmonization process effectively reduced batch effects. No significant clustering by cohort was observed, suggesting that population stratification was well-controlled in subsequent analyses. Download figure Open in new tab Figure 5. PCA Analysis of Genomic Harmonization. Scatter plot of the first two principal components (PC1: 26.18%; PC2: 19.65%), illustrating substantial overlaps between GDBF and GWAS datasets, indicating effective reduction of batch effects and successful genomic harmonization. Discussion This study implemented a robust framework for harmonizing phenotypic and genomic data across three major U.S. osteoporosis datasets: GDBF, GWAS, and NHANES. The harmonized dataset addressed key challenges, including variability in phenotypic definitions, demographic imbalances, and missing data, enabling comprehensive analyses of BMD and its determinants[ 34 ]. Key findings include the successful harmonization of BMI, age, and race, reduction of inter-study variability, and validation of known associations between BMD and predictors such as age, BMI, and race. The harmonization framework advanced previous approaches by comprehensively addressing inconsistencies in variable definitions and implementing robust statistical techniques, including multiple imputation and residual regression analysis[ 25 , 35 ]. This approach ensured that the harmonized dataset was suitable for pooled analyses, while including diverse populations, particularly from NHANES, improved the generalizability of findings. Furthermore, genomic data were harmonized using stringent quality control and imputation protocols, enabling reliable genotype-phenotype association analyses. This study builds upon and extends prior research in several key areas. Unlike previous studies[ 36 – 40 ] that often rely on single-cohort datasets, this work leveraged multiple datasets to enhance statistical power and representativeness. While genomic data integration has been extensively explored in other disease areas, its application to osteoporosis research remains relatively limited, primarily due to constraints in sample size[ 41 , 42 ]. By harmonizing genomic and phenotypic data, we established a critical resource for investigating BMD and fracture risk across diverse populations. Additionally, the inclusion of NHANES improved demographic diversity compared with prior studies, which have predominantly focused on populations of European ancestry [ 43 , 44 ]. The findings have significant implications for both clinical practice and research. The harmonized dataset lays the groundwork for developing more equitable and accurate fracture risk prediction models, which can account for population-specific differences, addressing long-standing disparities in osteoporosis diagnosis and treatment[ 45 , 46 ]. Integrating genomic data also provides opportunities to explore gene-environment interactions, which may reveal new biological pathways underlying BMD and fracture risk [ 47 ]. By creating a comprehensive data harmonization framework, this study provides a valuable methodological resource that can facilitate future research, including meta-analyses and replication studies. This study has several limitations. The study results generated from cross-sectional data may not be fully applicable to cohort data with phenotypic and genomic factors. Although multiple imputations effectively addressed missing data, some variables, such as alcohol consumption, were inconsistently recorded across datasets, potentially limiting the comprehensiveness of the analyses. Additionally, while NHANES improved demographic diversity, the overrepresentation of White participants in the GDBF and GWAS datasets highlights the need for further inclusion of underrepresented populations in future research. Furthermore, while harmonization aimed to enhance model performance by reducing variability and aligning definitions across datasets, the observed improvement in predictive accuracy was relatively modest. Several factors could potentially contribute to this limited increase, including prior adjustments for key confounders (age and weight) during harmonization and the inherently complex nature of osteoporosis risk. Future directions include expanding the harmonization framework to longitudinal datasets and providing insights into the dynamic relationships between genomic factors and BMD over time. Efforts to include more racially and ethnically diverse populations are critical for ensuring the generalizability and equity of osteoporosis research. Additionally, incorporating other omics data, such as transcriptomics, methylomics, proteomics, and metabolomics, could further help to elucidate the molecular mechanisms underlying osteoporosis and related conditions[ 48 , 49 ]. This study provides a robust and effective framework for harmonizing phenotypic and genomic data, facilitating the integration of multiple large-scale datasets. The harmonized data enhances the accuracy and generalizability of assessments, offering potential to develop more precise and equitable fracture risk prediction tools. By addressing methodological challenges such as inconsistent variable definitions, missing data, and demographic imbalances, this work establishes a foundation for future precision medicine research in osteoporosis, particularly for populations historically underrepresented in genomic studies. The availability of these harmonized datasets presents significant opportunities for advancing osteoporosis research, enabling more precise identification of genetic and environmental determinants of bone health. Future studies can leverage this approach to investigate diverse populations, develop targeted preventive strategies, and ultimately reduce disparities in osteoporosis diagnosis and treatment. Data Sharing The data/analyses presented in the current publication are based on study data downloaded from the dbGaP website under phs000138.v2. p1 and phs000390.v1. p1 , following the corresponding data use agreement. The NHANES data used in this study are publicly available from the National Center for Health Statistics (NCHS) at https://www.cdc.gov/nchs/nhanes/index.html . Disclosures of conflicts of interest Anqi Liu, Jianing Liu, and Qing Wu declare that they have no conflict of interest. L.W. provided consulting services to Pupil Bio Inc., Techspert, and Galiher DeRobertis & Waxman LLP, and reviewed manuscripts for Gastroenterology Report, activities unrelated to this study, and received an honorarium. Funding This study was supported by the National Institute on Minority Health and Health Disparities (R21MD013681, awarded to Dr. Q. Wu) and the National Institute on Aging (R01AG080017, awarded to Dr. Q. Wu). Role of the Funder/Sponsor The funding sources had no role in the design and conduct of the study; collection, management, analysis, or interpretation of the data; preparation, review, or approval of the manuscript; or the decision to submit the manuscript for publication. Authors’ Contributions Conceptualization: Qing Wu. Data curation: Anqi Liu. Formal analysis: Anqi Liu. Funding acquisition: Qing Wu. Investigation: Qing Wu, Anqi Liu. Methodology: Anqi Liu, Qing Wu, Jianing Liu, Lang Wu Project administration: Qing Wu. Resources: Qing Wu. Software: Anqi Liu. Supervision: Qing Wu. Validation: Anqi Liu. Visualization: Anqi Liu. Writing –original draft: Anqi Liu, Jianing Liu, Qing Wu. Writing –review & editing: Qing Wu, Jianing Liu, Anqi Liu, Lang Wu. Reference [1]. ↵ Sozen T , Ozisik L , Calik Basaran N . An overview and management of osteoporosis . Eur J Rheumatol 2017 ; 4 : 46 – 56 . OpenUrl CrossRef PubMed [2]. ↵ on behalf of the Scientific Advisory Board of the European Society for Clinical and Economic Aspects of Osteoporosis (ESCEO) and the Committees of Scientific Advisors and National Societies of the International Osteoporosis Foundation (IOF) , Kanis JA , Cooper C , et al. European guidance for the diagnosis and management of osteoporosis in postmenopausal women . Osteoporos Int 2019 ; 30 : 3 – 44 . OpenUrl CrossRef PubMed [3]. ↵ Singer AJ , Sharma A , Deignan C , et al. Closing the gap in osteoporosis management: the critical role of primary care in bone health . Curr Med Res Opin 2023 ; 39 : 387 – 398 . OpenUrl PubMed [4]. ↵ Cummings SR , Melton LJ . Epidemiology and outcomes of osteoporotic fractures . The Lancet 2002 ; 359 : 1761 – 1767 . OpenUrl [5]. ↵ Yuan J , Tickner J , Mullin BH , et al. Advanced Genetic Approaches in Discovery and Characterization of Genes Involved With Osteoporosis in Mouse and Human . Front Genet 2019 ; 10 : 288 . OpenUrl PubMed [6]. Estrada K , Styrkarsdottir U , Evangelou E , et al. Genome-wide meta-analysis identifies 56 bone mineral density loci and reveals 14 loci associated with risk of fracture . Nat Genet 2012 ; 44 : 491 – 501 . OpenUrl CrossRef PubMed [7]. ↵ Zheng H , Forgetta V , Hsu Y , et al. Whole genome sequencing identifies EN1 as a determinant of bone density and fracture . Nature 2015 ; 526 : 112 – 117 . OpenUrl CrossRef PubMed [8]. ↵ Brookes AJ , Robinson PN . Human genotype–phenotype databases: aims, challenges and opportunities . Nat Rev Genet 2015 ; 16 : 702 – 715 . OpenUrl CrossRef PubMed [9]. ↵ Fortier I , Dragieva N , Saliba M , et al. Harmonization of the Health and Risk Factor Questionnaire data of the Canadian Partnership for Tomorrow Project: a descriptive analysis . CMAJ Open 2019 ; 7 : E272 – E282 . OpenUrl Abstract / FREE Full Text [10]. ↵ Luningham JM , McArtor DB , Hendriks AM , et al. Data Integration Methods for Phenotype Harmonization in Multi-Cohort Genome-Wide Association Studies With Behavioral Outcomes . Front Genet 2019 ; 10 : 1227 . OpenUrl PubMed [11]. ↵ Albrechtsen A , Nielsen FC , Nielsen R . Ascertainment Biases in SNP Chips Affect Measures of Population Divergence . Mol Biol Evol 2010 ; 27 : 2534 – 2547 . OpenUrl CrossRef PubMed Web of Science [12]. Hinrichs AL , Culverhouse RC , Suarez BK . Genotypic discrepancies arising from imputation . BMC Proc 2014 ; 8 : S17 . [13]. Landry LG , Ali N , Williams DR , et al. Lack Of Diversity In Genomic Databases Is A Barrier To Translating Precision Medicine Research Into Practice . Health Aff (Millwood) 2018 ; 37 : 780 – 785 . OpenUrl CrossRef PubMed [14]. ↵ Popejoy AB , Fullerton SM . Genomics is failing on diversity . Nature 2016 ; 538 : 161 – 164 . OpenUrl CrossRef PubMed [15]. ↵ Pan K , Bazzano LA , Betha K , et al. Large-Scale Data Harmonization Across Prospective Studies . Am J Epidemiol 2023 ; 192 : 2033 – 2049 . OpenUrl PubMed [16]. Wong M , Day N , Luan J , et al. The detection of gene–environment interaction for continuous traits: should we deal with measurement error by bigger studies or better measurement? Int J Epidemiol 2003 ; 32 : 51 – 57 . OpenUrl CrossRef PubMed Web of Science [17]. Stilp AM , Emery LS , Broome JG , et al. A System for Phenotype Harmonization in the National Heart, Lung, and Blood Institute Trans-Omics for Precision Medicine (TOPMed) Program . Am J Epidemiol 2021 ; 190 : 1977 – 1992 . OpenUrl PubMed [18]. ↵ Sung YJ , De Las Fuentes L , Winkler TW , et al. A multi-ancestry genome-wide study incorporating gene–smoking interactions identifies multiple new loci for pulse pressure and mean arterial pressure . Hum Mol Genet 2019 ; 28 : 2615 – 2633 . OpenUrl PubMed [19]. ↵ Schmidt CO , Struckmann S , Enzenbach C , et al. Facilitating harmonized data quality assessments. A data quality framework for observational health research data collections with software implementations in R . BMC Med Res Methodol 2021 ; 21 : 63 . OpenUrl PubMed [20]. ↵ Little R , Rubin D . Statistical Analysis with Missing Data, Third Edition . 1st ed . Wiley . Epub ahead of print 17 April 2019 . DOI: 10.1002/9781119482260 . OpenUrl CrossRef [21]. ↵ National Center for Biotechnology Information . GWAS for Genetic Determinants of Bone Fragility in European-American Premenopausal Women , https://www.ncbi.nlm.nih.gov/projects/gap/cgi-bin/study.cgi?study_id=phs000138.v2.p1 . [22]. ↵ National Center for Biotechnology Information . Genomic Wide Scans for Female Osteoporosis Genes , https://www.ncbi.nlm.nih.gov/projects/gap/cgi-bin/study.cgi?study_id=phs000390.v1.p1 . [23]. ↵ Akinbam L , Chen T-C , Davy O , et al. National Health and Nutrition Examination Survey, 2017–March 2020 Prepandemic File: Sample Design, Estimation, and Analytic Guidelines . National Center for Health Statistics (U.S.) . Epub ahead of print 19 May 2022 . DOI: 10.15620/cdc:115434 . OpenUrl CrossRef [24]. ↵ Liang J. Data on the relationship between chloroform and bone mineral density . 2023 ; 1127026 . [25]. ↵ White IR , Royston P , Wood AM . Multiple imputation using chained equations: Issues and guidance for practice . Stat Med 2011 ; 30 : 377 – 399 . OpenUrl CrossRef PubMed [26]. ↵ Azur MJ , Stuart EA , Frangakis C , et al. Multiple imputation by chained equations: what is it and how does it work? Int J Methods Psychiatr Res 2011 ; 20 : 40 – 49 . OpenUrl CrossRef PubMed [27]. ↵ the Haplotype Reference Consortium . A reference panel of 64,976 haplotypes for genotype imputation . Nat Genet 2016 ; 48 : 1279 – 1283 . OpenUrl CrossRef PubMed [28]. ↵ Durbin R . Efficient haplotype matching and storage using the positional Burrows–Wheeler transform (PBWT) . Bioinformatics 2014 ; 30 : 1266 – 1272 . OpenUrl CrossRef PubMed Web of Science [29]. ↵ Abdi H , Williams LJ . Principal component analysis . WIREs Comput Stat 2010 ; 2 : 433 – 459 . OpenUrl CrossRef [30]. ↵ Price AL , Patterson NJ , Plenge RM , et al. Principal components analysis corrects for stratification in genome-wide association studies . Nat Genet 2006 ; 38 : 904 – 909 . OpenUrl CrossRef PubMed Web of Science [31]. ↵ Lakens D . Calculating and reporting effect sizes to facilitate cumulative science: a practical primer for t-tests and ANOVAs . Front Psychol ; 4 . Epub ahead of print 2013 . DOI: 10.3389/fpsyg.2013.00863 . OpenUrl CrossRef PubMed [32]. ↵ Liang K-Y , Zeger SL . Longitudinal data analysis using generalized linear models . Biometrika 1986 ; 73 : 13 – 22 . OpenUrl CrossRef Web of Science [33]. ↵ Laird NM , Ware JH . Random-effects models for longitudinal data . Biometrics 1982 ; 38 : 963 – 974 . OpenUrl CrossRef PubMed Web of Science [34]. ↵ Doiron D , Burton P , Marcon Y , et al. Data harmonization and federated analysis of population-based studies: the BioSHaRE project . Emerg Themes Epidemiol 2013 ; 10 : 12 . OpenUrl CrossRef PubMed [35]. ↵ Freedman D. Statistical Models: Theory and Practice . Cambridge: Cambridge University Press . Epub ahead of print 2005 . DOI: 10.1017/CBO9781139165495 . OpenUrl CrossRef [36]. ↵ Wang L , Yu W , Yin X , et al. Prevalence of Osteoporosis and Fracture in China: The China Osteoporosis Prevalence Study . JAMA Netw Open 2021 ; 4 : e2121106 . OpenUrl [37]. Karasik D , Hsu Y-H , Zhou Y , et al. Genome-wide pleiotropy of osteoporosis-related phenotypes: The framingham study . J Bone Miner Res 2010 ; 25 : 1555 – 1563 . OpenUrl CrossRef PubMed [38]. Mullin BH , Tickner J , Zhu K , et al. Characterisation of genetic regulatory effects for osteoporosis risk variants in human osteoclasts . Genome Biol 2020 ; 21 : 80 . OpenUrl CrossRef PubMed [39]. Richards JB , Zheng H-F , Spector TD . Genetics of osteoporosis from genome-wide association studies: advances and challenges . Nat Rev Genet 2012 ; 13 : 576 – 588 . OpenUrl CrossRef PubMed [40]. ↵ Kiel DP , Ferrari SL , Cupples LA , et al. Genetic variation at the low-density lipoprotein receptor-related protein 5 (LRP5) locus modulates Wnt signaling and the relationship of physical activity with bone mineral density in men . Bone 2007 ; 40 : 587 – 596 . OpenUrl CrossRef PubMed Web of Science [41]. ↵ Richards JB , Zheng H-F , Spector TD . Genetics of osteoporosis from genome-wide association studies: advances and challenges . Nat Rev Genet 2012 ; 13 : 576 – 588 . OpenUrl CrossRef PubMed [42]. ↵ Graat-Verboom L , Wouters EFM , Smeenk FWJM , et al. Current status of research on osteoporosis in COPD: a systematic review . Eur Respir J 2009 ; 34 : 209 – 218 . OpenUrl Abstract / FREE Full Text [43]. ↵ Vásquez E , Alam MT , Murillo R. Race and ethnic differences in physical activity, osteopenia, and osteoporosis: results from NHANES 2009–2010, 2013–2014, 2017–2018 . Arch Osteoporos 2023 ; 19 : 7 . OpenUrl PubMed [44]. ↵ Martin AR , Kanai M , Kamatani Y , et al. Clinical use of current polygenic risk scores may exacerbate health disparities . Nat Genet 2019 ; 51 : 584 – 591 . OpenUrl CrossRef PubMed [45]. ↵ Noel SE , Santos MP , Wright NC . Racial and Ethnic Disparities in Bone Health and Outcomes in the United States . J Bone Miner Res 2020 ; 36 : 1881 – 1905 . OpenUrl [46]. ↵ Thomas PA. Racial and Ethnic Differences in Osteoporosis : J Am Acad Orthop Surg 2007 ; 15 : S26 – S30 . OpenUrl PubMed [47]. ↵ Sambrook P , Nguyen T . Bone mineral density and gene-environment interactions in the search for osteoporosis genes . Environ Health Perspect ; 107 . Epub ahead of print March 1999 . DOI: 10.1289/ehp.107-1566373 . OpenUrl CrossRef [48]. ↵ Qiu C , Yu F , Su K , et al. Multi-omics Data Integration for Identifying Osteoporosis Biomarkers and Their Biological Interaction and Causal Mechanisms . iScience 2020 ; 23 : 100847 . OpenUrl PubMed [49]. ↵ Zhang X , Xu H , Li GH , et al. Metabolomics Insights into Osteoporosis Through Association With Bone Mineral Density . J Bone Miner Res Off J Am Soc Bone Miner Res 2021 ; 36 : 729 – 738 . OpenUrl View the discussion thread. Back to top Previous Next Posted May 13, 2025. Download PDF Supplementary Material Email Thank you for your interest in spreading the word about medRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following Integrative Harmonization of Phenotypic and Genomic Data Improves Bone Mineral Density Prediction in Multi-Study Osteoporosis Research Message Subject (Your Name) has forwarded a page to you from medRxiv Message Body (Your Name) thought you would like to see this page from the medRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share Integrative Harmonization of Phenotypic and Genomic Data Improves Bone Mineral Density Prediction in Multi-Study Osteoporosis Research Anqi Liu , Jianing Liu , Lang Wu , Qing Wu medRxiv 2025.05.12.25327471; doi: https://doi.org/10.1101/2025.05.12.25327471 Share This Article: Copy Citation Tools Integrative Harmonization of Phenotypic and Genomic Data Improves Bone Mineral Density Prediction in Multi-Study Osteoporosis Research Anqi Liu , Jianing Liu , Lang Wu , Qing Wu medRxiv 2025.05.12.25327471; doi: https://doi.org/10.1101/2025.05.12.25327471 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Epidemiology Subject Areas All Articles Addiction Medicine (567) Allergy and Immunology (863) Anesthesia (296) Cardiovascular Medicine (4409) Dentistry and Oral Medicine (443) Dermatology (380) Emergency Medicine (606) Endocrinology (including Diabetes Mellitus and Metabolic Disease) (1505) Epidemiology (15202) Forensic Medicine (30) Gastroenterology (1119) Genetic and Genomic Medicine (6568) Geriatric Medicine (666) Health Economics (994) Health Informatics (4509) Health Policy (1365) Health Systems and Quality Improvement (1608) Hematology (537) HIV/AIDS (1262) Infectious Diseases (except HIV/AIDS) (15902) Intensive Care and Critical Care Medicine (1103) Medical Education (619) Medical Ethics (144) Nephrology (665) Neurology (6572) Nursing (345) Nutrition (998) Obstetrics and Gynecology (1139) Occupational and Environmental Health (954) Oncology (3319) Ophthalmology (967) Orthopedics (369) Otolaryngology (420) Pain Medicine (435) Palliative Medicine (129) Pathology (662) Pediatrics (1686) Pharmacology and Therapeutics (691) Primary Care Research (710) Psychiatry and Clinical Psychology (5420) Public and Global Health (9203) Radiology and Imaging (2190) Rehabilitation Medicine and Physical Therapy (1367) Respiratory Medicine (1191) Rheumatology (593) Sexual and Reproductive Health (709) Sports Medicine (529) Surgery (709) Toxicology (99) Transplantation (288) Urology (265) (function(){function c(){var b=a.contentDocument||a.contentWindow.document;if(b){var d=b.createElement('script');d.innerHTML="window.__CF$cv$params={r:'9fe6a907b8d24807',t:'MTc3OTIzMTAyMw=='};var a=document.createElement('script');a.src='/cdn-cgi/challenge-platform/scripts/jsd/main.js';document.getElementsByTagName('head')[0].appendChild(a);";b.getElementsByTagName('head')[0].appendChild(d)}}if(document.body){var a=document.createElement('iframe');a.height=1;a.width=1;a.style.position='absolute';a.style.top=0;a.style.left=0;a.style.border='none';a.style.visibility='hidden';document.body.appendChild(a);if('loading'!==document.readyState)c();else if(window.addEventListener)document.addEventListener('DOMContentLoaded',c);else{var e=document.onreadystatechange||function(){};document.onreadystatechange=function(b){e(b);'loading'!==document.readyState&&(document.onreadystatechange=e,c())}}}})();

Text is read by the "Ask this paper" AI Q&A widget below. Extraction quality varies by source — PMC NXML preserves structure cleanly, OA-HTML may include some navigation residue, and OA-PDF can have broken hyphenation. The publisher copy (via DOI) is the canonical version.

My notes (saved in your browser only)

Ask this paper AI returns verbatim quotes from the full text · source: preprint-html

Answers must be backed by verbatim quotes from this paper's full text. Hallucinated quotes are dropped automatically; if no verbatim passage answers the question, we say so. How this works

Citation neighborhood (no data yet)

We don't have any in-corpus citations linked to this paper yet. This is a recent paper (2025) — citers typically take a year or two to land, and the OpenAlex reference graph may still be filling in.

Source provenance

europepmc
last seen: 2026-05-20T01:45:00.602351+00:00