Early Prediction of Gestational Diabetes Using Integrated Cell-free DNA Features and Omics-derived Genetic Scores

preprint OA: closed CC-BY-NC-ND-4.0
📄 Open PDF Full text JSON View at publisher

Abstract

Background Gestational diabetes mellitus (GDM) affects 15.6% of pregnancies globally, with Vietnam exhibiting one of the highest prevalences at 21%. Current diagnostic approaches at 24-28 weeks limit early intervention opportunities. We developed a multi-modal machine learning framework integrating cell-free DNA (cfDNA) structural features and genetic information for early GDM prediction at 10-12 weeks of gestation in Vietnamese women. Methods We analyzed blood samples from 1,086 pregnant women (435 GDM cases, 651 controls) collected at 9-12 weeks. Two parallel analytical pathways were employed: cfDNA profiling extracting cfDNA-specific features (fragment length, end motifs, GC content, nucleosome patterns), and whole-genome imputation generating predictions for ∼19,000 omics traits. Component scores were developed using TabPFN classifier and integrated via logistic regression into a unified master score. Results Genome-wide analysis identified five omics traits with significant GDM associations: HSD11B1 , NEK7 , COMMD10 , KLRC4 , and OCEL1 . Component score optimization revealed distinct patterns—cfDNA scores peaked at 200 features (AUC=71.53), while genetics-based scores improved with up to 2,000 omics traits (AUC=77.21). The final master score, integrating three components (gbSC 2000 , gbSC BH , cfSC200), achieved AUCs of 86.82 - 87.19 across validation cohorts with 70% sensitivity and 89% specificity. Addition-deletion analysis confirmed that both cfDNA and genetic components provided essential, non-redundant contributions. Conclusions This multi-modal framework demonstrates superior performance compared to single-biomarker approaches, enabling risk stratification from very low (4% GDM prevalence) to very high risk (90% prevalence). At the cutoff 0.4, the model identifies 78% of future GDM cases at 10-12 weeks while maintaining an 18% false-positive rate, potentially enabling early interventions to prevent GDM development and associated complications.
Full text 58,857 characters · extracted from preprint-html · click to expand
Early Prediction of Gestational Diabetes Using Integrated Cell-free DNA Features and Omics-derived Genetic Scores | medRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-P4HH5NV'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search Early Prediction of Gestational Diabetes Using Integrated Cell-free DNA Features and Omics-derived Genetic Scores Vinh Nguyen Dao , Nhat-Thang Tran , Ta-Son Vo , Hong-Thinh Le , Thu-Ha Thi Nguyen , Quoc-Huy Vu Nguyen , Minh-Thi Thi Ha , Tam Minh Le , Diem-Tuyet Thi Hoang , Khanh-Trang Nguyen Huynh , Nhan Viet Nguyen , Chuong Canh Nguyen , Thuong Chi Bui , Xuan Thanh Nguyen , Sa Viet Le , Vinh Dinh Tran , My-Nhi Ba Nguyen , Thong Van Nguyen , Tuyet-Anh Thi Nguyen , Ba Phuoc Hoang , Trong Van Nguyen , Thuy-Ai Thuy Nguyen , Toa Tri Nguyen , Thang Duc Duong , Cuong Huy Pham , Kim-Oanh Thi Luong , Cuong Ngoc Dao , Khanh Van Hoang , Thu-Thanh Thi Huynh , Khuong Manh Nguyen , Son-Tra Thi Tran , Hoanh Trung Tran , Son Canh Nguyen , Thuy Dinh Tran , Phương Thi Lan Nguyen , Thanh Viet Pham , Kong Chi Pham , Minh Doan Thai , My-Hang Thi Truong , Hieu Ha Pham , Thanh-Thuy Thi Do , Sang Hung Tang , Hoai-Nghia Nguyen , Minh-Duy Phan , Hoa Thi Dao , Hoa Giang doi: https://doi.org/10.1101/2025.09.03.25334985 Vinh Nguyen Dao 1 Medical Genetics Institute , Ho Chi Minh city, Vietnam 2 Gene Solutions , Ho Chi Minh city, Vietnam Find this author on Google Scholar Find this author on PubMed Search for this author on this site Nhat-Thang Tran 3 University Medical Center , Ho Chi Minh city, Vietnam Find this author on Google Scholar Find this author on PubMed Search for this author on this site Ta-Son Vo 4 Vinmec Health Care System , Hanoi, Vietnam Find this author on Google Scholar Find this author on PubMed Search for this author on this site Hong-Thinh Le 5 Can Tho Obstetrics and Gynecology Hospital , Can Tho, Vietnam Find this author on Google Scholar Find this author on PubMed Search for this author on this site Thu-Ha Thi Nguyen 6 National Hospital of Obstetrics and Gynecology , Hanoi, Vietnam Find this author on Google Scholar Find this author on PubMed Search for this author on this site Quoc-Huy Vu Nguyen 7 Hue University of Medicine and Pharmacy, Hue University , Hue, Vietnam Find this author on Google Scholar Find this author on PubMed Search for this author on this site Minh-Thi Thi Ha 7 Hue University of Medicine and Pharmacy, Hue University , Hue, Vietnam Find this author on Google Scholar Find this author on PubMed Search for this author on this site Tam Minh Le 7 Hue University of Medicine and Pharmacy, Hue University , Hue, Vietnam Find this author on Google Scholar Find this author on PubMed Search for this author on this site Diem-Tuyet Thi Hoang 8 Hung Vuong Hospital , Ho Chi Minh city, Vietnam Find this author on Google Scholar Find this author on PubMed Search for this author on this site Khanh-Trang Nguyen Huynh 9 Pham Ngoc Thach University of Medicine , Ho Chi Minh city, Vietnam Find this author on Google Scholar Find this author on PubMed Search for this author on this site Nhan Viet Nguyen 4 Vinmec Health Care System , Hanoi, Vietnam Find this author on Google Scholar Find this author on PubMed Search for this author on this site Chuong Canh Nguyen 10 Hanoi Obstetrics & Gynecology Hospital , Hanoi, Vietnam Find this author on Google Scholar Find this author on PubMed Search for this author on this site Thuong Chi Bui 11 Nhan dan Gia dinh Hospital , Ho Chi Minh city, Vietnam Find this author on Google Scholar Find this author on PubMed Search for this author on this site Xuan Thanh Nguyen 12 Hue Central General Hospital , Hue, Vietnam Find this author on Google Scholar Find this author on PubMed Search for this author on this site Sa Viet Le 12 Hue Central General Hospital , Hue, Vietnam Find this author on Google Scholar Find this author on PubMed Search for this author on this site Vinh Dinh Tran 13 Danang Obstetrics and Pediatrics Hospital , Danang, Vietnam Find this author on Google Scholar Find this author on PubMed Search for this author on this site My-Nhi Ba Nguyen 14 Tam Anh General Hospital , Ho Chi Minh city, Vietnam Find this author on Google Scholar Find this author on PubMed Search for this author on this site Thong Van Nguyen 8 Hung Vuong Hospital , Ho Chi Minh city, Vietnam Find this author on Google Scholar Find this author on PubMed Search for this author on this site Tuyet-Anh Thi Nguyen 15 Hanh Phuc International Hospital , Ho Chi Minh city, Vietnam Find this author on Google Scholar Find this author on PubMed Search for this author on this site Ba Phuoc Hoang 16 Vung Tau General Hospital , Ho Chi Minh city, Vietnam Find this author on Google Scholar Find this author on PubMed Search for this author on this site Trong Van Nguyen 17 Ba Ria General Hospital , Ho Chi Minh city, Vietnam Find this author on Google Scholar Find this author on PubMed Search for this author on this site Thuy-Ai Thuy Nguyen 5 Can Tho Obstetrics and Gynecology Hospital , Can Tho, Vietnam Find this author on Google Scholar Find this author on PubMed Search for this author on this site Toa Tri Nguyen 18 A Thai Nguyen Hospital , Thai Nguyen, Vietnam Find this author on Google Scholar Find this author on PubMed Search for this author on this site Thang Duc Duong 19 Obstetrics and Gynecology Clinics , Vietnam Find this author on Google Scholar Find this author on PubMed Search for this author on this site Cuong Huy Pham 20 Bac Ninh 2 Obstetrics & Gynecology Hospital , Bac Ninh, Vietnam Find this author on Google Scholar Find this author on PubMed Search for this author on this site Kim-Oanh Thi Luong 21 Nghe An Obstetrics and Pediatrics Hospital , Nghe An, Vietnam Find this author on Google Scholar Find this author on PubMed Search for this author on this site Cuong Ngoc Dao 19 Obstetrics and Gynecology Clinics , Vietnam Find this author on Google Scholar Find this author on PubMed Search for this author on this site Khanh Van Hoang 22 Andrology and Fertility Hospital of Ha Noi , Hanoi, Vietnam Find this author on Google Scholar Find this author on PubMed Search for this author on this site Thu-Thanh Thi Huynh 23 Khanh Hoa General Hospital , Khanh Hoa, Vietnam Find this author on Google Scholar Find this author on PubMed Search for this author on this site Khuong Manh Nguyen 19 Obstetrics and Gynecology Clinics , Vietnam Find this author on Google Scholar Find this author on PubMed Search for this author on this site Son-Tra Thi Tran 24 Vietnam-Cuba Dong Hoi Friendship Hospital , Quang Binh, Vietnam Find this author on Google Scholar Find this author on PubMed Search for this author on this site Hoanh Trung Tran 19 Obstetrics and Gynecology Clinics , Vietnam Find this author on Google Scholar Find this author on PubMed Search for this author on this site Son Canh Nguyen 19 Obstetrics and Gynecology Clinics , Vietnam Find this author on Google Scholar Find this author on PubMed Search for this author on this site Thuy Dinh Tran 25 Dong Nai 2 Hospital , Dongnai, Vietnam Find this author on Google Scholar Find this author on PubMed Search for this author on this site Phương Thi Lan Nguyen 26 Long Khanh General Hospital Dongnai , Vietnam Find this author on Google Scholar Find this author on PubMed Search for this author on this site Thanh Viet Pham 27 Mekong Obstetrics and Gynecology Hospital , Ho Chi Minh city, Vietnam Find this author on Google Scholar Find this author on PubMed Search for this author on this site Kong Chi Pham 13 Danang Obstetrics and Pediatrics Hospital , Danang, Vietnam Find this author on Google Scholar Find this author on PubMed Search for this author on this site Minh Doan Thai 28 MyDuc Hospital , Ho Chi Minh city, Vietnam Find this author on Google Scholar Find this author on PubMed Search for this author on this site My-Hang Thi Truong 2 Gene Solutions , Ho Chi Minh city, Vietnam Find this author on Google Scholar Find this author on PubMed Search for this author on this site Hieu Ha Pham 2 Gene Solutions , Ho Chi Minh city, Vietnam Find this author on Google Scholar Find this author on PubMed Search for this author on this site Thanh-Thuy Thi Do 1 Medical Genetics Institute , Ho Chi Minh city, Vietnam 2 Gene Solutions , Ho Chi Minh city, Vietnam Find this author on Google Scholar Find this author on PubMed Search for this author on this site Sang Hung Tang 1 Medical Genetics Institute , Ho Chi Minh city, Vietnam 2 Gene Solutions , Ho Chi Minh city, Vietnam Find this author on Google Scholar Find this author on PubMed Search for this author on this site Hoai-Nghia Nguyen 1 Medical Genetics Institute , Ho Chi Minh city, Vietnam 2 Gene Solutions , Ho Chi Minh city, Vietnam Find this author on Google Scholar Find this author on PubMed Search for this author on this site Minh-Duy Phan 1 Medical Genetics Institute , Ho Chi Minh city, Vietnam 2 Gene Solutions , Ho Chi Minh city, Vietnam Find this author on Google Scholar Find this author on PubMed Search for this author on this site Hoa Thi Dao 6 National Hospital of Obstetrics and Gynecology , Hanoi, Vietnam Find this author on Google Scholar Find this author on PubMed Search for this author on this site For correspondence: hoagiang{at}genesolutions.vn dr.daohoa{at}nhog.vn Hoa Giang 1 Medical Genetics Institute , Ho Chi Minh city, Vietnam 2 Gene Solutions , Ho Chi Minh city, Vietnam Find this author on Google Scholar Find this author on PubMed Search for this author on this site For correspondence: hoagiang{at}genesolutions.vn dr.daohoa{at}nhog.vn Abstract Full Text Info/History Metrics Supplementary material Data/Code Preview PDF Abstract Background Gestational diabetes mellitus (GDM) affects 15.6% of pregnancies globally, with Vietnam exhibiting one of the highest prevalences at 21%. Current diagnostic approaches at 24-28 weeks limit early intervention opportunities. We developed a multi-modal machine learning framework integrating cell-free DNA (cfDNA) structural features and genetic information for early GDM prediction at 10-12 weeks of gestation in Vietnamese women. Methods We analyzed blood samples from 1,086 pregnant women (435 GDM cases, 651 controls) collected at 9-12 weeks. Two parallel analytical pathways were employed: cfDNA profiling extracting cfDNA-specific features (fragment length, end motifs, GC content, nucleosome patterns), and whole-genome imputation generating predictions for ∼19,000 omics traits. Component scores were developed using TabPFN classifier and integrated via logistic regression into a unified master score. Results Genome-wide analysis identified five omics traits with significant GDM associations: HSD11B1 , NEK7 , COMMD10 , KLRC4 , and OCEL1 . Component score optimization revealed distinct patterns—cfDNA scores peaked at 200 features (AUC=71.53), while genetics-based scores improved with up to 2,000 omics traits (AUC=77.21). The final master score, integrating three components (gbSC 2000 , gbSC BH , cfSC200), achieved AUCs of 86.82 - 87.19 across validation cohorts with 70% sensitivity and 89% specificity. Addition-deletion analysis confirmed that both cfDNA and genetic components provided essential, non-redundant contributions. Conclusions This multi-modal framework demonstrates superior performance compared to single-biomarker approaches, enabling risk stratification from very low (4% GDM prevalence) to very high risk (90% prevalence). At the cutoff 0.4, the model identifies 78% of future GDM cases at 10-12 weeks while maintaining an 18% false-positive rate, potentially enabling early interventions to prevent GDM development and associated complications. Introduction Gestational diabetes mellitus (GDM) affects approximately 15.6% of pregnancies worldwide, ranking as one of the most common metabolic complications during pregnancy ( H. Wang et al., 2022 ). This condition significantly increases the risk of adverse maternal and neonatal outcomes such as preeclampsia, operative delivery, macrosomia, and birth trauma ( Buchanan et al., 2012 ; Wicklow & Retnakaran, 2023 ). The consequences of GDM extend far beyond pregnancy: more than 40% of affected women develop type 2 diabetes within ten years, and their children carry lifelong elevated diabetes risk, perpetuating an intergenerational cycle of metabolic disease ( Sheiner, 2020 ). Current diagnostic approaches rely on glucose tolerance testing at 24-28 weeks of gestation, limiting the window for meaningful early intervention during critical developmental periods. This diagnostic timing constraint is increasingly seen as a fundamental limitation, especially since emerging evidence suggests that GDM-associated metabolic changes can be detected much earlier in pregnancy ( Zhang & Yang, 2022 ; Zhu et al., 2022 ). Furthermore, current screening approaches provide only binary classification at 24-28 weeks, missing opportunities for the more nuanced risk stratification that early prediction could enable. Effective early GDM prediction requires not only high accuracy but also clear risk stratification to guide clinical decision-making—identifying women at various risk levels who might benefit from different intensities of intervention. Cell-free DNA (cfDNA) circulating in maternal blood offers a promising non-invasive approach for early detection of pregnancy complications. The cfDNA, which comes from both maternal and placental sources, provides molecular insights into gestational processes throughout pregnancy. Recent studies have shown that deep learning models analyzing cfDNA sequencing data—including copy number variations, fragmentation patterns, and transcription start site accessibility—can predict GDM with impressive accuracy as early as 12 weeks of gestation ( Tang et al., 2024 ; Y. Wang et al., 2023 ). These methods significantly outperform traditional screening approaches and enable detection weeks before the current diagnostic standards. However, cfDNA features may represent only one component of GDM complex etiology. Emerging evidence suggests that genetic predisposition plays a substantial role in GDM development, with genome-wide association studies identifying numerous susceptibility loci across diverse populations (Pervjakova et al., 2022). The challenge lies in translating this genetic knowledge into clinically actionable prediction models, particularly given the polygenic nature of GDM where thousands of variants or genes may contribute small individual effects. Recent advances in omics prediction models offer a promising approach to capture this genetic complexity by integrating information across approximately 19,000 molecular traits, potentially bridging the gap between genomic discovery and clinical application ( Xu et al., 2022 ). Single-biomarker approaches may be insufficient for complex diseases like GDM, where multiple biological pathways—including insulin signaling, inflammatory responses, and placental metabolism—contribute to disease pathogenesis. Vietnam exhibits one of the world’s highest GDM prevalences at 21.0% ( Global Prevalence of Gestational Diabetes (GDM) | IDF Atlas , n.d.). This elevated prevalence may reflect unique genetic susceptibility patterns, environmental factors, or gene-environment interactions specific to Vietnamese populations. Population-specific prediction models are increasingly recognized as essential, given that genetic risk variants and their effect sizes can vary substantially across ancestries, potentially limiting the transferability of prediction models developed in other populations. Yet Vietnam currently lacks population-specific early detection tools that account for these unique genetic and environmental factors. In this study, we developed a comprehensive machine learning framework that integrates multiple data modalities—including cfDNA features and genetic-derived omics trait predictions—to predict GDM at 10-12 weeks of gestation in Vietnamese women. Our two-stage approach first develops specialized component scores for each data type, then combines them into a unified master score optimized for clinical risk stratification. The framework was validated using independent datasets to ensure generalizability and clinical utility. This precision medicine approach may address a critical unmet clinical need while establishing a framework for population-specific, multi-modal GDM prediction that could potentially transform prenatal care and reduce the substantial burden of diabetes-related complications in this high-risk population. Methodology Participants This retrospective case-control study enrolled pregnant women from multiple leading maternity hospitals and clinics across Vietnam. Eligible participants had undergone non-invasive prenatal testing (NIPT) between 9 th and 12 th week of their gestation and were subsequently assessed for GDM during routine antenatal screening at 24–28 weeks, with GDM defined according to national guidelines. Women carrying fetuses with chromosomal abnormalities were excluded to reduce potential confounding from genetic anomalies. A total of 1,086 pregnant women met the inclusion criteria and completed the study protocol. Of these, 435 women were diagnosed with gestational diabetes mellitus (GDM) based on standard screening conducted between the 24th and 28th weeks of gestation, while 651 women with normal glucose tolerance served as the control group. Sequencing & Imputation All of 1086 NIPT samples were sequenced at low coverage (0.05x – 0.8x). Sequencing was performed using the AVITI platform, with detailed procedural information available in ( Tran et al., 2020 ). Whole-genome imputation, with the exception of chromosomes X and Y, was conducted using QUILT2 ( Li et al., 2024 ). The fetal fraction — a required input for QUILT2 — was estimated by the SeqFF program ( Kim et al., 2015 ). Reference genotype maps were generated from the KHV genome and aligned to the hg38 assembly. Score construction & alignment To construct a comprehensive GDM risk score, we first developed multiple component scores, each capturing distinct types of information available for pregnant women. These component scores were derived independently using the TabPFN classifier, a state-of-the-art AutoML algorithm for small datasets ( Hollmann et al., 2023 , 2025 ). Subsequently, these component scores were aligned onto a common risk scale through simple logistic regression, yielding a unified master score. To support this two-stage framework, the dataset was partitioned into three mutually exclusive subsets: TRAIN (n = 652; 40.34% GDM), VALIDATION (n = 217; 36.41% GDM), and DISCOVERY (n = 217; 42.86% GDM). The TRAIN set was used to develop component scores, the VALIDATION set was employed to calibrate and align them on a unified risk scale, and the DISCOVERY set was reserved for independent performance evaluation of the master score. Cf-DNA score (cfSC) Characteristics of cell-free DNA (cfDNA) were systematically profiled using key attributes, including fragment length distributions, end motif frequencies, GC content (rounded to two decimal places), nucleosome spacing, and chromosomal origin. Fragments with nucleosome distances greater than 400 bp were aggregated, and those exceeding 450 bp in length were truncated. In total, 1,848 distinct cfDNA features were derived for downstream analyses. To construct cfDNA scores (cfSC), we employed two distinct feature selection strategies. In the first approach, features were ranked using p-values from the Wilcoxon rank-sum test comparing GDM to CTRL samples in the TRAIN dataset. Features were then selected based on predefined thresholds, including the top 20, 50, 100, 200, 300, 500, 800, 1000, 1500, and all 1,848 features. In the second approach, Gaussian Random Projection was employed to reduce the dimensionality of the feature matrix to 500 components while preserving essential data structure. Subsequently, Hierarchical Density-Based Spatial Clustering of Applications with Noise (HDBSCAN) clustering technique ( Campello et al., 2013 ) was applied to group features by similarity within this reduced space, with the minimum cluster size parameter optimized iteratively. Features identified as noise were excluded from further analysis. For each cluster, the feature exhibiting the greatest absolute mean difference between GDM and control groups was selected as a representative, facilitating the identification of a non-redundant subset strongly associated with the outcome. Following selection of cfDNA structural features, three previously established markers— motif diversity scores (MDS), methylation-associated (MA) value, and fetal fraction—were incorporated into the analysis as described in ( Tang et al., 2024 ). To differentiate among structural scores, each is denoted by the cfSC prefix followed by the number of selected features or the feature selection method employed. For example, cfSC HDBSCAN stands for the cfDNA score in which features are selected via HDBSCAN clustering. Genetics-based score (gbSC) Omics prediction scores and omics selection A comprehensive atlas of genetic prediction models for approximately 19,000 omics traits was obtained from the OmicsPred database ( https://www.omicspred.org/ ) primarily developed by ( Xu et al., 2022 ). Using this resource, we generated a high-dimensional dataset comprising predicted scores for each trait across all 1,086 individuals in our cohort. In parallel with the approach employed for cfDNA structural scores, we applied two distinct omics selection strategies. First, we evaluated the association between each omics-derived prediction score and gestational diabetes mellitus (GDM) using the Wilcoxon rank-sum test across our cohort of 1,086 samples. Omics traits demonstrating statistical significance following Benjamini–Hochberg false discovery rate (FDR) correction were retained for the construction of genetic scores. In a second strategy, we re-assessed the associations between omics traits and GDM within the TRAIN dataset alone. Subsequently, omics traits which are corresponding to the top 20, 50, 100, 200, 300, 500, 800, 1,000, 1,500, 2,000, 2,500, and 3,000 lowest p-values were selected for genetics-based scores development. Each genetics-based score is denoted by the gbSC prefix followed by the number of selected features or the feature selection method employed. For example, gbSC BH stands for the genetics-based score in which omics traits are selected by Bejamin-Hochberg correction. Master score The master score was constructed by aligning individual component scores using logistic regression, trained on the VALIDATION dataset. Each component score was designed such that higher values indicate an increased likelihood of gestational diabetes mellitus (GDM); accordingly, we expected positive regression coefficients in the final model. We began with an initial logistic regression model including all component scores. To refine the model, we employed a stepwise elimination procedure: at each iteration, a single component score was removed if its regression coefficient was negative or if its p -value exceeded 0.05. The performance of the master score is then verified through AUC addition-deletion analysis. Result Study Design and Workflow Our study employed a comprehensive multi-modal approach to develop an early GDM prediction model using blood samples collected from 1,086 pregnant women between 9-12 weeks of gestation ( Figure 1 ). The cohort comprised 435 women who were diagnosed with GDM and 651 controls with normal glucose tolerance, with GDM diagnosis confirmed through standard screening protocols conducted between 24-28 weeks of gestation. Download figure Open in new tab Figure 1: Study design and analytical workflow. Blood samples from 1,086 pregnant women were collected at 9-12 weeks of gestation, with GDM diagnosis confirmed at 24-28 weeks. The cfDNA underwent two parallel processes: cfDNA profiling to extract cfDNA specific features (fetal fraction, end motifs, fragment length, GC content, nucleosome distance, MA, MDS), and whole-genome imputation to derive multi-omic trait prediction scores. A two-stage machine learning framework integrated these complementary data modalities to develop component scores, which were subsequently combined into a unified master score for early GDM prediction. The analytical workflow incorporated two parallel data generation pathways from circulating cell-free DNA. The first pathway involved cfDNA profiling, extracting cfDNA features including fetal fraction, end motifs, fragment length distributions, GC content, nucleosome distance patterns, along with previously established markers MA and MDS ( Tang et al., 2024 ). These features captured the dynamic, pregnancy-specific characteristics of cfDNA that may reflect ongoing metabolic changes associated with GDM development. The second pathway utilized whole-genome imputation to rebuild individual genetic profiles. These profiles were subsequently used to calculate multi-omics genetic prediction scores for about 19,000 molecular traits, thereby capturing the wider polygenic architecture associated with GDM susceptibility. These diverse data streams—cfDNA features and genetic information—were integrated using a machine learning framework designed to develop component scores for each data modality before combining them into a unified classification model. This two-stage approach allowed us to harness the complementary strengths of both dynamic cfDNA structural changes and underlying genetic predisposition, ultimately generating a master score optimized for early GDM prediction and clinical risk stratification. Omics traits association Our genome-wide analysis of ∼19,000 omics-derived trait prediction scores across 1,086 samples identified five genes with statistically significant associations with GDM after Benjamini-Hochberg correction for multiple testing ( Figure 2 ). These genes— HSD11B1 , NEK7 , COMMD10 , KLRC4 , and OCEL1 —showed the strongest evidence for association with GDM risk, with -log₁₀(p-values) ranging from approximately 4.5 to 6.0. While the Manhattan plot revealed numerous additional signals approaching nominal significance thresholds (P < 0.05 and P < 0.01), the relatively small number of variants surviving multiple testing correction underscores the polygenic nature of GDM, where many genes may contribute modest individual effects. These results suggest the utility of omics-informed predictive models in capturing the complex molecular architecture of GDM and underscore their potential for improving early classification and risk stratification in pregnancy. Download figure Open in new tab Figure 2: Genome-wide association analysis of omics traits with GDM. Manhattan plot showing the genomic distribution of ∼19,000 omics traits and their association with GDM risk. Each point represents an omics trait positioned by its corresponding gene location or mean SNP location (hg38 assembly). Red points indicate traits showing statistical significance after Benjamini–Hochberg correction (uppermost blue line). Two lower blue lines indicate nominal significance thresholds (P < 0.05 and P < 0.01). Component score optimization The performance of individual component scores demonstrated distinct optimization patterns that varied substantially between cfDNA features and genetics-based predictors ( Figure 3 and Table S1). For cfDNA scores (cfSC), we observed a clear performance peak when incorporating 200 features, achieving an AUC of approximately 71.52. Beyond this optimal point, model performance declined progressively, plateauing around 66-68 when including 500 or more features. This performance degradation suggests that adding less discriminative cfDNA structural features introduces noise that outweighs any marginal predictive benefit. Notably, the HDBSCAN-based feature selection approach yielded an AUC of approximately 68.94, which was notably lower than the optimal cfSC 200 model. This indicates that the unsupervised clustering approach may sacrifice some discriminative power by not directly optimizing GDM classification performance. Download figure Open in new tab Figure 3: Performance optimization of component scores. AUC values (combined validation and discovery cohorts) plotted against the number of features included in cfSC and gbSC models. Blue curves show performance trends as feature numbers increase. Red dashed lines indicate baseline model performance: cfSC HDBSCAN for cfDNA scores (cfSC) and gbSC BH for genetics-based scores (gbSC). Optimal performance occurs at 200 features for cfSC and about 2,000 features for gbSC. In contrast, genetics-based scores (gbSC) exhibited markedly different optimization dynamics. Performance improved steadily from an AUC of approximately 63.62 with 20 omics traits to 71.58 with 200 traits. The AUC then remained relatively stable around 72-73 until approximately 800 traits were included, after which performance continued to increase gradually, ultimately reaching peak performance near 77.21 when incorporating 2,000 omics traits. Notably, the genetics-based score built exclusively from the five statistically significant omics traits (gbSC BH ) achieved only moderate performance with an AUC of 68.96, substantially lower than models incorporating larger numbers of nominally significant traits. This finding underscores a key insight: while individual omics traits may not reach genome-wide significance thresholds, their collective contribution in a polygenic framework provides substantially greater predictive value than focusing solely on the most statistically significant signals. Master score performance The master score construction process began with an initial logistic regression model incorporating all 24 component scores, but stepwise elimination based on statistical significance and coefficient direction ultimately retained only three components: gbSC 2000 , gbSC BH , and cfSC 200 ( Figure 4 and Table S2). This parsimonious final model represents the optimal balance between predictive performance and model complexity, with gbSC 2000 and cfSC 200 being the highest-performing representatives from the genetics and structural feature domains, respectively. While individual component scores achieved AUCs ranging from 63 to 77, the master score demonstrated substantially enhanced performance with AUCs of 86.82 (95% CI: 80.98-91.24) in the discovery cohort and 87.19 (95% CI: 81.94-91.72) in the validation cohort. The ROC analysis revealed a sensitivity of 70% (95% CI: 63%-77%) and specificity of 89% (95% CI: 84%-92%) when evaluated across the combined validation and discovery datasets ( Figure 4A ). Detailed results of the logistic regression are provided in (Table S3). Download figure Open in new tab Figure 4: Master score performance and component contribution analysis. (A) AUC values for individual component scores (colored lines) and the integrated master score (black line) in validation and discovery cohorts. Sensitivity and specificity values are shown for the combined validation and discovery datasets. (B) Impact of component score removal (Deletion sub-panel) and addition (Addition sub-panel) on master score AUC. The master score baseline AUC is indicated at the bottom. Removing cfSC 200 or gbSC 2000 substantially reduces performance (7-8 percentage points), while adding other components provides minimal improvement. The addition-deletion analysis provided crucial insights into each component’s contribution to overall performance ( Figure 4B ). Removal of either gbSC 2000 or cfSC 200 from the master score resulted in substantial performance decrements of 7.26 and 8.62 percentage points, respectively, in the validation dataset, with similar magnitudes observed in the discovery cohort. This demonstrates that both genetics-based and cfDNA structural information provide essential, non-redundant contributions to GDM prediction. Interestingly, the removal of gbSC BH showed a more modest impact (1.18 percentage point reduction), suggesting its role may be more complementary than fundamental. Conversely, the addition analysis revealed that incorporating any of the remaining 21 component scores produced minimal improvements in AUC, with most changes being less than 0.5 percentage points. This finding validates the effectiveness of the stepwise elimination procedure and confirms that the three-component master score captures the essential predictive information available in our multi-modal dataset without introducing unnecessary complexity or overfitting. Risk Stratification The master score demonstrated excellent capacity for clinical risk stratification, with clear separation between GDM cases and controls across the full range of score values ( Figure 5 ). The cumulative distribution curves revealed that approximately 88% of controls scored below 0.5, while only 30% of GDM cases fell within this low-risk range. This substantial separation provides a strong foundation for clinical decision-making at early gestational stages. Download figure Open in new tab Figure 5: Clinical risk stratification using the master score. Cumulative distribution curves showing the proportion of control (gray) and GDM (black) populations exceeding each master score threshold in combined validation and discovery cohorts. Red dots indicate GDM prevalence within score bins, assuming 21% population prevalence. For example, 75% of GDM cases score above 0.4 compared to 18% of controls, and women scoring above 0.9 have 90% GDM prevalence (95% CI: 79%-100%). The red dashed line shows the relationship between GDM risk and master score. Using the established Vietnamese population GDM prevalence of 21% as a baseline reference ( Global Prevalence of Gestational Diabetes (GDM) | IDF Atlas , n.d.), we identified clinically meaningful risk thresholds that could guide prenatal care strategies. Women with master scores of 0.3 or higher demonstrated GDM prevalence rates at least similar to the population baseline. At a cutoff of 0.4, approximately 75% of GDM cases would be identified, while maintaining a relatively low false positive rate of 18% among controls. The risk stratification analysis revealed four distinct risk categories based on master score ranges. Women scoring between 0.0-0.3 represented a low-risk group with GDM prevalence is below 10%. The moderate-risk category (scores 0.3-0.5) encompassed women with GDM prevalence ranging from 10% to 30%, remaining at similar level of population baseline risk. Women in the high-risk category (scores 0.5-0.8) showed GDM prevalence of approximately 45%, nearly double the population baseline. Most strikingly, women in the severe-risk category (scores 0.8-1.0) exhibited GDM prevalence exceeding 70%, reaching 90% (95% CI: 78%-100%) for those scoring above 0.9. Discussion The multi-modal machine learning framework developed in this study demonstrates how integrating complementary data types can substantially enhance early GDM prediction compared to single-biomarker approaches. Our findings reveal that genetics-based scores (gbSC) and cfDNA features (cfSC) capture fundamentally different aspects of GDM pathogenesis, with each contributing essential, non-redundant information to the final prediction model. The genetics-based (gbSC) scores reflect the stable, time-invariant genetic predisposition of pregnant women to GDM development. These scores derive entirely from germline genetic variation and remain constant throughout pregnancy, representing the underlying constitutional risk that each woman carries. In contrast, cfSC scores capture dynamic, pregnancy-specific changes in cfDNA fragmentation patterns that may reflect ongoing metabolic perturbations, placental dysfunction, or inflammatory processes associated with early GDM pathogenesis. The complementary nature of these two data modalities is clearly demonstrated by our addition-deletion analysis, which showed that removing either gbSC 2000 or cfSC 200 resulted in substantial performance decrements of 6-8 percentage points. Importantly, our logistic regression suggests that a substantial proportion of women who will develop GDM exhibit normal cfDNA structural profiles at 12 weeks of gestation but are nevertheless identified through their genetic risk profiles captured by gbSC 2000 . This finding has significant clinical implications, as it indicates that cfDNA structural features alone—while valuable—may be insufficient to capture the full complexity of GDM risk. Future research should explore integrating additional data modalities, including clinical parameters such as maternal age, BMI, and family history (Lyu et al., 2025; Zhu et al., 2025), copy number variation profiles from cfDNA ( Y. Wang et al., 2023 ), and transcription start site accessibility patterns ( Tang et al., 2024 ) to further enhance prediction accuracy. Several technical limitations warrant consideration. Our imputation approach was restricted to biallelic variants, potentially limiting the completeness of genetic variant representation, particularly for structural variants or rare mutations that might contribute to GDM risk. Additionally, the omics prediction atlas was constructed using the hg19 reference genome while our imputation utilized hg38, creating occasional mismatches that resulted in missing SNPs for some omics trait predictions. Although QUILT2 showed robustness to down sampling (Figure S1), the absence of maternal whole-genome sequencing data prevented direct validation of QUILT2 imputation accuracy within our Vietnamese cohort. Despite these constraints, our classifier demonstrated consistent performance across independent discovery datasets, indicating reasonable generalizability within the target population. Literature validation of our five genome-wide significant omics traits provides compelling biological support for their roles in GDM pathogenesis. NEK7 , a critical regulator of NLRP3 inflammasome activation, shows upregulated expression in GDM pregnancies and correlates with glycemic control markers and systemic inflammation (Kumar et al., 2025). Its expression is further elevated by high glucose exposure, directly linking NEK7 to the inflammatory cascade characteristic of GDM. HSD11B1 , encoding 11β-hydroxysteroid dehydrogenase type 1, exhibits dysregulated placental expression in GDM, affecting local cortisol metabolism and contributing to the insulin resistance central to disease pathogenesis (Konstantakou et al., 2017; Ondřejíková et al., 2021). COMMD10’s involvement in membrane trafficking and cellular signaling includes modulation of basal insulin secretion, with variants linked to metabolic dysfunction and diabetic vascular complications (L. Zhang et al., 2023). KLRC4 , encoding a natural killer cell receptor, harbors deleterious variants associated with type 2 diabetes in large-scale genetic studies, suggesting immune-mediated metabolic regulation pathways. OCEL1 has been consistently identified in genomic screens as a type 2 diabetes susceptibility candidate, though its specific mechanistic role requires further investigation. The convergence of statistical significance and biological plausibility for these genes strengthens confidence in our omics-based approach and suggests that the broader set of 2,000 omics traits likely captures additional relevant biological pathways, even if individual effects are modest. Our sensitivity analysis illuminates important principles for polygenic disease prediction that extend beyond GDM. While only five omics traits achieved genome-wide significance after multiple testing correction, models incorporating 2,000 nominally significant traits substantially outperformed those built exclusively from statistically significant features. This finding underscores the value of polygenic methods that aggregate modest effects across many loci. The complex polygenic landscape of GDM likely involves thousands of genes contributing small individual effects through interconnected expression networks. Our omics atlas approach effectively reduces the search space from approximately eight million germline SNPs to ∼19,000 interpretable molecular traits, providing a more tractable and biologically meaningful framework for polygenic prediction. This methodology may prove valuable for other complex diseases with substantial polygenic components, including type 2 diabetes, cardiovascular disease, and cancer. The risk stratification framework developed here offers a practical approach for clinical decision-making, with four well-defined risk categories spanning low (0-10%) to severe (70-90%) GDM probability. Setting a threshold at 0.4 enables identification of 78% of future GDM cases while maintaining an acceptable false positive rate of 18% among controls. This balance appears suitable for a screening context where the goal is early identification to enable preventive interventions. Women classified as moderate risk (scores 0.3-0.5) might benefit from enhanced nutritional counseling, lifestyle modifications, and more frequent monitoring. Those at high risk (scores 0.5-0.8) could warrant earlier glucose tolerance testing, intensified dietary interventions, and consideration of metformin or other pharmacological approaches. The severe-risk women (scores >0.8) might benefit from immediate endocrinology consultation and aggressive management strategies typically reserved for established diabetes. Importantly, this prediction model was designed for apparently healthy pregnant women at 10-12 weeks of gestation and should be interpreted as a probabilistic risk assessment rather than a diagnostic tool. Women classified as high-risk may benefit from preventive interventions that could potentially modify their trajectory and prevent GDM development by 24-28 weeks, when traditional diagnosis occurs. Vietnam’s exceptionally high GDM prevalence (21%) may reflect unique genetic susceptibility patterns, environmental factors, or gene-environment interactions specific to Southeast Asian populations. Our population-specific model addresses this need while establishing a framework that could be adapted to other high-risk populations with appropriate validation studies. Future research directions should include: (1) prospective validation in independent Vietnamese cohorts to confirm clinical utility; (2) investigation of the model’s performance in other Southeast Asian populations to assess broader generalizability; (3) integration of additional data modalities including metabolomics, proteomics, and detailed clinical parameters; (4) development of interventional strategies tailored to different risk categories; and (5) cost-effectiveness analyses to guide implementation decisions. In conclusion, the framework established in this study demonstrates the potential for precision medicine approaches in maternal-fetal medicine, where early identification of high-risk pregnancies could enable targeted interventions to prevent GDM and its associated complications. As sequencing costs continue to decline and omics technologies become more accessible, such multi-modal prediction models may become increasingly feasible for routine clinical implementation, potentially transforming prenatal care from a reactive to a proactive, personalized approach. Author Contributions The core writing group (DNV, PMD, HG) designed the study, coordinated the multicenter collaboration, performed the data analysis, and drafted the manuscript. All named authors substantially contributed to the interpretation of results, critical revision of the manuscript, and approved the final version for submission. Collaborating investigators at participating sites were responsible for patient recruitment and data collection. The lead author had full access to all study data and takes responsibility for the integrity of the data and the accuracy of the analysis. Funding This study was funded by Gene Solutions, Vietnam. The funder did not have any additional role in the study design, data collection and analysis, decision to publish, or preparation of the manuscript. Data Availability The data that support the findings of this study are available from the corresponding author upon reasonable request. Competing Interests DNV, HHP, TTTD, SHT, HNN, PMD, and HG are employees of Gene Solutions, Vietnam. The other authors declare no competing interests. Ethics Statement Ethical approval for this study was obtained from the appropriate institutional review board. Written informed consent was secured from all participating subjects prior to inclusion in the study. All procedures were performed in compliance with relevant guidelines and regulations to protect the rights and privacy of human participants. Acknowledgment We gratefully acknowledge the guidance and valuable contributions of Robert W. Davies and Zilong Li, co-authors of the QUILT2 algorithm, in the development and execution of imputation procedures. Reference ↵ Buchanan , T. A. , Xiang , A. H. , & Page , K. A . ( 2012 ). Gestational diabetes mellitus: risks and management during and after pregnancy . Nature Reviews Endocrinology , 8 ( 11 ), 639 – 649 . doi: 10.1038/nrendo.2012.96 OpenUrl CrossRef PubMed ↵ Campello , R. J. G. B. , Moulavi , D. , & Sander , J . ( 2013 ). Density-Based Clustering Based on Hierarchical Density Estimates (pp. 160 – 172 ). doi: 10.1007/978-3-642-37456-2_14 OpenUrl CrossRef Global Prevalence of Gestational Diabetes (GDM) | IDF Atlas . (n.d.). Retrieved July 26, 2025, from https://diabetesatlas.org/data-by-indicator/hyperglycaemia-in-pregnancy-hip-20-49-y/prevalence-of-gestational-diabetes-mellitus-gdm/ ↵ Hollmann , N. , Müller , S. , Eggensperger , K. , & Hutter , F . ( 2023 ). TabPFN: A Transformer That Solves Small Tabular Classification Problems in a Second . http://arxiv.org/abs/2207.01848 ↵ Hollmann , N. , Müller , S. , Purucker , L. , Krishnakumar , A. , Körfer , M. , Hoo , S. Bin , Schirrmeister , R. T. , & Hutter , F. ( 2025 ). Accurate predictions on small data with a tabular foundation model . Nature , 637 ( 8045 ), 319 – 326 . doi: 10.1038/s41586-024-08328-6 OpenUrl CrossRef PubMed ↵ Kim , S. K. , Hannum , G. , Geis , J. , Tynan , J. , Hogg , G. , Zhao , C. , Jensen , T. J. , Mazloom , A. R. , Oeth , P. , Ehrich , M. , van den Boom , D. , & Deciu , C. ( 2015 ). Determination of fetal DNA fraction from the plasma of pregnant women using sequence read counts . Prenatal Diagnosis , 35 ( 8 ), 810 – 815 . doi: 10.1002/pd.4615 OpenUrl CrossRef PubMed ↵ Li , Z. , Albrechtsen , A. , & Davies , R. W . ( 2024 ). Rapid and accurate genotype imputation from low coverage short read, long read, and cell free DNA sequence . doi: 10.1101/2024.07.18.604149 OpenUrl Abstract / FREE Full Text ↵ Sheiner , E . ( 2020 ). Gestational Diabetes Mellitus: Long-Term Consequences for the Mother and Child Grand Challenge: How to Move on Towards Secondary Prevention? Frontiers in Clinical Diabetes and Healthcare , 1 . doi: 10.3389/fcdhc.2020.546256 OpenUrl CrossRef PubMed ↵ Tang , Z. , Wang , S. , Li , X. , Hu , C. , Zhai , Q. , Wang , J. , Ye , Q. , Liu , J. , Zhang , G. , Guo , Y. , Su , F. , Liu , H. , Guan , L. , Jiang , C. , Chen , J. , Li , M. , Ren , F. , Zhang , Y. , Huang , M. , … Liu , G . ( 2024 ). Longitudinal integrative cell-free DNA analysis in gestational diabetes mellitus . Cell Reports Medicine , 5 ( 8 ), 101660 . doi: 10.1016/j.xcrm.2024.101660 OpenUrl CrossRef PubMed ↵ Tran , N. H. , Vo , T. B. , Nguyen , V. T. , Tran , N.-T. , Trinh , T.-H. N. , Pham , H.-A. T. , Dao , T. H. T. , Nguyen , N. M. , Van , Y.-L. T. , Tran , V. U. , Vu , H. G. , Bui , Q.-T. N. , Vo , P.-A. N. , Nguyen , H. N. , Nguyen , Q.-T. T. , Do , T.-T. T. , Lam , N. V. , Ngoc , P. C. T. , Truong , D. K. , … Phan , M.-D . ( 2020 ). Genetic profiling of Vietnamese population from large-scale genomic analysis of non-invasive prenatal testing data . Scientific Reports , 10 ( 1 ), 19142 . doi: 10.1038/s41598-020-76245-5 OpenUrl CrossRef PubMed ↵ Wang , H. , Li , N. , Chivese , T. , Werfalli , M. , Sun , H. , Yuen , L. , Hoegfeldt , C. A. , Elise Powe , C. , Immanuel , J. , Karuranga , S. , Divakar , H. , Levitt , Na ., Li , C. , Simmons , D. , & Yang , X . ( 2022 ). IDF Diabetes Atlas: Estimation of Global and Regional Gestational Diabetes Mellitus Prevalence for 2021 by International Association of Diabetes in Pregnancy Study Group’s Criteria . Diabetes Research and Clinical Practice , 183 , 109050 . doi: 10.1016/j.diabres.2021.109050 OpenUrl CrossRef PubMed ↵ Wang , Y. , Sun , P. , Zhao , Z. , Yan , Y. , Yue , W. , Yang , K. , Liu , R. , Huang , H. , Wang , Y. , Chen , Y. , Li , N. , Feng , H. , Li , J. , Liu , Y. , Chen , Y. , Shen , B. , Zhao , L. , & Yin , C . ( 2023 ). Identify gestational diabetes mellitus by deep learning model from cell-free DNA at the early gestation stage . Briefings in Bioinformatics , 25 ( 1 ). doi: 10.1093/bib/bbad492 OpenUrl CrossRef ↵ Wicklow , B. , & Retnakaran , R . ( 2023 ). Gestational Diabetes Mellitus and Its Implications across the Life Span . Diabetes & Metabolism Journal , 47 ( 3 ), 333 – 344 . doi: 10.4093/dmj.2022.0348 OpenUrl CrossRef PubMed ↵ Xu , Y. , Ritchie , S. C. , Liang , Y. , Timmers , P. R. H. J. , Pietzner , M. , Lannelongue , L. , Lambert , S. A. , Tahir , U. A. , May-Wilson , S. , Johansson , Å. , Surendran , P. , Nath , A. P. , Persyn , E. , Peters , J. E. , Oliver-Williams , C. , Deng , S. , Prins , B. , Foguet , C. , Luan , J. , … Inouye , M . ( 2022 ). An atlas of genetic scores to predict multi-omic traits . doi: 10.1101/2022.04.17.488593 OpenUrl Abstract / FREE Full Text ↵ Zhang , M. , & Yang , H . ( 2022 ). Perspectives from metabolomics in the early diagnosis and prognosis of gestational diabetes mellitus . Frontiers in Endocrinology , 13 . doi: 10.3389/fendo.2022.967191 OpenUrl CrossRef ↵ Zhu , Y. , Barupal , D. K. , Ngo , A. L. , Quesenberry , C. P. , Feng , J. , Fiehn , O. , & Ferrara , A . ( 2022 ). Predictive Metabolomic Markers in Early to Mid-pregnancy for Gestational Diabetes Mellitus: A Prospective Test and Validation Study . Diabetes , 71 ( 8 ), 1807 – 1817 . doi: 10.2337/db21-1093 OpenUrl CrossRef PubMed View the discussion thread. Back to top Previous Next Posted September 05, 2025. Download PDF Supplementary Material Data/Code Email Thank you for your interest in spreading the word about medRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following Early Prediction of Gestational Diabetes Using Integrated Cell-free DNA Features and Omics-derived Genetic Scores Message Subject (Your Name) has forwarded a page to you from medRxiv Message Body (Your Name) thought you would like to see this page from the medRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share Early Prediction of Gestational Diabetes Using Integrated Cell-free DNA Features and Omics-derived Genetic Scores Vinh Nguyen Dao , Nhat-Thang Tran , Ta-Son Vo , Hong-Thinh Le , Thu-Ha Thi Nguyen , Quoc-Huy Vu Nguyen , Minh-Thi Thi Ha , Tam Minh Le , Diem-Tuyet Thi Hoang , Khanh-Trang Nguyen Huynh , Nhan Viet Nguyen , Chuong Canh Nguyen , Thuong Chi Bui , Xuan Thanh Nguyen , Sa Viet Le , Vinh Dinh Tran , My-Nhi Ba Nguyen , Thong Van Nguyen , Tuyet-Anh Thi Nguyen , Ba Phuoc Hoang , Trong Van Nguyen , Thuy-Ai Thuy Nguyen , Toa Tri Nguyen , Thang Duc Duong , Cuong Huy Pham , Kim-Oanh Thi Luong , Cuong Ngoc Dao , Khanh Van Hoang , Thu-Thanh Thi Huynh , Khuong Manh Nguyen , Son-Tra Thi Tran , Hoanh Trung Tran , Son Canh Nguyen , Thuy Dinh Tran , Phương Thi Lan Nguyen , Thanh Viet Pham , Kong Chi Pham , Minh Doan Thai , My-Hang Thi Truong , Hieu Ha Pham , Thanh-Thuy Thi Do , Sang Hung Tang , Hoai-Nghia Nguyen , Minh-Duy Phan , Hoa Thi Dao , Hoa Giang medRxiv 2025.09.03.25334985; doi: https://doi.org/10.1101/2025.09.03.25334985 Share This Article: Copy Citation Tools Early Prediction of Gestational Diabetes Using Integrated Cell-free DNA Features and Omics-derived Genetic Scores Vinh Nguyen Dao , Nhat-Thang Tran , Ta-Son Vo , Hong-Thinh Le , Thu-Ha Thi Nguyen , Quoc-Huy Vu Nguyen , Minh-Thi Thi Ha , Tam Minh Le , Diem-Tuyet Thi Hoang , Khanh-Trang Nguyen Huynh , Nhan Viet Nguyen , Chuong Canh Nguyen , Thuong Chi Bui , Xuan Thanh Nguyen , Sa Viet Le , Vinh Dinh Tran , My-Nhi Ba Nguyen , Thong Van Nguyen , Tuyet-Anh Thi Nguyen , Ba Phuoc Hoang , Trong Van Nguyen , Thuy-Ai Thuy Nguyen , Toa Tri Nguyen , Thang Duc Duong , Cuong Huy Pham , Kim-Oanh Thi Luong , Cuong Ngoc Dao , Khanh Van Hoang , Thu-Thanh Thi Huynh , Khuong Manh Nguyen , Son-Tra Thi Tran , Hoanh Trung Tran , Son Canh Nguyen , Thuy Dinh Tran , Phương Thi Lan Nguyen , Thanh Viet Pham , Kong Chi Pham , Minh Doan Thai , My-Hang Thi Truong , Hieu Ha Pham , Thanh-Thuy Thi Do , Sang Hung Tang , Hoai-Nghia Nguyen , Minh-Duy Phan , Hoa Thi Dao , Hoa Giang medRxiv 2025.09.03.25334985; doi: https://doi.org/10.1101/2025.09.03.25334985 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Endocrinology (including Diabetes Mellitus and Metabolic Disease) Subject Areas All Articles Addiction Medicine (567) Allergy and Immunology (863) Anesthesia (297) Cardiovascular Medicine (4411) Dentistry and Oral Medicine (443) Dermatology (380) Emergency Medicine (606) Endocrinology (including Diabetes Mellitus and Metabolic Disease) (1505) Epidemiology (15205) Forensic Medicine (30) Gastroenterology (1119) Genetic and Genomic Medicine (6574) Geriatric Medicine (666) Health Economics (994) Health Informatics (4511) Health Policy (1365) Health Systems and Quality Improvement (1608) Hematology (537) HIV/AIDS (1263) Infectious Diseases (except HIV/AIDS) (15903) Intensive Care and Critical Care Medicine (1103) Medical Education (620) Medical Ethics (144) Nephrology (666) Neurology (6573) Nursing (345) Nutrition (998) Obstetrics and Gynecology (1139) Occupational and Environmental Health (954) Oncology (3319) Ophthalmology (968) Orthopedics (369) Otolaryngology (420) Pain Medicine (435) Palliative Medicine (129) Pathology (662) Pediatrics (1689) Pharmacology and Therapeutics (691) Primary Care Research (710) Psychiatry and Clinical Psychology (5422) Public and Global Health (9205) Radiology and Imaging (2191) Rehabilitation Medicine and Physical Therapy (1367) Respiratory Medicine (1191) Rheumatology (593) Sexual and Reproductive Health (709) Sports Medicine (529) Surgery (709) Toxicology (99) Transplantation (288) Urology (265) (function(){function c(){var b=a.contentDocument||a.contentWindow.document;if(b){var d=b.createElement('script');d.innerHTML="window.__CF$cv$params={r:'9fe96d79083b41e2',t:'MTc3OTI2MDA0MQ=='};var a=document.createElement('script');a.src='/cdn-cgi/challenge-platform/scripts/jsd/main.js';document.getElementsByTagName('head')[0].appendChild(a);";b.getElementsByTagName('head')[0].appendChild(d)}}if(document.body){var a=document.createElement('iframe');a.height=1;a.width=1;a.style.position='absolute';a.style.top=0;a.style.left=0;a.style.border='none';a.style.visibility='hidden';document.body.appendChild(a);if('loading'!==document.readyState)c();else if(window.addEventListener)document.addEventListener('DOMContentLoaded',c);else{var e=document.onreadystatechange||function(){};document.onreadystatechange=function(b){e(b);'loading'!==document.readyState&&(document.onreadystatechange=e,c())}}}})();

Text is read by the "Ask this paper" AI Q&A widget below. Extraction quality varies by source — PMC NXML preserves structure cleanly, OA-HTML may include some navigation residue, and OA-PDF can have broken hyphenation. The publisher copy (via DOI) is the canonical version.

My notes (saved in your browser only)

Ask this paper AI returns verbatim quotes from the full text · source: preprint-html

Answers must be backed by verbatim quotes from this paper's full text. Hallucinated quotes are dropped automatically; if no verbatim passage answers the question, we say so. How this works

Citation neighborhood (no data yet)

We don't have any in-corpus citations linked to this paper yet. This is a recent paper (2025) — citers typically take a year or two to land, and the OpenAlex reference graph may still be filling in.

Source provenance

europepmc
last seen: 2026-05-20T01:45:00.602351+00:00
unpaywall
last seen: 2026-05-27T02:00:06.600101+00:00
License: CC-BY-NC-ND-4.0