Full text
48,057 characters
· extracted from
preprint-html
· click to expand
Development and Validation of Machine Learning-Based Prediction of Depression Progression Using EHR Data: A Multi-Institutional Retrospective Cohort Study | medRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-P4HH5NV'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search Development and Validation of Machine Learning-Based Prediction of Depression Progression Using EHR Data: A Multi-Institutional Retrospective Cohort Study View ORCID Profile Pegah Ahadian , View ORCID Profile Angela Fragano , View ORCID Profile Tianyuan Guan , View ORCID Profile Qiang Guan , View ORCID Profile Sophia Z. Shalhout doi: https://doi.org/10.1101/2025.11.28.25341207 Pegah Ahadian 1 Department of Computer Science, Kent State University , Kent, Ohio, USA MS Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Pegah Ahadian Angela Fragano 2 Department of Otolaryngology-Head and Neck Surgery, Division of Surgical Oncology, Mike Toth Cancer Research Center, Massachusetts Eye and Ear , Boston, MA, USA BSc Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Angela Fragano Tianyuan Guan 3 College of Public Health, Kent State University , Kent, Ohio, USA MD, MPH, PhD Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Tianyuan Guan Qiang Guan 1 Department of Computer Science, Kent State University , Kent, Ohio, USA PhD Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Qiang Guan For correspondence: sophia_shalhout{at}meei.harvard.edu qguan{at}kent.edu Sophia Z. Shalhout 2 Department of Otolaryngology-Head and Neck Surgery, Division of Surgical Oncology, Mike Toth Cancer Research Center, Massachusetts Eye and Ear , Boston, MA, USA 4 Department of Otolaryngology-Head and Neck Surgery, Harvard Medical School , Boston, MA, USA PhD Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Sophia Z. Shalhout For correspondence: sophia_shalhout{at}meei.harvard.edu qguan{at}kent.edu Abstract Full Text Info/History Metrics Data/Code Preview PDF ABSTRACT Background Depression is a leading cause of global disability. Timely identification of patients at risk for clinical worsening remains a major challenge. Electronic health records (EHRs) facilitate large-scale, real-world analyses of disease trajectories. However, standardized symptom scale data such as the Patient Health Questionnaire-9 are often unavailable or recorded only as unstructured text. In this context, International Classification of Diseases (ICD10) diagnostic-code based severity progression provides a pragmatic alternative for developing predictive tools to identify worsening depression. Objective We aim to develop and evaluate machine-learning and deep-learning models for predicting ICD10-defined progression from mild to moderate/severe depression using EHR data curated by the MedStar Health Research Institute (MHRI). Methods We conducted a multi-institutional retrospective cohort analysis using the MHRI EHR database, which integrates data from 10 hospitals and 300 outpatient sites across the mid-Atlantic. Adults (≥18 years) with an initial ICD10 diagnosis of mild depression between 2017 and 2023 were included (N=2131). Nonprogressors were defined as patients whose mild major depressive disorder remained mild for 24 months (N=270). Progressors were defined as patients who developed moderate or severe ICD10 depression within 24 months of the index diagnosis (N=533). Data were stratified and split into (60%) training, (20%) validation, and (20%) test subsets. A heterogeneous feature set spanning demographics, healthcare utilization, socioeconomic indices, diagnostic context, and laboratory measurements were available. Logistic regression utilized elastic net regularization with fivefold cross validation, and random forest hyperparameters were tuned by grid search. XGBoost, CatBoost, and a deep neural network (DNN) were trained with standard learning rate, depth, class weighting, and early stopping. A deterministic top model selection framework applied prespecified thresholds of sensitivity at least 0.70 and AUC at least 0.70, and composite rankings integrated accuracy, sensitivity, specificity, and the overfitting gap. Results The analytic cohort included 803 patients with complete two-year follow-up. Under the selection criteria, the DNN failed to meet the AUC threshold (0.671) and was excluded. Among the remaining models, XGBoost achieved the top composite score (accuracy = 0.72; AUC = 0.776; sensitivity = 0.77; specificity = 0.63; overfit gap = 0.112). Logistic regression ranked second (accuracy = 0.71; AUC = 0.797; sensitivity = 0.79; specificity = 0.61; overfit gap = 0.052), followed by CatBoost and random forest, the latter penalized for overfitting (gap = 0.278). The TinyLlama audit note, generated through a local Hugging Face pipeline, confirmed XGBoost as the most balanced model. Conclusions Using EHR data from a multi-institutional regional health system, we developed and validated machine-learning models that predicted progression of depression. XGBoost demonstrated the most reliable composite performance. These findings support the feasibility of leveraging socioeconomic and EHR data to predict worsening depression and emphasize the importance of transparent model-selection frameworks for trustworthy clinical artificial intelligence. INTRODUCTION Depression is one of the largest burdens of disease worldwide. 1 Globally, the prevalence of major depressive disorder (MDD) continues to increase and ranks among the highest causes of disability-adjusted life years (DALYs) and years lived with disability (YLDs). 2 Unfortunately, a notable acceleration of this trend was observed due to the COVID-19 pandemic, which was associated with a substantial increase in new and persistent depressive episodes worldwide. 3 The widespread adoption of electronic health records (EHRs) has enabled large-scale, real-world investigations of mental health and clinical trajectories of depression. Multiple systematic reviews highlight the growing utility of artificial intelligence/machine learning (AI/ML) and deep learning (DL) methods applied to EHR data for the diagnosis, risk stratification, and prognosis of depressive disorders. 4 , 5 , 6 In addition, utilizing the International Classification of Diseases (ICD10) depression diagnostic codes in EHRs has demonstrated the feasibility of identifying depression phenotypes and the onset or recurrence of depression. 4 , 7 However, several gaps remain. While many studies focus on onset or presence of depression, relatively few have addressed progression of depression defined as the transition from mild to moderate or severe disease states, especially using longitudinal EHR data with ICD10 severity coding. An important unmet need in mental-health care is the development of predictive models that enable early identification of patients whose depression will progress. Timely intervention by providers can help prevent chronicity, reduce functional disability and suicide risk, and lower health-care utilization and costs. 8 , 9 However, many large-scale EHR datasets lack complete standardized depression symptom-scale measures such as the Patient Health Questionnaire-9 (PHQ-9) which can indicate progression of depression. PHQ-9 outcomes are often unstructured, incomplete, unavailable, or inconsistently documented, limiting their use for predictive modeling. To address this limitation, investigators have increasingly turned to ICD diagnostic codes as pragmatic proxies for depression severity and longitudinal progression. Prior studies have demonstrated that ICD-coded diagnosis sequences can approximate symptom-based trajectories and facilitate large-scale phenotyping when structured depression survey data are lacking. 10 Furthermore, validation studies have shown that ICD10 diagnostic codes can reliably identify depression and track its progression in administrative or EHR data, achieving high specificity (>99%) and positive predictive value (∼90%), thereby providing a robust alternative to symptom-scale instruments like the PHQ-9 when such data are sparse or unavailable. 11 Together, these findings suggest that ICD-based phenotyping provides a reliable and scalable foundation for developing predictive models of depression progression, particularly in real-world datasets where standardized symptom instruments are incomplete. In this context, our study leverages data from the MedStar Health Research Institute (MHRI) EHR repository to develop and evaluate machine-learning and deep-learning models for predicting ICD10-defined progression from mild to moderate/severe depression within two years of mild MDD diagnosis. By focusing on severity progression within 24 months in a real-world, multi-institutional EHR cohort and applying a transparent model-selection framework, our work aims to extend the literature on depression prognosis and EHR-driven predictive modeling for early intervention applications. METHODS Study cohort and outcomes We performed a retrospective cohort study approved by the Kent State University and Mass General Brigham Institutional Review Boards under a Data Usage Agreement with MHRI. We conducted a multi-institutional analysis using the MHRI EHR database, which integrates longitudinal patient data from ten hospitals and more than 300 affiliated outpatient facilities across the mid-Atlantic region of the United States. The MHRI database includes structured and semi-structured data encompassing demographics, laboratory test values, Social Vulnerability Index (SVI) and Area Deprivation Index (ADI) features, encounter records, ICD10 diagnostic codes, and procedures. Adult patients aged 18 years or older with at least one MDD ICD10 diagnosis recorded between January 1, 2017 and December 31, 2023 were identified (N=47,292) in MHRI. Subjects with an initial qualifying ICD10 diagnosis of mild MDD (F32.0 or F33.0) and at least two depression-related encounters (F32.0, F32.1, F32.2, F32.3, F33.0, F33.1, F33.2, F33.3) within two years of the index mild episode were identified. To ensure sufficient longitudinal follow-up, patients were required to have a minimum of two years of continuous EHR activity following their index diagnosis. A total of 2,131 patients met these initial inclusion criteria. The target outcomes were defined by ICD10-coded severity depression progression over 24 months. Nonprogressors were defined as patients whose diagnostic trajectory remained limited to mild MDD codes throughout the 24-month observation window, without any recorded moderate, severe, or unspecified MDD codes, nor any diagnostic codes suggestive of bipolar disorder, psychotic depression, or other affective disorders that could confound depression severity classification. Progessors were defined as patients who developed moderate or severe MDD (F32.1, F32.2, F33.1, F33.2, F33.3) within 24 months of the mild index diagnosis. Patients lacking an initial mild MDD diagnosis, and/or continuous follow-up were excluded. The final analytic cohort included 803 patients who met all inclusion criteria. This framework operationalizes ICD10-encoded severity change as a reproducible marker of clinically meaningful worsening. Predictor Variables From the MHRI EHR database, 292 candidate features were extracted. These included demographic, and socioeconomic variables (age, sex, race, ethnicity, insurance, zip code, ADI, SVI), encounter-level utilization metrics (total encounters, encounter type, reason for visit, admission type), laboratory test results (metabolic, hematologic, thyroid, endocrine, inflammatory panels, chemistry, specialty assays) and comorbidity indicators for common chronic conditions (hypertension, diabetes, chronic kidney disease, asthma, chronic obstructive pulmonary disease, heart failure, coronary artery disease, liver disease, arthritis, cancer, epilepsy, multiple sclerosis, Parkinson’s disease, Alzheimer’s disease, and dysthymia). This multi-domain feature offers a comprehensive representation of each patient’s biological, clinical, and contextual risk profile for depression progression. 12 Continuous missing values were imputed using the mean. Categorical missing values were imputed using the most frequent category. Categorical features were one-hot encoded, and continuous features were standardized using Z-score normalization. Feature Selection Data were partitioned using stratified sampling into 60 percent training, 20 percent validation, and 20 percent test subsets. Feature selection was conducted exclusively within the training subset to avoid information leakage. Interaction-only polynomial feature construction and univariate screening reduced the initial set to the top 100 features for the Logistic Regression (LR) model. A Random Forest (RF) model trained on the training split provided additional feature ranking. Fixed random seeds and the predefined split ensured reproducibility. Stability checks using alternative importance measures showed consistent ranking among the strongest predictors. The final feature set contained 20 predictors. Development of AI and ML Models Five supervised binary classifiers were trained to predict progression including LR, RF, XGBoost, CatBoost, and a DNN. The training split was used for model fitting, and the validation split guided hyperparameter tuning. LR used elastic net regularization with five-fold cross-validation (CV). RF tuning explored tree number, depth, and split and leaf thresholds. XGBoost and CatBoost were trained with fixed learning rate, depth, subsampling, class weighting, and early stopping or calibration on the validation split. The DNN consisted of four fully connected layers with batch normalization, dropout, Adam optimization, and early stopping. Final performance was evaluated exclusively on the held-out test data split. Evaluation Metrics and Model Selection Evaluation metrics included accuracy, area under the receiver operating characteristic curve (AUC), sensitivity, specificity, and F1-score. To prevent overfitting and ensure transparent ‘best’ model selection, a deterministic selection framework was applied, requiring models to meet prespecified performance thresholds of AUC ≥ 0.70 and sensitivity ≥ 0.70. Models that failed to meet these criteria were excluded from final comparison. For models meeting threshold criteria, a composite ranking score was computed. This score integrates accuracy, sensitivity, specificity, and the overfitting gap, defined as the difference in AUC between training and test sets. The ranking approach prioritized models with balanced discrimination and generalization performance. Model interpretability and explainability were further assessed by generating SHapley Additive exPlanations (SHAP) feature importance plots for the top three models and by performing a TinyLlama audit via a local Hugging Face pipeline to confirm reproducibility of results. Code and Analysis All preprocessing and predictive models were implemented in Python 3.13.2 using pandas, NumPy, scikit-learn, TensorFlow, Matplotlib, and Seaborn, with all dependencies managed via pip 25.2. The AI ReAct-style judge and audit note were run in the same Python 3.13.2 environment, using the Hugging Face transformers stack to call TinyLlama/TinyLlama-1.1B-Chat-v1.0 through a CPU-only text-generation pipeline. All preprocessing, model-training, and evaluation steps adhered to the Transparent Reporting of a Multivariable Prediction Model for Individual Prognosis or Diagnosis (TRIPOD+AI) guidelines. 13 RESULTS Participant Characteristics A total of 47,292 adult patients with any depression diagnosis were identified in the MHRI EHR database between 2017 and 2023. Of these, 2,172 patients had an initial diagnosis of mild depression. After restricting patients to those with at least two subsequent depression-related encounters recorded within two years of their initial mild episode, 2,131 patients remained eligible for analysis. The study cohort selection process is shown in Figure 1 . Patients were classified into two outcome groups based on longitudinal ICD10 coding. Cohort 1 (Nonprogessors) were patients whose follow-up codes remained mild throughout the two-year observation window (N = 270). Cohort 2 (Progressors) were patients who progressed to moderate or severe MDD within two years (N = 533) of the index mild depression diagnosis. The final analytic cohort included 803 patients, representing those with feature availability and continuous EHR activity. The final dataset was divided into training (60 %), validation (20 %), and test (20 %) splits using stratified sampling to preserve class proportions. Baseline demographic, and socioeconomic characteristics of each cohort are summarized in Table 1 . Progressors and nonprogressors were predominantly female, consistent with the known higher prevalence of depression in women. 14 Both groups showed broadly similar age and race distributions. Across social vulnerability measures (SVI themes and ADI national rank) and healthcare utilization (total encounters), the variables displayed large standard deviations with substantial overlap, indicating that these characteristics were not strongly differentiated between groups on an absolute basis. Download figure Open in new tab Figure 1: Study Cohorts. Flow diagram illustrating sequential inclusion criteria applied to the MHRI EHR dataset. From 47,292 adults with any depression diagnosis, 2,172 individuals had an initial ICD10 diagnosis of mild MDD. Of these, 2,131 had at least two follow-up depression-related encounters within two years. Patients were classified into ‘Nonprogressors’ (mild-only follow-up codes; N = 270) and ‘Progressors’ (progressed to moderate or severe ICD-10 depression; N = 533). The final study cohort (N = 803) was stratified and divided into training (60 percent), validation (20 percent), and held out test (20 percent) sets for model development and evaluation. View this table: View inline View popup Download powerpoint Table 1: Patient Characteristics Model Training and Evaluation The LR, RF, XGBoost, CatBoost, and DNN were trained on selected features. Hyperparameter tuning on the validation split yielded stable convergence. Early stopping was used for XGBoost, CatBoost, and the DNN, while LR and RF hyperparameters were optimized using cross-validation. Deterministic preprocessing, fixed random seeds, and a predefined data split were employed to ensure reproducibility. Models were evaluated against prespecified performance thresholds of AUC ≥ 0.70 and sensitivity ≥ 0.70. The evaluator loads the metrics per-split, applies our chosen policy on test sensitivity and AUC, rejects models that are considered ‘violators’ with explicit reasons, then scores the remaining models using a weighted composite of AUC, F1-weighted, specificity, accuracy, and an overfit penalty based on the train-test AUC gap. Finally, models are sorted and ranked by total score, emitting a ranked report. In addition, a brief rationale contrasting the models ranked as #1 and #2 is provided. The large language model (LLM) produces a concise audit note; its prompt contains the fixed policy and the top-k results with exact metrics, with a brief explanation of why the model ranked first is chosen and how the model ranked second compares. Under these settings, the DNN did not meet the AUC requirement (AUC = 0.671 on the test set) and was excluded from further comparison. The remaining LR, CatBoost, XGBoost and RF models exceeded threshold criteria and were retained for composite ranking. Test-set performance metrics are summarized in Table 2 . XGBoost demonstrated the most balanced performance (accuracy = 0.72, AUC = 0.78, sensitivity = 0.77, specificity = 0.63) and the train, validation, and test AUC curves for the XGBoost model ( Figure 2A ) illustrate moderate overfitting consistent with a train-test AUC gap of 0.11. The corresponding test-set confusion matrix ( Figure 2B ) demonstrates moderate sensitivity and specificity. Logistic regression achieved the highest AUC (0.80) with a sensitivity = 0.79. AUC curves for training, validation and test show minimal divergence, consistent with the small overfitting gap (0.05) ( Figure 3A ). The test-set confusion matrix for the LR model ( Figure 3B ) illustrates balanced classification performance across the two outcome classes. CatBoost demonstrated effective control of overfitting via ordered boosting and early stopping ( Figure 4A ) with similar accuracy (0.71) and sensitivity (0.72) ( Figure 4B ). Random forest produced high sensitivity (0.85) but lower specificity and a larger overfitting gap (0.28), reducing its composite ranking. XGBoost received the highest composite score (TOTAL = 70.36), followed by LR (TOTAL = 65.85), CatBoost (TOTAL = 65.73), and RF (TOTAL = 56.00). The audit note generated by the pipeline ( Figure 5 ) summarized the trade-offs between the top models, highlighting the stronger F1 and accuracy of XGBoost relative to logistic regression, and the sensitivity-specificity balance that distinguished the three surviving boosted or linear models. These rankings confirmed XGBoost as the top-scoring model and provided a concise explanation for its selection over the logistic regression model based on differences in weighted-F1, accuracy, and specificity profiles. Download figure Open in new tab Figure 2: XGBoost Model Performance. A. Receiver operating characteristic curves for the XGBoost classifier evaluated on the training, validation, and test splits. B. Test-set confusion matrix for XGBoost is depicted. The model correctly identified 33 true negatives and 78 true positives, with 18 false positives and 25 false negatives, reflecting the model’s observed test-set sensitivity (0.77) and specificity (0.63). Download figure Open in new tab Figure 3: Logistic Regression Model Performance. A. Receiver operating characteristic curves for the LR classifier evaluated on the training, validation, and test splits. The model achieved AUC = 0.85 on the training set, AUC = 0.86 on the validation set, and AUC = 0.80 on the independent test set, indicating stable generalization across splits. B. Test-set confusion matrix for LR model is shown. Download figure Open in new tab Figure 4: CatBoost Model Performance. A. The ROC curves display CatBoost performance across the training, validation, and test sets. The model achieved AUC = 0.86 on the training set, AUC = 0.66 on the validation set, and AUC = 0.76 on the test set, reflecting moderate generalization. B. The CatBoost classifier produced 35 true negatives, 16 false positives, 29 false negatives, and 74 true positives on the held-out test set. These results illustrate a balanced recall profile. Download figure Open in new tab Figure 5: Model Ranking and ReAct Audit Output. This figure presents the full output of the ReAct-style model evaluation and auditing pipeline used to rank classifiers. The top block lists the enforced hard constraints for model acceptance, including minimum sensitivity (recall) and minimum ROC-AUC. The middle section displays the ranked models that satisfied these constraints, with each model’s composite score, AUC, F1-weighted score, sensitivity, specificity, accuracy, overfitting gap, and test set size. The bottom section summarizes the audit rationale comparing the top-ranked models, explaining why XGBoost was selected as the preferred model and how logistic regression performed in comparison. View this table: View inline View popup Download powerpoint Table 2: Summary of Model Performance on Test Data Model Explainability The XGBoost SHAP summary plot ( Figure 6A ) highlights ‘Total Encounters’, ‘Age at Visit Group’, and ‘Number of Social Categories’ exerting the largest influence on predicted risk of progression. Several SVI-related features (RPL themes) demonstrate modest but directionally consistent effects, indicating that social vulnerability contributes incremental predictive signal within the model’s multivariate context. The SHAP distribution for logistic regression ( Figure 6B ) highlights ‘Age at Visit Group’, ‘Total Encounters’, ‘Immature Platelet Fraction’, and ‘Total Bilirubin’ as the strongest contributors to model output. The CatBoost SHAP summary plot ( Figure 6C ) identifies ‘Total Encounters’, ‘Diagnostic Priority’, and SVI-related variables (RPL Theme 1, RPL Theme 3, RPL Theme 4) as the most influential predictors. CatBoost produced more compact SHAP ranges than XGBoost, consistent with its smoother boosting procedure. Download figure Open in new tab Figure 6: SHAP Feature Importance Across Top Models. (A) XGBoost SHAP summary plot which depicts ‘Total encounters’ and ‘Age group at visit’ with the largest positive contributions to predicted progression, and multiple SVI-related variables (RPL_THEME1, RPL_THEME3) exerting smaller but consistent effects. Higher feature values (pink) generally increase predicted risk, while lower values (blue) reduce it. (B) Logistic Regression SHAP summary plot depicting ‘Age group at visit’ and ‘Total encounters’ as the dominate factors influencing the model. (C) CatBoost SHAP summary plot is shown. Predictions are driven by a blend of encounter frequency, diagnostic priority, RPL themes, and laboratory values. CatBoost displays tighter SHAP distributions, reflecting stronger regularization and monotonic handling of categorical features. DISCUSSION We evaluated machine-learning and deep-learning models to predict ICD10-defined progression from mild to moderate or severe MDD using real-world EHR data. Depression is a heterogeneous disorder influenced by biological, social, and healthcare-access factors. Early prediction and identification of high-risk patients remain a major clinical priority. 15 Our classification framework operationalized severity changes strictly through ICD10 codes, enabling reproducible labeling within large EHR systems despite the well-known absence or inconsistency of symptom-scale data such as PHQ-9. 16 – 18 Using a curated cohort of 803 adults, we demonstrated that structured socioeconomic and EHR data can provide meaningful predictive signal for clinically relevant worsening of depression. XGBoost achieved the most balanced performance across discrimination, calibration stability, and sensitivity-specificity trade-offs compared to the remaining models. Logistic regression achieved the highest AUC, likely retaining competitive performance due to strong and well-selected predictors. SHAP-based explainability analyses highlighted heterogeneous risk signatures distributed across laboratory markers, healthcare-utilization patterns, chronic comorbidities, and socioeconomic vulnerability indices. This aligns with our current understanding of depression as a multi-system disorder influenced by several factors. 12 Notably, SVI and ADI measures contributed non-trivially to model predictions, consistent with prior studies linking such socioeconomic factors with depression. 19 – 20 However, the large standard deviations and overlapping distributions across groups underscore that no single factor is sufficient to discriminate progression risk, at least in this cohort, reinforcing the need for multivariable modeling ( Table 1 ). Several important limitations must be emphasized. First, as previously mentioned, the dataset did not include standardized symptom scales such as PHQ-9, which limits granularity in measuring depressive severity. ICD10 codes provide an imperfect proxy for symptom-level severity, since coding decisions may vary by clinician, healthcare setting, and reimbursement policies. ICD10 misclassification bias, for example, may attenuate observed associations or obscure true predictive signal. Second, although the cohort was multi-institutional, the final analytic sample (N=803) reflects only patients with continuous two-year follow-up, potentially introducing selection bias toward patients with higher healthcare utilization. Third, EHR-derived features are influenced by patterns of care rather than purely underlying pathophysiology, and some predictors may reflect differential access or utilization rather than intrinsic risk such as ‘total hospital encounters’ and ‘insurance type’. In addition, a larger, external, multicenter cohort is required to test our models’ robustness and confirm that the performance observed, generalize across populations and EHR systems. While trained and evaluated on a multi-institutional cohort, MHRI is a single regional health system, which may limit external generalizability, particularly given geographic, socioeconomic, and practice-pattern differences across regions. We applied strict data-splitting and leakage controls. However, overfitting remains possible, as reflected by the overfitting gaps observed in our tree-based models. Finally, the study period overlapped with the COVID-19 pandemic, which introduced major disruptions to healthcare utilization, diagnostic coding patterns, and follow-up frequency. These system-level changes may have influenced both the exposure features, such as encounter counts, and the progression labels derived from ICD10 codes. As a result, some observed lack of progressions may partially reflect pandemic-related care patterns rather than true underlying clinical trajectory. Future models should assess performance in post-pandemic cohorts. 21 – 23 Despite these constraints, the study demonstrates that gradient-boosted tree approaches applied to socioeconomic and EHR data can identify individuals at risk for clinically coded worsening of depression with reasonable accuracy. The methodological framework, stratified splitting, leakage-free feature selection, calibration checks, standardized model auditing, and composite model ranking, aligns with recommended practices for clinical machine-learning transparency and reproducibility. 24 Future work should incorporate symptom-scale data, natural-language-processing extraction of narrative clinical assessments, and causal-inference frameworks to distinguish true clinical deterioration from healthcare-utilization artifacts. External validation across other health systems and post-pandemic cohorts will be essential for assessing transportability. Ultimately, integrating interpretable risk models into clinical workflows may support proactive monitoring for targeted intervention aimed at patients most likely to experience worsening depressive illness. CONCLUSIONS In this multi-institutional retrospective cohort study, we demonstrated that structured socioeconomic and EHR data can be used to predict ICD10-defined progression from mild to moderate or severe depression with meaningful accuracy. After standardized preprocessing, transparent data-splitting, and deterministic model selection, XGBoost achieved the strongest overall balance of discrimination, sensitivity, specificity, and generalization, while logistic regression provided the highest AUC and CatBoost offered competitive performance with a distinct operating profile. These findings show that depression-progression risk can be predicted from real-world clinical, laboratory, and socioeconomic features. Our results highlight the feasibility of building reproducible, interpretable machine-learning pipelines for mental-health risk stratification using EHR data. Although further validation in external health systems and prospective settings is needed, the approach provides a scalable framework for future predictive tools that could support earlier identification of patients at risk for worsening depressive illness. AUTHORS’ CONTRIBUTIONS Conceptualization: SZS, PA Data curation: SZS, PA Formal analysis: SZS, PA, AF, TG, QG Funding acquisition: SZS, PA, QG Investigation: SZS, PA Methodology: SZS, PA Project administration: SZS Resources: SZS, PA Supervision: SZS, QG Validation: SZS, PA Visualization: SZS, PA, AF Writing-original draft: SZS, AF, PA Writing-review and editing: SZS, AF, PA, TG, QG Funding Research reported in this publication was supported by the Office of the Director, National Institutes of Health Common Fund under award number 1OT2OD032581-01. The work is solely the responsibility of the authors and does not necessarily represent the official view of the National Institutes of Health. Conflict of Interest Statement The authors have no conflicts to report. Data Availability The MedStar Health Research Institute electronic health record data used in this study cannot be publicly shared due to institutional policy and patient-privacy regulations. Access to this data requires a formal Data Use Agreement with MedStar Health and approval from the relevant Institutional Review Boards. Researchers interested in accessing the underlying data should contact the MedStar Health Research Institute Research Administration. All analysis code and model scripts used in this study are available via GitHub. Acknowledgements This work was supported by the AIM-AHEAD Coordinating Center, funded by the NIH. The data used in this study were obtained from the MedStar Health Research Institute (MHRI) electronic health record (EHR) data warehouse. MedStar Health System includes an extensive network of clinical facilities in the mid-Atlantic region, including 10 hospitals (33% rural hospitals) and over 300 points-of-care connected by MedStar’s EHR system, built on the Cerner Millennium platform. The authors thank the MHRI teams for data curation, accession, and training. ABBREVIATIONS ADI Area Deprivation Index AI Artificial Intelligence AUC Area Under the Receiver Operating Characteristic Curve CPU Central Processing Unit CV Cross-Validation DALY Disability-Adjusted Life Year DNN Deep Neural Network EHR Electronic Health Record ICD10 International Classification of Diseases, Tenth Revision LLM Large Language Model LR Logistic Regression MDD Major Depressive Disorder MHRI MedStar Health Research Institute ML Machine Learning PHQ9 Patient Health Questionnaire-9 RF Random Forest ROC Receiver Operating Characteristic Curve SVI Social Vulnerability Index TRIPOD+AI Transparent Reporting of a Multivariable Prediction Model for Individual Prognosis or Diagnosis (Artificial Intelligence extension) YLD Years Lived With Disability REFERENCES 1. ↵ Chen X-D , Li F , Zuo H , Zhu F . Trends in Prevalent Cases and Disability-Adjusted Life-Years of Depressive Disorders Worldwide: Findings From the Global Burden of Disease Study From 1990 to 2021 . Depress Anxiety . 2025; 2025 : 5553491 . doi: 10.1155/da/5553491 OpenUrl CrossRef 2. ↵ Yan G , Zhang Y , Wang S , et al. Global, regional, and national temporal trend in burden of major depressive disorder from 1990 to 2019: An analysis of the global burden of disease study . Psychiatry Res . 2024 ; 337 : 115958 . doi: 10.1016/j.psychres.2024.115958 OpenUrl CrossRef PubMed 3. ↵ Wang Y , Qin C , Chen H , Liang W , Liu M , Liu J . Global, regional, and national burden of major depressive disorders in adults aged 60 years and older from 1990 to 2021, with projections of prevalence to 2050: Analyses from the Global Burden of Disease Study 2021 . J Affect Disord . 2025 ; 374 : 486 – 494 . doi: 10.1016/j.jad.2025.01.086 OpenUrl CrossRef PubMed 4. ↵ Nickson D , Meyer C , Walasek L , Toro C . Prediction and diagnosis of depression using machine learning with electronic health records data: a systematic review . BMC Med Inform Decis Mak . 2023 ; 23 ( 1 ): 271 . doi: 10.1186/s12911-023-02341-x OpenUrl CrossRef PubMed 5. ↵ Phiri D , Makowa F , Amelia VL , Phiri YVA , Dlamini LP , Chung M-H . Text-Based Depression Prediction on Social Media Using Machine Learning: Systematic Review and Meta-Analysis . J Med Internet Res . 2025 ; 27 : e59002 . doi: 10.2196/59002 OpenUrl CrossRef PubMed 6. ↵ Curtiss J , DiPietro C . Machine learning in the prediction of treatment response for emotional disorders: A systematic review and meta-analysis . Clin Psychol Rev . 2025 ; 120 : 102593 . doi: 10.1016/j.cpr.2025.102593 OpenUrl CrossRef 7. ↵ Xu Z , Wang F , Adekkanattu P , et al. Subphenotyping depression using machine learning and electronic health records . Learn Health Sys . 2020 ; 4 ( 4 ): e10241 . doi: 10.1002/lrh2.10241 OpenUrl CrossRef 8. ↵ Kraus C , Kadriu B , Lanzenberger R , Zarate CA , Kasper S . Prognosis and improved outcomes in major depression: A review . Focus (Am Psychiatr Publ ). 2020 ; 18 ( 2 ): 220 – 235 . doi: 10.1176/appi.focus.18205 OpenUrl CrossRef PubMed 9. ↵ Reynolds CF , Cuijpers P , Patel V , et al. Early intervention to reduce the global health and economic burden of major depression in older adults . Annu Rev Public Health . 2012 ; 33 : 123 – 135 . doi: 10.1146/annurev-publhealth-031811-124544 OpenUrl CrossRef PubMed Web of Science 10. ↵ Kessing LV . Severity of depressive episodes according to ICD-10: prediction of risk of relapse and suicide . Br J Psychiatry . 2004 ; 184 : 153 – 156 . doi: 10.1192/bjp.184.2.153 OpenUrl Abstract / FREE Full Text 11. ↵ Fiest KM , Jette N , Quan H , et al. Systematic review and assessment of validated case definitions for depression in administrative data . BMC Psychiatry . 2014 ; 14 ( 1 ): 289 . doi: 10.1186/s12888-014-0289-5 OpenUrl CrossRef PubMed 12. ↵ Remes O , Mendes JF , Templeton P . Biological, psychological, and social determinants of depression: A review of recent literature . Brain Sci . 2021 ; 11 ( 12 ). doi: 10.3390/brainsci11121633 OpenUrl CrossRef 13. ↵ TRIPOD+AI statement: updated guidance for reporting clinical prediction models that use regression or machine learning methods . BMJ . 2024 ; 385 :q 902 . doi: 10.1136/bmj.q902 OpenUrl CrossRef 14. ↵ Albert PR . Why is depression more prevalent in women? J Psychiatry Neurosci . 2015 ; 40 ( 4 ): 219 – 221 . doi: 10.1503/jpn.150205 OpenUrl FREE Full Text 15. ↵ van Krugten FCW , Goorden M , van Balkom AJLM , et al. Indicators to facilitate the early identification of patients with major depressive disorder in need of highly specialized care: A concept mapping study . Depress Anxiety . 2018 ; 35 ( 4 ): 346 – 352 . doi: 10.1002/da.22741 OpenUrl CrossRef PubMed 16. ↵ Robinson J , Khan N , Fusco L , Malpass A , Lewis G , Dowrick C . Why are there discrepancies between depressed patients’ Global Rating of Change and scores on the Patient Health Questionnaire depression module? A qualitative study of primary care in England . BMJ Open . 2017 ; 7 ( 4 ): e014519 . doi: 10.1136/bmjopen-2016-014519 OpenUrl Abstract / FREE Full Text 17. Ma S , Kang L , Guo X , et al. Discrepancies between self-rated depression and observed depression severity: The effects of personality and dysfunctional attitudes . Gen Hosp Psychiatry . 2021 ; 70 : 25 – 30 . doi: 10.1016/j.genhosppsych.2020.11.016 OpenUrl CrossRef PubMed 18. ↵ Mitchell AJ , Yadegarfar M , Gill J , Stubbs B . Case finding and screening clinical utility of the Patient Health Questionnaire (PHQ-9 and PHQ-2) for depression in primary care: a diagnostic meta-analysis of 40 studies . BJPsych Open . 2016 ; 2 ( 2 ): 127 – 138 . doi: 10.1192/bjpo.bp.115.001685 OpenUrl Abstract / FREE Full Text 19. ↵ Savitz ST , Chamberlain AM , Jiang R , Sarwar S , Williams MD . Impact of rurality and the area deprivation index on outcomes of collaborative care for depression . J Rural Health . 2025 ; 41 ( 3 ): e70044 . doi: 10.1111/jrh.70044 OpenUrl CrossRef PubMed 20. ↵ Rivera KM , Mollalo A . Spatial analysis and modelling of depression relative to social vulnerability index across the United States . Geospat Health . 2022 ; 17 ( 2 ). doi: 10.4081/gh.2022.1132 OpenUrl CrossRef 21. ↵ Xiong J , Lipsitz O , Nasri F , et al. Impact of COVID-19 pandemic on mental health in the general population: A systematic review . J Affect Disord . 2020 ; 277 : 55 – 64 . doi: 10.1016/j.jad.2020.08.001 OpenUrl CrossRef PubMed 22. Shetty PA , Ayari L , Madry J , Betts C , Robinson DM , Kirmani BF . The Relationship Between COVID-19 and the Development of Depression: Implications on Mental Health . Neurosci Insights . 2023 ; 18 : 26331055231191510 . doi: 10.1177/26331055231191513 OpenUrl CrossRef 23. ↵ Gao Y , Duan C , Miao T , et al. Association between depression and three key COVID-19-related outcomes: The SHARE study . Hum Vaccin Immunother . 2025 ; 21 ( 1 ): 2551930 . doi: 10.1080/21645515.2025.2551930 OpenUrl CrossRef PubMed 24. ↵ Collins GS , Moons KGM , Dhiman P , et al. TRIPOD+AI statement: updated guidance for reporting clinical prediction models that use regression or machine learning methods . BMJ . 2024 ; 385 : e078378 . doi: 10.1136/bmj-2023-078378 OpenUrl FREE Full Text View the discussion thread. Back to top Previous Next Posted December 01, 2025. Download PDF Data/Code Email Thank you for your interest in spreading the word about medRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following Development and Validation of Machine Learning-Based Prediction of Depression Progression Using EHR Data: A Multi-Institutional Retrospective Cohort Study Message Subject (Your Name) has forwarded a page to you from medRxiv Message Body (Your Name) thought you would like to see this page from the medRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share Development and Validation of Machine Learning-Based Prediction of Depression Progression Using EHR Data: A Multi-Institutional Retrospective Cohort Study Pegah Ahadian , Angela Fragano , Tianyuan Guan , Qiang Guan , Sophia Z. Shalhout medRxiv 2025.11.28.25341207; doi: https://doi.org/10.1101/2025.11.28.25341207 Share This Article: Copy Citation Tools Development and Validation of Machine Learning-Based Prediction of Depression Progression Using EHR Data: A Multi-Institutional Retrospective Cohort Study Pegah Ahadian , Angela Fragano , Tianyuan Guan , Qiang Guan , Sophia Z. Shalhout medRxiv 2025.11.28.25341207; doi: https://doi.org/10.1101/2025.11.28.25341207 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Psychiatry and Clinical Psychology Subject Areas All Articles Addiction Medicine (570) Allergy and Immunology (864) Anesthesia (302) Cardiovascular Medicine (4445) Dentistry and Oral Medicine (444) Dermatology (383) Emergency Medicine (609) Endocrinology (including Diabetes Mellitus and Metabolic Disease) (1515) Epidemiology (15236) Forensic Medicine (30) Gastroenterology (1127) Genetic and Genomic Medicine (6610) Geriatric Medicine (669) Health Economics (1000) Health Informatics (4549) Health Policy (1370) Health Systems and Quality Improvement (1613) Hematology (543) HIV/AIDS (1266) Infectious Diseases (except HIV/AIDS) (15926) Intensive Care and Critical Care Medicine (1104) Medical Education (623) Medical Ethics (147) Nephrology (668) Neurology (6613) Nursing (346) Nutrition (999) Obstetrics and Gynecology (1147) Occupational and Environmental Health (957) Oncology (3341) Ophthalmology (975) Orthopedics (369) Otolaryngology (420) Pain Medicine (436) Palliative Medicine (130) Pathology (665) Pediatrics (1694) Pharmacology and Therapeutics (693) Primary Care Research (714) Psychiatry and Clinical Psychology (5458) Public and Global Health (9244) Radiology and Imaging (2205) Rehabilitation Medicine and Physical Therapy (1370) Respiratory Medicine (1197) Rheumatology (596) Sexual and Reproductive Health (715) Sports Medicine (530) Surgery (713) Toxicology (99) Transplantation (289) Urology (265) (function(){function c(){var b=a.contentDocument||a.contentWindow.document;if(b){var d=b.createElement('script');d.innerHTML="window.__CF$cv$params={r:'a02431db2c4f6151',t:'MTc3OTg3NjI1OQ=='};var a=document.createElement('script');a.src='/cdn-cgi/challenge-platform/scripts/jsd/main.js';document.getElementsByTagName('head')[0].appendChild(a);";b.getElementsByTagName('head')[0].appendChild(d)}}if(document.body){var a=document.createElement('iframe');a.height=1;a.width=1;a.style.position='absolute';a.style.top=0;a.style.left=0;a.style.border='none';a.style.visibility='hidden';document.body.appendChild(a);if('loading'!==document.readyState)c();else if(window.addEventListener)document.addEventListener('DOMContentLoaded',c);else{var e=document.onreadystatechange||function(){};document.onreadystatechange=function(b){e(b);'loading'!==document.readyState&&(document.onreadystatechange=e,c())}}}})();
Text is read by the "Ask this paper" AI Q&A widget below.
Extraction quality varies by source — PMC NXML preserves structure
cleanly, OA-HTML may include some navigation residue, and OA-PDF can
have broken hyphenation. The publisher copy
(via DOI)
is the canonical version.