Full text
45,859 characters
· extracted from
preprint-html
· click to expand
Assessment of Medication Adherence in Patients: Development and Validation of a Machine Learning Model | medRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-P4HH5NV'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search Assessment of Medication Adherence in Patients: Development and Validation of a Machine Learning Model Jifan Zhang , Jason Zhensheng Qu doi: https://doi.org/10.1101/2025.09.14.25335382 Jifan Zhang 1 Nanjing Foreign Language School , Nanjing 210008, China Find this author on Google Scholar Find this author on PubMed Search for this author on this site Jason Zhensheng Qu 1 Nanjing Foreign Language School , Nanjing 210008, China Find this author on Google Scholar Find this author on PubMed Search for this author on this site For correspondence: Jqu{at}mgh.Harvard.edu Abstract Full Text Info/History Metrics Data/Code Preview PDF Abstract Background This study addresses limitations of traditional medication adherence assessment tools by developing a machine learning model to evaluate post-discharge medication compliance in patients using drugs. The research was conducted at Nanjing Drum Tower Hospital from February 2024 to December 2024. Methods We collected clinical data from 240 patients through questionnaires and developed a multi-class machine learning model. Feature selection employed manual screening and polynomial logistic regression. Six ML models were evaluated, with the Random Forest Classifier (RFC) demonstrating optimal performance (bad_AUC = 0.979, fine_AUC = 0.973, good_AUC = 0.917). SHAP analysis was used to explain the best-performing model. Results The RFC model showed superior predictive capability across all adherence levels. Model interpretation revealed key clinical factors influencing adherence patterns. The tool enables early identification of non-compliance and supports intervention strategies. Conclusions This RFC-based model represents a significant advancement in medication adherence assessment, offering clinicians a practical tool for monitoring compliance. The approach shows particular promise for enhancing mental health management in this patient population by fostering better medication awareness and establishing scientific medication habits during early treatment stages. As a core determinant of treatment efficacy, medication adherence has become a significant research topic in the field of precision medicine [ 1 ]. The World Health Organization (WHO) has highlighted that treatment failures, complications, and increased healthcare costs caused by long-term non-adherence to medication regimens have evolved into major societal issues [ 2 ]. Medication adherence is influenced by multi-dimensional factors: patient age, health literacy, and psychological state form internal drivers; medication dosage frequency and side effects create direct barriers; while the quality of medical communication and socioeconomic conditions constitute external constraints [ 3 ]. However, existing studies primarily focus on specific disease populations, lacking systematic attention to medication behaviors in the general population, which fails to meet the needs of the precision medicine era. Current mainstream medication adherence assessment tools (e.g. MMAS-8, MARS scale) have significant limitations [ 4 ]. The MMAS-8 scale measures patients’ medication-taking behavior over the past four weeks through 8 items, but patients may overestimate adherence due to social desirability bias, while clinical records often overlook asymptomatic medication interruptions [ 5 - 6 ]. Traditional scales rely on patient self-reports or clinical observations, making them susceptible to recall bias and subjective judgments, and most are static assessments unable to capture dynamic changes in medication behaviors [ 7 - 8 ]. This study innovatively constructs a predictive model: using machine learning algorithms to identify high-risk non-adherent populations, combined with other technologies to implement precision interventions, ultimately achieving full-cycle management of treatment. Materials and methods Study design The study protocol strictly follows the Transparent Reporting of a Multivariable Prediction Model for Individual Prognosis or Diagnosis (TRIPOD) reporting guidelines. Participants This study was conducted from February 2024 to December 2024 in the inpatient department of Nanjing Drum Tower Hospital. All subjects in this study were patients who had used drugs. The inclusion criteria were: (1) Chinese citizens aged ≥ 18 years, (2)voluntary participation in the survey and (3) being able to complete the related questionnaires independently. The exclusion criteria were: (1) prior medical history with severe mental illnesses, (2) prior use of medical drugs to treat mental disorders, (3)pregnant and lactating women, (4) medical history of malignant tumor. Data collection We collected comprehensive patient data in a structured format, including demographics (age, height, weight, gender, marital status, educational attainment, type of medical insurance), comorbidity history (including hypertension, diabetes, coronary heart disease, gout, renal insufficiency, liver dysfunction, pulmonary hypertension, heart failure), medication use (aspirin, clopidogrel, amiodarone, NASIDs (including tramadol), digoxin, beraprost, ACEI/ARB, β-blockers, statins), laboratory test results (TTR, variability classification, regular follow-up, location, number of comorbid diseases, number of concomitant medications, polypharmacy, disability, multiple diseases, history of thrombosis, history of stroke, history of bleeding, whether taking medication), and other relevant metrics. For the assessment of medication adherence, we utilized a well-validated and reliable tool to measure the adherence behavior of patients over the past two weeks. Medication adherence was categorized based on standard criteria: bad adherence, fine adherence, and good adherence. In this study, we formulated medication adherence prediction as a three-class classification task: bad adherence, fine adherence, and good adherence. Data preprocessing We performed min–max normalization on the entire dataset by mapping all features to the range of [0, 1 ] using the maximum and minimum values of each feature. This normalization eliminates scale differences among different features and ensures comparability and analysis within the same numerical range, thereby enhancing the performance of the algorithm model.In machine learning(ML) modeling, a sample class ratio lower than 9:1 is typically regarded as imbalanced. Our dataset had a notably lower proportion for one outcome event, indicating an extremely skewed class distribution. To address this, we used the Synthetic Minority Over - sampling Technique (SMOTE) to generate a balanced dataset. This approach involves synthesizing samples for the minority class to balance the number of samples across different classes, thus creating a dataset where each class has a similar representation for more effective modeling. Feature selection A two-step feature selection process was employed. First, manual screening identified potentially relevant features associated with medication adherence. Subsequently, polynomial logistic regression was applied to further refine the feature subset. This method helped in selecting the most significant features that could potentially influence medication adherence. Data imputation and normalization Missing values were imputed using K-Nearest Neighbors (KNN) for continuous variables and mode imputation for categorical variables in both the training and test sets. Additionally, continuous features were normalized using Z-score standardization, while multi-categorical features were one-hot encoded to create dummy variables, prior to model training. Model development A total of 240 patients were involved in this study.The normality of continuous variables was evaluated using the Shapiro-Wilk test. For normally distributed variables, the mean and standard deviation were utilized to express their values, whereas non-normally distributed variables were represented by the median (interquartile range [IQR]). When comparing two groups, Student’s t-test or the Wilcoxon rank-sum test was selected based on the distribution characteristics of the data. Categorical variables were presented in terms of counts and percentages, with comparisons between groups conducted using the chi-square test. Variables with more than 10% missing values were excluded from the analysis. The remaining missing values were imputed using a random forest-based multiple imputation method. To address the class imbalance between the negative and positive groups, the Synthetic Minority Over-sampling Technique (SMOTE) was applied to balance the dataset. Finally, the dataset was divided into training and validation sets at a ratio of 80% to 20%.Grid search was employed for hyperparameter tuning. For variable selection, classical polynomial logistic regression was utilized. Variables with a p - value less than 0.05 were included in the analysis, while the rest were excluded. To enhance the reliability and generalization ability of the model, a five - fold cross - validation was implemented. In this process, the training set was further divided into five non - overlapping subsets. In each iteration, four subsets were used for training, and the remaining one subset was used for validation. This approach was repeated five times, with each subset serving as the validation set exactly once. By averaging the evaluation metrics across these five iterations, a more comprehensive and accurate assessment of the model performance was obtained. Six machine learning models were adopted, namely logistic regression, support vector machine, K - Nearest Neighbors (KNN), Random Forest Classifier (RFC), CatBoost, and LightGBM (LBGM). These models were trained and evaluated to predict the anticoagulation outcomes, aiming to establish a reliable prediction model for better understanding and predicting the efficacy of anticoagulation therapy. Model evaluation The hold-out test set was used exclusively to evaluate the generalizability of the developed models. Considering the inherent characteristics of the multi-classification task, we employed a comprehensive suite of evaluation metrics to assess multi-classification model performance. This included: (1) Confusion Matrix: Visualized the distribution of true and predicted medication adherence classifications across categories (bad, fine, good); (2) Multi-class ROC Curves: Illustrated the tradeoff between sensitivity and specificity for each adherence class (see Fig. 2 for the ROC curves of the six models); (3) Macro-average AUC: Evaluated overall model performance by averaging AUC scores across all adherence categories; (4) Kappa Statistic: Assessed model agreement with the true classifications, accounting for class imbalance; (5) Average Precision, Recall, and F1-score. In a clinical setting, the ideal approach would be to develop a predictive model that is more effective at identifying patients with poor medication adherence. Download figure Open in new tab Download figure Open in new tab Fig. 1. Overview of the study workflow. Download figure Open in new tab Fig. 2. ROC curve of the six multi-classification ML models in the test cohort. (a) LR; (b) SVM; (c) RFC; (d) LGBM; (e) Catboost; (f) ANN. Notes: ROC, receiver operating curve; ML, machine learning; LR, logistic regression; SVM, support vector machine; RFC, random forest classifier; LGBM, light gradient boosting machine; Catboost, categorical boosting; ANN, artificial neural network. Furthermore, the validity of the proposed models was assessed using the test cohort. The performance of the six models within the test cohort is detailed in Table 1 and illustrated in Fig. 2 . Confusion matrix of the six multi-classification ML models in the test cohort is shown in Fig. 3 . Among these, RFC model demonstrating strong predictive capabilities across different levels of medication adherence (bad_AUC = 0.979412, fine_AUC = 0.973499, good_AUC = 0.917184). These results highlight the robust performance of the RFC model. View this table: View inline View popup Download powerpoint Table 1. The performance of the following six multi-classification ML models in the test cohort (n=48). ML, machine learning; LR, logistic regression; SVM, support vector machine; LGBM, light gradient boosting machine; Catboost, categorical boosting; ANN, artificial neural network. View this table: View inline View popup Download powerpoint Table 2. The demographic, clinical and laboratory characteristics of the different depression levels groups in the training cohort. Download figure Open in new tab Fig. 3. Confusion matrix of the six multi-classification ML models in the test cohort. (a) LR; (b) SVM; (c) RFC; (d) LGBM; (e) Catboost; (f) ANN. Notes: ML, machine learning; LR, logistic regression; SVM, support vector machine; RFC, random forest classifier; LGBM, light gradient boosting machine; Catboost, categorical boosting; ANN, artificial neural network. Model interpretation The SHapley Additive exPlanations (SHAP) technique, demonstrating each variable’s influence on the overall model, was further applied to gain insight into the best-performance model. We utilized the SHAP technique to evaluate the contribution of features in the best-performing model to the prediction outcomes of the test set. In the three-class classification task, the model’s objective is to accurately assign samples to one of the three categories. Consequently, we calculated the SHAP values for each category. For each sample, the SHAP value is represented as a matrix of size (1, k, 3), where each matrix element corresponds to the contribution of a feature to a specific category (with k representing the number of features in the model). We generated separate SHAP bar and dot plots for each category using the test set data (see Fig.4 for an example of SHAP plots). Furthermore, to visualize the distribution of key features across different adherence levels in a sample of patients, we generated a grouped-clustered heatmap for ten randomly selected patients using the “pheatmap” package in R (version 4.2.1). Finally, for enhanced clinical usability, the best-performance multi-classification model was deployed as an R Shiny application, facilitating user-friendly model interaction. Statistical analysis This study aimed to compare the differences in demographic, clinical, and laboratory characteristics across three levels of medication adherence. The initial step involved evaluating the normality of continuous variables utilizing the Shapiro-Wilk test. Continuous variables that adhered to a normal distribution were presented as mean ± standard deviation (SD) and were analyzed using the Chi-square test. In contrast, those not following normal distribution were depicted as median ± interquartile range (IQR) and assessed with the Kruskal-Wallis test. Subsequently, categorical variables were compared using either the Chi-square (χ2) or Fisher’s exact test, depending on their distribution and sample size. A p-value of less than 0.05 was considered statistically significant for all tests. The entire statistical analysis process was performed using the “compareGroups” package in R version 4.2.1. Results Participants’ characteristics The Morisky Medication Adherence Scale was used to assess the patients’ medication adherence. By filling out the questionnaires distributed by the hospital, a total of 240 eligible patients were included.The median age of the participants was 60.5 years (IQR: 51.2–69.8 years), with a predominance of female patients, accounting for 75.2%.To address the class imbalance between the negative and positive groups, the Synthetic Minority Over-sampling Technique (SMOTE) was applied to balance the dataset. Finally, the dataset was divided into training and validation sets at a ratio of 80% to 20%. These 240 patients were randomly divided into two groups: the training cohort (n = 192) and the test cohort (n = 48). Comparative analysis of baseline characteristics between the two cohorts indicated a general balance, as detailed in Supplementary Table 2. The MMAS-8 scale was used to assess the medication adherence of patients. The medication adherence of patients was classified into three levels: bad adherence (below 6 points), fine adherence (6-8 points), and good adherence (8 points). Within the training cohort, the prevalence rates for these categories were 33.3% (n = 64/192), 33.3% (n = 64/192), and 33.3% (n = 64/192), respectively. Correspondingly, in the test cohort, the prevalence rates were 33.3% (n = 16/48), 33.3% (n = 16/48), and 33.3% (n = 16/48), respectively. Feature selection In this research, forty-eight potential variables related to medication adherence in patients using drugs were initially considered. Of these, twenty-four variables were identified as having missing data. Twenty-four variables had missing values, and six variables with missing values over 25% were excluded from the analysis . Manual screening subsequently pinpointed fifteen variables markedly associated with the medication adherence outcome, including age, height, weight, BMI, number of comorbid diseases, number of concomitant medications, polypharmacy, disability, gender, regular follow-up, location, educational attainment, type of medical insurance, disease onset, and history of stroke. These identified fifteen variables were further analyzed using polynomial logistic regression to discern the optimal subset. The polynomial logistic regression ultimately determined that ten variables had a significant association with the multi-classification medication adherence outcome: age, height, weight, number of comorbid diseases, number of concomitant medications, gender, location, educational attainment, history of stroke, and whether taking medication. Model performance The aforementioned ten variables were used to construct the following six different multi-classification ML models: LR, SVM, RFC, LGBM, Catboost, and KNN. The optimal hyperparameters for each model are documented in Supplementary Table 3. Supplementary Table 4 and Figure 2 show the performance metrics of these models within the training cohort. Figure 3 shows the confusion matrix of each multi-class model on the training cohort. Notably, except for the LR and the KNN models, all other models demonstrated superior performance, ranking in the top four across various metrics, including bad_AUC, fine_AUC, and good_AUC. These models have the potential to correctly detect more positive cases of poor medication adherence. View this table: View inline View popup Table 3. The best hyperparameters for the six multi-classification ML models View this table: View inline View popup Download powerpoint Table 4. The performance of the six multi-classification ML models with 15-fold cross validation in the training cohort (n = 400) Furthermore, the validity of the proposed models was assessed using the test cohort. The performance of the six models within the test cohort is detailed in Table 1 and illustrated in Fig. 2 . Confusion matrix of the six multi-classification ML models in the test cohort is shown in Fig. 3 . Among these, the RFC model exhibited the highest levels of bad_AUC, fine_AUC, and good_AUC. These results highlight the robust performance of the RFC model and its effectiveness in classifying medication adherence levels in patients using anticoagulant drugs. Model interpretation The RFC model was explained through the application of the SHAP algorithm, which calculated the average absolute SHAP values to quantify the impact of each variable. Figure 4 illustrates the distributions of feature importance analyses and variable impact model outputs for the RFC model across the bad, fine, and good adherence categories. Notably, the analysis highlights that higher age, lower height, higher weight, increased number of comorbid diseases, higher number of concomitant medications, male gender, urban location, lower educational attainment, and history of stroke were associated with increased SHAP values. This suggests these factors correlate with an elevated risk of poor medication adherence, as indicated by the model. Download figure Open in new tab Fig. 4. SHAP summary plots for explaining the nine variables contributing to the RFC model. Model deployment To enhance the practical applicability of our proposed RFC model in clinical settings, the model was deployed to the cloud, making it accessible to healthcare professionals for real-time medication adherence assessment in patients using drugs. The model is hosted on a web-based platform available at https://patient-medication-adherence-calculator.streamlit.app/ . The user interface of this online tool is shown in Figure 5 , aiming to provide a user-friendly experience for clinicians, patients and researchers. Download figure Open in new tab Fig. 5. The screenshot of the web-based tool. Discussion This study introduces the Random Forest Classifier (RFC) model[ 9 ]as a predictive tool for assessing medication adherence in patients using drugs, demonstrating superior performance over competing models across multiple metrics such as AUC, accuracy, precision, recall, F1 score, and kappa coefficient[ 10 ]. The model’s robustness in parsing complex medical data reveals its potential for accurate medication adherence risk predictions[ 11 ]. Unlike traditional assessment tools, which often rely on clinical judgment[ 12 ], the RFC model integrates and analyzes a broad range of clinical and laboratory data, automating the analysis process[ 13 ]. This circumvents the need for manual scoring by clinicians, facilitating quicker and more objective assessments[ 14 ]. The development of a web-based calculator further enhances the model’s clinical utility, providing healthcare professionals with a readily accessible tool for real-time medication adherence assessment[ 15 ]. This predictive model can assist the general population and potential drug users in prospectively assessing their own ability for drug management. By intelligently analyzing their daily medication behaviors and life routines, it can accurately identify potential tendencies of poor medication compliance. When the assessment results indicate a risk of insufficient compliance, users can promptly notice the weak links in their health management and actively optimize their behavior patterns through personalized methods such as setting medication reminders and integrating medication taking with fixed daily behaviors. This mechanism not only transforms abstract health management into concrete visual feedback but also forms a virtuous cycle through continuous behavioral guidance, helping users cultivate rigorous medication awareness subconsciously and establish scientific and standardized medication habits in the early stages of disease prevention or treatment. This research contributes to the literature on machine learning (ML) applications in refining medication adherence assessment strategies[ 16 ], expanding upon previous studies[ 17 ]. It acknowledges the multifactorial nature of medication non-adherence, arising from various factors such as patient demographics, comorbidities, medication regimens, and psychosocial factors[ 18 ], which manifest the need for comprehensive approaches to understanding and managing this condition[ 19 ]. The RFC model marks a notable advancement in applying ML to patient care, offering a nuanced and efficient method for identifying at-risk individuals[ 20 ]. The relationship between demographic factors (such as age and gender) and medication adherence is well-documented[ 21 ]. Our study highlights that higher age and male gender are associated with poorer medication adherence, which is consistent with previous findings[ 22 ]. Additionally, the presence of multiple comorbid diseases and higher BMI also correlate with non-adherence[ 23 ], suggesting that these factors may complicate medication management and patient engagement[ 24 ]. The impact of medication regimen complexity on adherence is another critical area explored in this study[ 25 ]. The findings indicate that a higher number of concomitant medications and polypharmacy are significant predictors of poor adherence[ 26 ]. This underscores the importance of optimizing medication regimens and minimizing unnecessary polypharmacy to improve patient outcomes[ 27 ]. Geographical location and educational attainment also play a role in medication adherence[ 28 - 29 ]. Patients from urban areas and those with higher educational levels tend to have better adherence, possibly due to improved access to healthcare resources and better understanding of treatment plans[ 30 ]. These findings suggest that targeted interventions may be needed for patients in rural areas or those with lower educational backgrounds[ 31 ]. Our study also highlights the importance of regular follow-up and the presence of a history of stroke in predicting medication adherence[ 32 ]. Regular follow-up visits provide opportunities for healthcare providers to reinforce the importance of adherence and address any barriers or side effects[ 33 ]. The association between a history of stroke and better adherence may reflect heightened awareness of the risks associated with non-adherence among these patients[ 34 ]. The utilization of SHAP (SHapley Additive exPlanations) for model interpretation further elucidates the contributions of specific variables to medication adherence risk, enhancing the model’s clinical relevance[ 35 ]. The SHAP values provide insights into how each feature influences the model’s predictions, allowing clinicians to identify key factors driving non-adherence in individual patients[ 36 ]. Most existing studies focus on compliance assessment based on patients’ current clinical characteristics or historical behaviors. Although such methods can identify the current situation, they are difficult to achieve early risk warning and dynamic intervention. The innovation of this study lies in integrating multi-dimensional patient data (such as demographic characteristics, comorbidities, medication complexity, etc.) to construct a machine learning model with forward-looking predictive capabilities. This model can not only assess the compliance status of patients in real time, but also predict their potential future compliance risk levels (such as low, medium, and high) based on individual characteristics. This predictive ability makes it possible for clinical intervention to shift from traditional passive response to active prevention. For example, for patients predicted to be at high risk, personalized intervention measures (such as simplifying medication regimens, strengthening medication reminders, or providing psychological support) can be implemented in advance to effectively block the risk chain before compliance deteriorates. In addition, the dynamic prediction feature of the model can be combined with the patient’s real-time updated clinical data to achieve dynamic adjustment of risk stratification and intervention strategies, providing technical support for full-cycle health management in the era of precision medicine. Compared with static assessment tools, this study opens up a new path for improving the treatment outcomes of long-term medication patients through the combination of predictive modeling and early intervention. Despite its promising results, our study has some limitations. The data were collected from a single center, which may introduce biases and limit the generalizability of the findings . Future research should focus on expanding the model’s generalizability through multi-center studies . Additionally, the exclusion of variables with excessive missing data points might have affected the model’s performance . Future studies should aim to improve data collection methods to minimize missing data . Furthermore, the model’s performance could be further enhanced by incorporating additional biomarkers and psychosocial factors that may influence medication adherence . In conclusion, this study demonstrates the potential of advanced ML techniques in transforming the approach to medication adherence assessment within the medical field . The RFC model outperformed existing assessment tools and other ML models in accuracy and efficiency . Deployed as a web-based application, the model offers a practical tool for healthcare professionals, facilitating early identification and management of medication non-adherence .Furthermore, this model can also assist users in cultivating a rigorous medication awareness unconsciously, and establish scientific and standardized medication habits at the early stage of disease prevention or treatment. Future research should focus on expanding the model’s generalizability and integrating additional factors to refine its predictive capability . Conclusions This study introduces a novel multi-classification machine learning (ML)-based model, employing the Random Forest Classifier (RFC) algorithm, for the assessment of medication adherence in patients using drugs. Through rigorous analysis of multidimensional patient data, including clinical information, laboratory test results, and medication regimens, we identified key variables substantially associated with medication adherence risk. The RFC model outperformed existing assessment tools and other ML models in accuracy and efficiency. The utilization of SHAP for model interpretation further elucidates the contributions of specific variables to medication adherence risk, enhancing the model’s clinical relevance. Deployed as a web-based application, the model offers a practical tool for healthcare professionals, facilitating early identification and management of medication non-adherence in patients using drugs. Despite its promising results, future research should focus on expanding the model’s generalizability through integrating other indicators to refine its predictive capability. This study displays the potential of advanced ML techniques in transforming the approach to medication adherence assessment within the medical field, particularly for patients with complex chronic conditions requiring long-term therapy. Data availability Data presented in this study were obtained from patients admitted to Nanjing Drum Tower Hospital who consented to participate in the study. Owing to data protection rules, we are not allowed to share personal-level data. The datasets used and/or analysed during the current study are available from the corresponding author on reasonable request. Data Availability All data produced in the present study are available upon reasonable request to the authors. References [1]. ↵ Jimmy B , Jose J. Patient medication adherence: measures in daily practice[J] . Oman medical journal , 2011 , 26 ( 3 ): 155 . OpenUrl PubMed [2]. ↵ World Health Organization (WHO ). ( 2003 ). Failure to take prescribed medicine for chronic diseases is a massive, world-wide problem . Retrieved from https://www.who.int/news/item/01-07-2003-failure-to-take-prescribed-medicine-for-chronic-diseases-is-a-massive-world-wide-problem [3]. ↵ Gast A , Mathes T. Medication adherence influencing factors—an (updated) overview of systematic reviews[J] . Systematic reviews , 2019 , 8 : 1 – 17 . OpenUrl PubMed [4]. ↵ Naqvi A A , Hassali M A. Limitations in contemporary self-reported medication adherence questionnaires: the concept and design of the General Medication Adherence Scale (GMAS) originating from a developing country[J] . Current Medical Research and Opinion , 2019 , 35 ( 1 ): 1 – 2 . OpenUrl [5]. ↵ Moon S J , Lee W Y , Hwang J S , et al. Accuracy of a screening tool for medication adherence: A systematic review and meta-analysis of the Morisky Medication Adherence Scale-8[J] . PloS one , 2017 , 12 ( 11 ): e0187139 . OpenUrl CrossRef PubMed [6]. ↵ Selvaraj K , Thekkur P , Chinnakali P. Comparing adherence in cardiac clinic versus general outpatient clinic: Few concerns and way forward[J] . Annals of Medical and Health Sciences Research , 2015 , 5 ( 6 ): 485 – 486 . OpenUrl [7]. ↵ Nguyen T M U , Caze A L , Cottrell N. What are validated self-report adherence scales really measuring?: a systematic review[J] . British journal of clinical pharmacology , 2014 , 77 ( 3 ): 427 – 445 . OpenUrl CrossRef PubMed [8]. ↵ La Caze A , Cottrell N. Validated adherence scales used in a measurement-guided medication management approach to target and tailor a medication adherence intervention: a randomised controlled trial[J] . BMJ open , 2016 , 6 ( 11 ): e013375 . OpenUrl Abstract / FREE Full Text [9]. ↵ L. Breiman , “ Random forests ,” Machine Learning , vol. 45 , no. 1 , pp. 5 – 32 , 2001 . OpenUrl [10]. ↵ T. Fawcett , “ An introduction to ROC analysis ,” Pattern Recognition Letters , vol. 27 , no. 8 , pp. 861 – 874 , 2006 . OpenUrl CrossRef [11]. ↵ J. H. Friedman , “ Greedy function approximation: A gradient boosting machine ,” Annals of Statistics , vol. 29 , no. 5 , pp. 1189 – 1232 , 2001 . OpenUrl CrossRef Web of Science [12]. ↵ J. A. Cramer et al. , “ Medication compliance and persistence: Terminology and definitions ,” Value in Health , vol. 11 , no. 1 , pp. 44 – 47 , 2008 . OpenUrl PubMed [13]. ↵ Z. Obermeyer and E. J. Emanuel , “ Predicting the future—big data, machine learning, and clinical medicine ,” The New England Journal of Medicine , vol. 375 , no. 13 , pp. 1216 – 1219 , 2016 . OpenUrl CrossRef PubMed [14]. ↵ A. Rajkomar et al. , “ Machine learning in medicine ,” The New England Journal of Medicine , vol. 380 , no. 14 , pp. 1347 – 1358 , 2019 . OpenUrl CrossRef PubMed [15]. ↵ M. P. Turakhia et al. , “ Feasibility of extended ambulatory electrocardiogram monitoring to identify silent atrial fibrillation in high-risk patients: The mHealth Screening to Prevent Strokes (mSToPS) trial ,” American Heart Journal , vol. 200 , pp. 102 – 109 , 2018 . OpenUrl PubMed [16]. ↵ Murphy , S. A. , & McKay , J. R. ( 2004 ). Adaptive treatment strategies: An emerging approach for improving treatment effectiveness . Clinical Science , 12 , 7 – 13 . OpenUrl [17]. ↵ J. A. Cramer et al. , “ How often is medication taken as prescribed? A novel assessment technique ,” JAMA , vol. 261 , no. 22 , pp. 3273 – 3277 , 1989 . OpenUrl CrossRef PubMed Web of Science [18]. ↵ Li , L. , Cui , Y. , Yin , R. , Chen , S. , Zhao , Q. , Chen , H. , & Shen , B. ( 2017 ). Medication adherence has an impact on disease activity in rheumatoid arthritis: a systematic review and meta-analysis . Patient preference and adherence , 1343 – 1356 . [19]. ↵ E. Sabaté , Adherence to Long-Term Therapies: Evidence for Action . World Health Organization , 2003 . [20]. ↵ A. L. Beam and I. S. Kohane , “ Big data and machine learning in health care ,” JAMA , vol. 319 , no. 13 , pp. 1317 – 1318 , 2018 . OpenUrl CrossRef PubMed [21]. ↵ M. R. DiMatteo , “ Variations in patients’ adherence to medical recommendations: A quantitative review of 50 years of research ,” Medical Care , vol. 42 , no. 3 , pp. 200 – 209 , 2004 . OpenUrl CrossRef PubMed Web of Science [22]. ↵ Lewey , J. , Shrank , W. H. , Bowry , A. D. , Kilabuk , E. , Brennan , T. A. , & Choudhry , N. K. ( 2013 ). Gender and racial disparities in adherence to statin therapy: a meta-analysis . American heart journal , 165 ( 5 ), 665 – 678 . OpenUrl CrossRef PubMed Web of Science [23]. ↵ D. E. Morisky et al. , “ Predictive validity of a medication adherence measure in an outpatient setting ,” The Journal of Clinical Hypertension , vol. 10 , no. 5 , pp. 348 – 354 , 2008 . OpenUrl PubMed [24]. ↵ M. R. DiMatteo et al. , “ Patient adherence and medical treatment outcomes: A meta-analysis ,” Medical Care , vol. 40 , no. 9 , pp. 794 – 811 , 2002 . OpenUrl CrossRef PubMed Web of Science [25]. ↵ Quirós López , R. , Formiga Pérez , F. , & Beyer-Westendorf , J. ( 2025 ). Adherence and persistence with direct oral anticoagulants by dose regimen: A systematic review . British Journal of Clinical Pharmacology . [26]. ↵ Mohammed , S. , Arabi , A. , El-Menyar , A. , Abdulkarim , S. , Aljundi , A. , Alqahtani , A. , … Al Suwaidi , J. ( 2016 ). Impact of polypharmacy on adherence to evidence-based medication in patients who underwent percutaneous coronary intervention . Current Vascular Pharmacology , 14 ( 4 ), 388 – 393 . OpenUrl PubMed [27]. ↵ S. M. Smith et al. , “ Interventions for improving outcomes in patients with multimorbidity in primary care and community settings ,” Cochrane Database of Systematic Reviews , no. 3 , 2016 . [28]. ↵ McGrath , J. J. , Saha , S. , Al-Hamzawi , A. , Alonso , J. , Bromet , E. J. , Bruffaerts , R. , … & Kessler , R. C. ( 2015 ). Psychotic experiences in the general population: a cross-national analysis based on 31 261 respondents from 18 countries . JAMA psychiatry , 72 ( 7 ), 697 – 705 . OpenUrl PubMed [29]. ↵ Marcolino , M. S. , Sales , T. L. S. , Oliveira , J. A. D. , Rios , D. R. A. , Pedroso , T. M. , Sá , L. C. D. , … & Ribeiro , A. L. P. ( 2023 ). Health literacy, patient knowledge and adherence to oral anticoagulation in primary care . International Journal of Cardiovascular Sciences , 36 , e20220158 . OpenUrl [30]. ↵ J. A. Gazmararian et al. , “ Health literacy and medication adherence among Medicare managed care enrollees ,” Journal of Health Communication , vol. 11 , no. 8 , pp. 791 – 806 , 2006 . OpenUrl [31]. ↵ Roter , D. L. , Hall , J. A. , Merisca , R. , Nordstrom , B. , Cretin , D. , & Svarstad , B. ( 1998 ). Effectiveness of interventions to improve patient compliance: a meta-analysis . Medical care , 36 ( 8 ), 1138 – 1161 . OpenUrl CrossRef PubMed Web of Science [32]. ↵ Bansilal , S. , Castellano , J. M. , Garrido , E. , Wei , H. G. , Freeman , A. , Spettell , C. , … & Fuster , V. ( 2016 ). Assessing the impact of medication adherence on long-term cardiovascular outcomes . Journal of the American College of Cardiology , 68 ( 8 ), 789 – 801 . OpenUrl FREE Full Text [33]. ↵ M. R. DiMatteo et al. , “ Patient adherence to pharmacotherapy: The importance of effective communication ,” Journal of the American Pharmacists Association , vol. 43 , no. 5 , pp. 607 – 616 , 2003 . OpenUrl [34]. ↵ Chapman , B. , & Bogle , V. ( 2014 ). Adherence to medication and self-management in stroke patients . British Journal of Nursing , 23 ( 3 ), 158 – 166 . OpenUrl PubMed [35]. ↵ Lundberg , S. M. , & Lee , S. I. ( 2017 ). A unified approach to interpreting model predictions . Advances in neural information processing systems , 30 . [36]. ↵ M. T. Ribeiro et al. , “ Why should I trust you? Explaining the predictions of any classifier ,” Proceedings of the 22nd ACM SIGKDD International Conference on Knowledge Discovery and Data Mining , pp. 1135 – 1144 , 2016 . View the discussion thread. Back to top Previous Next Posted September 18, 2025. Download PDF Data/Code Email Thank you for your interest in spreading the word about medRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following Assessment of Medication Adherence in Patients: Development and Validation of a Machine Learning Model Message Subject (Your Name) has forwarded a page to you from medRxiv Message Body (Your Name) thought you would like to see this page from the medRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share Assessment of Medication Adherence in Patients: Development and Validation of a Machine Learning Model Jifan Zhang , Jason Zhensheng Qu medRxiv 2025.09.14.25335382; doi: https://doi.org/10.1101/2025.09.14.25335382 Share This Article: Copy Citation Tools Assessment of Medication Adherence in Patients: Development and Validation of a Machine Learning Model Jifan Zhang , Jason Zhensheng Qu medRxiv 2025.09.14.25335382; doi: https://doi.org/10.1101/2025.09.14.25335382 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Pharmacology and Therapeutics Subject Areas All Articles Addiction Medicine (568) Allergy and Immunology (863) Anesthesia (300) Cardiovascular Medicine (4435) Dentistry and Oral Medicine (444) Dermatology (382) Emergency Medicine (608) Endocrinology (including Diabetes Mellitus and Metabolic Disease) (1509) Epidemiology (15229) Forensic Medicine (30) Gastroenterology (1124) Genetic and Genomic Medicine (6600) Geriatric Medicine (668) Health Economics (997) Health Informatics (4536) Health Policy (1368) Health Systems and Quality Improvement (1613) Hematology (541) HIV/AIDS (1264) Infectious Diseases (except HIV/AIDS) (15916) Intensive Care and Critical Care Medicine (1103) Medical Education (623) Medical Ethics (146) Nephrology (667) Neurology (6599) Nursing (346) Nutrition (998) Obstetrics and Gynecology (1144) Occupational and Environmental Health (957) Oncology (3332) Ophthalmology (974) Orthopedics (369) Otolaryngology (420) Pain Medicine (436) Palliative Medicine (130) Pathology (663) Pediatrics (1693) Pharmacology and Therapeutics (691) Primary Care Research (711) Psychiatry and Clinical Psychology (5447) Public and Global Health (9232) Radiology and Imaging (2198) Rehabilitation Medicine and Physical Therapy (1370) Respiratory Medicine (1196) Rheumatology (593) Sexual and Reproductive Health (712) Sports Medicine (530) Surgery (712) Toxicology (99) Transplantation (289) Urology (265) (function(){function c(){var b=a.contentDocument||a.contentWindow.document;if(b){var d=b.createElement('script');d.innerHTML="window.__CF$cv$params={r:'a00c0f05fb4473a2',t:'MTc3OTYyMzE3NA=='};var a=document.createElement('script');a.src='/cdn-cgi/challenge-platform/scripts/jsd/main.js';document.getElementsByTagName('head')[0].appendChild(a);";b.getElementsByTagName('head')[0].appendChild(d)}}if(document.body){var a=document.createElement('iframe');a.height=1;a.width=1;a.style.position='absolute';a.style.top=0;a.style.left=0;a.style.border='none';a.style.visibility='hidden';document.body.appendChild(a);if('loading'!==document.readyState)c();else if(window.addEventListener)document.addEventListener('DOMContentLoaded',c);else{var e=document.onreadystatechange||function(){};document.onreadystatechange=function(b){e(b);'loading'!==document.readyState&&(document.onreadystatechange=e,c())}}}})();
Text is read by the "Ask this paper" AI Q&A widget below.
Extraction quality varies by source — PMC NXML preserves structure
cleanly, OA-HTML may include some navigation residue, and OA-PDF can
have broken hyphenation. The publisher copy
(via DOI)
is the canonical version.