Machine Learning Prediction of Pharmacogenetic Test Uptake Among Opioid-Prescribed Patients Using Electronic Health Records: A Retrospective Cohort Study

preprint OA: closed
📄 Open PDF Full text JSON View at publisher

Abstract

Background Opioids are a widely prescribed class of medication for pain management. However, they have variable efficacy and adverse effects among patients, due to complex interplay between biological and clinical factors. Pharmacogenetic (PGx) testing can be utilized to match patients’ genetic profiles to individualize opioid therapy, improving pain relief and reducing the risk of adverse effects. Despite its potential, PGx uptake—utilization of PGx testing—remains low due to a range of barriers at the patient, health care provider, infrastructure, and financial levels. Since testing typically involves a shared decision between the provider and patient, predicting likelihood of patient undergoing PGx testing and understanding the factors influencing that decision can help optimize resource use and improve outcomes in pain management. Objective To develop machine learning (ML) models, identifying patients’ likelihood of PGx uptake based on their demographics, clinical variables, medication use, and social determinants of health (SDoH). Methods We utilized electronic health records (EHR) data from a single center healthcare system to identify patients prescribed opioids. We extracted patients’ demographics, clinical variables, medication use, and SDoH, and developed and validated ML models, including neural networks (NN), logistic regression (LR), random forests (RF), gradient boosting (XGB), naïve bayes (NB), and support vector machines (SVM) for PGx uptake prediction based on procedure codes. We performed 5-fold cross validation (CV) and created an ensemble probability-based classifier using the best-performing ML models for PGx uptake prediction. Various performance metrics, uptake stratification analysis, and feature importance analysis were employed to evaluate the performance of the models. Results The ensemble model using XGB and SVM-RBF classifiers had the highest C-statistics at 79.61%, followed by XGB (78.94%), and NN (78.05%). While XGB was the best-performing model, the ensemble model achieved a high accuracy (67.38%), recall (76.50%), specificity (67.25%), and negative predictive value (99.49%). The uptake stratification analysis using the ensemble model indicated that it can effectively distinguish across uptake probability deciles, where those in the higher strata are more likely to undergo PGx in real-world (6.59% in the highest decile compared to 0.12% in the lowest). Furthermore, SHAP value analysis using the XGB model indicated age, hypertension, and household income as the most influential factors for PGx uptake prediction. Conclusions The proposed ensemble model demonstrated a high performance in PGx uptake prediction among patients using opioids for pain. This model can be utilized as a decision support tool, assisting clinicians in identifying patients’ likelihood of PGx uptake and guiding appropriate decision-making.
Full text 48,416 characters · extracted from preprint-html · click to expand
Machine Learning Prediction of Pharmacogenetic Test Uptake Among Opioid-Prescribed Patients Using Electronic Health Records: A Retrospective Cohort Study | medRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-P4HH5NV'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search Machine Learning Prediction of Pharmacogenetic Test Uptake Among Opioid-Prescribed Patients Using Electronic Health Records: A Retrospective Cohort Study Mohammad Yaseliani , Je-Won Hong , Jiang Bian , Larisa Cavallari , Julio Duarte , Danielle Nelson , Wei-Hsuan Lo-Ciganic , Khoa Anh Nguyen , Md Mahmudul Hasan doi: https://doi.org/10.1101/2025.09.26.25336591 Mohammad Yaseliani 1 Department of Pharmaceutical Outcomes and Policy, College of Pharmacy, University of Florida , Gainesville, FL, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site Je-Won Hong 2 Department of Pharmacotherapy and Translational Research, College of Pharmacy, University of Florida , Gainesville, FL, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site Jiang Bian 3 School of Medicine, Indiana University , Indianapolis, IN, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site Larisa Cavallari 2 Department of Pharmacotherapy and Translational Research, College of Pharmacy, University of Florida , Gainesville, FL, USA 4 University of Florida Center for Pharmacogenomics and Precision Medicine , Gainesville, FL, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site Julio Duarte 2 Department of Pharmacotherapy and Translational Research, College of Pharmacy, University of Florida , Gainesville, FL, USA 4 University of Florida Center for Pharmacogenomics and Precision Medicine , Gainesville, FL, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site Danielle Nelson 5 College of Medicine, University of Florida , Gainesville, FL, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site Wei-Hsuan Lo-Ciganic 6 Division of General Internal Medicine, School of Medicine, University of Pittsburgh , Pittsburgh, PA, USA 7 Center for Pharmaceutical Policy and Prescribing, University of Pittsburgh , Pittsburgh, PA, USA 8 North Florida/South Georgia Veterans Health System; Geriatric Research Education and Clinical Center , Gainesville, FL, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site Khoa Anh Nguyen 2 Department of Pharmacotherapy and Translational Research, College of Pharmacy, University of Florida , Gainesville, FL, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site Md Mahmudul Hasan 1 Department of Pharmaceutical Outcomes and Policy, College of Pharmacy, University of Florida , Gainesville, FL, USA 9 Department of Information Systems and Operations Management, Warrington College of Business , Gainesville, FL, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site For correspondence: hasan.mdmahmudul{at}ufl.edu Abstract Full Text Info/History Metrics Data/Code Preview PDF Abstract Background Opioids are a widely prescribed class of medication for pain management. However, they have variable efficacy and adverse effects among patients, due to complex interplay between biological and clinical factors. Pharmacogenetic (PGx) testing can be utilized to match patients’ genetic profiles to individualize opioid therapy, improving pain relief and reducing the risk of adverse effects. Despite its potential, PGx uptake—utilization of PGx testing—remains low due to a range of barriers at the patient, health care provider, infrastructure, and financial levels. Since testing typically involves a shared decision between the provider and patient, predicting likelihood of patient undergoing PGx testing and understanding the factors influencing that decision can help optimize resource use and improve outcomes in pain management. Objective To develop machine learning (ML) models, identifying patients’ likelihood of PGx uptake based on their demographics, clinical variables, medication use, and social determinants of health (SDoH). Methods We utilized electronic health records (EHR) data from a single center healthcare system to identify patients prescribed opioids. We extracted patients’ demographics, clinical variables, medication use, and SDoH, and developed and validated ML models, including neural networks (NN), logistic regression (LR), random forests (RF), gradient boosting (XGB), naïve bayes (NB), and support vector machines (SVM) for PGx uptake prediction based on procedure codes. We performed 5-fold cross validation (CV) and created an ensemble probability-based classifier using the best-performing ML models for PGx uptake prediction. Various performance metrics, uptake stratification analysis, and feature importance analysis were employed to evaluate the performance of the models. Results The ensemble model using XGB and SVM-RBF classifiers had the highest C-statistics at 79.61%, followed by XGB (78.94%), and NN (78.05%). While XGB was the best-performing model, the ensemble model achieved a high accuracy (67.38%), recall (76.50%), specificity (67.25%), and negative predictive value (99.49%). The uptake stratification analysis using the ensemble model indicated that it can effectively distinguish across uptake probability deciles, where those in the higher strata are more likely to undergo PGx in real-world (6.59% in the highest decile compared to 0.12% in the lowest). Furthermore, SHAP value analysis using the XGB model indicated age, hypertension, and household income as the most influential factors for PGx uptake prediction. Conclusions The proposed ensemble model demonstrated a high performance in PGx uptake prediction among patients using opioids for pain. This model can be utilized as a decision support tool, assisting clinicians in identifying patients’ likelihood of PGx uptake and guiding appropriate decision-making. 1. Introduction Although effective for many, opioid prescribing presents a complex therapeutic challenge due to variable efficacy and the risk of adverse effects often resulting in a trial-and-error approach to opioid prescribing. These variations in opioid response often arise from interindividual differences in pharmacokinetics and pharmacodynamics, influenced in part by genetic variations [ 1 ]. Over recent decades, pharmacogenetic testing (PGx) has emerged as a promising strategy to tailor opioid therapy to an individual genetic profile, with the goal of enhancing pain relief, minimizing adverse effects, and improving overall patient outcomes. Several opioids, including codeine, tramadol, hydrocodone, oxycodone are metabolized—processed by the body–to varying extents by CYP2D6, an enzyme involved in the metabolism of many drugs. CYP2D6 is a highly polymorphic gene with over 130 identified variants that can result in varying enzyme activity, categorized into four phenotypes: poor metabolizer, intermediate metabolizer, normal metabolizer, and ultrarapid metabolizer. These phenotypes can influence the efficacy and side effects of drugs metabolized by CYP2D6 [ 2 ]. A pragmatic trial demonstrated improved composite pain intensity outcomes with CYP2D6-guided pain management, highlighting the potential benefits of personalized therapy based on pharmacogenetic testing [ 2 ]. Based on current evidence, the Clinical Pharmacogenetics Implementation Consortium (CPIC) provides guidelines on using CYP2D6 genotype result for prescribing tramadol, hydrocodone and codeine [ 2 ]. Despite increasing evidence supporting the benefits of PGx in improving patient outcome, its routine implementation faces significant challenges at multiple levels, including patient, healthcare provider, infrastructure, and financial barriers. A major obstacle is the knowledge gap among healthcare professionals. Many providers lack sufficient training in pharmacogenetics, hindering their ability to effectively integrate genetic testing into their prescribing practice [ 3 ]. This educational deficit is further complicated by the absence of standardized clinical workflows to incorporate pharmacogenetic data seamlessly [ 4 , 5 ]. Additionally, regional- and patient-level social determinants of health (SDoH), such as socioeconomic status, insurance coverage, and access to healthcare, can limit patient access to PGx testing and personalized care [ 6 ]. As a result, PGx remains underutilized in opioid prescribing. Predicting a likelihood of patient undergoing PGx testing and identifying the underlying factors influencing that decision is crucial for enhancing the clinical utility and cost-effectiveness of PGx. Since testing often involves a shared decision between the provider and patient following a discussion of potential benefits and risks, understanding these dynamics can help healthcare providers develop targeted strategies to encourage appropriate use of PGx testing. This, in turn, can lead to improved patient outcomes in pain management [ 7 ]. Predicting PGx uptake using electronic health records (EHRs) presents significant challenges because of the complexity and diversity of clinical data, including, laboratory results, genetic information, and patient-reported outcomes. Additionally, demographics and SDoH—such as race, gender, socioeconomic status, environmental exposures, access to healthcare, and lifestyle choices—significantly impact PGx uptake [ 8 ]. These diverse data sources provide valuable insights but also pose substantial challenges because their relationships with PGx uptake are often non-linear and complex [ 9 ]. Traditional regression models assume linear relationships, which limits their predictive accuracy in this context. This limitation can lead to inaccurate predictions, as conventional methods fail to capture the intricate nonlinear relationships present in complex data. Machine learning (ML) techniques such as neural networks (NN), random forests (RF), and logistic regression (LR) have been increasingly employed to enhance predictive accuracy by identifying and understanding nonlinear relationships within high-dimensional data [ 10 ]. While ML has gained significant attention in medicine, particularly in pharmacogenomics, its application has primarily focused on analyzing genetic or proteomic data to identify patterns associated with drug response [ 11 ]. This narrow focus overlooks the potential value of integrating a broader range of relevant data types to better facilitate PGx uptake. By leveraging ML algorithms to process extensive and complex data, we can more accurately identify patients with the highest likelihood of undergoing PGx, enabling a more personalized and effective approach to opioid therapy. In this study, we aimed to develop and validate ML algorithms to identify opioid-prescribed patients who are most likely to undergo PGx test. By analyzing clinical data, demographics, medication use, and SDoH, our models seek to identify factors influencing the uptake of PGx in opioid therapy. We hypothesize that ML-driven predictions can significantly enhance the targeted use of PGx in patients receiving opioids, leading to improved pain management and optimized therapeutic outcomes. Ultimately, our models may facilitate better utilization of PGx testing for opioid use in pain management. This paper outlines our methodological approach, findings, and potential implications for clinical practice in the field of pain management. 2. Method The overall ML pipeline for PGx uptake prediction is shown in Fig. 1 . The pipeline involves four steps, including study design, data pre-processing and feature selection, model development, and comprehensive evaluation. Each step involves several tasks that are critical to building and evaluating the final predictive model. The details of each step are described in this section. Download figure Open in new tab Fig. 1. Overall ML pipeline for PGx uptake prediction. 2.1. Data source and study design We utilized real world electronic health record (EHR) data available from the University of Florida Health Integrated Data Repository (UF Health IDR) to develop the models. Patients’ age range was 18 to 89 years, and we did not exclude any patients based on age. Patients were included if they were between 18 and 89 years had an opioid prescription order for non-cancer treatment from 2011 to 2020. Patients were stratified into an intervention group (patient with a minimum of one PGx order) and control group (those with no history of PGx order). We used current procedural terminology (CPT) codes to determine patients’ PGx test. To create the cohort for ML model development, we defined the index dates separately for intervention and control groups. The index date for intervention group was defined as the most recent date of an opioid prescription prior to PGx testing. The index date for control group was defined as the earliest date of opioid prescription. All baseline covariates were collected in the one-year period prior to the index date. The outcome was a binary target variable, indicating whether the patient was in the intervention or control group based on their PGx test record. If a patient was in the intervention group, their outcome value was encoded as 1 and otherwise, 0. 2.2. Feature selection and data pre-processing We selected four different categorizes of features for our study sample: (i) demographics, (ii) clinical history, (iii) medication use, and (iv) SDoH. Demographics included age, gender, and race, where age was continuous, sex was binary, and race was multi-category which was one-hot-encoded to get separate binary inputs for each racial group and prevent the ordinality assumption by the model. We used the international classification of disease (ICD-9 and ICD-10) codes to determine the clinical history variables, including a diagnosis of AIDS/HIV, alcohol abuse, blood loss anemia, cardiac arrhythmias, chronic pulmonary disease, coagulopathy, systolic (congestive) heart failure, deficiency anemia, depression, diabetes with complications, diabetes without complications, drug abuse, fluid and electrolyte disorder, hypertension, hypothyroidism, liver disease, lymphoma, metastatic cancer, neurodegenerative disorders, obesity, paralysis, peptic ulcer disease, peripheral vascular disease, psychosis, pulmonary circulation disorders, renal failure, rheumatoid arthritis/collagen, solid tumor without metastasis, valvular disease, and weight loss. Furthermore, we included Elixhauser comorbidity index [ 12 ] to calculate the overall healthcare burden for each patient. In addition, the SDoH information included average household size, Gini index as a measure of income inequality, median household income, and median rent, all linked to UF Health EHR based on year and ZIP code using the Agency for Healthcare Research and Quality (AHRQ) SDoH database [ 13 ]. We removed all the missing data, so all patients had complete information for model development. 2.3. Model development and hyperparameter optimization We adopted a comprehensive strategy for model development using LR, XGBoost [ 14 ], RF [ 15 ], support vector machines (SVM) [ 16 ] with linear and radial basis function (RBF) kernels (i.e., LSVM and SVM-RBF), and NNs to capture predictive performance across a wide range of models for PGx uptake prediction. LR was first developed due to its ability to capture linear relationships in the data and providing baseline performance. Similarly, Naïve bayes (NB) model was developed due to its simplicity and effectiveness in high-dimensional spaces. However, considering LR’s and NB’s limitations in capturing complex non-linear relationships, we trained tree-based models, including XGBoost and RF, given their ability to capture non-linear interactions that may be critical to accurately predicting the PGx uptake. Linear SVM and SVM-RBF were developed to capture high-dimensional complex interactions among input features. Finally, NNs were trained to explore more complex interactions among features that may be missed by traditional models. To train the models, we randomly partitioned the data with stratified 80/20 train/test split. To mitigate the risk of overfitting and get the best performance out of ML models, we conducted 5-fold cross validation. For LR, we tuned the parameter C with values of 0.01, 0.1, 1, and 10. To tune NB, we used smoothing parameters ranging from 10 −9 to 10 −5 with 10x increments. Both XGBoost and RF models were tuned with 100, 200, 300, 400, and 500 trees to explore the effect of different numbers of trees on performance. Linear SVM and SVM-RBF were optimized with C values equal to 0.01, 0.1, 1, and 10. Finally, NNs were trained with all possible combinations of various optimizers and learning rates. The optimizers included stochastic gradient descent (SGD) [ 17 ], Adam [ 18 ], and Nadam [ 19 ], and learning rates ranged from 10 −5 to 10 −1 with 10x increases. The best hyperparameters were selected based on the highest C-statistics achieved on the training set during cross-validation process. We trained all the models using the best-performing hyperparameters and evaluated their performance on the test set. Notably, we employed a balanced class weighting approach, assigning proportionally higher weights to minority class, to decrease the risk of overfitting when training each of the models. To improve the performance and enhance generalizability of the models, we created a weighted probability-based ensemble classifier. To this end, we searched for the best individual model weights using AUC-ROC as the performance metrics, conditioned on having a total sum of weights equal to 1. This ensemble model was selected as the final best model for PGx uptake prediction. 2.4. Comprehensive evaluation We followed TRIPOD+AI guidelines [ 20 ] to evaluate the performance of ML models. Area under receiver operating characteristics curve (AUC-ROC) or C-statistics demonstrates the discriminative ability of the model, considering the trade-off between sensitivity and specificity at different threshold values. DeLong test [ 21 , 22 ] is used to identify whether there is a statistically significant difference between AUC values. Accuracy provides the overall performance of the model by calculating proportion of true negative and true positive cases compared to all the samples in the data. However, this measure may be misleading for rare events and not suitable for highly imbalanced datasets. To address this, we calculated specificity, which measures the model’s ability to correctly identify true negative cases and recall or sensitivity, indicating model’s performance for identifying true positive cases. Since model performance may not be optimal, we use Youden index [ 23 – 25 ] to obtain the best performing classification threshold for each model, where the highest value of ‘specificity+sensitivity-1’ is achieved. To further evaluate model performance for real-world applications, we calculated the number needed to evaluate and predictive positives per 100 patients, showing how many predictions must be made to identify one actual positive case and the number of PGx uptake predictions per 100 individuals, respectively. Moreover, we did stratification analysis, where PGx uptake probabilities were categorized into 10 deciles in ascending order and the percentage of actual PGx uptake in the data was identified in each decile to evaluate the ability of the model in identifying more uptakes in the higher strata. In addition, we used Shapley Additive exPlanations (SHAP) [ 26 ] analysis to report feature importance and identify the features that are most influential to PGx uptake. 3. Results We used data from UF Health IDR with patients from a wide of range of demographic and clinical backgrounds. The data included a total of 455,773 patients, 7,645 (1.68%) with a recorded PGx test and 448,128 (98.32%) without. After removing 160,914 further patients who had no opioid prescription, there were 294,859 patients using opioids in the cohort. We further excluded patients with no race information and SDoH variables (after matching with AHRQ database). Overall, the final cohort included 242,640 patients, where 3,510 had a recorded PGx uptake and 239,130 did not. The cohort had an average age of 58 with a standard deviation of 17.47 years, a gender distribution comprising 41.49% as males and 58.51% as females, and a racial distribution of 64.94% White, 28.77% Black, and 6.29% belonging to other racial groups (i.e., Hispanic, White Hispanic, Black Hispanic, Asian, Pacific Islander, American Indian, Multiracial, and other). In addition, the cohort had an average household size of 2.58, a Gini index of 0.45, and an average median income of $46,324. Table 1 presents the key sociodemographic characteristics of patients in the cohort. Moreover, the distribution of patient characteristics in the training and test sets are provided in Table 2 . View this table: View inline View popup Table 1. Sociodemographic summary of patients. View this table: View inline View popup Download powerpoint Table 2. Distribution of patient characteristics in the training and test sets. 3.1. Model performance The ROC curves of all models and their corresponding AUC or C-statistics with 95% confidence intervals (CIs) are provided in Fig. 2 . The ensemble model achieved the highest C-statistics of 79.61% using 0.7 and 0.3 weights for XGB and SVM-RBF, respectively. In contrast, NB achieved the lowest C-statistics at 72.49%. Other models had comparable C-statistics, equal to 75.38% (RF), 76.72% (LR), 76.46% (LSVM), 77.73% (SVM-RBF), 78.05% (NN), and 78.94% (XGB). Additionally, the DeLong test results indicated that there was a statistically significant difference (p <0.05) between the C-statistics of ensemble model and all other classifiers, showing its higher performance is less unlikely to be due to chance. Download figure Open in new tab Fig. 2. ROC curves of developed models. Since model performance was not optimal, we used Youden index [ 23 – 25 ] to obtain the best performing classification threshold for each model. Fig. 3 portrays the confusion matrices for all developed models based on their Youden index threshold. Table 3 provides a summary of performance for all developed models. Accuracy representing the proportion of correctly classified cases, was highest for XBG (72.54%), followed by RF (71.31%), LSVM (71.10%), and SVM-RBF (70.60%). Other classifiers had 65-70% accuracy with ensemble model having an accuracy of 67.38%. View this table: View inline View popup Download powerpoint Table 3. Pefromance metrics of developed models based on Youden index. Download figure Open in new tab Fig. 3. Confusion matrix of models based on Youden index. (A) NN; (B) LR; (C) RF; (D) XGB; (E) NB; (F) LSVM; (G) SVM_RBF; (H) Ensemble model. To assess the model performance in identifying true positive cases (i.e., those with a recorded PGx uptake) and true negative cases (i.e., those without a recorded PGx uptake), recall and specificity were calculated. Recall values ranged from 66.24-77.78% with NN model achieving the highest recall and ensemble model achieving 76.50%. Specificity values ranged from 65.57-72.57%, with XGB model achieving the highest specificity. To further assess models’ performance in distinguishing between true positives and true negatives, negative predictive value (NPV) and positive predictive value (PPV) were calculated. NPV values ranged from 99.28-99.51% with NN model having the highest NPV, while PPV values ranged from 2.99-3.43% with SVM-RBF achieving the highest value. Stratification analysis results are presented in Fig. 4 . The analysis was performed by dividing the predicted probabilities of PGx uptake into deciles and calculating observed uptake rates within each to assess performance across the probability groups. The lowest decile was decile 1 with lowest probability values, while decile 10 represented the highest probability values. Within each decile, observed uptake rates were calculated to plot against the probability deciles. The observed event rates demonstrated an increasing trend across the probability deciles (except for the 5 th ). The event rate in the highest probability decile was 6.59% while 0.12% in the lowest decile. Download figure Open in new tab Fig. 4. Stratification analysis for the probability of PGx uptake. Fig. 5 demonstrates the plots of recall values against predictive positives per 100 and number needed to evaluate. Predictive positives per 100 represents the number of individuals per 100 who are predicted to undergo PGx across different recall values. The number needed to evaluate indicates the number of individuals that need to be assessed by the model in order to find one true positive (i.e., recorded PGx uptake). Download figure Open in new tab Fig. 5. Recall vs. (A) predicted positives per 100, (B) number needed to evaluate. SHAP analysis for feature importance using the XGB model has been shown in Fig. 6 . Age had the highest impact on PGx uptake, where older patients are more likely to undergo PGx. Hypertension and median household income were the next most important features, both associated with a higher likelihood of PGx uptake. Other SDoH factors were among the top 7 important factors for PGx uptake. In contrast, racial group variables (i.e., Hispanic, Pacific Islander, and White Hispanic) had the lowest influence on the outcome. Download figure Open in new tab Fig. 6. SHAP analysis for feature importance (The x-axis shows the impact on prediction and the color represents the feature value (i.e., red for high and blue for low)). Discussion The results of this study demonstrate that the ensemble model achieved the highest AUC compared to other models. Although the XGB model, when evaluated at its Youden index performed better than other models in terms of accuracy, recall, specificity, and NPV, we selected the ensemble model as the final classifier. This was because the ensemble model harnesses the strengths of both high-performing XGB and SVM-RBF models for prediction. It is notable that although the study used PPV as a performance metric, it is highly dependent on outcome prevalence and may not generalize across different patient populations or clinical settings, reducing its applicability in decision-making. Overall, the ensemble model proves to be a viable tool for accurately predicting PGx uptake, thereby supporting more informed, data-driven decision-making in clinical settings. By providing precise and individualized PGx uptake probabilities, clinicians will be able to identify individuals who are less likely to undergo PGx testing. By prioritizing these patients, the model has the potential to assist clinicians in optimizing resource allocation based on patient needs and facilitating PGx uptake, ultimately improving patients’ pain management outcomes. The dataset used for model development exhibited class imbalance, with a relatively low percentage of patients having recorded PGx uptake. Despite utilizing 5-fold CV for hyperparameter optimization and a balanced weighting approach for final model development, this imbalance resulted in a lower PPV compared to other performance metrics. This means the model has a higher number of false positives, meaning that many of patients who are predicted to undergo PGx testing will not, potentially leading to inappropriate opioid treatment for the patients. Conversely, the class imbalance led to a very high NPV, meaning that most no-uptake predictions are accurate. This high NPV could be advantageous in clinical decision-making, as it allows for the allocation of more resources and encouragement plans for patients predicted not to undergo PGx testing. By identifying characteristics of these patients, targeted strategies can be developed to increase their PGx uptake rate, ultimately leading to improved patient outcomes. This study utilized various clinical and demographic variables to develop more-informed ML models. However, the underrepresentation of certain racial groups, such as Hispanic and Pacific Islander could potentially affect models’ outcomes. This raises concerns about the bias and fairness issues, as ML models may exhibit lower performance for these groups compared to others. Additionally, while the study incorporated SDoH information as potential predictors of PGx uptake, this information was based on ZIP code-level data, which may lack the necessary granularity required for precise prediction. Detailed individual-level data on variables such as income, education, and household size could enhance model’s reliability and improve the accurate identification of each feature’s contribution on the outcome. Another limitation of the study is its generalizability to other populations in different clinical settings. The model’s performance may not directly translate to external populations with varying demographic and clinical backgrounds. Specifically, the model was trained on data from a single EHR system including minority racial groups and the results may not be applicable to other racial groups. Therefore, caution is warranted when interpreting the model’s outcomes for these populations. In addition, some of the patients in the data who had a recorded PGx uptake were part of a clinical trial where the cost of testing was covered. This would potentially influence the ability of the model in identifying the true impact of patients’ economic and income status on undergoing PGX testing. To demonstrate model’s broader applicability, future work should focus on external validation using independent datasets from different clinical settings. This would help evaluate model’s robustness and clinical utility for real-world decision-making. The SHAP value analysis using the XGB model highlighted the contribution of each feature to individual predictions, enhancing model’s clinical credibility and trustworthiness. The most influential features were age, hypertension, and household income, suggesting that these features require more attention for increasing PGx uptake among patients. Notably, PGx uptake is not necessarily due to opioid prescribing and could be due to prescribing of other medications and participation in a research project, potentially affecting the study results. Thus, although the results demonstrate model’s transparency and clinical utility, more research is needed before integrating the model into a real-world clinical decision support system. While SHAP diagram ranks features based on their overall importance, it is crucial to examine critical factors for each patient on a case-by-case basis when making clinical decisions. Overall feature importance does not always reflect individual patient’s conditions and their influence on the outcome. The uptake stratification analysis showed that the ensemble model effectively differentiated patients across PGx uptake deciles, with higher uptake probabilities corresponding to higher observed uptake rates. Such an analysis can assist clinicians in categorizing patients into different deciles based on their likelihood of PGx uptake and prioritize those in lower deciles (less likely for the PGx uptake) for optimal resource allocation and improved access to testing. In addition, it can help clinicians authorize insurance coverage and provide post-test consultation for patients in higher uptake deciles. While there was an overall increasing trend in uptake rates as the probabilities increased, there was a slight decrease (i.e., 0.002) in 5 th decile, indicating that the higher probabilities in this decile could not capture a higher uptake rate. In addition, stratification analysis often relies on pre-defined thresholds (i.e., 10% in this study) to categorize probabilities into deciles, which may not be aligned with the way clinicians make decisions and could limit its utility in clinical decision-making. Despite these limitations, the stratification analysis supports the potential use of the ensemble model in prioritizing patients based on their likelihood of undergoing PGx, leading to optimized resource allocation and improved patient outcomes. The implementation of the PGx uptake prediction model addresses the barriers to PGx testing in opioid therapy. By providing probability of PGx uptake at the individual patient level, clinicians can strategically optimize resource allocation across their patient population. For patients with a high predicted probability of PGx uptake, the clinical workflows can be expedited by focusing on authorizing insurance coverage or allocating clinical pharmacist time for post-test consultation. This efficient deployment of resources where they are most likely to be utilized allows practices to simultaneously redirect access support toward patients with lower uptake probabilities. Such support may include patient education on the benefits and implications of PGx testing or addressing financial barriers. The model’s ability to stratify patient’s likelihood of PGx uptake transforms the current trial-and-error paradigm of opioid prescribing into a more systematic framework for clinical decision-making. Conclusions PGx is a viable tool for matching patient’s genetic profile to suitable opioids for pain treatment. This study proposed ML models for PGx uptake prediction using data from an EHR system. Results demonstrated that the ensemble ML model combining XGB and SVM-RBF classifiers achieved the highest AUC at 79.61%, making it a reliable prediction model for PGx uptake prediction. Additionally, the uptake stratification and feature importance analysis using SHAP values further indicated model’s utility for real-world applications. Following further validation using external dataset, this model can be integrated into a clinical decision support system, enabling clinicians in more informed, risk-averse clinical decisions, ultimately improving patient outcomes. Data Availability The data cannot be made available to the readers. Conflicts of Interest None declared. Footnotes ↵ * Khoa Anh Nguyen and Md Mahmudul Hasan share senior authorship of this manuscript References [1]. ↵ O. Dale , K. Moksnes , and S. Kaasa , “ European Palliative Care Research Collaborative pain guidelines: opioid switching to improve analgesia or reduce side effects. A systematic review ,” (in eng), Palliat Med , vol. 25 , no. 5 , pp. 494 – 503 , Jul 2011 , doi: 10.1177/0269216310384902 . OpenUrl CrossRef PubMed [2]. ↵ K. R. Crews et al. , “ Clinical Pharmacogenetics Implementation Consortium Guideline for CYP2D6, OPRM1, and COMT Genotypes and Select Opioid Therapy ,” (in eng), Clin Pharmacol Ther , vol. 110 , no. 4 , pp. 888 – 896 , Oct 2021 , doi: 10.1002/cpt.2149 . OpenUrl CrossRef PubMed [3]. ↵ A. A. Lemke et al. , “ Primary care physician experiences with integrated pharmacogenomic testing in a community health system ,” (in eng), Per Med , vol. 14 , no. 5 , pp. 389 – 400 , Sep 2017 , doi: 10.2217/pme-2017-0036 . OpenUrl CrossRef PubMed [4]. ↵ B. M. Vest , L. O. Wray , M. E. Thase , L. A. Brady , S. R. Chapman , and D. W. Oslin , “ Providers’ Use of Pharmacogenetic Testing to Inform Antidepressant Prescribing: Results of Qualitative Interviews ,” (in eng), Psychiatr Serv , vol. 74 , no. 12 , pp. 1270 – 1276 , Dec 1 2023 , doi: 10.1176/appi.ps.20220537 . OpenUrl CrossRef PubMed [5]. ↵ A. M. Zebrowski et al. , “ Qualitative study of system-level factors related to genomic implementation ,” (in eng), Genet Med , vol. 21 , no. 7 , pp. 1534 – 1540 , Jul 2019 , doi: 10.1038/s41436-018-0378-9 . OpenUrl CrossRef PubMed [6]. ↵ S. Shaaban and Y. Ji , “ Pharmacogenomics and health disparities, are we helping? ,” (in eng), Front Genet , vol. 14 , p. 1099541 , 2023 , doi: 10.3389/fgene.2023.1099541 . OpenUrl CrossRef PubMed [7]. ↵ C. Slomp et al. , “ Pharmacogenomic Testing for Major Depression: A Qualitative Study of the Perceptions of People with Lived Experience and Professional Stakeholders ,” The Canadian Journal of Psychiatry , vol. 68 , no. 6 , pp. 436 – 452 , 2023/06/01 2022 , doi: 10.1177/07067437221140383 . OpenUrl CrossRef [8]. ↵ B. H. Park , A. A. Macias , K. M. Fisch , S. Simpson , J. Chen , and R. A. Gabriel , “The Association of Social Determinants of Health With Reported Daily Pain Scores Among Patients With a History of Chronic Pain Using the “All of Us” Research Program ,” (in eng), Anesth Analg , vol. 140 , no. 3 , pp. 732 – 735 , Mar 1 2025 , doi: 10.1213/ane.0000000000007211 . OpenUrl CrossRef PubMed [9]. ↵ H. Chen Jonathan and M. Asch Steven , “ Machine Learning and Prediction in Medicine Beyond the Peak of Inflated Expectations ,” New England Journal of Medicine , vol. 376 , no. 26 , pp. 2507 – 2509 , doi: 10.1056/NEJMp1702071 . OpenUrl CrossRef PubMed [10]. ↵ R. C. Deo , “ Machine Learning in Medicine ,” (in eng), Circulation , vol. 132 , no. 20 , pp. 1920 – 30 , Nov 17 2015 , doi: 10.1161/circulationaha.115.001593 . OpenUrl Abstract / FREE Full Text [11]. ↵ E. Lin , C.-H. Lin , and H.-Y. Lane , “ Machine Learning and Deep Learning for the Pharmacogenomics of Antidepressant Treatments ,” (in eng), Clinical Psychopharmacology and Neuroscience , vol. 19 , no. 4 , pp. 577 – 588 , 2021 , doi: 10.9758/cpn.2021.19.4.577 . OpenUrl CrossRef [12]. ↵ C. van Walraven , P. C. Austin , A. Jennings , H. Quan , and A. J. Forster , “ A modification of the Elixhauser comorbidity measures into a point system for hospital death using administrative data ,” (in eng), Med Care , vol. 47 , no. 6 , pp. 626 – 33 , Jun 2009 , doi: 10.1097/MLR.0b013e31819432e5 . OpenUrl CrossRef PubMed [13]. ↵ Agency for Healthcare Research and Quality (AHRQ ). “Social Determinants of Health Database.” https://www.ahrq.gov/sdoh/data-analytics/sdoh-data.html#download (accessed April 4, 2025 ). [14]. ↵ T. Chen and C. Guestrin , “ XGBoost: A Scalable Tree Boosting System ,” presented at the Proceedings of the 22nd ACM SIGKDD International Conference on Knowledge Discovery and Data Mining, San Francisco, California, USA , 2016 . [Online]. Available : doi: 10.1145/2939672.2939785 . OpenUrl CrossRef [15]. ↵ L. Breiman , “ Random Forests ,” Machine Learning , vol. 45 , no. 1 , pp. 5 – 32 , 2001/10/01 2001 , doi: 10.1023/A:1010933404324 . OpenUrl CrossRef [16]. ↵ J. Cervantes , F. Garcia-Lamont , L. Rodríguez-Mazahua , and A. Lopez , “ A comprehensive survey on support vector machine classification: Applications, challenges and trends ,” Neurocomputing , vol. 408 , pp. 189 – 215 , 2020/09/30/ 2020 , doi: 10.1016/j.neucom.2019.10.118 . OpenUrl CrossRef [17]. ↵ S. Ruder , “An overview of gradient descent optimization algorithms,” 09/15 2016 , doi: 10.48550/arXiv.1609.04747 . OpenUrl CrossRef [18]. ↵ D. Kingma and J. Ba , “ Adam: A Method for Stochastic Optimization ,” International Conference on Learning Representations , 12/22 2014 . [19]. ↵ T. Dozat , “Incorporating Nesterov Momentum into,” 2015 . [20]. ↵ G. S. Collins et al. , “ TRIPOD+AI statement: updated guidance for reporting clinical prediction models that use regression or machine learning methods ,” BMJ , vol. 385 , p. e078378 , 2024 , doi: 10.1136/bmj-2023-078378 . OpenUrl FREE Full Text [21]. ↵ E. R. DeLong , D. M. DeLong , and D. L. Clarke-Pearson , “ Comparing the areas under two or more correlated receiver operating characteristic curves: a nonparametric approach ,” (in eng), Biometrics , vol. 44 , no. 3 , pp. 837 – 45 , Sep 1988 . OpenUrl CrossRef PubMed Web of Science [22]. ↵ X. Sun and W. Xu , “ Fast Implementation of DeLong’s Algorithm for Comparing the Areas Under Correlated Receiver Operating Characteristic Curves ,” IEEE Signal Processing Letters , vol. 21 , no. 11 , pp. 1389 – 1393 , 2014 , doi: 10.1109/LSP.2014.2337313 . OpenUrl CrossRef [23]. ↵ W. J. Youden , “ Index for rating diagnostic tests ,” (in eng), Cancer , vol. 3 , no. 1 , pp. 32 – 5 , Jan 1950 , doi: 10.1002/1097-0142(1950)3:13.0.co;2-3 . OpenUrl CrossRef PubMed Web of Science [24]. R. Fluss , D. Faraggi , and B. Reiser , “ Estimation of the Youden Index and its associated cutoff point ,” (in eng), Biom J , vol. 47 , no. 4 , pp. 458 – 72 , Aug 2005 , doi: 10.1002/bimj.200410135 . OpenUrl CrossRef PubMed Web of Science [25]. ↵ E. F. Schisterman , D. Faraggi , B. Reiser , and J. Hu , “ Youden Index and the optimal threshold for markers with mass at zero ,” (in eng), Stat Med , vol. 27 , no. 2 , pp. 297 – 315 , Jan 30 2008 , doi: 10.1002/sim.2993 . OpenUrl CrossRef PubMed Web of Science [26]. ↵ S. M. Lundberg and S.-I. Lee , “ A unified approach to interpreting model predictions ,” presented at the Proceedings of the 31st International Conference on Neural Information Processing Systems , Long Beach, California, USA , 2017 . View the discussion thread. Back to top Previous Next Posted September 28, 2025. Download PDF Data/Code Email Thank you for your interest in spreading the word about medRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following Machine Learning Prediction of Pharmacogenetic Test Uptake Among Opioid-Prescribed Patients Using Electronic Health Records: A Retrospective Cohort Study Message Subject (Your Name) has forwarded a page to you from medRxiv Message Body (Your Name) thought you would like to see this page from the medRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share Machine Learning Prediction of Pharmacogenetic Test Uptake Among Opioid-Prescribed Patients Using Electronic Health Records: A Retrospective Cohort Study Mohammad Yaseliani , Je-Won Hong , Jiang Bian , Larisa Cavallari , Julio Duarte , Danielle Nelson , Wei-Hsuan Lo-Ciganic , Khoa Anh Nguyen , Md Mahmudul Hasan medRxiv 2025.09.26.25336591; doi: https://doi.org/10.1101/2025.09.26.25336591 Share This Article: Copy Citation Tools Machine Learning Prediction of Pharmacogenetic Test Uptake Among Opioid-Prescribed Patients Using Electronic Health Records: A Retrospective Cohort Study Mohammad Yaseliani , Je-Won Hong , Jiang Bian , Larisa Cavallari , Julio Duarte , Danielle Nelson , Wei-Hsuan Lo-Ciganic , Khoa Anh Nguyen , Md Mahmudul Hasan medRxiv 2025.09.26.25336591; doi: https://doi.org/10.1101/2025.09.26.25336591 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Health Informatics Subject Areas All Articles Addiction Medicine (570) Allergy and Immunology (864) Anesthesia (302) Cardiovascular Medicine (4445) Dentistry and Oral Medicine (444) Dermatology (383) Emergency Medicine (609) Endocrinology (including Diabetes Mellitus and Metabolic Disease) (1515) Epidemiology (15236) Forensic Medicine (30) Gastroenterology (1127) Genetic and Genomic Medicine (6610) Geriatric Medicine (669) Health Economics (1000) Health Informatics (4549) Health Policy (1370) Health Systems and Quality Improvement (1613) Hematology (543) HIV/AIDS (1266) Infectious Diseases (except HIV/AIDS) (15926) Intensive Care and Critical Care Medicine (1104) Medical Education (623) Medical Ethics (147) Nephrology (668) Neurology (6613) Nursing (346) Nutrition (999) Obstetrics and Gynecology (1147) Occupational and Environmental Health (957) Oncology (3341) Ophthalmology (975) Orthopedics (369) Otolaryngology (420) Pain Medicine (436) Palliative Medicine (130) Pathology (665) Pediatrics (1694) Pharmacology and Therapeutics (693) Primary Care Research (714) Psychiatry and Clinical Psychology (5458) Public and Global Health (9244) Radiology and Imaging (2205) Rehabilitation Medicine and Physical Therapy (1370) Respiratory Medicine (1197) Rheumatology (596) Sexual and Reproductive Health (715) Sports Medicine (530) Surgery (713) Toxicology (99) Transplantation (289) Urology (265) (function(){function c(){var b=a.contentDocument||a.contentWindow.document;if(b){var d=b.createElement('script');d.innerHTML="window.__CF$cv$params={r:'a024bba218de1b23',t:'MTc3OTg4MTkwMg=='};var a=document.createElement('script');a.src='/cdn-cgi/challenge-platform/scripts/jsd/main.js';document.getElementsByTagName('head')[0].appendChild(a);";b.getElementsByTagName('head')[0].appendChild(d)}}if(document.body){var a=document.createElement('iframe');a.height=1;a.width=1;a.style.position='absolute';a.style.top=0;a.style.left=0;a.style.border='none';a.style.visibility='hidden';document.body.appendChild(a);if('loading'!==document.readyState)c();else if(window.addEventListener)document.addEventListener('DOMContentLoaded',c);else{var e=document.onreadystatechange||function(){};document.onreadystatechange=function(b){e(b);'loading'!==document.readyState&&(document.onreadystatechange=e,c())}}}})();

Text is read by the "Ask this paper" AI Q&A widget below. Extraction quality varies by source — PMC NXML preserves structure cleanly, OA-HTML may include some navigation residue, and OA-PDF can have broken hyphenation. The publisher copy (via DOI) is the canonical version.

My notes (saved in your browser only)

Ask this paper AI returns verbatim quotes from the full text · source: preprint-html

Answers must be backed by verbatim quotes from this paper's full text. Hallucinated quotes are dropped automatically; if no verbatim passage answers the question, we say so. How this works

Citation neighborhood (no data yet)

We don't have any in-corpus citations linked to this paper yet. This is a recent paper (2025) — citers typically take a year or two to land, and the OpenAlex reference graph may still be filling in.

Source provenance

europepmc
last seen: 2026-05-20T01:45:00.602351+00:00