Full text
46,770 characters
· extracted from
preprint-html
· click to expand
Machine learning approach to dissect the clinical heterogeneity of IBD-associated fatigue | medRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-P4HH5NV'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search Machine learning approach to dissect the clinical heterogeneity of IBD-associated fatigue View ORCID Profile Cher S Chuah , Rebecca Hall , View ORCID Profile Robert J Whelan , Peter D Cartlidge , View ORCID Profile Beatriz Gros , View ORCID Profile Eva Iglesias-Flores , View ORCID Profile Nikita Parkash , View ORCID Profile Ray K Boyapati , View ORCID Profile Clara Ramos-Belinchon , View ORCID Profile Solomon Ong , View ORCID Profile Emily F Brownson , Iona AM Campbell , View ORCID Profile Craig Mowat , View ORCID Profile John P Seenan , View ORCID Profile Jonathan C MacDonald , View ORCID Profile Gwo-Tzer Ho doi: https://doi.org/10.1101/2025.08.14.25333676 Cher S Chuah 1 School of Infection and Immunity, University of Glasgow , Scotland, United Kingdom Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Cher S Chuah Rebecca Hall 2 Institute of Regeneration and Repair, University of Edinburgh , Scotland, United Kingdom Find this author on Google Scholar Find this author on PubMed Search for this author on this site Robert J Whelan 2 Institute of Regeneration and Repair, University of Edinburgh , Scotland, United Kingdom Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Robert J Whelan Peter D Cartlidge 2 Institute of Regeneration and Repair, University of Edinburgh , Scotland, United Kingdom Find this author on Google Scholar Find this author on PubMed Search for this author on this site Beatriz Gros 3 Reina Sofía University Hospital. IMIBIC. University of Cordoba. Córdoba , Spain 4 Centro de Investigación Biomédica en Red Enfermedades Hepáticas y Digestivas , CIBEREHD, Madrid, Spain Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Beatriz Gros Eva Iglesias-Flores 3 Reina Sofía University Hospital. IMIBIC. University of Cordoba. Córdoba , Spain Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Eva Iglesias-Flores Nikita Parkash 5 Department of Gastroenterology , Monash Health, Clayton, Victoria, Australia Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Nikita Parkash Ray K Boyapati 6 Faculty of Medicine, Nursing and Health Sciences, Monash University , Clayton, Victoria, Australia Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Ray K Boyapati Clara Ramos-Belinchon 7 Western General Hospital , Edinburgh, Scotland, United Kingdom Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Clara Ramos-Belinchon Solomon Ong 2 Institute of Regeneration and Repair, University of Edinburgh , Scotland, United Kingdom Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Solomon Ong Emily F Brownson 8 NHS Greater Glasgow and Clyde , Scotland, United Kingdom Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Emily F Brownson Iona AM Campbell 8 NHS Greater Glasgow and Clyde , Scotland, United Kingdom Find this author on Google Scholar Find this author on PubMed Search for this author on this site Craig Mowat 9 University of Dundee School of Medicine , Dundee, Scotland, United Kingdom Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Craig Mowat John P Seenan 8 NHS Greater Glasgow and Clyde , Scotland, United Kingdom Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for John P Seenan Jonathan C MacDonald 8 NHS Greater Glasgow and Clyde , Scotland, United Kingdom Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Jonathan C MacDonald Gwo-Tzer Ho 1 School of Infection and Immunity, University of Glasgow , Scotland, United Kingdom Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Gwo-Tzer Ho For correspondence: Gwo-tzer.ho{at}glasgow.ac.uk Abstract Full Text Info/History Metrics Data/Code Preview PDF Abstract Extreme fatigue is a clinical symptom that affects >50% of individuals with Inflammatory Bowel Disease (IBD), with a similar prevalence across many common immune-mediated inflammatory diseases (IMIDs). Despite its ubiquity, human scientific studies have yet to explain the mechanistic basis of this pervasive and complex symptom. One fundamental reason for this, is our inability to account for the clinical heterogeneity of fatigue with its multifactorial nature. We present the conceptual machine learning (ML) framework to dissect the complex nature of fatigue using one of the largest prospectively captured, real-world patient-reported outcome (PROs) on wellbeing from three contemporaneous cohorts (2020-present), totalling 2,970 responses from 2,290 participants across the UK and internationally, including non-IBD controls with 100 lines of clinical metadata. We systematically defined the (1) threshold of fatigue as our primary outcome (≥10/14 fatigue days) to build our ML approach, (2) utilised routinely available clinical data that can be used at a population-level analysis, (3) employed seven different ML methods with external validation in 3 different cohorts in UK, Spain and Australia (n=252), (4) employed Shapley Additive Explanations (SHAP) analysis to break down the clinical heterogeneity to allow the examination of clinical predictive factors at an individual level; and finally (5), investigate whether there are distinct clusters of fatigue patients. We found that ML models performed comparably (AUC/C-index ∼0.7) on external validation with SHAP analysis showing interpretable, individualised fatigue drivers and five distinct fatigue phenotypes, including a subgroup of young males with significantly lower fatigue burden. Our data therefore provides the ML ‘roadmap’ to predict and deconstruct fatigue in IBD and potentially also more widely in IMIDs, enabling patient-level dissection beyond symptom-based classification with the ability to integrate deep molecular data. This is a step towards future clinical-scientific AI models with the immediate clinical application to stratify patients to human experimental studies to better understand the dominant mechanisms that drive fatigue at an individual level. Introduction Many patients with Inflammatory Bowel Diseases (IBD) suffer from extreme fatigue, even when in remission 1 – 3 . From a patient’s perspective, fatigue is a major priority area for further research 4 . Extreme fatigue is also a common symptom across many immune-mediated inflammatory conditions (IMIDs) such as rheumatoid arthritis (RA), systemic lupus erythematosus (SLE) and sarcoidosis 5 – 7 . This suggest there may be unrecognised cross disease-mechanisms that are potentially independent of organ-specific inflammation 8 and hence also poorly treated by conventional immune-suppression. There are increasingly powerful scientific tools from functional neuroimaging to metabolic assays to study the mechanistic basis of this pervasive symptom in IMIDs 9 , 10 . However, human experimental and interventional studies have been stymied by patient heterogeneity. Pertinently in IBD, few interventional studies have specifically targeted fatigue, and those that have, often show poor results 11 – 13 , are small and open-labelled in design 14 , and include ill-defined patient groups with fatigue. Our aim is to develop a machine-learning (ML) approach that can dissect the clinical heterogeneity of this complex symptom construct of fatigue. Here, we present an end-to-end machine learning roadmap in IBD utilising routinely available clinical and laboratory data to do this at a population level whilst retaining the ability to investigate characteristics on the individual level with the objective of deeper patient stratification - ‘to find the right patients, for the right studies’. Methods Patient data We utilised patient-reported outcome (PRO) fatigue data from two prospective mechanistic biomarker studies, namely (1) Investigation into Gastrointestinal Damage Associated Molecular Patterns (GI-DAMPs) that is a cross-sectional IBD study with ethical approvals from the East Scotland Ethics Committee (REC 18/ES/0090); and (2) Mitochondrial DAMPs as mechanistic biomarkers of mucosal inflammation in Crohn’s Disease (MUSIC study; www.musicstudy.uk ), a prospective IBD cohort study with ethical approvals by East Scotland Ethics Committee (REC 19/ES/0087). Both are cohort studies carried out in Scotland (Glasgow, Edinburgh and Dundee; 2020-present) that recruit participants with IBD along with ∼100 lines of clinical metadata including IBD activity, treatment, comorbidities and laboratory parameters. In addition to this, we conducted an online survey to further collect PRO data on fatigue (2023-2024; n=1,643 within UK, n=112 internationally). In all 3 studies ( Table 1 ), we used the validated Crohn’s and Ulcerative Colitis Questionnaire-32 (CUCQ32) questionnaire that is applicable to Crohn’s disease (CD) and Ulcerative colitis (CD), the 2 sub-types of IBD 15 . CUCQ32 contains 32 questions that measures 4 domains of wellbeing (gastrointestinal, social, psychological and general wellbeing) with 5 main questions pertaining to fatigue, generating a score ranging from 0-272. Collectively, these studies involved 2,290 total participants (including 336 non-IBD participants who responded online, and 27 non-IBD symptomatic controls) and provide the scale of data to establish a clinical threshold of patient-reported fatigue as a baseline for our ML approach. View this table: View inline View popup Table 1: Demographics and clinical characteristics of Cohorts 1 (MUSIC), 2 (GI-DAMPs) and 3 (self-reported online). Model development and evaluation of machine learning pipeline Input features include clinical (e.g. diagnosis group, Montreal classification), demographics, exposome (smoking, alcohol, seasonality), laboratory (e.g. CRP, faecal calprotectin) and IBD drug exposure data (Full list in Supplementary Table 3 ). Seven supervised ML models were employed: XGBoost, Random Forest, AdaBoost, multilayer perceptron (MLP), support vector machine, logistic regression ( scikit-learn implementation), and a custom feedforward deep neural network implemented in TensorFlow ( Supplementary Figure 4c ). Conventional logistic regression ( statsmodels ) was included as a reference comparator. Analyses were conducted using Python (version 3.11.9), with the following libraries: scikit-learn (v1.5.2), scipy (v1.14.1), xgboost (v2.1.2), pandas (v2.2.3), numpy (v2.0.2), shap (v0.46.0), tensorflow (v2.18.0), pytorch (v2.5.1), statsmodels (v0.14.4), seaborn (v0.13.2), and matplotlib (v3.10.0). Missing data were imputed according to variable type. Continuous variables with non-normal distributions (e.g., C-reactive protein (CRP), albumin, faecal calprotectin) were imputed using the median to minimise bias. Categorical variables were transformed using one-hot encoding. Numerical variables were standardised (zero mean, unit variance) using StandardScaler ( scikit-learn ). Derived features included seasonality (calculated from the date of CUCQ32 completion) and disease duration (expressed in weeks since diagnosis). To avoid information leakage from repeated measures, group-aware data splitting was performed using GroupKFold and GroupShuffleSplit ( scikit-learn ), with participant identifiers used as grouping variables. For conventional ML models, data were partitioned into training-validation (80%) and independent testing (20%) sets, with hyperparameter tuning conducted via 5-fold group cross-validation. For deep neural networks, data were split into training (65%), validation (15%), and testing (20%) sets using the same group-aware strategy. Model interpretability was evaluated using Shapley Additive Explanations (SHAP) 16 . Model generalisability was evaluated using fully anonymised datasets from three prospectively collected independent validation cohorts based in Scotland, Spain, and Australia ( Supplementary Table 5 , predominantly recruited in the outpatient IBD clinic setting). These datasets were entirely independent of model training and hyperparameter optimisation. Identical preprocessing steps were applied to the external datasets prior to prediction. Model performance was assessed using receiver operator (ROC) curves, area under the curve (AUC), sensitivity, specificity, positive and negative predictive values (PPV and NPV respectively). Unsupervised clustering was performed using K-means. Following exclusion of a single outlier datapoint, the optimal number of clusters was empirically determined by evaluating solutions for k = 2 to k = 10, informed by prior clinical knowledge that the cohort contained a subgroup with active IBD. A solution with k = 5 was selected, balancing cohort heterogeneity while avoiding over-fragmentation or generation of spurious small clusters. All anonymised datasets, dependencies, and code to reproduce this study are publicly available at: https://github.com/1-gut/machine_learning_for_ibd_fatigue . Results Defining the threshold of fatigue We modelled our ML algorithms to predict IBD patients with patient-reported outcome fatigue of ≥10 days over the past 14 days at time of data entry, as the primary outcome variable (Fatigue PRO). This is based on CUCQ32 question: “On how many days over the last 2 weeks did you feel tired?” (0–14 days). Across the combined dataset, IBD patients with active disease reported significantly more fatigued days than those in remission (medians 14 vs 7 days; p<0.001). In IBD patients in remission, fatigue days remained significantly elevated compared to non-IBD controls. (7 vs 4 days; p<0.001; Supplementary Figure 1a ). These findings were consistent across individual cohorts. Fatigue days were strongly correlated with total CUCQ32 scores (r=0.73, p<0.0001; Supplementary Figure 1b ). The distribution of fatigue days was skewed ( Supplementary Figure 1c ). Therefore, we used the median value of ≥10 days of fatigue over 14 days and defined it as the threshold for clinically significant fatigue (Fatigue high ). Using this definition, 53.9% (1602/2970) of observations met criteria for Fatigue high , comprising 75.2% of active IBD patients, 43.6% of those in remission, and 14.2% of non-IBD controls. Patients in the Fatigue high group had significantly higher overall CUCQ32 scores than those below this threshold (median 130 vs 38; p<0.0001; Supplementary Figure 1d ). Fatigue PRO is highly correlated to all measures of CUCQ32 domains (all p<0.001). This shows that a simple question is relevant in the complex construct of fatigue. PROs (‘ what patients tells clinicians ’) are strongly correlated with clinician-based assessment (‘ what clinicians think of patients ’) and are associated with clinical parameters such as CRP and faecal calprotectin ( Supplementary Figure 2 , both p<0.0001). Among 147 MUSIC patients with longitudinal follow-up over 12 months, serial CUCQ32 measurements demonstrated heterogeneity between individuals. Patients achieving clinical remission (Harvey Bradshaw Index, HBI≤5 for CD and Simple Clinical Colitis Activity Index SCCAI≤2 for UC) at 12 months tended to show improvements in both overall CUCQ32 scores and fatigue-specific domains ( Supplementary Figure 3 ). However, no significant difference in CUCQ32 was observed between those with or without mucosal healing at 12 months (median 68 vs 62; Supplementary Figure 2e ), indicating that endoscopic remission does not fully capture patient-reported wellbeing. Collectively, this demonstrates the clinical heterogeneity of fatigue when assessed in accordance with conventional measures of inflammation in IBD. Development of machine learning algorithms and external validation to predict fatigue Having defined our threshold for fatigue, we incorporated self-reported fatigue assessments and all available clinical parameters such as disease activity indices (HBI and SCCAI), laboratory parameters (C-Reactive Protein [CRP], faecal calprotectin, full blood count), body mass index (BMI), smoking status, drug therapy, and seasonality. Here, we generated an initial training dataset of 1,215 observations from Cohorts 1 and 2 for seven ML algorithms: XGBoost, random forest, AdaBoost, multilayer perceptron classifier, support vector classifier, logistic regression, and a custom-built deep neural network (DNN). GroupKFold cross-validation ensured that repeated measures from individual patients remained in the same dataset split. Across classical ML models, predictive performance was broadly comparable with AUC values ranging from 0.69-0.73 for Fatigue high prediction ( Figure 1b ). To explore whether complex nonlinear patterns might enhance predictive accuracy, we developed a custom Deep Neural Network (DNN) implemented in TensorFlow and re-coded in PyTorch. The network architecture comprised four layers (two dense layers interspersed with two dropout layers; Supplementary Figure 4c ). On internal test data, the DNN outperformed classical ML models, achieving an AUC of 0.89 for all IBD cases and 0.94 in patients in biochemical remission ( Figure 1b ). However, external validation using an independent cohort (n = 252; Córdoba, Spain n = 101; Melbourne, Australia n = 90; Edinburgh, Scotland n = 61) revealed that the DNN’s performance did not generalise, with AUCs falling back to the 0.69-0.73 range observed across all models ( Figure 1c ). This suggests that the superior internal performance likely reflected model overfitting rather than true generalisability. No single ML model demonstrated consistent superiority across all datasets. Download figure Open in new tab Figure 1 Machine learning workflow, performance, and interpretability of models predicting fatigue in IBD patients (a) Overview of the machine learning workflow. Two pipelines were implemented: one including all IBD patients (Cohorts 1-2), and another restricted to patients in biochemical remission (defined as calprotectin <250 µg/g and CRP <5 mg/dL). Model performance was subsequently evaluated on an external validation cohort comprising patients with IBD from Spain, Australia, and Scotland. (b) Model performance (AUG) in the full IBD cohort. The deep neural network (DNN) achieved the highest test set performance compared to other models, motivating external validation. (c) Model performance (AUC) in the external validation cohort. All models performed similarly, though the DNN showed a notable decrease in AUC, suggesting possible overfitting in the initial test dataset. (d) SHAP summary plot for the DNN, illustrating the impact of individual features on model predictions. Each dot represents a patient’s data point for a specific feature. The horizontal position of the dot shows how much that feature pushes the prediction toward fatigue (positive values) or away from fatigue (negative values) Dot color reflects the actual feature value (e.g., red for high platelet counts, blue for low). Features are listed vertically in order of their overall impact on the model, with the most influential features at the top. This visualization reveals which features most strongly influence fatigue predictions and highlights that the impact of features can vary between patients—for example, very high platelet counts increase predicted fatigue in some individuals but not in others. Assessments conducted in autumn also increased the likelihood of fatigue prediction. When restricting the analysis to patients in biochemical remission (CRP < 5 mg/L and calprotectin < 250 μg/g), model performance declined (AUCs 0.61-0.66 ( Supplementary Figure 4a ). In this subgroup, inflammatory markers contributed less to fatigue prediction, while features such as anaemia, elevated lymphocytes, reduced urea (potentially reflecting sarcopenia), age, and seasonality emerged as more influential ( Supplementary Figure 4b ). Personalised Fatigue Profiling Using SHAP-Driven Model Interpretation To better understand how individual factors contributed to model predictions, we applied SHAP analysis 16 . SHAP values quantify the relative contribution of each input variable to individual model predictions, offering a transparent, patient-specific decomposition of the model’s decision-making process. SHAP interpretation revealed distinct patient-level patterns, underscoring the clinical (and possibly biological) heterogeneity of fatigue in IBD. For some individuals, fatigue was closely linked to inflammatory markers, while for others, factors such as weight, medication use, or seasonal variation were more influential. In the DNN model, key predictive features for all IBD patients included clinical symptom activity, steroid use, IBD diagnosis (CD vs UC), platelet count, and autumn season ( Figure 1d ). By enabling granular, patient-specific insights into the drivers of fatigue, SHAP analysis provides a proof-of-concept pathway toward personalised mechanistic understanding of fatigue in IBD ( Figure 2 ). Download figure Open in new tab Figure 2 Individualised decomposition of IBD-related fatigue using SHAP. SHAP force plots for true positive cases (i.e., instances where the patient reports fatigue and the model predicts fatigue). f(x) represents the model-predicted probability of Fatigue High ; for example, f(x) = 0.76 corresponds to a 76% probability of Fatigue High · Values closer to 1 indicate higher model confidence in predicting fatigue, while values closer to O indicate confidence in predicting low fatigue. This proof of concept illustrates the model’s ability to decompose feature contributions at the individual level. For example, in patient A, active IBD is a major contributor to fatigue; in patient B, azathioprine is a more influential factor. Patients C and D exhibit multifactorial contributions, with an important weight-related component, while patient E’s smoking status emerges as a key driver of fatigue. Identification of Distinct Fatigue Phenotypes Using K-Means Clustering To further characterise potential fatigue subtypes, we applied unsupervised k-means clustering ( Figure 3 ). Five distinct patient clusters emerged ( Table 2 ) namely those with (1) Active IBD, (2) Young tall males, (3) Crohn’s disease patients with long disease duration, (4) More elderly UC patients; and (5) Young female IBD patients. Cluster 0, representing active IBD, exhibited the highest prevalence of Fatigue high (81.4%, n=113). In contrast, Cluster 1 (young tall males) demonstrated a significantly lower fatigue prevalence (34.9%) compared to all other groups (ANOVA F=22.6, p<0.0001; Tukey HSD post hoc comparisons; Figure 4c). The remaining clusters exhibited intermediate fatigue prevalence ranging from 50–56%. The low-fatigue male cluster (Cluster 1) was distinct not only in sex (96% male), height (median 1.80 m), and weight (median 84 kg), but also exhibited higher haemoglobin levels (145 g/L vs 120-137 g/L in other clusters). The reasons for their relative fatigue resistance remain unclear but may reflect biological resilience, under-reporting, or unidentified protective mechanisms. Notably, deviations from low baseline fatigue within this group may represent meaningful signals warranting focused mechanistic investigation. Download figure Open in new tab Figure 3 K-means clustering identifies five distinct fatigue clusters. Fatigue prevalence by cluster: cluster O shows high fatigue prevalence (81%), cluster 1 shows low fatigue prevalence (35%), and clusters 2--4 show intermediate fatigue prevalence (50-56%). (a) Visualization of fatigue clusters using principal components analysis (PCA). (b) Tukey’s HSD test indicates that cluster O has significantly higher fatigue compared to the other clusters, while cluster 1 has significantly lower fatigue. No significant differences are observed among clusters 2--4 . (c) Heatmap visualization of the five identified subgroups. View this table: View inline View popup Download powerpoint Table 2 Summary Cluster Characteristics and Descriptive Label Together, these findings demonstrate that IBD-associated fatigue remains prevalent despite disease control, exhibits marked individual variability and can be partially modelled using supervised and unsupervised ML approaches that integrate multidimensional clinical data. Discussion Artificial intelligence and powerful machine learning are now firmly established and are increasingly utilised in medicine 17 . A key consideration is how we construct the practical steps and apply these tools to a complex and relevant medical patient-centric problem, such as fatigue. Here, we present the conceptual ML framework ‘roadmap’ to dissect the fatigue towards a practical outcome of better patient stratification to aid mechanistic studies; using one of the largest prospectively captured, real-world PROs on wellbeing from three contemporaneous cohorts, totalling 2,970 responses from 2,290 participants across the UK and internationally, including non-IBD controls. We systematically defined the relevance of the (1) primary outcome of fatigue to based our ML approach, (2) utilised routinely available clinical data that can be used at a population-level analysis, (3) employed 7 different ML methods, (4) employed SHAP analysis to break down the clinical heterogeneity at an individual level; and finally (5), investigate whether there are distinct clusters of fatigue patients based simply on routine clinical observations. We found that our ML approach can predict IBD patients with fatigue (AUC 0.70) with external validation across 3 independent cohorts in different geographical locations, but the performance drops within the IBD patient group that is in remission. We systematically compared seven ML models against traditional logistic regression (Appendix 4) to evaluate their ability to incorporate deep clinical data. Deep neural networks demonstrated superior internal performance but similar external validation performance across all models. However, substantial unexplained variance remains, particularly in remission. The use of our simplified outcome threshold of ≥10/14 fatigue days deserves further discussion. Firstly, the prevalences of fatigue defined by this agrees with clinical studies of IBD and IMID fatigue 18 , 19 . Secondly, this allows us to harmonise fatigue measurement across cohorts, scale up our dataset and facilitates predictive machine learning analyses that can be applicable to different IMID cohorts in the future. Thirdly, our data from the broader CUCQ32 shows that this simple PRO is highly correlated to other domains of wellbeing and therefore, a good representation of the deeper construct of the symptom fatigue that may be linked to poor sleep, depression, anxiety and organic disease-related causes. Hence, our goal is to build a ML-approach that can capture these complexities from a simple starting point that is easily captured in the clinic. A potential critique is that fatigue prediction models may merely recapitulate the fatigue question itself i.e. reinforcing a circular argument. We address this by explicitly excluding the fatigue question and other CUCQ questions as input variables in our ML algorithms. Instead, fatigue was predicted from a broad range of independently collected clinical metadata, including laboratory markers (e.g., CRP, calprotectin), medication exposure, BMI, smoking status, and seasonality. These predictors are biologically and clinically orthogonal to self-reported fatigue. Furthermore, SHAP analysis revealed interpretable, participant-specific predictors such as inflammatory markers and smoking, providing face validity to the modelling. External validation across independent cohorts confirmed that fatigue can be predicted to an extent without direct fatigue self-report, demonstrating the capacity of machine learning to uncover latent patterns underlying fatigue. This suggests that (1) machine learning is equivalent to classical methods for structured clinical data; and (2) the consistent performance ceiling likely reflects a ‘hidden’ multifactorial pathobiological compartment of factors currently unmeasured. We, therefore, present the first comprehensive framework utilising ML-algorithms to dissect the complex presentation of fatigue. This is initially based on using routinely available clinical data but molecular data such as genetics, nutritional factors, microbiome and metabolomics for examples, can be easily incorporated. We acknowledge that our dataset lacked key variables known to influence fatigue, including sleep quality (we excluded the use of CUCQ sleep question as it was highly correlated to the outcome), mental health, socioeconomic status, and more detailed comorbidities. We think that with further training of the models, there will be direct clinical application by identifying ‘clusters of patients’ or individualised stratification based on SHAP analysis to human experimental studies of fatigue. These insights may ultimately support patient stratification to targeted interventional studies. In conclusion, fatigue profoundly impacts wellbeing in IBD, even in remission. Our data provides credible support for the utility of patient reported outcomes as endpoints for future translational scientific research. In addressing IBD-associated fatigue, we show the conceptual ability to reduce the dimensionality via modelling (thus flattening out the heterogeneity of IBD-associated fatigue) as the first-pass mechanism to identify fatigue subgroups. This simplified approach will allow the future incorporation of multiple streams of complex scientific metadata (such as genetics and microbiome for example) for larger scale analyses at a cohort level. There are many better scientific tools to study central and peripheral fatigue - from imaging, metabolism and mitochondrial function. We envisage that our work provides a step towards identifying clear subgroups for targeted mechanistic studies and therapeutic development, shifting beyond symptom-based classification towards data-driven and personalised fatigue management in IBD. Data Availability All data produced are available online at https://github.com/1-gut/machine_learning_for_ibd_fatigue Supplementary Material View this table: View inline View popup Download powerpoint Supplementary Table 1: The CUCQ32 questions. View this table: View inline View popup Download powerpoint Supplementary Table 2: Paired correlation analyses of CUCQ32 questions against question 5, (‘How many days in the last 2 weeks did you feel tired?’). Spearman-rank analyses, Benjamini-Hochberg corrected. View this table: View inline View popup Download powerpoint Supplementary Table 3: List of input features for machine learning after data transformation View this table: View inline View popup Download powerpoint Supplementary Table 4: Model metrics for the 7 different machine learning approaches (All IBD) ranked by area under receiver-operator curve (AUC) View this table: View inline View popup Supplementary Table 5: Summary of Validation Cohorts View this table: View inline View popup Supplementary Table 6 Full Cluster Characteristics Download figure Open in new tab Supplementary Figure 1 Fatigue days differ by disease status and correlate with CUCQ32 scores (a) Median number of fatigue days differed significantly between patients with active I8D, patients in remission, and non-I8D controls (all cohorts combined). (b) Fatigue days reported in the CUCQ fatigue question correlated significantly with overall CUCQ32 scores. (c) Reported fatigue days were non-normally distributed, with a median of 10 used to classify fatigue as Fatigue High or Fatigue Low (d) Overall CUCQ32 scores were significantly higher in the Fatigue-high group compared to the Fatigue-low group. Download figure Open in new tab Supplementary Figure 2 CUCQ32 scores correlate with clinical activity scores and inflammatory biomarkers but not with endoscopic healing. (a) CUCQ32 scores are highly correlated with Harvey-Bradshaw Index scores in Crohn’s disease. (b) CUCQ32 scores are highly correlated with Simple Clinical Colitis Activity Index scores in ulcerative colitis. (c) CUCQ32 scores are significantly higher in patients with elevated faecal calprotectin (>250 µg/g). (d) CUCQ32 scores are significantly higher in patients with elevated C-reactive protein (>10 mg/L). (e) No significant difference in CUCQ32 scores was observed between patients with complete mucosa! healing and those without, in Cohort 3 (MUSIC). Download figure Open in new tab Supplementary Figure 3 CUCQ Longitudinal Characteristics (a) & {b) Evolution of CUCQ32 total scores (top) and CUCQ32 fatigue question scores {bottom) over time in patients with ongoing disease activity versus remission in the MUSIC cohort. Red lines represent the mean for CUCQ32 total scores and the median for the fatigue question scores (Cohort 1). Download figure Open in new tab Supplementary Figure 4 Model performance, features associated with fatigue prediction in the biochemical remission cohort and final DNN architecture (a) Model performance (ALIC) in the biochemical remission cohort. The dataset size for this cohort is smaller, and predicting fatigue is more challenging. (b) SHAP summary plot for the deep neural network (DNN) in the biochemical remission cohort. The top features associated with fatigue include active IBD symptoms, low red blood cell counts, high lymphocyte counts, low urea levels, and older age. Seasonality factors (summer and winter) have smaller contributions but are captured within the model. (c) Schematic overview of final deep neural network architecture. Acknowledgements This work is part of the MUSIC IBD study, funded by The Leona M. and Harry B. Helmsley Charitable Trust (G-1911-03343) and the Anne Ferguson Memorial Fund to GTH. References 1. ↵ Farrell D , McCarthy G , Savage E . Self-reported Symptom Burden in Individuals with Inflammatory Bowel Disease . J Crohns Colitis 2016 ; 10 ( 3 ): 315 – 22 . DOI: 10.1093/ecco-jcc/jjv218 . OpenUrl CrossRef PubMed 2. D’Silva A , Fox DE , Nasser Y , et al. Prevalence and Risk Factors for Fatigue in Adults With Inflammatory Bowel Disease: A Systematic Review With Meta-Analysis . Clin Gastroenterol Hepatol 2022 ; 20 ( 5 ): 995 – 1009 e7 . DOI: 10.1016/j.cgh.2021.06.034 . OpenUrl CrossRef PubMed 3. ↵ Borren NZ , van der Woude CJ , Ananthakrishnan AN . Fatigue in IBD: epidemiology, pathophysiology and management . Nat Rev Gastroenterol Hepatol 2019 ; 16 ( 4 ): 247 – 259 . DOI: 10.1038/s41575-018-0091-9 . OpenUrl CrossRef PubMed 4. ↵ Jagt JZ , van Rheenen PF , Thoma SMA , et al. The top 10 research priorities for inflammatory bowel disease in children and young adults: results of a James Lind Alliance Priority Setting Partnership . Lancet Gastroenterol Hepatol 2023 . DOI: 10.1016/S2468-1253(23)00140-1 . OpenUrl CrossRef 5. ↵ Arnaud L , Gavand PE , Voll R , et al. Predictors of fatigue and severe fatigue in a large international cohort of patients with systemic lupus erythematosus and a systematic review of the literature . Rheumatology (Oxford ) 2019 ; 58 ( 6 ): 987 – 996 . DOI: 10.1093/rheumatology/key398 . OpenUrl CrossRef 6. Drent M , Lower EE , De Vries J. Sarcoidosis-associated fatigue . Eur Respir J 2012 ; 40 ( 1 ): 255 - 63 . DOI: 10.1183/09031936.00002512 . OpenUrl Abstract / FREE Full Text 7. ↵ Nikolaus S , Bode C , Taal E , van de Laar MA . Fatigue and factors related to fatigue in rheumatoid arthritis: a systematic review . Arthritis Care Res (Hoboken ) 2013 ; 65 ( 7 ): 1128 – 46 . DOI: 10.1002/acr.21949 . OpenUrl CrossRef 8. ↵ Davies K , Dures E , Ng WF . Fatigue in inflammatory rheumatic diseases: current knowledge and areas for future research . Nat Rev Rheumatol 2021 ; 17 ( 11 ): 651 – 664 . DOI: 10.1038/s41584-021-00692-1 . OpenUrl CrossRef PubMed 9. ↵ Stefanov K , Al-Wasity S , Parkinson JT , Waiter GD , Cavanagh J , Basu N . Brain mapping inflammatory-arthritis-related fatigue in the pursuit of novel therapeutics . Lancet Rheumatol 2023 ; 5 ( 2 ): e99 – e109 . DOI: 10.1016/S2665-9913(23)00007-3 . OpenUrl CrossRef 10. ↵ Naviaux RK , Naviaux JC , Li K , et al. Metabolic features of chronic fatigue syndrome . Proc Natl Acad Sci U S A 2016 ; 113 ( 37 ): E5472 – 80 . DOI: 10.1073/pnas.1607571113 . OpenUrl Abstract / FREE Full Text 11. ↵ Bager P , Hvas CL , Rud CL , Dahlerup JF . Randomised clinical trial: high-dose oral thiamine versus placebo for chronic fatigue in patients with quiescent inflammatory bowel disease . Aliment Pharmacol Ther 2021 ; 53 ( 1 ): 79 – 86 . DOI: 10.1111/apt.16166 . OpenUrl CrossRef PubMed 12. Farrell D , Artom M , Czuber-Dochan W , Jelsness-Jorgensen LP , Norton C , Savage E . Interventions for fatigue in inflammatory bowel disease . Cochrane Database Syst Rev 2020 ; 4 ( 4 ):CD012005. DOI: 10.1002/14651858.CD012005.pub2 . OpenUrl CrossRef 13. ↵ Truyens M , Lobaton T , Ferrante M , et al. Effect of 5-Hydroxytryptophan on Fatigue in Quiescent Inflammatory Bowel Disease: A Randomized Controlled Trial . Gastroenterology 2022 ; 163 ( 5 ): 1294 – 1305 e3 . DOI: 10.1053/j.gastro.2022.07.052 . OpenUrl CrossRef PubMed 14. ↵ Moulton CD , Young AH , Hart AL . Modafinil for Severe Fatigue in Inflammatory Bowel Disease: A Prospective Case Series . Clin Gastroenterol Hepatol 2024 ; 22 ( 8 ): 1737 – 1740 . DOI: 10.1016/j.cgh.2023.12.030 . OpenUrl CrossRef PubMed 15. ↵ Alrubaiy L , Cheung WY , Dodds P , et al. Development of a short questionnaire to assess the quality of life in Crohn’s disease and ulcerative colitis . J Crohns Colitis 2015 ; 9 ( 1 ): 66 – 76 . DOI: 10.1093/ecco-jcc/jju005 . OpenUrl CrossRef PubMed 16. ↵ Ponce-Bobadilla AV , Schmitt V , Maier CS , Mensing S , Stodtmann S . Practical guide to SHAP analysis: Explaining supervised machine learning model predictions in drug development . Clin Transl Sci 2024 ; 17 ( 11 ): e70056 . DOI: 10.1111/cts.70056 . OpenUrl CrossRef PubMed 17. ↵ Maddox TM , Embi P , Gerhart J , Goldsack J , Parikh RB , Sarich TC . Generative AI in Medicine - Evaluating Progress and Challenges . N Engl J Med 2025 ; 392 ( 24 ): 2479 – 2483 . DOI: 10.1056/NEJMsb2503956 . OpenUrl CrossRef PubMed 18. ↵ Villoria A , Garcia V , Dosal A , et al. Fatigue in out-patients with inflammatory bowel disease: Prevalence and predictive factors . PLoS One 2017 ; 12 ( 7 ): e0181435 . DOI: 10.1371/journal.pone.0181435 . OpenUrl CrossRef PubMed 19. ↵ Cohen BL , Zoega H , Shah SA , et al. Fatigue is highly associated with poor health-related quality of life, disability and depression in newly-diagnosed patients with inflammatory bowel disease, independent of disease activity . Aliment Pharmacol Ther 2014 ; 39 ( 8 ): 811 – 22 . DOI: 10.1111/apt.12659 . OpenUrl CrossRef PubMed View the discussion thread. Back to top Previous Next Posted August 15, 2025. Download PDF Data/Code Email Thank you for your interest in spreading the word about medRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following Machine learning approach to dissect the clinical heterogeneity of IBD-associated fatigue Message Subject (Your Name) has forwarded a page to you from medRxiv Message Body (Your Name) thought you would like to see this page from the medRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share Machine learning approach to dissect the clinical heterogeneity of IBD-associated fatigue Cher S Chuah , Rebecca Hall , Robert J Whelan , Peter D Cartlidge , Beatriz Gros , Eva Iglesias-Flores , Nikita Parkash , Ray K Boyapati , Clara Ramos-Belinchon , Solomon Ong , Emily F Brownson , Iona AM Campbell , Craig Mowat , John P Seenan , Jonathan C MacDonald , Gwo-Tzer Ho medRxiv 2025.08.14.25333676; doi: https://doi.org/10.1101/2025.08.14.25333676 Share This Article: Copy Citation Tools Machine learning approach to dissect the clinical heterogeneity of IBD-associated fatigue Cher S Chuah , Rebecca Hall , Robert J Whelan , Peter D Cartlidge , Beatriz Gros , Eva Iglesias-Flores , Nikita Parkash , Ray K Boyapati , Clara Ramos-Belinchon , Solomon Ong , Emily F Brownson , Iona AM Campbell , Craig Mowat , John P Seenan , Jonathan C MacDonald , Gwo-Tzer Ho medRxiv 2025.08.14.25333676; doi: https://doi.org/10.1101/2025.08.14.25333676 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Gastroenterology Subject Areas All Articles Addiction Medicine (568) Allergy and Immunology (863) Anesthesia (300) Cardiovascular Medicine (4440) Dentistry and Oral Medicine (444) Dermatology (383) Emergency Medicine (608) Endocrinology (including Diabetes Mellitus and Metabolic Disease) (1510) Epidemiology (15229) Forensic Medicine (30) Gastroenterology (1126) Genetic and Genomic Medicine (6605) Geriatric Medicine (668) Health Economics (998) Health Informatics (4541) Health Policy (1369) Health Systems and Quality Improvement (1613) Hematology (543) HIV/AIDS (1265) Infectious Diseases (except HIV/AIDS) (15921) Intensive Care and Critical Care Medicine (1103) Medical Education (623) Medical Ethics (147) Nephrology (668) Neurology (6604) Nursing (346) Nutrition (998) Obstetrics and Gynecology (1145) Occupational and Environmental Health (957) Oncology (3334) Ophthalmology (974) Orthopedics (369) Otolaryngology (420) Pain Medicine (436) Palliative Medicine (130) Pathology (663) Pediatrics (1693) Pharmacology and Therapeutics (692) Primary Care Research (711) Psychiatry and Clinical Psychology (5448) Public and Global Health (9234) Radiology and Imaging (2199) Rehabilitation Medicine and Physical Therapy (1370) Respiratory Medicine (1196) Rheumatology (594) Sexual and Reproductive Health (712) Sports Medicine (530) Surgery (712) Toxicology (99) Transplantation (289) Urology (265) (function(){function c(){var b=a.contentDocument||a.contentWindow.document;if(b){var d=b.createElement('script');d.innerHTML="window.__CF$cv$params={r:'a01565756b5d41e2',t:'MTc3OTcyMTA4Ng=='};var a=document.createElement('script');a.src='/cdn-cgi/challenge-platform/scripts/jsd/main.js';document.getElementsByTagName('head')[0].appendChild(a);";b.getElementsByTagName('head')[0].appendChild(d)}}if(document.body){var a=document.createElement('iframe');a.height=1;a.width=1;a.style.position='absolute';a.style.top=0;a.style.left=0;a.style.border='none';a.style.visibility='hidden';document.body.appendChild(a);if('loading'!==document.readyState)c();else if(window.addEventListener)document.addEventListener('DOMContentLoaded',c);else{var e=document.onreadystatechange||function(){};document.onreadystatechange=function(b){e(b);'loading'!==document.readyState&&(document.onreadystatechange=e,c())}}}})();
Text is read by the "Ask this paper" AI Q&A widget below.
Extraction quality varies by source — PMC NXML preserves structure
cleanly, OA-HTML may include some navigation residue, and OA-PDF can
have broken hyphenation. The publisher copy
(via DOI)
is the canonical version.