Stratifying Eating Disorders with Clustering: From Diagnosis to Phenotypic Diversity

preprint OA: closed
📄 Open PDF Full text JSON View at publisher

Abstract

Psychiatric disorders, including eating disorders (EDs), are characterized by heterogeneity that limits diagnostic accuracy and treatment personalization. Precision psychiatry calls for tools to identify data-driven phenotypes beyond traditional categories. In a large cohort (N=809), we applied an unsupervised clustering pipeline to self-report and clinical variables to uncover ED subgroups. Repeated Spectral Clustering revealed four phenotypes: two aligned with DSM-5 diagnoses (anorexia nervosa restricting-type; bulimia nervosa), and two diagnostically mixed clusters, one characterized by higher psychopathology and trauma exposure, the other by prolonged illness duration. These clusters showed distinct profiles and outcomes. The higher predictability of data-driven clusters compared to DSM-5 categories suggests that unsupervised stratification may offer a complementary perspective on relevant heterogeneity. Our findings highlight how computational phenotyping can advance precision psychiatry by revealing meaningful patterns, informing prognosis, and guiding personalized interventions. This approach may help bridge the gap between clinical presentations and the need for stratified, patient-centered care.
Full text 68,051 characters · extracted from preprint-html · click to expand
Stratifying Eating Disorders with Clustering: From Diagnosis to Phenotypic Diversity | medRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-P4HH5NV'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search Stratifying Eating Disorders with Clustering: From Diagnosis to Phenotypic Diversity View ORCID Profile Andrea Zanola , View ORCID Profile Louis Fabrice Tshimanga , View ORCID Profile Valentina Meregalli , View ORCID Profile Angela Favaro , View ORCID Profile Manfredo Atzori , View ORCID Profile Enrico Collantoni doi: https://doi.org/10.1101/2025.08.25.25334344 Andrea Zanola 1 Department of Neuroscience, University of Padua , Padua, 35128, Italy 2 Padua Neuroscience Center , Padua, 35128, Italy Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Andrea Zanola For correspondence: andrea.zanola{at}studenti.unipd.it enrico.collantoni{at}unipd.it Louis Fabrice Tshimanga 1 Department of Neuroscience, University of Padua , Padua, 35128, Italy 2 Padua Neuroscience Center , Padua, 35128, Italy 3 Department of Information Engineering, University of Padua , Padua, 35128, Italy Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Louis Fabrice Tshimanga Valentina Meregalli 1 Department of Neuroscience, University of Padua , Padua, 35128, Italy 2 Padua Neuroscience Center , Padua, 35128, Italy Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Valentina Meregalli Angela Favaro 1 Department of Neuroscience, University of Padua , Padua, 35128, Italy 2 Padua Neuroscience Center , Padua, 35128, Italy Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Angela Favaro Manfredo Atzori 1 Department of Neuroscience, University of Padua , Padua, 35128, Italy 2 Padua Neuroscience Center , Padua, 35128, Italy 4 Information Systems Institute, University of Applied Sciences Western Switzerland (HES-SO Valais) , 3960 Sierre Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Manfredo Atzori Enrico Collantoni 1 Department of Neuroscience, University of Padua , Padua, 35128, Italy 2 Padua Neuroscience Center , Padua, 35128, Italy Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Enrico Collantoni For correspondence: andrea.zanola{at}studenti.unipd.it enrico.collantoni{at}unipd.it Abstract Full Text Info/History Metrics Supplementary material Data/Code Preview PDF Abstract Psychiatric disorders, including eating disorders (EDs), are characterized by heterogeneity that limits diagnostic accuracy and treatment personalization. Precision psychiatry calls for tools to identify data-driven phenotypes beyond traditional categories. In a large cohort (N=809), we applied an unsupervised clustering pipeline to self-report and clinical variables to uncover ED subgroups. Repeated Spectral Clustering revealed four phenotypes: two aligned with DSM-5 diagnoses (anorexia nervosa restricting-type; bulimia nervosa), and two diagnostically mixed clusters, one characterized by higher psychopathology and trauma exposure, the other by prolonged illness duration. These clusters showed distinct profiles and outcomes. The higher predictability of data-driven clusters compared to DSM-5 categories suggests that unsupervised stratification may offer a complementary perspective on relevant heterogeneity. Our findings highlight how computational phenotyping can advance precision psychiatry by revealing meaningful patterns, informing prognosis, and guiding personalized interventions. This approach may help bridge the gap between clinical presentations and the need for stratified, patient-centered care. 1 Introduction Psychiatric disorders are characterized by substantial clinical heterogeneity, posing challenges for diagnosis, prognosis, and treatment planning. Eating disorders (EDs), including anorexia nervosa and bulimia nervosa, exemplify this complexity, exhibiting diverse trajectories, comorbidities, and levels of treatment response [ 1 ]. Although these conditions share several elements of psychopathological and behavioral homogeneity, as outlined in the Diagnostic and Statistical Manual of Mental Disorders, Fifth Edition (DSM-5) [ 2 ], patients with EDs often demonstrate a wide range of symptom profiles, comorbidities, and clinical trajectories that extend beyond conventional diagnostic subtypes [ 3 ]. This inherent variability complicates efforts to develop targeted therapeutic interventions, accurately predict outcomes, and refine clinical strategies. In this context, increasing attention has been devoted to identifying more nuanced and empirically defined subgroups—or phenotypes—within clinical populations with EDs [ 4 , 5 ]. To this end, data-driven methods such as network modeling and cluster analysis have been employed to examine a broad range of psychological variables, personality dimensions, demographic factors, and clinical indicators [ 6 ]. These approaches aim to reveal distinct patterns that transcend conventional nosological frameworks, placing emphasis on how dimensions beyond core EDs symptoms can critically shape the clinical picture. Recent evidence suggests that, within EDs samples, discernible groups emerge on the basis of factors such as disordered eating behaviors, body image disturbances, interpersonal functioning, and underlying temperament and personality traits [ 7 ]. In addition, a growing body of research emphasizes the importance of examining symptoms beyond the confines of rigid diagnostic categories, underscoring the critical role of transdiagnostic mechanisms in EDs. These mechanisms may influence not only the underlying patho-physiology but also the persistence, clinical complexity, and treatment responsiveness of these conditions. This evidence appears to align with the well-established theoretical view that conceptualizes EDs within a transdiagnostic framework, which underpins cognitive-behavioral therapies for these conditions [ 8 ]. In recent years, machine learning (ML) has emerged as a powerful and transformative approach within the psychiatric context, enabling the analysis of complex, high-dimensional data and facilitating the discovery of latent patient subgroups [ 9 ]. This capability is particularly relevant for heterogeneous disorders such as anorexia nervosa and bulimia nervosa, where conventional diagnostic categories often fail to capture the underlying variability and where the boundaries defined by diagnostic criteria do not prevent high rates of transdiagnostic migration and the emergence of atypical phenotypes [ 10 ]. Unsuper-vised ML techniques, such as clustering, are especially promising as they can identify subgroups without relying on predefined categories, offering a more flexible and data-driven framework for understanding EDs subtypes. Despite this promise, existing ML applications to EDs populations have been somewhat fragmented [ 11 ]. Previous research has pursued a variety of goals, from early detection of those at risk to predicting treatment responses, using a broad range of methodologies, from basic regression models to advanced neural networks [ 12 ]. For instance, Rosenfield and Linstead [ 12 ] employed cluster analysis to examine the relationships among the Eating Disorder Examination Questionnaire, Clinical Impairment Assessment, and Autism Quotient, showcasing the potential of ML to identify vulnerability to EDs. Using a different approach, Haynos and colleagues [ 13 ] demonstrated that ML techniques outperformed logistic regression in predicting the longitudinal course of EDs. Furthermore, Juarascio et al .[ 14 ] emphasized the potential of just-in-time adaptive interventions, which use ML or predictive algorithms to identify moments of heightened risk for disordered eating behaviors and deliver personalized, real-time strategies to improve treatment outcomes. Notably, many of these studies remain anchored in traditional diagnostic labels, and fewer have fully explored unsupervised methods that might reveal previously unrecognized patient profiles or confirm the presence and the possible clinical utility of novel phenotypes. The utility of ML in psychiatry is not limited to EDs. Similar approaches have already shown promise in other complex mental health conditions, such as mood disorders and schizophrenia, where data-driven subgroup identification has led to the discovery of clinically meaningful phenotypes and improved stratification for interventions [ 15 ]. In this context, specific unsupervised machine learning methods, such as Gaussian mixture models and clustering algorithms, have been used to discover distinct neuroanatomical and cognitive subtypes within schizophrenia populations [ 16 , 17 , 18 ], revealing subgroups with differing brain structural and cognitive patterns [ 19 ]. Similarly, in the field of depression research, ML approaches have been increasingly utilized to address the substantial heterogeneity of the disorder, which often hinders precise diagnosis and treatment. For instance, latent variable models and clustering techniques have been applied to large datasets to identify empirically derived depressive subtypes, enabling a better understanding of symptom variability and its relationship to clinical outcomes [ 20 ]. This cross-diagnostic evidence supports the application of ML-based strategies in the psychiatric context, where there remains a pressing need to enhance diagnostic precision and treatment personalization. Building on the ED-specific and cross-diagnostic work outlined above, unsupervised techniques represent a valuable tool for exploring latent phenotypic structures that extend beyond the boundaries of established diagnostic classifications. By clustering readily available psychometric and clinical variables, such models can simultaneously refine diagnostic boundaries, furnish pragmatic decision tools for clinicians, and generate falsifiable predictions about prognosis and treatment response. Building on growing evidence that ML can reveal clinically relevant patterns in psychiatric populations, we applied an unsupervised clustering pipeline to a large cohort of individuals with EDs. The primary objectives were twofold: first, to examine whether the resulting clusters correspond to traditional DSM-5 diagnostic categories or instead reflect novel, clinically meaningful phenotypes; and second, to investigate whether these data-driven subgroups are associated with differential treatment outcomes. By addressing these aims, the study investigates whether cluster-based phenotyping can enhance clinical understanding and prognostic accuracy, thereby supporting more personalized and effective interventions. 2 Materials and Methods 2.1 Participants The cohort under analysis consists of 809 patients admitted to the Eating Disorders Unit of Padua Hospital between 2005 and 2023. Inclusion criteria were: (1) female gender and (2) a diagnosis of acute AN or BN according to DSM-5 criteria [ 2 ] at the time of the initial clinical evaluation. Among these patients, 351 (43%) were diagnosed with the restrictive subtype of AN (AN-R), 150 (19%) with the binge-purging subtype of AN (AN-BP), and 308 (38%) with bulimia nervosa (BN). The mean age was 23.1 years (standard deviation, SD = 6.8 years), with a range between 12.7 and 57.0 years. All subjects gave their informed written consent to the use of data in an anonymous form. The Ethical Committee of Padova Hospital approved the study (protocol n ◦ 1598P). 2.2 Procedure and Measures At their first admission to the center, all patients underwent a routine baseline assessment. This included a semi-structured interview covering sociodemographic and clinical variables, as well as the completion of a series of self-administered questionnaires and scales. From the interview, the following variables were extracted: diagnosis according to DSM-5 criteria [ 2 ], age, age of onset (defined as the age at first occurrence of an eating disorder), illness duration, weekly hours of physical activity, current Body Mass Index (BMI), lowest lifetime BMI, weight suppression (defined as the difference between the highest lifetime BMI and current BMI), presence of amenorrhea, and history of either sexual or physical childhood abuse. BMI was measured in kg/m2; henceforth, the unit will be omitted for brevity. The semi-structured interview also included a checklist of 27 stressful events. Patients were asked to indicate whether they had experienced any of these events in the six months preceding the onset of the eating disorder, and to rate the subjective severity of each event on a scale from 1 (slight) to 5 (catastrophic). The total severity score was used as a variable in the analyses. At the time of discharge, each patient’s clinical status was recorded and classified as full remission, partial remission, or persistent illness. Full remission was defined as the absence of eating disorder symptoms in the last month, including restored weight (BMI ≥ 18 kg/m2), absence of amenorrhea (when not due to other medical causes), no binge eating or purging behaviors, and no severe body image disturbance. Partial remission was defined as the persistence of only one of the following symptoms: amenorrhea (not explained by other medical conditions), BMI lower than 18, presence of binge eating, purging, or severe body image disturbance. No remission (persistent illness) was defined by the presence of at least two of the aforementioned symptoms at the end of the treatment [ 21 ]. The self-administered questionnaires included: Eating Disorder Inventory-1 (EDI-1) [ 22 ] The EDI-1 is a 64-item self-report measure to assess the severity of eating disorder psychopathology. Higher scores indicate greater symptom severity. All the 8 subscales were considered in this study: drive for thinness (DR-THIN), bulimia (BUL), body dissatisfaction (BODY-DISS), ineffectiveness (INEFF), perfectionism (PERF), interpersonal distrust (INT-DIS), interoceptive awareness (INT-AW), and maturity fears (MAT-FEAR). Temperament and Character Inventory (TCI) [ 23 ] The TCI is a 240-item self-report questionnaire designed to evaluate four dimensions of temperament—Novelty Seeking (NS), Harm Avoidance (HA), Reward Dependence (RD), and Persistence (P)— and three dimensions of character—Self-Directedness, Cooperativeness, and Self-Transcendence. In the present study, only the four temperament dimensions were considered. Symptoms Check-list-90 (SCL-90) [ 24 ] The SCL-90 is a 90-item self-report measure designed to assess a broad range of psychopathological symptoms. Higher scores indicate greater symptoms severity. In the present study, we considered only the Global Severity Index of the SCL-90 as a measure of overall psychopathological distress. 2.3 Data processing The analysis begins from dividing the features derived from clinical data and pathological assessments, into a set of clustering features and a set of validating features. The clustering features constitute the space in which lie the data fed to the clustering algorithm of choice. The validating features constitute the space in which the separation of data performed by clustering can be statistically tested. This distinction is important, since every clustering algorithm explicitly or implicitly optimizes a measure of separation of dissimilar data or aggregation of similar data. Evaluating clustering results within the same feature space used for clustering can lead to overly optimistic conclusions. This is because clustering algorithms are inherently capable of partitioning data into groups, even when there is no true underlying structure or distinct generative processes. To assess the validity and relevance of the identified clusters, one can instead test the null hypothesis that cluster assignments are random with respect to a separate set of validating features i.e., features not used in the clustering process. If the cluster groupings exhibit statistically significant differences in these external validating features, this supports the validity of the clustering solution. Table 1 reports the 25 available features, categorized into clustering and validation features. For each feature, four summary statistics (minimum, mean, standard deviation and maximum) are provided, stratified by the three DSM-5 diagnostic groups. From the 25 features available, BMI, minimum BMI, weight suppression and amenorrhea, have been intentionally omitted from the clustering input. These variables are highly discriminative across DSM-5 diagnostic categories and would have led to a trivial separation of the bulimia nervosa cluster from anorexia nervosa subtypes. To ensure that the clustering process captured latent phenotypic structure beyond overt diagnostic distinctions, we chose to omit these features from the unsupervised model. View this table: View inline View popup Download powerpoint Table 1: Dataset Summary. Summary statistics for all dataset features, grouped according to the three DSM-5 diagnostic categories. Features are divided into those used for cluster definition and those reserved for validation. AN-R refers to anorexia nervosa, restricting type; AN-BP refers to anorexia nervosa binge/purge type; BN denotes bulimia nervosa; MD indicates mean; SD indicates standard deviation. 2.4 Pre-clustering analyses: diagnosis classification Before applying the clustering procedure, which is an unsupervised approach, we first employed a supervised method to assess how well the DSM-5 diagnostic categories could be predicted from the clustering features described in subsection 2.3 and listed in Table 1 . XGBoost (eXtreme Gradient Boosting) is a widely used, high-performance ensemble learning algorithm, based on gradient-boosted decision trees. It iteratively constructs decision trees that correct errors made by previous ones, using gradient descent to minimize a specified loss function, with built-in regularization to mitigate overfitting. The dataset was split into 80% training and 20% testing subsets. To tune hyper-parameters, a grid search with 5-fold stratified cross-validation on the training set was used; the set of parameters tested is reported in the supplementary material section 1. A fixed seed (42) was randomly chosen to ensure reproducibility of the experiment. Due to class imbalance, the average balanced accuracy was used as the evaluation metric. By definition, for a multi-class classification problem with C classes (here the three DSM-5 diagnosis), the balanced accuracy is: where TP i and FN i stands respectively for the true positive and the false negative for class i . 2.5 Clustering Clustering is a popular unsupervised technique, particularly useful in the identification of phenotypes, which can be effectively used to find groups of patients with common traits. The following procedure was applied to cluster the data according to the selected features described in subsection 2.3 . The distance between subjects is calculated using the General Distance Measure (GDM) [ 25 ] on the 19 selected features. The GDM is a proximity measure inspired by Kendall’s General Correlation Coefficient, and thus apt both for numerical and ordinal scales. Between two subjects i and j , the distance d ij is calculated. The matrix D ( ij ) = d ij is obtained from the pairwise distances between patients. After that, the similarity matrix ( W = 𝕀 − D ) is calculated and used as input for the clustering algorithm. The application of the Repeated Spectral Clustering (RSC) [ 26 ] algorithm to the similarity matrix W allows the identification of groups of patients with similar traits and common characteristics. The RSC algorithm is a blend of spectral clustering [ 27 ], known to work well with similarities matrices representing graphs, and consensus clustering [ 28 ], known to be powerful since it allows to deal with the variability of results that stem from random initializations. In brief, RSC takes the matrix W as input to the spectral clustering procedure. Instead of running k -means [ 29 ] just once, it is performed N RSC times to account for randomness in centroid initialization. Across these N RSC iterations, the number of times two patients are assigned to the same cluster is recorded. The more frequently two subjects are clustered together, the stronger the evidence that they should remain in the same cluster in the final assignment. The final cluster assignment is based on the co-occurrence matrix C , where each element C ij represents the number of times subjects i and j were clustered together across all runs. For a complete description of the algorithm, including technical details and its application to NIHSS scores, refer to Tshimanga and colleagues [ 26 ]. The optimal number of clusters is determined by calculating the maximum spectral gap for different cluster configurations (i.e. different k ) [ 26 ]. The elbow criterion is a simple and effective way to find the optimal number of clusters. In particular, for k = 4, the maximum spectral gap is at the end of a high plateau, indicating well-separated groups, while for k = 5, it decreases sharply, indicating subjects that are worst separated in groups. Thus, k = 4 is chosen as the optimal number of clusters. For more details, see the supplementary material section 2. 2.6 Statistical tests After applying the RSC clustering algorithm and determining the optimal number of clusters, statistical tests were conducted on the validation features to assess differences between the identified groups. The null hypothesis is that of no association between cluster label and diagnosis or remission, respectively. Both these categorical variables, have been compared between the four groups with the χ 2 independence test [ 30 ] as omnibus and post hoc for pairwise comparisons, in order to compare the observed counts against the expectations under the null hypothesis. Expected frequencies are based on marginal distributions, assuming independence. Significant deviations between observed and expected distributions indicate non-random associations and support the clinical relevance of the identified clusters. The Cramér’s V metric [ 31 ] was used as effect size for the χ 2 test; the significance level α was set to 0.05. While boolean and categorical features have been tested with the χ 2 test, numerical and ordinal features (e.g BMI) between the four groups have been compared with the Kruskal-Wallis test [ 32 ] as omnibus, and then the Mann-Whiteny U [ 33 ] as post hoc for pairwise comparisons. Finally, p-values have been corrected for multiple comparisons using the Holm [ 34 ] procedure. 2.7 Post-clustering analyses: clusters classification The XGBoost algorithm, using the same technical setup described in subsection 2.4 , was applied to the clustering features selected in subsection 2.3 and listed in Table 1 , this time to predict the cluster labels. Comparing the performance of predicting DSM-5 diagnoses versus predicting the cluster labels highlights the separability achieved by the RSC algorithm. This comparison also illustrates the extent to which the newly identified clusters, or phenotypes, are more strongly described by the selected clustering features. 3 Results The RSC clustering procedure is applied to the 19 clustering features, giving an optimal number of clusters equal to 4. The maximum gap criterium, usually used in spectral-like clustering algorithms [ 26 ], suggests k = 4 as the optimal number of classes (see the supplementary material section 2). 3.1 Clusters and DSM-5 diagnosis The partition of subjects into four clusters obtained via RSC is illustrated in Figure 1 , Panel A. The visualization is based on the Fruchterman-Reingold force-directed layout algorithm [ 35 ], where the nodes connected by edges of higher weight are positioned closer together. The pairwise values of the edge weights correspond to the entries in the co-occurrence matrix C ij , which is the number of times two patients i and j have been clustered together over the N RSC iterations performed by the RSC algorithm (see subsection 2.5 ). This representation highlights a high degree of cluster separability, with only six subjects occupying intermediate positions between clusters (IDs: 155, 271, 398, 692, 751 and 778). The same cluster layout is used in Figure 1 Panel B to illustrate the association between diagnostic categories and the identified clusters. Download figure Open in new tab Figure 1. RSC results. Graph visualization of the clustering co-occurrence matrix C . Nodes represent patients; node numbers indicate patient IDs. In Panel A node colors identifies clusters, while in Panel B identifies one of the three DSM-5 diagnosis. Darker red and thicker links patients frequently clustered together. Weak links occurring in < 5% of repetitions are omitted. The relationship between cluster assignments and both DSM-5 diagnoses and treatment remission was formally assessed using statistical tests of independence. This was done by comparing the observed frequencies of diagnostic and remission labels within each cluster to the expected frequencies under the assumption of random cluster assignment. For diagnosis, the omnibus χ 2 independent test turns out to be significant ( χ 2 (6, 809) = 248, p < .001, V =.39), with strong effect size. The strongest associations were observed between Cluster 1 and restricting anorexia, and between Cluster 2 and bulimia nervosa. Specifically, 81.8% of individuals in Cluster 1 were diagnosed with restricting anorexia, while 67.6% of those in Cluster 2 were diagnosed with bulimia nervosa. Pairwise post hoc comparisons between these two clusters and all the others were highly significant. Finally, no significant differences were found between Cluster 3 and Cluster 4, both of which included a heterogeneous mix of subjects with different diagnoses. Concerning remission, the omnibus χ 2 independent test turns out to be significant ( χ 2 (6, 809) = 13.9, p =.03, V =.09), but with weak effect size. Although none of the pairwise post hoc comparisons reached statistical significance, a descriptive pattern emerged: Cluster 1 was characterized by a higher proportion of partial remission (36.5%) and a lower rate of full remission, whereas Cluster 2 showed the opposite trend, with more individuals achieving full remission (44.7%) and fewer cases of no or partial remission. Clusters 3 and 4 displayed a more heterogeneous pattern, with few cases of partial remission and a comparable number of patients in full or no remission. 3.2 Phenotypes of the clusters Each cluster has a characteristic profile across the input clustering features. By design, the clustering algorithm yields a partition that separates the data across features. In Table 2 it is reported a summary table describing the four clusters, respect the 25 features, divided in clustering and validation features. Cluster 1 (AN-R phenotype) is composed predominantly (81.8%) by patients with restricting-type anorexia nervosa. Individuals in this group are younger than those in other clusters (18.2 ± 3.6 years), with an adolescent age of onset (15.7 ± 2.1 years), short illness duration (1.4 ± 2.3 years), markedly low BMI (16.4 ± 2.2) and minimum BMI (15.7 ± 1.8). This cluster also show the lowest levels of self-reported psychopathology, both general and ED-specific, across the sample. From a temperamental standpoint, patients in this cluster exhibit the lowest level of harm avoidance (17.7 ± 5.7). Individuals in this cluster also report the lowest number of stressful life events (6.8 ± 7.0) and experiences of childhood abuse (3.12%). Cluster 2 (BN phenotype) is composed predominantly (70.7%) of patients diagnosed with bulimia nervosa. Consistently with their diagnostic profile, individuals in this group show the highest scores on the EDI bulimia scale (12.2 ± 4.7), as well as higher current BMI (21.2 ± 4.9) and minimum lifetime BMI (17.8 ± 3.3). Notably, the latter two variables had not been explicitly included in the clustering View this table: View inline View popup Download powerpoint Table 2. RSC Clusters Summary. Summary statistics for all dataset features, grouped according to the four clusters found by the RSC algorithm. Features are divided into those used for cluster definition and those reserved for validation. AN-R refers to anorexia nervosa, restricting type; AN-BP refers to anorexia nervosa binge/purge type; BN denotes bulimia nervosa; MD indicates mean; SD indicates standard deviation. The cluster with the highest mean value for each feature is highlighted in red, while the lowest is shown in blue. analysis, further supporting the meaningfulness of the identified groups. Regarding temperamental traits, the individuals in this group display the highest levels of novelty seeking (20.4 ± 4.7) and reward dependence (15.1 ± 3.1) and the lowest level of persistence (5.4 ± 2.0). This profile suggests a tendency toward impulsivity and a low tolerance for frustration. This cluster also show the highest rates of stressful life events (12.6 ± 9.9). Cluster 3 (Mixed phenotype) include a diagnostically mixed sample of patients-comprising individuals with restrictive anorexia nervosa (39.9%), binge-purging anorexia nervosa (22.3%), and bulimia nervosa (37.8%). It is characterized by the highest levels of psychopathology, encompassing both eating disorder–specific and general psychiatric symptoms. Specifically, it shows the highest average scores across the entire EDI inventory (excluding the Bulimia subscale) and the SCL-90 (2.0 ± 0.6), compared to all other clusters. Regarding temperamental traits, the individuals in this cluster present the lowest level of reward dependence (11.3 ± 3.6) and the highest levels of harm avoidance (24.6 ± 5.7) and persistance (6.3 ± 2.1). This group also shows the highest prevalence of childhood trauma (15.54%) and recent stressful life events (11.9 ± 10.1). Patients in this group also exhibit the highest average physical activity (5.0 ± 4.0 hours/week). Cluster 4 (Older Mixed phenotype) includes patients with a mixed phenotype, as both anorexia nervosa (69.4%) and bulimia nervosa (30.6%) diagnoses are represented. Compared to the other clusters, this group is characterized by older age (28.9 ± 7.0 years), later illness onset (21.9 ± 5.7 years), and longer illness duration (4.9 ± 6.0 years). Psychopathology levels are intermediate relative to the other subgroups and the temperamental profile is marked by the lowest level of novelty seeking (13.6 ± 5.2 years). This group also shows the highest proportion of patients who did not achieve remission (38.3%). The distribution of values for each feature and cluster is shown in Figure 2 . To ensure comparability across all features, each variable was rescaled using min-max normalization to the range [0, 1 ]. For features with well-defined bounds (e.g. those from EDI and TCI), normalization was performed using their known minimum and maximum values. In contrast, features without predefined bounds (e.g. age, BMI etc.) were normalized using the empirical minimum and maximum observed inside the cohort. For boolean variables (e.g. amenorrhea) and categorical variables (e.g. remission), a stacked bar plot was used to visualize their distribution across clusters. The height of each segment within a bar represents the percentage of occurrences for each category within the corresponding cluster. Download figure Open in new tab Figure 2. Clustering phenotypes. Each panel corresponds to one cluster identified by the RSC algorithm. The x-axis lists the available features—those used for clustering appear on the left, while validation features are shown on the right. All features are normalized to the 0–1 range for comparability. The plotted values represent the mean (cross) and standard deviation (line) for each feature. Individual data points represent subjects. 3.3 Supervised learning comparison A supervised learning analysis using XGBoost [ 36 ] was conducted prior to clustering to evaluate whether the 19 selected features could accurately predict DSM-5 diagnostic categories. The model, appropriately trained and tuned, achieved a balanced accuracy of 66.1%, indicating only moderate classification performance. This limited predictive power suggests that the selected features do not fully capture the heterogeneity inherent in the DSM-5 diagnoses. In contrast, when predicting the data-driven cluster labels derived from the RSC algorithm, the same XGBoost model, appropriately trained and tuned, achieved a balanced accuracy of 92.4%, indicating great classification performance. This substantial improvement highlights that while both unsupervised and supervised approaches struggle to replicate DSM-5 groupings, they may converge on similar discriminative patterns that support more robust classification when alternative, data-driven phenotypes are employed. The optimal hyper-parameters found for both tasks are reported in the supplementary material section 1. 4 Discussion In this study, an unsupervised machine learning approach was applied to a large and phenotypically heterogeneous sample of individuals with EDs, yielding four distinct clusters. Two of these clusters demonstrated a high degree of alignment with established DSM-5 diagnostic categories, specifically, patients with anorexia nervosa restricting-type and those with bulimia nervosa. In contrast, the remaining two clusters reflected more complex and diagnostically mixed phenotypes, suggesting the presence of latent clinical subtypes not fully captured by current nosological systems. This pattern of findings is consistent with the overarching aim of the study: to examine the extent to which data-driven classifications converge with, or diverge from, traditional categorical diagnoses. On one hand, the emergence of diagnostically coherent clusters reaffirms the clinical validity of existing DSM-5 constructs. On the other, the identification of diagnostically heterogeneous subgroups highlights the phenotypic complexity and dimensional variability that characterize eating disorders, an aspect which is increasingly recognized in contemporary psychiatric research [ 37 ]. A noteworthy feature of this study is the nature of the input data. The clustering procedure relied exclusively on self-report instruments that can be easily employed in clinical practice, complemented by easily accessible clinical and demographic variables. This highlights the pragmatic potential of clustering approaches, which can be implemented using data commonly available in real-world settings [ 38 ]. Such findings support the notion that data-driven, ML-based phenotyping may offer a cost-effective and scalable strategy for enhancing diagnostic precision, guiding personalized treatment, and informing risk stratification. Future research should aim to refine these models and evaluate their integration into clinical decision-making pathways, with particular attention to their ability to delineate clinically meaningful profiles that extend beyond conventional diagnostic boundaries. 4.1 Phenotypes and targeted intervention Beyond validating the effectiveness of the clustering approach, the present findings revealed four phenotypically distinct subgroups, each characterized by a specific configuration of demographic, psychopathological, and temperamental features. These profiles allow for clinically grounded considerations regarding the phenotypic variability of EDs and their potential implications for diagnostic refinement and therapeutic planning. Among the four clusters, Cluster 1 and Cluster 2 demonstrated the highest degree of alignment with categorical DSM-5 diagnoses; Clusters 3 and 4, in contrast, showed greater diagnostic heterogeneity and phenotypic complexity, and did not correspond as clearly to established diagnostic categories. Cluster 1 (AN-R phenotype) is composed predominantly by patients with restricting-type anorexia nervosa. The clinical characteristics described in subsection 3.2 and in Table 2 are indicative of an early-stage presentation of the disorder, marked by significant physical severity (e.g. low BMI and amenorrhea). Despite this physical severity, individuals in this cluster report relatively low levels of psychopathology. These features, combined with young age and short illness duration, are consistent with previous findings indicating that, in the early stages of the disorder, egosyntonic symptoms and limited insight may attenuate subjective distress [ 39 ]. Such a profile may have therapeutic relevance, supporting the use of interventions aimed at enhancing illness awareness, strengthening motivation for change, and improving engagement in treatment tailored to restrictive anorexia nervosa. Cluster 2 (BN phenotype) is composed predominantly of patients diagnosed with bulimia nervosa. The clinical and psychological profile of this cluster reflects a pattern consistent with an affective–impulsive subtype previously described in eating disorder populations, which is characterized by the presence of emotional instability, impulsivity, and lower interpersonal avoidance. The notably higher rate of full remission observed in this cluster suggests a relatively favorable treatment response. These findings are in line with prior literature indicating that individuals with similar profiles may particularly benefit from structured, symptom-focused interventions [ 40 ], potentially accounting for the more positive outcomes seen in this group. Cluster 3 (Mixed phenotype) was defined by the highest burden of psychopathology, encompassing both ED-specific and general psychiatric symptoms. The combination of diagnostic heterogeneity and severe psychopathology, suggests a transdiagnostic profile marked by broad dysregulation across emotional, identity-related, and interpersonal domains. The elevated rates of trauma and life stressors are consistent with prior evidence linking trauma exposure to greater clinical complexity in EDs [ 41 ]. From a temperamental point of view, this cluster is characterized by avoidant and rigid personality traits, frequently described in patients with anorexia nervosa. These features are well documented in the literature and may also have relevant therapeutic implications [ 42 ]. Clinically, this subgroup may benefit from integrated, multimodal interventions that target both ED-specific symptoms and broader comorbid dimensions, such as affective instability, cognitive rigidity, and trauma-related distress. Cluster 4 (Older Mixed phenotype) includes patients with different diagnoses, characterized by a combination of older age, longer illness duration, and moderate psychopathology. These aspects suggest a more chronic and potentially stabilized clinical presentation. The presence of both anorexia nervosa and bulimia nervosa diagnoses, along with a controlling–compulsive temperament, points to a heterogeneous but less acutely severe profile. These characteristics align with patterns of persistent impairment and may indicate the need for long-term, rehabilitative treatment approaches focused on functional recovery and symptom management rather than short-term resolution. One of the aims of this study was to explore whether the data-driven phenotypic clusters were meaningfully associated with clinical outcome, defined categorically as full remission, partial remission, or no remission. Although the overall distribution of outcomes appeared relatively balanced across clusters, statistical testing rejected the hypothesis of independence, suggesting that remission status is not randomly distributed with respect to cluster assignment. Notably, Cluster 2 showed the highest proportion of full remission cases. In contrast, Cluster 1, largely composed of individuals with restrictingtype anorexia nervosa, was more frequently associated with partial remission. Clusters 3 and 4, both marked by diagnostic heterogeneity and prolonged illness duration, displayed a more unfavorable outcome profile. In particular, Cluster 4 had the highest proportion of individuals who did not reach remission, pointing to a potentially more chronic and treatment-resistant trajectory. While these patterns did not reach significance in pairwise comparisons, they may reflect clinically relevant trends that warrant further investigation. The categorical nature of the remission variable and the cross-sectional assessment limit definitive conclusions. Future studies with longitudinal designs and more nuanced outcome indicators, such as relapse rates, psychosocial functioning, or quality of life, will be necessary to assess the prognostic value of these phenotypic distinctions more robustly. 4.2 Toward a transdiagnostic understanding of EDs Beyond the formal clustering structure, our findings contribute meaningfully to the growing body of literature advocating for a dimensional and transdiagnostic reconceptualization of EDs. While two of the identified clusters closely aligned with prototypical diagnostic categories, restricting-type anorexia nervosa and bulimia nervosa, the remaining two clusters revealed more heterogeneous profiles, marked by overlapping symptomatology and divergent profiles. This duality illustrates a key point: although established diagnostic constructs retain clinical validity for a subset of patients, they may obscure important variability in others whose presentation is not adequately captured by current nosological boundaries. The identification of diagnostically mixed clusters speaks to the multifaceted nature of transdiagnostic complexity. Importantly, this complexity may take at least two distinct forms. The first may be considered a form of cross-sectional transdiagnosticity, exemplified by individuals who simultaneously exhibit elevated levels of general psychopathology, emotional dysregulation, and traumarelated vulnerability. These cases may reflect a shared vulnerability profile cutting across EDs subtypes, consistent with models that emphasize impairments in affect regulation, identity coherence, and interpersonal functioning as common etiological threads [ 43 , 44 ]. The second may represent a form of temporally diffuse transdiagnosticity, as suggested by the presence of individuals with older age and prolonged illness duration. It is possible to speculate that this pattern aligns with the phenomenon of diagnostic migration described in longitudinal studies, and may reflect the cumulative transformation or substitution of symptoms over time within the spectrum of ED [ 45 ]. These two dimensions of transdiagnosticity highlight the limitations of static diagnostic systems in fully capturing the evolving nature of eating disorders. Although categorical diagnoses can be informative at specific time points, they often do not account for symptom progression, accumulation of comorbidities, and resistance to treatment. In this regard, data-driven approaches, such as the one employed in this study, may help uncover latent phenotypes that better reflect both the underlying mechanisms and the real-world complexity of these conditions. Moreover, from a clinical standpoint, these findings suggest that different phenotypic profiles may call for distinct therapeutic strategies. Individuals belonging to diagnostically coherent clusters may respond well to targeted, diagnosis-specific interventions. In contrast, patients with mixed or complex profiles may require broader, integrative approaches that target transdiagnostic mechanisms such as emotion regulation or behavioral rigidity, and that can be flexibly adapted over time. Moreover, the ability to empirically distinguish between stable, early-stage phenotypes and long-standing, treatment-resistant trajectories offers a promising avenue for improving treatment stratification and personalizing care pathways. Ultimately, this study reinforces the relevance of transdiagnostic models in both research and clinical practice. By identifying subgroups that are not fully captured by current diagnostic labels, it supports the need for classification systems that can integrate both categorical and dimensional information, reflect clinical reality, and guide the development of mechanism-based interventions tailored to individual needs. 4.3 Strengths and limitations This study presents some important strengths. First, it is based on a large, phenotypically diverse sample of patients with eating disorders, all assessed at a single specialized center using standardized diagnostic and psychometric procedures. This uniformity in data acquisition minimizes methodological heterogeneity and enhances the internal validity of the findings. Second, the application of a novel unsupervised machine learning pipeline represents a methodological innovation within the field. Unlike more commonly used clustering techniques, this approach enhances robustness and interpretability, allowing for the identification of coherent phenotypic subgroups without relying on a priori diagnostic assumptions. The fact that these clusters emerged from easily accessible self-report and clinical data further underscores the potential for translational applications in both research and practice. Despite its strengths, some limitations of the research must also be acknowledged. First, the crosssectional design precludes any inference about the temporal stability or prognostic value of the identified phenotypes. Longitudinal data would be necessary to determine whether these clusters remain stable over time or evolve in response to treatment and illness progression. Second, the study sample consisted exclusively of female patients, limiting the generalizability of findings to male or gender-diverse populations, who may exhibit different symptom profiles, risk factors, and trajectories of illness. Third, while the use of standardized self-report instruments enhances the scalability of the clustering procedure, it also introduces potential biases related to self-perception and reporting accuracy. Behavioral or biological markers were not included and could enrich the phenotypic profiles in future studies. Finally, although the ML pipeline yielded clinically interpretable clusters, its application to other independent samples is important to evaluate reproducibility, generalizability, and clinical utility. At present, the integration of such unsupervised models into real-world clinical workflows remains exploratory, and further research is needed to determine whether these tools can meaningfully inform diagnostic decision-making, treatment planning, or prognostic stratification in diverse clinical settings. 5 Conclusions In recent years, machine learning have advanced significantly, offering new prospects across many domains, including eating disorders. Despite this promise, existing ML applications to these clinical populations have been somewhat fragmented. Notably, many of the existing studies remain anchored in traditional diagnostic labels, and fewer have fully explored unsupervised methods. In this paper, we employed RSC to a large cohort of female patients with eating disorders, relying on self-report and clinical variables. This approach yielded four robust and clinically interpretable clusters. Two of these aligned with DSM-5 prototypical diagnoses, supporting their empirical validity. However, the other two revealed diagnostically mixed profiles marked by greater psychopathological burden, prolonged illness duration, or trauma exposure. These findings underscore the value of unsupervised methods for uncovering latent phenotypic structures that may transcend conventional nosological boundaries. From a clinical perspective, our results suggest that patients fitting categorical profiles may benefit from targeted interventions, whereas those with complex or overlapping features may require more integrative and flexible treatment strategies. Overall, this work illustrates how data-driven phenotyping can inform a more nuanced, stratified understanding of EDs, ultimately contributing to the development of personalized and scalable models of care. Data Availability All data produced in the present study are available upon reasonable request to the authors. Funding This work was supported by the STARS@UNIPD funding program of the University of Padova, Italy, through the project: MEDMAX. This project has received funding from the European Union’s Horizon Europe research and innovation programme under grant agreement no 101137074 - HEREDITARY. Conflict of interest The authors have no competing interests to declare that are relevant to the content of this article. Code availability The code is available at MedMaxLab/clusteringED. Data availability The data that support the findings of this study are available upon reasonable request. Ethical statement The research was conducted in accordance with the principles embodied in the Declaration of Helsinki and in accordance with local statutory requirements. All participants (or their parent or legal guardian in the case of children under 18) gave written informed consent to participate in the study. All subjects gave their informed written consent to the use of data in an anonymous form. The Ethical Committee of Padova Hospital approved the study (protocol n ◦ 1598P). Authors’ contributions AZ - Conceptualization, Methodology, Statistical Analysis, Writing - Original Draft. LFT - Methodology, Writing - Original Draft. VM - Medical Interpretation, Writing - Original Draft. AF - Conceptualization, Medical Interpretation, Supevision. MA - Conceptualization, Methodology, Project Administration, Supervision, Writing - Original Draft. EC - Conceptualization, Medical Interpretation, Supervision, Writing - Original Draft. All authors reviewed and edited previous versions of the manuscript. 6 Acknowledgment References [1]. ↵ M. Solmi et al. , “ Outcomes in people with eating disorders: a transdiagnostic and disorder-specific systematic review, meta-analysis and multivariable meta-regression analysis ,” World Psychiatry , vol. 23 , no. 1 , pp. 124 – 138 , 2024 . OpenUrl CrossRef PubMed [2]. ↵ “Diagnostic and Statistical Manual of Mental Disorders | Psychiatry Online.” [3]. ↵ S. Zipfel , K. E. Giel , C. M. Bulik , P. Hay , and U. Schmidt , “ Anorexia nervosa: aetiology, assessment, and treatment ,” The Lancet Psychiatry , vol. 2 , no. 12 , pp. 1099 – 1111 , Dec . 2015 . OpenUrl [4]. ↵ P. K. Keel et al. , “ Application of a Latent Class Analysis to Empirically Define EatingDisorder Phenotypes ,” Archives of General Psychiatry , vol. 61 , no. 2 , pp. 192 – 200 , Feb . 2004 . OpenUrl CrossRef PubMed Web of Science [5]. ↵ C. J. Foldi and K. R. Griffiths , “ Examining the biological causes of eating disorders to inform treatment strategies ,” Nature Reviews Neuroscience , Jun 2025 . [6]. ↵ Z. Li , J. Leppanen , J. Webb , P. Croft , S. Byford , and K. Tchanturia , “ Analysis of symptom clusters amongst adults with anorexia nervosa: Key severity indicators ,” Psychiatry Research , vol. 326 , p. 115272 , Aug . 2023 . OpenUrl [7]. ↵ M. Solmi , E. Collantoni , P. Meneguzzo , D. Degortes , E. Tenconi , and A. Favaro , “ Network analysis of specific psychopathology and psychiatric symptoms in patients with eating disorders ,” International Journal of Eating Disorders , vol. 51 , no. 7 , pp. 680 – 692 , July 2018 , epub 2018 May 30. OpenUrl [8]. ↵ C. G. Fairburn , Z. Cooper , and R. Shafran , “ Cognitive behaviour therapy for eating disorders: a “transdiagnostic” theory and treatment ,” Behaviour Research and Therapy , vol. 41 , no. 5 , pp. 509 – 528 , May 2003 . OpenUrl CrossRef PubMed Web of Science [9]. ↵ D. Bennett , S. M. Silverstein , and Y. Niv , “ The Two Cultures of Computational Psychiatry ,” JAMA Psychiatry , vol. 76 , no. 6 , pp. 563 – 564 , Jun . 2019 . OpenUrl [10]. ↵ M. Vervaet , L. Puttevils , R. H. A. Hoekstra , E. Fried , and M.-A. Vanderhasselt , “ Transdiagnostic vulnerability factors in eating disorders: A network analysis ,” European Eating Disorders Review , vol. 29 , no. 1 , pp. 86 – 100 , 2021 . OpenUrl [11]. ↵ J. Fardouly , R. D. Crosby , and S. Sukunesan , “ Potential benefits and limitations of machine learning in the field of eating disorders: current research and future directions ,” Journal of Eating Disorders , vol. 10 , no. 1 , p. 66 , May 2022 . OpenUrl [12]. ↵ N. Stewart Rosenfield and E. Linstead , “ Exploring the Eating Disorder Examination Questionnaire, Clinical Impairment Assessment, and Autism Quotient to Identify Eating Disorder Vulnerability: A Cluster Analysis ,” Machine Learning and Knowledge Extraction , vol. 2 , no. 3 , pp. 347 – 360 , Sep . 2020 , number: 3 Publisher: Multidisciplinary Digital Publishing Institute . OpenUrl [13]. ↵ A. F. Haynos et al. , “ Machine learning enhances prediction of illness course: a longitudinal study in eating disorders ,” Psychological Medicine , vol. 51 , no. 8 , pp. 1392 – 1402 , Jun . 2021 . OpenUrl [14]. ↵ A. S. Juarascio , M. N. Parker , M. A. Lagacey , and K. M. Godfrey , “ Just-in-time adaptive interventions: A novel approach for enhancing skill utilization and acquisition in cognitive behavioral therapy for eating disorders ,” International Journal of Eating Disorders , vol. 51 , no. 8 , pp. 826 – 830 , 2018 . OpenUrl [15]. ↵ A. M. Chekroud et al. , “ The promise of machine learning in predicting treatment outcomes in psychiatry ,” World Psychiatry , vol. 20 , no. 2 , pp. 154 – 170 , 2021 , _eprint: https://onlinelibrary.wiley.com/doi/pdf/10.1002/wps.20882 . OpenUrl CrossRef PubMed [16]. ↵ E. Millgate et al. , “ Cross-sectional study comparing cognitive function in treatment responsive versus treatment non-responsive schizophrenia: evidence from the strata study ,” BMJ open , vol. 11 , no. 11 , p. e054160 , November 2021 . OpenUrl Abstract / FREE Full Text [17]. ↵ D. B. Dwyer et al. , “ Brain subtyping enhances the neuroanatomical discrimination of schizophrenia ,” Schizophrenia Bulletin , vol. 44 , no. 5 , pp. 1060 – 1069 , 02 2018 . OpenUrl CrossRef [18]. ↵ N. Bak et al. , “ Two subgroups of antipsychotic-naive, first-episode schizophrenia patients identified with a gaussian mixture model on cognition and electrophysiology ,” Translational Psychiatry , vol. 7 , no. 4 , pp. e1087 – e1087 , Apr 2017 . OpenUrl [19]. ↵ Y. Jiang et al. , “Two neurostructural subtypes: results of machine learning on brain images from 4,291 individuals with schizophrenia,” Oct . 2023 , pages: 2023.10.11.23296862. [20]. ↵ C. M. Ulbricht , S. A. Chrysanthopoulou , L. Levin , and K. L. Lapane , “ The use of latent class analysis for identifying subtypes of depression: A systematic review ,” Psychiatry Research , vol. 266 , pp. 228 – 246 , 2018 . OpenUrl CrossRef PubMed [21]. ↵ P. Santonastaso , R. Bosello , P. Schiavone , E. Tenconi , D. Degortes , and A. Favaro , “ Typical and atypical restrictive anorexia nervosa: Weight history, body image, psychiatric symptoms, and response to outpatient treatment ,” International Journal of Eating Disorders , vol. 42 , no. 5 , pp. 464 – 470 , 2009 . OpenUrl CrossRef PubMed [22]. ↵ D. M. Garner , M. P. Olmstead , and J. Polivy , “ Development and validation of a multidimensional eating disorder inventory for anorexia nervosa and bulimia ,” International Journal of Eating Disorders , vol. 2 , no. 2 , pp. 15 – 34 , 1983 . OpenUrl CrossRef Web of Science [23]. ↵ C. R. Cloninger , D. M. Svrakic , and T. R. Przybeck , “ A psychobiological model of temperament and character ,” Archives of General Psychiatry , vol. 50 , no. 12 , pp. 975 – 990 , 12 1993 . OpenUrl CrossRef PubMed Web of Science [24]. ↵ L. R. Derogatis , “ Scl-90-r: Administration, scoring and procedures ,” Manual II for the R (evised) version and other instruments of the psychopathology rating scale series , 1983 . [25]. ↵ M. Schwaiger and O. Opitz K. Jajuga , M. Walesiak , and A. Bak , “ On the general distance measure ,” in Exploratory Data Analysis in Empirical Research , M. Schwaiger and O. Opitz , Eds. Berlin, Heidelberg : Springer Berlin Heidelberg , 2003 , pp. 104 – 109 . [26]. ↵ L. F. Tshimanga , A. Zanola , S. Facchini , A. L. Bisogno , L. Pini , M. Atzori , and M. Corbetta , “ Behavioral clusters and lesion distributions in ischemic stroke, based on nihss similarity network ,” Journal of Healthcare Informatics Research , Mar 2025 . [27]. ↵ U. von Luxburg , “ A tutorial on spectral clustering ,” Statistics and Computing , vol. 17 , no. 4 , pp. 395 – 416 , Dec 2007 . OpenUrl CrossRef [28]. ↵ A. Fred and A. Jain , “ Data clustering using evidence accumulation ,” in 2002 International Conference on Pattern Recognition , vol. 4 , 2002 , pp. 276 – 280 vol.4. OpenUrl [29]. ↵ J. MacQueen , “ Multivariate observations ,” in Proceedings ofthe 5th Berkeley Symposium on Mathematical Statisticsand Probability , vol. 1 , 1967 , pp. 281 – 297 . OpenUrl [30]. ↵ A. Agresti , An Introduction to Categorical Data Analysis , 2nd ed. Hoboken, NJ : Wiley , 2007 . [31]. ↵ H. Cramér , Mathematical Methods of Statistics, ser. Goldstine Printed Materials . Princeton University Press , 1946 . [32]. ↵ W. H. Kruskal and W. A. Wallis , “ Use of ranks in one-criterion variance analysis ,” Journal of the American Statistical Association , vol. 47 , no. 260 , pp. 583 – 621 , 1952 . OpenUrl CrossRef Web of Science [33]. ↵ H. B. Mann and D. R. Whitney , “ On a test of whether one of two random variables is stochastically larger than the other ,” The annals of mathematical statistics , pp. 50 – 60 , 1947 . [34]. ↵ S. Holm , “ A simple sequentially rejective multiple test procedure ,” Scandinavian Journal of Statistics , vol. 6 , no. 2 , pp. 65 – 70 , 1979 . OpenUrl CrossRef PubMed Web of Science [35]. ↵ T. M. Fruchterman and E. M. Reingold , “ Graph drawing by force-directed placement ,” Software: Practice and experience , vol. 21 , no. 11 , pp. 1129 – 1164 , 1991 . OpenUrl CrossRef Web of Science [36]. ↵ T. Chen and C. Guestrin , “ Xgboost: A scalable tree boosting system ,” in Proceedings of the 22nd ACM SIGKDD International Conference on Knowledge Discovery and Data Mining, ser. KDD ‘16 . New York, NY, USA : Association for Computing Machinery , 2016 , p. 785 – 794 . [37]. ↵ J. E. Wildes , M. D. Marcus , R. D. Crosby , R. M. Ringham , M. Marin Dapelo , J. A. Gaskill , and K. T. Forbush , “ The clinical utility of personality subtypes in patients with anorexia nervosa ,” Journal of Consulting and Clinical Psychology , vol. 79 , no. 5 , p. 665 – 674 , 2011 . OpenUrl CrossRef PubMed [38]. ↵ Y. Wang et al. , “ Unsupervised machine learning for the discovery of latent disease clusters and patient subgroups using electronic health records ,” Journal of Biomedical Informatics , vol. 102 , p. 103364 , 2020 . OpenUrl CrossRef PubMed [39]. ↵ J. E. Steinglass , D. R. Glasofer , M. Dalack , and E. Attia , “ Between wellness, relapse, and remission: Stages of illness in anorexia nervosa ,” International Journal of Eating Disorders , vol. 53 , no. 7 , pp. 1088 – 1096 , 2020 . OpenUrl [40]. ↵ L. M. Schaefer , G. Forester , E. N. Dougherty , A. R. Bottera , E. E. Forbes , and J. E. Wildes , “ Do empirically-derived personality subtypes relate to cognitive inflexibility in anorexia nervosa and bulimia nervosa? ” Journal of Eating Disorders , vol. 12 , p. 212 , 2024 . OpenUrl [41]. ↵ I. Moroshko , A. Raspovic , J. Liu , and L. Brennan , “ Trauma and eating disorders: an integrated umbrella and scoping review ,” Clinical Psychology Review , vol. 119 , p. 102592 , 2025 . OpenUrl [42]. ↵ K. L. Klump , M. Strober , C. M. Bulik , L. Thornton , C. Johnson , B. Devlin , and W. H. Kaye , “ Personality characteristics of women before and after recovery from an eating disorder ,” Psychological Medicine , vol. 34 , no. 8 , pp. 1407 – 1418 , 2004 . OpenUrl PubMed [43]. ↵ J. M. Lavender , S. A. Wonderlich , S. G. Engel , K. H. Gordon , W. H. Kaye , and J. E. Mitchell , “ Dimensions of emotion dysregulation in anorexia nervosa and bulimia nervosa: A conceptual review of the empirical literature ,” Clinical Psychology Review , vol. 40 , pp. 111 – 122 , 2015 . OpenUrl [44]. ↵ C. G. Fairburn , Z. Cooper , and R. Shafran , “ Cognitive behaviour therapy for eating disorders: a “transdiagnostic” theory and treatment ,” Behaviour Research and Therapy , vol. 41 , no. 5 , pp. 509 – 528 , 2003 . OpenUrl CrossRef PubMed Web of Science [45]. ↵ K. T. Eddy , D. J. Dorer , D. L. Franko , K. Tahilani , H. Thompson-Brenner , and D. B. Herzog , “ Diagnostic crossover in anorexia nervosa and bulimia nervosa: implications for dsm-v ,” American Journal of Psychiatry , vol. 165 , no. 2 , pp. 245 – 250 , 2008 . OpenUrl CrossRef PubMed Web of Science View the discussion thread. Back to top Previous Next Posted August 27, 2025. Download PDF Supplementary Material Data/Code Email Thank you for your interest in spreading the word about medRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following Stratifying Eating Disorders with Clustering: From Diagnosis to Phenotypic Diversity Message Subject (Your Name) has forwarded a page to you from medRxiv Message Body (Your Name) thought you would like to see this page from the medRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share Stratifying Eating Disorders with Clustering: From Diagnosis to Phenotypic Diversity Andrea Zanola , Louis Fabrice Tshimanga , Valentina Meregalli , Angela Favaro , Manfredo Atzori , Enrico Collantoni medRxiv 2025.08.25.25334344; doi: https://doi.org/10.1101/2025.08.25.25334344 Share This Article: Copy Citation Tools Stratifying Eating Disorders with Clustering: From Diagnosis to Phenotypic Diversity Andrea Zanola , Louis Fabrice Tshimanga , Valentina Meregalli , Angela Favaro , Manfredo Atzori , Enrico Collantoni medRxiv 2025.08.25.25334344; doi: https://doi.org/10.1101/2025.08.25.25334344 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Health Informatics Subject Areas All Articles Addiction Medicine (568) Allergy and Immunology (863) Anesthesia (300) Cardiovascular Medicine (4435) Dentistry and Oral Medicine (444) Dermatology (382) Emergency Medicine (608) Endocrinology (including Diabetes Mellitus and Metabolic Disease) (1509) Epidemiology (15227) Forensic Medicine (30) Gastroenterology (1124) Genetic and Genomic Medicine (6597) Geriatric Medicine (668) Health Economics (997) Health Informatics (4534) Health Policy (1368) Health Systems and Quality Improvement (1613) Hematology (540) HIV/AIDS (1264) Infectious Diseases (except HIV/AIDS) (15916) Intensive Care and Critical Care Medicine (1103) Medical Education (623) Medical Ethics (146) Nephrology (667) Neurology (6599) Nursing (346) Nutrition (998) Obstetrics and Gynecology (1144) Occupational and Environmental Health (957) Oncology (3332) Ophthalmology (974) Orthopedics (369) Otolaryngology (420) Pain Medicine (436) Palliative Medicine (130) Pathology (663) Pediatrics (1693) Pharmacology and Therapeutics (691) Primary Care Research (711) Psychiatry and Clinical Psychology (5447) Public and Global Health (9230) Radiology and Imaging (2198) Rehabilitation Medicine and Physical Therapy (1370) Respiratory Medicine (1196) Rheumatology (593) Sexual and Reproductive Health (712) Sports Medicine (530) Surgery (712) Toxicology (99) Transplantation (289) Urology (265) (function(){function c(){var b=a.contentDocument||a.contentWindow.document;if(b){var d=b.createElement('script');d.innerHTML="window.__CF$cv$params={r:'a0037fc44eaadfa9',t:'MTc3OTUzMzQyMA=='};var a=document.createElement('script');a.src='/cdn-cgi/challenge-platform/scripts/jsd/main.js';document.getElementsByTagName('head')[0].appendChild(a);";b.getElementsByTagName('head')[0].appendChild(d)}}if(document.body){var a=document.createElement('iframe');a.height=1;a.width=1;a.style.position='absolute';a.style.top=0;a.style.left=0;a.style.border='none';a.style.visibility='hidden';document.body.appendChild(a);if('loading'!==document.readyState)c();else if(window.addEventListener)document.addEventListener('DOMContentLoaded',c);else{var e=document.onreadystatechange||function(){};document.onreadystatechange=function(b){e(b);'loading'!==document.readyState&&(document.onreadystatechange=e,c())}}}})();

Text is read by the "Ask this paper" AI Q&A widget below. Extraction quality varies by source — PMC NXML preserves structure cleanly, OA-HTML may include some navigation residue, and OA-PDF can have broken hyphenation. The publisher copy (via DOI) is the canonical version.

My notes (saved in your browser only)

Ask this paper AI returns verbatim quotes from the full text · source: preprint-html

Answers must be backed by verbatim quotes from this paper's full text. Hallucinated quotes are dropped automatically; if no verbatim passage answers the question, we say so. How this works

Citation neighborhood (no data yet)

We don't have any in-corpus citations linked to this paper yet. This is a recent paper (2025) — citers typically take a year or two to land, and the OpenAlex reference graph may still be filling in.

Source provenance

europepmc
last seen: 2026-05-20T01:45:00.602351+00:00