Machine Learning Models for Predicting Multiple Myeloma Staging and MGUS Progression Using Gene Expression Data

preprint OA: closed
📄 Open PDF Full text JSON View at publisher

Abstract

In this study, we developed and evaluated Machine Learning (ML) models aimed at predicting the stage of multiple myeloma (MM) and the progression of monoclonal gammopathy of undetermined significance (MGUS) to MM. Accurate staging of MM is critical for determining appropriate treatment strategies, and our models, employing algorithms such as ElasticNet, Random Forest, Boosting, and Support Vector Machines, demonstrated high efficacy in capturing the biological differences across disease stages. Among these, the ElasticNet model exhibited strong generalizability, achieving consistent multiclass AUC values across various datasets and data transformations. Predicting MGUS progression to MM presents a significant challenge due to the scarcity of MGUS cases that have progressed. We employed a two-pronged approach to address this: developing models using a limited dataset containing progressing MGUS patients and training models on combined MGUS and MM datasets. The models achieved AUC values slightly above 0.8, particularly with ElasticNet, Boosting and Support Vector Machines, indicating their potential in stratifying MGUS patients by progression risk. This study is original in integrating MM data with MGUS cases to enhance the predictive accuracy of MGUS progression, offering a novel methodology with potential clinical applications in patient monitoring and early intervention. Our feature selection and enrichment analyses further revealed that the identified genes are involved in key signaling pathways, including PI3K-Akt, MAPK, Wnt, and mTOR, all of which play crucial roles in MM pathogenesis. These findings align with established biological knowledge, suggest possible therapeutic targets and increase the explainability of our models.
Full text 57,056 characters · extracted from preprint-html · click to expand
Machine Learning Models for Predicting Multiple Myeloma Staging and MGUS Progression Using Gene Expression Data | bioRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-M677548'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search New Results Machine Learning Models for Predicting Multiple Myeloma Staging and MGUS Progression Using Gene Expression Data View ORCID Profile Nestoras Karathanasis , View ORCID Profile George M. Spyrou doi: https://doi.org/10.1101/2024.11.12.623149 Nestoras Karathanasis 1 Bioinformatics Department, The Cyprus Institute of Neurology & Genetics , 6 Iroon Avenue, 2371 Ayios Dometios, Nicosia, Cyprus Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Nestoras Karathanasis For correspondence: nestorask{at}cing.ac.cy George M. Spyrou 1 Bioinformatics Department, The Cyprus Institute of Neurology & Genetics , 6 Iroon Avenue, 2371 Ayios Dometios, Nicosia, Cyprus Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for George M. Spyrou Abstract Full Text Info/History Metrics Supplementary material Preview PDF Abstract In this study, we developed and evaluated Machine Learning (ML) models aimed at predicting the stage of multiple myeloma (MM) and the progression of monoclonal gammopathy of undetermined significance (MGUS) to MM. Accurate staging of MM is critical for determining appropriate treatment strategies, and our models, employing algorithms such as ElasticNet, Random Forest, Boosting, and Support Vector Machines, demonstrated high efficacy in capturing the biological differences across disease stages. Among these, the ElasticNet model exhibited strong generalizability, achieving consistent multiclass AUC values across various datasets and data transformations. Predicting MGUS progression to MM presents a significant challenge due to the scarcity of MGUS cases that have progressed. We employed a two-pronged approach to address this: developing models using a limited dataset containing progressing MGUS patients and training models on combined MGUS and MM datasets. The models achieved AUC values slightly above 0.8, particularly with ElasticNet, Boosting and Support Vector Machines, indicating their potential in stratifying MGUS patients by progression risk. This study is original in integrating MM data with MGUS cases to enhance the predictive accuracy of MGUS progression, offering a novel methodology with potential clinical applications in patient monitoring and early intervention. Our feature selection and enrichment analyses further revealed that the identified genes are involved in key signaling pathways, including PI3K-Akt, MAPK, Wnt, and mTOR, all of which play crucial roles in MM pathogenesis. These findings align with established biological knowledge, suggest possible therapeutic targets and increase the explainability of our models. 1 Introduction Multiple myeloma constitutes approximately 1% of all cancer cases and about 10% of hematologic malignancies 1 , 2 . Annually, more than 32,000 new cases are diagnosed in the United States, with nearly 13,000 resulting in fatalities 3 . The yearly age-adjusted incidence has remained steady for decades, hovering around 4 per 100,000 individuals 4 . It shows a slight preference for men over women and is twice as prevalent among African Americans compared to Caucasians 5 . The median age at diagnosis is typically around 65 years 6 . Nearly all multiple myeloma patients progress from an asymptomatic precursor stage known as monoclonal gammopathy of undetermined significance (MGUS) 7 , 8 . MGUS is found in roughly 5% of individuals aged over 50, with a prevalence around twice as high among Blacks compared to Whites 9 – 12 . MGUS transitions to multiple myeloma or related malignancies at a rate of 1% per year 13 , 14 . As MGUS is asymptomatic, over 50% of those diagnosed with it have likely harboured the condition for over a decade before clinical diagnosis 15 . In particular cases, an intermediate asymptomatic but more advanced pre-malignant stage, termed smoldering multiple myeloma (SMM), may be clinically recognisable 16 . SMM progresses to multiple myeloma at a rate of approximately 10% per year within the first five years post-diagnosis, followed by 3% annually over the subsequent five years and 1.5% per year thereafter. This progression rate is influenced by the underlying cytogenetic profile, with patients harbouring specific translocations at a higher risk of progressing from MGUS or SMM to multiple myeloma 17 – 19 . Despite notable therapeutic advancements in recent years, multiple myeloma (MM) remains an uncurable disease. Enhanced insights into MM’s biology and pathogenesis have prompted a transformative shift in managing MM and its precursor states, monoclonal gammopathy of undetermined significance (MGUS) and smoldering multiple myeloma (SMM) 20 . The conventional notion that MM treatment should only start upon the onset of symptoms has been challenged by the introduction of novel therapies characterised by both safety and efficacy. Clinical trials have underscored the significance of initiating treatment early in high-risk asymptomatic cases, demonstrating a marked delay in disease progression and improved progression-free survival outcomes for patients 21 , 22 . Yet, a critical challenge persists in identifying individuals with asymptomatic myeloma at the highest risk of progression, thereby maximising the benefits of early treatment strategies. While risk stratification models such as the Mayo Clinic model 23 and the Spanish model 24 have been valuable, they still possess notable limitations, particularly in the context of modern therapies. Studies have revealed that patients with high-risk cytogenetic MM, including del17p, t(4;14), or t(14;20), may achieve survival rates comparable to standard-risk patients through intensified treatment regimens involving a combination of proteasome inhibitors, immunomodulatory drugs, and autologous stem cell transplantation 25 . Consequently, there is an urgent imperative to deepen our comprehension of the molecular mechanisms underpinning disease progression and refine risk stratification models for asymptomatic MM concurrently with endeavours to optimise early treatment strategies. Over the past several years, there has been a notable increase in the utilization of machine learning (ML) algorithms and deep learning (DL) procedures for tumor detection. To manage cancer patients, these methods leverage diverse data sources such as proteomic, genomic, histopathological data, or images. These techniques have proven to be beneficial not only in the realm of solid tumors but also in the management of hematological malignancies. Recent reports on multiple myeloma have highlighted the significance of machine learning in diagnosing, prognosticating, response to treatment and evaluating therapeutic responses in hematological neoplasms 26 . Currently, serum markers are employed to categorize MGUS patients into different clinical risk groups. However, no established molecular signature can reliably predict the progression of MGUS. To address this gap, Sun et al. 27 conducted a study utilizing gene expression profiling to stratify the risk of MGUS and devised a signature based on extensive samples with long-term follow-up. They analyzed microarrays of plasma cell mRNA from 334 MGUS patients with stable disease and 40 MGUS patients who progressed to multiple myeloma (MM) within a decade and identified a thirty-six-gene molecular signature indicative of MGUS risk. The objectives of this study are: (1) develop machine-learning models capable of accurately predicting the stage of multiple myeloma (MM) based on microarray datasets. We utilized advanced algorithms to analyse molecular data to classify patients into different stages of the disease, thereby aiding clinicians in making more informed treatment decisions. (2) To investigate the feasibility of using machine-learning techniques to predict disease progression from monoclonal gammopathy of undetermined significance (MGUS) to multiple myeloma (MM). By leveraging microarray datasets containing gene expression profiles and clinical information from patients at the MGUS stage, the study aims to develop predictive models that identify individuals at high risk of progressing to MM. Models trained to distinguish MGUS from MM were tested for their effectiveness in identifying progressing MGUS cases 27 , with results indicating similar or better performance to models explicitly trained for this task. This proactive approach aims to enable early intervention strategies and improve patient outcomes by potentially delaying or preventing disease progression. 2 Method 2.1 Source of Microarray Datasets and Description of Data Variables and Features We downloaded seven microarray datasets, two from ArrayExpress and five from the Gene Expression Omnibus. In all cases, the samples were CD-138+ bone marrow plasma cells from patients with different stages of multiple myeloma (MGUS, MM) and Healthy. The datasets come from four different platforms (A-AFFY-33, A-AFFY-44, GPL96, GPL570) and contain different numbers of patients in total and per stage, see Table 1 . View this table: View inline View popup Download powerpoint Table 1. The number of samples per dataset and disease stage. The table is sorted by the total number of samples. The empty cells correspond to a zero number of samples. For each dataset, we downloaded the raw .cel files. We calculated the expression matrix using the “Robust Multi-Array Average” expression measure via the rma() function of the affy or oligo R packages, depending on the requirements of each dataset, with background correction. At this step of the analysis, data was not normalized. Datasets from different platforms have different numbers of probes. GLP96, A-AFFY-33/A-AFFY-34 contain ∼22.000 probes, whereas GLP570 and A-AFFY-44 contain ∼55.000. We retained only the 22.277 shared probes across all datasets. Also, datasets contained samples corresponding to disease stages outside of the scope of this study (for example, SMM, relapse MM, PCL, and HUVEC) which we removed from our analysis. 2.2 Data Cleaning and Pre-processing Techniques We employed several data transformation and normalization techniques to prepare our datasets for analysis, (refer to Figure 1 ): Download figure Open in new tab Figure 1. Flowchart of the analysis. A) The flowchart illustrates the process used for predicting the stage of Multiple Myeloma. The method encompasses multiple steps: data preprocessing, model training, and performance evaluation, applied across various datasets. Preprocessing includes several data transformations and the training phase incorporates a variety of machine learning models. After predictions, the model’s key features were interpreted through enrichment analyses. In the figure, (ps) indicates per-sample preprocessing, (train) denotes that normalization was applied to training samples, and (test) refers to applying the parameters learned from training to the test set. B) The flowchart outlines the process used for predicting the progression of MGUS to MM using machine learning techniques. The method involves preprocessing, model training, and performance evaluation using different datasets similar to A. The boxes with a black background indicate the use of the GSE235356 dataset for training and testing in a 10-fold nested cross-validation fashion. In contrast, grey background boxes represent training on various datasets and testing on the GSE235356 dataset. Robust Multi-Array Average (rma) We utilized rma function with data background correction, which is implemented in affy or oligo R packages, depending on the requirements of each dataset. Binary Conversion Expression values from rma were converted to binary (0-1) using two quantile thresholds, 0 (binary_0) and 0.5 (binary_0.5) per sample. Values exceeding the quantile threshold were set to 1, while those equal to or below the threshold were set to 0. In the case of binary_0, all values except the minimum were set to 1, and the minimum value was set to 0. Binary_0 was used as a negative control, where we expected the machine learning algorithms to perform as random classifiers, offering a baseline for performance comparison. Ranking (ranking) Expression values were ranked from 0 to 1, with the highest value assigned a rank of 1. This ranking system provided a relative measure of gene expression levels within each sample. Ratios (ratio) We selected only healthy samples from the GSE6477 dataset, which served solely as a training set. We calculated the ratios performing the following steps. First, we used the ranks from the ranking transformation, and calculated the standard deviation of each probe. We kept 210 probes with the lowest standard deviation. This number was chosen to minimize feature combinations, as the total number of combinations when selecting two genes each time was 21945 features—close to the total number of features from the other pre-processing approaches. Then we calculated all possible ratios of these probes. Quantile Normalization (qnorm) Quantile normalization was applied in a train-test fashion using the preprocess R package. The training set underwent quantile normalization, and the parameters learned from this process were then applied to the test set. This approach ensured consistency in data distribution between the training and test datasets. 2.3 Overview of Machine Learning (ML) Algorithms We evaluated the following parametric and non-parametric methods, (see Figure 1 ). ElasticNet (glmnet) is a parametric method that fits generalized linear and similar models via penalized maximum likelihood 28 . We employed its implementation in the glmnet package in R. ElasticNet’s advantage is that it is the most interpretable ML method 29 among these mentioned here. Random Forest (rf) is a non-parametric tree-based method. We utilized its implementation in the randomForest R package. RF is somewhat interpretable as it provides information on which features are more important for the model by calculating variable importance scores 29 . Boosting (gbm) is a non-parametric method. We used gradient boosting machines implemented in the gbm package in R. Like RF, boosting is somewhat interpretable and provides the most important features 30 . Support Vector Machines (SVM) is a non-parametric method that has the advantage of projecting the data to a different feature space 29 . However, even though SVMs can produce very accurate models, they lack interpretability. We used the implementation of SVMs in the e1071 31 R package to fit an SVM with the linear kernel (svmLinear2) and the implementation in the kernlab 32 R package to fit an SVM with the radial kernel (svmRadial). 2.4 Models Training and Interpretation We utilised the caret R package, which stands for classification and regression training 33 , to train, optimise, and test our models, (see Figure 1 ). In order to optimize the model’s hyperparameters, we employed a ten-fold cross-validation repeated ten times. As the performance metric to determine the best model, we used the multiclass area under the ROC curve (multiclass_auc) for multiclass problems (see task 1 below) and the area under the ROC curve (AUC) for two-class problems (see task 2 below). For all models, except svmRadial, we tuned our models in a set of ten hyperparameters by setting caret’s tuneLength argument to ten. For the svmRadial model, we used the sigest() function from the kernlab R package to calculate the range of the sigma hyperparameter. The cost hyperparameter was set to the following values: 0.25, 0.50, 2, 4, 8, 16, 32, 64, 128, 256, 512, 1024. In all cases, the data were centred and scaled. To interpret our models, we calculated the importance of each feature by utilizing the varImp() function from the caret R package. We then performed enrichment analysis for GO biological processes, KEGG and Reactome pathways, and disease ontology semantics using the clusterProfiler R package 34 .Last, we filtered the results with the following terms related to multiple myeloma, MAPK, RAS, RAF, MEK, ERK, ERK1, ERK2, PI3K, AKT, NF-KB, Jak-STAT, Wnt, Hedgehog, TNFa, mTOR, multiple myeloma, myeloid, leukemia, myeloma, Plasmacytoma, Amyloidosis, Chronic Lymphocytic Leukemia, Heavy Chain Disease, Lymphoma 35 , 36 . 3 Results 3.1 Task 1 - Predicting the Stage of Multiple Myeloma 3.1.1 Model Development for Disease Staging We trained our models using the GSE6477 microarray dataset 37 . This dataset comprises 162 samples representing various stages of myeloma. Specifically, it includes 15 samples classified as Normal, 21 as MGUS (Monoclonal Gammopathy of Undetermined Significance), 23 as SMM (Smoldering Multiple Myeloma), 75 as MM (newly diagnosed myeloma), and 28 as RMM (relapsed myeloma samples). We focused on 110 samples after excluding the SMM and RMM categories. We trained our models to separate the three classes: Normal, MGUS and MM. We used all other datasets (see Table 1 ) only for testing. 3.1.2 Evaluation of Model Performance During training, all models consistently achieved a multiclass_auc with a cross-validation median ranging from 0.9 to 1 across various data transformations and machine-learning methods (refer to Supplementary Figure 1). For the binary_0 transformation, where all expressions except from the lowest were set to 1, the median training performance of all models was around 0.5, that corresponds to random classifier, as expected. Subsequently, we evaluated the multiclass_auc for all test datasets using all models and data transformations (see Figure 2 and Supplementary Figure 2). Download figure Open in new tab Figure 2. Models multiclass auc in the external validation sets. A) The performance to the external dataset used across all data transformations and machine learning algorithms. B) The relation of performance to the data transformations across datasets generated in GLP96 or A.AFFY.34 platforms and all machine learning algorithms. C) The relation of performance to the machine learning algorithms across datasets generated in GLP96 or A.AFFY.34 platforms and all data transformations. Focusing on the platform of origin for the data, we noted that datasets (GSE13591, GSE2113) originating from the same platform (GPL96) as the training set exhibited similar multiclass_auc scores as seen during training, (see Figure 2A ). We observed a slight decline in performance, approximately 0.1 (see Supplementary Figure 3). EMTAB316 originated from A-AFFY-34, which is very close to GPL96, showed similar multiclass_auc scores with training. In contrast, for datasets generated using different platforms (EMTAB317 from A-AFFY-44 and GSE5900, GSE235356 from GPL570), our models experienced a more significant decrease in performance. This suggests that performance variability across datasets may be attributed to differences in the platforms used for data generation. Regarding GSE235356, it’s important to note that this dataset includes only MGUS and progressing MGUS samples, which could contribute to the observed decline in performance. Download figure Open in new tab Figure 3. The number of features utilized by each model across different data transformations. The plot shows the variation in feature selection for each model, highlighting the range of features used in the analysis. For the subsequent phases of our analysis, we focused on datasets generated from the GPL96 and A-AFFY-34 platforms. Concerning data transformations, we found that binary_0.5 yielded the highest multiclass_auc in the test datasets and exhibited less performance degradation compared to training results across all machine learning algorithms (refer to Figure 2B ). Following binary_0.5, ranking, qnorm, rma, and ratios transformations were observed. In terms of machine learning algorithms, we observed that glmnet demonstrated the highest multiclass_auc in the test datasets across all data transformations and showed less performance degradation relative to the training phase (see Figure 2C ). Succeeding glmnet, rf, svmlinear2, gbm and svmRadial were observed. 3.1.3 Models Interpretation We calculated the importance of each feature for each model and data transformation combination. The svmLinear2 and svmRadial models utilized all available features (22,277). RandomForest models used between 1,260 and 4,866 features, while gbm models employed fewer features, ranging from 228 to 3,371. The glmnet models used the least features, with counts between 197 and 798 (see Figure 3 ). We observed that most selected features were specific to the model and transformation used. For the rma transformation, 72 probes were selected across glmnet, gbm, and rf models. Similarly, 62 probes, 58 probes, 101 probes, and 37 ratios were selected for the qnorm, ranking, binary_0.5, and ratios transformations, respectively (see Supplementary Figure 3A). Within the same model type (glmnet, gbm, rf), there was minimal overlap of probes across different normalizations. No probes overlapped across all five normalizations. The number of common probes for 4 out of 5 transformations was 8 for gbms, 30 for glmnet, and 31 for rf (see Supplementary Figure 3B). Next, using the probes selected by at least one data transformation for each method, we performed enrichment analysis of biological processes via Gene Ontology (GO) terms (see Supplementary Figure 4), Reactome pathways (see Supplementary Figure 5), KEGG pathways (see Figure 4 ), and disease ontology semantics (see Figure 4 ). Our models identified probes whose respective genes are involved in pathways highly related to multiple myeloma, such as the PI3K-Akt, MAPK, JAK-STAT, and Wnt signalling pathways, the RAF/MAP kinase cascade, NF-kB related pathways, and signalling by RAS and BRAF mutants. The disease ontology semantics highlighted several blood cancers, including multiple myeloma, across all methods. Download figure Open in new tab Figure 4. Enrichment analysis for the selected probes. Top: KEGG Pathways Associated with Identified Genes. This figure illustrates the KEGG pathways enriched for the genes identified by the machine learning models across different data transformations and training datasets. The pathways displayed are significantly associated with the probes selected by at least one model. Key pathways related to multiple myeloma, such as PI3K-Akt, MAPK, and Wnt signaling, are highlighted. Bottom: Disease-related terms associated with Identified Genes. The figure illustrates the distribution of disease-related terms associated with the genes identified by the models. The chart highlights how different methods and data transformations reveal connections to various cancers, including multiple myeloma. Each term represents a disease category. In both figures, the size and color indicate the strength of the association and statistical significance. Download figure Open in new tab Figure 5. Performance of machine learning algorithms on the GSE235356 dataset. The figure displays the distribution of the mean cross-validation AUC (auc_cvmean, shown in red) and the distribution of the AUC from the outer hold of the nested cross-validation (auc_test, shown in cyan) for each algorithm when the GSE235356 dataset was used for training and testing. The auc_cvmean represents the performance across the cross-validation folds, while the auc_test indicates the model’s generalizability on unseen data. The comparison of these distributions highlights the algorithm’s generalization and stability. 3.2 - Task 2. Predicting Progression from MGUS to MM 3.2.1 Model Development for Disease Progression Prediction We trained our models employing the following datasets: GSE235356, GSE6477, EMTAB317 alone and all datasets generated from the GLP96 and A-AFFY-33 platforms: GSE6477, GSE2113, EMTAB316 and GSE13591. For GSE235356 27 , we trained our models to distinguish between MGUS and progressing MGUS , which refers to MGUS cases that progressed to MM. We utilized a 10-fold cross-validation to optimize our model’s hyperparameters and employed a 10-fold nested cross-validation protocol to assess performance. In other scenarios, we trained our models to differentiate between MGUS and MM . This involved optimizing the hyperparameters using a ten-fold cross-validation repeated ten times. For consistency, we applied the same machine learning models and data transformations as in task 1, see Figure 1 . The scope of these two training approaches was twofold. First, we aimed to assess whether models trained to differentiate MGUS from MM could effectively distinguish MGUS from progressing MGUS, using the GSE235356 dataset for testing. Second, we sought to compare the performance of these models with those specifically trained to separate MGUS from progressing MGUS patients. This comparison would provide insights into whether models generalized well across related conditions or if specialized training was required for optimal performance in predicting MGUS progression. 3.2.2 Evaluation of Model Performance Using the GSE235356 dataset for training, we calculated the cross-validation Area Under the ROC Curve (AUC) during model optimization (auc_cv), the mean cross-validation AUC (auc_cvmean), and the outer cross-validation AUC from the nested cross-validation protocol (auc_test). Among the models, glmnet achieved the best performance, followed by gbm, rf, svmRadial, and svmLinear2 (refer to Figure 5 ). Specifically, glmnet with rma, qnorm, or ranking transformations showed the highest performance, with both auc_cvmean and auc_test around 0.8 (refer to Figure 5 ). All algorithms and data transformations also demonstrated good generalization performance in the outer cross-validation fold. The mean auc_test across all outer cross-validation folds fell within the AUC distribution achieved during training cross-validation (refer to Figure 6 ). Download figure Open in new tab Figure 6. Model performance in differentiating MGUS from progressing MGUS across different datasets. The boxplots show the distribution of the mean cross-validation AUC for models trained to differentiate MGUS from progressing MGUS using the GSE235356 dataset. The colored points represent the performance of each algorithm-data transformation combination across various training datasets: models trained with the EMTAB317 dataset are shown in red; those trained with the GSE235356 dataset are in green; models trained with the GSE6477 dataset are shown in cyan; and those trained with the combined GSE6477 + GSE2113 + EMTAB316 + GSE13591 datasets are depicted in purple. Notably, in all cases except for the second (GSE235356), the models were specifically trained to distinguish MGUS from MM. We trained our models to distinguish MGUS from MM using the GSE6477 dataset. These models achieved a training cross-validation AUC median ranging from 0.93 to 1 across all data transformations and machine-learning methods (refer to Supplementary Figure 6). Most models demonstrated good generalization performance when applied to other datasets (EMTAB316, EMTAB317, GSE13591, GSE2113) for identifying MGUS from MM (refer to Supplementary Figure 7). For EMTAB316 and GSE2113, the test AUC median was 0.9 and 0.86 across all data transformations and machine learning methods. For GSE13591, the test AUC median was 0.8 across all methods and transformations. The models achieved the lowest test AUC for the EMTAB317 dataset, with an AUC median of 0.7. This result is consistent with our findings in task 1 and likely due to the different microarray platforms used to generate the data. Download figure Open in new tab Figure 7. Disease-related terms associated with Identified Genes. The figure illustrates the distribution of disease-related terms associated with the genes identified by the models. The chart highlights how different methods across all data transformations and the different training datasets reveal connections to various cancers, including multiple myeloma. Each term represents a disease category. The size and color indicate the strength of the association and statistical significance. “all GLP96” refers to the combined dataset of GSE6477 + GSE2113 + EMTAB316 + GSE13591, and “GSE” to the GSE235356 dataset. When we applied our models to separate MGUS from progressing MGUS , the models differentiated the two classes. Specifically, the test AUC achieved by gbm, glmnet, rf, and svmLinear2 ranged from 0.7 to 0.8, falling within the AUC distribution achieved with cross-validation during training with the GSE235356 dataset (refer to Figure 6 ) and within the outer cross-validation auc_test distribution of the GSE235356 dataset (refer to Supplementary Figure 11). In the case of svmRadial, the auc_test ranged from 0.54 to 0.69; in all cases except rma normalization, it was below the training AUC cross-validation distribution but inside the outer cross-validation auc_test distribution of the GSE235356 dataset. Next, we trained our models to distinguish MGUS from MM , employing the EMTAB317 dataset. These models achieved a training cross-validation AUC median ranging from 0.79 to 0.94 across all data transformations and machine-learning methods (refer to Supplementary Figure 8). Most models demonstrated good generalization performance when applied to other datasets (EMTAB316, GSE13591, GSE2113, GSE6477) for identifying MGUS from MM (refer to Supplementary Figure 9). For EMTAB316 and GSE6477, the median test AUC was close to 0.75 and 0.82 across all data transformations and machine learning methods. For GSE13591 and GSE2113, the median test AUC was close to 0.86 and 0.91 across all methods and transformations. When we applied our models to separate MGUS from progressing MGUS , the models showed a test AUC performance ranging from 0.5 to 0.76, with a median of 0.65. The test AUC achieved by gbm, glmnet, rf, and svmRadial fell below the cross-validation AUC distribution achieved during training with the GSE235356 dataset (refer to Figure 6 ) but within the outer cross-validation auc_test distribution of the GSE235356 dataset (refer to Supplementary Figure 11). Interestingly, for svmLinear2, the test AUC fell within the AUC cross-validation distribution during training with the GSE235356 dataset for all data transformations. Last, we trained our models to separate MGUS from MM using all datasets generated from the GLP96 or A-AFFY-33 platforms ( GSE6477 + GSE2113 + EMTAB316 + GSE13591 ). Our models achieved a training cross-validation AUC median ranging from 0.94 to 0.97 across all data transformations and machine-learning methods (refer to Supplementary Figure 10). Similarly, we applied our models to separate MGUS from progressing MGUS . The models’ test AUC performance ranged from 0.55 for svmRadial using ranking to 0.82 for glmnet employing rma, with a median performance across all models and data transformations of 0.77. Importantly, the test AUC achieved by gbm, glmnet, and rf fell within the cross-validation AUC distribution achieved during training with the GSE235356 dataset (refer to Figure 6 ) and within the outer cross-validation auc_test distribution of the GSE235356 dataset (refer to Supplementary Figure 11). For svmRadial, the test AUCs fell below the cross-validation AUC distribution for all data transformations except binary_0.5, and at the lower end of the outer cross-validation auc_test distribution. Interestingly, for svmLinear2, the test AUC fell above the cross-validation AUC distribution for all data transformations except binary_0.5, and on the upper end of the outer cross-validation auc_test distribution. Additionally, with the inclusion of the GSE2113, EMTAB316, and GSE13591 datasets, glmnet and svmLinear2 showed a 0.05 increase in median test AUC across all data transformations compared to when only the GSE6477 was used for training; however, these differences were not statistically significant. Next, we conducted a permutation test to assess the statistical significance of the observed model performances in the test dataset in comparison to a random classification. In this analysis, we permuted the class labels ( MGUS, progressing MGUS ) and recalculated the auc_test for each model. Models with auc_test values close to 0.5, which correspond to a random classifier, did not demonstrate statistically significant different performance from random, as exprected. Conversely, models with auc_test values exceeding 0.7 showed highly significant results, clearly falling outside the permutation distribution (refer to Supplementary 12). 3.2.3 Models Interpretation We focused on interpreting the models trained using either the GSE235356 dataset or all GPL96 datasets combined. The svmLinear2 and svmRadial models utilized all available features. When all GPL96 datasets were used for training, the rf models employed between 2,169 and 10,269 features, glmnet selected between 236 and 859 probes, and gbm chose between 214 and 685 probes. In contrast, when the GSE235356 dataset was used for training, the rf models utilized between 3,090 and 11,650 features, glmnet selected between 10 and 792 probes, and gbm chose between 47 and 2,084 probes (see Supplementary Figure 13). We also assessed the overlap of probes selected across the two training datasets. For gbm and glmnet, only a small number of probes (ranging from 1 to 99) were selected in both cases. In contrast, the rf models showed a higher degree of overlap, with common probes ranging from 482 to 5,459 (refer to Supplementary Figure 14). Using the probes selected by at least one data transformation for each method and training dataset, we conducted enrichment analyses on Gene Ontology (GO) biological processes, KEGG pathways, Reactome pathways, and disease ontology semantics. The analysis revealed that our models identified probes associated with genes involved in pathways closely related to multiple myeloma, such as PI3K-Akt (see Supplementary Figure 17), MAPK (see Supplementary Figure 15, 16 and 17), Wnt signalling (see Supplementary Figure 16), BRAF and RAF1 fusion signalling (see Supplementary Figure 17), and mTOR pathways (see Supplementary Figure 16). The disease ontology analysis also underscored the relevance of several blood cancers, including multiple myeloma, across most methods and training datasets (see Figure 7 ). 4 Discussion In this study, we utilized advanced machine learning (ML) techniques to tackle two critical challenges in multiple myeloma (MM): predicting the disease stage and predicting disease progression from monoclonal gammopathy of undetermined significance (MGUS) to MM. Through comprehensive data preprocessing, model training, and evaluation across multiple datasets, we aimed to enhance diagnostic precision and offer valuable prognostic insights for hematologic malignancies. The first focus of our study was on predicting the stage of MM. Accurate staging is crucial for determining the appropriate treatment strategy and prognosis. We developed models using various ML algorithms, including ElasticNet, Random Forest, Boosting, and Support Vector Machines. These models were trained on a dataset comprising samples from different stages of MM and healthy samples, and their performance was evaluated on external validation datasets. The multiclass area under the curve values obtained during cross-validation and testing consistently demonstrated that the selected features and ML algorithms effectively capture the biological differences across disease stages. Among the models evaluated, gbm achieved the highest performance in training, and glmnet showed minimal degradation across different data transformations and datasets, indicating its robustness and generalizability. Our findings align with the growing body of literature that supports the use of ML in oncology, particularly in hematologic malignancies. Previous studies have shown the effectiveness of ML algorithms in improving diagnostic accuracy and risk stratification in MM 26 , 38 . The variability in model performance across different platforms, observed in datasets from GPL96, A-AFFY-34, GPL570, and A-AFFY-44, underscores the challenges of integrating data from diverse sources. This issue has been documented in the literature, where differences in data generation methods significantly affect model performance 39 , 40 . Predicting the progression of monoclonal gammopathy of undetermined significance (MGUS) to multiple myeloma (MM) remains one of the most pressing challenges in managing plasma cell disorders. Early identification of high-risk MGUS patients could significantly enhance clinical outcomes by enabling timely interventions that might delay or even prevent the onset of MM. A significant obstacle in this effort is the limited availability of datasets that include progressing MGUS patients, as these cases are inherently rare and difficult to procure. To address this challenge, we employed a two-pronged approaches. First, we developed machine-learning models using a dataset specifically containing MGUS and progressing MGUS patients, achieving a maximum AUC of 0.8 with the glmnet model combined with quantile normalization. Other models and data transformations demonstrated good generalization performance, with AUC values around 0.75. This result highlights the potential of machine learning in identifying high-risk MGUS patients even with limited data availability. Second, to evade the scarcity of progressing MGUS samples, we trained our models using multiple datasets containing both MGUS and MM patients. These models were then evaluated for their ability to distinguish MGUS from progressing MGUS cases. Our findings indicate that machine learning models, including ElasticNet, Boosting, SVM with linear kernel, and Random Forest, achieved AUC values close to 0.8, suggesting a strong potential for these models in risk stratification. Although some models, such as SVM with radial kernel, demonstrated lower performance, the overall results underscore the utility of incorporating both MGUS and MM data in predictive modeling. To our knowledge, this study is the first to develop comprehensive machine-learning models specifically designed to predict the progression of MGUS to MM by leveraging datasets from both MGUS and MM cases. Our innovative approach of integrating MM data to train models that predict MGUS progression offers a novel and potentially more accurate method for risk assessment. This methodology could have significant clinical implications, particularly in distinguishing MGUS patients who require closer monitoring from those who may not. The novelty and potential impact of our approach are further emphasized by recent reviews in the field, such as the one by Awada et al. 41 , which highlight the need for more sophisticated predictive models that integrate data across disease stages to enhance prognostication and treatment planning. The feature selection and enrichment analyses conducted in this study provided significant insights into the molecular pathways and biological processes involved in the progression of multiple myeloma. Our models consistently identified genes involved in critical signaling pathways, such as PI3K-Akt, MAPK, Wnt, and mTOR. These pathways are well-known for their roles in cell growth, survival, and proliferation, and their involvement in MM pathogenesis is well-documented 42 . For instance, the PI3K-Akt pathway has been widely recognized as a key player in MM, influencing proliferation, migration, apoptosis, and autophagy 43 . Similarly, the MAPK pathway is involved in the regulation of cell proliferation, survival, and differentiation, and its dysregulation has been implicated in various cancers, including MM 44 , 45 . The Wnt pathway, which is crucial for cell differentiation and proliferation, has also been associated with MM progression, particularly in the context of bone disease 46 . The consistency of our results with established biological knowledge validates our models and suggests potential therapeutic targets that could be explored in future research. While the results of this study are promising, several limitations should be considered when interpreting our findings. One significant challenge is the variability in model performance across different microarray platforms. This variability suggests a need for more comprehensive cross-platform validation to ensure the robustness of our models in different clinical settings. Moreover, the relatively small number of datasets used in this study and the focus on a limited set of machine learning algorithms may have constrained our ability to explore other potentially valuable approaches. Future research should aim to include a more extensive variety of datasets, especially those generated from different omics technologies, to enhance the generalizability and robustness of the models. Also, to improve model robustness, an ensemble classification approach could be explore. By combining multiple machine learning algorithms, ensemble methods can reduce performance variability across platforms and enhance prediction accuracy, offering more reliable generalizability for clinical use. Additionally, integrating clinical data, such as patient demographics and treatment history, could provide a more comprehensive understanding of disease progression and improve the clinical applicability of the models. These challenges are well-recognized in the literature 47 – 49 . All studies emphasize the need for cross-platform validation and standardization in ML models, particularly in the context of precision medicine, where the ability to generalize across different datasets is crucial for clinical implementation. 5 Funding Funded by the European Union. Views and opinions expressed are however those of the author(s) only and do not necessarily reflect those of the European Union or ELMUMY (Project: 101097094). Neither the European Union nor the granting authority can be held responsible for them. 6 References 1. ↵ Rajkumar , S. V. et al. International Myeloma Working Group updated criteria for the diagnosis of multiple myeloma . Lancet Oncol . 15 , e538 – e548 ( 2014 ). OpenUrl CrossRef PubMed 2. ↵ Rajkumar , S. V. & Kumar , S. Multiple myeloma current treatment algorithms . Blood Cancer J . 10 , 94 ( 2020 ). OpenUrl CrossRef PubMed 3. ↵ Siegel , R. L. , Miller , K. D. , Fuchs , H. E. & Jemal , A. Cancer Statistics, 2021 . CA. Cancer J. Clin . 71 , 7 – 33 ( 2021 ). OpenUrl CrossRef PubMed 4. ↵ Kyle , R. A. et al. Incidence of multiple myeloma in Olmsted County, Minnesota . Cancer 101 , 2667 – 2674 ( 2004 ). OpenUrl CrossRef PubMed Web of Science 5. ↵ Landgren , O. & Weiss , B. M. Patterns of monoclonal gammopathy of undetermined significance and multiple myeloma in various ethnic/racial groups: support for genetic factors in pathogenesis . Leukemia 23 , 1691 – 1697 ( 2009 ). OpenUrl CrossRef PubMed Web of Science 6. ↵ Kyle , R. A. et al. Review of 1027 Patients With Newly Diagnosed Multiple Myeloma . Mayo Clin. Proc . 78 , 21 – 33 ( 2003 ). OpenUrl CrossRef PubMed Web of Science 7. ↵ Landgren , O. et al. Monoclonal gammopathy of undetermined significance (MGUS) consistently precedes multiple myeloma: A prospective study . Blood 113 , 5412 – 5417 ( 2009 ). OpenUrl Abstract / FREE Full Text 8. ↵ Weiss , B. M. , Abadie , J. , Verma , P. , Howard , R. S. & Kuehl , W. M. A monoclonal gammopathy precedes multiple myeloma in most patients . Blood 113 , 5418 – 5422 ( 2009 ). OpenUrl Abstract / FREE Full Text 9. ↵ Kyle , R. A. et al. Prevalence of Monoclonal Gammopathy of Undetermined Significance . N. Engl. J. Med . 354 , 1362 – 1369 ( 2006 ). OpenUrl CrossRef PubMed Web of Science 10. Dispenzieri , A. et al. Prevalence and risk of progression of light-chain monoclonal gammopathy of undetermined significance: a retrospective population-based cohort study . Lancet 375 , 1721 – 1728 ( 2010 ). OpenUrl CrossRef PubMed Web of Science 11. Murray , D. et al. Detection and prevalence of monoclonal gammopathy of undetermined significance: a study utilizing mass spectrometry-based monoclonal immunoglobulin rapid accurate mass measurement . Blood Cancer J . 9 , ( 2019 ). 12. ↵ Landgren , O. et al. Prevalence of myeloma precursor state monoclonal gammopathy of undetermined significance in 12 372 individuals 10–49 years old: a population-based study from the National Health and Nutrition Examination Survey . Blood Cancer J . 7 , ( 2017 ). 13. ↵ Kyle , R. A. et al. A Long-Term Study of Prognosis in Monoclonal Gammopathy of Undetermined Significance . N. Engl. J. Med . 346 , 564 – 569 ( 2002 ). OpenUrl CrossRef PubMed Web of Science 14. ↵ Kyle , R. A. et al. Long-Term Follow-up of Monoclonal Gammopathy of Undetermined Significance . N. Engl. J. Med . 378 , 241 – 249 ( 2018 ). OpenUrl CrossRef PubMed 15. ↵ Therneau , T. M. et al. Incidence of monoclonal gammopathy of undetermined significance and estimation of duration before first clinical recognition . Mayo Clin. Proc . 87 , 1071 – 1079 ( 2012 ). OpenUrl CrossRef PubMed 16. ↵ Kyle , R. A. et al. Clinical Course and Prognosis of Smoldering (Asymptomatic) Multiple Myeloma . N. Engl. J. Med . 356 , 2582 – 2590 ( 2007 ). OpenUrl CrossRef PubMed Web of Science 17. ↵ Rajkumar , S. V et al. Impact of primary molecular cytogenetic abnormalities and risk of progression in smoldering multiple myeloma . Leukemia 27 , 1738 – 1744 ( 2013 ). OpenUrl CrossRef PubMed Web of Science 18. Neben , K. et al. Progression in smoldering myeloma is independently determined by the chromosomal abnormalities del(17p), t(4;14), gain 1q, hyperdiploidy, and tumor load . J. Clin. Oncol . 31 , 4325 – 4332 ( 2013 ). OpenUrl Abstract / FREE Full Text 19. ↵ Rajkumar , S. V. Multiple myeloma: 2022 update on diagnosis, risk stratification, and management . Am. J. Hematol . 97 , 1086 – 1107 ( 2022 ). OpenUrl CrossRef PubMed 20. ↵ Ho , M. et al. Changing paradigms in diagnosis and treatment of monoclonal gammopathy of undetermined significance (MGUS) and smoldering multiple myeloma (SMM) . Leukemia 34 , 3111 – 3125 ( 2020 ). OpenUrl CrossRef PubMed 21. ↵ Mateos , M.-V. et al. Lenalidomide plus Dexamethasone for High-Risk Smoldering Multiple Myeloma . N. Engl. J. Med . 369 , 438 – 447 ( 2013 ). OpenUrl CrossRef PubMed Web of Science 22. ↵ Lonial , S. et al. Randomized Trial of Lenalidomide Versus Observation in Smoldering Multiple Myeloma . J. Clin. Oncol . 38 , 1126 – 1137 ( 2020 ). OpenUrl CrossRef PubMed 23. ↵ Rajkumar , S. V. et al. Serum free light chain ratio is an independent risk factor for progression in monoclonal gammopathy of undetermined significance . Blood 106 , 812 – 817 ( 2005 ). OpenUrl Abstract / FREE Full Text 24. ↵ Pérez-Persona , E. et al. New criteria to identify risk of progression in monoclonal gammopathy of uncertain significance and smoldering multiple myeloma based on multiparameter flow cytometry analysis of bone marrow plasma cells . Blood 110 , 2586 – 2592 ( 2007 ). OpenUrl Abstract / FREE Full Text 25. ↵ Dispenzieri , A. et al. Treatment of Newly Diagnosed Multiple Myeloma Based on Mayo Stratification of Myeloma and Risk-Adapted Therapy (mSMART): Consensus Statement . Mayo Clin. Proc . 82 , 323 – 341 ( 2007 ). OpenUrl CrossRef PubMed Web of Science 26. ↵ Allegra , A. et al. Machine Learning and Deep Learning Applications in Multiple Myeloma Diagnosis, Prognosis, and Treatment Selection . Cancers (Basel) . 14 , 1 – 16 ( 2022 ). OpenUrl 27. ↵ Sun , F. et al. A gene signature can predict risk of MGUS progressing to multiple myeloma . J. Hematol. Oncol . 16 , 1 – 5 ( 2023 ). OpenUrl CrossRef PubMed 28. ↵ Friedman , J. , Hastie , T. & Tibshirani , R. Regularization Paths for Generalized Linear Models Via Coordiante Descent . J. Stat. Softw . 33 , ( 2010 ). 29. ↵ Hastie , T. , Tibshirani , R. , James , G. & Witten , D. An Introduction to Statistical Learning, with Applications in R. Springer Texts vol. 102 ( 2021 ). 30. ↵ Natekin , A. & Knoll , A. Gradient boosting machines, a tutorial . Front. Neurorobot . 7 , ( 2013 ). 31. ↵ Meyer , D. Support vector machines: the interface to libsvm in package e1071 . … Syst. their … 1 , 1 – 8 ( 2014 ). OpenUrl 32. ↵ Karatzoglou , A. , Smola , A. , Hornik , K. & Zeileis , A. kernlab - An S4 Package for Kernel Methods in R . J. Stat. Softw . 11 , 389 – 393 ( 2004 ). OpenUrl 33. ↵ Kuhn , M. Building predictive models in R using the caret package . J. Stat. Softw . 28 , 1 – 26 ( 2008 ). OpenUrl CrossRef PubMed 34. ↵ Yu , G. , Wang , L. G. , Yan , G. R. & He , Q. Y. DOSE: An R/Bioconductor package for disease ontology semantic and enrichment analysis . Bioinformatics 31 , 608 – 609 ( 2015 ). OpenUrl CrossRef PubMed Web of Science 35. ↵ Yang , P. et al. Pathogenesis and treatment of multiple myeloma . MedComm 3 , 1 – 27 ( 2022 ). OpenUrl PubMed 36. ↵ John , L. , Krauth , M. T. , Podar , K. & Raab , M. S. Pathway-directed therapy in multiple myeloma . Cancers (Basel) . 13 , 1 – 19 ( 2021 ). OpenUrl 37. ↵ Chng , W. J. et al. Molecular dissection of hyperdiploid multiple myeloma by gene expression profiling . Cancer Res . 67 , 2982 – 2989 ( 2007 ). OpenUrl Abstract / FREE Full Text 38. ↵ Zhong , H. et al. 18F-FDG PET/CT based radiomics features improve prediction of prognosis: multiple machine learning algorithms and multimodality applications for multiple myeloma . BMC Med. Imaging 23 , 1 – 12 ( 2023 ). OpenUrl CrossRef PubMed 39. ↵ Franks , J. M. , Cai , G. & Whitfield , M. L. Feature specific quantile normalization enables cross-platform classification of molecular subtypes using gene expression data . Bioinformatics 34 , 1868 – 1874 ( 2018 ). OpenUrl CrossRef PubMed 40. ↵ Foltz , S. M. , Greene , C. S. & Taroni , J. N. Cross-platform normalization enables machine learning model training on microarray and RNA-seq data simultaneously . Commun. Biol . 6 , ( 2023 ). 41. ↵ Awada , H. et al. A Comprehensive Review of the Genomics of Multiple Myeloma: Evolutionary Trajectories, Gene Expression Profiling, and Emerging Therapeutics . Cells 10 , 1961 ( 2021 ). OpenUrl CrossRef 42. ↵ Lu , Q. , Yang , D. , Li , H. , Niu , T. & Tong , A. Multiple myeloma: signaling pathways and targeted therapy. Molecular Biomedicine vol. 5 ( Springer Nature Singapore , 2024 ). 43. ↵ Isa , R. et al. The Rationale for the Dual-Targeting Therapy for RSK2 and AKT in Multiple Myeloma . Int. J. Mol. Sci . 23 , ( 2022 ). 44. ↵ Bahar , M. E. , Kim , H. J. & Kim , D. R. Targeting the RAS/RAF/MAPK pathway for cancer therapy: from mechanism to clinical studies . Signal Transduct. Target. Ther . 8 , ( 2023 ). 45. ↵ Song , Y. et al. Targeting RAS–RAF–MEK–ERK signaling pathway in human cancer: Current status in clinical trials . Genes Dis . 10 , 76 – 88 ( 2023 ). OpenUrl CrossRef PubMed 46. ↵ Spaan , I. , Raymakers , R. A. , van de Stolpe , A. & Peperzak , V. Wnt signaling in multiple myeloma: a central player in disease with therapeutic potential . J. Hematol. Oncol . 11 , 67 ( 2018 ). OpenUrl CrossRef PubMed 47. ↵ Wang , T. et al. Molecular precision medicine: Multi-omics-based stratification model for acute myeloid leukemia . Heliyon 10 , e36155 ( 2024 ). OpenUrl CrossRef 48. Correa-Aguila , R. , Alonso-Pupo , N. & Hernández-Rodríguez , E. W. Multi-omics data integration approaches for precision oncology . Mol. Omi . 18 , 469 – 479 ( 2022 ). OpenUrl CrossRef 49. ↵ Li , Y. , Wu , X. , Fang , D. & Luo , Y. Informing immunotherapy with multi-omics driven machine learning . npj Digit. Med . 7 , ( 2024 ). View the discussion thread. Back to top Previous Next Posted November 15, 2024. Download PDF Supplementary Material Email Thank you for your interest in spreading the word about bioRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following Machine Learning Models for Predicting Multiple Myeloma Staging and MGUS Progression Using Gene Expression Data Message Subject (Your Name) has forwarded a page to you from bioRxiv Message Body (Your Name) thought you would like to see this page from the bioRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share Machine Learning Models for Predicting Multiple Myeloma Staging and MGUS Progression Using Gene Expression Data Nestoras Karathanasis , George M. Spyrou bioRxiv 2024.11.12.623149; doi: https://doi.org/10.1101/2024.11.12.623149 Share This Article: Copy Citation Tools Machine Learning Models for Predicting Multiple Myeloma Staging and MGUS Progression Using Gene Expression Data Nestoras Karathanasis , George M. Spyrou bioRxiv 2024.11.12.623149; doi: https://doi.org/10.1101/2024.11.12.623149 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Bioinformatics Subject Areas All Articles Animal Behavior and Cognition (7643) Biochemistry (17717) Bioengineering (13910) Bioinformatics (42018) Biophysics (21480) Cancer Biology (18629) Cell Biology (25537) Clinical Trials (138) Developmental Biology (13392) Ecology (19935) Epidemiology (2067) Evolutionary Biology (24356) Genetics (15617) Genomics (22531) Immunology (17755) Microbiology (40438) Molecular Biology (17200) Neuroscience (88706) Paleontology (667) Pathology (2840) Pharmacology and Toxicology (4832) Physiology (7657) Plant Biology (15171) Scientific Communication and Education (2046) Synthetic Biology (4304) Systems Biology (9828) Zoology (2272)

Text is read by the "Ask this paper" AI Q&A widget below. Extraction quality varies by source — PMC NXML preserves structure cleanly, OA-HTML may include some navigation residue, and OA-PDF can have broken hyphenation. The publisher copy (via DOI) is the canonical version.

My notes (saved in your browser only)

Ask this paper AI returns verbatim quotes from the full text · source: preprint-html

Answers must be backed by verbatim quotes from this paper's full text. Hallucinated quotes are dropped automatically; if no verbatim passage answers the question, we say so. How this works

Citation neighborhood (no data yet)

We don't have any in-corpus citations linked to this paper yet. This is a recent paper (2024) — citers typically take a year or two to land, and the OpenAlex reference graph may still be filling in.

Source provenance

europepmc
last seen: 2026-05-20T01:45:00.602351+00:00