multiDEGGs: a multi-omic differential network analysis package for biomarker discovery and predictive modeling

preprint OA: gold CC-BY-NC-ND-4.0
📄 Open PDF Full text JSON View at publisher

Abstract

ABSTRACT Modern clinical trials increasingly leverage high-throughput omic data for patient stratification and biomarker discovery. While traditional differential gene expression analysis disregards the networked nature of molecular entities and produces extensive gene lists with limited interpretability, differential network analysis has emerged as a crucial complementary analysis for comparative studies. Here we present multiDEGGs, a CRAN R package that enables differential network analysis in multi-omic scenarios. multiDEGGs uses a multi-layer graph framework to model omic data by leveraging an internal network of over 10 000 literature-validated biological interactions. For each data type, differential networks are generated, and the statistical significance of each link (p-values or adjusted p-values) is evaluated through robust linear regression with interaction terms. These networks are then integrated into a comprehensive visualisation that allows interactive exploration of cross-omic patterns. Beyond network visualization and exploration, multiDEGGs extends its utility to predictive modelling applications. The package facilitates seamless integration into cross-validation machine learning pipelines, serving as feature selection and augmentation tool. We validated multiDEGGs using two cohorts of rheumatoid arthritis patients who underwent tocilizumab and rituximab therapy, respectively. For each treatment group, multi-layer differential interactions were identified, and seven machine learning models were trained to predict treatment resistance using synovial RNA-seq data. We systematically compared multiDEGGs against five traditional feature selection methods. On average, AUC values obtained with multiDEGGs showed an improvement of 0.10 compared to conventional filters. KEY POINTS Traditional gene expression analysis leaves researchers with hundreds of ‘significant’ genes but no clear biological story. The multiDEGGs CRAN package shifts the focus: instead of asking which genes change, it asks which gene relationships change. It can be used with single or multi-omic data: differential networks are calculated separately for each data type, with results integrated into a comprehensive, interactive view. multiDEGGs can be combined with the nestedcv CRAN package (nested cross-validation) to serve as feature selection and augmentation tool. In comparative evaluations, machine learning models trained with multiDEGGs-selected features showed AUC improvements of 0.10 compared to other feature selection methods.
Full text 50,284 characters · extracted from preprint-html · click to expand
multiDEGGs: a multi-omic differential network analysis package for biomarker discovery and predictive modeling | bioRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-M677548'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search New Results multiDEGGs: a multi-omic differential network analysis package for biomarker discovery and predictive modeling View ORCID Profile Elisabetta Sciacca , View ORCID Profile Susan Wang , View ORCID Profile Costantino Pitzalis , View ORCID Profile Myles J Lewis doi: https://doi.org/10.1101/2025.09.25.678475 Elisabetta Sciacca 1 Department of Biomedical Sciences, Humanitas University , Via Rita Levi Montalcini 4, 20090 Pieve Emanuele, Milan, Italy 2 Centre for Experimental Medicine & Rheumatology, William Harvey Research Institute, Barts and The London School of Medicine and Dentistry, Queen Mary University of London , London, UK Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Elisabetta Sciacca For correspondence: elisabetta.sciacca{at}humanitasresearch.it myles.lewis{at}qmul.ac.uk c.pitzalis{at}qmul.ac.uk Susan Wang 2 Centre for Experimental Medicine & Rheumatology, William Harvey Research Institute, Barts and The London School of Medicine and Dentistry, Queen Mary University of London , London, UK 3 Barts Health NHS Trust and National Institute for Health and Care Research (NIHR) Barts Biomedical Research Centre (BRC) , London, UK Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Susan Wang Costantino Pitzalis 2 Centre for Experimental Medicine & Rheumatology, William Harvey Research Institute, Barts and The London School of Medicine and Dentistry, Queen Mary University of London , London, UK 3 Barts Health NHS Trust and National Institute for Health and Care Research (NIHR) Barts Biomedical Research Centre (BRC) , London, UK 4 IRCCS Istituto Clinico Humanitas Rozzano (MI) , Italy Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Costantino Pitzalis For correspondence: elisabetta.sciacca{at}humanitasresearch.it myles.lewis{at}qmul.ac.uk c.pitzalis{at}qmul.ac.uk Myles J Lewis 2 Centre for Experimental Medicine & Rheumatology, William Harvey Research Institute, Barts and The London School of Medicine and Dentistry, Queen Mary University of London , London, UK 3 Barts Health NHS Trust and National Institute for Health and Care Research (NIHR) Barts Biomedical Research Centre (BRC) , London, UK Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Myles J Lewis For correspondence: elisabetta.sciacca{at}humanitasresearch.it myles.lewis{at}qmul.ac.uk c.pitzalis{at}qmul.ac.uk Abstract Full Text Info/History Metrics Supplementary material Data/Code Preview PDF ABSTRACT Modern clinical trials increasingly leverage high-throughput omic data for patient stratification and biomarker discovery. While traditional differential gene expression analysis disregards the networked nature of molecular entities and produces extensive gene lists with limited interpretability, differential network analysis has emerged as a crucial complementary analysis for comparative studies. Here we present multiDEGGs, a CRAN R package that enables differential network analysis in multi-omic scenarios. multiDEGGs uses a multi-layer graph framework to model omic data by leveraging an internal network of over 10 000 literature-validated biological interactions. For each data type, differential networks are generated, and the statistical significance of each link (p-values or adjusted p-values) is evaluated through robust linear regression with interaction terms. These networks are then integrated into a comprehensive visualisation that allows interactive exploration of cross-omic patterns. Beyond network visualization and exploration, multiDEGGs extends its utility to predictive modelling applications. The package facilitates seamless integration into cross-validation machine learning pipelines, serving as feature selection and augmentation tool. We validated multiDEGGs using two cohorts of rheumatoid arthritis patients who underwent tocilizumab and rituximab therapy, respectively. For each treatment group, multi-layer differential interactions were identified, and seven machine learning models were trained to predict treatment resistance using synovial RNA-seq data. We systematically compared multiDEGGs against five traditional feature selection methods. On average, AUC values obtained with multiDEGGs showed an improvement of 0.10 compared to conventional filters. KEY POINTS Traditional gene expression analysis leaves researchers with hundreds of ‘significant’ genes but no clear biological story. The multiDEGGs CRAN package shifts the focus: instead of asking which genes change, it asks which gene relationships change. It can be used with single or multi-omic data: differential networks are calculated separately for each data type, with results integrated into a comprehensive, interactive view. multiDEGGs can be combined with the nestedcv CRAN package (nested cross-validation) to serve as feature selection and augmentation tool. In comparative evaluations, machine learning models trained with multiDEGGs-selected features showed AUC improvements of 0.10 compared to other feature selection methods. INTRODUCTION Modern clinical trials increasingly leverage high-throughput omics data to characterize distinct patient phenotypes and elucidate differential biological mechanisms across various conditions. In this context, differential gene expression analysis (DGE) represents the most basic analytical approach, despite its limitations in efficacy and interpretability( Nance et al. 2022 , Cui et al. 2021 , Stretch et al. 2013 , Ein-Dor and Domany 2006 ). Differential network analysis has emerged as a crucial complement to single-gene differential methods by identifying alterations in biomolecular relationships, thereby providing deeper insights into the molecular mechanisms underlying specific phenotypes or disease states. The past decade has witnessed significant advancements in the development of R packages dedicated to differential network analysis of high-throughput omics data. These tools can be broadly categorised based on their statical and mathematical strategies. One group of approaches primarily focuses on co-expression patterns derived directly from the omics data and identifying differential co-expressed modules. Methods in this category typically employ correlation measures, such as Pearson’s correlation coefficient, to construct networks and subsequently compare them between conditions. For example, the DiffCorr( Fukushima 2013 ) package calculates correlation matrices and employs Fisher’s z-test to identify differential correlations, while DiffCoEx( Tesson, Breitling and Jansen 2010 ) leverages the WGCNA framework to detect differentially co-expressed gene modules. WGCNA is also used – among other options – in the DCGL package( Liu et al. 2010 ), and the INDEED package( Li et al. 2018 ) allows users to choose between partial, Pearson, and Spearman correlation. In contrast, another class of tools integrates prior knowledge of biological interactions to guide the differential network analysis. NetOmics( Bodein et al. 2022 ), DIABLO( Singh et al. 2019 ) and DEGGs( Sciacca, Alaimo, Silluzio, Ferro, Latora, Pitzalis, Pulvirenti and Myles J. Lewis 2023 ) are examples of this approach. This strategy addresses the computational challenges of analysing all possible gene-gene combinations in NGS data while enhancing interpretability by focusing exclusively on functionally relevant interactions. Of those tools, DEGGs is the only one providing a user-friendly interactive interface, and differential statistical significance for each gene-gene link via robust linear regression with interaction term. Here we present multiDEGGs (multi-omics Differentially Expressed Gene-Gene pairs)( Sciacca and Lewis 2025 ), an evolution of the DEGGs package that extends differential network analysis to multiomic scenarios using multi-layer differential networks (Video 1). We applied multiDEGGs to two rheumatoid arthritis case studies to identify multi-layer differential networks that revealed key biological mechanisms distinguishing therapy responders from non-responders, offering valuable insights for clinical investigation. In the second part of the paper, we also demonstrate how the resulting differential interactions can be effectively leveraged for feature selection and augmentation in machine learning applications. To this aim, the original code has undergone substantial speed and efficiency improvements from its previous version and has been expanded with new features. Finally, to enable seamless integration in nested cross-validation processes, we also extended the nestedCV package( Lewis et al. 2023 , Myles Lewis et al. 2025 ) with complementary functionalities. The features selected by multiDEGGs led to better model performances when compared with other well-known filters (e.g. univariate filters, glmnet filter, ect.). MATERIALS AND METHODS Differential multi-omic analyses through multi-layer networks The multiDEGGs package extracts differential multilayer networks following the workflow depicted in Figure 1 . Download figure Open in new tab Figure 1 Workflow diagram showing the multiDEGGs internal implementation for multi-omic differential network analysis. In particular, let be a reference biological network with vertex set v representing molecular entities and edge set E representing known molecular interactions from literature, where | E | ≈10 4 . From this reference network, multiDEGGs extracts group-specific differential multilayer networks for each experimental group or condition k where each layer represents a specific omic data type (e.g. RNA-seq, proteomics, Olink, etc.) filtered for group k . The key distinction between groups arises from the group-specific activation patterns of molecular entities. During step 1, for each group k and layer i , multiDEGGs defines a node activation function as a step function: Here, represents the normalised median expression or abundance of the biological entity v in group within the omic layer i , and τ ( i ) is the activation threshold selected by solving: where is the set of significant edges at threshold τ, as per percolation procedure previously detailed in ( Sciacca, Alaimo, Silluzio, Ferro, Latora, Pitzalis, Pulvirenti and Myles J Lewis 2023 ), where the first version of the package was introduced. The group-specific active vertex set for layer i is then defined as Since varies between groups, during the first step different nodes will be activated for each group k , creating group-specific network topologies. To retain only the group-specific edges, in step 2 multiDEGGs identifies and removes edges that are common to all k groups within each omic layer. For layer i , the group-specific edge set is then defined as where represents the edges connecting active nodes in group k and layer i . In step 3, an edge activation function is applied to each remaining group-specific edge in layer i . multiDEGGs employs robust linear regression with interaction term to assess whether the differential connectivity between groups is statistically significant. For each potential edge ( u, v ) connecting active nodes within the same layer i and group k , let u, v ∈ R n be the corresponding normalised expression or abundance vectors of the two biological entities across all samples, and G ∈ 1,2, …, K n be the categorical group vector indicating group membership for each sample. multiDEGGs formulates the regression model as: where 1 n is the vector of ones of dimension n , ⊙ denotes the element-wise (Hadamard) product, β 3 captures the differential interaction effect between the two biological entities in group k relative to other groups, and ε ∈ R n represents the error term vector. The differential edge activation function is then defined as where p ( β 3 ) is the adjusted p value for the interaction term β 3 , and α adj is the multiple testing corrected significance threshold defined by the user. The differential edge set for layer i is then defined as: Therefore, the complete group-specific differential multilayer network is represented as: where is the differential graph for group k and layer i . It is worth noting that, in this final representation, each molecular entity is active in only a subset of layers. Being able to visually identify those entities active in multiple omic layers – forming bridging connections – is important to shed light on possible cross-layer signal propagation associated to certain conditions. For visualization purposes, the final multi-layer differential networks are integrated into a comprehensive display where each omic layer i is assigned a distinct colour, enabling users to easily gain an overview of all layers and explore cross-omic patterns. The interactive visualization is implemented through the View_diffNetworks() function using R Shiny and includes several features: a slider for filtering network links by p-value (or adjusted p-value), a search box to locate specific genes or molecular entities, clickable links that display the associated differential regression plot when selected, and clickable nodes that generate boxplots showing expression differences between experimental groups (Video 1). Overall, this framework ensures group-specific differential network extraction from a common reference network, layer-specific activation through percolation-optimised thresholds, removal of common connectivity to highlight group-specific patterns, differential edge weighting based on the continuous significance values of interaction effects within each layer, and robustness through multiple testing correction. RESULTS Multi-omic differential network analysis in rheumatoid arthritis We applied multiDEGGs to two rheumatoid arthritis multi-omic datasets. Rheumatoid arthritis (RA) is a chronic autoimmune disease characterized by synovial inflammation and joint destruction. Despite the availability of diverse biologic disease modifying anti-rheumatic drugs targeting distinct immune pathways, 30–40% of RA patients exhibit inadequate response ( Taylor et al. 2022 ), and approximately 10% develop multi-refractory disease, failing more than two biologic classes in longitudinal studies ( Novella-Navarro et al. 2020 ). To better understand the biological signatures underlying treatment response, multiDEGGs has been applied to two cohorts of rheumatoid arthritis patients who underwent tocilizumab and rituximab therapy, respectively ( Rivellese et al. 2022 , Humby et al. 2021 , Myles J. Lewis et al. 2025 , Rivellese et al. 2023 ). In both studies, synovial biopsies were collected before treatment initiation, and therapeutic response was assessed following treatment completion. The resulting multi-layer differential networks highlighted important mechanisms distinguishing patients who respond to therapy from those who do not, providing crucial insights for clinical investigation. Multi-omic differential network analysis in RA patients not responding to tocilizumab therapy The RA cohort of patients treated with tocilizumab consisted of 65 individuals with available synovial RNAseq (n=65), O-link data (n=64), and lower numbers of mass spectrometry proteomic (n=17) and phosphoproteomic (n=17) data. The main cluster of the resulting multi-omic differential network centres on PIK3R1 , which shows extensive synovial RNA-seq co-expression (green arrows) with key cytokine-driven inflammatory pathway signalling molecules including JAK1, JAK3, TYK2, IL2RG, CSF2RB , and AKT2 ( Figure 2 ). This indicates possible coordinated activation of the PI3K–AKT and JAK/STAT pathways, both of which are downstream of cytokine receptors, including the IL-6 receptor (IL-6R) ( Malemud 2018 ). The co-expression of ADCY4 and PRKACA (Padj = 0.0074) in responders suggests coordinated activation of the cAMP–PKA pathway. ADCY4 encodes adenylate cyclase 4, which generates cAMP as a second messenger, while PRKACA encodes the catalytic subunit of protein kinase A, the primary effector of cAMP signalling. Together, their co-expression implies enhanced cAMP-driven phosphorylation of downstream targets, which is known to increase secretion of cytokines including IL-6 ( Hershko et al. 2002 ). Phosphoproteomic links (orange arrows) involving ITGB2 and ICAM3 indicate active integrin-mediated adhesion and immune cell retention, which may reinforce cytokine-driven inflammation. Download figure Open in new tab Figure 2 Network overview and magnified view of a cluster of multi-omic differential interactions between tocilizumab responders and non-responders in rheumatoid arthritis patients. The network displays differential interactions with edge colours representing different omics data types (green edges for synovial RNA-seq, orange edges for phosphoproteomics data). All interactions shown have adjusted p-values < 0.05. Edge thickness is proportional to statistical significance, with thicker edges representing lower adjusted p-values. Arrow directions reflect literature-reported regulatory relationships and are not derived from data. From a clinical perspective, the presence of this integrated network prior to therapy suggests that the pathology in responders is highly dependent on cytokine, including IL-6, mediated PI3K–AKT and JAK/STAT signalling for synovial inflammation. Tocilizumab, by blocking IL-6R signalling, likely disrupts this central network, leading to a more profound therapeutic effect in these patients. Conversely, the absence of this transcriptional and phosphoproteomic configuration in non-responders suggests that their synovial inflammation is driven by alternative, IL-6-independent pathways. Dysregulated multi-omic interactions in RA patients treated with rituximab The RA cohort of patients receiving rituximab treatment comprised 72 patients with accessible synovial RNAseq and O-link data (n=72) along with fewer mass spectrometry-based proteomic and phosphoproteomic (n=21) profiles. The resulting multi-omic differential network highlights ITGB3 and ITGA3 as central genes ( Figure 3 ). They encode integrin subunits (β3 and α3) that mediate cell-extracelluar matrix (ECM) adhesion and signalling, processes central to synovial fibroblast activation and immune cell trafficking in RA ( Lowin and Straub 2011 ). Download figure Open in new tab Figure 3 Network overview with magnified view of a cluster of multi-omic differential interactions between rituximab responders and non-responders in rheumatoid arthritis patients. The network displays differential interactions with edge colours representing different omics data types (green edges for synovial RNA-seq, orange edges for phosphoproteomics data). All interactions shown have adjusted p-values < 0.05. Edge thickness is proportional to statistical significance, with thicker edges representing lower adjusted p-values. Arrow directions reflect literature-reported regulatory relationships and are not derived from data. Their presence suggests integrin-mediated (ECM) interactions, which represent a key distinguishing factor for rituximab responders in RA. ITGB3 expression is strongly associated with multiple extracellular matrix component genes ( LAMA4, LAMB1, LAMB2, COL1A1, COL6A2, THBS3, SPP1 ) in responders, consistent with its role in modulating fibroblast adhesion, migration, and tissue remodelling in the synovium. Notably, the association of ITGB3 with SRC and PLCG1 through phosphoproteomics data suggests downstream activation of focal adhesion and MAPK signalling pathways, which regulate pro-inflammatory and survival signals. Through SRC, this phosphoproteomic network extends to ITGB2, a leukocyte-specific integrin central to immune cell adhesion and migration ( Lowin and Straub 2011 ). The ITGB3–SRC–ITGB2 axis may facilitate immune cell, particularly B cell, recruitment into the inflamed synovium, aligning with the higher B cell burden typically observed in rituximab responders ( Rivellese et al. 2022 ). The upper right plot demonstrates a significant positive correlation between ITGB3 and CRKL (an adaptor protein involved in integrin and growth factor signalling) expression in good/moderate responders but absent in non-responders. The strong coupling of ITGB3–CRKL expression in responders indicates that integrin-mediated signalling dynamics, potentially linked to cellular adhesion and immune cell trafficking, are more transcriptionally coordinated in patients who benefit from rituximab. The ITGA3 network also shows strong co-expression with ECM-related genes, including COL1A1, COL1A2, COL6A1, COL6A2, COL6A3, LAMB1, THBS2 , and THBS3 , indicating active matrix remodelling and cell–matrix adhesion in rituximab responders. This pattern suggests that fibroblast-like synoviocytes are engaged in strong integrin–ECM interactions, promoting a structural microenvironment conducive to immune cell infiltration and retention. Collectively, these networks may indicate that responders possess a synovial microenvironment preconditioned for immune cell (including B cell) migration, retention, and survival via integrin– ECM interactions. As rituximab targets CD20+ B cells, its efficacy is enhanced in B cell–rich synovium, which may explain why responders display an ITGB3/ITGA3 -centred network of transcriptional and phosphoproteomic associations that is absent in non-responders. MACHINE LEARNING APPLICATIONS multiDEGGs as feature selection and feature augmentation method in machine learning DEGGs( Sciacca, Alaimo, Silluzio, Ferro, Latora, Pitzalis, Pulvirenti and Myles J. Lewis 2023 ) was initially conceived as differential network analysis tool, with a focus on the interactive exploration of networks to allow easy and quick identification of important nodes, clusters and biological interactions. However, beyond discovery, the identified key differential interactions between biomolecules can subsequently be utilised to predict clinical outcomes in new patients. In the context of machine learning with high-throughput data, researchers frequently encounter scenarios where the number of predictors (P) substantially exceeds the sample size (s). This dimensional imbalance requires selective feature inclusion, both for mathematical and clinical reasons. Models trained in P >> s conditions are prone to overfitting, resulting in high variance, poor generalisability, and spurious correlations that lead to mathematically unreliable predictions with inflated performances on trained data but poor real-world accuracy. Secondary, the use of all variables available in high-throughput data would not be feasible in clinical practise beyond research studies. For example, the implementation of targeted panels with selected, predictive biomarkers represents a more feasible approach than collecting whole transcriptome RNA sequencing in clinical routine. Traditional feature selection approaches focus on identifying individual predictors that demonstrate the strongest association with the outcome of interest (e.g. univariate filters such as t-test or Wilcoxon test). However, since biological systems operate through complex networks of interacting components, high predictive power may also reside in feature combinations and interactions. In general, feature engineering involves a set of techniques that enables the creation of new features by combining or transforming the existing ones( Zheng and Casari 2018 , Nargesian et al. 2017 ). For example, interaction terms are usually captured by multiplying or dividing two or more original features, while polynomial features extend this concept to include powers of individual variables and their combinations. Such a mathematical framework allows models to capture complex, non-additive relationships where the effect of one variable depends on the value of another. The use of combined predictors and polynomial transformations thus offers significant potential to enhance machine learning models because they enable the integration of non-linear biological mechanisms that individual features fail to represent. The higher-order informative content carried by differential interactions, combined with the multiDEGGs’ ability to extract only differential links validated in literature, positions it as particularly suitable both for single feature selection and for guiding the creation of modified predictors in machine learning applications, where conventional black-box algorithms may select single predictors lacking biological significance, thereby compromising the credibility and explainability of the resulting model in clinical settings. DEGGs has already demonstrated potential in identifying gene-gene interactions predictive of treatment response in rheumatoid arthritis( Sciacca et al. 2022 ), however the validation of its efficacy as both a feature selection method and a predictor modification strategy requires evaluation in larger cohorts with rigorous cross-validation protocols. Previous research has shown that applying filtering across the entire dataset introduces bias when assessing model accuracy( Vabalas et al. 2019 ). Feature selection and predictor modification should be conducted exclusively on training data within cross-validation loops to prevent information leakage from the test set. To address this methodological requirement, we have expanded the nestedCV R package( Lewis et al. 2023 ) with a new functionality that enables the nested modification of predictors within each outer fold, ensuring that the attributes learned from the training part are applied to the test data without prior knowledge of the test data itself ( Figure 4 ). The selected and combined features, and corresponding model, can then be evaluated on the hold-out test data without introducing bias. Download figure Open in new tab Figure 4 Nested cross-validation machine learning framework using multiDEGGs-based feature engineering. Specifically, the nestedCV code has been extended to accept any user-defined function that filters or transforms the feature matrix by passing the function name to the modifyX parameter of the nestcv.train() function. A boolean flag ( modifyX_useY ) determines how the transformation is applied. When set to TRUE , the transformation is fitted using both the training response labels (i.e. the dependent variable) and the training feature matrix. The fitted model is then stored and used by another user-defined predict() function to modify the feature space in both training and testing data ( Figure 5 ). This approach is suitable for supervised transformations that need to learn from the training data, such as PCA or the differential network analysis discussed in this paper. For unsupervised transformations like scaling, normalization, or polynomial feature generation that do not require knowledge of the response variable, modifyX_useY must be set to FALSE . Download figure Open in new tab Figure 5 Code architecture for integrating multiDEGGs-based feature processing with the nestedcv cross-validation framework. For this implementation, we then defined – and included in the multiDEGGs package – the filtering and predict functions as follows. The multiDEGGs_filter() function identifies the differential molecular interactions, extracts both individual and paired corresponding variables, and stores them into an S3 model object containing the indices of each selected feature. The second function ( predict.multiDEGGs_filter() ) is the predict method that applies the learned transformation to new data. It takes the paired variables identified during training and creates ratio features (feat. A / feat. B), while also retaining the individual predictors involved in the differential interaction. This ensures that both training and test data of each outer fold undergo the same feature transformation based on the patterns learned from the training set ( Figure 5 ). Finally, it is also worth noting that while traditional filtering functions require the number of selected features to be set by the user, multiDEGGs automatically establishes the best number of predictors thanks to the internal percolation process. Therefore, although a maximum threshold can still be set to prevent excessive feature inclusion, the exact number of selected features is automatically fine-tuned, eliminating arbitrary decision-making. Comparing multiDEGGs with other traditional filters To evaluate the impact of multiDEGGs feature selection on machine learning performance, we trained seven machine learning models to predict treatment resistance using synovial RNA-seq data from the two aforementioned cohorts. For each model, we systematically compared multiDEGGs against five traditional feature selection methods: partial least squares (PLS), t-test, Wilcoxon rank-sum test, random forest, and regularized linear models (glmnet). To ensure fair comparison, we set the same maximum number of final features for all six filtering approaches (40 for the tocilizumab cohort and 50 for the larger rituximab cohort). Finally, to guarantee robust evaluation and prevent data leakage, each model/filter combination has been trained 16 times using nested cross-validation with 8 outer folds and 10 inner folds ( Figure 4 ). Figure 6 and 7 report the area under the receiver operating characteristic curve (AUC) obtained for each combination of model and filter. Download figure Open in new tab Figure 6 Boxplots showing AUC values of each trained model in the prediction of the tocilizumab resistant state of rheumatoid arthritis patients. Seven different models were tested: Generalized Linear Model with Elastic Net Regularization (glmnet), Penalized Discriminant Analysis (pda), Partial Least Squares (pls), Random Forest (rf), Support Vector Machine with Polynomial Kernel (svmPoly), Extreme Gradient Boosting with Linear Booster (xgbLinear), and Extreme Gradient Boosting with Tree Booster (xgbTree). For each model, multiDEGGs was sistematically compared against five traditional filtering methods: pls, T-test, glmnet, rf, and Wilcoxon filter. The maximum number of selected features was set to 40 for all filters. Download figure Open in new tab Figure 7 Boxplots showing AUC values of each trained model in the prediction of the rituximab resistance for rheumatoid arthritis patients. Seven different models were tested: Generalized Linear Model with Elastic Net Regularization (glmnet), Penalized Discriminant Analysis (pda), Partial Least Squares (pls), Random Forest (rf), Support Vector Machine with Polynomial Kernel (svmPoly), Extreme Gradient Boosting with Linear Booster (xgbLinear), and Extreme Gradient Boosting with Tree Booster (xgbTree). For each model, multiDEGGs was sistematically compared against five traditional filtering methods: pls, T-test, glmnet, rf, and Wilcoxon filter. The maximum number of selected features was set to 50 for all filters. On average, starting from 22’975 protein coding genes available in synovial RNAseq, multiDEGGs reduced the feature space to 29 predictors for the tocilizumab cohort and 30 features for the rituximab cohort. Model performances obtained with multiDEGGs consistently outperformed the models trained with the alternative feature selection methods, particularly when using random forest and support vector machines. On average, AUC values obtained with multiDEGGs incremented by 0.10. To ensure this performance improvement is not inflated due to the presence of predictors as both single and combined variables, we repeated the same analysis by setting the keep_single_genes parameter of the predict.multiDEGGs() function to FALSE . This ensures that only combined predictors derived from differential pairs are selected as features, and the list of single variables stored by multiDEGGs_filter_train() function is ignored. As the feature space considerably decrease in this case, the filtering threshold was set to 15 for the tocilizumab cohort and to 25 for the larger rituximab group. Although its overall power decreased as a consequence of the smaller feature space, multiDEGGs confirmed better feature selection in most models (Supplementary Figure 1 and 2). In particular, in the tocilizumab cohort models trained with multiDEGGs obtained higher AUC values in five models out of eight, while the random forest filter’s AUC values were comparable or slightly higher for the remaining models. On the other hand, the rituximab dataset showed higher performances in four models out of eight, but performed poorly for the remaining models (Random Forest, Support Vector Machine with Polynomial Kernel, Extreme Gradient Boosting with Linear Booster, and Extreme Gradient Boosting with Tree Booster), indicating higher sensitivity to feature space reduction and insufficient filter power with fewer features. OTHER TECHNICAL CONSIDERATIONS Package Dependency Analysis and Optimization The complexity of package dependencies has emerged as a critical consideration in R package development, particularly within the bioinformatics ecosystem where numerous interdependent packages can significantly impact installation time, maintenance burden, and computational resource requirements( Gu and Hübschmann 2022 ). Dependency heaviness, defined as the number of additional packages that a parent uniquely brings to a child package, provides a quantitative framework for assessing and optimizing package dependency structures. To evaluate and minimize the dependency burden of our package, we employed the pkgndep (v1.99.3), which systematically analyzes all packages listed in the Depends, Imports, LinkingTo, and Suggests/Enhances fields of the DESCRIPTION file. The initial dependency analysis performed on the original DEGGs code revealed the complete dependency tree with 86 packages directly required for installation (Supplementary Figure 2). For the development of multiDEGGs, we identified potential optimization targets and reduced the dependency heaviness by replacing heavy parent packages with lighter alternatives and removing non-essential dependencies, resulting in a substantially streamlined package architecture with 54 packages directly required (Supplementary Figure 3). This optimization process not only reduces the installation footprint and potential for dependency conflicts but also enhances the package’s long-term maintainability and accessibility for end users. CONCLUSIONS multiDEGGs performs multi-omic differential network analysis by revealing differential interactions between molecular entities (genes, proteins, transcription factors, or other biomolecules). For each omic dataset provided, a differential network is constructed where links represent statistically significant differential interactions between entities. These networks are then integrated into a comprehensive visualisation that allows interactive exploration of cross-omic patterns, such as differential interactions present at both transcript and protein levels. For each link, users can access differential statistical significance metrics (p values or adjusted p values) and differential regression plots. multiDEGGs can also be used as feature selection/augmentation tool in machine learning pipelines. Models trained with features engineered my multiDEGGs constantly improved their performances compared to traditional filters. DATA AVAILABILITY All RNA-Seq data used in this article is available in ArrayExpress , and can be accessed with accession ID E-MTAB-13733 and E-MTAB-11611. STUDY FUNDING This work was supported by Fondazione Ceschina (HFR083 to C.P.); and the European Commission Innovative Medicines Initiative grant 3TR (831434 to C.P.). CONFLICT OF INTEREST All authors declare no competing interests. ACKNOWLEDGEMENTS The authors acknowledge the investigators and participants of the STRAP trial jointly funded by the UK Medical Research Council (MRC) and Versus Arthritis (grant no. MR/K015346/1); and the R4RA trial funded by the UK National Institute for Health and Care Research (NIHR) Efficacy and Mechanism Evaluation (EME) Programme (grant no. 11/100/76) for supporting this study and making their data available for this research. This work also acknowledges the support of the NIHR Barts Biomedical Research Centre (NIHR203330). Funder Information Declared Fondazione Ceschina , HFR083 European Commission Innovative Medicines Initiative , 831434 Footnotes ↵ # Co-senior last authors https://cran.r-project.org/web/packages/multiDEGGs/index.html REFERENCES ↵ Bodein , Antoine , Scott-Boyer , Marie Pier , Perin , Olivier , Lê Cao , Kim Anh , and Droit , Arnaud , ‘ Interpretation of Network-Based Integration from Multi-Omics Longitudinal Data’ , Nucleic Acids Research , 50/ 5 ( 2022 ), E27 OpenUrl PubMed ↵ Cui , Weitong , Xue , Huaru , Wei , Lei , Jin , Jinghua , Tian , Xuewen , and Wang , Qinglu , ‘ High Heterogeneity Undermines Generalization of Differential Expression Results in RNA-Seq Analysis’ , Human Genomics , 15/ 1 ( 2021 ) ↵ Ein-Dor , Liat , and Domany , Eytan , Thousands of Samples Are Needed to Generate a Robust Gene List for Predicting Outcome in Cancer, PNAS , 2006 , CIII ↵ Fukushima , Atsushi , ‘ DiffCorr: An R Package to Analyze and Visualize Differential Correlations in Biological Networks’ , Gene , 518/ 1 ( 2013 ), 209 – 14 OpenUrl CrossRef PubMed Web of Science ↵ Gu , Zuguang , and Hübschmann , Daniel , ‘ Pkgndep: A Tool for Analyzing Dependency Heaviness of R Packages’ , Bioinformatics , 38/ 17 ( 2022 ), 4248 – 51 OpenUrl PubMed ↵ Hershko , Dan D , Robb , Bruce W , Luo , Guangju , and Hasselgren , Per-olof , ‘ Multiple Transcription Factors Regulating the IL-6 Gene Are Activated by CAMP in Cultured Caco-2 Cells’ , Am J Physiol Regul Integr Comp Physiol , 283 ( 2002 ), 1140 – 48 OpenUrl ↵ Humby , Frances , Durez , Patrick , Buch , Maya H , Lewis , Myles J , Rizvi , Hasan , Rivellese , Felice , et al. , ‘ Rituximab versus Tocilizumab in Anti-TNF Inadequate Responder Patients with Rheumatoid Arthritis (R4RA): 16-Week Outcomes of a Stratified, Biopsy-Driven, Multicentre, Open-Label, Phase 4 Randomised Controlled Trial’ , The Lancet , 397/ 10271 ( 2021 ), 305 – 17 OpenUrl Lewis , Myles J. , Çubuk , Cankut , Surace , Anna E. A. , Sciacca , Elisabetta , Lau , Rachel , Goldmann , Katriona , et al. , ‘ Deep Molecular Profiling of Synovial Biopsies in the STRAP Trial Identifies Signatures Predictive of Treatment Response to Biologic Therapies in Rheumatoid Arthritis’ , Nature Communications , 16/ 1 ( 2025 ), 5374 OpenUrl PubMed ↵ Lewis , Myles J , Spiliopoulou , Athina , Goldmann , Katriona , Pitzalis , Costantino , McKeigue , Paul , and Barnes , Michael R , ‘ Nestedcv: An R Package for Fast Implementation of Nested Cross-Validation with Embedded Feature Selection Designed for Transcriptomics and High-Dimensional Data’ , Bioinformatics Advances , 3/ 1 ( 2023 ) ↵ Lewis , Myles , Spiliopoulou , Athina , Cubuk , Cankut , Goldmann , Katriona , and Thompson , Ryan C. , ‘Nestedcv’ ( 2025 ) [accessed 24 September 2025 ] ↵ Li , Zhenzhi , Zuo , Yiming , Xu , Chaohui , Varghese , Rency S , and Ressom , Habtom W , ‘ INDEED: R Package for Network Based Differential Expression Analysis’ , in 2018 IEEE International Conference on Bioinformatics and Biomedicine (BIBM) , 2018 , 2709 – 12 ↵ Liu , Bao Hong , Yu , Hui , Tu , Kang , Li , Chun , Li , Yi Xue , and Li , Yuan Yuan , ‘ DCGL: An R Package for Identifying Differentially Coexpressed Genes and Links from Gene Expression Microarray Data’ , Bioinformatics , 26/ 20 ( 2010 ), 2637 – 38 OpenUrl CrossRef PubMed Web of Science ↵ Lowin , Torsten , and Straub , Rainer H , ‘ Integrins and Their Ligands in Rheumatoid Arthritis.’ , Arthritis Research & Therapy , 13/ 5 ( 2011 ), 244 OpenUrl PubMed ↵ Malemud , Charles J. , ‘ The Role of the JAK/STAT Signal Pathway in Rheumatoid Arthritis’ , Therapeutic Advances in Musculoskeletal Disease ( 2018 ), 117 – 27 ↵ Nance , Rebecca L. , Cooper , Sara J. , Starenki , Dmytro , Wang , Xu , Matz , Brad , Lindley , Stephanie , et al. , ‘ Transcriptomic Analysis of Canine Osteosarcoma from a Precision Medicine Perspective Reveals Limitations of Differential Gene Expression Studies’ , Genes , 13/ 4 ( 2022 ) ↵ Nargesian , Fatemeh , Samulowitz , Horst , Khurana , Udayan , Khalil , Elias B , and Turaga , Deepak , Learning Feature Engineering for Classification , 2017 ↵ Novella-Navarro , Marta , Plasencia , Chamaida , Tornero , Carolina , Navarro-Compán , Victoria , Cabrera-Alarcón , José L. , Peiteado-López , Diana , et al. , ‘ Clinical Predictors of Multiple Failure to Biological Therapy in Patients with Rheumatoid Arthritis’ , Arthritis Research and Therapy , 22/ 1 ( 2020 ) ↵ Rivellese , Felice , Nerviani , Alessandra , Giorli , Giovanni , Warren , Louise , Jaworska , Edyta , Bombardieri , Michele , et al. , ‘ Stratification of Biological Therapies by Pathobiology in Biologic-Naive Patients with Rheumatoid Arthritis (STRAP and STRAP-EU): Two Parallel, Open-Label, Biopsy-Driven, Randomised Trials’ , The Lancet Rheumatology , 5/ 11 ( 2023 ), e648 – e659 OpenUrl ↵ Rivellese , Felice , Surace , Anna E A , Goldmann , Katriona , Sciacca , Elisabetta , Çubuk , Cankut , Giorli , Giovanni , et al. , ‘ Rituximab versus Tocilizumab in Rheumatoid Arthritis: Synovial Biopsy-Based Biomarker Analysis of the Phase 4 R4RA Randomized Trial’ , Nature Medicine , 28/ 6 ( 2022 ), 1256 – 68 OpenUrl CrossRef PubMed Rivellese , Felice , Surace , Anna E A , Goldmann , Katriona , Sciacca , Elisabetta , Çubuk , Cankut , Giorli , Giovanni , ‘ Rituximab versus Tocilizumab in Rheumatoid Arthritis: Synovial Biopsy-Based Biomarker Analysis of the Phase 4 R4RA Randomized Trial’ , Nature Medicine , 28/ 6 ( 2022 ), 1256 – 68 OpenUrl CrossRef PubMed ↵ Sciacca , Elisabetta , Alaimo , Salvatore , Silluzio , Gianmarco , Ferro , Alfredo , Latora , Vito , Pitzalis , Costantino , Pulvirenti , Alfredo , and Lewis , Myles J. , ‘ DEGGs: An R Package with Shiny App for the Identification of Differentially Expressed Gene-Gene Interactions in High-Throughput Sequencing Data’ , Bioinformatics , 39/ 4 ( 2023 ) ↵ Sciacca , Elisabetta , Alaimo , Salvatore , Silluzio , Gianmarco , Ferro , Alfredo , Latora , Vito , Pitzalis , Costantino , Pulvirenti , Alfredo , and Lewis, Myles J , ‘ DEGGs: An R Package with Shiny App for the Identification of Differentially Expressed Gene–Gene Interactions in High-Throughput Sequencing Data’ , Bioinformatics , 39/ 4 ( 2023 ), btad192. OpenUrl CrossRef PubMed ↵ Sciacca , Elisabetta , and Lewis , Myles , ‘MultiDEGGs’ ( 2025 ) [accessed 24 September 2025 ] ↵ Sciacca , Elisabetta , Surace , Anna E.A. , Alaimo , Salvatore , Pulvirenti , Alfredo , Rivellese , Felice , Goldmann , Katriona , et al. , ‘ Network Analysis of Synovial RNA Sequencing Identifies Gene-Gene Interactions Predictive of Response in Rheumatoid Arthritis.’ , Arthritis Research & Therapy , 24/ 1 ( 2022 ), 1 – 14 OpenUrl PubMed ↵ Singh , Amrit , Shannon , Casey P. , Gautier , Benoît , Rohart , Florian , Vacher , Michaël , Tebbutt , Scott J. , et al. , ‘ DIABLO: An Integrative Approach for Identifying Key Molecular Drivers from Multi-Omics Assays’ , Bioinformatics , 35/ 17 ( 2019 ), 3055 – 62 OpenUrl CrossRef PubMed ↵ Stretch , Cynthia , Khan , Sheehan , Asgarian , Nasimeh , Eisner , Roman , Vaisipour , Saman , Damaraju , Sambasivarao , et al. , ‘ Effects of Sample Size on Differential Gene Expression, Rank Order and Prediction Accuracy of a Gene Signature’ , PLoS ONE , 8/ 6 ( 2013 ) ↵ Taylor , Peter C. , Matucci Cerinic , Marco , Alten , Rieke , Avouac , Jérôme , and Westhovens , Rene , ‘ Managing Inadequate Response to Initial Anti-TNF Therapy in Rheumatoid Arthritis: Optimising Treatment Outcomes’ , Therapeutic Advances in Musculoskeletal Disease ( 2022 ) ↵ Tesson , Bruno M. , Breitling , Rainer , and Jansen , Ritsert C. , ‘ DiffCoEx: A Simple and Sensitive Method to Find Differentially Coexpressed Gene Modules’ , BMC Bioinformatics , 11 ( 2010 ) ↵ Vabalas , Andrius , Gowen , Emma , Poliakoff , Ellen , and Casson , Alexander J. , ‘ Machine Learning Algorithm Validation with a Limited Sample Size’ , PLoS ONE , 14/ 11 ( 2019 ) ↵ Zheng , Alice , and Casari , Amanda , Feature Engineering for Machine Learning: Principles and Techniques for Data Scientists ( 2018 ) View the discussion thread. Back to top Previous Next Posted September 27, 2025. Download PDF Supplementary Material Data/Code Email Thank you for your interest in spreading the word about bioRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following multiDEGGs: a multi-omic differential network analysis package for biomarker discovery and predictive modeling Message Subject (Your Name) has forwarded a page to you from bioRxiv Message Body (Your Name) thought you would like to see this page from the bioRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share multiDEGGs: a multi-omic differential network analysis package for biomarker discovery and predictive modeling Elisabetta Sciacca , Susan Wang , Costantino Pitzalis , Myles J Lewis bioRxiv 2025.09.25.678475; doi: https://doi.org/10.1101/2025.09.25.678475 Share This Article: Copy Citation Tools multiDEGGs: a multi-omic differential network analysis package for biomarker discovery and predictive modeling Elisabetta Sciacca , Susan Wang , Costantino Pitzalis , Myles J Lewis bioRxiv 2025.09.25.678475; doi: https://doi.org/10.1101/2025.09.25.678475 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Bioinformatics Subject Areas All Articles Animal Behavior and Cognition (7635) Biochemistry (17697) Bioengineering (13894) Bioinformatics (41951) Biophysics (21455) Cancer Biology (18592) Cell Biology (25507) Clinical Trials (138) Developmental Biology (13380) Ecology (19903) Epidemiology (2067) Evolutionary Biology (24321) Genetics (15610) Genomics (22509) Immunology (17737) Microbiology (40398) Molecular Biology (17182) Neuroscience (88618) Paleontology (667) Pathology (2833) Pharmacology and Toxicology (4825) Physiology (7641) Plant Biology (15158) Scientific Communication and Education (2046) Synthetic Biology (4296) Systems Biology (9825) Zoology (2271)

Text is read by the "Ask this paper" AI Q&A widget below. Extraction quality varies by source — PMC NXML preserves structure cleanly, OA-HTML may include some navigation residue, and OA-PDF can have broken hyphenation. The publisher copy (via DOI) is the canonical version.

My notes (saved in your browser only)

Ask this paper AI returns verbatim quotes from the full text · source: preprint-html

Answers must be backed by verbatim quotes from this paper's full text. Hallucinated quotes are dropped automatically; if no verbatim passage answers the question, we say so. How this works

Citation neighborhood (no data yet)

We don't have any in-corpus citations linked to this paper yet. This is a recent paper (2025) — citers typically take a year or two to land, and the OpenAlex reference graph may still be filling in.

Source provenance

europepmc
last seen: 2026-05-20T01:45:00.602351+00:00
unpaywall
last seen: 2026-05-21T05:10:58.409756+00:00
License: CC-BY-NC-ND-4.0