Transfer Learning and Permutation-Invariance improving Predicting Genome-wide, Cell-Specific and Directional Interventions Effects of Complex Systems

preprint OA: closed
📄 Open PDF Full text JSON View at publisher

Abstract

With the advent of precision medicine, single-drug treatments may not fully satisfy the pursuit of precision medicine. However, single-drug treatments have accumulated a large amount of data and a considerable number of deep learning models. In this context, using transfer learning to effectively leverage the existing vast amount of single-compound intervention effect data to build models that can accurately predict the intervention effects of complex systems is highly worth investigating. In this study, we used a deep model based on permutation-invariance as the core module, pre-trained on a large amount of single-compound intervention data in cell lines, and fine-tuned on a small amount of complex system (like natural products) intervention data in cell lines, resulting in a predictive model named SETComp (the Concat version with ~200M parameters and the Add version with ~173M parameters). The two versions of SETComp achieved an accuracy of 93.86% and 92.70%, respectively, on the complex system-cell-gene association test set, improving by 5.82% to 27.59% compared to the baseline. When predicting the intervention effects of those complex systems the model had never encountered before, the accuracy increased by up to 24.83% compared to the baseline. In our in vitro validation, up to 88.65% of the predictions were confirmed to be correct, and the model’s output showed a significant positive correlation with the real-world foldchange. We further observed SETComp’s potential in various biomedical scenarios, achieving good performance in applications such as mechanism uncovering, repositioning, and compound synergy discovery.
Full text 77,528 characters · extracted from preprint-html · click to expand
Transfer Learning and Permutation-Invariance improving Predicting Genome-wide, Cell-Specific and Directional Interventions Effects of Complex Systems | bioRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-M677548'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search New Results Transfer Learning and Permutation-Invariance improving Predicting Genome-wide, Cell-Specific and Directional Interventions Effects of Complex Systems View ORCID Profile Boyang Wang , Pan Boyu , Tingyu Zhang , Qingyuan Liu , Shao Li doi: https://doi.org/10.1101/2025.04.07.647536 Boyang Wang a Institute for TCM-X, Department of Automation, Tsinghua University , Beijing 100084, China Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Boyang Wang Pan Boyu b Department of Molecular Pharmacology, Tianjin Medical University Cancer Institute & Hospital; National Clinical Research Center for Cancer; Key Laboratory of Cancer Prevention and Therapy, Tianjin; Tianjin’s Clinical Research Center for Cancer , Tianjin 300060, China Find this author on Google Scholar Find this author on PubMed Search for this author on this site Tingyu Zhang a Institute for TCM-X, Department of Automation, Tsinghua University , Beijing 100084, China Find this author on Google Scholar Find this author on PubMed Search for this author on this site Qingyuan Liu a Institute for TCM-X, Department of Automation, Tsinghua University , Beijing 100084, China Find this author on Google Scholar Find this author on PubMed Search for this author on this site Shao Li a Institute for TCM-X, Department of Automation, Tsinghua University , Beijing 100084, China Find this author on Google Scholar Find this author on PubMed Search for this author on this site For correspondence: shaoli{at}mail.tsinghua.edu.cn Abstract Full Text Info/History Metrics Supplementary material Preview PDF Abstract With the advent of precision medicine, single-drug treatments may not fully satisfy the pursuit of precision medicine. However, single-drug treatments have accumulated a large amount of data and a considerable number of deep learning models. In this context, using transfer learning to effectively leverage the existing vast amount of single-compound intervention effect data to build models that can accurately predict the intervention effects of complex systems is highly worth investigating. In this study, we used a deep model based on permutation-invariance as the core module, pre-trained on a large amount of single-compound intervention data in cell lines, and fine-tuned on a small amount of complex system (like natural products) intervention data in cell lines, resulting in a predictive model named SETComp (the Concat version with ~200M parameters and the Add version with ~173M parameters). The two versions of SETComp achieved an accuracy of 93.86% and 92.70%, respectively, on the complex system-cell-gene association test set, improving by 5.82% to 27.59% compared to the baseline. When predicting the intervention effects of those complex systems the model had never encountered before, the accuracy increased by up to 24.83% compared to the baseline. In our in vitro validation, up to 88.65% of the predictions were confirmed to be correct, and the model’s output showed a significant positive correlation with the real-world foldchange. We further observed SETComp’s potential in various biomedical scenarios, achieving good performance in applications such as mechanism uncovering, repositioning, and compound synergy discovery. Introduction With the advancement of artificial intelligence and high-throughput omics technologies, researchers have gained a deeper understanding of life science or biomedical issues such as disease mechanisms and drug development. It has gradually become evident that single drugs may not adequately address the complex mechanisms underlying complex diseases. In this context, combination therapies involving multiple drugs or treatments based on complex systems like natural products (NP) have gradually become more popular, with predicting the targets of multiple drugs or complex systems with high throughput and high accuracy emerging as a hot topic in biomedicine. However, compared to target prediction for single compounds 1 – 5 , target prediction for complex systems requires more consideration on how to integrate compound–compound interactions, compound side effects and other information. Additionally, compared to target prediction for combinations of two or three compounds 6 – 8 , predicting targets for complex systems like NP, which involve an indeterminate number of compounds, appears more challenging. Currently, there are few existing algorithms for target prediction of complex systems, particularly network-based approaches. Li et al. 9 proposed statistical modeling based on the targets of all compounds to predict the holistic targets of complex systems. Zhou et al. 10 – 12 treated complex systems as whole entities and utilized heterogeneous graph neural networks for prediction. However, these prediction methods have certain limitations, including the inability to make predictions based on specific entities such as target cells, low prediction accuracy, no ability for directional prediction and a reliance on extensive prior knowledge, such as the need for the targets of compounds contained within NP. Therefore, developing a cell-specific target prediction model and a target promotion/inhibition direction prediction model suitable for complex systems with a large number of compounds is highly valuable and holds great potential for biomedical application. Transfer learning, as a deep learning technique 13 , 14 , has excellent applicability in scenarios with limited sample sizes. Till now, there is a large amount of data accumulated on compound-intervened transcriptomics like Connectivity Map (CMap) for The Library of Integrated Network-Based Cellular Signatures (LINCS) project 15 . In comparison, transcriptomic data related to complex systems is relatively scarce in comparison. Therefore, applying transfer learning to complex systems-related applications is highly meaningful. Similarly, set-based deep learning possesses the property of permutation invariance, with Deep Sets 16 and Set Transformer 17 being classic models in this domain. Set-based deep learning methods are designed to handle unordered data, such as sets, by ensuring that the model’s output remains invariant regardless of the input data’s order. This property makes them particularly effective for tasks where the order of elements in the set is irrelevant, such as in complex systems with multiple compounds. These models provide a powerful approach to understanding and predicting the interactions and effects of various components within these systems. In this study, we employed transfer learning techniques by treating complex systems as sets of compound combinations and utilizing permutation-invariant (set-based) deep learning as the core model framework, named Set Embedding and Transfer learning model for Complex systems (SETComp), to predict genome-wide, cell-specific and directional targets for both compounds and complex systems. We pre-trained the model on 970,481,750 compound–cell–gene association data obtained from transcriptomics data processed from LINCS and further fine-tuned it on 2,579,488 natural product–cell–gene association data collected and processed from GEO datasets and literature. Two versions of SETComp (Concat version with ~200M parameters and Add version with ~173M parameters) achieved 93.86% and 92.70% accuracy in predicting NP-cell-gene associations, respectively, of which the performance was also tested in real-world in vitro assays, achieving high accuracy. Besides, in multiple downstream application scenarios, SETComp has demonstrated good performance, including revealing the potential mechanisms of action of complex systems on different cell lines corresponding to tumors, repositioning complex systems to discover new potential target diseases, and predicting the gene-level synergistic interactions of complex systems and their constituent compounds. In summary, we have trained a model based on set-based deep learning using vast amounts of compound-cell-gene data and applied transfer learning to achieve high-precision predictions of complex system-cell-gene associations, which demonstrates strong application potential in various downstream scenarios related to biomedical issues. Result Overview of the study In this study, we proposed a model named Set Embedding and Transfer learning model for Complex systems (SETComp), cored by the combination of transfer learning and set-based deep learning. Aiming at predicting transcriptome-level targets in different types of cells, SETComp was designed for stimulating the genome-wide changes in the cellular level after the intervening of complex compound systems like drug combinations or NP ( Figure 1a ). Download figure Open in new tab Figure 1. Model schematic of the SETComp model. (a) The workflow of SETComp. The model is pre-trained on compound-cell-gene association extracted from CMap in LINCS project, and fine-tuned on complex system (like natural products)-cell-gene association processed from collected GEO datasets and literature. The fine-tuned SETComp model demonstrates scalability across multiple biomedical-related application scenarios, including repositioning, mechanism uncovering, and discovery of compound synergy. (b) The architecture of SETComp. The core module of SETComp lied in the permutation-invariance Set Embedding Module, which permutes-invariantly characterizes the complex system packaged as a set (with initial characterization through Deep Sets and further deep characterization through both Deep Sets and Set Transformer, combined at the end of the module in a concatenation/addition form). The set-based embedding is then combined with gene representations and cell line representations, followed by a self-attention module to learn the relationships between features. Finally, the prediction module, consisting of an MLP, performs association classification, including upregulation, downregulation, and no significant association. The compounds are jointly represented by the self-supervised pre-trained InfoGraph model and the Finger printer model, while genes are represented by a combination of PPI network-based node2vec and the pre-trained protein language model Protflash. Cell lines are represented based on the latent layer of the trained VAE. During feature extraction, all models in these sub modules are frozen. To achieve modeling the effects at the transcriptomic level after the intervention of a single compound or complex systems like NP in different cell conditions (Fig S1), SETComp consists of three frozen encoders and three internal modules for feature exacting, feature integration and performing prediction ( Figure 1b ). The three frozen encoders are: a compound encoder, which combines the trained Infograph graph neural network and a Fingerprinter encoder; a gene encoder, integrating a lightweight pre-trained large language model (LLM) with a node2vec model derived from protein-protein interaction (PPI) networks; and a cell state encoder, which is a Variational Auto-Encoders (VAE) model trained on expression profiles from The Cancer Genome Atlas (TCGA). As for the internal modules, centered around the set embedding module, which consists of an initial embedding based on Deep Sets 16 and a further embedding based on Set Transformer 17 , the three internal modules also include a self-supervised module and a prediction module composed of multiple progressively dimension-reduced Multilayer Perceptron (MLP). Compounds with SMILES description, genes, and cell lines were embedded into feature space with the three frozen encoders, respectively. SETComp was firstly pre-trained on 970,481,750 compound–cell–gene association data obtained from 1,805,898 transcriptomics samples with a single intervention of 39,321 compounds, to achieve accurate prediction of transcriptomics-level changes. With the collected relationships between compounds and NP, every natural product was considered as a set of compounds, and the model was then fine-tuned on 2,579,488 natural product-cell-gene pairs extracted from collected data obtained from both the GEO database (update to 2024 May) and literature for predicting transcriptomics-level targets for complex compound systems like NP. In addition, in vitro experimental validations were conducted on different cell lines and NPs to estimate the performance of SETComp in real-world assays. In the downstream, SETComp has been tested for applications in multiple biomedical-related scenarios, including the analysis of intervention mechanisms at the molecular and pathway levels, drug repositioning, and the analysis of synergistic effects between multiple compound components in complex systems. Pre-training of the SETComp model Although RNA-seq technology has been around for over a decade and has accumulated a large amount of data in the context of single compound interventions, such as the CMap from LINCS project, data in the case of complex system interventions, such as NP interventions, is relatively lacking compared to single compounds. We have collected transcriptomic data from compound interventions in CMap which then enhanced by a modified CycleGAN 18 , transcriptomic data from NP interventions from GEO and literature ( Fig 2a ), original transcriptomics data from TCGA and CCLE, gene association networks and protein coding information from STRING 19 , compound structure information from PubChem 20 , and compound composition information for NP from HERB (v2.0) 21 (Fig S2a). After cleaning, screening, enhancement, and preprocessing the transcriptomic data of single compound interventions, we obtained gene differential expression data for over 20,000 compounds (Fig S2b) across more than 90 cell lines (Fig S2c) and constructed 970,481,750 compound-cell line-gene association pairs. Download figure Open in new tab Figure 2. Preprocessing and performance of the SETComp model. (a) Illustration of the dataset construction of SETComp. The compound intervened expression profiles on multiple cell lines for pre-training were obtained from CMap in LINCS project, which were further cleaned and enhanced by a modified CycleGAN model. The compound-cell-gene associations were extract based on the differential expression analysis of the enhanced expression profile. The complex system (like natural products) intervened expression profiles on multiple cell lines for fine-tuning were obtained from GEO datasets and literature. And the complex system-cell-gene associations were also processed from the differential expression analysis. The sets of every complex system were constructed according to the recorded composition information from HERB (v2.0). (b) Detailed view of the embeddings of cell lines with largest number of intervened profiles involved in the pre-training data. After PCA dimensionality reduction of the original expression profiles of the top 10 cell lines with the most interventions, the data were further reduced using UMAP and visualized in a 2-dimensional space. (c) ROC curves of fine-tuned SETComp and baseline models on the test set. The baseline models include traditional machine learning models such as K-Nearest Neighbors (KNN), Linear Discriminant Analysis (LDA), and Decision Trees (DT), as well as a vanilla neural network (MLP) and the NoSet version consisting solely of the attention module and prediction module. (d) The performance of fine-tuned SETComp and baseline models on the test set. In the comparison of accuracy, F1 score, AUC, and AUPR, both the Concat version and Add version of the SETComp model achieved the highest levels, surpassing traditional machine learning models, vanilla neural network, and the NoSet version. The enhanced transcriptomic expression data of untreated cell lines showed a discernible data distribution in UMAP space, indicating that the data-enhanced cell lines possess improved representational characteristics, better clustering, and more robust differentiation in gene expression patterns ( Fig 2b , S2d). The core module of the SETComp model, the Set embedding module (Fig S3), consisted of two set-based deep learning model, Deep Sets 16 and Set Transformer 17 (see supplementary materials). Deep sets initially characterize the input sets (including single-element sets composed of individual compounds and multi-element sets composed of complex systems) as fixed-length sets, providing a preliminary representation of the input. The output after Deep sets representation is then used as a new set input, which is fed into the set Transformer. Through its unique Multihead Attention Block (MAB), Set Attention Block (SAB), Induced Set Attention Block (ISAB), and Pooling by Multihead Attention (PMA) blocks, further deep representation is obtained. Meanwhile, we employ two strategies, concatenation and addition, to combine the shallow set-based representation from Deep sets and the deep set-based representation obtained from the collaborative characterization of Deep sets and the set Transformer by either concatenating or adding them, to construct the Concat version (~200M parameters) and the Add version (~173M parameters) of the SETComp model. After concatenating the output of the set embedding module with the gene features and the cell line features obtained from the VAE, self-attention is applied to further learn the relationships between the features. Finally, the set-cell-gene prediction is performed through a prediction module composed of three layers of MLP with progressively reduced dimensions. After performing a grid search on the training parameters, including model parameters such as the size of the set Transformer, the size of MLP in the prediction module, as well as training parameters like batch size, learning rate, and regularization parameters such as L2 regularization and dropout, the optimal model parameters for the Concat version were used to determine the model structure (Table S1). Subsequently, the Add version also underwent a grid search on the training and regularization parameters (Table S2). Both versions of the SETComp model approach convergence after 5 epochs (Table S3, 4), and without overfitting, we trained both versions of the model to 10 epochs with the optimal parameters. In our training set division for pre-training, all compounds belonging to NP components were extracted separately as the test set. By training on the entire training set, we found that the vanilla neural network composed only of the prediction module outperforms the two versions of the SETComp model (with AUCs of 0.9480 and 0.9417, respectively) in predicting compound-cell-gene associations, and further outperforms the model (the NoSet version), which consists of the self-attention module and the prediction module (Fig S4 a-d, Table S5). This may be because when predicting a single compound, treating it as an individual rather than as a set is more suitable and sufficiently saturated for the model. The two versions of the SETComp model perform better in predicting up-regulation and down-regulation, with an AUC of 0.96 for both (Fig S5 a-d). Further, on a smaller-scale training set (1/100 size of the full training set), we found that deep learning-based models performed better on the test set compared to traditional machine learning models (Fig S6 a-d, Table S6), including K-Nearest Neighbors (KNN), Linear Discriminant Analysis (LDA), and Decision Trees (DT). Fine-tuning improving the performance of prediction on complex systems After completing the model pre-training, we focused on our research goal—achieving high-precision complex system-cell-gene association prediction. We pre-processed various complex system sets with compound features as elements, with the largest set containing up to 406 elements (Fig S7a). At the same time, we cleaned, pre-processed, and analyzed expression data of complex systems from the GEO database and literature across different cell lines (Fig S7b, c). We found that there was no significant correlation between the total expression levels of cell lines after complex system interventions, such as NP, and the number of compounds (Fig S7d) contained in NP (P=0.4331). During the fine-tuning procedure, we also performed a grid search on the training parameters, including batch size and learning rate, as well as regularization parameters such as L2 regularization and dropout for the two versions of the SETComp model, to determine the optimal parameter combination (Table S7). We found that the performance of the two versions of the SETComp model, after pre-training and fine-tuning, surpassed traditional deep learning models such as KNN, LDA, and DT, as well as the vanilla neural network and the NoSet version in deep learning, and their performance when trained only on fine-tuning ( Fig 2c ). The two versions of the SETComp model achieved 93.86% and 92.70% in accuracy (Table S8), as well as 0.9888 and 0.9856 in AUC, respectively, improving the accuracy compared to the baseline machine learning model, increasing from 5.82% to 27.59%, while the AUC increases from 7.83% to 15.63% (Fig S8a, c) and the AUPR increases from 8.20% to 24.4% (Fig S8b). Compared to the vanilla neural network, the accuracy can be improved by 10.10%, and compared to the NoSet version, it shows an improvement of 5.74%. Compared to the model that only underwent pre-training, we found that on the test set, the model that underwent both pre-training and fine-tuning performed better than the model that only underwent fine-tuning, and both outperformed the model that only underwent pre-training. This finding also demonstrates the benefit of fine-tuning on the prediction model, as well as the effectiveness of pre-training on compound-cell-gene in enhancing the model’s performance ( Fig 3a , S9a-c, Table S9). At the same time, this finding also indicates that, beyond the superiority of deep learning itself in this task, the attention module and set embedding module progressively contribute to the improvement of the model’s performance. Then, based on the pre-trained model, we gradually added the size of the fine-tuning training set from 0% to 100%, and observed that as the size of the fine-tuning training set increased, the model’s performance on the fine-tuning test set gradually improved for both the Concat version ( Fig 3b , S10a, b) and the Add version (Fig S10c) of the SETComp model. Furthermore, by adding models pre-trained on a small-scale pre-training dataset, we also performed fine-tuning and found that a larger pre-training dataset led to better performance on the fine-tuning test set, while accelerating the improvement of Accuracy and AUC ( Fig 3c , S11c, d) as well as the reduction of loss (Fig S11a, b), and improving Accuracy and AUC of the converged models ( Fig 3e , Fig S12a-d, Table S10). Download figure Open in new tab Figure 3. Ablation experiments and generalization study of the SETComp model. (a) The performance of fine-tuned SETComp and other deep learning models on the test set trained with different training datasets. In the comparison of accuracy, F1 score, AUC, and AUPR, both the Concat version and Add version of the SETComp model, after pre-training and fine-tuning, achieved the highest levels, surpassing the vanilla neural network and the NoSet version. They also outperformed their own performance when trained only on pre-training data or only on fine-tuning data. (b) A line chart showing how the AUC score of the Concat version of SETComp on the test set changes with the increasing percentage of the fine-tuning dataset in the total fine-tuning dataset. As the percentage of the fine-tuning dataset in the total fine-tuning dataset increases, i.e., with the increase in the size of the fine-tuning dataset, the model’s AUC on the test set also increases. (c) A line chart displaying how the Accuracy and AUC scores of the Concat version of SETComp on the test set change with variations in pre-training conditions. Compared to models pre-trained on a small-scale pre-training dataset (1/100) and models with no pre-training, the model fine-tuned and trained on the full pre-training dataset achieved higher Accuracy and AUC scores. (d) The performance of the model and baseline models on the test set extracted from intervention data from complex systems that the model had not encountered. In the comparison of accuracy, F1 score, AUC, and AUPR, both the Concat version and Add version of the SETComp model achieved the highest levels, surpassing traditional machine learning models and vanilla neural network. (e) The relationship between the model’s predicted output and the Foldchange in real-world data. After the softmax transformation, the model’s predicted outputs for class 0 (up-regulated) and class 1 (down-regulated) associations show a significant positive correlation with the corresponding real-world data’s Foldchange. Furthermore, to assess the model’s generalization ability, we re-divided the fine-tuning dataset based on NP, using 10% of the NP-related transcriptomic data as the test set, and the remaining data as the training and test sets for fine-tuning the model, and re-trained the pre-trained model. When predicting NP that the model has not seen by the model before, it can achieve accuracies of 82.75% and 82.66% for the two versions of the model ( Fig 3d , Table S11), respectively, which is an improvement of up to 24.83% in accuracy compared to the baseline machine learning models and 5.59% to the vanilla neural network (Fig S13a-c). Finally, to explore the potential for quantitative prediction in the future, we analyzed the model output values (predicted as down-regulated if negative) after applying softmax on all NP-cell-gene associations in the fine-tuning test set, and their actual statistical test logFoldChange ( Fig 3e ). We found a significant correlation, and this correlation varied across different NPs, ranging from 0.6734 to 0.9226 (Fig S14). This finding suggests that the model’s predicted results have a significant positive correlation with the actual transcriptomic logFoldChange, and it somewhat indicates the feasibility of extending the model to quantitative prediction. In vitro experiments validating the performance of the SETComp model To further validate the predictive performance of the model, we conducted real-world transcriptomics assays to verify the model’s predictions. In multiple cell lines, we tested the effects of several NPs, including cinnamon, codonopsis, and astragalus ( Fig 4a ), and obtained their RNA-seq counts through transcriptome sequencing for differential expression analysis. Download figure Open in new tab Figure 4. Case study of the SETComp model. (a) Case study illustration. We conducted 24-hour interventions of multiple complex systems (natural products) on cell lines and extracted RNA for RNA sequencing. (b) Accuracy of SETComp’s prediction of the intervention effects of Astragalus and Codonopsis in the A549 cell line within the differential expressed genes. In the A549 cell line, the accuracy of SETComp’s prediction of the intervention effects of Astragalus and Codonopsis for class 0 and 1, increases as the threshold increases, reaching a maximum of 88.65%. For the prediction of class 2, the accuracy can reach up to 94.01%. (c) The correlation coefficient between the SETComp predicted scores and the logFC obtained from real-world experiments within the differential expressed genes. As the threshold increases, the correlation coefficient between the SETComp predicted scores and the logFC from real-world experiments gradually increases, reaching a maximum of 0.4665. (d) Potential target GPX2 identified by combining model predictions, real transcriptomic data, and TCGA tumor data. After integrating the model’s predicted results (threshold 0.7) with transcriptomic intervention results, it was found that the three NPs collectively downregulated GPX2 and many other genes. Among them, six genes were found to be significantly overexpressed in tumor tissues of LUAD and LUSC. The upper bubble plot shows the downregulation of GPX2 in A549 cells after treatment with three NP in real-world transcriptomics assay. After conventional bioinformatics processing (Fig S15a), we obtained the differential expression of each gene after cell line intervention. Among the differential genes (P value < 0.05), for the model-predicted upregulated targets of codonopsis in A549, 75.44% were found to be truly upregulated (with a threshold of 0.9), while 75.28% of the downregulated targets were confirmed to be downregulated; for the model-predicted upregulated targets of astragalus in A549, 76.25% were found to be truly upregulated (with the threshold of 0.9), and 83.95% of the downregulated targets were confirmed to be downregulated ( Fig 4b ). These accuracies improve as the threshold increases, reaching up to 81.72% for the upregulated prediction and 81.51% for the downregulated prediction for codonopsis at the threshold of 0.99; for astragalus, the upregulated prediction reaches 86.25% and the downregulated prediction reaches 88.65%. For cinnamon, the maximum accuracy in predicting downregulation can reach 83.12% (Fig S15b). Among all genes, in the prediction of upregulated genes, the accuracy under the intervention of codonopsis or astragalus can reach a maximum of 76.84% (Fig S15c), with accuracy showing an increasing trend as the threshold increases. In the prediction of negative samples (class 2), the model’s predictions remain relatively stable, with a maximum of 94.01% of predictions for class samples being confirmed as having no significant differences for astragalus, and 90.85% for codonopsis. Similarly, we also observed the relationship between the scores predicted by the model for each gene under different NP interventions after softmax and the actual fold change of gene expression under each NP intervention. We found that the expression of intervention genes predicted by the model (classified as 0 or 1) is consistently significantly positively correlated with their true fold change, and as the threshold increases, the correlation score between them gradually increases ( Fig 4c ). In the differential genes of the A549 cell line, this correlation can reach a maximum of 0.4665, and in all genes, it can reach 0.2409 (Fig S16a). Focusing on the results of individual NP interventions, the positive correlation coefficient remains significant, and this phenomenon of increasing positive correlation with increasing threshold is still maintained. In the differential genes of the A549 cell line, the regularization coefficients under the interventions of astragalus, codonopsis, and cinnamon can reach 0.5788, 0.5486, and 0.4425, respectively (Fig S16b); in the full gene range of the A549 cell line, the regularization coefficients under the interventions of astragalus, codonopsis, and cinnamon can reach 0.2306, 0.2146, and 0.2778, respectively (Fig S16c). This finding further demonstrates that the model’s predictions not only enable qualitative directional predictions and predictions of intervention, but also possess the potential for quantitative prediction of relative expression levels, providing evidence support for subsequent quantitative predictive analyses in application scenarios. Combining the prediction results of the SETComp model of multiple complex systems on A549, real-world transcriptomics assays and transcriptomics data from TCGA, we attempted to identify potential targets of some complex systems in the treatment of the A549-related clinical cancers, specifically non-small cell lung cancer LUAD and LUSC. Among the genes that were predicted by the model with high scores (threshold of 0.7), that genes showed significant differential expression in the transcriptome, and those were significantly different in both LUAD and LUSC, we found that some potential targets such as GPX2 ( Fig 4d ), PRR13, and APOC1, which are highly expressed in cancer samples of LUAD and LUSC (Fig S17b, c), were significantly downregulated under the intervention of all these three NP (Fig S17a). GPX2 has been studied in lung cancer, with reports indicating its involvement in apoptosis 22 , 23 , immune regulation 24 and oxidative stress 25 – 27 , with clinical significance 28 in lung cancer. The above findings, along with relevant literature, provide some evidence that SETComp’s predictive capability demonstrates strong generalization, robustness, and potential for application expansion. The extensive applications of the SETComp model in various biomedical issues Last but not least, based on the model validated in both the test set and real-world assays, we use multiple biomedical application scenarios as downstream tasks to explore the practical application value of the model. First, we attempted to apply SETComp to explore the molecular and pathway mechanisms of complex system interventions in cell lines. Mechanism uncovering Based on the aforementioned finding, that the model’s predicted values quantitatively reflect the activation/inhibition intensity of genes after intervention to some extent, we can apply Gene Set Enrichment Analysis (GSEA) to perform enrichment analysis on the prediction results and observe the potential pathway-level changes they may induce. We predicted the intervention effects of three NPs, including cinnamon, codonopsis, and astragalus (Fig S18a) on two cell lines, MCF-7 and A549, and constructed their potential regulatory molecular networks (Fig S18b-d). After GSEA based on the predicted up-regulated and down-regulated genes, astragalus was found to affect certain pathways in multiple modules in both MCF-7 and A549 cell lines ( Fig 5a, b ). Although astragalus potentially intervenes in pathways belonging to modules such as signal transduction, cell growth and death, and the immune system in both MCF-7 and A549 cell lines, the specific pathways within each module vary. For example, in cell growth and death, astragalus potentially regulates tumor cell growth and proliferation in MCF-7 by promoting the p53 signaling pathway 29 – 31 , while in the A549 cell line, it potentially promotes tumor cell apoptosis by enhancing the Apoptosis signaling pathway 32 – 34 . Additionally, in some classic signaling pathways, astragalus potentially inhibits tumor cell growth and proliferation in MCF-7 through the FOXO signaling pathway 35 – 37 , while in A549-related non-small cell lung cancer, it potentially alters the tumor immune microenvironment by promoting the Toll-like receptor signaling pathway 38 , 39 . Similar analysis also revealed that in the potential pathways of cinnamon and codonopsis interventions in MCF-7 (Fig S19a, b) and A549 (Fig S19a, b), the model is able to identify different potential molecules, pathways, and modular mechanisms for different complex systems intervening in different cell lines. Additionally, we observed the impact of different thresholds on the robustness of the analysis results. Taking the intervention of astragalus in MCF-7 and A549 as an example, we found that when the threshold was set to 0.7 (Fig S21a, b) or 0.9 (Fig S21c, d), the analysis results did not change significantly. The variations were likely observed in the same pathways but with different enrichment levels. Download figure Open in new tab Figure 5. Application of SETComp in various biomedical scenarios. (a)-(b) Mechanism uncovering of potential intervention pathways of Astragalus in MCF-7 (a) and A549 cell lines (b) based on the prediction of SETComp. The GSEA enrichment analysis revealed that SETComp predicted Astragalus to be involved in multiple modules in both MCF-7 and A549 cell lines, like signal transduction, cell growth and death. However, the specific effects on MCF-7 and A549 cells differ. For example, Astragalus primarily affects the p53 signaling pathway in MCF-7, whereas in A549, it mainly potentially upregulates the apoptosis signaling pathway. (c)-(d) Reposition of certain complex systems like Red Ginseng (c) and American Ginseng (d) based on the prediction of SETComp. KEGG and DO enrichment analysis based on SETComp prediction results revealed that Red Ginseng and American Ginseng have potential interventions for multiple diseases, and this was supported by evidence from NCT clinical trials. (e) Discovery of compound synergy in complex systems intervening in BRCA based on SETComp. Based on SETComp’s predictions, multiple compounds in Astragalus (up) and Codonopsis (down) synergistically promote the expression of CEBPD and inhibit the expression of HMGB3 in MCF-7, respectively. It was further found that CEBPD was inhibited and HMGB3 was highly-expressed in cancer samples of BRCA, and BRCA patients with low expression of CEBPD have poor prognosis (P = 0.025), as well as BRCA patients with high expression of HMGB3 (P = 0.017). (f) Discovery of compound synergy in complex systems intervening in LUAD and LUSC based on SETComp. Based on the predictions of the SETComp model, multiple compounds in Astragalus synergistically inhibit the expression of SPP1 in A549, which was also highly expressed in cancer samples in both LUAD and LUSC, while high expression of SPP1 in LUAD (P = 0.015) and LUSC (P = 0.052) patients has been found to be associated with poor prognosis. NP repositioning Next, we explored the potential of SETComp in the field of drug repositioning. We analyzed several NPs with substantial clinical reports, including ginseng, American ginseng, red ginseng, and cinnamon, and, after masking the clinical reports in advance, we observed whether the model could predict repositioning with clinical evidence. We first performed Kyoto Encyclopedia of Genes and Genomes (KEGG) pathway analysis and Disease Ontology (DO) analysis on the model’s output. Based on different KEGG disease classifications, we organized the potential diseases that these NPs may intervene in (Fig S22a-d). According to the predicted KEGG and DO disease terms, red ginseng was found to liver function-related issues like hepatitis (DOID:1575, hsa05160, has05161) validated by clinical trial NCT0395412, and certain glucose-related diseases like Type II diabetes mellitus (hsa04930), Insulin resistance (hsa04931) and hyperglycemia (DOID:4195), validated by many trials like NCT03775733 and NCT01911663 , as well as blood pressure-related diseases like essential hypertension ( Fig 5c ). American ginseng was found to potentially affect hypertension, multiple sclerosis, respiratory infection and upper respiratory tract infections, which were also found in many clinical trials ( Fig 5d ), while cinnamon potentially intervened diseases related to diabetes or insulin resistance (Fig S22e) and ginseng potentially affected various diseases, including Alzheimer’s disease, atherosclerosis, glucose-related disorders, liver function issues, and multiple sclerosis (Fig S22f). Discovery of compound synergy in NP Finally, for the most critical issues of complex systems, we provide a potentially feasible application based on our model. We attempted to establish the relationship between the NP-cell-gene associations predicted by the model and the compound-cell-gene associations of the compounds that make up the NP (Fig S23a, b). In the prediction results of Astragalus intervention in MCF-7, we found that the compounds CID: 46906036 and CID: 644263 potentially exert synergistic effects on the gene CEBPD by jointly promoting the expression of CEBPD ( Fig 5e ), contributing to the upregulation of CEBPD expression in MCF-7 induced by Astragalus. Furthermore, we observed that CEBPD is significantly down-regulated in cancer tissues corresponding to Breast Cancer (BRCA) in MCF-7, and its low expression predicts a significantly worse prognosis (HR=0.69, p=0.025). Similarly, the compounds CID: 3083834 and CID: 644263 in Codonopsis, potentially synergistically inhibit HMGB3 ( Fig 5f ), which is highly expressed in tumor tissues of BRCA, and patients with high expression of HMGB3 have a significantly poorer prognosis (HR=1.5, p=0.017). Besides, in the prediction of Astragalus intervention in MCF-7, four compounds including CID: 644263 synergistically regulate TPD52 (Fig S23c), a gene that is highly expressed in BRCA tumor tissues, and its high expression is associated with poor prognosis in patients. This suggests that genes such as CEBPD and TPD52 may be potential targets for Astragalus to improve the prognosis of BRCA patients, while the compound CID: 644263 could be a key compound in Astragalus for combating BRCA. On the other hand, in the prediction of Astragalus intervention in A549, compounds CID: 73611 and CID: 5318203 potentially synergistically inhibit the expression of the gene SPP1 ( Fig 5g ), which is significantly upregulated in both Lung Adenocarcinoma (LUAD) and Lung Squamous Cell Carcinoma (LUSC), the two types of non-small cell lung cancer corresponding to A549. Additionally, in patients with LUAD and LUSC, high expression of SPP1 is associated with poorer prognosis, enhancing its potential as a target for intervention in LUAD and LUSC. The downregulation of the SPP1 gene was also observed in the transcriptomic analysis of Astragalus intervention in A549 (P=0.04, log2FoldChange=-0.28). Furthermore, there are some potentially more complex compound synergies. For example, in the prediction of Astragalus intervention in A549, compounds such as CID: 442813 potentially regulate B3GNT3 (Fig S23d), a gene that is highly expressed in LUSC and associated with poor prognosis in patients with high expression. Discussion As the demand for precision medicine continues to increase 40 – 44 , the intervention of a single drug has gradually become insufficient to achieve perfect precision medicine, while complex systems composed of multiple compounds offer a promising and potential direction for precision medicine 45 . However, despite the accumulation of vast amounts of multi-omics data from single-compound interventions 46 and the development of numerous deep models focused on single-compound predictions and generation 47 – 49 , data reflecting the effects of complex system interventions and models capable of predicting or generating complex system-related functions are scarce. Against this backdrop, there is an urgent need to develop a model capable of predicting the intervention effects of various complex systems, such as NP, on different cell conditions. Thus, we proposed a model named SETComp, which is capable of predicting genome-wide, cell-specific directed intervention effects for complex systems, such as NP. The two versions of the model (the Concat version with ~200M parameters and the Add version with ~173M parameters) are based on transfer learning and permutation-invariance, with pre-trained compounds as single-element set representations from 970,481,750 compound-cell-gene associations, and further fine-tuned on 2,579,488 compound-cell-gene associations derived from GEO data and literature. The model achieved an accuracy of 93.86% and 92.70%, AUC of 0.9888 and 0.9856, respectively, on the test set of complex system-cell-gene associations. In the ablation experiments, we demonstrated the improvement in model prediction performance brought by permutation-invariance, and in tasks involving complex systems that the prediction model had never encountered before, we achieved prediction accuracies of 82.75% and 82.66%. We also explored the relationship between the model’s predicted output and the true fold change, and found a strong correlation between them. This provides a new perspective for the subsequent development of quantitative prediction or generation models. The model was further validated through in vitro real-world transcriptomic assays that we personally conducted, achieving an accuracy of up to 88.65% among the intervention of multiple NP. We further demonstrated SETComp’s potential in various biomedical application scenarios, including mechanism uncovering of NP, repositioning of NP, and discovery of compound synergistic effects. There were still some limitations in our studies, including the lack of consideration for the quality ratio of each compound in the complex system, and the inability to achieve quantitative prediction of complex system-cell-gene associations. Fortunately, the set embedding module based on set-based composition, compared to directly summing the feature coefficients of the compounds that make up the complex system (which assumes each compound has the same proportion), can, to some extent, address this issue through various attention mechanisms. Regarding quantitative prediction, although we have not directly achieved quantitative prediction of complex system-cell-gene associations, we have deeply explored the relationship between the model’s predicted output and the actual fold change, observing a strong positive correlation. This provides a promising perspective for the future development of quantitative prediction or generation models. For future research, on one hand, we plan to collect more single-cell level compound intervention effect data based on the recently published Tahoe-100M dataset 46 , combined with the current 180M-level cell-line-based compound intervention expression profiles, for pre-training generative models. On the other hand, we are currently conducting single-cell sequencing experiments on the intervention effects of multiple complex systems in animals to provide data from a single-cell perspective, which will be used to fine-tune the generative model, ultimately achieving the generation of cell-specific whole-genome expression profiles after complex system interventions. We believe that such a generative model for cell-specific whole-genome expression profiles of complex system intervention will provide better basis for prediction in multiple biomedical application scenarios, such as complex system mechanism uncovering, repositioning, and compound synergy discovery. Methods and materials Construction and training of the SETComp model Data collection, preprocess and generation In this work, we collected transcriptomics data of various compounds on different cell lines in CMap from LINCS in NIH Common Fund program 15 , preprocessing with python package cmapPy v4.0.1, and the data was then enhanced using a modified CycleGAN model 18 to improve the feature dimension from the original 978 to the genome-wide 23,614. Consider the mapping function G : X → Y , where X represents the original L1000 assay and Y denotes the inferred RNA-seq assay, along with its corresponding discriminator D Y . Accordingly, the adversarial loss function is defined as: Finally, 1,805,898 samples with a single intervention of 39,321 compounds were used for further study. Compounds with at least 3 technical replications in one cell-line were kept. And the compounds in the processed matrix were annotated in PubChem database 20 and the chemical structures of these annotated compounds were obtained, represented by SMILES. NP were collected in HERB (v2.0) 21 database, and information like chemical compositions and Latin names were collected. Compounds in these NP were also annotated in PubChem database for standard PubChem CID and chemical structures. A total of 46,419 compounds remained, among which 25,751 served as chemical compositions in at least one of 6198 NP. For compounds, the enhanced transcriptomic data uniformly after 24h of intervention at the highest concentration of the compound was uniformly used to calculate expression pattern of each gene, while transcriptomic data or microarray in which cell lines were intervened by NP were collected from GEO database (update to 2024 May) and literature 50 , 51 . The expression data was further processed with limma 52 and DESeq2 53 to calculate DEGs with statistical models. The expression pattern of each gene in different cell lines under different interventions by compounds or NP was classified into 3 categories, including Up (log2FoldChange > 0, P value < 0.05), Down (log2FoldChange < 0, P value 0.05). The data and information of cell lines were collected from CCLE and TCGA database. The expressions matrix of CCLE and TCGA were processed with log transform, if necessary. The data and information of genes were collected from STRING database 19 , including the interactions between two genes and the corresponding encoded protein sequence. Finally, 970,481,750 compound-cell line-gene pairs and 2,579,488 NP-cell line-gene pairs were obtained for further data pre-training and fine-tuning in 3 classes (up-regulated, down-regulated and no change), respectively. Feature extraction for compounds, genes and cell lines Based on the processed data, every compound was firstly embedded in a self-supervised pre-training model, Infograph 54 , which proposed to maximize the mutual information between the graph-level and node-level representations and was implemented in TorchDrug. In the InfoGraph model, the mutual information estimator I ϕ , ψ , is modeled using the discriminator, T ψ which is parameterized by a neural network with parameters ψ : Here, x denotes an input sample, while the negative sample x ′ is drawn from the distribution , which matches the empirical distribution of the input space. Additionally, the softplus function is defined as sp ( z ) = log (1 + e z ). And the InfoGraph model was trained on the ZINC 2M database using the recommended parameters, with five hidden layers and 300 neurons in each layer. Another embedding for compounds was based on the PubChem fingerprint calculation implemented with scyjava package in Python. Finally, each compound was represented by a 1,181-dimensional feature vector, consisting of an 881-dimensional fingerprint embedding and a 300-dimensional Infograph embedding. The features of NP were composed of the corresponding features of chemical compositions and assembled into a set for each natural product. Genes were also embedded in two ways, including Node2Vec based on network relationships and ProtFlash 55 , A lightweight protein language model, based on corresponding encoded protein sequence. The loss function of the masked training of ProtFlash is defined as: In which s i is the true amino acid and s M is the masked sequence as context. Node2Vec algorithm was implemented by torch geometric to achieve parallel training on GPU and every gene was embedded into 256 dimensions, while ProtFlash model was used with pre-trained weights and every gene was embedded into 768 dimensions. For cell line embedding, a Variational Autoencoder (VAE) model was first trained on the TCGA expression matrix, and the CCLE expression matrix was then embedded with the trained VAE model. The loss of the VAE model is defined as the sum of MSE loss and KL loss: In this function, N represents the total number of samples in the dataset, where each sample x i is the original expression profile and is the corresponding reconstructed profile generated by the decoder. For each sample, the approximate posterior distribution is characterized by the mean μ ij and standard deviation σ ij for the j -th dimension of the latent variable z , with d denoting the overall dimensionality of the latent space. A hyperparameter β is introduced to control the weight of the KL divergence term, allowing for a balance between reconstruction fidelity and latent space regularization. Grid search was applied to find the best hyper-parameter combination, in which the search ranges were set as follows: batch size: 32, 64, 128, 256; learning rate: 1e-3, 1e-4, 1e-5; dimension in MLP layer 1: 4096, 2048, 1024; dimension in MLP layer 2: 1024, 512, 256; dimension in MLP layer 3: 256, 128, 64; dimension in latent layer: 64, 32, 16. For each combination, the model was trained for 20 epochs. And the loss function was composed of two parts: Mean Squared Error loss as the reconstruction loss and Kullback-Leibler Divergence loss. Finally, every cell line was embedded into 64 dimensions, in which VAE showed the best reconstruction performance. Main architecture of the model Apart from the feature extraction modules for compounds, genes and cell lines, the main architecture of the model was constructed with three modules including the set embedding module, the attention module and the prediction module. In the set embedding module, we utilized Deep Sets 16 and Set Transformer 17 (See Supplementary materials) to effectively model set-structured data inherent in our problem domain. The set embedding module consisted of two Deep Sets models and one Set transformer model, both implemented in Pytorch environment ( https://github.com/juho-lee/set_transformer/tree/master ). Traditional neural network architectures often assume a fixed-size input or a specific ordering of elements, which is unsuitable for sets that are inherently unordered and variable in size. Deep Sets address this challenge by providing a framework that is permutation-invariant to the input set elements. For the Concat version and Add version of the SETComp model, the outputs of the set embedding are as follows, respectively: In addition, the attention module further refines the representations obtained from the concatenate of the set embedding module for compounds or NP and the extracted gene embedding and cell line embedding. It utilizes advanced attention mechanisms to focus on the most relevant features within the data, enhancing the model’s predictive capabilities by capturing intricate patterns and dependencies. Finally, the prediction module takes the refined representations from the attention module to perform the final prediction task. This module typically consists of three fully connected layers that map the high-dimensional embeddings to the desired output space, enabling the model to generate accurate predictions based on the learned representations. In vitro cell lines intervention by multiple NP for validation Cell Culture and Related Reagents The human-derived breast cancer cell line MCF-7 and human-derived lung cancer cell line A549 were both purchased from the China Infrastructure of Cell Line Resources (China). These cells were then cultured in DMEM and RPMI-1640 media, respectively, with 10% (v/v) FBS and 100 U/ml streptomycin/penicillin at 37°C in a 5% CO2 environment. The Chinese medicinal herbs Astragalus, Cinnamon, and Codonopsis were purchased from Jiangyin Tianjiang Pharmaceutical Co., Ltd. (China). All the herbal granules were stored in a dry, light-protected environment. Cell Treatment and Drug Administration Method Human-derived breast cancer cells (MCF-7) and human-derived lung cancer cells (A549) were digested with trypsin, resuspended, and 10 μL of cell suspension (1:1 dilution) was counted using a cell counting plate. The cells were then seeded in a 10 cm cell culture dish at a density of 4 × 10^6 cells per well, and placed in the cell incubator to allow attachment. The following day, after the cells attached, the culture medium was replaced with 8 mL of fresh serum-free medium. Astragalus, Cinnamon, and Codonopsis were adjusted to final concentrations of 0, 100, and 200 μ g/mL, respectively, and then cultured in a 37°C, 5% CO2 cell incubator for 24 hours. On the third day, cells from each group were collected into 1.5 mL sterile, enzyme-free, Eppendorf tubes using Trizol reagent, and stored at −80°C for sequencing. RNA extraction, library construction, and sequencing Total RNA was extracted from the cell lines using TRIzol® reagent (Magen). The A260/A280 absorbance ratio of the RNA samples was measured using a Nanodrop ND-2000 (Thermo Scientific, USA), and the RNA Integrity Number (RIN) was determined using an Agilent Bioanalyzer 4150 (Agilent Technologies, CA, USA). Only RNA samples that passed quality control were used for library construction. The PE library was prepared according to the instructions of the ABclonal mRNA-seq Lib Prep Kit (ABclonal, China). mRNA was purified from 1μg of total RNA using oligo(dT) magnetic beads, and then fragmented in the ABclonal First Strand Synthesis Reaction Buffer. Subsequently, mRNA fragments were used as templates to synthesize the first strand of cDNA using random primers and reverse transcriptase (RNase H). The second strand of cDNA was synthesized using DNA polymerase I, RNase H, buffers, and dNTPs. The synthesized double-strand cDNA fragments were ligated with adapter sequences for PCR amplification. The PCR products were purified and the library quality was assessed using an Agilent Bioanalyzer 4150. Finally, sequencing was performed on the Illumina Novaseq 6000 / MGISEQ-T7 sequencing platforms. Application of the SETComp model in biomedical scenarios Mechanism uncovering for NP For the prediction of complex system intervention effects, we selected genes from class 0 and 1 with softmax scores greater than or equal to 0, 0.7, and 0.9 as potential upregulated or downregulated genes under different thresholds. These genes were then sorted by their scores (with negative scores assigned to genes predicted as class 1) for subsequent enrichment analysis. Next, we performed GSEA analysis using the R package clusterProfiler (v4.9.1) to identify pathways potentially intervened by the model, and classified the pathways into activation or inhibition based on Normalized Enrichment Score (NES). This enrichment analysis was repeated for potential intervention genes under different thresholds to obtain mechanism uncovering results for each threshold. For visualizing specific pathways, we used the gseaplot2() function from the R package enrichplot. Repositioning for NP with clinical evidences We collected clinical reports of various complex systems from the HERB (v2.0) database to validate our findings in drug repositioning. Based on the predicted effects of complex system interventions, we selected genes with softmax scores greater than or equal to 0.7 as potential intervention genes. Using the R package clusterProfiler (v4.9.1), we performed KEGG enrichment analysis and DO enrichment analysis on the potential intervention genes to identify diseases or disease-related terms potentially intervened by each complex system. These disease or disease-related terms were then modularized according to KEGG classifications. We matched the disease or disease-related terms with clinical diseases and observed whether there were relevant clinical trial reports. Sankey diagrams were visualized using the R package ggalluvial (v0.12.5). Gene-level compound synergistic effects finding in NP For the prediction of compound intervention effects and complex system intervention effects, we selected genes from class 0 and 1 with softmax scores greater than or equal to 0.7 as potential upregulated or downregulated genes. Tumor and normal samples of BRCA, LUAD, and LUSC were obtained from the UCSC Xena project 56 , which has normalized the expression matrices of TCGA tumor samples and GTEx normal samples. After processing with DESeq2, differential genes for BRCA, LUAD, and LUSC were obtained. Based on the clinical prognostic information of individual tumor patients from TCGA and the GEPIA2 platform 57 , we analyzed the prognosis differences between high and low expression groups of each gene (grouped by the median gene expression). Code availability The code related to this manuscript has been uploaded to GitHub and set to private. It will be made public after the manuscript is officially submitted for review. Conflict of interest statement The authors state no conflicts of interest or financial support influenced the outcome of this publication. Author contributions S.L. contributed to the conception and design of the work. B.W. developed the concept of the work, contributed to design and implementation of the algorithm and drafted the initial version of the manuscript. P.Y. contributed to conducting experimental validation. B.W., T.Z., and Q.L. contributed to the data collection and preprocessing. Acknowledgements This work was supported by National Administration of Traditional Chinese Medicine (GZY-KJS-2024-03). References ↵ Pham , T. H. , Qiu , Y. , Zeng , J. , Xie , L. & Zhang , P. A deep learning framework for high-throughput mechanism-driven phenotype compound screening and its application to COVID-19 drug repurposing . Nat Mach Intell 3 , 247 – 257 ( 2021 ). doi: 10.1038/s42256-020-00285-9 OpenUrl CrossRef Paggi , J. M. , Pandit , A. & Dror , R. O. The Art and Science of Molecular Docking . Annu Rev Biochem 93 , 389 – 410 ( 2024 ). doi: 10.1146/annurev-biochem-030222-120000 OpenUrl CrossRef Chen , H. et al. Drug target prediction through deep learning functional representation of gene signatures . Nat Commun 15 , 1853 ( 2024 ). doi: 10.1038/s41467-024-46089-y OpenUrl CrossRef PubMed You , Y. et al. Artificial intelligence in cancer target identification and drug discovery . Signal Transduct Target Ther 7 , 156 ( 2022 ). doi: 10.1038/s41392-022-00994-0 OpenUrl CrossRef ↵ Ye , Q. et al. A unified drug-target interaction prediction framework based on knowledge graph and recommendation system . Nat Commun 12 , 6775 ( 2021 ). doi: 10.1038/s41467-021-27137-3 OpenUrl CrossRef PubMed ↵ Liu , Q. et al. Leveraging Network Target Theory for Efficient Prediction of Drug-Disease Interactions: A Transfer Learning Approach . Adv Sci (Weinh) 12 , e2409130 ( 2025 ). doi: 10.1002/advs.202409130 OpenUrl CrossRef Cheng , F. , Kovacs , I. A. & Barabasi , A. L. Network-based prediction of drug combinations . Nat Commun 10 , 1197 ( 2019 ). doi: 10.1038/s41467-019-09186-x OpenUrl CrossRef PubMed ↵ Julkunen , H. et al. Leveraging multi-way interactions for systematic prediction of pre-clinical drug combination effects . Nat Commun 11 , 6136 ( 2020 ). doi: 10.1038/s41467-020-19950-z OpenUrl CrossRef ↵ Liang , X. , Li , H. & Li , S. A novel network pharmacology approach to analyse traditional herbal formulae: the Liu-Wei-Di-Huang pill as a case study . Mol Biosyst 10 , 1014 – 1022 ( 2014 ). doi: 10.1039/c3mb70507b OpenUrl CrossRef PubMed ↵ Yang , K. et al. Heterogeneous network propagation for herb target identification . BMC Med Inform Decis Mak 18 , 17 ( 2018 ). doi: 10.1186/s12911-018-0592-z OpenUrl CrossRef PubMed Duan , P. et al. HTINet2: herb-target prediction via knowledge graph embedding and residual-like graph neural network . Brief Bioinform 25 ( 2024 ). doi: 10.1093/bib/bbae414 OpenUrl CrossRef ↵ Gan , X. et al. Network medicine framework reveals generic herb-symptom effectiveness of traditional Chinese medicine . Sci Adv 9 , eadh0215 ( 2023 ). doi: 10.1126/sciadv.adh0215 OpenUrl CrossRef PubMed ↵ Lin , Y. et al. scJoint integrates atlas-scale single-cell RNA-seq and ATAC-seq data with transfer learning . Nat Biotechnol 40 , 703 – 710 ( 2022 ). doi: 10.1038/s41587-021-01161-6 OpenUrl CrossRef PubMed ↵ Chen , J. et al. Deep transfer learning of cancer drug responses by integrating bulk and single-cell RNA-seq data . Nat Commun 13 , 6494 ( 2022 ). doi: 10.1038/s41467-022-34277-7 OpenUrl CrossRef PubMed ↵ Subramanian , A. et al. A Next Generation Connectivity Map: L1000 Platform and the First 1,000,000 Profiles . Cell 171 , 1437 – 1452 e1417 ( 2017 ). doi: 10.1016/j.cell.2017.10.049 OpenUrl CrossRef PubMed ↵ Zaheer , M. et al. Deep sets . Advances in neural information processing systems 30 ( 2017 ). ↵ Lee , J. et al. in International conference on machine learning . 3744 – 3753 ( PMLR ). ↵ Jeon , M. et al. Transforming L1000 profiles to RNA-seq-like profiles with deep learning . BMC Bioinformatics 23 , 374 ( 2022 ). doi: 10.1186/s12859-022-04895-5 OpenUrl CrossRef PubMed ↵ Szklarczyk , D. et al. The STRING database in 2023: protein-protein association networks and functional enrichment analyses for any sequenced genome of interest . Nucleic Acids Res 51 , D638 – D646 ( 2023 ). doi: 10.1093/nar/gkac1000 OpenUrl CrossRef PubMed ↵ Kim , S. et al. PubChem 2023 update . Nucleic Acids Res 51 , D1373 – D1380 ( 2023 ). doi: 10.1093/nar/gkac956 OpenUrl CrossRef PubMed ↵ Gao , K. et al. HERB 2.0: an updated database integrating clinical and experimental evidence for traditional Chinese medicine . Nucleic Acids Res 53 , D1404 – D1414 ( 2025 ). doi: 10.1093/nar/gkae1037 OpenUrl CrossRef PubMed ↵ Wang , Y. et al. GPX2 suppression of H(2)O(2) stress regulates cervical cancer metastasis and apoptosis via activation of the beta-catenin-WNT pathway . Onco Targets Ther 12 , 6639 – 6651 ( 2019 ). doi: 10.2147/OTT.S208781 OpenUrl CrossRef PubMed ↵ Derakhshan Nazari , M. H. et al. GPX2 and BMP4 as Significant Molecular Alterations in The Lung Adenocarcinoma Progression: Integrated Bioinformatics Analysis . Cell J 24 , 302 – 308 ( 2022 ). doi: 10.22074/cellj.2022.7930 OpenUrl CrossRef PubMed ↵ Ahmed , K. M. et al. Glutathione peroxidase 2 is a metabolic driver of the tumor immune microenvironment and immune checkpoint inhibitor response . J Immunother Cancer 10 ( 2022 ). doi: 10.1136/jitc-2022-004752 OpenUrl Abstract / FREE Full Text ↵ Cao , P. et al. BRMS1L confers anticancer activity in non-small cell lung cancer by transcriptionally inducing a redox imbalance in the GPX2-ROS pathway . Transl Oncol 41 , 101870 ( 2024 ). doi: 10.1016/j.tranon.2023.101870 OpenUrl CrossRef PubMed Huang , H. et al. YAP Suppresses Lung Squamous Cell Carcinoma Progression via Deregulation of the DNp63-GPX2 Axis and ROS Accumulation . Cancer Res 77 , 5769 – 5781 ( 2017 ). doi: 10.1158/0008-5472.CAN-17-0449 OpenUrl Abstract / FREE Full Text ↵ Wang , M. , Chen , X. , Fu , G. & Ge , M. Glutathione peroxidase 2 overexpression promotes malignant progression and cisplatin resistance of KRAS□mutated lung cancer cells . Oncol Rep 48 ( 2022 ). doi: 10.3892/or.2022.8422 OpenUrl CrossRef ↵ Hashinokuchi , A. et al. Clinical and Prognostic Significance of Glutathione Peroxidase 2 in Lung Adenocarcinoma . Ann Surg Oncol 31 , 4822 – 4829 ( 2024 ). doi: 10.1245/s10434-024-15116-z OpenUrl CrossRef PubMed ↵ Vaddavalli , P. L. & Schumacher , B. The p53 network: cellular and systemic DNA damage responses in cancer and aging . Trends Genet 38 , 598 – 612 ( 2022 ). doi: 10.1016/j.tig.2022.02.010 OpenUrl CrossRef PubMed Hu , J. et al. Targeting mutant p53 for cancer therapy: direct and indirect strategies . J Hematol Oncol 14 , 157 ( 2021 ). doi: 10.1186/s13045-021-01169-0 OpenUrl CrossRef PubMed ↵ Hassin , O. & Oren , M. Drugging p53 in cancer: one protein, many targets . Nat Rev Drug Discov 22 , 127 – 144 ( 2023 ). doi: 10.1038/s41573-022-00571-8 OpenUrl CrossRef PubMed ↵ Carneiro , B. A. & El-Deiry , W. S. Targeting apoptosis in cancer therapy . Nat Rev Clin Oncol 17 , 395 – 417 ( 2020 ). doi: 10.1038/s41571-020-0341-y OpenUrl CrossRef PubMed Moyer , A. , Tanaka , K. & Cheng , E. H. Apoptosis in Cancer Biology and Therapy . Annu Rev Pathol 20 , 303 – 328 ( 2025 ). doi: 10.1146/annurev-pathmechdis-051222-115023 OpenUrl CrossRef PubMed ↵ Wong , R. S. Apoptosis in cancer: from pathogenesis to treatment . J Exp Clin Cancer Res 30 , 87 ( 2011 ). doi: 10.1186/1756-9966-30-87 OpenUrl CrossRef PubMed ↵ Farhan , M. et al. FOXO Signaling Pathways as Therapeutic Targets in Cancer . Int J Biol Sci 13 , 815 – 827 ( 2017 ). doi: 10.7150/ijbs.20052 OpenUrl CrossRef PubMed Ma , J. , Matkar , S. , He , X. & Hua , X. FOXO family in regulating cancer and metabolism . Semin Cancer Biol 50 , 32 – 41 ( 2018 ). doi: 10.1016/j.semcancer.2018.01.018 OpenUrl CrossRef PubMed ↵ Jiramongkol , Y. & Lam , E. W. FOXO transcription factor family in cancer and metastasis . Cancer Metastasis Rev 39 , 681 – 709 ( 2020 ). doi: 10.1007/s10555-020-09883-w OpenUrl CrossRef PubMed ↵ Bourquin , C. , Pommier , A. & Hotz , C. Harnessing the immune system to fight cancer with Toll-like receptor and RIG-I-like receptor agonists . Pharmacol Res 154 , 104192 ( 2020 ). doi: 10.1016/j.phrs.2019.03.001 OpenUrl CrossRef PubMed ↵ Khan , A. A. , Khan , Z. & Warnakulasuriya , S. Cancer-associated toll-like receptor modulation and insinuation in infection susceptibility: association or coincidence? Ann Oncol 27 , 984 – 997 ( 2016 ). doi: 10.1093/annonc/mdw053 OpenUrl CrossRef PubMed ↵ Ashley , E. A. Towards precision medicine . Nat Rev Genet 17 , 507 – 522 ( 2016 ). doi: 10.1038/nrg.2016.86 OpenUrl CrossRef PubMed Drevets , W. C. , Wittenberg , G. M. , Bullmore , E. T. & Manji , H. K. Immune targets for therapeutic development in depression: towards precision medicine . Nat Rev Drug Discov 21 , 224 – 244 ( 2022 ). doi: 10.1038/s41573-021-00368-1 OpenUrl CrossRef Mateo , J. et al. Delivering precision oncology to patients with cancer . Nat Med 28 , 658 – 665 ( 2022 ). doi: 10.1038/s41591-022-01717-2 OpenUrl CrossRef PubMed Manzari , M. T. et al. Targeted drug delivery strategies for precision medicines . Nat Rev Mater 6 , 351 – 370 ( 2021 ). doi: 10.1038/s41578-020-00269-6 OpenUrl CrossRef ↵ Denny , J. C. & Collins , F. S. Precision medicine in 2030-seven ways to transform healthcare . Cell 184 , 1415 – 1419 ( 2021 ). doi: 10.1016/j.cell.2021.01.015 OpenUrl CrossRef PubMed ↵ Berlow , N. et al. in Journal of Clinical Oncology . ( LIPPINCOTT WILLIAMS & WILKINS TWO COMMERCE SQ, 2001 MARKET ST, PHILADELPHIA … ). ↵ Zhang , J. et al. Tahoe-100M: A Giga-Scale Single-Cell Perturbation Atlas for Context-Dependent Gene Function and Cellular Modeling . bioRxiv , 2025.2002. 2020.639398 ( 2025 ). ↵ Qi , X. et al. Predicting transcriptional responses to novel chemical perturbations using deep generative model for drug discovery . Nat Commun 15 , 9256 ( 2024 ). doi: 10.1038/s41467-024-53457-1 OpenUrl CrossRef PubMed Li , C. et al. scRank infers drug-responsive cell types from untreated scRNA-seq data using a target-perturbed gene regulatory network . Cell Rep Med 5 , 101568 ( 2024 ). doi: 10.1016/j.xcrm.2024.101568 OpenUrl CrossRef PubMed ↵ Hao , M. et al. Large-scale foundation model on single-cell transcriptomics . Nat Methods 21 , 1481 – 1491 ( 2024 ). doi: 10.1038/s41592-024-02305-7 OpenUrl CrossRef ↵ Qiao , L. et al. Evaluation of the immunomodulatory effects of anti-COVID-19 TCM formulae by multiple virus-related pathways . Signal Transduct Target Ther 6 , 50 ( 2021 ). doi: 10.1038/s41392-021-00475-w OpenUrl CrossRef ↵ Luo , K. et al. Dendrocalamus latiflorus and its component rutin exhibit glucose-lowering activities by inhibiting hepatic glucose production via AKT activation . Acta Pharm Sin B 12 , 2239 – 2251 ( 2022 ). doi: 10.1016/j.apsb.2021.11.017 OpenUrl CrossRef PubMed ↵ Ritchie , M. E. et al. limma powers differential expression analyses for RNA-sequencing and microarray studies . Nucleic Acids Res 43 , e47 ( 2015 ). doi: 10.1093/nar/gkv007 OpenUrl CrossRef PubMed ↵ Love , M. I. , Huber , W. & Anders , S. Moderated estimation of fold change and dispersion for RNA-seq data with DESeq2 . Genome Biol 15 , 550 ( 2014 ). doi: 10.1186/s13059-014-0550-8 OpenUrl CrossRef PubMed ↵ Sun , F.-Y. , Hoffmann , J. , Verma , V. & Tang , J. Infograph: Unsupervised and semi-supervised graph-level representation learning via mutual information maximization . arXiv preprint arxiv: 1908.01000 ( 2019 ). ↵ Wang , L. , Zhang , H. , Xu , W. , Xue , Z. & Wang , Y. Deciphering the protein landscape with ProtFlash, a lightweight language model . Cell Reports Physical Science 4 ( 2023 ). ↵ Goldman , M. J. et al. Visualizing and interpreting cancer genomics data via the Xena platform . Nat Biotechnol 38 , 675 – 678 ( 2020 ). doi: 10.1038/s41587-020-0546-8 OpenUrl CrossRef PubMed ↵ Tang , Z. , Kang , B. , Li , C. , Chen , T. & Zhang , Z. GEPIA2: an enhanced web server for large-scale expression profiling and interactive analysis . Nucleic Acids Res 47 , W556 – W560 ( 2019 ). doi: 10.1093/nar/gkz430 OpenUrl CrossRef PubMed View the discussion thread. Back to top Previous Next Posted April 11, 2025. Download PDF Supplementary Material Email Thank you for your interest in spreading the word about bioRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following Transfer Learning and Permutation-Invariance improving Predicting Genome-wide, Cell-Specific and Directional Interventions Effects of Complex Systems Message Subject (Your Name) has forwarded a page to you from bioRxiv Message Body (Your Name) thought you would like to see this page from the bioRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share Transfer Learning and Permutation-Invariance improving Predicting Genome-wide, Cell-Specific and Directional Interventions Effects of Complex Systems Boyang Wang , Pan Boyu , Tingyu Zhang , Qingyuan Liu , Shao Li bioRxiv 2025.04.07.647536; doi: https://doi.org/10.1101/2025.04.07.647536 Share This Article: Copy Citation Tools Transfer Learning and Permutation-Invariance improving Predicting Genome-wide, Cell-Specific and Directional Interventions Effects of Complex Systems Boyang Wang , Pan Boyu , Tingyu Zhang , Qingyuan Liu , Shao Li bioRxiv 2025.04.07.647536; doi: https://doi.org/10.1101/2025.04.07.647536 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Systems Biology Subject Areas All Articles Animal Behavior and Cognition (7636) Biochemistry (17704) Bioengineering (13897) Bioinformatics (41963) Biophysics (21460) Cancer Biology (18598) Cell Biology (25525) Clinical Trials (138) Developmental Biology (13383) Ecology (19908) Epidemiology (2067) Evolutionary Biology (24325) Genetics (15613) Genomics (22512) Immunology (17738) Microbiology (40422) Molecular Biology (17190) Neuroscience (88634) Paleontology (667) Pathology (2835) Pharmacology and Toxicology (4825) Physiology (7645) Plant Biology (15158) Scientific Communication and Education (2046) Synthetic Biology (4302) Systems Biology (9825) Zoology (2271)

Text is read by the "Ask this paper" AI Q&A widget below. Extraction quality varies by source — PMC NXML preserves structure cleanly, OA-HTML may include some navigation residue, and OA-PDF can have broken hyphenation. The publisher copy (via DOI) is the canonical version.

My notes (saved in your browser only)

Ask this paper AI returns verbatim quotes from the full text · source: preprint-html

Answers must be backed by verbatim quotes from this paper's full text. Hallucinated quotes are dropped automatically; if no verbatim passage answers the question, we say so. How this works

Citation neighborhood (no data yet)

We don't have any in-corpus citations linked to this paper yet. This is a recent paper (2025) — citers typically take a year or two to land, and the OpenAlex reference graph may still be filling in.

Source provenance

europepmc
last seen: 2026-05-20T01:45:00.602351+00:00
unpaywall
last seen: 2026-08-10T06:43:36.850308+00:00