Transfer learning applied in predicting small molecule bioactivity

preprint OA: closed
📄 Open PDF Full text JSON View at publisher

Abstract

Despite over a half-century of effort by computational chemists, developing accurate empirical QSAR (Quantitative Structure-Activity Relationships) models for predicting bioactivity directly from chemical structure has remained elusive. The difficulties have been especially pronounced for virtual screening, finding new active compounds substantially different from the known chemical matter used to train the models. Recent breakthroughs have been achieved by employing transfer of learning across huge numbers of bioactivity assays, greatly increasing the amount and diversity of chemical and biological information that informs each model. An early example was Profile-QSAR (pQSAR), a 2-level stacked model, where level-2 PLS (Partial Least Squares) models characterize compounds by their profile of bioactivity predictions from individual level-1 random forest regression QSAR models built on up to 10,000+ other assays. This study introduces metaNN, a meta-learner that trains deep neural networks (DNN) for each individual assay initialised from a well-generalized consensus DNN optimized across all assays. Comparison of the results suggested that while Profile-QSAR and metaNN perform similarly overall, metaNN works slightly better for smaller assays which were well-predicted by the consensus DNN; whereas pQSAR struggled more with smaller assays, due to the large number of level-1 models but was less sensitive to similarity to an overall consensus. An ensemble average of both methods combined the strengths of each, working better than either alone. The similar performance of the 2 largely orthogonal algorithms raises questions about whether we are approaching a limit of prediction accuracy in transfer learning, for this application scenario.
Full text 42,533 characters · extracted from preprint-html · click to expand
Transfer learning applied in predicting small molecule bioactivity | bioRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-M677548'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search New Results Transfer learning applied in predicting small molecule bioactivity Li Tian , Hugh Hongkai Huang , Eric Martin , Judy Ying Wei , Shaoyong Xu , Junzhou Huang doi: https://doi.org/10.1101/2025.01.09.632140 Li Tian 1 Laboratory for Synthetic Chemistry and Chemical Biology Limited (LSCCB) , Hong Kong SAR Find this author on Google Scholar Find this author on PubMed Search for this author on this site For correspondence: tianli{at}hksccb.hk ying.wei{at}zju.edu.cn Hugh Hongkai Huang 2 Tencent AI Lab , Shenzhen, P.R.China Find this author on Google Scholar Find this author on PubMed Search for this author on this site Eric Martin 3 Novartis Institute for Biomedical Research , California, United States Find this author on Google Scholar Find this author on PubMed Search for this author on this site Judy Ying Wei 4 College of Computer Science and Technology , Zhejiang University, P.R.China Find this author on Google Scholar Find this author on PubMed Search for this author on this site For correspondence: tianli{at}hksccb.hk ying.wei{at}zju.edu.cn Shaoyong Xu 2 Tencent AI Lab , Shenzhen, P.R.China Find this author on Google Scholar Find this author on PubMed Search for this author on this site Junzhou Huang 5 Computer Science and Engineering department at the University of Texas at Arlington , United States Find this author on Google Scholar Find this author on PubMed Search for this author on this site Abstract Full Text Info/History Metrics Preview PDF Abstract Despite over a half-century of effort by computational chemists, developing accurate empirical QSAR (Quantitative Structure-Activity Relationships) models for predicting bioactivity directly from chemical structure has remained elusive. The difficulties have been especially pronounced for virtual screening, finding new active compounds substantially different from the known chemical matter used to train the models. Recent breakthroughs have been achieved by employing transfer of learning across huge numbers of bioactivity assays, greatly increasing the amount and diversity of chemical and biological information that informs each model. An early example was Profile-QSAR (pQSAR), a 2-level stacked model, where level-2 PLS (Partial Least Squares) models characterize compounds by their profile of bioactivity predictions from individual level-1 random forest regression QSAR models built on up to 10,000+ other assays. This study introduces metaNN, a meta-learner that trains deep neural networks (DNN) for each individual assay initialised from a well-generalized consensus DNN optimized across all assays. Comparison of the results suggested that while Profile-QSAR and metaNN perform similarly overall, metaNN works slightly better for smaller assays which were well-predicted by the consensus DNN; whereas pQSAR struggled more with smaller assays, due to the large number of level-1 models but was less sensitive to similarity to an overall consensus. An ensemble average of both methods combined the strengths of each, working better than either alone. The similar performance of the 2 largely orthogonal algorithms raises questions about whether we are approaching a limit of prediction accuracy in transfer learning, for this application scenario. Introduction Recently, significant interest has been focused on applying artificial intelligence (AI) algorithms to predict small molecule bioactivity and protein-binding affinity. Structure-based methods are supposed to learn the physics of interactions between the small molecule ligands and the protein pockets, and thus provide accurate prediction with no training data. However, for many drug discovery projects, the target protein or even biological mechanism is not completely clear, or a high-resolution structure of the binding complex may not be available. These factors limit the applicability of structure-based prediction methods. Furthermore, physics-based methods typically ignore many important factors that influence activity: concentrations of cofactors, post-translational modifications, allosteric modulations such as a protein complex formation, identity of second substrates (e.g., the particular peptide substrate for a protein kinases ATP-site inhibitor), and all the complexity of cellular assays. These factors complicate the states of the protein and limit the accuracy or sometimes even the relevance of the structure-based prediction. One alternative approach starts from the compounds’ chemical structures and bioactivity data, i.e., the “ligand-based” data sets. These data sets are much more widely available, including functional and cellular assays. These data sets can be used to build a ligand-based model for activity prediction without prior knowledge of the target protein or even the biological mechanism. How to best leverage this significantly bigger data pool and convert these data sets to knowledge and predictive models, is of great interest and has wide application in drug discovery. Chemical informaticians have been building ligand-based models for decades, from very traditional QSAR models based on physicochemical properties to random forest models based on molecular fingerprints. Some of these models may achieve impressive prediction accuracy on random test sets, yet fail to demonstrate comparable accuracy in realistic applications such as virtual screening to identify novel hit compounds or off-target prediction. Although the reasons behind this limitation may be complicated, the “stretched” extrapolation of the “local model” trained from small data sets to much larger chemical space for prediction, may account for the major contribution. On the other hand, the limitations rooted from sparse known data have also been well studied by machine learning (ML) scientists: how to effectively learn from a very small amount of data and apply to larger prediction sets. Recently, several attempts have been made leveraging knowledge transfer to remedy the lack of bioactivity data, and thus more effectively train predictive models. These studies demonstrate two major strands of algorithms: transfer learning ( 1 ) and multitask learning( 2 ). Transfer learning aims to improve the bioactivity prediction accuracy of a target assay via transferring the knowledge learned from previous source assays; whereas multitask learning jointly learns all the assays together, without differentiating source and target assays. Girschick et al. ( 3 ) proposed to learn a distance measure that reflects the similarities between instances and transfer that measure to the target assy. Profile-QSAR (pQSAR) [4] uses a 2-level stacked model ( 4 ), where a compound is characterized by its profile of bioactivity predictions from QSAR models built on many other assays. Similar to pQSAR, Alchemite also uses stacked models, but is based on neural networks rather than older random forest and PLS algorithms.( 5 ) Multi-target QSAR models were built on a subset of the human kinome with taxonomy-based multitask methods proposed in a recent study( 6 ). Recent advances in deep neural networks have motivated deep transfer learning (DTL) and deep multitask learning (DMTL) studies, where deep neural networks with greater expressiveness are taken as predictive models. The most prevalent DTL strategy is fine-tuning--a neural network that predicts an activity or property is pre-trained on all source assays and fine-tuned to the target assay ( 7 , 8 ). Another state-of-the-art method( 9 ) learns an attention function that measures the similarity between instances and transfers the function to the target assay, which shares the spirit of Girschic’s work ( 3 ). Despite its superiority over single-assay neural networks, DTL ignores the difference between different assays. Instead, DMTL learns different predictors for different assays, while sharing the feature extractor at low-level layers among all assays. By virtue of the high predictive capacity of deep neural networks, and knowledge sharing between assays, both Dahl et al.( 10 ) and Ramsundar et al.( 11 ) have reported that DMTL outperforms single-assay neural networks. To further improve the effectiveness of DMTL with fine-grained knowledge sharing, binding domain sequence similarity( 12 ) and amino-acid sequence similarity( 13 ) have been leveraged to model the relationship between targets. The downside of DMTL lies in the risk of an assay-specific predictor being overfitted to extremely scarce bioactivity data. In this paper, we seek an approach to address the problems of both DTL and DMTL, where a neural network is globally shared across all assays, while it is explicitly trained to adapt to each in only a few steps with low risk of overfitting. Transferring and adapting the global neural network quickly trains the model for any specific target assay. Here we describe a new meta-learning algorithm, named metaNN, discuss the similarity and dissimilarity in predictions between metaNN and pQSAR, analyze the results in detail, and show that a combination or ensemble method achieves even higher accuracy. 2. MetaNN algorithm Our proposed algorithm, metaNN, is inspired by the recent success of gradient-based meta-learning( 14 ), which leverages the meta-knowledge learned from previous tasks to facilitate the learning of a novel task. This paradigm creates task-specific neural networks, while requiring minimum learning for each individual task, thus minimizing the risk of overfitting given limited training data. Figure 1 presents an overview of the workflow of gradient-based meta-learning algorithms for the problem of bioactivity prediction: 1) we train a neural network whose initialization for the weights is shared across all n source assays; 2) for a source assay, e.g. source assay 1, we build the initialized neural network for its training-set compounds on bioactivity values for its assay; 3) we evaluate the adapted neural network on test-set compounds of each source assay, so that the error rates from all source assays are averaged and back-propagated to improve the generalization capability of the initialization; 4) the learned initialization, called meta-initialization, is transferred to facilitate the training of the target assay. Download figure Open in new tab Figure 1 Pictorial illustration of the workflow of the metaNN algorithm. Mathematically, we denote each assay as T i with its training set as and testing set as denotes the j -th compound for the i -th assay, and is its pIC 50 activity value. and are the numbers of training and testing compounds in the i -th assay, respectively. Suppose that the neural network is represented by f W , where W denotes all weights for the neural network. We evaluate the performance of the neural network with the squared loss between the predicted and experimental bioactivity values, i.e., . W 0 is the meta-initialization for the neural network f W which we aim to learn. The workflow of gradient-based meta-learning, therefore, boils down to the following two alternating optimization problems. The first equation corresponds to step 2) where all training compounds are used to adapt the meta-initialization W 0 to be assay-specific via several steps of gradient descent, while the second equation for step 3) learns a well-generalized meta-initialization such that the error rates of adapted neural networks starting from W 0 on test compounds of all source assays are minimized. µ and α are the learning rates for updating W i and W 0 , respectively. Provided with a target assay T t , the weights for its specific neural network are learned in a manner similar to the first equation, i.e., . Considering the high generalization capacity of the optimized meta-initialization W 0 for source assays, fast learning and satisfactory predictive accuracy of W t on the target assay can be anticipated. In the first set of calculations, this pipeline of gradient-based meta-learning did not achieve the accuracies we expected on the target assay. This implies that the learned meta-initialization does not generalize well. A well-qualified meta-initialization W 0 should be consistently under-performing on either the training or the test set, leaving sufficient room for training compounds to learn the assay-specific weights W i that suffice to achieve high prediction accuracies for both sets. Empirical results in Figure 2 show that even without adaptation to each individual assay, the meta-initialization W 0 achieves extremely high accuracies on the test sets of all source assays, but quite low accuracies on the training sets. This suggests that the meta-initialization W 0 simply memorizes the test sets of all source assays, rather than being a well-generalized starting point for the neural network to be optimized for each source assay through its training set. Download figure Open in new tab Figure 2 Inconsistent behaviors of W 0 on training sets and testing sets. We addressed this issue by producing more data to evaluate and update the meta-initialization in Eqn. ( 2 ). We list three desiderata for producing such a set of extra data: 1) the incorporation of such data would encourage consistent predictive performances of the meta-initialization across training and testing sets; 2) the data should contain more than the training set, as the training set has been used for training and fails to evaluate the generalization capability; 3) the data should contain more than the testing set, otherwise the memorization problem persists. Inspired by( 15 ), we produce extra data by mixing both training and testing sets, where we mix-up not only hidden representations of compounds, but also their activity values. Specifically, for each randomly selected pair of training a compound and a testing compound in the i -th source assay, we produce a new compound as where f l (·) is the l -th layer hidden representation by the neural network, and λ ∼ Beta ( α, β ) is a randomly sampled coefficient from the Beta distribution. By sampling different pairs of training and testing compounds and sampling various values of λ , we produce a new collection of data that can be joined with the testing set to update the meta-initialization in Eqn. ( 2 ). With the above-described algorithm, we used the same compound featurization as proposed by the pQSAR developers -- Morgan radius 2 substructure fingerprints of 1024 bits from RDkit v2018.03, trained the models, predicted bioactivities and evaluated prediction performance. Two data sets had been downloaded from ChEMBL and published in two earlier pQSAR papers: namely IC 50 s for 159 kinase assays( 16 ) and for 4276 assays from all diverse families( 17 ). The assay statistics of these two data sets are summarized in Table 1 . The 159 kinase data set is smaller--focused on the most studied protein family in drug discovery. The 4276 assay set includes assays from 5 protein families, including kinases, proteases, GPCR, ion channels, nuclear hormone receptors, as well as phenotypic assays without clear targets registered. These two data sets both cover different assay formats, both biochemical and cellular, and the data maintains a wide dynamic range as a good reflection of real-life compound optimization projects. View this table: View inline View popup Download powerpoint Table 1. Summary statistics of the two datasets we evaluated. In our calculations, we chose to use the same two public domain data sets for direct comparisons to pQSAR and easier access for other researchers in this field. When evaluating the model performance, we mainly use squared Pearson correlation coefficient between the experimental and predicted pIC 50 values for the compounds in the external test sets, noted as R 2 ext . A detailed discussion on why squared Pearson correlation coefficient was preferred for virtual screening use-cases, and its comparison to other evaluation metrics such as RMSE and MAE, was presented in earlier publications( 18 ). We also continued using the realistic training/test set splits discussed in the earlier papers( 16 ), and used the exact published training test sets. When evaluating model quality across many assays, the median R 2 ext , and the number of models with R 2 ext >= 0.3, were used as the two most important indicators. For metaNN calculation, the neural network for predicting pIC 50 activity was a two-layer Multi-layer Perceptron (MLP) with 500 hidden neurons in each layer. Also, each fully connected layer was followed by a batch normalization layer and a non-linear activation of leakyReLU (negative slope is 0.01). We updated the meta-initialization in a batch-wise manner, with each batch consisting of 8 randomly selected source assays, i.e., N s = 8. The learning rates to update the meta-initialization (i.e., α ) and to learn assay-specific weights (i.e., µ ) were 0.001 and 0.01, respectively. We set the Beta distribution to sample values of λ as Beta (0.5,0.5), i.e., α = β = 0.5. We iteratively ran 50 epochs to update the meta-initialization, with each epoch including 500 iterations, while we took only 5 gradient steps to quickly learn assay-specific weights for the neural network in Eqn. ( 1 ). 3. Results As the pQSAR code was not yet published, we first programmed and reproduced the pQSAR 2.0 result, and then compared its predictions with our methods. We modeled 4266 of the 4276 assays, having removed 10 assays whose test set compounds all have the same experimentally measured values. Tables 2 and 3 summarize the results. For both the 159 and the 4266 data sets, metaNN shows very similar overall prediction performance as pQSAR 2.0, evaluated by various summarizing statistics of R 2 ext . For the 4266 data set, the median R 2 ext reported in the original publication of pQSAR 2.0 for the 4266 assays was at 0.30. In reduced profile pQSAR, the authors only used the random forest models whose predicted pIC 50 s correlate (r 2 ) with the experimental measurements in the training sets above select thresholds. Models were built using three r 2 thresholds: which corresponds to the full profile of all random forest models, and . In our reproduced pQSAR 2.0 calculation, these 3 thresholds, 0.0, 0.05 and 0.2, provide R 2 ext median at 0.26, 0.30, and 0.33, respectively. In max2 pQSAR( 17 ), the authors used the test set R 2 ext to select the best of the 3 models. The improvement brought by this model selection, (here named as max3) in Table 2 , boosted the median R 2 ext over the 0.2 threshold model from 0.33 to 0.41, and added 357 more successful models (R 2 ext >0.3). This reproduces the reported results from the publication. However, using the test set to choose among 3 models introduces a multiple-hypothesis testing issue, i.e. tripling the opportunities for chance correlations. The authors repeated the calculations with Y-scrambled data sets to demonstrate that even with the model selection strategy there were very few successful models simply achieved by chance. We likewise only observed an increase in the y-scrambled median R 2 ext from 0.02 to 0.05. The number of y-scrambled models with R 2 ext >0.3 only increased by 107. We thus concluded that the published performance is very well reproduced and similar improvement over chance from model selection was observed. View this table: View inline View popup Download powerpoint Table 2 Summary of correlations R 2 ext between experiment and prediction on 4266 ChEMBL assays for various multitask models. View this table: View inline View popup Download powerpoint Table 3 Performance comparison for the 159 data set. Here we also created an ensemble average of these 3 models, to adapt the benefit from multiple models, without the complication of multiple hypotheses. The simple equal weight averaged ensemble of these 3 models provided a marginally improved median R 2 ext of 0.34 vs. 0.33 with the fixed correlation threshold . This ensemble did not change the median R 2 ext (0.02) on the y-scrambled data set. In the following discussion and comparison with metaNN predictions, we use the predictions of this ensemble of the 3 pQSAR models as the “pQSAR result”. For the same 4266 assays, the median R 2 ext for metaNN is 0.33, almost exactly the same as obtained from the pQSAR ensemble (0.34). Similarly, metaNN had the same R 2 ext in the y-scrambled data set of 0.02. The rank order plots in Figure 3 display R 2 ext rank-order curves for metaNN and pQSAR, as well as the y-scrambled results. The overlay in the plots of metaNN and ensemble pQSAR demonstrates their comparable performance. Download figure Open in new tab Figure 3 Left: pQSAR result; right: metaNN compared to ensemble pQSAR, and an ensemble of both methods. We divided the 4266 assays into sub-groups to compare pQSAR and metaNN based on 4 criteria: protein families, assay size, assay format and activity distribution. Overall, the performance was comparable for different sub-groups. While each algorithm performed better for some types of assays, the difference was not large. As shown in Figure 4 , each assay was assigned to 1 of 3 bins: equal for R 2 ext difference between metaNN and pQSAR ensemble between -0.05 and 0.05; metaNN if the difference is > 0.05 and pQSAR if < -0.05. For many cases, the 3 bins have very similar numbers of assays. MetaNN performs slightly better for GPCR and cellular assays; ensemble pQSAR trains slightly stronger models for kinases. Both methods show a clear bimodal distribution with mean pIC 50 in Figure 4d . However, pQSAR ensemble works better for assays with lower mean pIC 50 around 5.0, whereas metaNN performs slightly better for assays with more potent compounds, peaking with mean pIC 50 at 7.0. The dynamic range of pIC 50 values as characterized by pIC 50 standard deviation, and chemical diversity, as characterized by Bemis-Murcko scaffold count and cpd/scaffold count, were two additional factors used in comparison, but no substantial differences were observed. Download figure Open in new tab Figure 4. Prediction performance compared for metaNN and ensemble profile-QSAR, plotted for different sub-groups of assays. We note the R2ext difference between metaNN and pQSAR ensemble between -0.05 and 0.05 as equal and shown in the figures as gray bars; metaNN if the difference is > 0.05 and shown in the figure as green bars; and pQSAR if < -0.05 and shown in the figure as blue bars. a. Assay format and assay type from as extracted from ChEMBL database; b, A is the shortcut for ADMET assays, B for binding assays, F for functional assays, and T is for toxicity assays, c. Protein family: slightly different for kinase, protease and GPCR, d. Compound count: metaNN shows better performance for smaller assays, e. Mean pIC50 distribution, f. pIC50 standard deviation within each assay One interesting observation is from the bins based on assay size. MetaNN gave overall higher R 2 ext for smaller assays, especially those with no more than 100 compounds, while pQSAR ensemble performed better for the assays tested on 100-1000 compounds. This aligns with the expected strengths of metaNN and pQSAR. The high generalization capacity of the learned meta-initialization by metaNN enables the neural network to be quickly adapted to a target assay in only a few (i.e., 5 in our experiments) gradient steps with even a very limited number of training compounds. Profile-QSAR relies on more training compounds to build the PLS model, especially when the number of random forest models is high. 4. Discussion We will first discuss appropriate measures of model “usefulness” for real-life drug discovery; then we will discuss the combination of different algorithms, the improved predictive ability from the combination, and whether there is a practical upper limit to prediction accuracy. 1). Realistic test set splitting and corresponding evaluating factors Many machine-learning applications assume the training set is a fair sampling of the prediction set. However, drug-discovery teams are most interested in predictions for novel chemical matter very different from the previousl studied compounds used to train the models. A special procedure was therefore designed to split the whole data set into training and test sets specifically designed to mirror such “virtual screening”. Described in an earlier publication on pQSAR ( 16 ), the compounds from each assay are clustered based on their Morgan 2 chemical fingerprints. Models are trained on the 75% of compounds from the largest clusters and tested on the remaining small clusters and singletons. The test sets from this procedure are much less like the training set than the widely used random test sets. They were shown to mimic the novelty of compounds selected for testing in 18 real virtual screening projects, and thus are called the “realistic test sets”. When the performance of the models trained on this training set are evaluated against this test set, R 2 ext should realistically indicate model performance in actual drug-discovery applications. In contrast, using random test sets dramatically overestimates prediction performance in real applications, which badly misleads project teams. In real practice, training and test sets are subsequently combined, and a final model is trained on all the available bioactivities, which is then used to predict activity of new compounds with no experimental data. Still, assessing performance on the realistic test set is not just for comparing modeling methodologies. Project teams need these results to assess the expected accuracy of the models in practice--to know whether and how each model can best be employed in their projects. E.g., with a poorer model, screen a large number of predicted active compounds at a single concentration, followed by dose-response refinement on the experimentally confirmed hits, vs. with a better model, to save time and costs by immediately screening a smaller number of predicted actives with the dose-response assay. Similarly, consider model-quality in deciding which predicted bioactivities to follow up in off-target or mechanism-of-action calculations, to balance the probability of success with the costs of pursuing false-positives from virtually screening a large panel of models. 2). MetaNN and pQSAR can be profitably combined As shown in the Results section, sometimes pQSAR and sometimes metaNN performs better for different assay sub-groups. We suggested that pQSAR can fish out complex combinations of relevant assays through the 2 nd level PLS modeling, but suffers from the curse of dimensionality, especially for smaller assays. MetaNN benefits more if the target assay aligns significantly with the consensus parameters. Since there is a theoretical reason for expecting one or the other to give better predictions for different assay types, we hypothesized that even better predictions might be achieved by a consensus of these very different algorithms. We tested two strategies for combining the predictions from 4 models (pQ_0.0, pQ_0.05, pQ_0.2 and metaNN): either by averaging the predictions, named “ensemble_all”, or by selecting the single model with the highest R 2 ext on the realistic test set, named “max4”. Using the test set to choose among 4 models again raises the multiple-hypothesis problem. Note that the size of the problem increases with the number of hypotheses. An extensive optimization, such as a grid search over many parameters, would lead to extensive chance correlations. A simple selection among 4 models might produce few. There are numerous theoretical approaches and discussions around how to adjust parametric statistical measures for multiple hypotheses, but little consensus on the best practice. An alternative is to numerically measure the false discovery rate by repeating the calculation with scrambled pIC 50 s, as was done in the earlier pQSAR publication. Both consensus approaches improved correlation with the test set more using real data than with Y-scrambled data, demonstrating the benefit of combining the methods. Median R 2 ext using Max4 increased from between 0.26 and 0.33 for the 4 individual models, to 0.47 for selecting the best of the 4 for each assay, an improvement of from 0.14 to 0.21. Y-scrambled R 2 ext also increased, but only from 0.02 for the individual models to 0.07 for Max4. Similarly, the number of additional models with R 2 ext >0.3 found by max4 increased by 611 over the best single model (pQSAR using threshold ), whereas the number of additional Y-scrambled Max4 models with R 2 ext >0.3 only increased by 159, for a net gain of 452 “successful” models. These improvements over chance support the conjecture that pQSAR and metaNN, although overall similar in individual performance, can be profitably combined. For the “ensemble_all” consensus of averaging the 4 models, where no test set-based choices are made, the Y-scrambled median R 2 ext remained at 0.02 as expected. With the real data, median R 2 ext increased by 0.06 from 0.33 for the best single model to 0.39 for the ensemble average of all 4. Ensemble_all built 256 more models with R 2 ext >0.3 than the best single model, also with virtually no change in Y-scrambled results. Thus, Max4 explained additional 3% of the variance and built 196 additional successful models vs. ensemble_all. This improvement is modest but significant, since it summarizes results of over 4 thousand models. An additional advantage of Max4 is that predictions on large compound sets are faster, since ensemble_all requires 4 models per assay, and always includes pQSAR_0.0, by far the slowest for prediction. In summary, both consensus approaches were improvements over a single model. The choice between ensemble_all and Max4 comes down to speed, accuracy and risk tolerance. The end-user must ultimately decide if the improved speed, larger number of successful models and higher average prediction accuracy of Max4 over the ensemble average is worth the slightly higher chance of being misled by a chance correlation. 3). Is there an upper limit of prediction accuracy? MetaNN and pQSAR individually show very similar prediction performance calculated by R 2 ext , for the 4266 data sets. We have also compared the models trained on the set of 4266 diverse assays to those from the much smaller and more homogeneous set of 159 kinase assays. We again calculated the 3 pQSAR models using thresholds of 0.0, 0.05 and 0.2, pQSAR_max3, pQSAR ensemble, metaNN, Max4, and ensemble_all consensus models, along with their y-scrambled controls. Table 3 summarizes the performance statistics. As expected, with much more homogeneous data, the transfer of learning, and thus the models for these kinase assays, are overall much stronger. The median R 2 ext values are all above 0.45. The results also maintain the same trends as the 4266 assay results discussed above: pQSAR ensemble is comparable to metaNN, both with median R 2 ext of 0.49-0.50, max4 and ensemble_all of metaNN and pQSAR further improve the median by 0.04, while the Y-scrambled result was almost unchanged. Although the trends are the same, the marginal improvement was significantly smaller. One underlying reason might be that the chemical diversity involved in the data set is much smaller. Another possible reason is that all targets share similar ATP binding sites, so all assays are correlated and contribute signal for pQSAR, and all are near the consensus model for metaNN. For these well-studied targets, both methods produced strong models for a much larger portion of the assays, with up to 88% of assays producing “successful” models with R 2 ext >0.3. This raises the question: would there be an upper bound of this prediction accuracy, and if so, how high is it? 5. Conclusion We presented a novel transfer-learning algorithm, metaNN, and reported its prediction performance on two public-domain bioactivity data sets. The performance was found to be surprisingly comparable to and even marginally better than the earlier pQSAR algorithm. Besides, it enjoys greater storage efficiency; a consensus DNN model is stored only instead of all previous random forest models required by pQSAR. We demonstrated that combining these two very different algorithms further improves the predictions: a 27% increase of median R 2 ext , and successful models for another 10% of the >4000 assays. We anticipate these newly developed algorithms will have wide applications in drug discovery. Future directions include how incorporating assay or target meta-data might further improve the models, and might also help to elucidate the relationships among experimental assays, e.g. mechanisms of action for the phenotypica assays. There are many other multitask algorithms applicable to ligand-based bioactivity prediction: alchemite, multitask fully-connected deep neural networks, proteochemometrics and content-based filtering are a few. The authors are working with collaborators to analyze results from these methods for a more extensive analysis. Acknowledgements Dr. Li Tian acknowledges “Laboratory for Synthetic Chemistry and Chemical Biology” under the Health@InnoHK Program launched by Innovation and Technology Commission, The Government of Hong Kong Special Administrative Region of the People’s Republic of China, for the funding support of this research. References 1. ↵ S. J. Pan , Q. Yang , A Survey on Transfer Learning . IEEE Transactions on Knowledge and Data Engineering 22 , 1345 – 1359 ( 2010 ). OpenUrl CrossRef 2. ↵ Y. Zhang , Q. Yang , A Survey on Multitask Learning . IEEE Transactions on Knowledge and Data Engineering 34 , 5586 – 5609 ( 2022 ). OpenUrl CrossRef 3. ↵ T. Girschick , U. Rückert , S. Kramer , Adapted Transfer of Distance Measures for Quantitative Structure-Activity Relationships and Data-Driven Selection of Source Datasets . The Computer Journal 56 , 274 – 288 ( 2013 ). OpenUrl CrossRef 4. ↵ I. Kuzborskij , F. Orabona , B. Caputo , in Proceedings of the IEEE conference on computer vision and pattern recognition . ( 2013 ), pp. 3358 – 3365 . 5. ↵ T. M. Whitehead , B. W. J. Irwin , P. Hunt , M. D. Segall , G. J. Conduit , Imputation of Assay Bioactivity Data Using Deep Learning . Journal of Chemical Information and Modeling 59 , 1197 – 1204 ( 2019 ). OpenUrl CrossRef PubMed 6. ↵ L. Rosenbaum , A. Dörr , M. R. Bauer , F. M. Boeckler , A. Zell , Inferring multi-target QSAR models with taxonomy-based multitask learning . Journal of Cheminformatics 5 , 33 ( 2013 ). OpenUrl CrossRef PubMed 7. ↵ W. Hu , B. Liu , J. Gomes , M. Zitnik , P. Liang , V. Pande , J. Leskovec , Strategies for pre-training graph neural networks . arXiv preprint arXiv: 1905.12265 , ( 2019 ). 8. ↵ X. Li , D. Fourches , Inductive transfer learning for molecular activity prediction: Next-Gen QSAR Models with MolPMoFiT . Journal of Cheminformatics 12 , 27 ( 2020 ). OpenUrl CrossRef PubMed 9. ↵ H. Altae-Tran , B. Ramsundar , A. S. Pappu , V. Pande , Low Data Drug Discovery with One-Shot Learning . ACS Central Science 3 , 283 – 293 ( 2017 ). OpenUrl CrossRef PubMed 10. ↵ G. E. Dahl , N. Jaitly , R. Salakhutdinov , Multitask neural networks for QSAR predictions . arXiv preprint arXiv: 1406.1231 , ( 2014 ). 11. ↵ B. Ramsundar , S. Kearnes , P. Riley , D. Webster , D. Konerding , V. Pande , Massively multitask networks for drug discovery . arXiv preprint arXiv: 1502.02072 , ( 2015 ). 12. ↵ K. Lee , D. Kim , In-silico molecular binding prediction for human drug targets using deep neural multitask learning . Genes 10 , 906 ( 2019 ). OpenUrl CrossRef 13. ↵ N. Sadawi , I. Olier , J. Vanschoren , J. N. Van Rijn , J. Besnard , R. Bickerton , C. Grosan , L. Soldatova , R. D. King , Multitask learning with a natural metric for quantitative structure activity relationship learning . Journal of Cheminformatics 11 , 1 – 13 ( 2019 ). OpenUrl CrossRef PubMed 14. ↵ C. Finn , P. Abbeel , S. Levine , in International conference on machine learning . ( PMLR , 2017 ), pp. 1126 – 1135 . 15. ↵ V. Verma , A. Lamb , C. Beckham , A. Najafi , I. Mitliagkas , D. Lopez-Paz , Y. Bengio , in International conference on machine learning . ( PMLR , 2019 ), pp. 6438 – 6447 . 16. ↵ E. J. Martin , V. R. Polyakov , L. Tian , R. C. Perez , Profile-QSAR 2.0: kinase virtual screening accuracy comparable to four-concentration IC50s for realistically novel compounds . Journal of chemical information and modeling 57 , 2077 – 2088 ( 2017 ). OpenUrl CrossRef PubMed 17. ↵ E. J. Martin , V. R. Polyakov , X.-W. Zhu , L. Tian , P. Mukherjee , X. Liu , All-assay-max2 pqsar: Activity predictions as accurate as four-concentration ic50s for 8558 novartis assays . Journal of chemical information and modeling 59 , 4450 – 4459 ( 2019 ). OpenUrl CrossRef PubMed 18. ↵ E. Martin , P. Mukherjee , D. Sullivan , J. Jansen , Profile-QSAR: a novel meta-QSAR method that combines activities across the kinase family to accurately predict affinity, selectivity, and cellular activity . Journal of chemical information and modeling 51 , 1942 – 1956 ( 2011 ). OpenUrl CrossRef PubMed View the discussion thread. Back to top Previous Next Posted January 10, 2025. Download PDF Email Thank you for your interest in spreading the word about bioRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following Transfer learning applied in predicting small molecule bioactivity Message Subject (Your Name) has forwarded a page to you from bioRxiv Message Body (Your Name) thought you would like to see this page from the bioRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share Transfer learning applied in predicting small molecule bioactivity Li Tian , Hugh Hongkai Huang , Eric Martin , Judy Ying Wei , Shaoyong Xu , Junzhou Huang bioRxiv 2025.01.09.632140; doi: https://doi.org/10.1101/2025.01.09.632140 Share This Article: Copy Citation Tools Transfer learning applied in predicting small molecule bioactivity Li Tian , Hugh Hongkai Huang , Eric Martin , Judy Ying Wei , Shaoyong Xu , Junzhou Huang bioRxiv 2025.01.09.632140; doi: https://doi.org/10.1101/2025.01.09.632140 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Bioinformatics Subject Areas All Articles Animal Behavior and Cognition (7622) Biochemistry (17648) Bioengineering (13871) Bioinformatics (41880) Biophysics (21423) Cancer Biology (18561) Cell Biology (25461) Clinical Trials (138) Developmental Biology (13364) Ecology (19866) Epidemiology (2067) Evolutionary Biology (24290) Genetics (15590) Genomics (22475) Immunology (17713) Microbiology (40328) Molecular Biology (17148) Neuroscience (88473) Paleontology (666) Pathology (2827) Pharmacology and Toxicology (4816) Physiology (7635) Plant Biology (15114) Scientific Communication and Education (2044) Synthetic Biology (4286) Systems Biology (9815) Zoology (2268)

Text is read by the "Ask this paper" AI Q&A widget below. Extraction quality varies by source — PMC NXML preserves structure cleanly, OA-HTML may include some navigation residue, and OA-PDF can have broken hyphenation. The publisher copy (via DOI) is the canonical version.

My notes (saved in your browser only)

Ask this paper AI returns verbatim quotes from the full text · source: preprint-html

Answers must be backed by verbatim quotes from this paper's full text. Hallucinated quotes are dropped automatically; if no verbatim passage answers the question, we say so. How this works

Citation neighborhood (no data yet)

We don't have any in-corpus citations linked to this paper yet. This is a recent paper (2025) — citers typically take a year or two to land, and the OpenAlex reference graph may still be filling in.

Source provenance

europepmc
last seen: 2026-05-20T01:45:00.602351+00:00