The state-of-the-art machine learning model for Plasma Protein Binding Prediction: computational modeling with OCHEM and experimental validation

preprint OA: closed
📄 Open PDF Full text JSON View at publisher

Abstract

ABSTRACT Plasma protein binding (PPB) is closely related to pharmacokinetics, pharmacodynamics and drug toxicity. Prediction of PPB is an alternative to experimental approaches that are known to be time-consuming and costly. Although there are various models and web servers for PPB prediction already available, they suffer from low prediction accuracy and poor interpretability, in particular for molecules with high values, and are most often not properly validated in prospective studies. Here, we carried out strict data curation, and applied consensus modeling to obtain a model with a coefficient of determination of 0.90 and 0.91 on the training set and the test set, respectively. This model was further validated in a prospective study to predict 63 poly-fluorinated and another 25 highly diverse compounds, and its performance for both these sets was superior to that of other previously reported models. To identify structural features related to PPB, we analyzed a model based on Morgan2 fingerprints and identified that features such as aromatic rings, halogen atoms, heterocyclic rings can discriminate high- and low-PPB molecules. In conclusion, we have established a PPB prediction model that showed state-of-the-art performance in prospective screening, which we have made publicly available in the OCHEM platform ( https://ochem.eu/article/29 ). Graphic Abstract
Full text 58,444 characters · extracted from preprint-html · click to expand
The state-of-the-art machine learning model for Plasma Protein Binding Prediction: computational modeling with OCHEM and experimental validation | bioRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-M677548'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search New Results The state-of-the-art machine learning model for Plasma Protein Binding Prediction: computational modeling with OCHEM and experimental validation View ORCID Profile Zunsheng Han , Zhonghua Xia , View ORCID Profile Jie Xia , View ORCID Profile Igor V. Tetko , Song Wu doi: https://doi.org/10.1101/2024.07.12.603170 Zunsheng Han a State Key Laboratory of Bioactive Substance and Function of Natural Medicines, Institute of Materia Medica, Chinese Academy of Medical Sciences and Peking Union Medical College , Beijing 100050, China Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Zunsheng Han Zhonghua Xia b Institute of Structural Biology, Helmholtz Munich - German Research Center for Environmental Health (GmbH) , Ingolstädter Landstraße 1, 85764 Neuherberg, Germany c BIGCHEM GmbH , Valerystr. 49, 85716 Unterschleißheim, Germany Find this author on Google Scholar Find this author on PubMed Search for this author on this site Jie Xia a State Key Laboratory of Bioactive Substance and Function of Natural Medicines, Institute of Materia Medica, Chinese Academy of Medical Sciences and Peking Union Medical College , Beijing 100050, China Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Jie Xia For correspondence: itetkoai{at}gmail.com jie.william.xia{at}hotmail.com ws{at}imm.ac.cn Igor V. Tetko b Institute of Structural Biology, Helmholtz Munich - German Research Center for Environmental Health (GmbH) , Ingolstädter Landstraße 1, 85764 Neuherberg, Germany c BIGCHEM GmbH , Valerystr. 49, 85716 Unterschleißheim, Germany Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Igor V. Tetko For correspondence: itetkoai{at}gmail.com jie.william.xia{at}hotmail.com ws{at}imm.ac.cn Song Wu a State Key Laboratory of Bioactive Substance and Function of Natural Medicines, Institute of Materia Medica, Chinese Academy of Medical Sciences and Peking Union Medical College , Beijing 100050, China Find this author on Google Scholar Find this author on PubMed Search for this author on this site For correspondence: itetkoai{at}gmail.com jie.william.xia{at}hotmail.com ws{at}imm.ac.cn Abstract Full Text Info/History Metrics Supplementary material Preview PDF ABSTRACT Plasma protein binding (PPB) is closely related to pharmacokinetics, pharmacodynamics and drug toxicity. Prediction of PPB is an alternative to experimental approaches that are known to be time-consuming and costly. Although there are various models and web servers for PPB prediction already available, they suffer from low prediction accuracy and poor interpretability, in particular for molecules with high values, and are most often not properly validated in prospective studies. Here, we carried out strict data curation, and applied consensus modeling to obtain a model with a coefficient of determination of 0.90 and 0.91 on the training set and the test set, respectively. This model was further validated in a prospective study to predict 63 poly-fluorinated and another 25 highly diverse compounds, and its performance for both these sets was superior to that of other previously reported models. To identify structural features related to PPB, we analyzed a model based on Morgan2 fingerprints and identified that features such as aromatic rings, halogen atoms, heterocyclic rings can discriminate high- and low-PPB molecules. In conclusion, we have established a PPB prediction model that showed state-of-the-art performance in prospective screening, which we have made publicly available in the OCHEM platform ( https://ochem.eu/article/29 ). Download figure Open in new tab 1. INTRODUCTION Binding of drugs to plasma proteins is one of the important parameters of pharmacokinetics, as it can affect many key properties of drugs, e.g. , distribution volume (Vss), drug-drug interaction (DDI), clearance rate (CL) and therapeutic index (TI) [ 1 - 3 ] . After drugs are administered, most drugs bind to plasma proteins to various degrees through blood circulation, forming drug-protein complexes. This binding is reversible and most of the time, it is non-specific [ 4 , 5 ] . Only free drugs (i.e., unbound drugs) can exert their biological effects. However, these drug-plasma protein complex can be used as a drug reservoir. When free drugs are metabolized and the concentration of free drugs is reduced, bound drugs can be released and become free drugs to play their role [ 4 , 6 ] . Drugs based on compounds with a high affinity to plasma proteins have an increased half-life, and in some cases higher doses of the drug may be required to achieve an effective concentration for treatment. In addition, drugs bind to plasma proteins competitively, and drugs with higher binding rates will occupy most of the plasma protein binding sites. Special attention should be paid to drug-drug interaction (DDI) [ 1 ] . Drugs with high PPB values may influence the binding of other drugs to the same plasma proteins, resulting in an increase or decrease in the (free) plasma concentration of the other drug, leading to toxicity or ineffectiveness. Situations like this mainly arise for drugs with narrow therapeutic windows, e.g., warfarin [ 3 , 7 ] . Consequently, the PPB property of a drug affects its absorption, distribution, metabolism, elimination, and toxicity (ADME-T). Thus, assessment of PPB is very important for the development of new drugs and the safe use of clinical drugs The level of binding of a drug to plasma proteins is usually evaluated as PPB rate (PPB%) or free fraction (fu) [ 6 , 8 ] . There are three commonly used methods for determining PPB,, i.e., equilibrium dialysis (ED) [ 9 - 11 ] , ultrafiltration (UF) [ 12 , 13 ] and ultracentrifugation (UC) [ 8 , 14 ] . The ED method is the gold standard and is often used as a reference for UF and UC. However, each of the three methods has its own advantages and disadvantages depending on the application [ 11 ] . For non-specific compounds with good adsorption, UC is the first choice, but the cost is high. For compounds that are not stable in plasma, UF is preferred, but the non-specific adsorption is more serious. ED is generally applied to most compounds, but its drawbacks are obvious, such as changes in initial equilibrium state, non-specific binding, volume transformation, Donnan effect and protein leakage, all of which can influence results [ 6 , 10 , 15 ] . In any case, experimental assays are complex, time-consuming, and expensive, while in silico prediction has the advantages of being economical, simple, and fast, and able to facilitate rapid screening of a large number of compounds. Over the past decade, significant advances have been made in ML based drug discovery [ 3 , 16 ] . In recent years, a number of regression models using ML have been built for PPB prediction. Table 1 summarizes some of the published models and performance metrics [ 17 - 24 ] . These models have made great progress in improving PPB prediction accuracy, but the performance of these models on test sets still needs some improvement. Moreover, all of the aforementioned models were not validated in prospective studies. It is also worth mentioning that none of the previous studies (with the exception of the study by Lou et al. [ 24 ] using IDL-PPBopt) discussed the substructure and physicochemical properties that are highly related to PPB. View this table: View inline View popup Table 1. Performance of the Published Models In addition, the successful use of these models usually requires a significant amount of programming expertise, meaning that many of them are not particularly user-friendly. [ 25 ] .To address this issue, several web servers offer free PPB prediction services, e.g. ADMETlab3.0 [ 26 ] , admetSAR3.0 [ 27 ] , DruMAP [ 28 ] , Pangu Drug [ 29 ] , pkCSM(Deep-PK) [ 30 ] , PreADMET [ 31 ] . Although these platforms are easy to use, it appears that the model accuracies of these platforms have not been validated in prospective studies. On-line CHEmical database and Modelling environment (OCHEM) is an algorithm-rich, automated, and simple model training and sharing platform [ 32 ] . In this study, we leveraged the services offered by OCHEM to train models for PPB prediction with a variety of ML algorithms and descriptors. Then, we picked out several better-performing models and continued to train them into a consensus model. In addition, we performed experimental validation of the consensus model, by using it to predict the PPB of a diverse set of compounds from the ChemDiv library as an external validation set and then tested the compounds with ED assay. We also compared it with other publicly available web servers and models for such a prospective study of PPB prediction. Finally, we analyzed the molecular features that were associated with high and low PPB values. The workflow adopted in this study is shown in Figure 1 . Download figure Open in new tab Figure 1. The workflow of the PPB prediction model. 2. Data 2.1 Training Data set Human PPB data points were collected from several sources: (1) literature data from Tang [ 24 ] , Basant [ 33 ] , Reiko [ 20 ] , Brandall [ 34 ] , etc.; and (2) from DrugBank for drugs on market and in clinical trials ( https://go.drugbank.com/ , accessed in May 2022). A total of 9,500 data points were collected, including SMILES and multiple different forms of PPB values (e.g. protein binding rate 90% or 0.9 or unbound ratio 0.1 or 10%). The compound data were carefully prepared according to the following steps: (1) de-duplication, and removing mixtures, inorganic and organometallic compounds; (2) stripping salts and water; (3) removing compounds with molecular weights more than 800 Da and rotatable bonds more than 20 degrees [ 35 , 36 ] . According to the histogram of %PPB (cf. Figure S1), the distribution of original %PPB values was heavily skewed toward the high-PPB region. To minimize the effects of this skew, %PPB data points were transformed into a pseudo-equilibrium constant parameter (LogIt) in OCHEM automatically (see Eq. 1 ) as suggested elsewhere. [ 37 ] After this conversion the distribution became Gaussian-like (cf. Figure 2A ). where f b is fraction bound. This transformation is better for addressing/investigating difference in compounds with very small (i.e., 99%) PPB values. Indeed, there is a very large discrepancy in compounds with a PPB 99% and those with a PPB of 99.9%, since the difference in free concentration of the compound in plasma capable of producing a therapeutic effect will be 10 folds. Finally, total of 3214 compounds with PPB values were collected. These data were randomly split into training (2571) and external validation (643) sets. Download figure Open in new tab Figure 2 (A) Distribution of LogIt transformation dataset, PCA distribution plots: (B) Training and test dataset, (C) Training and PFC dataset, (D) Training and new dataset, (E) Pairwise Tanimoto coefficients distribution of 3,214 compounds. 2.2 Retrospective validation: PFC dataset A dataset containing PPB values for Per- and Polyfluoroalkyl Substances was used for model validation [ 38 ] . As the article was published in 2023, this dataset was not yet encountered by many models published by other groups, nor by our own model, which collected data from earlier publications. We found that 5 out of 68 compounds from the novel dataset were also part of our training dataset. These molecules were excluded and the remaining 63 molecules were used as a validation set, which we called the “PFC dataset”. Furthermore, 36 compounds from the PFC set with PPB >99% formed the PFC99 set, which was used to compare methods for compounds that bind very strongly to plasma proteins. 2.3 Prospective validation: new dataset A prospective study was performed to estimate model performance for a set of compounds selected using an experimental design method. The 10k compounds of the ChemDiv diverse subset purchased by Institute of Materia Medica, Chinese Academy of Medical Sciences (structure information can be retrieved at Zenodo: https://zenodo.org/records/12641856 ), were clustered using the “Cluster Ligands” module of Discovery Studio (version 2016). Specifically, the fixed “Number of Clusters” was set to 25 and FCFP_6 fingerprints [ 39 ] were used to represent compounds. The cluster centers (molecules) were selected to form the external prospective set. These were all new compounds and there was no overlap between these molecules and the modeling set (training and retrospective test set). The experimental method used to measure PPB was chosen based on the premise that free drugs can pass through the semi-permeable membrane (cellulose material), while the drugs bound to plasma proteins cannot. To an appropriate volume of compound stock solution, plasma (Human, IPHASE Pharmaceutical Technology Co., Ltd. Ethics No. ZYBFZ-LL-SOP-020-02) was added to yield a 10 μg/mL concentration of compound solution in plasma. 200 μL of compound solution in plasma was transferred to the dialysis bag of the RED device (Thermo Fisher Scientific, Single-Use RED Plate with Inserts, 8K MWCO), and a 400μL PBS solution was added to the corresponding well and the plate was covered. The RED dialysis unit was incubated at 37 °C in a constant-temperature incubator and shaken at 220 rpm for 4 h. After the incubation was completed, the acetonitrile protein precipitation method followed by HPLC or LC/MS/MS were used to determine the compound concentration in the dialysis bag and the compound concentration in the corresponding well. The results from the HPLC and LC/MS/MS experiments are shown in Table S1. The plasma protein binding rates PPB% were calculated according to the formula 100% - B/A * 100%. For the chemical space analysis, the Mordred package(1.2.0) [ 40 ] was used to calculate the descriptors, since it contributed the model with highest accuracy as shown in Results section. The Tanimoto coefficients (Tc) [ 41 ] based on Morgan2 fingerprints (implementation of ECFP4 [ 39 ] in RDKit package [ 42 ] ), were calculated for each pair of compounds in the initial set of 3,214 compounds and their distribution. 3 Methods The PPB regression models were built by using the Online CHEmistry database and Modeling environment (OCHEM) [ 32 ] , a web server that integrates a wide variety of ML algorithms and molecular representations. During model training, a number of algorithms implemented in OCHEM were applied and the models giving the best performance were combined into a consensus model [ 43 ] which provided the final predictions. 3.1 Analyzed machine approaches We initially analyzed several machine learning approaches available in OCHEM, including traditional methods such as Partial Least Squares, Multiple Linear Regression Analysis, k-Nearest Neighbors as well as more advanced methods based on decision trees, such as Random Forest, XGboost, CatBoos, shallow and deep neural networks, which develop models based on calculated descriptors. Among all investigated methods, the Associative Neural Network (ASNN) method [ 44 ] generally provided higher performances across analyzed descriptors and as such this method was selected for model development. The ASNN is a combination of an ensemble of single hidden layer neural networks and k Nearest Neighbors. It was inspired by thalamo-cortical organization [ 45 ] of the brain and improves performance of ensemble by correcting its bias using errors of the most similar records from the training set. Of the representation learning methods we selected Transformer Convolutional Neural Networks (TransCNN [ 46 ] ), along with its variation Transformer Convolutional Neural Fingerprint (CNF [ 47 ] ) as well as two Graph Neural networks methods: AttentiveFP [ 48 ] and ChemProp [ 49 ] , both of which are implemented as part of the KGCNN [ 50 ] package in OCHEM. 3.2 Model validation The models were validated in two steps. The first one was the five-fold cross-validation (5CV) based on the training set during model construction. During this step training dataset (n=2571) was randomly split into five non-overlapping parts. Four of five parts were used to develop the model and predict the remaining part, which was used as the validation set for the respective model. This procedure was repeated five times and predictions for individual validation sets were collected to form the prediction for the initial training set. The final model was developed with the whole initial training set according to the same procedure as was used for each individual model during the 5CV. In addition to 5CV result the prediction performance of the final model was validated using prospective and retrospective test sets, which are described in the Data section. 3.3 Descriptor filtering OCHEM contains around 20 types of descriptors. For this analysis we decided to use only 2D descriptors. Indeed, we noticed that 3D based descriptors did not increase quality of models, e.g. models developed with full and 2D sets of Mordred [ 40 ] descriptors had a similar accuracy, which is why we decided to proceed with 2D descriptors. On each validation fold, as well as during the final model development, an unsupervised filtering of descriptors was performed. The pairwise de-correlation cutoff of descriptors was set to 0.95, and descriptors with low variance and nearly constant values were eliminated. The filtering procedure and descriptor and model hyperparameters were used with default values as specified in OCHEM. 3.4 Assessment of Model Performance Three statistical parameters were used to evaluate model performance (cf. Eqs. 2 - 4 ). MAE and RMSE were used to evaluate the difference between the predicted values and the observed values. In addition to these parameters, the coefficient of determination R 2 , which measures explained variance of the model, was also used. A good model usually has small MAE and RMSE values, and an R 2 value close to 1. where is the observed value, is the predicted value, ̅ is the average of the observed values, N is the number of data points. 3.5 Applicability domain The application domain (AD) is an important concept in quantitative structure activity relationships (QSAR) [ 51 ] . The measurement of AD was based on the distance to model (DM) of the compound, where DM is a numerical measure proportional to the model’s prediction uncertainty for a given compound [ 52 ] . A large DM value corresponds to a low prediction accuracy for the target compound. In our study, the DM corresponds to the standard deviation of models in consensus. A DM that covered 95% of compounds in the training set was used to define the AD of a model. Each prediction was given a confidence interval to help users judge the reliability of the prediction. 3.6 PPB prediction using webservers The PPB values for retrospective and prospective test set molecules were predicted using different computational platforms (ADMETlab3.0 [ 26 ] ( https://admetlab3.scbdd.com ), admetSAR3.0 [ 27 ] ( http://lmmd.ecust.edu.cn/admetsar3 ), DruMAP [ 28 ] ( https://drumap.nibiohn.go.jp ), Pangu Drug [ 29 ] ( http://pangu-drug.com ), PreADMET [ 31 ] ( https://preadmet.qsarhub.com/ ), pkCSM(Deep-PK) [ 30 ] ( https://biosig.lab.uq.edu.au/deeppk/data ). The prediction accuracies of these platforms were compared with performance of the PPB model developed in this study. 3.7 Feature analysis related to PPB 3.7.1 Physicochemical descriptors analysis At first, the physicochemical descriptors of all compounds (3,214) were calculated using the Mordred package (1.2.0) [ 40 ] , and then the Person correlation coefficient ( r 2 ) of each descriptor with PPB was calculated. Lastly, top 17 descriptors with r 2 >0.4 were used to plot the heat map. In order to gain insight into the model’s interpretability, the relationship between each of the aforementioned descriptors with high, medium and low PPB values were analyzed, respectively. All compounds were divided into three categories according to their PPB values, i.e. low PPB class (≤ 50%), medium PPB class (50-90%), high PPB class (≥90%). In each class, the compounds were further divided into 3 groups according to the descriptor being studied. Then, the number of compounds in each descriptor group in the categories of low, medium and high PPB was counted. The number of compounds in a certain category was defined as 100%, and the percentage of molecules in each descriptor group in all compounds in this category was calculated. The data were converted into graphs for visualization after calculation. Taking the descriptor of SlogP as an example, the high PPB compounds were divided into three subgroups (SLogP≥3, 1<SLogP<3, SLogP≤1), and the proportion of compounds in each subgroup was determined. This facilitated the identification of physicochemical properties associated with high and low-PPB compounds. 3.7.2 Privileged substructures analysis with Similarity Map and Setcompare in OCHEM A PPB classification model was built and was used to produce a similarity map. [ 53 ] Specifically, the compounds from the initial set (n=3128) were divided into high and low PPB sets with thresholds of PPB >90% and PPB<50% respectively. The compounds with PPB values between 50% and 90% were removed. Each compound was represented with Morgan2 fingerprints. The classification model was built based on the training set and was evaluated on the test set. In the similarity map [ 53 ] , the atoms of each compound were marked with different colors according to the contribution value of the atom. The substructures composed of atoms with positive contributions or negative contributions were visualized. The high/low PPB data were also analyzed using the SetCompare utility [ 54 ] in combination with the Extended Functional Groups [ 55 ] descriptor type. SetCompare uses hypergeometric distribution with Bonferroni correction to identify overrepresented descriptors amid compounds with high/low PPB data. 4. Results and discussion 4.1 The Data Set After careful pre-processing, 3,214 PPB data points were obtained. The LogIt transformed data exhibited a Gaussian-like distribution, as shown in Figure 2(A) . As mentioned in the Data section, the data points were randomly split between the training set and test set. The chemical space of the training, test, PFC and new compound datasets are shown in Figure 2 (B-D) based on principal component analysis (PCA) of the 17 Mordred descriptors (see Supporting Material). The scatter plot shows substantial overlap in chemical space between training and test compounds. The Tanimoto coefficient (Tc) values of most pairs of compounds were less than 0.2 (see Figure 2(E) ), indicating high chemical diversity in the PPB data set. 4.2 Models constructed in OCHEM OCHEM offers a number of ML algorithms and various molecular representations. Almost all of the ML algorithms implemented in the platform were tested in this study. ASNN [ 44 ] provided on average better performance compared to other descriptor-bases machine learning methods and were used for further analysis. Amid ASNN models, the model based on Mordred descriptors achieved the highest accuracy for the training set ( Table 2 ). After comparing all models, we selected those with RMSE equal or less than 0.33 to build the consensus model. There were five ASNN models developed with ALogPS-Oestate [ 56 - 58 ] , EPA ( https://www.epa.gov/comptox-tools/toxicity-estimation-software-tool-test ), Fragmentor [ 59 ] , MOLD2 [ 60 ] and MORDRED (2D) [ 40 ] descriptors as well as four models based on representation learning: transformer convolutional neural networks (Transformer CNN) [ 61 ] , Convolutional neural network Fingerprint (CNF2) [ 62 ] , ChemProp [ 63 ] and AttFP [ 64 ] (as implemented in KGCNN [ 50 ] package). The performance of the consensus model, calculated as a simple average of these selected models, was superior to any individual model for both training and external test sets (see Table 2 and Figure 3 ). Download figure Open in new tab Figure 3. The measured PPB vs the PPB predicted by the consensus model for the training and test sets. View this table: View inline View popup Table 2. Performance of individual and the consensus models 4.3 Prospective and retrospective study: Model performance and comparison with other models We clustered a diverse ChemDiv set (10,000) based on FCFP_6 fingerprints into 25 cluster and selected 25 cluster centers to comprise an external set for the prospective study. The PPB of the compounds were predicted using several representative open platforms, including ADMETlab3.0, admetSAR3.0, DruMAP, Pangu Drug, pkCSM(Deep-PK), PreADMET and the newly developed consensus model in OCHEM. To compare the prediction accuracy of these models in practice, we determined the true PPB values of 25 new compounds using equilibrium dialysis combined with HPLC or LC-MS/MS. The structures and measured PPB values of the 25 compounds are shown in Figure 4 . The measured values and results of each prediction platform are shown in Supporting Material. The predictive performance parameters for the different platforms are listed in Table 3 . The consensus model developed in this study achieved a higher accuracy than the other published models. Download figure Open in new tab Figure 4. The structure and experimental results of the 25 new compounds. (*MV: measured value). View this table: View inline View popup Table 3. The tested platforms’ prediction performance for external validation sets. For the retrospective study, we also compared the predictive performances of the models with the 63 PFC dataset (See Table 3 and Supporting Material for more detail) and found that the consensus model developed in this study also achieved a higher accuracy for this set than the other platforms. In addition, the results for compounds with PPB>99% (PFC99 set) predicted by the developed model had significantly better RMSE values (0.4%) compared to the results obtained by other models, which had RMSE values 3 to 100 times higher for the same compounds. Accurately predicting compounds with PPB>99% is very important and significant for drug discovery projects, because for binding rates of 99%, 99.9% or 99.99%, the free drug concentrations will differ by a factor of 10 or 100. Thus, despite the fact that existing models have a relatively good overall prediction performance when using %PPB as an overall performance measure, most of them were not able to distinguish compounds with very high PPBs. 4.4 Importance of LogIt function for predicting compounds with high PPB values We used LogIt units to develop the individual and consensus models and estimate their performances. Of course, one can use different units, e.g. percentage, to estimate performance of the developed models by converting predicted and experimental values to the respective unit. For the consensus model we first converted LogIt to a percentage (%) for each individual model and then built a consensus by averaging predictions of individual models given in percentages. This was done in OCHEM, which allow the user to select a different target unit when creating a consensus model. As shown in Table 2 and Table 4 , this conversion had a generally negative impact on the models: it decreased R 2 and also led to widened confidence intervals for RMSE and MAE. Therefore, we also developed individual models using the % unit and created a consensus model with % or LogIt units, respectively (See Table 4 ). View this table: View inline View popup Download powerpoint Table 4. RMSE of consensus models developed with LogIt and percentage units The consensus model based on individual models developed with the same unit (LogIt-Logit model) had a significantly lower RMSE compared to the consensus model based on the individual models developed with the % unit (%-LogIt model) for all but the PFC set. For the latter set, RMSEs of both consensus models were not significantly different due to their large confidence intervals. There were no significant differences for all but the PFC99 subset (see discussion below) when we compared performances of LogIt-% and %-% models using the % unit. The results for the PFC set had significantly large errors (RMSE=0.79 and 0.74 in LogIt unit) compared to the training set compounds (RMSE=0.25 and 0.29) for LogIt-LogIt and %-LogIt models, thus indicating that this set was particularly difficult to predict. This was the expected results, since the distribution of the compounds in the PFC set was markedly different to that of the training set, as shown on PCA plot ( Figure 2C ). However, there were no significant differences in RMSEs for the training and test sets when predicted values were compared using % units. However, for the subset of the PFC set with PPB>99% (PFC99, n = 36), the consensus model LogIt-* models yielded significantly smaller errors than the %-* consensus model for both unit types ( Table 4 ). Thus, the LogIt-* consensus models were much better at predicting the PPB of compounds with very high binding. The ability to differentiate compounds with high PPB values is important for drug discovery, and we have shown that developing models using LogIt units allows for a deeper exploration of the data and more accurate predictions of PPB. 4.5 Physicochemical descriptors highly related to PPB We calculated the physicochemical descriptors of the molecules, along with their corresponding Person correlation coefficients r 2 with PPB using the Mordred program [ 40 ] . A total of 397 related physicochemical descriptors ( r 2 > 0.05) were obtained. Among them, 17 descriptors (see Supporting Material Table S2) were highly correlated to PPB, with r 2 values greater than 0.4. The heatmap below ( Figure 5 ) illustrates the correlations between the selected descriptors and PPB. Download figure Open in new tab Figure 5. A heatmap of r 2 between any two descriptors or between any descriptor and PPB. The heatmap was plotted based on the absolute value of r 2 . As mentioned in the Methods section, we analyzed the relationships between each of the aforementioned descriptors and high, medium and low PPB values. We observed that the SLogP and LogS are very important physicochemical indicators for PPB, with R 2 values are greater than 0.6. When considering three subgroups of SLogP (SLogP≥3, 1<SLog P<3, SLog P≤1) and three subgroups of LogS (LogS<-6, -6≤logS<-4, -4≤logS), we observe that a large proportion (around 80%) of high-PPB compounds had high SLogP (≥3) and low LogS (<-4), as shown in Figure 6A and 6B . In medium- or low-PPB compounds, the proportion of compounds with SLogP≥3 and LogS<-4 decreased significantly. Therefore, SLogP was positively correlated with PPB and LogS were negatively correlated with PPB. Download figure Open in new tab Figure 6. The differences in physicochemical properties of high-, medium- and low-PPB compounds. (A) SLogP, (B) LogS, (C) AromAtom or nAromBond, (D) nX, (E) nAcid, (F) nBas. The number of AromAtom or nAromBond was also correlated with PPB (see Figure 6C .). A reduction in the number of AromAtom or nAromBond led to a significant decrease in the value of PPB – not a surprising observation, as it is generally known that the increase in the number of aromatic rings can improve the fat-soluble properties of a drug. In addition, we found that a higher number halogen atom (nX) may be associated with higher PPB values ( Figure 6D ). The number of acid (nAcid) and basic (nBase) groups is closely related to the pKa of compounds. It was found that, for compounds with 0-1 acid group, nAcid had no influence on PPB. For nAcid ≤ 2, the increase of nAcid may lead to lower PPB values. As for nBase, the effect of basic groups on PPB was more significant than that of acidic groups. In general, the higher the number of basic groups, the lower the PPB value ( Figure 6E and 6F ). Atom-bond Connectivity Index (ABC Index) refers to the strong interaction between two or more adjacent atoms, including covalent bond,s ionic bonds and metallic bonds. The study found that lower ABC Indices were associated with lower PPB values, as shown in Supporting Material Figure S2A. Hydrogen bonding is known to be a very important intermolecular force, and our results showed that the more hydrogen bond donors a molecule contained (i.e. >2), the more likely it was to have a lower PPB, whereas the number of hydrogen bond receptors had no significant influence on PPB (see Supporting Material Figure S2B and S2C). Furthermore, we found that some feature descriptors also showed some positive correlation with PPB. For example: KappaShapeIndex, sum of atomic volume parameters (McGowanVolume), atomic polarizability, rotatable bond (nRot), van der Waals volume (VdwVolumeABC), and some topological indicators, such as Wiener index and ZagrebIndex, see Supporting Material Figure S2D-J. 4.6 Substructures that affect PPB To identify structures that significantly influence PPB, we built a classification model (PPB>90% and PPB< 50%) using Morgan2 fingerprints and generated a similarity map for several representative compounds [ 53 ] . The representative similarity maps of 3 low-PPB and 3 high-PPB compounds are displayed in Figure 7 . In low-PPB compounds (see Figure 7A-C ), we can conclude that: (1) amino groups often appeared in low-PPB molecules, and secondary and primary amines were dominant, which was exactly consistent with the conclusion of the descriptor analysis (the more basic groups, the lower the PPB). Also, the presence of five-membered nitrogen heterocyclic rings, saturated polycyclic rings, carbonyl groups, hydroxyl groups and carboxyl groups appeared to play an important role in the reduction of PPB. (2) In high-PPB compounds (see Figure 7D-F ), the presence of aromatic rings, halogen atoms (F, Cl, Br) in a benzene ring or alkyl chain, alkyl chains, sulfonyl groups, thiazoles, oxazoles and oxadiazoles etc. were associated with higher PPB values. More details are shown in Supporting Material Figure S3. and Figure S4. Download figure Open in new tab Figure 7. Structural contribution diagrams of partially representative low(A, B and C) and high (D, E, F) PPB compounds. Atoms colored red have a greater likelihood of increasing PPB, while green atoms are associated with lower PPB values. We further analyzed the frequency of occurrence of substructures of high/low PPB compounds using SetCompare tool in the OCHEM platform. It was found that several functional groups were more strongly associated with one of the two analyzed classes (see Figure 8a ). For example, aromatic and halogen derivatives occurred significantly more often in high PPB compounds. Additionally, further functional groups, e.g., arenes and aryl chlorides derivatives, were overrepresented in the high PPB dataset. At the same time, primary amines and secondary aliphatic amines occurred more often in the low PPB dataset (see Figure 8b ). Other groups associated with low PPB are secondary alcohols, heterocycles,α, β-Unsaturated carboxylic acids and tetrahydrofurans. The results from this analysis can provide useful suggestions for chemists seeking to design compounds with lower PPB. The full list of calculated groups is show in Supporting Material. Download figure Open in new tab Figure 8. Functional groups that are overrepresented in high (a) / low (b) PPB compound datasets. Appearance counts are listed, as well as the p-value of the respective distribution. Negative and positive p-values indicate groups overrepresented in high/low PPB datasets, respectively. 5. Conclusions PPB is one of the key parameters for evaluating the efficacy of potential drugs, and the accurate prediction of PPB plays an important role in the screening stages of drug discovery. We compiled a comprehensive collection of PPB values for known compounds, rigorously curated the data, trained a variety of ML models with OCHEM and built a consensus model based on the best performing models. The consensus model performed well in both internal and external validation, and yielded results that were superior to those given by other online webservers for PPB prediction in the prospective study, i.e., predicting PPB values of diverse, unknown compounds, with experimental validation. The consensus model is available for free on the OCHEM website at https://ochem.eu/article/29 . In addition, we analyzed and identified the physicochemical descriptors and fragments that have a significant influence on PPB, which may facilitate lead optimization or drug development. These results of this study are particularly useful for the prediction of high PPB compounds (>99%), as excessively high PPB poses many problems in drug discovery, such as: narrow treatment window, obvious individual differences, drug-drug interactions, and the need for clinicians to precisely measure drug concentrations. Hence, we can conclude that the current maximum achievable accuracy of PPB predictions for high PPB compounds provided by existing models is generally not satisfactory. While model described in this article demonstrated good performance for such compounds, large number of experimental measurements are required to further improve models for compounds with high PPB. Supporting Information The distribution of original %PPB values, ChemDiv library of 25 compounds, detection methods with HPLC or LC/MS/MS, experimental and predicted results. Mordred descriptors determined via feature selection, differences in molecular and substructural properties between high and low PPB chemicals (Supporting Material.docx); model output by six online web servers for 25 new measured dataset, PFC dataset, PFC99 dataset, and substructure feature analysis by Setcompare results (Supporting Material.xlsx). Author Contributions J.X. led the project, with the support from S.W. and I.V.T. Z.H. performed data collection and curation, and built ML models. Z.H., I.V.T., J.X. and Z.X jointly performed the data analysis. Z.H. tested the wet verification. Z.H. and J.X. wrote the manuscript. All authors reviewed the final version of the manuscript and provided important revisions. Funding Funding sources of the study are Chinese Academy of Medical Sciences (CAMS) Innovation Fund for Medical Sciences (No. 2021-I2M-1-069), the National Science and Technology Major Projects for Major New Drugs Innovation and Development (No. 2018ZX09711001-012-003) and the Program for Foreign Talent of Ministry of Science and Technology of the People’s Republic of China (No. G2021194015L). Notes The authors declare no competing financial interest. Acknowledgements We thank Dr. Katya Ahmad for proof-reading the manuscript for English and her comments. REFERENCES [1]. ↵ Di L . An update on the importance of plasma protein binding in drug discovery and development [J] . Expert Opin Drug Discov , 2021 , 16 ( 12 ): 1453 – 1465 . OpenUrl CrossRef [2]. Smith D A , Di L , Kerns E H . The effect of plasma protein binding on in vivo efficacy: misconceptions in drug discovery [J] . Nat Rev Drug Discov , 2010 , 9 ( 12 ): 929 – 939 . OpenUrl CrossRef PubMed [3]. ↵ Lambrinidis G , Vallianatou T , Tsantili-Kakoulidou A. In vitro, in silico and integrated strategies for the estimation of plasma protein binding. A review [J] . Adv Drug Deliv Rev , 2015 , 86 ( 27 – 45 . OpenUrl CrossRef [4]. ↵ Liu X , Wright M , Hop C E . Rational use of plasma protein and tissue binding data in drug design [J] . J Med Chem , 2014 , 57 ( 20 ): 8238 – 8248 . OpenUrl CrossRef [5]. ↵ Levy G . Effect of plasma protein binding on renal clearance of drugs [J] . J Pharm Sci , 1980 , 69 ( 4 ): 482 – 483 . OpenUrl CrossRef PubMed Web of Science [6]. ↵ Seyfinejad B , Ozkan S A , Jouyban A . Recent advances in the determination of unbound concentration and plasma protein binding of drugs: Analytical methods [J] . Talanta , 2021 , 225 ( 122052 . [7]. ↵ Di L , Breen C , Chambers R , et al. Industry Perspective on Contemporary Protein-Binding Methodologies: Considerations for Regulatory Drug-Drug Interaction and Related Guidelines on Highly Bound Drugs [J] . J Pharm Sci , 2017 , 106 ( 12 ): 3442 – 3452 . OpenUrl CrossRef [8]. ↵ Vuignier K , Schappler J , Veuthey J L , et al. Drug-protein binding: a critical review of analytical tools [J] . Anal Bioanal Chem , 2010 , 398 ( 1 ): 53 – 66 . OpenUrl CrossRef PubMed Web of Science [9]. ↵ Chen Y C , Kenny J R , Wright M , et al. Improving Confidence in the Determination of Free Fraction for Highly Bound Drugs Using Bidirectional Equilibrium Dialysis [J] . J Pharm Sci , 2019 , 108 ( 3 ): 1296 – 1302 . OpenUrl CrossRef [10]. ↵ Van Liempd S , Morrison D , Sysmans L , et al. Development and validation of a higher-throughput equilibrium dialysis assay for plasma protein binding [J] . J Lab Autom , 2011 , 16 ( 1 ): 56 – 67 . OpenUrl CrossRef PubMed [11]. ↵ Dimitrijevic D , Fabian E , Funk-Weyer D , et al. Rapid equilibrium dialysis, ultrafiltration or ultracentrifugation? Evaluation of methods to quantify the unbound fraction of substances in plasma [J] . Biochem Biophys Res Commun , 2023 , 651 ( 114 – 120 . OpenUrl [12]. ↵ Toma C-M , Imre S , Vari C-E , et al. Ultrafiltration Method for Plasma Protein Binding Studies and Its Limitations [J] . Processes , 2021 , 9 ( 2 ): [13]. ↵ Resztak M , Kosicka K , Zalewska P , et al. Determination of total and free voriconazole in human plasma: Application to pharmacokinetic study and therapeutic monitoring [J] . J Pharm Biomed Anal , 2020 , 178 ( 112952 . [14]. ↵ Turner N A , Xu A , Zaharoff S , et al. Determination of plasma protein binding of dalbavancin [J] . J Antimicrob Chemother , 2022 , 77 ( 7 ): 1899 – 1902 . OpenUrl [15]. ↵ Ryu S , Riccardi K , Patel R , et al. Applying Two Orthogonal Methods to Assess Accuracy of Plasma Protein Binding Measurements for Highly Bound Compounds [J] . J Pharm Sci , 2019 , 108 ( 11 ): 3745 – 3749 . OpenUrl [16]. ↵ Vallianatou T , Lambrinidis G , Tsantili-Kakoulidou A. In silico prediction of human serum albumin binding for drug leads [J] . Expert Opin Drug Discov , 2013 , 8 ( 5 ): 583 – 595 . OpenUrl [17]. ↵ Votano J R , Parham M , Hall L M , et al. Qsar Modeling of Human Serum Protein Binding with Several Modeling Techniques Utilizing Structure−Information Representation [J] . Journal of Medicinal Chemistry , 2006 , 49 ( 24 ): 7169 – 7181 . OpenUrl CrossRef PubMed [18]. Zhu X W , Sedykh A , Zhu H , et al. The use of pseudo-equilibrium constant affords improved Qsar models of human plasma protein binding [J] . Pharm Res , 2013 , 30 ( 7 ): 1790 – 1798 . OpenUrl CrossRef [19]. Wang N-N , Deng Z-K , Huang C , et al. Adme properties evaluation in drug discovery: Prediction of plasma protein binding using NSGA-Ii combining Pls and consensus modeling [J] . Chemometrics and Intelligent Laboratory Systems , 2017 , 170 ( 84 – 95 . OpenUrl [20]. ↵ Watanabe R , Esaki T , Kawashima H , et al. Predicting Fraction Unbound in Human Plasma from Chemical Structure: Improved Accuracy in the Low Value Ranges [J] . Mol Pharm , 2018 , 15 ( 11 ): 5302 – 5311 . OpenUrl [21]. Sun L , Yang H , Li J , et al. In Silico Prediction of Compounds Binding to Human Plasma Proteins by Qsar Models [J] . ChemMedChem , 2018 , 13 ( 6 ): 572 – 581 . OpenUrl CrossRef [22]. Yuan Y , Chang S , Zhang Z , et al. A novel strategy for prediction of human plasma protein binding using machine learning techniques [J] . Chemometrics and Intelligent Laboratory Systems , 2020 , 199 ( [23]. Jimenez-Luna J , Skalic M , Weskamp N , et al. Coloring Molecules with Explainable Artificial Intelligence for Preclinical Relevance Assessment [J] . J Chem Inf Model , 2021 , 61 ( 3 ): 1083 – 1094 . OpenUrl [24]. ↵ Lou C , Yang H , Wang J , et al. IDL-PPBopt: A Strategy for Prediction and Optimization of Human Plasma Protein Binding of Compounds via an Interpretable Deep Learning Method [J] . J Chem Inf Model , 2022 , 62 ( 11 ): 2788 – 2799 . OpenUrl [25]. ↵ Tetko I V , Maran U , Tropsha A. Public (Q)Sar Services, Integrated Modeling Environments, and Model Repositories on the Web: State of the Art and Perspectives for Future Development [J] . Mol Inform , 2017 , 36 ( 3 ): [26]. ↵ Xiong G , Wu Z , Yi J , et al. ADMETlab 2.0: an integrated online platform for accurate and comprehensive predictions of Admet properties [J] . Nucleic Acids Res , 2021 , 49 ( W1 ): W5 – W14 . OpenUrl [27]. ↵ Yang H , Lou C , Sun L , et al. admetsar 2.0: web-service for prediction and optimization of chemical Admet properties [J] . Bioinformatics , 2019 , 35 ( 6 ): 1067 – 1069 . OpenUrl CrossRef [28]. ↵ Kawashima H , Watanabe R , Esaki T , et al. DruMAP: A Novel Drug Metabolism and Pharmacokinetics Analysis Platform [J] . J Med Chem , 2023 , 66 ( 14 ): 9697 – 9709 . OpenUrl [29]. ↵ Xinyuan L , Chi X , Zhaoping X , et al. PanGu Drug Model: Learn a Molecule Like a Human [J] . bioRxiv , 2022 , 2022.03.31.485886. [30]. ↵ Pires D E , Blundell T L , Ascher D B . pkCSM: Predicting Small-Molecule Pharmacokinetic and Toxicity Properties Using Graph-Based Signatures [J] . J Med Chem , 2015 , 58 ( 9 ): 4066 – 4072 . OpenUrl CrossRef PubMed [31]. ↵ Lee S , Lee I H , Kim H J , et al. The Preadme Approach: Web-based program for rapid prediction of physico-chemical, drug absorption and drug-like properties [J]. euro Qsar 2002 - Designing Drugs and Crop Protectants: Processes Problems and Solutions , 2002 , 418 – 420 . [32]. ↵ Sushko I , Novotarskyi S , Korner R , et al. Online chemical modeling environment (OCHEM): web platform for data storage, model development and publishing of chemical information [J] . J Comput Aided Mol Des , 2011 , 25 ( 6 ): 533 - 54 . OpenUrl CrossRef PubMed [33]. ↵ Basant N , Gupta S , Singh K P . Predicting binding affinities of diverse pharmaceutical chemicals to human serum plasma proteins using Qspr modelling approaches [J] . Sar Qsar Environ Res , 2016 , 27 ( 1 ): 67 – 85 . OpenUrl [34]. ↵ Ingle B L , Veber B C , Nichols J W , et al. Informing the Human Plasma Protein Binding of Environmental Chemicals by Machine Learning in the Pharmaceutical Space: Applicability Domain and Limits of Predictability [J] . J Chem Inf Model , 2016 , 56 ( 11 ): 2243 – 2252 . OpenUrl [35]. ↵ Li S , Ding Y , Chen M , et al. HDAC3i-Finder: A Machine Learning-based Computational Tool to Screen for HDAC3 Inhibitors [J] . Molecular Informatics , 2020 , 40 ( 3 ): [36]. ↵ Mysinger M M , Carchia M , Irwin J J , et al. Directory of Useful Decoys, Enhanced (DUD-E): Better Ligands and Decoys for Better Benchmarking [J] . Journal of Medicinal Chemistry , 2012 , 55 ( 14 ): 6582 – 6594 . OpenUrl CrossRef PubMed [37]. ↵ Gao H , Yao L , Mathieu H W , et al. In Silico Modeling of Nonspecific Binding to Human Liver Microsomes [J] . Drug Metabolism and Disposition , 2008 , 36 ( 10 ): 2130 – 2135 . OpenUrl Abstract / FREE Full Text [38]. ↵ Smeltz M , Wambaugh J F , Wetmore B A . Plasma Protein Binding Evaluations of Per- and Polyfluoroalkyl Substances for Category-Based Toxicokinetic Assessment [J] . Chemical Research in Toxicology , 2023 , 36 ( 6 ): 870 – 881 . OpenUrl [39]. ↵ Rogers D , Hahn M . Extended-Connectivity Fingerprints [J] . Journal of Chemical Information and Modeling , 2010 , 50 ( 5 ): 742 – 754 . OpenUrl CrossRef PubMed Web of Science [40]. ↵ Moriwaki H , Tian Y-S , Kawashita N , et al. Mordred: a molecular descriptor calculator [J] . Journal of Cheminformatics , 2018 , 10 ( 1 ): 4 . OpenUrl [41]. ↵ Bajusz D , Rácz A , Héberger K . Why is Tanimoto index an appropriate choice for fingerprint-based similarity calculations? [J] . Journal of Cheminformatics , 2015 , 7 ( 1 ): 20 . OpenUrl [42]. ↵ RDKit: Open-source cheminformatics . https://www.rdkit.org , accessed day 3 July 2024 . [M]. [43]. ↵ Zhu H , Tropsha A , Fourches D , et al. Combinatorial Qsar Modeling of Chemical Toxicants Tested against Tetrahymena pyriformis [J] . Journal of Chemical Information and Modeling , 2008 , 48 ( 4 ): 766 – 784 . OpenUrl PubMed [44]. ↵ Tetko I V. Associative Neural Network [M]//Livingstone D J. Artificial Neural Networks: Methods and Applications . Totowa, NJ ; Humana Press . 2009 : 180 - 97 . [45]. ↵ Villa A E P , Tetko I V , Dutoit P , et al. Corticofugal modulation of functional connectivity within the auditory thalamus of rat, guinea pig and cat revealed by cooling deactivation [J] . Journal of Neuroscience Methods , 1999 , 86 ( 2 ): 161 – 178 . OpenUrl CrossRef PubMed Web of Science [46]. ↵ Karpov P , Godin G , Tetko I V . Transformer-CNN: Swiss knife for Qsar modeling and interpretation [J] . Journal of Cheminformatics , 2020 , 12 ( 1 ): 17 . OpenUrl [47]. ↵ Makarov D M , Fadeeva Y A , Shmukler L E , et al. Beware of proper validation of models for ionic Liquids! [J] . Journal of Molecular Liquids , 2021 , 344 ( 117722 . [48]. ↵ Xiong Z , Wang D , Liu X , et al. Pushing the Boundaries of Molecular Representation for Drug Discovery with the Graph Attention Mechanism [J] . Journal of Medicinal Chemistry , 2020 , 63 ( 16 ): 8749 – 8760 . OpenUrl [49]. ↵ Yang K , Swanson K , Jin W , et al. Analyzing Learned Molecular Representations for Property Prediction [J] . Journal of Chemical Information and Modeling , 2019 , 59 ( 8 ): 3370 – 3388 . OpenUrl CrossRef PubMed [50]. ↵ Reiser P , Eberhard A , Friederich P . Graph neural networks in TensorFlow-Keras with RaggedTensor representation (kgcnn) [J] . Software Impacts , 2021 , 9 ( [51]. ↵ Weaver S , Gleeson M P . The importance of the domain of applicability in Qsar modeling [J] . J Mol Graph Model , 2008 , 26 ( 8 ): 1315 – 1326 . OpenUrl PubMed [52]. ↵ Tetko I V , Sushko I , Pandey A K , et al. Critical assessment of Qsar models of environmental toxicity against Tetrahymena pyriformis: focusing on applicability domain and overfitting by variable selection [J] . J Chem Inf Model , 2008 , 48 ( 9 ): 1733 – 1746 . OpenUrl CrossRef PubMed Web of Science [53]. ↵ Riniker S , Landrum G A . Similarity maps - a visualization strategy for molecular fingerprints and machine-learning methods [J] . Journal of Cheminformatics , 2013 , 5 ( 1 ): 43 . OpenUrl [54]. ↵ Vorberg S , Tetko I V . Modeling the Biodegradability of Chemical Compounds Using the Online CHEmical Modeling Environment (OCHEM) [J] . Molecular Informatics , 2014 , 33 ( 1 ): 73 – 85 . OpenUrl [55]. ↵ Salmina E S , Haider N , Tetko I V. Extended Functional Groups (EFG): An Efficient Set for Chemical Characterization and Structure-Activity Relationship Studies of Chemical Compounds [J/OL] 2016 , 21 ( 1 ): [56]. ↵ Hall L H , Kier L B . Electrotopological State Indices for Atom Types: A Novel Combination of Electronic, Topological, and Valence State Information [J] . Journal of Chemical Information and Computer Sciences , 1995 , 35 ( 6 ): 1039 – 1045 . OpenUrl CrossRef [57]. Huuskonen J J , Villa A E P , Tetko I V . Prediction of partition coefficient based on atom-type electrotopological state indices [J] . Journal of Pharmaceutical Sciences , 1999 , 88 ( 2 ): 229 – 233 . OpenUrl PubMed [58]. ↵ Tetko I V , Tanchuk V Y , Kasheva T N , et al. Estimation of Aqueous Solubility of Chemical Compounds Using E-State Indices [J] . Journal of Chemical Information and Computer Sciences , 2001 , 41 ( 6 ): 1488 – 1493 . OpenUrl PubMed Web of Science [59]. ↵ Varnek A , Fourches D , Horvath D , et al. Isida - Platform for Virtual Screening Based on Fragment and Pharmacophoric Descriptors [J] . Current Computer - Aided Drug Design , 2008 , 4 ( 191 - 8 . OpenUrl [60]. ↵ Hong H , Xie Q , Ge W , et al. Mold2, Molecular Descriptors from 2d Structures for Chemoinformatics and Toxicoinformatics [J] . Journal of Chemical Information and Modeling , 2008 , 48 ( 7 ): 1337 – 44 . OpenUrl CrossRef PubMed [61]. ↵ Liu Y , Wu Y-H , Sun G , et al. Vision Transformers with Hierarchical Attention [J/OL] 2021 , arXiv:2106.03180[ https://ui.adsabs.harvard.edu/abs/2021arXiv210603180L . [62]. ↵ Kimber T B , Engelke S , Tetko I V , et al. Synergy Effect between Convolutional Neural Networks and the Multiplicity of Smiles for Improvement of Molecular Prediction [J/OL] 2018 , arXiv:1812.04439[ https://ui.adsabs.harvard.edu/abs/2018arXiv181204439K . [63]. ↵ Yang K , Swanson K , Jin W , et al. Analyzing Learned Molecular Representations for Property Prediction [J] . J Chem Inf Model , 2019 , 59 ( 8 ): 3370 – 3388 . OpenUrl CrossRef PubMed [64]. ↵ Xiong Z , Wang D , Liu X , et al. Pushing the Boundaries of Molecular Representation for Drug Discovery with the Graph Attention Mechanism [J] . Journal of Medicinal Chemistry , 2019 , 63 ( 16 ): 8749 – 8760 . OpenUrl View the discussion thread. Back to top Previous Next Posted July 16, 2024. Download PDF Supplementary Material Email Thank you for your interest in spreading the word about bioRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following The state-of-the-art machine learning model for Plasma Protein Binding Prediction: computational modeling with OCHEM and experimental validation Message Subject (Your Name) has forwarded a page to you from bioRxiv Message Body (Your Name) thought you would like to see this page from the bioRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share The state-of-the-art machine learning model for Plasma Protein Binding Prediction: computational modeling with OCHEM and experimental validation Zunsheng Han , Zhonghua Xia , Jie Xia , Igor V. Tetko , Song Wu bioRxiv 2024.07.12.603170; doi: https://doi.org/10.1101/2024.07.12.603170 Share This Article: Copy Citation Tools The state-of-the-art machine learning model for Plasma Protein Binding Prediction: computational modeling with OCHEM and experimental validation Zunsheng Han , Zhonghua Xia , Jie Xia , Igor V. Tetko , Song Wu bioRxiv 2024.07.12.603170; doi: https://doi.org/10.1101/2024.07.12.603170 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Bioinformatics Subject Areas All Articles Animal Behavior and Cognition (7651) Biochemistry (17746) Bioengineering (13928) Bioinformatics (42066) Biophysics (21499) Cancer Biology (18650) Cell Biology (25579) Clinical Trials (138) Developmental Biology (13409) Ecology (19947) Epidemiology (2067) Evolutionary Biology (24374) Genetics (15633) Genomics (22557) Immunology (17775) Microbiology (40505) Molecular Biology (17217) Neuroscience (88796) Paleontology (667) Pathology (2845) Pharmacology and Toxicology (4836) Physiology (7664) Plant Biology (15179) Scientific Communication and Education (2047) Synthetic Biology (4304) Systems Biology (9839) Zoology (2272)

Text is read by the "Ask this paper" AI Q&A widget below. Extraction quality varies by source — PMC NXML preserves structure cleanly, OA-HTML may include some navigation residue, and OA-PDF can have broken hyphenation. The publisher copy (via DOI) is the canonical version.

My notes (saved in your browser only)

Ask this paper AI returns verbatim quotes from the full text · source: preprint-html

Answers must be backed by verbatim quotes from this paper's full text. Hallucinated quotes are dropped automatically; if no verbatim passage answers the question, we say so. How this works

Citation neighborhood (no data yet)

We don't have any in-corpus citations linked to this paper yet. This is a recent paper (2024) — citers typically take a year or two to land, and the OpenAlex reference graph may still be filling in.

Source provenance

europepmc
last seen: 2026-05-20T01:45:00.602351+00:00