Full text
65,733 characters
· extracted from
preprint-html
· click to expand
Prediction of Antibody Non-Specificity using Protein Language Models and Biophysical Parameters | bioRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-M677548'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search New Results Prediction of Antibody Non-Specificity using Protein Language Models and Biophysical Parameters View ORCID Profile Laila I. Sakhnini , View ORCID Profile Ludovica Beltrame , View ORCID Profile Simone Fulle , View ORCID Profile Pietro Sormanni , View ORCID Profile Anette Henriksen , View ORCID Profile Nikolai Lorenzen , View ORCID Profile Michele Vendruscolo , Daniele Granata doi: https://doi.org/10.1101/2025.04.28.650927 Laila I. Sakhnini 1 Therapeutics Discovery , Novo Nordisk A/S, Copenhagen, Denmark 3 Centre for Misfolding Diseases, Department of Chemistry, University of Cambridge , UK Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Laila I. Sakhnini For correspondence: llsh{at}novonordisk.com mv245{at}cam.ac.uk dngt{at}novonordisk.com Ludovica Beltrame 2 Digital Chemistry and Design , Novo Nordisk A/S, Copenhagen, Denmark Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Ludovica Beltrame Simone Fulle 2 Digital Chemistry and Design , Novo Nordisk A/S, Copenhagen, Denmark Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Simone Fulle Pietro Sormanni 3 Centre for Misfolding Diseases, Department of Chemistry, University of Cambridge , UK Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Pietro Sormanni Anette Henriksen 1 Therapeutics Discovery , Novo Nordisk A/S, Copenhagen, Denmark Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Anette Henriksen Nikolai Lorenzen 1 Therapeutics Discovery , Novo Nordisk A/S, Copenhagen, Denmark Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Nikolai Lorenzen Michele Vendruscolo 3 Centre for Misfolding Diseases, Department of Chemistry, University of Cambridge , UK Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Michele Vendruscolo For correspondence: llsh{at}novonordisk.com mv245{at}cam.ac.uk dngt{at}novonordisk.com Daniele Granata 2 Digital Chemistry and Design , Novo Nordisk A/S, Copenhagen, Denmark Find this author on Google Scholar Find this author on PubMed Search for this author on this site For correspondence: llsh{at}novonordisk.com mv245{at}cam.ac.uk dngt{at}novonordisk.com Abstract Full Text Info/History Metrics Supplementary material Preview PDF Abstract The development of therapeutic antibodies requires optimizing target binding affinity and pharmacodynamics, while ensuring high developability potential, including minimizing non-specific binding. In this study, we address this problem by predicting antibody non-specificity by two complementary approaches: (i) antibody sequence embeddings by protein language models (PLMs), and (ii) a comprehensive set of sequence-based biophysical descriptors. These models were trained on human and mouse antibody data from Boughter et al . (2020) and tested on three public datasets: Jain et al . (2017), Shehata et al . (2019) and Harvey et al . (2022). We show that non-specificity is best predicted from the heavy variable domain and heavy-chain complementary variable regions (CDRs). The top performing PLM, a heavy variable domain-based ESM 1v LogisticReg model, resulted in 10-fold cross-validation accuracy of up to 71%. Our biophysical descriptor-based analysis identified the isoelectric point as a key driver of non-specificity. Our findings underscore the importance of biophysical properties in predicting antibody non-specificity and highlight the potential of protein language models for the development of antibody-based therapeutics. To illustrate the use of our approach in the development of lead candidates with high developability potential, we show that it can be extended to therapeutic antibodies and nanobodies. 1. Introduction Monoclonal antibodies (mAbs) continue to be one of the leading drug modalities in the pharmaceutical industry, with more than 100 unique mAbs approved by the FDA since 2021 1 and global sales forecasted to 300 billion US dollars by 2025 2 , 3 . The success of mAbs for therapeutic application is the result of advances in in vivo and in vitro discovery platforms, which have enabled fast generation of high-affinity binders towards a highly diverse set of targets 4 , 5 . Recently, de novo design has gained increasing interest in the field as a third-generation discovery approach with the potential of significantly accelerating drug discovery and development timelines 6 , 7 . Moreover, with the advances in mAb engineering, there has also been an increased interest in the development of mAbs with ultra-high target affinity (pM to fM) 8 , context-dependent target binding (e.g. pH-dependent 9 , 10 or ligand induced target binding 11 ), and various multi-specific functionalities 12 . To reach optimal binding affinity and potency, mAb hits identified during discovery are often subjected to comprehensive screening campaigns using display platforms (libraries of 10 3 to 10 10 variants) of and recombinant well-plate variant generation workflows (libraries of 10 2 to 10 4 variants) 13 . When selecting an antibody lead candidate for development towards clinical testing, optimal target binding affinity and/or pharmacodynamics are key selection parameters. In addition, in the last decade there has been an increased focus on the importance of progressing antibodies to clinical stage which also possess a good developability potential. Antibody developability requires the intersection of multiple disciplines, in which diverse parameters such as expression levels, immunogenicity, processability, and formulation feasibility are addressed, to ensure optimal potential for successful clinical development of lead candidates. Non-specific binding, i.e. weak non-covalent interactions with off-target molecules or interfaces, has emerged as one of the key developability parameters to increase the chance for clinical success 14 , 15 . Specifically, several studies reported that high tendency for non-specific binding can translate into faster in vivo clearance, thereby compromising pharmacokinetics 16 - 22 . Furthermore, there is an inherent risk that non-specific interactions can translate into undesirable side-effects 23 . Non-specific binding is not a rare phenomenon, as recent reports suggest the presence of a trade-off between affinity and specificity. Thus, optimization of affinity and potency comes with an inherent risk of compromising target specificity 24 - 27 . Given the high level of interest in measuring non-specific interactions, there are several in vitro screening assays available for this purpose. A commonly used assay is the Enzyme-Linked Immunosorbent Assay (ELISA) with a panel of common antigens, typically insulin, DNA, albumin, cardiolipin and lipopolysaccharide (LPS) 15 , 27 . Initially, these biomolecules were studied as model antigens for autoimmune responses and diseases. For example, insulin is a self-antigen for autoantibodies associated with type 1 diabetes 28 , 29 . In addition, DNA, albumin and cardiolipin are self-antigens for autoantibodies associated with several diseases such as systemic lupus erythematosus 30 , 31 and anti-phospholipid syndrome 32 , and LPS is an antigen for immune responses to bacterial infections 33 . Moreover, ELISA is widely used in immunology, where non-specific antibodies are often referred to as poly-reactive antibodies. Such antibodies are characterised as having low-affinity binding to multiple distinct antigens, including self-antigens, and they have been widely studied for targets such as HIV and Influenza viruses, as they can be broadly neutralizing 34 , 35 . While this feature is beneficial for immunity to infectious diseases, to potentially confer broad protection against viruses, it is a highly undesirable feature for therapeutic mAbs. Other common non-specificity assays include baculovirus particle (BVP) ELISA 36 , poly-specific reagent (PSR) 37 , and cross-interaction chromatography with ligands such as heparin 18 and human IgG from serum 17 . In addition to these in vitro assays, in silico methods have been gaining interest, and several tools have been reported recently for prediction of non-specific interactions 38 - 41 . The development and implementation of predictive computational methods for the prediction and re-design of monoclonal antibody non-specificity at an early stage is of great interest, as it facilitates the generation of safe and efficacious lead candidates with high developability potential. Several studies reported on the identification of non-specificity by in silico approaches. Short, linear sequence motifs (e.g. GG, RR, VG, VV, YY, WW and WxW, where x can be any amino acid) enriched in non-specific antibodies, as reported by Kelly et al . 25 , have been utilized to create synthetic antibody libraries free from such motifs in the CDRs 42 . Moreover, AI/ML models that classifies non-specific antibodies, leveraging experimental data and sequence-based information, have been reported. Boughter et al . 38 developed a classifier to identify non-specific antibodies based on experimental data acquired from ELISA with a panel of common antigens. Harvey et al . 39 developed a one-hot LogisticReg model based on a naïve Nb library assessed by the PSR assay. In this study, we developed machine learning (ML) models to estimate the non-specificity of antibodies ( Figure 1A ). Commonly used biophysical properties were tested alongside protein language models (PLMs) to embed antibody sequences. PLMs have emerged as powerful tools for extracting informative features from raw protein sequences by leveraging patterns learned from massive sequence databases 43 - 46 . Among these, Evolutionary Scale Modeling (ESM) models have shown particular promise in capturing structural, functional, and physico-chemical properties (including antibody specificity) without requiring explicit structural data 47 - 50 . ESM models, such as ESM-1v, encode sequences into high-dimensional embeddings that reflect residue context, conservation, and evolutionary information, which are all factors known to influence antibody behaviour. These features make ESM models well-suited for predicting complex properties like non-specificity, which can arise from subtle sequence-dependent effects not easily captured by traditional descriptors. By applying ESM models to antibody variable regions, we aim to harness its representation power to identify sequence signatures of non-specific binding and improve early-stage developability assessment. Download figure Open in new tab Figure 1. Performance Evaluation of Machine Learning Models for Predicting Antibody Non-Specificity. (A) Schematic workflow of the study. Publicly available datasets containing antibody sequences were used. These sequences were annotated andembedded using sequence-based biophysical descriptors and protein language models (PLMs). Different ML models were trained and evaluated using k-fold cross-validation, sensitivity-specificity analysis, and external datasets. (B) Histogram showing the distribution of antibody sequences based on the number of flags in the Boughter dataset. Sequences are categorized as Influenza reactive (blue), HIV non-reactive (red), and HIV reactive (light blue). (C) Bar plots showing the validation performance (k-Fold CV and Leave-One Family-Out) for a top performing ML algorithm (LogisticReg) and PLMs across various validation schemes (k-fold CV and leave-one-family-out). (D) Bar plot of 10-fold CV accuracy for different antibody sequences embedded by top performing language model (ESM 1v mean-mode). (E) Histogram of predicted probabilities of antibody non-specificity using the VH-based Logistic Regression (ESM 1v) model for the Jain dataset. Antibodies are classified into non-specific (dark blue), mildly non-specific (light blue), and specific (red) categories. (F) Boxplot comparing the predicted non-specificity probabilities for antibodies across different datasets using VH-based Logistic Regression (ESM 1v). The boxplot displays the median, interquartile range, and outliers, with significant differences indicated by SCC and p-values (*** indicate p-value <0.001). Besides testing which encoding provided the best prediction performance, one aspect of the study was to identify which part of the antibody contributes to non-specificity. Furthermore, to gain biophysical insight, sequence-based biophysical descriptors were analysed to support the predictive models. Our results indicate that the computational models that we considered enable the prediction of non-specific interactions that can be used to guide the design and selection of antibodies with improved specificity and efficacy. 2. Results & Discussion 2.1 Public antibody data Four different datasets were retrieved from public sources; (i) a curated dataset of >1000 mouse IgA, influenza-reactive and HIV-reactive antibodies with their respective non-specificity flag from ELISA with a panel of common antigens 38 , (ii) 137 clinical-stage IgG1-formatted antibodies with their respective non-specificity flag from ELISA with a panel of common antigens 15 , (iii) 398 antibodies, originating from naïve, IgG memory and long-lived plasma cells, with their respective poly-specific reagent score 51 , and (iv) 140 000 nanobody (Nb) clones assessed by the PSR assay from a naïve Nb library 39 . These four datasets are referred to as the Boughter, the Jain, the Shehata and the Harvey datasets, respectively.As therapeutic antibodies are engineered to be closely related to human antibodies to avoid immunogenic responses, it is important to exploit human antibody data for development of optimal ML models. As the Boughter dataset partly consists of mouse IgA antibodies, the sequence similarity of these mouse antibodies was compared to the human antibodies to ensure that there are not too large sequence differences within the dataset. The mouse IgA antibodies appear to differ mostly in the H/L-CDRs ( Figure S1B ). Another notable difference within the Boughter dataset is that the mouse IgA antibodies have a slightly shorter H-CDR3 loop relative to the human antibodies ( Figure S1C ). Ideally, training data sets for classification ML-Models should be balanced when it comes to positive and negative data points. The distribution of non-specificity for the three datasets is visualised in Figure 1B . The Boughter and the Jain datasets are relatively balanced in terms of specific (zero flags), mildly non-specific (1-3 flags) and non-specific (>4 flags) antibodies, while the Shehata dataset is unbalanced, with 7 out of 398 antibodies characterised as non-specific only. In this study, the most balanced dataset (i.e. Boughter one) was selected for training of ML models, while the remining three (i.e. Jain, Shehata and Harvey, which consists exclusively of VHH sequences) were used for testing. 2.2 Protein language models enable the representation of antibody non-specificity Following the study original study 38 , the Boughter dataset was first parsed into two groups: specific (0 flags) and non-specific group (>3 flags), leaving out the mildly non-specific antibodies (1-3 flags) ( Figure 1A ). The amino acid sequences of the parsed dataset were then annotated in the CDRs, and various fragments of the antibody sequences were embedded into vectors representing their physico-chemical and structural properties (i.e. ESM 1b, ESM 1v, ESM 2, Protbert bfd, AntiBERTy, and AbLang2). This procedure resulted into the training of 12 different antibody fragment-specific binary classification models were trained (see Table 4 ). The performance of all the generated models from 10-fold cross-validation (CV) can be seen in Figure S2-S8 . Overall, all of the protein language models (PLMs) performed well with 66-71% 10-fold CV accuracy, including the antibody-specific ones AntiBERTy and AbLang2 ( Figure 1C ). These deep learning models were trained on large datasets of protein sequences in the million-to-billion range, encoding protein sequences into vectors for representation of their physiological properties, remote homology, and secondary/tertiary structure. PLMs were originally developed for the prediction of protein contacts and structure 52 , 53 . Going forward, the Evolutionary Scale Modelling (ESM) 1v was selected as the embedder of choice for this study. 2.3 The highest PLM-based predictability is achieved by encoding the VH domain An overview of the different antibody fragment-specific models based on the ESM 1v embedder is shown in Figure 1D . Highest predictability (71% 10-fold CV accuracy of non-specificity) was obtained for the models trained on the VH and H-CDRs sequences. These results suggest that the non-specificity primarily originates from the VH domain, with main contributions from the H-CDR loops. When looking at the models based on the individual H-CDR loops, the order of low-to-high predictability of non-specificity follows H-CDR2, H-CDR1 and H-CDR3 ( Figure 1D ). H-CDR3 has the highest predictability of non-specificity among all the H-CDR loops. The importance of the H-CDR3 loop for non-specificity is in agreement with Guthmiller et al . 35 , who showed by using MD simulation that the flexibility of the H-CDR3 loop plays an important role in the non-specific behaviour of antibodies. Accuracy of around 70% was consistently observed across 3, 5 and 10-Fold CV for the top performing models ( Figure 1C ), and similar performance obtained for sensitivity and specificity. Moreover, when looking at the predictability of one antibody family to another, the overall accuracy was consistently above >60% for the Leave-One Family-Out validations. However, when comparing sensitivity and specificity, classifiers trained on human antibodies performed poorly when tested on mouse antibodies. This is not surprising as mouse and human antibodies have notable sequence differences, such as mouse IgA having a shorter H-CDR3 and larger sequence differences in the CDRs relative to human antibodies ( Figure S1B ). Moreover, classifiers trained on mouse IgA and HIV reactive antibodies perform well across all evaluation metrics (accuracy, sensitivity and specificity) when tested on Influenza reactive antibodies, while classifiers trained on mouse IgA and Influenza reactive antibodies seem to be better in predicting non-specific HIV reactive antibodies than specific ones. 2.4 Classification probability of non-specificity against non-specificity ELISA flags mimics regression behaviour A prediction probability of non-specificity was computed in addition to the binary output from the binary classification models. When comparing the prediction probability of non-specificity for all the antibodies from the Boughter dataset (test antibodies sampled from the 10-Fold CV and the mildly non-specific antibodies), three distinct distributions of prediction probabilities for the specific, mildly non-specific and non-specific antibody groups appeared ( Figure 1E ). The premise that non-specificity is not a binary property is exemplified by the overlaps between the distributions. This illustrates that the prediction probability can be used beyond the binary output to assess antibodies of varying degree of non-specificity. When comparing those to non-specificity ELISA flags, a significant regression-like behaviour (SCC 0.43) was observed for one of the top performing classifiers, ESM 1v mean-mode VH-based LogisticReg model ( Figure 1F ) . The prediction probability for non-specificity followed an uptrend when compared to the non-specificity ELISA flags. An exception to this trend was the antibodies with seven non-specificity ELISA flags, as those were exclusively mouse IgA antibodies (differences discussed in previous section). 2.5 The isoelectric point: a key biophysical driver of non-specificity To gain insight into the biophysical origins of antibody non-specificity, a set of 68 sequence descriptors ( Table S1 ) was computed for the parsed Boughter dataset. These descriptors encompass a wide range of biophysical properties derived from the antibodies sequence, including theoretical isoelectric point (pI), secondary structure propensity, and hydrophobicity. To assess the presence of redundancy among the descriptors, we constructed a Spearman’s correlation matrix, which revealed that several descriptors, such as hydrophobicity descriptors, exhibited strong correlation among each other (SCC > 0.5, Figure S9 ), thus indicating redundancy. All the descriptors were ranked according to the absolute logistic regression coefficients ( Table S2 ), whereafter top 25 descriptors were selected, and used for training of a VH-based LogisticReg model. The in-depth analysis of the importance of the 25 descriptors is shown in four different plots in Figure 2A : Download figure Open in new tab Figure 2. Analysis of Descriptor Importance and Model Performance for VH-Based Logistic Regression. (A) Analysis of descriptor importance using various metrics for the VH-based Logistic Regression model; (first panel) Logistic regression coefficients indicating the relative importance of different features, (second panel) permutation importance showing the change in 10-fold CV accuracy when each descriptor is permuted, (third panel) 10-fold CV accuracy of models based on single descriptors, and (fourth panel) 10-fold CV accuracy of leave-one-feature-out models compared to the mean accuracy of model using the top 25 descriptors. (B) Heatmap displaying the Spearman’s correlation coefficient (SCC) between the top 25 descriptors selected based on highest Logistic Regression coefficient in model with all descriptors. The dendrogram shows hierarchical clustering of descriptors based on their SCC. (C) Bar plot comparing the 10-fold CV accuracy of different models in predicting antibody non-specificity across various validation schemes: k-fold CV and leave-one-family-out validation. Models compared include VH-based sequences embedded by ESM 1v, all descriptors, PCA with 3, 5 and 10 components, and top 2, 3, 4, and 5 descriptors. Sensitivity and specificity bar plots can be found in Figure S12 . The first plot shows the LogisticReg coefficients, indicating the relative importance of each descriptor. Notably, Disorder_Propensity_DisProt, Aggrescan_a4v, and theoretical pI show significant positive coefficients, suggesting that they are strong drivers of non-specificity. The second plot displays the permutation importance, highlighting the change in accuracy when each descriptor is permuted. Descriptors like theoretical pI, bulkiness, Hplc_Hfba_retention and Polarity_Zimmerman demonstrate substantial decrease in accuracy upon permutation. The third plot illustrates the accuracy of models based on single descriptors to underscore their individual predictive power. Theoretical pI resulted in the highest accuracy compared to the other descriptors, confirming its critical role in the prediction of non-specificity. The fourth plot shows the leave-one-feature-out accuracy, revealing how the exclusion of each descriptor affects the overall model performance. Most of the descriptors result in a minimal drop in accuracy, indicating that the model performance remains unaffected when a certain descriptor is left out. This can be explained by that there remains a certain level of redundancy among the 25 descriptors, e.g. theoretical pI appears to be negatively correlated with Polarity_Zimmerman, according to the Spearman’s correlation matrix in Figure 2B To further narrow down the redundancy, the 25 descriptors were tested in all possible combinations of 2, 3, 4 and 5 descriptors for training of new LogisticReg models. The results of the top descriptors from this analysis are shown in Table 1 . The results indicate that, among the top 5 descriptors, the theoretical pI appear to be the most important driver for non-specificity. This conclusion is also supported by the frequency of this particular descriptor among the top models ( Figure S10 ). Additionally, as a parallel check, we performed Principal Component Analysis (PCA) for dimensionality reduction and feature selection. The primary objective of this study was again to address multicollinearity among the descriptors and to identify the most significant features contributing to the variance in the dataset. We thus evaluated the performance of the LogisticReg models trained on the 3, 5, and 10 principal components identified by PCA. In agreement with our previous findings regarding the theoretical pI, the presence of this descriptor among the selected features significantly influenced the model performance, particularly when it was included in PCA 5 and 10 components (see Figure 2C ). The importance of theoretical pI was captured at ≥5 components, as seen by the magnitudes of Eigenvalues of theoretical pI for PCA 5 as compared to PCA 1-4 (see Figure S11 ). The isoelectric point is known to influence PKPD behaviour/clearance of antibodies 54 , 55 . Altogether, a comparison between the PLM-based and descriptor-based ML models in terms of accuracy of across different validation schemes is shown in Figure 2C . The results indicate that the ESM 1v model consistently achieves high accuracy across all validation schemes. The VH-based Logistic Regression model using all descriptors also performs well, though slightly lower than the ESM 1v model. Notably, the PCA-based models show comparable performance, demonstrating the effectiveness of dimensionality reduction in maintaining model accuracy. Interestingly, the VH-based Logistic Regression models using the top descriptors, 2, 3, 4, and 5 combinations ( Table 1 ), exhibit robust performance. This finding suggests that a smaller subset of key descriptors can achieve similar predictive power as using the full set of descriptors, highlighting the potential for model simplification without compromising accuracy. View this table: View inline View popup Download powerpoint Table 1. Top 2, 3, 4 and 5 combined VH-based sequence descriptors. 2.6 VH-based LogisticReg classification model is applicable to clinical-stage therapeutic antibodies To show applicability of the non-specificity classification model on therapeutic antibodies, the ESM 1v mean-mode VH-based LogisticReg model was tested on the Jain dataset. As in the Boughter dataset, the Jain dataset was parsed into two groups, specific (0 flags) and non-specific (>3 flags), leaving out the mildly non-specific antibodies (1-3 flags). An accuracy of 69% was obtained for the parsed Jain dataset (see confusion matrix in Figure S14A ). This value is comparable to the mean accuracy of 71% obtained for the same classifier across 3, 5 and 10-Fold CV for the parsed Boughter dataset. Moreover, as in the case of the Boughter dataset, a similar distribution of prediction probability of non-specificity was obtained for the full Jain dataset, and it appears to mimic regression-like behaviour when compared to the non-specificity ELISA flags, although to a slightly weaker extent ( Figure 3A and S13 ). The same trend can be observed for the top 5 descriptors model ( Figure 3C ). The overall performance of the classifier on the Jain dataset illustrates that it can be applied to therapeutic antibodies. Download figure Open in new tab Figure 3. Logistic Regression Models Predicting Antibody Non-Specificity Across Different Datasets. (A-F) Distributions of predicted probabilities of antibody non-specificity for three different datasets using two logistic regression models: predictions for the Jain dataset (A,B), for the Shehata dataset (C,D), and for the Harvey dataset (E,F). (A, C, and E) depict results from the ESM 1v VH-based logistic regression model, while (B, D, and F) depict results from the top 5 descriptors VH-based logistic regression model. For each dataset, antibodies are classified into specific, mildly non-specific (only in the Jain dataset), and non-specific categories, represented by different colours. 2.7 Antibodies characterised by the PSR assay appear to be on a different non-specificity spectrum than that from the non-specificity ELISA assay During recent years, alternative assays to ELISA have been developed to meet the demand of high-throughput screening during drug discovery, and such one is the poly-specific reagent (PSR) assay 37 , where antibodies displayed on the surface of yeast cells are counter-selected when non-specifically bound to soluble membrane protein in a flow cytometry-setup. Several studies have been reported using this assay for assessing antibody non-specificity 15 , 17 , 51 . Recently, Harvey and co-authors 39 developed a one-hot LogisticReg model based on >140 000 clones assessed by the PSR assay from a naïve Nb library. They found a significant correlation of the PSR with the gold-standard ELISA based on six Nbs. To find out whether our ESM 1v mean-mode VH-based LogisticReg model can extend its applicability further to the non-specificity scored by the PSR assay, the Shehata dataset and the VH-based Nb dataset by Harvey and co-authors 39 , here referred to as the Harvey dataset, were tested. The classifier did not appear to separate the PSR-scored specific and non-specific antibodies well. All the specific PSR-scored antibodies of the Shehata dataset were distributed along the entire prediction probability scale, while the few non-specific ones were on the probability end towards higher non-specificity ( Figure 3C,D ). A similar forecast was observed for the Harvey dataset; all the specific PSR-scored Nbs resulted in a broad probability distribution, while the non-specific PSR-scored ones resulted in a narrower probability distribution towards higher non-specificity ( Figure 3E,F ). Thus, the classifier appears to be better at predicting non-specific PSR-scored antibodies, than specific PSR-scored antibodies. This result suggests that the spectrum of non-specificity from the PSR assay is different than the one from the non-specificity ELISA assay, of which the classifier is trained on. Thus, a specific antibody classified by the PSR assay may necessarily not translate into a specific antibody classified by the non-specificity ELISA assay. The specific PSR-scored Nbs could partly consist of mildly non-specific clones in addition to specific ones, thus resulting in this broad probability distribution. An interesting remark can be made about the distributions of predicted probabilities obtained from the two different LogisticReg models tested on the Harvey dataset in Figure 3E,F . The ESM 1v VH-based LogisticReg model produces a more uniform distribution of predicted probabilities across the dataset, while the top 5 descriptors VH-based LogisticReg model exhibits a clear biphasic distribution. It is no surprise that this bimodal pattern closely resembles the distribution of pI ( Figures S15-S18 ), as this descriptor is the main driver of non-specificity in the top 2, 3, 4 and 5 descriptors LogisticReg models ( Table 1 and Figure S10 ). Nonetheless, the distribution of non-specific antibodies in the Harvey dataset appears to exclusively be of high pI (>8) according to Figure S18A . The distributions of the other descriptors do not appear to differ significantly between specific and non-specific antibodies see ( Figures S15-S18 ). The distinct separation suggests that the pI plays a crucial role in differentiating between specific and non-specific antibodies. 2.8 VH-based LogisticReg models performs on par or better than existing predictors The ESM 1v VH-based LogisticReg model can be compared to two existing predictors in the literature - the predictors reported in the Boughter et al . study 38 , and the Harvey et al . study 39 . Boughter and colleagues stated that while no notable difference could be observed between specific and non-specific antibodies in both gene usage level and amino acid-usage level in CDRs, the positional context of biophysical properties can show the differences. They showed to be successful in developing a binary classifier based on a position-sensitive biophysical matrix with accuracy up to 75%. Their reported performance is on par with the achieved performance of the PLM-based classifier (ESM 1v VH-based LogisticReg), which was trained on the same data (parsed Boughter dataset). Furthermore, Harvey and colleagues developed a one-hot LogisticReg model based on >140 000 clones assessed by the PSR assay from a naïve Nb library with an accuracy >80%. Using their published web-based predictor 56 , we tested its performance on the Boughter, Jain, and Shehata datasets. The results show that the Harvey predictor does not separate well the different antibody groups in the Boughter dataset ( Figure S19A , B ), with overlapping distributions of prediction scores. Similarly, the Jain and the Shehata datasets demonstrate significant overlap between specific and non-specific antibodies ( Figure S19C-F ), indicating some limitations for the Harvey method in predicting non-specificity of antibodies as compared to Nbs. 2.9 Prediction of non-specificity in antibody drug development programs Consequences of non-specificity in the clinic Non-specific binding of therapeutic antibodies can lead to significant adverse effects in the clinic. Such antibodies can bind to structurally unrelated off-targets and thereby potentially result in unwanted toxicity 57 or reduced efficacy 23 . They can also interact with tissues like subcutis and thereby result in faster clearance via pinocytosis independently of FcRn 54 , 55 , 58 . Ultimately, non-specificity can compromise the safety and efficacy of therapeutic antibodies, potentially resulting in clinical trial failures and increased development costs. Thus, early-stage prediction of non-specificity, such as during selection and optimisation stage, is essential to reduce the risk of failing at late-stage during clinical trials. Otherwise, the further into the development program, the harder it becomes to allow additional protein engineering to mitigate biophysical liabilities, as new in vitro and in vivo data must be reproduced. A powerful strategy to address this problem is to combine in silico prediction with in vitro developability assessment. To identify and flag non-specific antibodies, we propose a combined strategy that integrates in silico prediction models with traditional in vitro developability assessments during the lead optimization stage in the drug development process. This hybrid approach leverages the strengths of computational predictions and experimental validations, ensuring the selection of lead candidates with high developability potential. 3. Conclusions In this study, we developed ML models to predict the non-specificity of antibodies, utilizing both PLMs and biophysical properties to embed antibody sequences. In agreement with previous reports on different datasets 59 , our results indicate that the VH domain, particularly the H-CDR loops, as the main contributor to non-specificity, and that the biophysical parameter pI is a key biophysical driver of non-specificity. The resulting computational models enable the prediction of non-specific interactions of antibodies with accuracy of 71% in 10-fold CV, thus providing a valuable tool to guide the design and selection of monoclonal antibodies with improved specificity and efficacy. These findings have important implications for the development of safe and efficacious lead candidates with high developability potential. 4. Methods 4.1 Data sources All antibody datasets used in this study were retrieved from public sources, as well as upon request from Prof. Debora Marks and Prof. Andrew Kruse (Harvey dataset), and Dr. Tushar Jain and Dr. Dane Wittrup (Jain dataset). A list of the datasets and their corresponding sources are reported Table 2 . View this table: View inline View popup Download powerpoint Table 2. List of public antibody datasets with their corresponding size, non-specificity assay and reference. 4.2 Python programming All coding was performed in Python using Spyder IDE and Jupyter Notebook (Anaconda software distribution) 60 , and a list of used Python modules is reported in Table 3 . View this table: View inline View popup Download powerpoint Table 3. List of software and Python modules. View this table: View inline View popup Download powerpoint Table 4. Type and description of sequence input for the binary classification models. 4.3 Training and validation of binary classification models First, the Boughter dataset was parsed into three groups as previously done in 38 : specific group (0 flags), mildly poly-reactive group (1-3 flags) and poly-reactive group (>3 flags). The primary sequences were annotated in the CDRs using ANARCI following the IMGT numbering scheme. Following this, 16 different antibody fragment sequences were assembled and embedded by three state-of-the-art protein language models (PLMs), ESM 1v 52 , Protbert bfd 53 , and AbLang2 44 , for representation of the physico-chemical properties and secondary/tertiary structure ( Table 4 ). For the embeddings from the PLMs, mean (average of all token vectors) was used. The vectorised embeddings were served as features for training of binary classification models (e.g. LogisticReg, RandomForest, GaussianProcess, GradeintBoosting and SVM algorithms) for non-specificity (class 0: specific group, and class 1: poly-reactive group). The mildly poly-reactive group was left out from the training of the models. The trained classification models were validated by (i) 3, 5 and 10-Fold cross-validation (CV), (ii) Leave-One Family-Out validation, e.g. training on HIV and Influenza reactive antibodies, while testing on mouse IgA antibodies, (iii) comparing probability of predicted poly-reactive class to true class, and (iv) testing on the Jain dataset. The evaluation metrics included accuracy, sensitivity and specificity ( Eqs. 1 - 3 ). Where TP is true positive (true poly-reactive), TN is true negative (true specific), FP is false positive (false poly-reactive), FN is false negative (false specific), P is all positives (all poly-reactive), and N is all negatives (all specific). 4.4 Data availability Data and models used in this publication are available at our GitHub repository: https://github.com/NovoNordisk-OpenSource/ML-predictions-of-Antibody-Non-Specificity . Funding Financial support from Novo Nordisk A/S is acknowledged. P.S. is a Royal Society University Research Fellow (grant no. URF\R1\201461) and acknowledges funding from UK Research and Innovation (UKRI), and Engineering and Physical Sciences Research Council (grant no. EP/X024733/1). The authors declare no competing financial interest. Acknowledgements We would like to thank Prof. Debora Marks and Prof. Andrew Kruse for kindly sharing the dataset on 140 000 Nb clones assessed by the PSR assay from 39 , as well as Dr. Tushar Jain and Dr. Dane Wittrup for kindly sharing the dataset on individual ELISA readouts from 15 . We would also like to thank Mauricio Augilar Rangel, Dillon Rinauro and Ross Taylor from the University of Cambridge for their valuable discussions to this work. ChatGPT-5 mini (OpenAI) is acknowledged for use as a tool to draft the figure legends of this paper. Funder Information Declared UK Research and Innovation, https://ror.org/001aqnf71 Engineering and Physical Sciences Research Council Novo Nordisk (Denmark), https://ror.org/0435rc536 Footnotes The manuscript have been revised, references updated and link to access data and models have been included. References 1. ↵ Mullard A. FDA approves 100th monoclonal antibody product . Nat Rev Drug Discov . 2021 ; 20 : 491 – 495 . doi: 10.1038/d41573-021-00079-7 . PMID: 33953368 . OpenUrl CrossRef PubMed 2. ↵ Kaplon H , Chenoweth A , Crescioli S , Reichert JM . Antibodies to watch in 2022 . MAbs . 2022 ; 14 : 2014296 . doi: 10.1080/19420862.2021.2014296 . PMID: 35030985 . OpenUrl CrossRef PubMed 3. ↵ Lu RM , Hwang YC , Liu IJ , Lee CC , Tsai HZ , Li HJ , Wu HC . Development of therapeutic antibodies for the treatment of diseases . J Biomed Sci . 2020 ; 27 : 1 . doi: 10.1186/s12929-019-0592-z . PMID: 31894001 . OpenUrl CrossRef PubMed 4. ↵ Pedrioli A , Oxenius A. Single B cell technologies for monoclonal antibody discovery . Trends Immunol . 2021 ; 42 : 1143 – 1158 . doi: 10.1016/j.it.2021.10.008 . PMID: 34743921 . OpenUrl CrossRef PubMed 5. ↵ Tabasinezhad M , Talebkhan Y , Wenzel W , Rahimi H , Omidinia E , Mahboudi F. Trends in therapeutic antibody affinity maturation: From in-vitro towards next-generation sequencing approaches . Immunol Lett . 2019 ; 212 : 106 – 113 . doi: 10.1016/j.imlet.2019.06.009 . PMID: 31247224 . OpenUrl CrossRef PubMed 6. ↵ Chidyausiku TM , Mendes SR , Klima JC , Nadal M , Eckhard U , Roel-Touris J , Houliston S , Guevara T , Haddox HK , Moyer A , et al. De novo design of immunoglobulin-like domains . Nat Commun . 2022 ; 13 : 5661 . doi: 10.1038/s41467-022-33004-6 . PMID: 36192397 . OpenUrl CrossRef PubMed 7. ↵ Watson JL , Juergens D , Bennett NR , Trippe BL , Yim J , Eisenach HE , Ahern W , Borst AJ , Ragotte RJ , Milles LF , et al. De novo design of protein structure and function with RFdiffusion . Nature . 2023 ; 620 : 1089 – 1100 . doi: 10.1038/s41586-023-06415-8 . PMID: 37433327 . OpenUrl CrossRef PubMed 8. ↵ Boder ET , Midelfort KS , Wittrup KD . Directed evolution of antibody fragments with monovalent femtomolar antigen-binding affinity . Proc Natl Acad Sci U S A . 2000 ; 97 : 10701 – 10705 . doi: 10.1073/pnas.170297297 . PMID: 10984501 . OpenUrl Abstract / FREE Full Text 9. ↵ Igawa T , Ishii S , Tachibana T , Maeda A , Higuchi Y , Shimaoka S , Moriyama C , Watanabe T , Takubo R , Doi Y , et al. Antibody recycling by engineered pH-dependent antigen binding improves the duration of antigen neutralization . Nat Biotechnol . 2010 ; 28 : 1203 – 1207 . doi: 10.1038/nbt.1691 . PMID: 20953198 . OpenUrl CrossRef PubMed 10. ↵ Klaus T , Deshmukh S. pH-responsive antibodies for therapeutic applications . J Biomed Sci . 2021 ; 28 : 11 . doi: 10.1186/s12929-021-00709-7 . PMID: 33482842 . OpenUrl CrossRef PubMed 11. ↵ Kamata-Sakurai M , Narita Y , Hori Y , Nemoto T , Uchikawa R , Honda M , Hironiwa N , Taniguchi K , Shida-Kawazoe M , Metsugi S , et al. Antibody to CD137 Activated by Extracellular Adenosine Triphosphate Is Tumor Selective and Broadly Effective In Vivo without Systemic Immune Activation . Cancer Discov . 2021 ; 11 : 158 – 175 . doi: 10.1158/2159-8290.CD-20-0328 . PMID: 32847940 . OpenUrl Abstract / FREE Full Text 12. ↵ Elshiaty M , Schindler H , Christopoulos P. Principles and Current Clinical Landscape of Multispecific Antibodies against Cancer . Int J Mol Sci . 2021 ; 22 . doi: 10.3390/ijms22115632 . PMID: 34073188 . OpenUrl CrossRef PubMed 13. ↵ Ostergaard H , Lund J , Greisen PJ , Kjellev S , Henriksen A , Lorenzen N , Johansson E , Roder G , Rasch MG , Johnsen LB , et al. A factor VIIIa-mimetic bispecific antibody, Mim8, ameliorates bleeding upon severe vascular challenge in hemophilia A mice . Blood . 2021 ; 138 : 1258 – 1268 . doi: 10.1182/blood.2020010331 . PMID: 34077951 . OpenUrl CrossRef PubMed 14. ↵ Starr CG , Tessier PM . Selecting and engineering monoclonal antibodies with drug-like specificity . Curr Opin Biotechnol . 2019 ; 60 : 119 – 127 . doi: 10.1016/j.copbio.2019.01.008 . PMID: 30822699 . OpenUrl CrossRef PubMed 15. ↵ Jain T , Sun T , Durand S , Hall A , Houston NR , Nett JH , Sharkey B , Bobrowicz B , Caffry I , Yu Y , et al. Biophysical properties of the clinical-stage antibody landscape . Proc Natl Acad Sci U S A . 2017 ; 114 : 944 – 949 . doi: 10.1073/pnas.1616408114 . PMID: 28096333 . OpenUrl Abstract / FREE Full Text 16. ↵ Avery LB , Wade J , Wang M , Tam A , King A , Piche-Nicholas N , Kavosi MS , Penn S , Cirelli D , Kurz JC , et al. Establishing in vitro in vivo correlations to screen monoclonal antibodies for physicochemical properties related to favorable human pharmacokinetics . MAbs . 2018 ; 10 : 244 – 255 . doi: 10.1080/19420862.2017.1417718 . PMID: 29271699 . OpenUrl CrossRef PubMed 17. ↵ Kelly RL , Sun T , Jain T , Caffry I , Yu Y , Cao Y , Lynaugh H , Brown M , Vasquez M , Wittrup KD , et al. High throughput cross-interaction measures for human IgG1 antibodies correlate with clearance rates in mice . MAbs . 2015 ; 7 : 770 – 777 . doi: 10.1080/19420862.2015.1043503 . PMID: 26047159 . OpenUrl CrossRef PubMed 18. ↵ Kraft TE , Richter WF , Emrich T , Knaupp A , Schuster M , Wolfert A , Kettenberger H. Heparin chromatography as an in vitro predictor for antibody clearance rate through pinocytosis . MAbs . 2020 ; 12 : 1683432 . doi: 10.1080/19420862.2019.1683432 . PMID: 31769731 . OpenUrl CrossRef PubMed 19. Dobson CL , Devine PW , Phillips JJ , Higazi DR , Lloyd C , Popovic B , Arnold J , Buchanan A , Lewis A , Goodman J , et al. Engineering the surface properties of a human monoclonal antibody prevents self-association and rapid clearance in vivo . Sci Rep . 2016 ; 6 : 38644 . doi: 10.1038/srep38644 . PMID: 27995962 . OpenUrl CrossRef PubMed 20. Datta-Mannan A , Lu J , Witcher DR , Leung D , Tang Y , Wroblewski VJ . The interplay of non-specific binding, target-mediated clearance and FcRn interactions on the pharmacokinetics of humanized antibodies . MAbs . 2015 ; 7 : 1084 – 1093 . doi: 10.1080/19420862.2015.1075109 . PMID: 26337808 . OpenUrl CrossRef PubMed 21. Datta-Mannan A , Thangaraju A , Leung D , Tang Y , Witcher DR , Lu J , Wroblewski VJ . Balancing charge in the complementarity-determining regions of humanized mAbs without affecting pI reduces non-specific binding and improves the pharmacokinetics . MAbs . 2015 ; 7 : 483 – 493 . doi: 10.1080/19420862.2015.1016696 . PMID: 25695748 . OpenUrl CrossRef PubMed 22. ↵ Igawa T , Tsunoda H , Tachibana T , Maeda A , Mimoto F , Moriyama C , Nanami M , Sekimori Y , Nabuchi Y , Aso Y , et al. Reduced elimination of IgG antibodies by engineering the variable region . Protein Eng Des Sel . 2010 ; 23 : 385 – 392 . doi: 10.1093/protein/gzq009 . PMID: 20159773 . OpenUrl CrossRef PubMed Web of Science 23. ↵ Bumbaca D , Wong A , Drake E , Reyes AE , 2nd . , Lin BC , Stephan JP , Desnoyers L , Shen BQ , Dennis MS . Highly specific off-target binding identified and eliminated during the humanization of an antibody against FGF receptor 4 . MAbs . 2011 ; 3 : 376 – 386 . doi: 10.4161/mabs.3.4.15786 . PMID: 21540647 . OpenUrl CrossRef PubMed Web of Science 24. ↵ Tiller KE , Li L , Kumar S , Julian MC , Garde S , Tessier PM . Arginine mutations in antibody complementarity-determining regions display context-dependent affinity/specificity trade-offs . J Biol Chem . 2017 ; 292 : 16638 – 16652 . doi: 10.1074/jbc.M117.783837 . PMID: 28778924 . OpenUrl Abstract / FREE Full Text 25. ↵ Kelly RL , L. D , Zhao J , Wittrup KD . Reduction of Nonspecificity Motifs in Synthetic Antibody Libraries . J Mol Biol . 2018 ; 430 : 119 – 130 . doi: 10.1016/j.jmb.2017.11.008 . PMID: 29183788 . OpenUrl CrossRef PubMed 26. Ausserwoger H , Schneider MM , Herling TW , Arosio P , Invernizzi G , Knowles TPJ , Lorenzen N. Non-specificity as the sticky problem in therapeutic antibody development . Nat Rev Chem . 2022 ; 6 : 844 – 861 . doi: 10.1038/s41570-022-00438-x . PMID: 37117703 . OpenUrl CrossRef PubMed 27. ↵ Cunningham O , Scott M , Zhou ZS , Finlay WJJ . Polyreactivity and polyspecificity in therapeutic antibody development: risk factors for failure in preclinical and clinical development campaigns . MAbs . 2021 ; 13 : 1999195 . doi: 10.1080/19420862.2021.1999195 . PMID: 34780320 . OpenUrl CrossRef PubMed 28. ↵ Palmer JP , Asplin CM , Clemons P , Lyen K , Tatpati O , Raghu PK , Paquette TL . Insulin antibodies in insulin-dependent diabetics before insulin treatment . Science . 1983 ; 222 : 1337 – 1339 . doi: 10.1126/science.6362005 . PMID: 6362005 . OpenUrl Abstract / FREE Full Text 29. ↵ Taplin CE , Barker JM . Autoantibodies in type 1 diabetes . Autoimmunity . 2008 ; 41 : 11 – 18 . doi: 10.1080/08916930701619169 . PMID: 18176860 . OpenUrl CrossRef PubMed Web of Science 30. ↵ Noble PW , Bernatsky S , Clarke AE , Isenberg DA , Ramsey-Goldman R , Hansen JE . DNA-damaging autoantibodies and cancer: the lupus butterfly theory . Nat Rev Rheumatol . 2016 ; 12 : 429 – 434 . doi: 10.1038/nrrheum.2016.23 . PMID: 27009542 . OpenUrl CrossRef PubMed 31. ↵ Nehring J , Schirmbeck LA , Friebus-Kardash J , Dubler D , Huynh-Do U , Chizzolini C , Ribi C , Trendelenburg M. Autoantibodies Against Albumin in Patients With Systemic Lupus Erythematosus . Front Immunol . 2018 ; 9 : 2090 . doi: 10.3389/fimmu.2018.02090 . PMID: 30333817 . OpenUrl CrossRef PubMed 32. ↵ Miyakis S , Lockshin MD , Atsumi T , Branch DW , Brey RL , Cervera R , Derksen RH , PG Deg , Koike T , Meroni PL , et al. International consensus statement on an update of the classification criteria for definite antiphospholipid syndrome (APS) . J Thromb Haemost . 2006 ; 4 : 295 – 306 . doi: 10.1111/j.1538-7836.2006.01753.x . PMID: 16420554 . OpenUrl CrossRef PubMed Web of Science 33. ↵ Poxton IR . Antibodies to lipopolysaccharide . J Immunol Methods . 1995 ; 186 : 1 – 15 . doi: 10.1016/0022-1759(95)00123-r . PMID: 7561138 . OpenUrl CrossRef PubMed Web of Science 34. ↵ Mouquet H , Scheid JF , Zoller MJ , Krogsgaard M , Ott RG , Shukair S , Artyomov MN , Pietzsch J , Connors M , Pereyra F , et al. Polyreactivity increases the apparent affinity of anti-HIV antibodies by heteroligation . Nature . 2010 ; 467 : 591 – 595 . doi: 10.1038/nature09385 . PMID: 20882016 . OpenUrl CrossRef PubMed Web of Science 35. ↵ Guthmiller JJ , Lan LY , Fernandez-Quintero ML , Han J , Utset HA , Bitar DJ , Hamel NJ , Stovicek O , Li L , Tepora M , et al. Polyreactive Broadly Neutralizing B cells Are Selected to Provide Defense against Pandemic Threat Influenza Viruses . Immunity . 2020 ; 53 : 1230 – 1244 e1235 . doi: 10.1016/j.immuni.2020.10.005 . PMID: 33096040 . OpenUrl CrossRef PubMed 36. ↵ Hotzel I , Theil FP , Bernstein LJ , Prabhu S , Deng R , Quintana L , Lutman J , Sibia R , Chan P , Bumbaca D , et al. A strategy for risk mitigation of antibodies with fast clearance . MAbs . 2012 ; 4 : 753 – 760 . doi: 10.4161/mabs.22189 . PMID: 23778268 . OpenUrl CrossRef PubMed Web of Science 37. ↵ Xu Y , Roach W , Sun T , Jain T , Prinz B , Yu TY , Torrey J , Thomas J , Bobrowicz P , Vasquez M , et al. Addressing polyspecificity of antibodies selected from an in vitro yeast presentation system: a FACS-based, high-throughput selection and analytical tool . Protein Eng Des Sel . 2013 ; 26 : 663 – 670 . doi: 10.1093/protein/gzt047 . PMID: 24046438 . OpenUrl CrossRef PubMed 38. ↵ Boughter CT , Borowska MT , Guthmiller JJ , Bendelac A , Wilson PC , Roux B , Adams EJ . Biochemical patterns of antibody polyreactivity revealed through a bioinformatics-based analysis of CDR loops . Elife . 2020 ; 9 . doi: 10.7554/eLife.61393 . PMID: 33169668 . OpenUrl CrossRef PubMed 39. ↵ Harvey EP , Shin JE , Skiba MA , Nemeth GR , Hurley JD , Wellner A , Shaw AY , Miranda VG , Min JK , Liu CC , et al. An in silico method to assess antibody fragment polyreactivity . Nat Commun . 2022 ; 13 : 7554 . doi: 10.1038/s41467-022-35276-4 . PMID: 36477674 . OpenUrl CrossRef PubMed 40. Robert PA , Akbar R , Frank R , Pavlovic M , Widrich M , Snapkov I , Slabodkin A , Chernigovskaya M , Scheffer L , Smorodina E , et al. Unconstrained generation of synthetic antibody-antigen structures to guide machine learning methodology for antibody specificity prediction . Nat Comput Sci . 2022 ; 2 : 845 – 865 . doi: 10.1038/s43588-022-00372-4 . PMID: 38177393 . OpenUrl CrossRef PubMed 41. ↵ Makowski EK , Kinnunen PC , Huang J , Wu L , Smith MD , Wang T , Desai AA , Streu CN , Zhang Y , Zupancic JM , et al. Co-optimization of therapeutic antibody affinity and specificity using machine learning models that generalize to novel mutational space . Nat Commun . 2022 ; 13 : 3788 . doi: 10.1038/s41467-022-31457-3 . PMID: 35778381 . OpenUrl CrossRef PubMed 42. ↵ Teixeira AAR , D’Angelo S , Erasmus MF , Leal-Lopes C , Ferrara F , Spector LP , Naranjo L , Molina E , Max T , DeAguero A , et al. Simultaneous affinity maturation and developability enhancement using natural liability-free CDRs . MAbs . 2022 ; 14 : 2115200 . doi: 10.1080/19420862.2022.2115200 . PMID: 36068722 . OpenUrl CrossRef PubMed 43. ↵ Hie BL , Shanker VR , Xu D , Bruun TUJ , Weidenbacher PA , Tang S , Wu W , Pak JE , Kim PS . Efficient evolution of human antibodies from general protein language models . Nat Biotechnol . 2024 ; 42 : 275 – 283 . doi: 10.1038/s41587-023-01763-2 . PMID: 37095349 . OpenUrl CrossRef PubMed 44. ↵ Olsen TH , Moal IH , Deane CM . AbLang: an antibody language model for completing antibody sequences . Bioinform Adv . 2022 ; 2 : vbac046 . doi: 10.1093/bioadv/vbac046 . PMID: 36699403 . OpenUrl CrossRef PubMed 45. Shanker VR , Bruun TUJ , Hie BL , Kim PS . Unsupervised evolution of protein and antibody complexes with a structure-informed language model . Science . 2024 ; 385 : 46 – 53 . doi: 10.1126/science.adk8946 . PMID: 38963838 . OpenUrl CrossRef PubMed 46. ↵ Singh R , Im C , Qiu Y , Mackness B , Gupta A , Joren T , Sledzieski S , Erlach L , Wendt M , Fomekong Nanfack Y , et al. Learning the language of antibody hypervariability . Proc Natl Acad Sci U S A . 2025 ; 122 : e2418918121 . doi: 10.1073/pnas.2418918121 . PMID: 39793083 . OpenUrl CrossRef PubMed 47. ↵ Mason DM , Friedensohn S , Weber CR , Jordi C , Wagner B , Meng SM , Ehling RA , Bonati L , Dahinden J , Gainza P , et al. Optimization of therapeutic antibodies by predicting antigen specificity from antibody sequence via deep learning . Nat Biomed Eng . 2021 ; 5 : 600 – 612 . doi: 10.1038/s41551-021-00699-9 . PMID: 33859386 . OpenUrl CrossRef PubMed 48. Wang Y , Lv H , Teo QW , Lei R , Gopal AB , Ouyang WO , Yeung YH , Tan TJC , Choi D , Shen IR , et al. An explainable language model for antibody specificity prediction using curated influenza hemagglutinin antibodies . Immunity . 2024 ; 57 : 2453 – 2465 e2457 . doi: 10.1016/j.immuni.2024.07.022 . PMID: 39163866 . OpenUrl CrossRef PubMed 49. Wang M , Patsenker J , Li H , Kluger Y , Kleinstein SH . Supervised fine-tuning of pre-trained antibody language models improves antigen specificity prediction . PLoS Comput Biol . 2025 ; 21 : e1012153 . doi: 10.1371/journal.pcbi.1012153 . PMID: 40163503 . OpenUrl CrossRef PubMed 50. ↵ Yu X , Vangjeli K , Prakash A , Chhaya M , Stanley SJ , Cohen N , Huang L. Protein language models enable prediction of polyreactivity of monospecific, bispecific, and heavy-chain-only antibodies . Antib Ther . 2024 ; 7 : 199 – 208 . doi: 10.1093/abt/tbae012 . PMID: 39036071 . OpenUrl CrossRef PubMed 51. ↵ Shehata L , Maurer DP , Wec AZ , Lilov A , Champney E , Sun T , Archambault K , Burnina I , Lynaugh H , Zhi X , et al. Affinity Maturation Enhances Antibody Specificity but Compromises Conformational Stability . Cell Rep . 2019 ; 28 : 3300 – 3308 e3304 . doi: 10.1016/j.celrep.2019.08.056 . PMID: 31553901 . OpenUrl CrossRef PubMed 52. ↵ Rives A , Meier J , Sercu T , Goyal S , Lin Z , Liu J , Guo D , Ott M , Zitnick CL , Ma J , et al. Biological structure and function emerge from scaling unsupervised learning to 250 million protein sequences . Proc Natl Acad Sci U S A . 2021 ; 118 . doi: 10.1073/pnas.2016239118 . PMID: 33876751 . OpenUrl Abstract / FREE Full Text 53. ↵ Elnaggar A , Heinzinger M , Dallago C , Rehawi G , Wang Y , Jones L , Gibbs T , Feher T , Angerer C , Steinegger M , et al. ProtTrans: Toward Understanding the Language of Life Through Self-Supervised Learning . IEEE Trans Pattern Anal Mach Intell . 2022 ; 44 : 7112 – 7127 . doi: 10.1109/TPAMI.2021.3095381 . PMID: 34232869 . OpenUrl CrossRef PubMed 54. ↵ Kelly RL , Yu Y , Sun T , Caffry I , Lynaugh H , Brown M , Jain T , Xu Y , Wittrup KD . Target-independent variable region mediated effects on antibody clearance can be FcRn independent . MAbs . 2016 ; 8 : 1269 – 1275 . doi: 10.1080/19420862.2016.1208330 . PMID: 27610650 . OpenUrl CrossRef PubMed 55. ↵ Li B , Tesar D , Boswell CA , Cahaya HS , Wong A , Zhang J , Meng YG , Eigenbrot C , Pantua H , Diao J , et al. Framework selection can influence pharmacokinetics of a humanized therapeutic antibody through differences in molecule charge . MAbs . 2014 ; 6 : 1255 – 1264 . doi: 10.4161/mabs.29809 . PMID: 25517310 . OpenUrl CrossRef PubMed 56. ↵ Web-based predictor of nanobody non-specificity : http://18.224.60.30:3000/. Date [accessed 17-Dec-2025 ]. 57. ↵ Finlay WJJ , Coleman JE , Edwards JS , Johnson KS . Anti-PD1 ‘SHR-1210’ aberrantly targets pro-angiogenic receptors and this polyspecificity can be ablated by paratope refinement . MAbs . 2019 ; 11 : 26 – 44 . doi: 10.1080/19420862.2018.1550321 . PMID: 30541416 . OpenUrl CrossRef PubMed 58. ↵ Boswell CA , Tesar DB , Mukhyala K , Theil FP , Fielder PJ , Khawli LA . Effects of charge on antibody tissue distribution and pharmacokinetics . Bioconjug Chem . 2010 ; 21 : 2153 – 2163 . doi: 10.1021/bc100261d . PMID: 21053952 . OpenUrl CrossRef PubMed 59. ↵ Chen HT , Zhang Y , Huang J , Sawant M , Smith MD , Rajagopal N , Desai AA , Makowski E , Licari G , Xie Y , et al. Human antibody polyreactivity is governed primarily by the heavy-chain complementarity-determining regions . Cell Rep . 2024 ; 43 : 114801 . doi: 10.1016/j.celrep.2024.114801 . PMID: 39392756 . OpenUrl CrossRef PubMed 60. ↵ Anaconda Software Distribution , Anaconda Documentation : https://docs.anaconda.com/ . Anaconda Inc ; 2020 xDate [accessed 17-Dec-2025 ]. 61. Harris CR , Millman KJ , van der Walt SJ , Gommers R , Virtanen P , Cournapeau D , Wieser E , Taylor J , Berg S , Smith NJ , et al. Array programming with NumPy . Nature . 2020 ; 585 : 357 – 362 . doi: 10.1038/s41586-020-2649-2 . PMID: 32939066 . OpenUrl CrossRef PubMed 62. Virtanen P , Gommers R , Oliphant TE , Haberland M , Reddy T , Cournapeau D , Burovski E , Peterson P , Weckesser W , Bright J , et al. SciPy 1.0: fundamental algorithms for scientific computing in Python . Nat Methods . 2020 ; 17 : 261 – 272 . doi: 10.1038/s41592-019-0686-2 . PMID: 32015543 . OpenUrl CrossRef PubMed 63. Seabold SP , J. statsmodels: Econometric and statistical modeling with python . In: Proceedings of the Proceedings of the 9th Python in Science Conference Conference 2010 ; 2010 . 64. McKinney W. Data Structures for Statistical Computing in Python . In: Proceedings of the Proceedings of the 9th Python in Science Conference Conference 2010 ; 2010 . 65. Hunter JD . Matplotlib: A 2D Graphics Environment . Computing in Science & Engineering . 2007 ; 9 : 90 – 95 . OpenUrl CrossRef 66. Waskom MB , O. ; O’Kane , D. ; et al. mwaskom/seaborn: v0.8.1 : doi: 10.5281/zenodo.883859 . Zenodo ; 2017 Date [accessed 17-Dec-2025 ]. OpenUrl CrossRef 67. Pedregosa FV , G. ; Gramfort , A. ; et al. Scikit-learn: Machine Learning in Python . Journal of Machine Learning Research . 2011 ; 12 : 2825 – 2830 . OpenUrl 68. Json, The Python Standard Standard (release 3.9.2 ): https://docs.python.org/3/library/json.html#rfc-errata . Python Software Foundation ; xDate [accessed 17-Dec-2025 ]. 69. Collections, The Python Standard Reference (release 3.9.2 ): https://docs.python.org/3/library/collections.html . Python Software Foundation ; xDate [accessed 17-Dec-2025 ]. 70. Itertools, The Python Standard Reference (release 3.9.2 ): https://docs.python.org/3/library/itertools.html?highlight=itertools . Python Software Foundation ; xDate [accessed 17-Dec-2025 ]. 71. Dunbar J , Deane CM . ANARCI: antigen receptor numbering and receptor classification . Bioinformatics . 2016 ; 32 : 298 – 300 . doi: 10.1093/bioinformatics/btv552 . PMID: 26424857 . OpenUrl CrossRef PubMed View the discussion thread. Back to top Previous Next Posted February 11, 2026. Download PDF Supplementary Material Email Thank you for your interest in spreading the word about bioRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following Prediction of Antibody Non-Specificity using Protein Language Models and Biophysical Parameters Message Subject (Your Name) has forwarded a page to you from bioRxiv Message Body (Your Name) thought you would like to see this page from the bioRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share Prediction of Antibody Non-Specificity using Protein Language Models and Biophysical Parameters Laila I. Sakhnini , Ludovica Beltrame , Simone Fulle , Pietro Sormanni , Anette Henriksen , Nikolai Lorenzen , Michele Vendruscolo , Daniele Granata bioRxiv 2025.04.28.650927; doi: https://doi.org/10.1101/2025.04.28.650927 Share This Article: Copy Citation Tools Prediction of Antibody Non-Specificity using Protein Language Models and Biophysical Parameters Laila I. Sakhnini , Ludovica Beltrame , Simone Fulle , Pietro Sormanni , Anette Henriksen , Nikolai Lorenzen , Michele Vendruscolo , Daniele Granata bioRxiv 2025.04.28.650927; doi: https://doi.org/10.1101/2025.04.28.650927 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Bioinformatics Subject Areas All Articles Animal Behavior and Cognition (7635) Biochemistry (17691) Bioengineering (13892) Bioinformatics (41937) Biophysics (21452) Cancer Biology (18588) Cell Biology (25504) Clinical Trials (138) Developmental Biology (13378) Ecology (19899) Epidemiology (2067) Evolutionary Biology (24320) Genetics (15609) Genomics (22506) Immunology (17736) Microbiology (40394) Molecular Biology (17181) Neuroscience (88605) Paleontology (666) Pathology (2832) Pharmacology and Toxicology (4824) Physiology (7641) Plant Biology (15156) Scientific Communication and Education (2045) Synthetic Biology (4294) Systems Biology (9825) Zoology (2271)
Text is read by the "Ask this paper" AI Q&A widget below.
Extraction quality varies by source — PMC NXML preserves structure
cleanly, OA-HTML may include some navigation residue, and OA-PDF can
have broken hyphenation. The publisher copy
(via DOI)
is the canonical version.