In Silico Tool for Predicting and Scanning Rheumatoid Arthritis-Inducing Peptides in an Antigen

preprint OA: closed CC-BY-4.0
📄 Open PDF Full text JSON View at publisher

Abstract

Rheumatoid arthritis (RA) is an autoimmune disorder in which the immune system mounts an abnormal response to self-antigens, leading to chronic inflammation and joint damage. Therefore, identifying antigenic regions in a protein that trigger RA is crucial for developing protein-based therapeutics. In this study, we developed models for predicting RA-inducing peptides using a dataset comprising of 291 experimentally confirmed RA-inducing peptides and 165 RA non-inducing peptides. Our initial analysis revealed that certain residues, such as glycine, proline, and tyrosine, are significantly enriched in RA-inducing peptides. While alignment-based techniques like BLAST and MERCI offered high precision, they suffered from limited coverage. We developed machine/deep learning based prediction and obtained highest performance (AUC = 0.75) using XGboost on an independent dataset. We also developed prediction methods using large language models and achieved highest performance (AUC 0.72) using ProtBERT. Our ensemble model achieved highest performance (AUC = 0.80 & MCC = 0.45) on an independent dataset that combine XGBoost and MERCI-derived motifs. All models were rigorously evaluated on an independent dataset not used during training or testing of models. This study will be valuable for assessing the risk of proteins used in probiotics, genetically modified foods, and protein-based therapeutics. Our most effective approach has been implemented in RAIpred, a web server and standalone software tool for predicting and scanning RA-inducing peptides. ( https://webs.iiitd.edu.in/raghava/raipred/ ). Highlights Rheumatoid arthritis (RA), an incurable chronic joint disorder with diverse systemic complications. An attempt to identify antigenic regions in a protein which trigger this severe disease. Utilizing sequence composition based features for developing models. Implementation of ML, DL and LLM based models for prediction of RA-inducing peptides. Development of webserver, standalone, pypi and GitHub package for users.
Full text 57,015 characters · extracted from preprint-html · click to expand
In Silico Tool for Predicting and Scanning Rheumatoid Arthritis-Inducing Peptides in an Antigen | bioRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-M677548'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search New Results In Silico Tool for Predicting and Scanning Rheumatoid Arthritis-Inducing Peptides in an Antigen View ORCID Profile Ritu Tomer , View ORCID Profile Shipra Jain , View ORCID Profile Pushpendra Singh Gahlot , View ORCID Profile Nisha Bajiya , View ORCID Profile Gajendra P. S. Raghava doi: https://doi.org/10.1101/2025.03.19.644081 Ritu Tomer 1 Department of Computational Biology, Indraprastha Institute of Information Technology , Okhla Phase 3, New Delhi-110020, India Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Ritu Tomer For correspondence: ritut{at}iiitd.ac.in Shipra Jain 1 Department of Computational Biology, Indraprastha Institute of Information Technology , Okhla Phase 3, New Delhi-110020, India Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Shipra Jain For correspondence: shipra{at}iiitd.ac.in Pushpendra Singh Gahlot 1 Department of Computational Biology, Indraprastha Institute of Information Technology , Okhla Phase 3, New Delhi-110020, India Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Pushpendra Singh Gahlot For correspondence: pushpendrag{at}iiitd.ac.in Nisha Bajiya 1 Department of Computational Biology, Indraprastha Institute of Information Technology , Okhla Phase 3, New Delhi-110020, India Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Nisha Bajiya For correspondence: nishab{at}iiitd.ac.in Gajendra P. S. Raghava 1 Department of Computational Biology, Indraprastha Institute of Information Technology , Okhla Phase 3, New Delhi-110020, India Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Gajendra P. S. Raghava For correspondence: raghava{at}iiitd.ac.in Abstract Full Text Info/History Metrics Supplementary material Preview PDF Abstract Rheumatoid arthritis (RA) is an autoimmune disorder in which the immune system mounts an abnormal response to self-antigens, leading to chronic inflammation and joint damage. Therefore, identifying antigenic regions in a protein that trigger RA is crucial for developing protein-based therapeutics. In this study, we developed models for predicting RA-inducing peptides using a dataset comprising of 291 experimentally confirmed RA-inducing peptides and 165 RA non-inducing peptides. Our initial analysis revealed that certain residues, such as glycine, proline, and tyrosine, are significantly enriched in RA-inducing peptides. While alignment-based techniques like BLAST and MERCI offered high precision, they suffered from limited coverage. We developed machine/deep learning based prediction and obtained highest performance (AUC = 0.75) using XGboost on an independent dataset. We also developed prediction methods using large language models and achieved highest performance (AUC 0.72) using ProtBERT. Our ensemble model achieved highest performance (AUC = 0.80 & MCC = 0.45) on an independent dataset that combine XGBoost and MERCI-derived motifs. All models were rigorously evaluated on an independent dataset not used during training or testing of models. This study will be valuable for assessing the risk of proteins used in probiotics, genetically modified foods, and protein-based therapeutics. Our most effective approach has been implemented in RAIpred, a web server and standalone software tool for predicting and scanning RA-inducing peptides. ( https://webs.iiitd.edu.in/raghava/raipred/ ). Highlights Rheumatoid arthritis (RA), an incurable chronic joint disorder with diverse systemic complications. An attempt to identify antigenic regions in a protein which trigger this severe disease. Utilizing sequence composition based features for developing models. Implementation of ML, DL and LLM based models for prediction of RA-inducing peptides. Development of webserver, standalone, pypi and GitHub package for users. Introduction Rheumatoid arthritis (RA) is an incurable chronic autoimmune joint disorder which exhibits significant clinical heterogeneity ( Ding et al., 2023 ; Wasserman, 2011 ; J. Zhao et al., 2021 ). RA is characterized by abnormal inflammation within the joint’s synovial tissue, which progressively damages both cartilage and bone ( Gibofsky, 2014 ; Smolen et al., 2016 ). The global prevalence of RA, affects approximately 1% of the world population, translating to millions of individuals ( Huang et al., 2021 ; Ngian, 2010 ). In literature several studies have reported, that RA contributes to a diverse range of systemic complications like, cardiovascular disease ( Gibofsky, 2014 ; D. Wu et al., 2022 ). The pathogenesis of RA is still unknown, but as reported it may arise from interactions among genetic predisposition and environmental factors. Several immune cells secrete immune regulatory molecules, which causes inflammation and joint damage, characteristic of autoimmune diseases ( More et al., 2024 ). Previously, in silico methods has been developed to predict binders of HLA-DRB1*04:01 as it plays critical role in rheumatoid arthritis ( Bhasin & Raghava, 2004 ; Patiyal et al., 2024 ). In response to genetic and environmental triggers, autoreactive CD4+ T cells become activated, which presents antigens to B cells, producing autoantibodies such as rheumatoid factor and anti-citrullinated protein antibodies (C.-Y. Wu et al., 2021 ). This autoimmune response is further fuelled by pro-inflammatory cytokines, like tumor necrosis factor-alpha (TNF-α) and interleukin-6 (IL-6), which promote elevated immune cell response and inflammation in the synovium ( McInnes & Schett, 2017 ). Additionally, macrophages and synovial fibroblasts releases the chemokines that perpetuates the inflammatory response ( Hammaker et al., 2019 ). The dysregulation of the Janus kinase (JAK)/signal transducer and activator of transcription (STAT) signalling pathway is crucial, as it mediates the signalling of several key cytokine receptors involved in RA ( Ding et al., 2023 ; Simon et al., 2021 ). Together, these pathways add in a persistent autoimmune response, leading to chronic joint inflammation and subsequent tissue destruction ( Radu & Bungau, 2021 ). In literature, traditional therapy for rheumatoid arthritis (RA) primarily focuses on disease-modifying anti-rheumatic drugs (DMARDs) ( Radu & Bungau, 2021 ). These drugs have been reported to reduce the pro-inflammatory cytokine production, thereby decreasing the underlying inflammation process in synovium, which slows down the disease progression ( Demoruelle & Deane, 2012 ). Notable progress has been observed in the DMARDs that target inflammation to prevent joint damage. Methotrexate is considered as first-line therapy response due to its efficacy and safety reported in literature ( Bedoui et al., 2019 ; Cronstein, 1997 ; Z. Zhao et al., 2022 ). In addition to this, hydroxychloroquine and sulfasalazine are also widely considered conventional DMARDs, that can be administered alone or in combination with methotrexate ( Conley et al., 2023 ; Moreira et al., 2022 ). Apart from DMARDs glucocorticoids, non-steroidal anti-inflammatory drugs (NSAIDs), and inflammatory cytokine inhibitors (ICI), are also used in the prevention of RA disease progression and management (J. Zhao et al., 2021 ). These traditionally used drugs have their own limitations that include inadequate response, intolerance, higher costs and number of side effects ( Angelini et al., 2020 ). In protein therapeutics era, one of the key challenge is identifying the antigens or antigenic regions that activate T-helper cells implicated in rheumatoid arthritis ( Ansari et al., 2010 ). In the past, numerous computational methods have been developed to predict T-helper epitopes responsible for inducing cytokines such as interferon-gamma, TNF-alpha, IL-4, and IL-5 ( Dhall et al., 2023 , 2024 ; Dhanda et al., 2013 ; Naorem et al., 2023 ). These cytokines may play a crucial role in the development of autoimmune diseases like rheumatoid arthritis. Till date, there is no in-silico computational tool available in market, which aids in predicting T-helper cell-inducing peptides or epitopes that trigger rheumatoid arthritis. In this study, we extracted only the peptides that activate T-cells responsible for inducing rheumatoid arthritis. We extracted experimentally validated 291 RA associated and 165 non-associated peptides for IEDB. In order to create a robust model, we have implemented both alignment based like BLAST, MOTIF and alignment free approaches such as machine learning algorithms, deep learning, large language models. In addition, we developed an ensemble model that combines our best machine learning models with a motif-based approach to achieve higher performance. Finally, we developed a web server and standalone software RAIpred for predicting, designing and scanning RA inducing peptides. Materials & Methods 1. Dataset Preparation & Preprocessing We gathered the experimentally validated data from the immune epitope database (IEDB) for our study and perform few preprocessing steps to further improve the quality of data used in the given study ( Vita et al., 2019 ). Firstly, we extracted a total of 344 unique Rheumatoid arthritis (RA) inducing peptides as a positive dataset and 176 unique Rheumatoid arthritis (RA) non-inducing peptides (which are not available in positive) as negative dataset from IEDB. Here, we have observed that RA-associated peptides are HLA-class I and HLA-class II binders (please refer figure 2 ). The HLA-class-I contains only 46 peptides while HLA-class II have 298 peptides. Due to less number of peptides in HLA-class I, we have selected HLA-class II peptides only. Download figure Open in new tab Figure 1: The figure depicts the etiology of rheumatoid arthritis. Download figure Open in new tab Figure 2: The figure shows the percentage of peptides binds to specific class/allele of MHC given in RA-inducers and RA non-inducers dataset. Download figure Open in new tab Figure 3: The figure depicts the complete operational flow of RAIpred. Secondly, we perform few preprocessing steps such as removal of duplicate sequences from negative dataset which are common in both (Positive as well as negative) datasets. Then we filtered out sequences whose length occurrences are only 1 or few, we kept sequences more than 9 or less than 20 length of amino acid. After pre-processing of our data, we left with 291 sequences in positive dataset and 165 sequences in negative dataset. 2. Feature Generation In present study, we have generated relevant features for both RA inducing as well as non-inducing peptides. We have calculated the different composition based features using Pfeature software ( Pande et al., 2023 ). The Pfeature software calculates around 9189 features using amino-acid sequences. We have extracted features such as AAC, DPC, DDR, BTC, PCP, ATC, RRI, CeTD, PAAC, APAAC and many more. In addition to composition based features, we extend the features extraction to binary profiles as well. With the intent of capturing maximum information for the amino acid peptide sequences, we deployed the Amino Acid Binary Profile (AABP) module of Pfeature software ( Pande et al., 2023 ). The generated features aids in implementing ML-based prediction algorithms. We have also included Embeddings of Large Language Models trained on ProtBERT developed by Rostlab (Ahmed Elnaggar, Michael Heinzinger, Christian Dallago, Ghalia Rihawi, Yu Wang, Llion Jones, Tom Gibbs, Tamas Feher, Christoph Angerer, Martin Steinegger, Debsindhu Bhowmik, 2020). Each and every feature have their own importance. 3. Preliminary Analysis 3.1. Positional Analysis We created a two sample logo implementing the Two Sample Logo software to comprehend the preference of amino acid residues at a certain location ( Vacic et al., 2006 ). A fixed length input sequence vector is required for this step. We identified 9-mers from the N-terminal and 9-mers from the C-terminal of each peptide as the minimum length of peptides in our datasets is nine residues. Then, to produce a fixed length vector of 18 amino acids, we re-join both areas. For the purpose of creating one sample logo plot, 18-residue sequences were provided for each peptide in the positive and negative datasets. 3.2. Compositional Analysis To understand the deeper insights of amino acid composition difference between RA-inducing and non-inducing peptides, we perform compositional analysis on both positive and negative datasets. Here, we have utilized Pfeature software to compute amino acid compositions of both datasets ( Pande et al., 2023 ). The Pfeature software calculates amino acid composition using formula; Where, AAC i is amino-acid composition of residue type i, Ri is the number of residues in i, and L is the length of peptide sequence. 3.3. Mean-based Univariate analysis In mean based analysis, we have calculated the absolute mean difference of every feature in both positive and negative class after normalizing the data. In order to understand the relevance of feature generated, we computed the mean difference of each feature among both groups. Then we have applied the applied independent t-test and identified top 5 features with significant p-value among both groups. 3.4. Logistic regression based analysis We have also employed Logistic regression (LR) based single feature analysis. In this case, we have used LR as a statistical model to evaluate the relationship between each feature and the target label. We have computed AUC values for each feature to understand the relevance of each feature. 4. Alignment-based approach 4.1. BLAST Search We have utilised a similarity-search based well known method called BLAST for the annotation of peptide sequences ( Altschul et al., 1990 ). In order to predict RA-inducing and RA non-inducing peptides, we used the peptide-peptide BLAST-based algorithm blastp-short (BLAST+ 2.2.28) in our study. 4.2. Motif Search Motifs are generally short amino acid patterns which might be responsible for a similar function. This analysis helps us to map signature patterns in inducing and non-inducing peptides. Motifs can act as targets to develop new drugs and therapies against RA. We deployed, Motif-EmeRging and with Classes-Identification (MERCI) tool ( Vens et al., 2011 ) developed over Perl script to search for exclusive motifs in positive and negative sequences using default and user-defined parameters. 5. Alignment free approach 5.1. Machine Learning Models The Python software “scikit-learn” was used for classification. The classification algorithm consists of decision trees (DT), random forest (RF), multi-layer perceptron (MLP), eXtreme gradient boosting (XGBoost), support vector with the kernel as a radial basis (SVR), ExtraTreesClassifier (ET), Logistic Regression (LR), K-nearest Neighbor (KNN), and Guassian Naïve Baise (GNB). The grid search CV approach was used to tune hyperparameters, with “ROC” serving as the optimization metric. 5.2. Deep Learning Models In order to handle sequential data and extract hierarchical features along the sequence, we have used the 1D CNN model for the DL technique. The 1D CNN Model recognizes patterns and dependencies throughout the peptide sequences, making it especially useful for sequential data processing. By modifying hyperparameters, the model was individually adjusted on the datasets to maximize its performance for the classification task. 6. Feature Selection As the literature suggests that not all descriptors are effective for developing machine learning models, two techniques were employed to select the best features: minimum Redundancy - Maximum Relevance (mRMR) and SVC-L1. Feature selection was performed on all combined composition based features. SVC-L1 based method able to select 34 features while we retrieve 50 features using mRMR based technique. In addition to this, we have selected top relevant features from mean based univariate analysis and logistic regression based analysis. After computing the mean difference among both groups of each feature, we selected the 3782 out of 9189 features. Secondly, we applied independent t-test to identify significant features from the stretch of 3782 features using p-value <= 0.05. Finally, we have obtained 305 features with maximum absolute mean difference ranges between 0.001 to 0.14 with significant p-value. Upon which, we have deployed machine learning classifiers over top 10, 20, 50, 100, 150, 200, 250 and 305 features. Similarly, we have developed machine learning models on top 10, 20 and 50 features obtained from logistics based regression analysis. 7. Reinforcement learning When it comes to text classification, LLM (large language model) excels because of their deep linguistic comprehension. LLMs can categorize data accurately even in situations where there is a lack of information because of their adaptability and remarkable capacity to extract important textual features. This study used LLMs to differentiate between RA inducing and RA non-inducing sequences. To fit the properties of the datasets, we used and fine-tuned the ProtBert model. After this fine-tuning, the model predicted whether each sequence belongs to the class of RA inducing and RA non-inducing. 8. Ensemble method To develop a more robust model which can classify RA-inducing peptides from non-inducers, we apply two approaches to create an ensemble method: 1. Combination of BLAST based approach and ML-based method 2. Combination of motif based approach and ML-based method. Firstly, BLAST based approach is used to identify disease causing peptides based on similarity hits and then machine learning approach is used for the prediction of those peptides which are not covered by the BLAST-based approach. Secondly, Motif-based approach is used to classify between disease causing peptides by identifying specific motifs and then machine learning approach is used for the prediction of those peptides which are not covered by the motif-based approach. 9. Model Evaluation In present study, we have followed best machine learning practices, in order to avoid overfitting, we have implemented five-fold cross validation technique ( Kumar et al., 2023 ; Tomer et al., 2023 ). The complete dataset was divided into 80:20 ratios, where 80% dataset used for the training and 20% used for the external validation. The model evaluation process was performed by employing standard parameters, here we estimated both threshold dependent and threshold independent parameters. In the case of threshold-dependent characteristics, we calculated sensitivity (Sens), specificity (Spec), accuracy (Acc), and Matthews correlation coefficient (MCC) using the equations (2–6). Furthermore, we evaluated the performance of models using a well-known and threshold-independent parameter Area Under the Receiver Operating Characteristic (AUROC) curve. Where, FP is false positive, FN is false negative, TP is true positive and TN is true negative. Results 1. Preliminary Analysis 1.1. Positional Analysis To determine the most significant position of an amino acid residue in a peptide, we use Two Sample Logo to perform positional analysis. It is worth mentioning that the first nine regions relate to peptide N-terminus residues, while the last nine correspond to the peptide C-terminus. We discovered that Glycine (G), glutamine (Q), and phenylalanine (F) residues are located in the N-terminus regions of positive peptides, whereas Isoleucine (I), glycine (G), tyrosine (Y), and proline (P) are found in the C-terminus region. However, threonine (T) and isoleucine (I) are found at the N-terminus, and alanine (A), leucin (L), arginine (R), and histidine (H) are identified in the C-terminus area, whereas glutamic acid (E) highly prominent at every position in negative dataset. 1.2. Compositional Analysis Here, we computed amino acid composition for RA-inducing (positive) and RA non-inducing (negative) peptides. In figure 4 , we can see that the glycine (G), proline (P) and tyrosine (Y) have the highest average composition in RA-inducing peptides with significant p-value as compared to the negative set. While the amino-acids alanine (A), aspartic acid (D), glutamic acid (E) and leucine (L) are significantly abundant in negative set. Download figure Open in new tab Figure 4: Positional analysis of RA inducing and non-inducing peptides generated using Two Sample Logo. Download figure Open in new tab Figure 5: Amino acid compositional analysis among RA inducers and RA non-inducers. 1.3. Mean-based univariate analysis We have finally selected 305 features based on their higher mean difference between positive and negative dataset with significant p-value. As depicted in Table 2 , the composition enhanced transition and distribution-based features such as CeTD_21_HB and CeTD_25_p_VW3 showed the highest mean difference of 0.084 & -0.140. Please refer Supplementary Table S1 for the complete list and their rankings. View this table: View inline View popup Download powerpoint Table 1: The table shows list of features calculated using Pfeature tool. View this table: View inline View popup Download powerpoint Table 2: The table shows the top 5 features with maximum mean differences and a significant p-value. 1.4. LR-based analysis In order to identify the best features based on their individual performance, we have also applied logistic regression classifier on 9189 features, which were calculated using Pfeature tool. Here, we have observed that top AUROC features are from the bond composition (BTC) & Composition enhanced Transition and Distribution (CeTD) based feature categories. The features named as BTC_T, BTC_S and BTC_H achieved the maximum AUROC 0.69 & features CeTD_75_p_VW3, CeTD_100_p_VW3 and CeTD_50_p_VW3 achieved the AUROC of 0.68. The top 10 features with their performance are shown in Table 3 , detailed results can be seen in Supplementary Table S2. View this table: View inline View popup Download powerpoint Table 3: The table shows the performance of top 10 features using single feature analysis. 2. Alignment based approach We used alignment-free methods (machine learning techniques) as well as alignment-based methods (Motif & BLAST), as shown in the previous sections. Every strategy has pros and cons of its own. Alignment-based techniques have a low sensitivity but a high specificity. The presence of motif or sequence similarity is essential to their performance. However, alignment-free approaches or models based on machine learning are more universal and performance is independent of similarity. We developed ensemble or hybrid approaches using BLAST & Motif to leverage the advantages of both alignment-free and alignment-based models. 2.1. BLAST In BLAST based approach Firstly, we have prepared a BLAST formatted database using training data. Then, we search query sequences (sequences in the test set) against the training database, to find hit at different e-values ranging from 1e-5 to 1e+3. The query sequence was categorized as positive if the top hit was positive and as negative if the top hit was negative. The detailed hits achieved by BLAST on validation data is shown below in Table 4 . View this table: View inline View popup Download powerpoint Table 4: The table showing the coverage of BLAST hits on validation dataset. Secondly, we have combined predicted labels developed using CeTD features with the BLAST score with the aim of enhancing the performance of our XGB models. We attained a maximum AUROC of 0.77 on the validation dataset at e-value 1.00E+01. As the e-value is too high so it could be the random chances of getting hits. The complete result table shown in Supplementary Table S3. In order to develop a more robust model, we further move-on to Motif-based approach. 2.2. Motif In Motif based approach, we have calculated K number of exclusive motifs using MERCI tool. The MERCI tool provides different parameters to generate specific motifs based on positive and negative sets. We have calculated exclusive motifs using None, KOOLMAN-ROHM and BETTS-RUSSELL classification methods and assign them +0.5 value if motif found in positive sequence and 0 if no match found. Finally, we combine predicted labels of best model with motif score. The best motifs with highest performance was obtained from BETTS-RUSSELL classification method. We have provided the detailed list of motifs with their occurrence in RA-inducing datasets in Table 5 . View this table: View inline View popup Download powerpoint Table 5: The table shows the abundance of motifs in RA-inducing and RA non-inducing datasets. 3. Alignment free Approach 3.1. Machine Learning based analysis We have applied multiple ML-based classifiers on composition based features and amino acid binary profile (AABP) based feature (Result given in Supplementary Table S4 .), Our finding demonstrated that among different composition based features, the CeTD performed exceptionally. Using CeTD based features, we have achieved maximum accuracy and AUROC of 71% & 0.75 over training dataset and maximum accuracy and AUROC of 66.30% & 0.75 over validation dataset with balanced sensitivity and specificity by applying XGB classifier. The Table 6 , below shows the performance of best model for all composition and binary profile based features over validation dataset. View this table: View inline View popup Download powerpoint Table 6: The table shows the performance of ML classifiers on validation dataset using composition and binary profile based features. 3.2. Deep Learning based analysis We have applied 1DCNN on different composition-based features as well as on amino acid binary profile-based features. Here, 1DCNN, performs well on tri-peptide composition based features by achieving AUROC of 0.69 on validation dataset. The detailed results of 1DCNN model on different features are given in Supplementary Table S5. The Table 7 of all feature performance is given below. View this table: View inline View popup Table 7: The table shows the performance of 1DCNN model over different type of features on validation data. 3.3. Feature selection techniques In order to select most relevant features, we have applied two feature selection techniques – SVC-L1 and mRMR. Using SVC-L1 technique, we were able to select 34 different composition based features. As depicted in Table 8 , we achieved maximum AUROC of 0.72 on validation dataset using support vector classifier (SVC). View this table: View inline View popup Download powerpoint Table 8: The table shows the performance of feature selected using SVC-L1 technique over validation dataset. In present study, we have also applied mRMR based feature selection technique, which were calculated 50 different composition based features and achieved a maximum AUROC of 0.73 on validation dataset using random forest (RF) classifier. The detailed results are depicted in Table 9 . View this table: View inline View popup Download powerpoint Table 9: The table shows the performance of feature selected using mRMR based technique over validation dataset. In addition to this, we have also implemented mean based univariate analysis and logistics regression based feature selection techniques. We have applied machine learning techniques on 305 features as well as on top 200, 150, 100, 50, 20, 10 set of features. Using this, we achieved highest AUC as 0.73 for top 100, top 150 and top 200 features. The results of all the selected features over validation dataset are shown below in Table 10 , for detailed results please refer Supplementary Table S6. Also, the performance of various machine learning algorithms on top 10, 20 and 50 features selected using logistics based regression analysis to discriminate RA-inducers from non RA-inducers, can be referred in Supplementary Table S7. View this table: View inline View popup Table 10: The table shows the best performing models over validation dataset on different set of features selected using Mean-based univariate analysis. 4. Reinforcement learning based analysis We have used protBERT and BioBERT, two pre-trained large language models, for this study. The number of epochs was changed to fine-tune the models for specific tasks. Among the epochs tested, epoch 10 on protBERT & 3 on BioBERT provided the best results, with a maximum AUC of 0.71 and 0.67 on finetune models. The Table 11 shows the finetune model result over validation dataset. View this table: View inline View popup Download powerpoint Table 11: The table shows the performance of finetune LLM models over validation dataset. Furthermore, since embeddings generated by the fine-tuned model can serve as valuable features, we have extracted embeddings using the above finetune model and applied various ML algorithms. The finetune model were not performed well on this data. The results for LLM models are given in Supplementary Table S8. 7. Ensemble Model An ensemble model is a combined model developed by merging two different approaches to enhance the model quality and robustness. Here, we have combined motif-based approach with best performing ML model using CeTD based features. We have observed that the BETTS-RUSSELL classification method of MERCI with fp-2, fn-0, g-0, k-20 parameters identify the best motifs which covers maximum validation data. We have combined these motif scores with best performing ML-classifier. As depicted in Table 12 , we achieved the highest AUROC of 0.80 with MCC of 0.45 on validation dataset. The exclusive motif of disease-inducing peptides would help easily identifying the specific regions in protein which leads to this disease. Finally. This ensemble model is also incorporated in the RAIpred webserver’s predict module for the providing the easy accessibility. View this table: View inline View popup Download powerpoint Table 12: The table shows the performance of hybrid model developed using exclusive positive motifs [with MERCI classification – None, KOOLMAN-ROHM and BETTS-RUSSELL]. 8. Webserver Designing We developed RAIpred, which is accessible at https://webs.iiitd.edu.in/raghava/raipred/ , to provide the scientific community a more user-friendly webserver. This platform employs our best-performing model to forecast peptides that cause RA. RAIpred provides “Prediction, Design, Protein Scan, and Motif Scan” modules. By efficiently differentiating between RA-inducing and RA-non-inducing proteins, the Prediction module enables users to submit protein sequences in FASTA format for prediction. Users can forecast which analogs will cause RA by using the Design module to produce all possible analogs of the input sequence. The Protein Scan module helps find the parts of a given protein sequence that cause RA. The “MERCI” program is used in the Motif Scan module to map the motifs in the query sequence that cause RA. The responsive framework used in the development of this platform allows it to be viewed on a variety of platforms, including computers, laptops, and smartphones. Furthermore, we have developed a standalone Python tool called “RAIpred” to help users identify areas of proteins or peptides that may cause sickness. The web server’s “download module” can be used to get this package at https://webs.iiitd.edu.in/raghava/raipred/download.html . Powered by HTML5, Java, CSS3, and PHP scripts, the webserver works with a number of devices, including desktop, tablet, mobile, and iMac. Discussion Rheumatoid arthritis is caused by gradual loss of self-tolerance in genetically vulnerable individuals due to various environmental stressors. Both the genetic as well as environment factors are significantly responsible for the onset of the disease. The ability of peptides to selectively bind arthritogenic peptide sequences for presentation to auto-reactive T lymphocytes is provided by a common epitope in the peptide binding groove area of MHC class II molecules ( Hill et al., 2003 ). These T-cells generate inflammatory responses by releasing excessive amounts of cytokines and by activating B-cells which are further responsible for the excessive production of autoantibodies and lead to destruction and bone erosion. As these arthritogenic peptides are responsible for inducing T-cell response, they can be targeted for therapeutic purposes by modifying their properties using peptide cyclization, chemical modification or other in-vitro approaches. Few machine learning-based techniques have been developed in the past to combat rheumatoid arthritis. One of these techniques, which was developed in order to forecast how well biologic drugs will work in treating patients with RA and AS (ankylosing spondylitis), with an AUC of 0.64 based on validation data. Furthermore, it demonstrates that the most significant predictors of therapy responses were patient self-reporting scales, the Bath Ankylosing Spondylitis Functional Index (BASFI) in AS patients, and the patient global assessment of disease activity (PtGA) in RA patients ( Lee et al., 2021 ). Another study uses machine learning (ML) to predict patient relapses based on blood test results and ultrasound examination data ( Matsuo et al., 2022 ). Prasad et al. developed ATRPred, an ML-based technique that uses clinical and demographic characteristics to predict how well RA patients will respond to anti-TNF treatment ( Prasad et al., 2022 ). One study uses genetic information from SNPs in non-HLA genes to predict RA ( Dudek et al., 2024 ). To the best of our knowledge, no method has yet been developed to anticipate the T-helper cell-inducing peptides or epitopes that trigger rheumatoid arthritis. In this study, we have made a systemic approach for the classification of RA-inducing peptides. We have extracted 291 RA-inducing peptides and 165 RA non-inducing peptides from IEDB. In the preceding sections, we observed few key insights from the comprehensive analysis among RA inducing and non-inducing peptides. Here, compositional analysis indicates that RA-inducing peptides have the highest average composition of glycine and proline as compared to non-inducing peptides, which might be responsible for peptide binding to MHC ( Rauscher et al., 2006 ). In addition to this, positional analysis further marks the preferences of distinct amino acid preferences for N-terminal and C-terminal regions in positive and negative peptide datasets, emphasizing the differential roles of residues such as glycine (G), threonine (T), and alanine (A). In classification modelling, both alignment-based (BLAST and Motif) and alignment-free methods were implemented. BLAST based approach demonstrated slightly increased performance at higher e-value which depicts the random chances of getting hits. While motif based approach, gives highest correct hits on validation set. In the present study, among different composition-based features, we have observed that CeTD composition-based features outperformed all. We have reported maximum accuracy and AUROC over training data as 71% & 0.75 and 66.30% & 0.75 over validation dataset with balanced sensitivity and specificity by applying XGB classifier. This result underscores that CeTD features capture physicochemical peptide properties for our dataset, which is critical for accurate predictions. Finally, we have combined motif analysis and an ML-based methodology to develop an ensemble method. On the validation dataset, an ensemble-based method gets a maximum AUROC of 0.80 and an MCC of 0.45. The integration of compositional insights and machine learning algorithms enabled the development of a robust tool ‘RAIPred’ for RA-inducing peptide prediction. In order to server the scientific community an easy and user-friendly approach, we have also developed a web server named RAIpred ( https://webs.iiitd.edu.in/raghava/raipred/ ), standalone package ( https://webs.iiitd.edu.in/ raghava/raipred/download.html) and Github ( https://github.com/raghava/raipred ) for the prediction of RA-inducing peptides. Applications of RAIpred This tool enables researchers to access RA causing risk involved in newly discovered proteins/ peptides, before deploying them in therapeutics or GMOs. This tool aids in designing therapeutic peptides by assessing their physiochemical properties as well as screening them as RA inducers or non-inducers. Using Protein scan module users can map disease causing antigenic epitopes in the query protein sequence. Researchers may produce new peptides by using the Design Module of RAIpred to execute single residue modifications and anticipate whether or not they would accelerate the pathogenesis of RA. Conclusion Nowadays, peptide-based therapeutics has become popular as they have high success rate in clinical trials due to their selectivity towards the target. Risk assessment of these proteins is essential for preventing them causing some severe side-effects or involve in disease development. There are several peptide-based drugs, approved by FDA for treatment of rheumatoid arthritis and other autoimmune disorders. This RA-inducing peptide prediction tool, with high accuracy and balanced sensitivity and specificity, making it a reliable resource for identifying RA-inducing peptides. This tool would aid in development of targeted therapeutic peptides treatment for RA. Additionally, it provides valuable insights into peptide functionality, aiding in the discovery of novel peptides with potential biological and pharmaceutical significance. Funding Source The current work has been supported by the Department of Biotechnology (DBT) grant BT/PR40158/BTIS/137/24/2021. Author Contributions RT and GPSR collected and processed the datasets. RT and GPSR implemented the algorithms and developed the prediction models. RT, SJ and GPSR analysed the results. RT and PSG created the front-end user interface and RT created the back-end of the webserver. RT, SJ and GPSR penned the manuscript. NB prepared the figures for the manuscript. RT, SJ, PSG, NB and GPSR reviewed the manuscript. GPSR conceived and coordinated the project. All authors read and approved the final manuscript. Conflict of Interest Statement The authors declare no conflicts of interest. Data Availability Statement All the datasets used in this study are available at the “RAIpred” webserver, https://webs.iiitd.edu.in/raghava/raipred/dd.html . Acknowledgments Authors are thankful to the Department of Science and Technology (DST-INSPIRE), University Grants Commission (UGC), Department of Bio-Technology (DBT), and Council of Scientific & Industrial Research (CSIR) for fellowships and financial support, and the Department of Computational Biology, IIITD New Delhi for infrastructure and facilities. We would like to acknowledge that Figures were created using BioRender.com. Abbreviations RA Rheumatoid arthritis IEDB Immune epitope database HLA Human leukocyte antigen XGBoost extreme gradient boosting classifier ProtBERT Protein BERT DMARDs Disease-modifying anti-rheumatic drugs NSAIDs Non-steroidal anti-inflammatory drugs GMO Genetically modified organisms Reference 1. Ahmed Elnaggar , Michael Heinzinger , Christian Dallago , Ghalia Rihawi , Yu Wang , Llion Jones , Tom Gibbs , Tamas Feher , Christoph Angerer , Martin Steinegger , Debsindhu Bhowmik , B. R. ( 2020 ). ProtTrans: Towards Cracking the Language of Life’s Code Through Self-Supervised Deep Learning and High Performance Computing . Arxiv , v3 . doi: 10.48550/arXiv.2007.06225 OpenUrl CrossRef 2. ↵ Altschul , S. F. , Gish , W. , Miller , W. , Myers , E. W. , & Lipman , D. J . ( 1990 ). Basic local alignment search tool . Journal of Molecular Biology , 215 ( 3 ), 403 – 410 . doi: 10.1016/S0022-2836(05)80360-2 OpenUrl CrossRef PubMed Web of Science 3. ↵ Angelini , J. , Talotta , R. , Roncato , R. , Fornasier , G. , Barbiero , G. , Dal Cin , L. , Brancati , S. , & Scaglione , F . ( 2020 ). JAK-Inhibitors for the Treatment of Rheumatoid Arthritis: A Focus on the Present and an Outlook on the Future . Biomolecules , 10 ( 7 ). doi: 10.3390/biom10071002 OpenUrl CrossRef 4. ↵ Ansari , H. R. , Flower , D. R. , & Raghava , G. P. S . ( 2010 ). AntigenDB: an immunoinformatics database of pathogen antigens . Nucleic Acids Research , 38 (Database issue), D847-53. doi: 10.1093/nar/gkp830 OpenUrl CrossRef PubMed Web of Science 5. ↵ Bedoui , Y. , Guillot , X. , Selambarom , J. , Guiraud , P. , Giry , C. , Jaffar-Bandjee , M. C. , Ralandison , S. , & Gasque , P . ( 2019 ). Methotrexate an Old Drug with New Tricks . International Journal of Molecular Sciences , 20 ( 20 ). doi: 10.3390/ijms20205023 OpenUrl CrossRef PubMed 6. ↵ Bhasin , M. , & Raghava , G. P. S . ( 2004 ). SVM based method for predicting HLA-DRB1*0401 binding peptides in an antigen sequence. Bioinformatics (Oxford , England ) , 20 ( 3 ), 421 – 423 . doi: 10.1093/bioinformatics/btg424 OpenUrl CrossRef PubMed Web of Science 7. ↵ Conley , B. , Bunzli , S. , Bullen , J. , O’Brien , P. , Persaud , J. , Gunatillake , T. , Nikpour , M. , Grainger , R. , Barnabe , C. , & Lin , I . ( 2023 ). What are the core recommendations for rheumatoid arthritis care? Systematic review of clinical practice guidelines . Clinical Rheumatology , 42 ( 9 ), 2267 – 2278 . doi: 10.1007/s10067-023-06654-0 OpenUrl CrossRef PubMed 8. ↵ Cronstein , B. N . ( 1997 ). The mechanism of action of methotrexate . Rheumatic Diseases Clinics of North America , 23 ( 4 ), 739 – 755 . doi: 10.1016/s0889-857x(05)70358-6 OpenUrl CrossRef PubMed Web of Science 9. ↵ Demoruelle , M. K. , & Deane , K. D . ( 2012 ). Treatment strategies in early rheumatoid arthritis and prevention of rheumatoid arthritis . Current Rheumatology Reports , 14 ( 5 ), 472 – 480 . doi: 10.1007/s11926-012-0275-1 OpenUrl CrossRef PubMed 10. ↵ Dhall , A. , Patiyal , S. , Choudhury , S. , Jain , S. , Narang , K. , & Raghava , G. P. S . ( 2023 ). TNFepitope: A webserver for the prediction of TNF-alpha inducing epitopes . Computers in Biology and Medicine , 160 , 106929 . doi: 10.1016/j.compbiomed.2023.106929 OpenUrl CrossRef PubMed 11. ↵ Dhall , A. , Patiyal , S. , & Raghava , G. P. S . ( 2024 ). A hybrid method for discovering interferon-gamma inducing peptides in human and mouse . Scientific Reports , 14 ( 1 ), 26859 . doi: 10.1038/s41598-024-77957-8 OpenUrl CrossRef PubMed 12. ↵ Dhanda , S. K. , Gupta , S. , Vir , P. , & Raghava , G. P. S . ( 2013 ). Prediction of IL4 inducing peptides . Clinical & Developmental Immunology , 2013 , 263952 . doi: 10.1155/2013/263952 OpenUrl CrossRef PubMed 13. ↵ Ding , Q. , Hu , W. , Wang , R. , Yang , Q. , Zhu , M. , Li , M. , Cai , J. , Rose , P. , Mao , J. , & Zhu , Y. Z . ( 2023 ). Signaling pathways in rheumatoid arthritis: implications for targeted therapy . Signal Transduction and Targeted Therapy , 8 ( 1 ), 68 . doi: 10.1038/s41392-023-01331-9 OpenUrl CrossRef PubMed 14. ↵ Dudek , G. , Sakowski , S. , Brzezinska , O. , Sarnik , J. , Budlewski , T. , Dragan , G. , Poplawska , M. , Poplawski , T. , Bijak , M. , & Makowska , J . ( 2024 ). Machine learning-based prediction of rheumatoid arthritis with development of ACPA autoantibodies in the presence of non-HLA genes polymorphisms . PloS One , 19 ( 3 ), e0300717 . doi: 10.1371/journal.pone.0300717 OpenUrl CrossRef PubMed 15. ↵ Gibofsky , A . ( 2014 ). Epidemiology, pathophysiology, and diagnosis of rheumatoid arthritis: A Synopsis . The American Journal of Managed Care , 20 ( 7 Suppl ), S128 – 35 . OpenUrl PubMed 16. ↵ Hammaker , D. , Nygaard , G. , Kuhs , A. , Ai , R. , Boyle , D. L. , Wang , W. , & Firestein , G. S . ( 2019 ). Joint Location-Specific JAK-STAT Signaling in Rheumatoid Arthritis Fibroblast-like Synoviocytes . ACR Open Rheumatology , 1 ( 10 ), 640 – 648 . doi: 10.1002/acr2.11093 OpenUrl CrossRef PubMed 17. ↵ Hill , J. A. , Wang , D. , Jevnikar , A. M. , Cairns , E. , & Bell , D. A . ( 2003 ). The relationship between predicted peptide-MHC class II affinity and T-cell activation in a HLA-DRbeta1*0401 transgenic mouse model . Arthritis Research & Therapy , 5 ( 1 ), R40 – 8 . doi: 10.1186/ar605 OpenUrl CrossRef PubMed Web of Science 18. ↵ Huang , J. , Fu , X. , Chen , X. , Li , Z. , Huang , Y. , & Liang , C . ( 2021 ). Promising Therapeutic Targets for Treatment of Rheumatoid Arthritis . Frontiers in Immunology , 12 , 686155 . doi: 10.3389/fimmu.2021.686155 OpenUrl CrossRef PubMed 19. ↵ Kumar , N. , Patiyal , S. , Choudhury , S. , Tomer , R. , Dhall , A. , & Raghava , G. P. S . ( 2023 ). DMPPred: a tool for identification of antigenic regions responsible for inducing type 1 diabetes mellitus . Briefings in Bioinformatics , 24 ( 1 ). doi: 10.1093/bib/bbac525 OpenUrl CrossRef 20. ↵ Lee , S. , Kang , S. , Eun , Y. , Won , H.-H. , Kim , H. , Lee , J. , Koh , E.-M. , & Cha , H.-S . ( 2021 ). Machine learning-based prediction model for responses of bDMARDs in patients with rheumatoid arthritis and ankylosing spondylitis . Arthritis Research & Therapy , 23 ( 1 ), 254 . doi: 10.1186/s13075-021-02635-3 OpenUrl CrossRef PubMed 21. ↵ Matsuo , H. , Kamada , M. , Imamura , A. , Shimizu , M. , Inagaki , M. , Tsuji , Y. , Hashimoto , M. , Tanaka , M. , Ito , H. , & Fujii , Y . ( 2022 ). Machine learning-based prediction of relapse in rheumatoid arthritis patients using data on ultrasound examination and blood test . Scientific Reports , 12 ( 1 ), 7224 . doi: 10.1038/s41598-022-11361-y OpenUrl CrossRef PubMed 22. ↵ McInnes , I. B. , & Schett , G . ( 2017 ). Pathogenetic insights from the treatment of rheumatoid arthritis. Lancet (London , England) , 389 ( 10086 ), 2328 – 2337 . doi: 10.1016/S0140-6736(17)31472-1 OpenUrl CrossRef PubMed 23. ↵ More , N. E. , Mandlik , R. , Zine , S. , Gawali , V. S. , & Godad , A. P . ( 2024 ). Exploring the therapeutic opportunities of potassium channels for the treatment of rheumatoid arthritis . Frontiers in Pharmacology , 15 , 1286069 . doi: 10.3389/fphar.2024.1286069 OpenUrl CrossRef PubMed 24. ↵ Moreira , P. M. , Correia , A. M. , Cerqueira , M. , & Gil , M. F . ( 2022 ). Perioperative management of disease-modifying antirheumatic drugs and other immunomodulators . ARP Rheumatology , 1 (ARP Rheumatology, n masculine3 2022), 218 – 224 . OpenUrl PubMed 25. ↵ Naorem , L. D. , Sharma , N. , & Raghava , G. P. S . ( 2023 ). A web server for predicting and scanning of IL-5 inducing peptides using alignment-free and alignment-based method . Computers in Biology and Medicine , 158 , 106864 . doi: 10.1016/j.compbiomed.2023.106864 OpenUrl CrossRef PubMed 26. ↵ Ngian , G.-S . ( 2010 ). Rheumatoid arthritis . Australian Family Physician , 39 ( 9 ), 626 – 628 . OpenUrl PubMed 27. ↵ Pande , A. , Patiyal , S. , Lathwal , A. , Arora , C. , Kaur , D. , Dhall , A. , Mishra , G. , Kaur , H. , Sharma , N. , Jain , S. , Usmani , S. S. , Agrawal , P. , Kumar , R. , Kumar , V. , & Raghava , G. P. S . ( 2023 ). Pfeature: A Tool for Computing Wide Range of Protein Features and Building Prediction Models . Journal of Computational Biology : A Journal of Computational Molecular Cell Biology , 30 ( 2 ), 204 – 222 . doi: 10.1089/cmb.2022.0241 OpenUrl CrossRef 28. ↵ Patiyal , S. , Dhall , A. , Kumar , N. , & Raghava , G. P. S . ( 2024 ). HLA-DR4Pred2: An improved method for predicting HLA-DRB1*04:01 binders. Methods (San Diego , Calif .) , 232 , 18 – 28 . doi: 10.1016/j.ymeth.2024.10.007 OpenUrl CrossRef 29. ↵ Prasad , B. , McGeough , C. , Eakin , A. , Ahmed , T. , Small , D. , Gardiner , P. , Pendleton , A. , Wright , G. , Bjourson , A. J. , Gibson , D. S. , & Shukla , P . ( 2022 ). ATRPred: A machine learning based tool for clinical decision making of anti-TNF treatment in rheumatoid arthritis patients . PLoS Computational Biology , 18 ( 7 ), e1010204 . doi: 10.1371/journal.pcbi.1010204 OpenUrl CrossRef 30. ↵ Radu , A.-F. , & Bungau , S. G . ( 2021 ). Management of Rheumatoid Arthritis: An Overview . Cells , 10 ( 11 ). doi: 10.3390/cells10112857 OpenUrl CrossRef 31. ↵ Rauscher , S. , Baud , S. , Miao , M. , Keeley , F. W. , & Pomes , R . ( 2006 ). Proline and glycine control protein self-organization into elastomeric or amyloid fibrils . Structure (London, England : 1993) , 14 ( 11 ), 1667 – 1676 . doi: 10.1016/j.str.2006.09.008 OpenUrl CrossRef PubMed 32. ↵ Simon , L. S. , Taylor , P. C. , Choy , E. H. , Sebba , A. , Quebe , A. , Knopp , K. L. , & Porreca , F . ( 2021 ). The Jak/STAT pathway: A focus on pain in rheumatoid arthritis . Seminars in Arthritis and Rheumatism , 51 ( 1 ), 278 – 284 . doi: 10.1016/j.semarthrit.2020.10.008 OpenUrl CrossRef PubMed 33. ↵ Smolen , J. S. , Aletaha , D. , & McInnes , I. B . ( 2016 ). Rheumatoid arthritis. Lancet (London , England ) , 388 ( 10055 ), 2023 – 2038 . doi: 10.1016/S0140-6736(16)30173-8 OpenUrl CrossRef PubMed 34. ↵ Tomer , R. , Patiyal , S. , Dhall , A. , & Raghava , G. P. S. ( 2023 ). Prediction of celiac disease associated epitopes and motifs in a protein . In Frontiers in Immunology (Vol. 14) . https://www.frontiersin.org/articles/10.3389/fimmu.2023.1056101 35. ↵ Vacic , V. , Iakoucheva , L. M. , & Radivojac , P . ( 2006 ). Two Sample Logo: a graphical representation of the differences between two sets of sequence alignments. Bioinformatics (Oxford , England ) , 22 ( 12 ), 1536 – 1537 . doi: 10.1093/bioinformatics/btl151 OpenUrl CrossRef PubMed Web of Science 36. ↵ Vens , C. , Rosso , M.-N. , & Danchin , E. G. J . ( 2011 ). Identifying discriminative classification-based motifs in biological sequences. Bioinformatics (Oxford , England ) , 27 ( 9 ), 1231 – 1238 . doi: 10.1093/bioinformatics/btr110 OpenUrl CrossRef PubMed 37. ↵ Vita , R. , Mahajan , S. , Overton , J. A. , Dhanda , S. K. , Martini , S. , Cantrell , J. R. , Wheeler , D. K. , Sette , A. , & Peters , B . ( 2019 ). The Immune Epitope Database (IEDB): 2018 update . Nucleic Acids Research , 47 ( D1 ), D339 – D343 . doi: 10.1093/nar/gky1006 OpenUrl CrossRef PubMed 38. ↵ Wasserman , A. M . ( 2011 ). Diagnosis and management of rheumatoid arthritis . American Family Physician , 84 ( 11 ), 1245 – 1252 . OpenUrl PubMed 39. ↵ Wu , C.-Y. , Yang , H.-Y. , Luo , S.-F. , & Lai , J.-H . ( 2021 ). From Rheumatoid Factor to Anti-Citrullinated Protein Antibodies and Anti-Carbamylated Protein Antibodies for Diagnosis and Prognosis Prediction in Patients with Rheumatoid Arthritis . International Journal of Molecular Sciences , 22 ( 2 ). doi: 10.3390/ijms22020686 OpenUrl CrossRef 40. ↵ Wu , D. , Luo , Y. , Li , T. , Zhao , X. , Lv , T. , Fang , G. , Ou , P. , Li , H. , Luo , X. , Huang , A. , & Pang , Y . ( 2022 ). Systemic complications of rheumatoid arthritis: Focus on pathogenesis and treatment . Frontiers in Immunology , 13 , 1051082 . doi: 10.3389/fimmu.2022.1051082 OpenUrl CrossRef PubMed 41. ↵ Zhao , J. , Guo , S. , Schrodi , S. J. , & He , D . ( 2021 ). Molecular and Cellular Heterogeneity in Rheumatoid Arthritis: Mechanisms and Clinical Implications . Frontiers in Immunology , 12 , 790122 . doi: 10.3389/fimmu.2021.790122 OpenUrl CrossRef PubMed 42. ↵ Zhao , Z. , Hua , Z. , Luo , X. , Li , Y. , Yu , L. , Li , M. , Lu , C. , Zhao , T. , & Liu , Y . ( 2022 ). Application and pharmacological mechanism of methotrexate in rheumatoid arthritis . Biomedicine & Pharmacotherapy = Biomedecine & Pharmacotherapie , 150 , 113074 . doi: 10.1016/j.biopha.2022.113074 OpenUrl CrossRef PubMed View the discussion thread. Back to top Previous Next Posted March 19, 2025. Download PDF Supplementary Material Email Thank you for your interest in spreading the word about bioRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following In Silico Tool for Predicting and Scanning Rheumatoid Arthritis-Inducing Peptides in an Antigen Message Subject (Your Name) has forwarded a page to you from bioRxiv Message Body (Your Name) thought you would like to see this page from the bioRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share In Silico Tool for Predicting and Scanning Rheumatoid Arthritis-Inducing Peptides in an Antigen Ritu Tomer , Shipra Jain , Pushpendra Singh Gahlot , Nisha Bajiya , Gajendra P. S. Raghava bioRxiv 2025.03.19.644081; doi: https://doi.org/10.1101/2025.03.19.644081 Share This Article: Copy Citation Tools In Silico Tool for Predicting and Scanning Rheumatoid Arthritis-Inducing Peptides in an Antigen Ritu Tomer , Shipra Jain , Pushpendra Singh Gahlot , Nisha Bajiya , Gajendra P. S. Raghava bioRxiv 2025.03.19.644081; doi: https://doi.org/10.1101/2025.03.19.644081 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Bioinformatics Subject Areas All Articles Animal Behavior and Cognition (7635) Biochemistry (17697) Bioengineering (13894) Bioinformatics (41951) Biophysics (21455) Cancer Biology (18592) Cell Biology (25507) Clinical Trials (138) Developmental Biology (13380) Ecology (19903) Epidemiology (2067) Evolutionary Biology (24321) Genetics (15610) Genomics (22509) Immunology (17737) Microbiology (40398) Molecular Biology (17182) Neuroscience (88618) Paleontology (667) Pathology (2833) Pharmacology and Toxicology (4825) Physiology (7641) Plant Biology (15158) Scientific Communication and Education (2046) Synthetic Biology (4296) Systems Biology (9825) Zoology (2271)

Text is read by the "Ask this paper" AI Q&A widget below. Extraction quality varies by source — PMC NXML preserves structure cleanly, OA-HTML may include some navigation residue, and OA-PDF can have broken hyphenation. The publisher copy (via DOI) is the canonical version.

My notes (saved in your browser only)

Ask this paper AI returns verbatim quotes from the full text · source: preprint-html

Answers must be backed by verbatim quotes from this paper's full text. Hallucinated quotes are dropped automatically; if no verbatim passage answers the question, we say so. How this works

Citation neighborhood (no data yet)

We don't have any in-corpus citations linked to this paper yet. This is a recent paper (2025) — citers typically take a year or two to land, and the OpenAlex reference graph may still be filling in.

Source provenance

europepmc
last seen: 2026-05-20T01:45:00.602351+00:00
unpaywall
last seen: 2026-05-22T02:00:06.705733+00:00
License: CC-BY-4.0