Full text
59,691 characters
· extracted from
preprint-html
· click to expand
A Simple Generative Model for the Prediction of T-Cell Receptor - Peptide Binding in T-cell Therapy for Cancer | bioRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-M677548'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search New Results A Simple Generative Model for the Prediction of T-Cell Receptor - Peptide Binding in T-cell Therapy for Cancer View ORCID Profile Athanasios Papanikolaou , View ORCID Profile Vladimir Sivtsov , View ORCID Profile Enrica Zereik , View ORCID Profile Eliana Ruggiero , View ORCID Profile Chiara Bonini , View ORCID Profile Fabio Bonsignorio doi: https://doi.org/10.1101/2025.03.18.643937 Athanasios Papanikolaou a University of Zagreb Faculty of Electrical Engineering and Computing , Zagreb, Croatia Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Athanasios Papanikolaou For correspondence: athanasios.papanikolaou{at}fer.hr Vladimir Sivtsov a University of Zagreb Faculty of Electrical Engineering and Computing , Zagreb, Croatia Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Vladimir Sivtsov Enrica Zereik b Italian National Research Council, INstitute of Marine Engineering (CNR-INM) , Genoa, Italy Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Enrica Zereik Eliana Ruggiero c IRCCS San Raffaele Scientific Institute, Experimental Hematology lab, Division of Immunology, Transplantation and Infectious Diseases , Milan, Italy d Experimental Hematology lab, Division of Immunology, Transplantation and Infectious Diseases , Milan, Italy Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Eliana Ruggiero Chiara Bonini c IRCCS San Raffaele Scientific Institute, Experimental Hematology lab, Division of Immunology, Transplantation and Infectious Diseases , Milan, Italy d Experimental Hematology lab, Division of Immunology, Transplantation and Infectious Diseases , Milan, Italy Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Chiara Bonini Fabio Bonsignorio a University of Zagreb Faculty of Electrical Engineering and Computing , Zagreb, Croatia Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Fabio Bonsignorio Abstract Full Text Info/History Metrics Preview PDF Abstract Objective To develop a deep learning model capable of predicting epitope peptides recognized by specific CDR3 (Complementarity-Determining Region 3) sequences of T-cell receptors (TCRs) in the context of Major Histocompatibility Complex (MHC) molecules, addressing the challenges of incomplete datasets and the need for novel sequence generation in adoptive T-cell therapy for cancer. Methods We implemented a sequence to sequence generative model named “GRIP” (Generative Reconstruction of antIgen Peptides) using a Long Short-Term Memory (LSTM) network with attention mechanisms. The model was trained and validated on publicly available datasets, employing data balancing, label smoothing, and dynamic learning rate scheduling to enhance performance and generalization. Accuracy was assessed at the amino acid level. Results The model achieved a training accuracy of 97% and a test accuracy of 85% for predicting epitope sequences at the amino acid level. Probabilistic sequence generation allowed GRIP to produce biologically plausible epitope sequences, even for unseen CDR3 inputs. Attention-based interpretability provided insights into the model’s focus on critical sequence elements. The model outperformed existing approaches in handling data imbalance and generalization to novel epitopes. Conclusion GRIP offers a novel solution to the TCR-epitope binding problem by generating potential epitope sequences instead of matching to known data, addressing a fundamental gap in existing models. This approach has significant implications for personalized immunotherapy, facilitating the design of targeted T-cell therapies for cancer. 1. Introduction The immune system’s ability to target and eliminate pathogens and cancer cells is a cornerstone of effective immunotherapy. Central to this process is the precise interaction between epitope peptides, CDR3 sequences of TCRs, and MHC molecules. As discussed in [ 1 ], these interactions are fundamental in determining the immune system’s recognition and response to antigens, which is crucial in activating T-cells. We will explore these terms in more detail later in the dataset subsection. However, predicting which epitope peptides can effectively bind to specific CDR3 sequences and MHC molecules remains a significant challenge. Traditional experimental methods for identifying these interactions are labourintensive, costly, and time-consuming [ 2 ]. Moreover, the vast diversity of potential epitopes and the specificity of TCR-MHC interactions further complicate this task. There is a growing need for computational models to address these challenges that can accurately and efficiently predict the interactions between epitope peptides, CDR3 sequences, and MHC molecules. Such models would significantly advance personalized immunotherapy, allowing for the design of targeted treatments tailored to individual patients’ immune profiles. This would enhance the efficacy of immunotherapy and reduce the time and cost associated with developing these treatments. In this study, we present a novel computational approach to predict epitope peptides based on given CDR3 sequences and MHC molecules using deep learning techniques. Our sequence-to-sequence model “GRIP – Generative Reconstruction of antIgen Peptides” leverages recurrent neural networks (RNNs) with LSTM layers, which are particularly effective in capturing the dependencies and patterns within sequential data, and generates “new” epitopes even with unseen CDR3 sequences. As highlighted in [ 3 ], RNNs and LSTMs are especially suited for tasks involving temporal or sequential data due to their ability to maintain memory over longer sequences, making them highly applicable for biological sequence analysis. The model also incorporates convolutional layers, which are adept at capturing local patterns within data, and attention mechanisms, which allow the model to focus on the most relevant parts of the input sequences, further enhancing its predictive capabilities. View this table: View inline View popup Download powerpoint Table 1: Summary 2. Related Work Many different approaches, mostly relying on deep learning methodologies, have been proposed in the literature. Key points, common to all of the proposed systems, are the attention to system interpretability and the problem of biased and insufficient datasets. These two can lead to actual non-explainability of the model behaviour and hide algorithmic errors [ 4 ]. Work in [ 5 ] discusses the importance of determining the main factors of TCR-epitope interactions at a molecular level, underlining how the 3D structure of the molecule could offer new insights on the binding occurency. To this aim, various methods are trying to model such 3D structure [ 6 ], as AlphaFold [ 7 ] or TCRmodel2 [ 8 ], and trying to embed it into systems predicting molecular bindings [ 9 ], even if much work is still to be done. Many approaches proposed in the literature present good performance when input data are not strictly divided, but the same performance is not observed when trying to extrapolate on unseen epitopes [ 10 ]. Authors of [ 11 ] try to embed physical and biochemical properties in their neural network model for binding prediction, arguing that amino acid physical proximity could not have the greatest importance in determining TCR-epitope specificity, in open contrast to other works [ 5 ], [ 12 ]. The problem of improving how a specific model generalizes to unseen data is addressed in [ 13 ], which proposes a method to split data in such a way that testing is only performed on sequences not used during training, and exploits a negative sample strategy. Negative data are included in [ 14 ] as well, which also discusses the employment of pretraining and concludes that, if not well-balanced, it leads to a drop in performance. On the other hand, authors of [ 15 ] and [ 16 ] concluded that self-supervised pretraining highly improved their model performance for some specific cancer types. A predictor for TCR-peptide binding only focusing on epitopes with enough known TCRs is presented in [ 17 ]; this work concludes that a good performance for generalization to unseen data can be obtained only for epitopes having a high similarity and the same MHC restriction to epitopes in the training set. Also, most models in the literature are made to pair CDR3s only with known epitope peptides. However, our knowledge of epitope peptides and our databases are incomplete, which means that a classification system that can deal only with known epitopes is fundamentally flawed from its conception. The real problem is not to just pair an unseen CDR3 with an existing epitope peptide but to find what type of sequence this peptide should have in order to pair with that specific CDR3. Besides deep learning approaches, more traditional machine learning methods have been exploited to try to solve the problem of TCR-epitope binding prediction. Physico-chemical features and negative samples are embedded, as well, in a Support Vector Machine model presented in [ 18 ], which highlights the need for considering the possible affinity among different peptides. The great variety and diversity of the cited work in the literature and the often opposite ideas and conclusions drawn in all of them underline how the binding prediction problem is still quite far from being solved. 3. Methodology 3.1. Dataset Description The dataset used in this study is pivotal for understanding the interactions between epitope peptides, CDR3 sequences and MHC molecules. This dataset is essential for immunotherapy techniques, particularly in the context of predicting which epitope peptide can bind to a specific CDR3 sequence when presented by a defined MHC molecule. By identifying the exact target of a specific TCR, such prediction enables the use of the given TCR in adoptive T-cell therapy approaches. Some of the dataset entries are reported in Figure 1 and include the following columns: Epitope peptide : This column lists the amino acid sequences that define each epitope peptide. Epitope peptides are the part of antigens that are recognized by T lymphocytes, a major arm of the immune system. As highlighted by [ 1 ], these peptides are recognized through the TCR molecule and are critical in initiating T cell-mediated immune responses, with MHC class I molecules typically binding peptides of 8-10 amino acids, and MHC class II molecules binding longer peptides of 13-18 amino acids. Understanding these sequences is crucial for designing vaccines and for cancer immunotherapy, as they can trigger an effective immune response. CDR3 beta aa : This column contains the amino acid sequences of the CDR3 region of the T-cell receptor (TCR) beta chain. The TCR is composed of an alpha and a beta chain. In each chain, the CDR3 (Complementarity-Determining Region 3) is a critical part as it directly interacts with the epitope presented by the MHC molecule. As discussed by [ 1 ], the CDR3 region’s high variability is essential for antigen specificity, as it directly contacts the central portion of the peptide within the MHC groove, allowing the immune system to recognize a vast array of antigens. MHC : This column specifies the type of MHC molecule. Major Histocompatibility Complex (MHC) molecules play a crucial role in the immune system by presenting peptide fragments (epitopes) to T cells. According to [ 1 ], class I MHC molecules present antigens derived from intracellular pathogens to CD8+ cytotoxic T cells, while class II MHC molecules present antigens from extracellular pathogens to CD4+ helper T cells. This differentiation is key to facilitating a broad range of immune responses. Download figure Open in new tab Figure 1: Initial and final part of the dataset In the context of immunotherapy, and cancer immunotherapy, understanding these interactions can help in predicting which peptides are likely to be recognized by T cells, thereby enabling the design of effective therapeutic strategies. For instance, as [ 1 ] indicates, identifying epitopes that can strongly bind to both MHC and TCR can lead to the development of personalized cancer vaccines or adoptive T-cell therapy that enhances the immune system’s ability to target and destroy cancer cells. 3.2. Data Preprocessing In this subsection, we describe the preprocessing steps taken to prepare the dataset for the model training phase. These steps ensured data quality and consistency, enhancing the model’s performance and reliability. First, the datasets containing information about epitope peptides, CDR3 sequences and MHC molecules were loaded and combined. Duplicate entries and rows with missing values were removed to maintain data integrity. Additionally, any row with special amino acids in key columns was excluded to prevent inconsistencies. Ensuring data integrity at this stage is critical, as emphasized by various studies such as [ 19 ], which link preprocessing quality directly to the reliability of machine learning outcomes. To address the imbalance in the dataset, epitopes that appeared with high frequency were downsampled to a threshold of 1% of the total entries. Conversely, rare epitopes that appeared less than 0.1% of the time were removed. This balancing act is vital to avoid model bias toward more frequent epitopes, a common issue in bioinformatics. For example, in [ 19 ], immuneML applied a similar balancing strategy when working with immune receptor repertoires to prevent the proposed models from being overly influenced by highly frequent sequences, thereby enhancing the models’ ability to generalize across a broader spectrum of data. Additionally, [ 20 ] explored balancing techniques such as 2-gram encoding, which we also tried to incorporate in some of our experiments but without significant results, and the 6-letter exchange group method to maintain a balance between computational efficiency and classification accuracy in protein sequence data. For further preparation, all text data was standardized to uppercase, and empty strings were replaced with missing values. This standardization ensured consistency across the dataset, which is particularly important when working with sequence data. While some approaches, such as the one described in [ 21 ], focus on transforming protein sequences into detailed numerical representations, we opted for a more straightforward standardization and tokenization method, given our model’s specific requirements. Next, the data was prepared for model training by separating the features (CDR3 sequences and MHC types) and the target variable (epitope peptides). The sequences were tokenized to convert them into numerical form and then padded to ensure uniform length across all data points. This method ensures that each sequence is properly formatted for input into the model. Although reducing computational load during the prediction phase is critical, as discussed in [ 22 ], our primary focus during preprocessing was to ensure data quality and consistency, rather than optimizing for computational efficiency at this stage. The CDR3 sequences and epitope peptides were tokenized using character-level tokenizers. These tokenizers were fitted on the training sequences and then used to transform the validation and testing sequences. Padding was then applied to standardize the sequence lengths, a common requirement for machine learning models dealing with sequence data. By following this approach, we ensured that the model would not encounter issues related to varying input lengths, which could complicate training. To ensure that the model could generalize to new data, the dataset was split into training, validation, and testing sets using K-Fold cross-validation. This technique, which involves creating multiple folds with distinct sets of CDR3 sequences, is a well-established method for preventing overfitting and ensuring that the model’s performance is robust. By dividing the data in this way, we could accurately assess the model’s ability to generalize to unseen data [ 22 ]. Additionally, one-hot encoding was applied to the MHC features to convert them into a numerical format suitable for model input. This encoding preserved the categorical nature of MHC types while ensuring compatibility with the machine-learning algorithms used in our model, an approach that remains a standard in the field [ 23 ], [ 19 ]. By implementing these preprocessing steps, we ensured a high-quality, balanced, and consistent dataset, ready for training a robust sequence-to-sequence model aimed at predicting epitope peptides based on given CDR3 sequences and MHC molecules. 3.3. Model Architecture The core of our model is based on a deep learning approach using recurrent neural networks (RNNs), specifically Long Short-Term Memory (LSTM) layers. LSTMs are particularly well-suited for sequence data as they can effectively capture long-range dependencies and patterns within the sequences, a capability that is crucial when dealing with the sequential nature of biological data, as highlighted in the study on pMTnet [ 24 ]. This model similarly employed LSTM networks for encoding MHC-peptide sequences, demonstrating the effectiveness of such an approach in capturing the dependencies necessary for predicting TCR-pMHC interactions. Our model architecture includes the following key components: Embedding Layer : this layer transforms the input sequences into dense vector representations, allowing the model to work with numerical data while preserving the contextual relationships between sequence elements. This approach aligns with methods discussed in [ 25 ], where embedding layers were effectively used to capture the nuances in TCR sequences and improve the model’s ability to distinguish between different antigen specificities. Convolutional Layers : two convolutional layers were added to capture local patterns and features within the sequences. These layers help in extracting relevant features before passing the data to the LSTM layers. The use of CNNs to capture local patterns in sequence data is a technique that has proven effective in other models like NetTCR-2.0 [ 26 ], where convolutional layers were pivotal in analyzing paired TCR α and β sequences, enhancing the model’s ability to predict peptide binding. Bidirectional LSTM Layers : two bidirectional LSTM layers were used to capture dependencies in both forward and backward directions. This bidirectional approach ensures that the model can learn from the entire context of the sequences, a strategy also employed in deep learning frameworks like DeepTCR [ 25 ], which utilizes such layers to analyze TCR sequence data comprehensively, improving classification performance. Attention Layer : an attention mechanism was incorporated to allow the model to focus on different parts of the input sequence when making predictions. This mechanism helps in improving the model’s ability to capture relevant features and relationships within the data. The use of attention mechanisms in models like ours is crucial, as they enable the model to weigh the importance of different sequence elements dynamically, which is particularly important in complex datasets where certain features may have more predictive power [ 3 ]. Dense Layers with Batch Normalization and Dropout : several dense layers were included to further process the data, with batch normalization to stabilize and accelerate training, and dropout layers to prevent overfitting by randomly dropping units during training. These techniques are essential for maintaining the model’s performance and have been widely validated in various deep-learning architectures, including those used in immunology research [ 3 ]. Time-Distributed Output Layer : the final layer is a time distributed layer that applies a dense layer to each time step of the input sequence. This layer produces the output predictions for each position in the sequence, ensuring that the model can make precise predictions across the entire length of the input. Label smoothing was applied to enhance the generalization of the model. This technique prevents the model from becoming too confident in its predictions, thereby improving its performance on unseen data. This method is particularly useful in scenarios where the training data is limited or noisy, as it can help the model avoid overfitting to specific patterns in the data, a challenge often encountered in biological sequence prediction tasks [ 3 ]. This architecture ensures that the model can effectively learn from the provided sequences and make accurate predictions about the epitope peptides based on given CDR3 sequences and MHC molecules. The detailed structure of the model is illustrated in Figure 2 . Download figure Open in new tab Figure 2: Model Architecture 3.4. Training Strategy The training procedure of our sequence-to-sequence model was meticulously designed to optimize its performance in predicting epitope peptides based on given CDR3 sequences and MHC molecules. We employed several strategies to ensure effective training and evaluation of the model, drawing on best practices in deep learning and computational immunology. The model was trained using a batch size of 64 and for a total of 200 epochs. These hyperparameters were carefully chosen based on preliminary experiments to balance between training time and model performance. The total number of training steps was calculated by multiplying the number of epochs by the number of batches per epoch, a standard approach that ensures comprehensive coverage of the dataset during training. To further enhance the model’s learning efficiency, we implemented the OneCycleScheduler, a dynamic learning rate adjustment technique. This scheduler gradually increases the learning rate during the initial phase of training and decreases it during the later phases, promoting efficient learning and convergence. The effectiveness of dynamic learning rate adjustments is well-documented, particularly in complex models dealing with biological sequences, such as those discussed in the TCRGP study [ 27 ], which also emphasizes the importance of adapting learning strategies to optimize model performance. The optimal learning rate identified for our model was 0.01. Several callbacks were employed to ensure robust training: ModelCheckpoint : this callback saved the best model during training based on validation loss, ensuring that we retained the most effective version of the model. The use of validation-based checkpoints is a common practice in deep learning, as it prevents overfitting and ensures that the final model is the best representation of the learning process. EarlyStopping : this callback monitored the validation loss and stopped training if no improvement was observed for 10 consecutive epochs, thereby preventing overfitting and saving computational resources. This approach is particularly valuable in scenarios with high-dimensional data, such as TCR sequence analysis, where overfitting can significantly degrade model performance, as highlighted in TCRMatch [ 28 ]. OneCycleScheduler : as mentioned, this callback adjusted the learning rate dynamically throughout the training process. This technique is increasingly recognized for its ability to accelerate convergence and improve model robustness, especially in models handling complex datasets like those involving TCR sequences [ 27 ]. The model was trained on the padded CDR3 sequences and one-hot encoded MHC features, with the epitope peptides as the target output. During training, the data was shuffled at each epoch to ensure that the model did not learn any unintended patterns from the data order, a technique that aligns with best practices in sequence-based model training to avoid introducing biases. Finally, the best model was saved during the training process and loaded at the end for evaluation on both the validation and test sets to assess its performance. This approach of saving the optimal model based on validation performance, rather than just the final model after all epochs, helps ensure that the best possible model configuration is used for predictions, as discussed in various studies [ 29 ]. 3.5. Comparative Analysis with Existing Models and Architectures As far as different models go, we can see the full picture of how our model compares to the rest of the literature by looking at Table 2 . In particular, our approach offers advantages and innovation with respect to the existing literature concerning the following aspects: Task : In the field of TCR-epitope interaction prediction, several models have been developed. However, most existing models focus primarily on binary classification tasks, such as determining whether a binding event is likely to occur between a TCR and an epitope. This approach, while valuable, provides a limited perspective, as it only answers if the binding is possible rather than predicting what the binding sequence might look like. In contrast, our model goes beyond binary classification by generating the amino acid sequence for the potential epitope itself. This distinction is critical: instead of merely predicting compatibility, our model constructs the precise sequence of amino acids that could successfully bind to a given CDR3 sequence. By focusing on this generative task, we aim to provide a more biologically informative prediction that aligns with the needs of personalized immunotherapy, where exact sequence matching is paramount for targeted treatment. Accuracy : Most existing models, such as DeepTCR, NetTCR-2.0, and TCRGP, excel in binary classification, reporting high AUC scores (e.g., 0.9+ for DeepTCR). However, these metrics reflect an easier task—determining whether binding occurs, not predicting which amino acids participate in the interaction. Our model achieves an 85% test accuracy for predicting the exact amino acids at each position, a more challenging and biologically informative task. Therefore, while their AUC scores are higher, our model addresses a more complex problem. Explainability : Our model also differentiates itself through attention-based interpretability, allowing researchers to understand which sequence elements drive predictions. Models like TCRGP and NetTCR-2.0, while accurate, lack such transparency. This makes our model more suitable for clinical applications, where understanding the reasons behind predictions is essential. Data Handling and Generalization : Another key advantage of our approach is how we handle data imbalance through down-sampling frequent epitopes, ensuring that our model generalizes well to unseen data. Models like TCRMatch rely on curated datasets and perform poorly when confronted with novel epitopes, whereas our model is built to generalize beyond the training data, which is crucial for real-world immunotherapy. Training Efficiency : We also prioritize efficiency. Our use of the OneCycleScheduler allows for faster convergence, making our model more practical for large-scale applications compared to models like TCRGP, which rely on computationally expensive methods like Gaussian Process optimization. View this table: View inline View popup Download powerpoint Table 2: Model Comparison Our architectural choices are driven by the distinct demands of the generative task at hand. We chose an LSTM network over alternatives like Transformers, Convolutional Neural Networks (CNNs), or even standard RNNs and Gated Recurrent Units (GRUs) because of the LSTM’s unique suitability for handling sequential data with long-range dependencies—essential for accurately modelling the complex interactions in TCR-epitope binding. Our model is based on a novel generative and probabilistic output approach. Unlike previous models that usually are limited to classifying between known epitopes, our model produces each amino acid in the predicted epitope sequence through a probabilistic process informed by learned patterns. At each position in the sequence, the model calculates the probability of each possible amino acid and then selects based on these probabilities. This means that each predicted epitope sequence is not a fixed, deterministic match to the training data but a new, context-driven sequence tailored to the specific CDR3 input. This generation aspect empowers our model to dynamically create sequences that are both plausible and mostly accurate, even for completely unseen CDR3 inputs. Moreover, the attention mechanism enhances this generative process by focusing on the most relevant sequence elements, and by refining the probability distributions for each position in the generated sequence. This combination of attention-driven focus and probabilistic generation allows the model to produce contextually accurate and adaptable outputs, distinguishing it as a tool for applications where precision and flexibility are essential, such as in personalized cancer immunotherapy. While Transformers are adept at modelling global relationships through self-attention, they can be computationally intensive, especially for long biological sequences like those in TCR-epitope interactions. CNNs, though efficient for identifying local patterns, cannot capture long-term dependencies. Additionally, standard RNNs and GRUs can struggle with vanishing gradient issues, making LSTMs an optimal choice for sustained sequential memory. In summary, our LSTM-based model, coupled with attention mechanisms and a probabilistic generative approach, combines the benefits of sequential memory, interpretability, and dynamic sequence generation. This architecture allows us not only to predict binding but also to generate biologically meaningful sequences, positioning our model as an advanced tool for predictive and generative tasks in immunotherapy and elsewhere. 4. Results 4.1. Model Performance and Metrics The performance of our sequence-to-sequence model was evaluated based on its ability to predict epitope peptides given specific CDR3 sequences and MHC molecules. The evaluation metrics include loss and accuracy on the train, validation and test datasets. After training, the best model achieved the following results: Train Loss : 0.5796 Train Accuracy : 97.24% Validation Loss : 0.9583 Validation Accuracy : 85.83% Test Loss : 0.9768 Test Accuracy : 85.16% These metrics indicate that the model performs well in generalizing to unseen data, making it a robust tool for predicting epitope peptides. To visualize the training process, we plotted the accuracy and loss over epochs for both the training and validation sets which helped in understanding its performance and convergence behaviour during training. More specifically, even though the validation accuracy could not reach the training accuracy, we still saw a remarkable performance for both curves and a very low loss. To provide a more comprehensive evaluation of the model’s performance, we analyzed additional metrics beyond the basic accuracy and loss. These metrics offer a deeper understanding of the model’s strengths and areas for future improvement. 4.1.1. Amino acid Level Accuracy per Position We calculated the accuracy for each character (amino acid) at every position across the entire testing dataset. This detailed analysis helps in understanding the model’s performance for each specific amino acid at each position within the epitope peptides. The overall character-level accuracy, combining all positions, was found to be 65%. For each position (rows), we created a detailed table, as shown in Figure 3 , showing the average accuracy of every amino acid (columns) in that specific position, as well as the overall accuracy of all amino acids in that position of the peptide, see the last column (Average Accuracy). This table provides insights about which amino acids the model predicts more accurately at specific positions. Understanding these accuracies can be particularly valuable for researchers and practitioners who need precise predictions for each position in the epitope peptides. For instance, if certain positions in the peptide sequence are critical for binding to CDR3 sequences or MHC molecules, knowing the model’s accuracy at these positions can guide further research and applications. Download figure Open in new tab Figure 3: Accuracy of each amino acid in each position 4.1.2. Incorrect Predictions by Position In addition to character-level accuracy, we analyzed the incorrect predictions made by the model for each amino acid at every position. This analysis reveals which amino acids are often mispredicted and what they are typically mistaken for. This is clearly shown in Figure 4 , which represents just one of all possible positions, specifically the first one. The rows represent the prediction that we make, which in this case is incorrect, and the columns are the true value, the one we should have predicted, which means that if we take A for example, in position 1 it is predicted correctly 53.67% on average, which means that 46.33% it isn’t. This 46.33% is now split into the rest of the amino acids and is presented in the cells of position 1, in row A, among all the columns. Specifically, the number one amino acid that we predict incorrectly as A is E with 10.31% and so on for every possibility. Download figure Open in new tab Figure 4: Incorrect Predictions This information is crucial for understanding the model’s behaviour when it makes errors. By identifying common mispredictions, researchers can gain insights into potential weaknesses in the model and address these issues in future iterations. Furthermore, this analysis helps users anticipate and interpret the model’s predictions, providing a clearer picture of the model’s reliability and potential improvements. 4.1.3. Epitope Frequency and Accuracy Apart from character-wise metrics, we thought that a more zoomed-out approach would also be insightful. As shown in Figure 5 , we present the frequency and accuracy percentages of all the amino acids in the epitopes of our testing dataset. It is immediately observed that even though some epitopes appear even 10 times less than others, their overall accuracy is not always affected by their smaller frequency. On the other hand, epitopes more often appearing can potentially be predicted with a smaller accuracy; this shows that higher frequency does not correlate to higher accuracy. Download figure Open in new tab Figure 5: Accuracy and Frequency of all Epitopes The combination of i) a non-over-fitting training model, ii) accuracy not bounded to frequency, and iii) different CDR3 between training, testing and validation datasets proves how unbiased and accurate our model is. 4.1.4. Prediction Probabilities The model’s binding predictions are based on the computed probabilities for each amino acid at every sequential position in the epitope peptides. To provide a more comprehensive view, we examined these probabilities for each epitope in the testing dataset. This allows users to see not only the top predictions but also the other potential predictions along with their probabilities. This detailed probability distribution, given as an example for the first two epitopes of the testing dataset, as seen in Figure 6 , can be especially beneficial for users with domain-specific knowledge. It shows exactly what probability the model gave to every amino acid in every position of all epitopes tested to be the correct prediction. By understanding the full range of probabilities, users can make more informed decisions based on the model’s evaluations, instead of just relying blindly upon the one with the highest number. For instance, in cases where the top prediction might not be the most biologically plausible, the second or third-highest probabilities might provide valuable alternatives. This level of detail empowers users to leverage the model’s predictions more effectively and apply their expertise to refine and interpret the results. Download figure Open in new tab Figure 6: All probabilities per amino acid per position 4.1.5. Multiple Predictions Inspired by the table of probabilities for each amino acid in each position, we created a different method of predicting them. More specifically, instead of always automatically picking the amino acid with the highest probability, we give the user the freedom to select the top N amino acids with the best probabilities. As we discussed before, the amino acid with the highest, given by our model, probability is occasionally not the correct prediction and there are instances where other amino acids, even with a lower estimated probability, are indeed the correct ones. Thus, we created a new way to predict accuracy which involves multiple amino acids. If the real amino acid, the one we are trying to predict is found in the group of these N -predicted amino acids, we count it as a correct prediction. Using this different method, we got some interesting results where the accuracy rose significantly both overall and per position. In more detail, for each amino acid added to the total predictions, the accuracy increased based on this mathematical expression: The more amino acids are included in the prediction group, the better accuracy we get in all metrics, but with a deceleration in growth. This behaviour was already predicted, but it is necessary to understand exactly when to stop including more amino acids in the prediction group. We speculate that there is a thin line between choosing a lower accuracy with fewer amino acids and higher accuracy with more amino acids, but we leave that decision to the experimentation of the user. Overall, these detailed metrics and analyses provide a comprehensive evaluation of our sequence-to-sequence model. They highlight the model’s strengths in accurately predicting epitope peptides and provide valuable insights into its behaviour and reliability for specific amino acids and positions within the peptides. These results underscore the model’s potential for advancing personalized immunotherapy by accurately identifying epitope targets based on immune receptor sequences. 5. Discussion Our research demonstrates the feasibility and effectiveness of using deep learning techniques to model complex biological interactions. The sequence-to-sequence model we developed captures the intricate dependencies between epitope peptides, CDR3 sequences, and MHC molecules. These results high-light the model’s potential as a robust tool for predicting peptide binding, thereby contributing to the advancement of targeted immunotherapy strategies. One of the key strengths of our approach is the use of generative deep learning to address the inherent complexity of TCR-epitope binding, which allows for novel epitope predictions rather than relying solely on existing databases. Our model’s use of attention mechanisms offers a significant advantage in interpretability, enabling insights into which features most influence binding predictions. Additionally, GRIP’s ability to synthesize new epitopes, even with unseen inputs, represents an important step forward in personalized immunotherapy, potentially improving patient outcomes by providing more tailored treatment options. These innovations position our model as a valuable tool for exploring immune interactions in a way that is both predictive and generative, setting it apart from conventional methods. Despite the promising results, there are some limitations and areas for improvement in our study. Accuracy Improvements: While the model achieves high accuracy, there is room for improvement, particularly in reducing the test loss. This is a common challenge in deep learning models for biological data, as seen in studies like NetTCR-2.0 [ 26 ], where fine-tuning for optimal accuracy across diverse datasets remained a significant hurdle. Data Imbalance: Although we addressed data imbalance through down-sampling and rare epitope removal, the problem persists. As noted in immuneML [ 19 ], data imbalance requires advanced solutions beyond simple downsampling. Future work could involve synthetic data generation techniques, such as SMOTE, to better balance datasets, as suggested in related studies [ 20 ]. Lack of Structural Information: The reliance on sequence data limits our model’s ability to leverage the rich structural and chemical properties of amino acids and peptides. This limitation, observed in other models like pMTnet [ 24 ], underscores the need for integrating biophysical and 3D structural data, which could enhance predictive power and biological relevance [ 3 , 23 ]. Generalization Challenges: Our model shows robustness but faces difficulties in generalizing to new data, particularly unseen TCR or MHC variants. Similar challenges are noted in TCRGP [ 27 ]. Strategies like transfer learning or biology-informed neural networks could improve adaptability. By addressing these limitations and incorporating additional layers of biological context, we aim to refine the model further, enhancing its utility for diverse immunotherapy applications. 6. Conclusion This study establishes a novel framework for computationally predicting TCR-epitope binding interactions, demonstrating the power of deep learning to solve complex biological problems. Our sequence-to-sequence generative model, GRIP, goes beyond existing classification-focused approaches by generating novel epitope sequences tailored to specific CDR3 inputs, a significant advancement in the field. The implications of this work are broad, particularly for personalized immunotherapy. GRIP’s ability to handle unseen data and generate biologically plausible epitope sequences highlights its potential for advancing adoptive T-cell therapy in cancer treatment. This capability could enable the design of more targeted and effective therapies, addressing the challenges posed by the vast diversity of potential epitopes and the limited availability of experimental data. While our results are promising, they also underscore the challenges inherent in modeling TCR-epitope interactions, such as data imbalance, reliance on sequence-based features, and generalization to unseen variants. Overcoming these challenges will require incorporating structural and chemical information, leveraging advanced data augmentation techniques, and exploring physics-informed neural networks tailored for biological systems. In summary, this work not only provides a strong foundation for computational immunology but also points toward exciting future directions. By addressing the outlined limitations and refining the model, GRIP could become an indispensable tool in personalized immunotherapy, paving the way for improved patient outcomes and novel insights into the immune system’s complex dynamics. CRediT Authorship Contribution Statement Athanasios Papanikolaou : Data curation, Investigation, Methodology, Software, Visualization, Writing – original draft, Writing – review and editing Vladimir Sivtsov : Data curation, Investigation, Methodology, Soft-ware, Visualization, Writing – original draft, Writing – review and editing Enrica Zereik : Data curation, Investigation, Methodology, Software, Validation, Writing – original draft, Writing – review and editing Elliana Ruggiero : Data curation, Investigation, Methodology, Validation, Writing – original draft, Writing – review and editing Chiara Bonini : Conceptualization, Data curation, Investigation, Methodology, Supervision, Writing – original draft, Writing – review and editing Fabio Bonsignorio : Conceptualization, Data curation, Investigation, Methodology, Software, Supervision, Writing – original draft, Writing – review and editing Ethics Statement This research accurately reflects the work conducted, ensuring that all presented data are correct and that the methodologies are described in sufficient detail to enable replication by others. This manuscript is entirely original, and any use of others’ work or words has been properly cited or quoted, with necessary permissions obtained where applicable. This material has not been published, in whole or in part, elsewhere and is not currently under consideration for publication in another journal. Generative AI and AI-assisted technologies have not been used to create or modify images. All authors have been actively involved in the substantive work leading to this manuscript and accept joint and individual responsibility for its content. Declaration of Competing Interest The authors have no competing interests. FB is the CEO and Founder of Heron Robots. CB and ER are inventors of different patents on cancer immunotherapy and genetic engineering. CB has been a member of the Advisory Board and Consultant for Intellia, Novartis, GSK, Allogene, Kite/Gilead, Miltenyi, Kiadis, Evir,and Janssen and received research support from Intellia Therapeutics. Acknowledgements VS, AP and FB are funded by the European Union’s Horizon 2020 Research and Innovation Programme under the Grant Agreement No. 952275. CB is supported by AIRC (Ig 24965) and AIRC 5xMille, Rif. 22737, Italian Ministry of Research and University (PRIN 2017WC8499, 2022SLL3YZ), Italian Ministry of Health (Research project on CAR-T cells for haematological malignancies and solid tumors and RF-2019–12370243) EU (T2Evolve, Join4ATMP) RF-2021-12373598 “Uncover and overcome senescence and dysfunction of genetically engineered T lymphocytes for cancer immunotherapy”. ER is supported by 2023 DKMS John Hansen Research Grant. Footnotes Email addresses: Athanasios.Papanikolaou{at}fer.unizg.hr (Athanasios Papanikolaou), Vladimir.Sivtsov{at}fer.unizg.hr (Vladimir Sivtsov), Enrica.Zereik{at}cnr.it (Enrica Zereik), Ruggiero.Eliana{at}hsr.it (Eliana Ruggiero), Bonini.Chiara{at}hsr.it (Chiara Bonini), Fabio.Bonsignorio{at}fer.unizg.hr (Fabio Bonsignorio) References [1]. ↵ Abul K. Abbas , Andrew H. H. Lichtman , and Shiv Pillai . Basic Immunology . Elsevier , 2019 . [2]. ↵ Francesco Manfredi , Lorena Stasi , Silvia Buonanno , Francesca Marzuttini , Maddalena Noviello , Sara Mastaglio , Danilo Abbati , Alessia Potenza , Chiara Balestrieri , Beatrice Claudia Cianciotti , Elena Tassi , Sara Feola , Cristina Toffalori , Marco Punta , Zulma Magnani , Barbara Camisa , Elena Tiziano , Maria Teresa Lupo-Stanghellini , Rui Mamede Branca , Janne Lehtïo , Tiina M. Sikanen , Markus J. Haapala , Vincenzo Cerullo , Monica Casucci , Luca Vago , Fabio Ciceri , Chiara Bonini , and Eliana Ruggiero . Harnessing t cell exhaustion and trogocytosis to isolate patient-derived tumor-specific tcr . Science Advances , 9 ( 48 ): eadg8014 , 2023 . OpenUrl CrossRef PubMed [3]. ↵ Simon J.D. Prince . Understanding Deep Learning . MIT Press , 2023 . [4]. ↵ Anna Weber , Aurélien Pélissier , and María Rodríguez Martínez . T-cell receptor binding prediction: A machine learning revolution . ImmunoInformatics, page 100040 , 2024 . [5]. ↵ Ceder Dens , Wout Bittremieux , Fabio Affaticati , Kris Laukens , and Pieter Meysman . Interpretable deep learning to uncover the molecular binding patterns determining TCR–epitope interaction predictions . ImmunoInformatics , 11 : 100027 , 2023 . OpenUrl CrossRef [6]. ↵ Philip Bradley . Structure-based prediction of T cell receptor: peptide-MHC interactions . Elife , 12 : e82813 , 2023 . OpenUrl CrossRef PubMed [7]. ↵ John Jumper , Richard Evans , Alexander Pritzel , Tim Green , Michael Figurnov , Olaf Ronneberger , Kathryn Tunyasuvunakool , Russ Bates , Augustin ídek , Anna Potapenko , et al. Highly accurate protein structure prediction with AlphaFold . Nature , 596 ( 7873 ): 583 – 589 , 2021 . OpenUrl CrossRef PubMed [8]. ↵ Rui Yin , Helder V Ribeiro-Filho , Valerie Lin , Ragul Gowthaman , Melyssa Cheung , and Brian G Pierce . TCRmodel2: high-resolution modeling of T cell receptor recognition using deep learning . Nucleic Acids Research , 51 ( W1 ): W569 – W576 , 2023 . OpenUrl CrossRef PubMed [9]. ↵ Yu Zhao , Bing He , Fan Xu , Chen Li , Zhimeng Xu , Xiaona Su , Haohuai He , Yueshan Huang , Jamie Rossjohn , Jiangning Song , et al. DeepAIR: A deep learning framework for effective integration of sequence and 3D structure to enable adaptive immune receptor analysis . Science Advances , 9 ( 32 ): eabo5128 , 2023 . OpenUrl CrossRef PubMed [10]. ↵ Junwei Chen , Bowen Zhao , Shenggeng Lin , Heqi Sun , Xueying Mao , Meng Wang , Yanyi Chu , Liang Hong , Dong-Qing Wei , Min Li , et al. TEPCAM: Prediction of T-cell receptor–epitope binding specificity via interpretable deep learning . Protein Science , 33 ( 1 ): e4841 , 2024 . OpenUrl CrossRef PubMed [11]. ↵ Alan M Luu , Jacob R Leistico , Tim Miller , Somang Kim , and Jun S Song . Predicting TCR-epitope binding specificity using deep metric learning and multimodal learning . Genes , 12 ( 4 ): 572 , 2021 . OpenUrl CrossRef [12]. ↵ Martina Milighetti , Yuta Nagano , James Henderson , Uri Hershberg , Andreas Tiffeau-Mayer , Anne-Florence Bitbol , and Benny Chain . Intra- and inter-chain contacts determine TCR specificity: applying protein co-evolution methods to TCR αβ pairing . bioRxiv , pages 2024 – 05 , 2024 . [13]. ↵ Filippo Grazioli , Anja Mösch , Pierre Machart , Kai Li , Israa Alqassem , Timothy J O’Donnell , and Martin Renqiang Min . On TCR binding predictors failing to generalize to unseen peptides . Frontiers in immunology , 13 : 1014256 , 2022 . OpenUrl CrossRef PubMed [14]. ↵ Yuepeng Jiang , Miaozhe Huo , and Shuai Cheng Li . TEINet: a deep learning framework for prediction of TCR–epitope binding specificity . Briefings in Bioinformatics , 24 ( 2 ): bbad086 , 2023 . OpenUrl CrossRef PubMed [15]. ↵ Min Zhang , Qi Cheng , Zhenyu Wei , Jiayu Xu , Shiwei Wu , Nan Xu , Chengkui Zhao , Lei Yu , and Weixing Feng . BertTCR: a bert-based deep learning framework for predicting cancer-related immune status based on T cell receptor repertoire . Briefings in Bioinformatics , 25 ( 5 ): bbae420 , 2024 . OpenUrl CrossRef PubMed [16]. ↵ Kevin E Wu , Kathryn Yost , Bence Daniel , Julia Belk , Yu Xia , Takeshi Egawa , Ansuman Satpathy , Howard Chang , and James Zou . TCR-BERT: learning the grammar of T-cell receptors for flexible antigen-binding analyses . In Machine Learning in Computational Biology , pages 194 – 229 . PMLR, 2024 . [17]. ↵ Giancarlo Croce , Sara Bobisse , Dana Ĺea Moreno , Julien Schmidt , Philippe Guillame , Alexandre Harari , and David Gfeller . Deep learning predictions of TCR-epitope interactions reveal epitope-specific chains in dual alpha T cells . Nature Communications , 15 ( 1 ): 3211 , 2024 . OpenUrl [18]. ↵ Omar NA Demerdash and Jeremy C Smith . TCR-H: explainable machine learning prediction of T-cell receptor epitope binding on unseen datasets . Frontiers in Immunology , 15 : 1426173 , 2024 . [19]. ↵ Milena Pavlovíc , Lonneke Scheffer , Keshav Motwani , Chakravarthi Kanduri , Radmila Kompova , et al. The immuneML ecosystem for machine learning analysis of adaptive immune receptor repertoires . Nature Machine Intelligence , 3 ( 11 ): 936 – 944 , nov 2021 . OpenUrl CrossRef PubMed [20]. ↵ Suprativ Saha . Application of data mining in protein sequence classification . International Journal of Database Management Systems , 4 ( 5 ): 103 – 118 , oct 2012 . OpenUrl CrossRef [21]. ↵ William R. Atchley , Jieping Zhao , Andrew D. Fernandes , and Tanja Drüke . Solving the protein sequence metric problem . Proceedings of the National Academy of Sciences , 102 ( 18 ): 6395 – 6400 , apr 2005 . OpenUrl Abstract / FREE Full Text [22]. ↵ Jayakishan K. Meher , Gananath N. Dash , Pramod Kumar Meher , and Mukesh Kumar Raval . A reduced computational load protein coding predictor using equivalent amino acid sequence of DNA string with period-3 based time and frequency domain analysis . American Journal of Molecular Biology , 01 ( 02 ): 79 – 86 , 2011 . OpenUrl CrossRef [23]. ↵ David Harding-Larsen , Jonathan Funk , Madsen , Carlos G. Acevedo-Rocha , and Mazurenko . Protein representations: Encoding biological information for machine learning in biocatalysis . apr 2024 . [24]. ↵ Tianshi Lu , Ze Zhang , James Zhu , Yunguan Wang , Peixin Jiang , Xue Xiao , Chantale Bernatchez , John V. Heymach , Don L. Gibbons , Jun Wang , Lin Xu , Alexandre Reuben , and Tao Wang . Deep learning-based prediction of the T cell receptor–antigen binding specificity . Nature Machine Intelligence , 3 ( 10 ): 864 – 875 , sep 2021 . OpenUrl CrossRef PubMed [25]. ↵ John-William Sidhom , H. Benjamin Larman , Drew M. Pardoll , and Alexander S. Baras . DeepTCR is a deep learning framework for revealing sequence concepts within t-cell repertoires . Nature Communications , 12 ( 1 ), mar 2021 . [26]. ↵ Alessandro Montemurro , Viktoria Schuster , Povlsen , Vanessa Jurtz , Chronister , Sine R. Hadrup , Ole Winther , Peters , and Morten Nielsen . NetTCR-2.0 enables accurate prediction of TCR-peptide binding by using paired TCR α and β sequence data . Communications Biology , 4 ( 1 ), sep 2021 . [27]. ↵ Emmi Jokinen , Jani Huuhtanen , Satu Mustjoki , and Heinonen. Predicting recognition between T cell receptors and epitopes with TCRGP . PLOS Computational Biology , 17 ( 3 ): e1008814 , mar 2021 . OpenUrl CrossRef [28]. ↵ William D. Chronister , Austin Crinklaw , Swapnil Mahajan , Randi Vita , and Koşal . TCRMatch: Predicting T-cell receptor specificity based on sequence similarity to previously characterized receptors . Frontiers in Immunology , 12 . [29]. ↵ Anton S. Smirnov , Anastasia V. Rudik , Dmitry A. Filimonov , and Alexey A. Lagunin . TCR-Pred: A new web-application for prediction of epitope and MHC specificity for CDR3 TCR sequences using molecular fragment descriptors . Immunology , 169 ( 4 ): 447 – 453 , mar 2023 . OpenUrl CrossRef View the discussion thread. Back to top Previous Next Posted March 19, 2025. Download PDF Email Thank you for your interest in spreading the word about bioRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following A Simple Generative Model for the Prediction of T-Cell Receptor - Peptide Binding in T-cell Therapy for Cancer Message Subject (Your Name) has forwarded a page to you from bioRxiv Message Body (Your Name) thought you would like to see this page from the bioRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share A Simple Generative Model for the Prediction of T-Cell Receptor - Peptide Binding in T-cell Therapy for Cancer Athanasios Papanikolaou , Vladimir Sivtsov , Enrica Zereik , Eliana Ruggiero , Chiara Bonini , Fabio Bonsignorio bioRxiv 2025.03.18.643937; doi: https://doi.org/10.1101/2025.03.18.643937 Share This Article: Copy Citation Tools A Simple Generative Model for the Prediction of T-Cell Receptor - Peptide Binding in T-cell Therapy for Cancer Athanasios Papanikolaou , Vladimir Sivtsov , Enrica Zereik , Eliana Ruggiero , Chiara Bonini , Fabio Bonsignorio bioRxiv 2025.03.18.643937; doi: https://doi.org/10.1101/2025.03.18.643937 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Bioinformatics Subject Areas All Articles Animal Behavior and Cognition (7636) Biochemistry (17704) Bioengineering (13898) Bioinformatics (41967) Biophysics (21460) Cancer Biology (18599) Cell Biology (25525) Clinical Trials (138) Developmental Biology (13384) Ecology (19909) Epidemiology (2067) Evolutionary Biology (24326) Genetics (15613) Genomics (22512) Immunology (17740) Microbiology (40423) Molecular Biology (17191) Neuroscience (88645) Paleontology (667) Pathology (2835) Pharmacology and Toxicology (4825) Physiology (7646) Plant Biology (15158) Scientific Communication and Education (2046) Synthetic Biology (4302) Systems Biology (9825) Zoology (2271)
Text is read by the "Ask this paper" AI Q&A widget below.
Extraction quality varies by source — PMC NXML preserves structure
cleanly, OA-HTML may include some navigation residue, and OA-PDF can
have broken hyphenation. The publisher copy
(via DOI)
is the canonical version.