Hybrid Transformer and Neural Network Configuration for Protein Classification Using Amino Acids

preprint OA: closed
📄 Open PDF Full text JSON View at publisher

Abstract

This study introduces a hybrid machine learning model for classifying proteins, developed to address the complexities of protein sequence and structural analysis. Utilizing an architecture that combines a lightweight transformer with a concurrent neural network, the hybrid model leverages both sequential and intrinsic physical properties of proteins. Trained on a comprehensive dataset from the Research Collaboratory for Structural Bioinformatics Protein Data Bank, the model demonstrates a classification accuracy of 95%, outperforming existing methods by at least 15%. The high accuracy achieved demonstrates the potential of this approach to innovate protein classification, facilitating advancements in drug discovery and the development of personalized medicine. By enabling precise protein function prediction, the hybrid model allows for specialized strategies in therapeutic targeting and the exploration of protein dynamics in biological systems. Future work will focus on enhancing the model’s generalizability across diverse datasets and exploring the integration of more machine learning techniques to refine predictive capabilities further. The implications of this research offer potential breakthroughs in biomedical research and the broader field of protein engineering.
Full text 34,293 characters · extracted from preprint-html · click to expand
Hybrid Transformer and Neural Network Configuration for Protein Classification Using Amino Acids | bioRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-M677548'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search New Results Hybrid Transformer and Neural Network Configuration for Protein Classification Using Amino Acids View ORCID Profile Nathan Labiosa , Aryan Kohli , View ORCID Profile Christian Chung , View ORCID Profile Christopher Korban doi: https://doi.org/10.1101/2024.10.31.621421 Nathan Labiosa 1 University of Wisconsin-Madison 3 Revilico Inc Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Nathan Labiosa For correspondence: nlabiosa{at}wisc.edu Aryan Kohli 2 University of California-Los Angeles 3 Revilico Inc Find this author on Google Scholar Find this author on PubMed Search for this author on this site Christian Chung 3 Revilico Inc Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Christian Chung Christopher Korban 2 University of California-Los Angeles 3 Revilico Inc Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Christopher Korban Abstract Full Text Info/History Metrics Preview PDF Abstract This study introduces a hybrid machine learning model for classifying proteins, developed to address the complexities of protein sequence and structural analysis. Utilizing an architecture that combines a lightweight transformer with a concurrent neural network, the hybrid model leverages both sequential and intrinsic physical properties of proteins. Trained on a comprehensive dataset from the Research Collaboratory for Structural Bioinformatics Protein Data Bank, the model demonstrates a classification accuracy of 95%, outperforming existing methods by at least 15%. The high accuracy achieved demonstrates the potential of this approach to innovate protein classification, facilitating advancements in drug discovery and the development of personalized medicine. By enabling precise protein function prediction, the hybrid model allows for specialized strategies in therapeutic targeting and the exploration of protein dynamics in biological systems. Future work will focus on enhancing the model’s generalizability across diverse datasets and exploring the integration of more machine learning techniques to refine predictive capabilities further. The implications of this research offer potential breakthroughs in biomedical research and the broader field of protein engineering. I. I ntroduction Proteins perform many roles in biological systems and their amino acid sequences dictate their structural conformations and function. For instance, proteins can function as enzymes, provide structural support to cells, facilitate material transport, serve as therapeutic agents like monoclonal antibodies, participate in DNA replication and transcription, and act as precursors for other protein formations. Further exploring categorization, proteins can be classified into several types, including signaling proteins, immune system proteins, and transcription proteins, etc. Additionally, enzymes are categorized into six principal groups: oxidoreductases, transferases, hydrolases, lyases, isomerases, and ligases. This paper introduces a novel computational approach to classify a vast array of protein types, leveraging only the primary structure of the protein and ten intrinsic properties. Traditional protein classification relies on a group of sophisticated laboratory techniques. Gel electrophoresis, for instance, separates peptides by molecular weight and uses staining methods such as Coomassie blue and silver stain for visualization. Complementing this, structural assessments frequently apply X-ray crystallography and electron microscopy to examine protein formations. Another technique, Static Light Scattering, determines molecular weight, solubility, and aggregation propensity through light scattering metrics. Further, methods like mass spectrometry and isoelectric help to enhance the accuracy of classifying distinct protein types. Despite their precision, these methods are often time-consuming and costly [ 1 ]. With the roadblock of extensive wet-lab data, there arises a substantial opportunity for efficiently applying machine learning to enhance protein classification. Particularly, if machine learning models can swiftly classify proteins, it could potentially speed up evaluating protein function up to a hundred times faster than traditional techniques [ 2 ]. The significance of rapid protein classification extends to several areas. Primarily, it drives the assessment of protein functions, which is crucial in the discovery and development of biologics. Efficient protein classification not only facilitates the examination of potential therapeutic impacts but also allows for the study of functional changes resulting from amino acid mutations. In the fields of proteomics and drug repurposing, accurate protein identification and expression level analysis are key steps towards discovering new therapeutic targets [ 3 ]. This combined approach holds promise for expediting the identification and utilization of novel protein targets in drug development research. Despite the deep knowledge base in the protein field, the demand for continued research in proteomics and biologics, particularly for personalized medicine, remains strong [ 4 ]. With the protein sector projected to grow at a compound annual growth rate of 15.53% through 2030, [ 5 ], there is a need for innovative solutions to keep pace with this expansion. In response, machine learning has emerged as a powerful tool to propel advancements. For instance, generative models like AlphaFold 2 leverage deep learning for high-accuracy structural predictions, while ProtBert addresses evolutionary tasks and sequence masking. Further, efforts to engineer new amino acid sequences use advanced techniques such as generative adversarial networks and variational autoencoders. On the discriminatory side, models like random forests and logistic regression are used to predict important physicochemical properties of proteins, including pH, water solubility, and thermal stability. Additionally, classification models are increasingly being deployed to identify subcellular localizations and facilitate protein identifications. Despite the potential of these methods, the increasing complexity and scope of protein research emphasizes a critical need for superior accuracy in protein identification tasks [ 3 ]. Recent advancements in natural language processing have catalyzed the development of new architectures, notably the transformer model. This architecture utilizes a mechanism of query, key, and value pairs, enabling the model to selectively ‘pay attention’ to different sequences within the input [ 6 ], [ 7 ]. This attention mechanism allows the model to adjust the significance it assigns to each part of the input based on its context, significantly enhancing the model’s ability to interpret complex data sequences [ 8 ], [ 9 ], [ 10 ]. The versatility of transformers has led to their widespread application across various domains, including the creation of expansive language models for text analysis and hybrid models that integrate computer vision with linguistic processing [ 11 ], [ 12 ], [ 13 ]. The attention mechanism of transformers holds immense promise for biological applications, particularly in the study of amino acids. In the context of proteomics, each amino acid sequence correlates directly to a physical molecule, with the sequence order fundamentally changing the molecule’s properties. By harnessing the attention mechanism, it becomes possible to analyze and classify amino acids based on their sequential characteristics. This capability is key for identifying and categorizing new amino acid sequences as specific protein types. Successfully classifying proteins through their amino acid sequences can revolutionize researcher’s understanding and manipulation of biological functions. It could lead to breakthroughs in designing novel proteins with desired properties, enhancing drug discovery, and tailoring therapeutic interventions more precisely to individual genetic profiles [ 14 ]. This could significantly impact the fields of personalized medicine and biotechnology, where accurate protein function prediction enables more targeted treatment strategies and the development of new biologic drugs [ 15 ]. In this paper, the following key contributions are presented: The introduction of an unique method that utilizes context-dependent machine learning tools to leverage both the sequential and physical properties of amino acids for classification. The details of the construction of a novel model architecture, designed to effectively integrate and process these dual features, resulting in a system capable of highaccuracy protein classification. II. R elated W orks The application of transformer models to protein analysis has been attempted; significant among these attempts is the development of ProtBERT [ 16 ], [ 17 ]. ProtBERT was trained on over 106 million protein sequences using a bidirectional encoding strategy. This training involves masking specific amino acids within a sequence and tasking the model with predicting these masked positions, thereby enabling it to learn the intricate relationships between amino acid patterns [ 17 ]. While ProtBERT demonstrates the capabilities of transformers in biological contexts, it is primarily optimized for tasks other than protein classification. Furthermore, its training, though less extensive than some of its counterparts, required four weeks of pre-training on significant computational resources, rendering it impractical for researchers with limited access to time and GPU capabilities to duplicate [ 17 ]. Another noteworthy model in the field is ProtTrans [ 18 ]. Like ProtBERT, ProtTrans was primarily developed for predicting protein sequencing and structural configurations. The ProtT5-XL variant of this model underwent training on a supercomputer, utilizing thousands of GPUs and TPUs. Stemming from its immense computational demand, ProtT5-XL’s architecture is too large for deployment on standard consumer GPUs and similar to ProtBERT, it was not tailored for direct protein classification tasks [ 19 ]. Efforts to classify proteins using other machine learning methods have been documented, particularly applied to the same dataset that was utilized in this study. Notable among these are models employing Long Short-Term Memory networks, convolutional neural networks, and Naive Bayes classifiers [ 20 ], [ 21 ], [ 22 ]. These models leverage a combination of deep learning and statistical techniques to approach protein classification. Of these, the Naive Bayes model has shown the most success, achieving approximately an 80% accuracy rate in classifying proteins into 30 distinct classes. III. M ethod The primary dataset features, aside from those derived computationally, were sourced from the Research Collaboratory for Structural Bioinformatics (RCSB) Protein Data Bank (PDB). The RCSB PDB is an open-source database in the U.S., offering structural data points for proteins, DNA, and RNA. All entries in the RCSB PDB undergo a careful review process, including consistency checks, cross-validation with other data sources, and verification with relevant literature [ 23 ]. Given the dataset volume, several preprocessing steps were necessary to prepare the data for the hybrid model. Initially, an analysis was conducted to determine an appropriate sequence length cutoff. With some sequences extending beyond 16,000 amino acids and the mean and median lengths around 350, this number was chosen as the cutoff to manage dataset variability Subsequently, the thirty most prevalent proteins were selected for inclusion in the model. This selection criterion was based on each protein having over 1,000 examples within the dataset, ensuring a significant sample size and reflecting the proteins’ regularity in biological research 3. Additionally, any duplicate sequences were excluded from the dataset. To counteract model bias, which simpler preliminary models indicated was skewed towards the five most prevalent classes, a strategy of sample over-selection was employed. The Random Over Sampler technique was utilized to balance the class distribution within the training dataset [ 24 ]. This method works by randomly duplicating examples in the underrepresented classes until all classes have a comparable number of samples. This approach not only helps in reducing the model’s bias towards more frequent classes but also enhances the generalizability of the model across less common protein types. To fully leverage the data available from protein analyses, amino acid sequences were utilized to extract additional physical and numerical features. The dataset initially included intrinsic features such as structureMolecularWeight, crystallizationTemperature (in Kelvin), Matthews coefficient (densityMatthews), percent solubility (densityPercentSol), and pH value. To augment this data, Biopython toolkit was employed, a tool for computational biology and bioinformatics. Biopython uses various biological computations, including the manipulation of protein sequences to extract biochemical properties [ 25 ]. A diagram of this extraction is shown in Figure 4 . From these sequences, critical features were extracted that are important in protein classification due to their impact on protein structure and function: Isoelectric Point: The pH at which a protein carries no net electric charge, significant for understanding protein solubility and interaction. Aromaticity: A measure of the frequency of aromatic amino acids, important for protein stability and function. Instability Index: A predictor of the stability of a protein in a test tube, with higher values indicating less stability. Helix, Turn, Sheet: Structural components of proteins, with helices and sheets forming the core structures and turns facilitating turns between them. These elements are necessary for defining the protein’s 3D conformation. To ensure uniformity and improve the model’s performance, all numerical features were normalized. Additionally, protein sequences were front-padded to align them for processing by the model. Building upon the integration of both sequential and numerical features, a novel model architecture was developed, as depicted in Figure 1 . Initially, the study aimed to maximize the utility of amino acids represented as character sequences. To achieve this, a lightweight transformer architecture was employed designed to capture bidirectional relationships within the data. This choice was determined on the understanding that specific amino acids often neighbor each other, and their sequential arrangement influences protein function and classification [ 26 ]. The transformer’s ability to focus on different parts of the sequence makes it suited for tasks where context significantly affects the output, which is the foundation of protein classification. Download figure Open in new tab Fig 1: Protein Model Diagram Download figure Open in new tab Fig 2: Sequence Length Download figure Open in new tab Fig 3: Class Distribution Download figure Open in new tab Fig 4: Numerical Data Diagram Simultaneously, a neural network was engineered, tasked with processing the numerical features. This network was designed to interpret the physical properties of proteins, identifying key relationships within the latent space. The outputs from both the transformer and the concurrent neural network were then converged into a fully connected layer. This layer acts as a fusion point that combines learned sequential and physical insights, subsequently feeding into a neural network that results in a classification output. This architecture represents a novel approach in protein classification, leveraging dual insights from amino acid sequences and their intrinsic physical properties. The combination of the sequential understanding provided by the transformer and the contextual insights from the numerical data processing allows for a more comprehensive classification. This method not only enhances accuracy but also offers a more holistic view of protein functionalities. The model underwent training on Kaggle, utilizing their free, accessible GPU resources. The training process spanned approximately 10 hours, during which the model completed 13 epochs. This duration was chosen as it marked the point of diminishing returns, where further training ceased to yield significant improvements in model performance. The decision to halt training at this stage ensured the optimal use of computational resources while preventing overfitting. IV. E xperimental S tudy The outcomes of the study are detailed in Table I , where this novel model was benchmarked against previously mentioned protein classification methods on the PDB dataset. This study’s model achieves a classification accuracy of 95%, which is an improvement of at least 15% over the next best-reported technique. This increase in accuracy highlights the effectiveness of this novel approach. View this table: View inline View popup Download powerpoint Table I: * Indicates classification on only 10 classes. Otherwise the models performed classification on 30 classes. Further demonstrating the robustness of the model, the evaluation metrics were expanded to include top 3 and top 5 accuracies. The results of these metrics are presented in Table II .This broader accuracy measurement further illustrates the model’s capability. View this table: View inline View popup Download powerpoint Table II: Model accuracy for top N predictions. To provide a practical idea of the model’s application, a detailed operation in a typical use case was documented. The process begins with extracting the amino acid sequences and initial numerical data from the dataset. Additional features are then derived using Biopython. This data undergoes normalization before being processed by the model, which then accurately outputs the protein’s class. V. C onclusion This study introduced a novel approach to protein classification, leveraging a unique architecture that combines the strengths of transformer models and neural networks. By combining sequential and numerical features of proteins, this novel model achieved a classification accuracy of 95%. This result not only demonstrates the model’s superior performance—surpassing existing methods by at least 15%—but also highlights the importance of leveraging physical properties. The significance of this study’s findings extends deep into the field of research. High-accuracy protein classification has implications in biomedical research and pharmaceutical development. By accurately categorizing proteins, researchers can better predict protein functions and interactions, which are crucial for drug discovery and the development of personalized medicine. This capability paves the way for more targeted therapies and faster identification of therapeutic targets, ultimately contributing to advances in treating complex diseases. Looking to the future, there are several avenues for enhancing this model. First, incorporating more diverse datasets, including those from different organisms and environmental conditions, could improve the model’s generalizability. Secondly, refinement of the model’s architecture to optimize computational efficiency would allow for scaling up to handle larger datasets. Additionally, exploring the integration of newer machine learning techniques, such as reinforcement learning, could provide deeper insights into dynamic protein behaviors and their implications on biological functions. The continued development and refinement of this model not only promise to elevate the standards of protein classification but also aims to contribute substantially to the broader field of proteomics, where understanding the landscape of proteins and their functions has great potential. VI. A vailability of D ata and M aterials Main database is available at RCSB’s website [ 23 ] All code and data available on Github [ 27 ]. VII. C ontributions N.L. and A.K. conceived the review article, collected data, organized figures, and performed all meta-analyses of the literature provided in the paper. C.C. and C.K. contributed to the oversight of this article as co-principal investigators. IX. C ompeting I nterests For N.L., A.K., C.C., C.K., and authors that are affiliated with Revilico Inc.: Revilico Inc. is a corporation focused on AI-driven drug discovery. We specialize in repurposing abandoned therapeutics with an emphasis on developing computational and predictive modeling for enhanced understanding of diseases and drugs. Revilico Inc. holds proprietary algorithms in the field of drug repurposing for several disease states and drug targets. The authors are engaged in creating and applying AI models to facilitate drug discovery and to provide greater insights into a broad array of conditions, including metabolic diseases. N.L., A.K., C.C., and C.K. are affiliated with Revilico Inc., and have contributed to the research and development of the disease/drug models discussed in this review. No other conflicts are reported. VIII. A cknowledgments The authors are grateful to the various research studies, clinical trials, and individuals behind the work of documenting protein structures. The authors would like to extend their gratitude to the entire Research and Development team at Revilico for their guidance, support, and contributions to the overall team dynamics. R eferences [1]. ↵ W. Chen , X. Liu , S. Zhang , and S. Chen , “Artificial intelligence for drug discovery: Resources, methods, and applications ,” Molecular Therapy , 2024 , wei Chen and Shilin Chen contributed equally to this work . [Online]. Available: https://www.ncbi.nlm.nih.gov/pmc/articles/PMC10009646/ [2]. ↵ U.S. Department of Energy , “Doe explains…machine learning,” https://www.energy.gov/science/doe-explainsmachine-learning , n.d., accessed: 2024-09-03 . [Online]. Available: https://www.energy.gov/science/doe-explainsmachine-learning [3]. ↵ H. Alquran , A. A. Fahoum , A. Zyout , and I. A. Qasmieh , “A comprehensive framework for advanced protein classification and function prediction using synergistic approaches: Integrating bispectral analysis, machine learning, and deep learning ,” PLOS ONE, 2024, data curation, Formal analysis, Methodology, Resources, Software, Validation, Visualization, Writing – original draft, Writing– review editing, corresponding author* . [Online]. Available:URL https://www.ncbi.nlm.nih.gov/pmc/articles/PMC10721063/ [4]. ↵ M. Puccetti , M. Pariano , A. Schoubben , S. Giovagnoli , and M. Ricci , “Biologics, theranostics, and personalized medicine in drug delivery systems ,” Pharmacological Research , vol. 195 , p. 107086 , 2024 . [Online]. Available: 10.1016/j.phrs.2024.107086 OpenUrl [5]. ↵ Grand View Research , “Research-grade proteins market report, 2030,” https://www.grandviewresearch.com/industry-analysis/research-grade-proteins-market-report , 2024 , accessed: 2024-09-03 . [On-line]. Available: https://www.grandviewresearch.com/industry-analysis/research-grade-proteins-market-report [6]. ↵ A. Vaswani , N. Shazeer , N. Parmar , J. Uszkoreit , L. Jones , A. N. Gomez , L. Kaiser , and I. Polosukhin , “Attention is all you need,” 2023 . [Online]. Available: https://arxiv.org/abs/1706.03762 [7]. ↵ T. Lin , Y. Wang , X. Liu , and X. Qiu , “A survey of transformers ,” AI Open , vol. 3 , pp. 111 – 132 , 2022 . [Online]. Available: https://www.sciencedirect.com/science/article/pii/S2666651022000146 OpenUrl CrossRef [8]. ↵ W.-C. Chang , H.-F. Yu , K. Zhong , Y. Yang , and I. S. Dhillon , “ Taming pretrained transformers for extreme multi-label text classification ,” in Proceedings of the 26th ACM SIGKDD International Conference on Knowledge Discovery & Data Mining, ser. KDD ‘20 . New York, NY, USA : Association for Computing Machinery , 2020 , p. 3163 – 3171 . [Online]. Available: 10.1145/3394486.3403368 [9]. ↵ S. Gao , M. Alawad , M. T. Young , J. Gounley , N. Schaefferkoetter , H. J. Yoon , X.-C. Wu , E. B. Durbin , J. Doherty , A. Stroup , L. Coyle , and G. Tourassi , “Limitations of transformers on clinical text classification ,” IEEE Journal of Biomedical and Health Informatics , vol. 25 , no. 9 , pp. 3596 – 3607 , 2021 . OpenUrl CrossRef [10]. ↵ S. Khan , M. Naseer , M. Hayat , S. W. Zamir , F. S. Khan , and M. Shah , “Transformers in vision: A survey ,” ACM Comput. Surv ., vol. 54 , no. 10s , sep 2022 . [Online]. Available: 10.1145/3505244 [11]. ↵ A. Oh , T. Naumann , A. Globerson , K. Saenko , M. Hardt , and S. Levine C. Li , C. Wong , S. Zhang , N. Usuyama , H. Liu , J. Yang , T. Naumann , H. Poon , and J. Gao , “ Llava-med: Training a large language-and-vision assistant for biomedicine in one day ,” in Advances in Neural Information Processing Systems , A. Oh , T. Naumann , A. Globerson , K. Saenko , M. Hardt , and S. Levine , Eds., vol. 36 . Curran Associates, Inc ., 2023 , pp. 28 541 – 28 564 . [Online]. Available: https://proceedings.neurips.cc/paperfiles/paper/2023/file/5abcdf8ecdcacba028c6662789194572-Paper-DatasetsandBenchmarks.pdf OpenUrl [12]. ↵ N. Labiosa , D. T. Huynh , D. S.-N. Lim , and Wisconsin-Madison , “Visual information and large language models: A deeper analysis,” 2023 . [Online]. Available: https://api.semanticscholar.org/CorpusID:260636471 [13]. ↵ S. Shen , L. H. Li , H. Tan , M. Bansal , A. Rohrbach , K.-W. Chang , Z. Yao , and K. Keutzer , “How much can clip benefit vision-and-language tasks?” 2021 . [Online]. Available: https://arxiv.org/abs/2107.06383 [14]. ↵ S. Diplaris , G. Tsoumakas , P. A. Mitkas , and I. Vlahavas , “Protein classification with multiple algorithms,” in Advances in Informatics: 10th Panhellenic Conference on Informatics, PCI 2005, Volas, Greece, November 11-13, 2005 . Proceedings 10 . Springer , 2005 , pp. 448 – 456 . OpenUrl [15]. ↵ K. Jain , “Role of pharmacoproteomics in the development of personal-ized medicine ,” Pharmacogenomics , vol. 5 , no. 3 , pp. 331 – 336 , 2004 . OpenUrl CrossRef PubMed Web of Science [16]. ↵ B. Ramsundar , P. Eastman , P. Walters , V. Pande , K. Leswing , and Z. Wu , Deep Learning for the Life Sciences . O’Reilly Media , 2019 , https://www.amazon.com/Deep-Learning-Life-Sciences-Microscopy/dp/1492039837 . [17]. ↵ N. Brandes , D. Ofer , Y. Peleg , N. Rappoport , and M. Linial , “ProteinBERT: a universal deep-learning model of protein sequence and function ,” Bioinformatics , vol. 38 , no. 8 , pp. 2102 – 2110 , 02 2022 . [Online]. Available: 10.1093/bioinformatics/btac020 OpenUrl CrossRef PubMed [18]. ↵ A. Elnaggar , M. Heinzinger , C. Dallago , G. Rehawi , Y. Wang , L. Jones , T. Gibbs , T. Feher , C. Angerer , M. Steinegger et al. , “Prottrans: Toward understanding the language of life through self-supervised learning ,” IEEE transactions on pattern analysis and machine intelligence , vol. 44 , no. 10 , pp. 7112 – 7127 , 2021 . OpenUrl [19]. ↵ A. Elnaggar , M. Heinzinger , C. Dallago , G. Rehawi , Y. Wang , L. Jones , T. Gibbs , T. Feher , C. Angerer , M. Steinegger , “Prottrans: Toward understanding the language of life through self-supervised learning ,” IEEE Transactions on Pattern Analysis and Machine Intelligence , vol. 44 , no. 10 , pp. 7112 – 7127 , 2022 . OpenUrl CrossRef PubMed [20]. ↵ D. Hjek , “Protein sequence classification,” https://www.kaggle.com/code/davidhjek/protein-sequence-classification/notebook , 022 , accessed: 2023-09-28 . [21]. ↵ A. Bhargava , “Predicting protein classification,” https://www.kaggle.com/code/abharg16/predicting-protein-classification/code , 2022 , accessed: 2023-09-28 . [22]. ↵ D. Ofer , “Deep protein sequence family classification,” https://www.kaggle.com/code/danofer/deep-protein-sequence-family-classification , 2022 , accessed: 2023-09-28 . [23]. ↵ RCSB Protein Data Bank , “Rcsb pdb: Protein data bank,” 2024 . [24]. ↵ imbalanced-learn developers , “Documentation of im-blearn.over sampling.randomoversampler,” 2024 . [25]. ↵ P. J. Cock , T. Antao , J. T. Chang , B. A. Chapman , C. J. Cox , A. Dalke Friedberg , T. Hamelryck , F. Kauff , B. Wilczynski et al. , “Biopython: freely available python tools for computational molecular biology and bioinformatics ,” Bioinformatics , vol. 25 , no. 11 , p. 1422 , 2009 . OpenUrl CrossRef PubMed Web of Science [26]. ↵ F. Sanger , “ The arrangement of amino acids in proteins ,” in Advances in protein chemistry . Elsevier , 1952 , vol. 7 , pp. 1 – 67 . OpenUrl CrossRef [27]. ↵ N. Labiosa , “Github repository containing all code.” [Online]. Available: https://github.com/NathanLabiosa/AttentionAminoAcids View the discussion thread. Back to top Previous Next Posted November 03, 2024. Download PDF Email Thank you for your interest in spreading the word about bioRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following Hybrid Transformer and Neural Network Configuration for Protein Classification Using Amino Acids Message Subject (Your Name) has forwarded a page to you from bioRxiv Message Body (Your Name) thought you would like to see this page from the bioRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share Hybrid Transformer and Neural Network Configuration for Protein Classification Using Amino Acids Nathan Labiosa , Aryan Kohli , Christian Chung , Christopher Korban bioRxiv 2024.10.31.621421; doi: https://doi.org/10.1101/2024.10.31.621421 Share This Article: Copy Citation Tools Hybrid Transformer and Neural Network Configuration for Protein Classification Using Amino Acids Nathan Labiosa , Aryan Kohli , Christian Chung , Christopher Korban bioRxiv 2024.10.31.621421; doi: https://doi.org/10.1101/2024.10.31.621421 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Bioengineering Subject Areas All Articles Animal Behavior and Cognition (7644) Biochemistry (17728) Bioengineering (13916) Bioinformatics (42037) Biophysics (21489) Cancer Biology (18637) Cell Biology (25553) Clinical Trials (138) Developmental Biology (13401) Ecology (19941) Epidemiology (2067) Evolutionary Biology (24367) Genetics (15622) Genomics (22547) Immunology (17764) Microbiology (40475) Molecular Biology (17208) Neuroscience (88747) Paleontology (667) Pathology (2842) Pharmacology and Toxicology (4834) Physiology (7659) Plant Biology (15175) Scientific Communication and Education (2047) Synthetic Biology (4304) Systems Biology (9835) Zoology (2272)

Text is read by the "Ask this paper" AI Q&A widget below. Extraction quality varies by source — PMC NXML preserves structure cleanly, OA-HTML may include some navigation residue, and OA-PDF can have broken hyphenation. The publisher copy (via DOI) is the canonical version.

My notes (saved in your browser only)

Ask this paper AI returns verbatim quotes from the full text · source: preprint-html

Answers must be backed by verbatim quotes from this paper's full text. Hallucinated quotes are dropped automatically; if no verbatim passage answers the question, we say so. How this works

Citation neighborhood (no data yet)

We don't have any in-corpus citations linked to this paper yet. This is a recent paper (2024) — citers typically take a year or two to land, and the OpenAlex reference graph may still be filling in.

Source provenance

europepmc
last seen: 2026-05-20T01:45:00.602351+00:00
unpaywall
last seen: 2026-08-09T06:42:26.407065+00:00