Full text
40,962 characters
· extracted from
preprint-html
· click to expand
Application of Quantum Tensor Networks for Protein Classification | bioRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-M677548'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search New Results Application of Quantum Tensor Networks for Protein Classification Debarshi Kundu , View ORCID Profile Archisman Ghosh , Srinivasan Ekambaram , View ORCID Profile Jian Wang , Nikolay Dokholyan , Swaroop Ghosh doi: https://doi.org/10.1101/2024.03.11.584501 Debarshi Kundu 1 Pennsylvania State University , State College, PA, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site Archisman Ghosh 1 Pennsylvania State University , State College, PA, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Archisman Ghosh Srinivasan Ekambaram 1 Pennsylvania State University , State College, PA, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site Jian Wang 1 Pennsylvania State University , State College, PA, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Jian Wang Nikolay Dokholyan 1 Pennsylvania State University , State College, PA, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site Swaroop Ghosh 1 Pennsylvania State University , State College, PA, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site For correspondence: szg212{at}psu.edu Abstract Full Text Info/History Metrics Preview PDF Abstract Computational methods in drug discovery significantly reduce both time and experimental costs. Nonetheless, certain computational tasks in drug discovery can be daunting with classical computing techniques which can be potentially overcome using quantum computing. A crucial task within this domain involves the functional classification of proteins. However, a challenge lies in adequately representing lengthy protein sequences given the limited number of qubits available in existing noisy quantum computers. We show that protein sequences can be thought of as sentences in natural language processing and can be parsed using the existing Quantum Natural Language framework into parameterized quantum circuits of reasonable qubits, which can be trained to solve various proteinrelated machine-learning problems. We classify proteins based on their sub-cellular locations—a pivotal task in bioinformatics that is key to understanding biological processes and disease mechanisms. Leveraging the quantum-enhanced processing capabilities, we demonstrate that Quantum Tensor Networks (QTN) can effectively handle the complexity and diversity of protein sequences. We present a detailed methodology that adapts QTN architectures to the nuanced requirements of protein data, supported by comprehensive experimental results. We demonstrate two distinct QTNs, inspired by classical recurrent neural networks (RNN) and convolutional neural networks (CNN), to solve the binary classification task mentioned above. Our top-performing quantum model has achieved a 94% accuracy rate, which is comparable to the performance of a classical model that uses the ESM2 protein language model embeddings. It’s noteworthy that the ESM2 model is extremely large, containing 8 million parameters in its smallest configuration, whereas our best quantum model requires only around 800 parameters. We demonstrate that these hybrid models exhibit promising performance, showcasing their potential to compete with classical models of similar complexity. 1 INTRODUCTION Importance of Proteins in drug discovery Proteins are essential, large biomolecules that serve numerous vital functions within the body. Constructed from sequences of amino acids, proteins are the end product of long chains comprised of these units. The human body utilizes 20 distinct amino acids, the specific ordering of which crafts a protein’s unique three-dimensional shape and determines its particular role. This order of amino acids is encoded by the DNA through sequences of genes, which specify the arrangement of three nucleotide building blocks to form each amino acid. Protein functions reflect their fundamental importance in biological processes. In the realm of drug discovery, the knowledge of protein structures and functions is crucial. For instance, understanding how a protein interacts with other molecules can guide the design of drugs that can modulate these interactions. A classic example is the development of inhibitors targeting HIV protease, a pivotal enzyme in the HIV life cycle. The structural and functional insights into the HIV protease have led to the creation of drugs that specifically inhibit this enzyme, significantly improving the treatment of HIV infection [ 1 ]. This approach highlights the importance of protein sequence and structure knowledge in designing therapeutic agents that can effectively target specific molecular pathways [ 2 ]. Machine learning for protein engineering The incorporation of computational methods, notably in protein design, has significantly shortened experimental timelines in drug discovery. The advent of machine learning (ML) has particularly transformed protein sequence analysis, as evidenced by innovations like ESM-2[ 3 ] and AlphaFold[ 4 ]. ESM-2 has improved our predictive capabilities for protein functions from sequences, while AlphaFold has revolutionized protein structure prediction, reaching near-experimental accuracy. AlphaFold’s success in the CASP competitions highlights this progress. These advancements underscore the efficiency of ML in enhancing our comprehension of protein structures and functions, leveraging large datasets to reveal previously unknown predictive relationships. Despite these achievements, the complexity of some computational biology problems exceeds the capabilities of classical computing, suggesting a pivotal role for quantum machine learning in navigating these challenges more effectively. Quantum Natural Language Processing (QNLP) The application of quantum computing to Natural Language Processing (NLP) involves representing word embeddings as quantum states in quantum Hilbert space to obtain a polynomial time speedup in classification-based problems. The word embeddings in QNLP are prepared using linear maps to the tensor product of the word vectors, to find the meaning of a sentence [ 5 ]. PQCs are the primary trainable component in the QNLP circuit comprising entangling operations between qubits and parameterized single-qubit rotations. The entanglement operations are mostly multi-qubit operations between all qubits to generate correlated states and the parameterized rotations are used to search the solution space. The QNLP circuit is a Quantum Tensor Network that represents sentences as high-dimensional tensors and parses them into a network of more manageable, lower-dimension tensors that aid classification. Protein Sequence as a sentence Viewing a protein sequence as a sentence, with each amino acid acting as a word, presents a powerful analogy for deciphering the structure and function of proteins [ 6 ]. This perspective highlights the complexity and specificity inherent in protein sequences, likening a protein to a carefully composed sentence where the amino acids are arranged in a precise sequence. Each “word” (amino acid) adds its unique characteristics to the “sentence” (protein sequence), influencing the protein’s folding, structure, and function within biological systems. For example, the positioning of amino acids such as lysine, arginine, and glutamate within a protein can be seen as words constructing a sentence, where the exact sequence and arrangement are critical for the protein’s capability to fulfill its designated roles. Similarly, altering a word in a sentence can change its entire meaning, just as modifying an amino acid in a protein sequence can drastically affect the protein’s functionality. This analogy emphasizes the critical nature of sequence fidelity in proteins, underscoring the finely tuned equilibrium of biological systems where each “word” plays a pivotal role in the “narrative” of life [ 7 ]. Motivation Integrating Quantum Natural Language Processing (QNLP) into protein sequence analysis offers the potential to revolutionize drug discovery and our understanding of biological processes. This approach leverages quantum computing’s capacity for high-dimensional space management, speed, and semantic analysis, enabling more accurate predictions of protein functions, structures, and interactions. By viewing amino acid sequences as sentences, Quantum NLP can provide deeper insights into the protein “language,” surpassing current bioinformatic tools in sequence alignment, functional motif identification, and protein function annotation[ 8 ], [ 9 ]. This integration could significantly advance drug discovery by exploiting quantum algorithms for enhanced efficiency and speed. However, there is a need to represent the long protein sequences in the quantum circuits with reasonably small qubits and reasonably deep quantum circuits in the NISQ-era quantum computers. There is a need to extract the signal from these sequences to solve important challenges in drug discovery. Contributions Building on the foundational work by [ 10 ], our study marks the proof of concept of classifying long protein sequences leveraging Quantum Tensor Networks (QTNs). This advancement is pivotal, as it not only extends the applicability of QTNs beyond their traditional domains but also introduces a novel methodology for handling the complexities inherent in protein sequence data. The utilization of quantum computing in this context is not merely for its computational prowess but also for its ability to capture the intricate patterns and relationships within biological sequences, which are often beyond the reach of classical computational techniques. In a nutshell, our contributions to this paper are as follows: (a) we have successfully demonstrated, for the first time, the potential of QTNs[ 10 ] in the classification of long protein sequences, (b) we showed that QTNs, inspired by convolutional and recurrent neural networks, are capable of learning representations of proteins using a relatively small qubit circuit and (c) our findings underscore a significant advancement over classical models, showcasing the inherent advantages of quantum computing in processing and classifying biological data. This comparison not only validates the effectiveness of our approach but also sets the stage for future explorations into quantum bioinformatics. 2 PROPOSED MODELS Here we first explain the entire process of generating a parametrized quantum circuit from a protein sequence and training the circuits for the binary classification task ( Fig. 1 ). Then we explain two important steps to develop the quantum models namely, (a) building compositional schemes to convert protein sequences into networks which are represented using wires and boxes and (b) defining semantic functor to map these networks into quantum circuits. Download figure Open in new tab Figure 1: A diagram describing the flow of binary classification of the protein sequence. Protein to quantum model pipeline The process begins with a protein sequence input that undergoes parsing using a state-ofthe-art neural-trained parser, resulting in a protein syntax tree. This tree is encoded into a string diagram, abstractly representing the relationships between elements in the sequence. These string diagrams are based on category theory [ 11 ] and can be simplified by rewriting rules to reduce redundancy and adapt the computation for quantum processors. After rewriting, the diagrams are parameterized and converted into a quantum circuit using specific parameterization schemes and ansätze choices. The quantum circuit is then classified using a QTN, resulting in binary labels, and is ready for training. This entire model is structured based on the grammatical construction of the input sequence and is optimized for implementation on quantum processing units. Compositional Schemes A compositional scheme is initially defined for a given sequence, employing the graphical language of process theories. The processes within these schemes are represented by boxes, which have input and output wires. These wires carry types either the ‘internal’ type τ or the ‘sentence’ type σ . The composition of these boxes, following type constraints, allows for the generation of process diagrams that represent the scheme for sequence analysis. Given a vocabulary 𝒱 = { m i } i comprising a finite set of words (or tokens), we consider compositional schemes for sequences 𝒮 of finite length over this vocabulary. The schemes are then semantically mapped onto QTN models ( Fig. 2 ). Download figure Open in new tab Figure 2: Assignment of parameterized quantum circuits U ( ϕi ) to boxes labeled i : This is an example of a function definition, where the Words are mapped to qubits (bits in case of blue wires) via the unitary matrices U ( ϕ ) . The ⊥ is either represented as an all-zeroes state, postselect, or discard. Semantic Functor It is a structure-preserving map that assigns Hilbert space semantics to the compositional schemes, thereby designing parameterized quantum circuits (PQCs) for various components of the scheme. For handling non-deterministic outcomes, two strategies are defined: postselect and discard. Postselect involves conditioning on a particular measurement outcome, typically the all-zeros state, while discard involves ignoring specific dimensions of the quantum state like a partial trace. Parameterized Quantum Circuits (PQCs) for QTNs The PQCs designed by ℱ involves Word-State Preparation ( Fig. 2A ): These boxes prepare a word-state of type τ ⊗ 0 → τ ⊗ 1 , which is associated with a parameterized quantum state prepared by applying the circuit U ( ϕ m )to the fixed input state |0 ⟩ ⊗q . Each word corresponds to a unique set of parameters ϕ m , Filter Application ( Fig. 2B ): The filter boxes, with type τ ⊗ 2 → τ ⊗ 2 , are associated with a U ( ϕ m ) operating on 2 q input and 2 q output qubits, simulating the filtering process within the protein sequence, Merge Operation ( Fig. 2C ): The m-box, typed τ ⊗ 2 → τ ⊗ 1 , is mapped to U ( ϕ m ), with 2 q input qubits and q output qubits. The reduction in qubits is achieved by either discarding or postselecting the redundant qubits via the ⊥ -effect, and Classification ( Fig. 2D ): The classifier boxes, typed τ ⊗ 1 → σ ⊗ 1 , is associated with a unitary U ( ϕ m ), which processes a q -qubit state input and outputs a q ′ -qubit state. This state is subsequently measured in the Z basis, yielding a vector in , representing the classification outcome based on the Born rule probabilities. These strategies influence the tensor network topology, allowing for efficient tensor contraction which encapsulates the sequence processing task within a quantum framework. Example: A protein sequence is represented as a sentence and broken down into independent Words and then run through the QTN ( Fig. 3 ). We dive deep into the idea by considering a protein sequence of length four ( AGSQ ) in Fig. 4 and define the aminoacids as different Words that are then converted to the respective unitaries (parameterized input states) based on the rules of the Functor mapping ( Fig. 2 ). These mapped states finally culminate into the Quantum Convolution Tensor Net ( Fig. 5 ). Download figure Open in new tab Figure 3: A sentence is broken into Words m 1 … mn , converted to corresponding unitaries based on the Functor rules in Fig. 2 and finally run through the Scheme ( QT N ). Download figure Open in new tab Figure 4: An example to parse an demonstrative protein sequence AGSQ into a protein syntax tree based on CTN. Download figure Open in new tab Figure 5: Convolutional tensor network (CTN) 2.1 Compositional scheme: Recurrent Neural Net inspired We examine QTNs with a model that adheres to the sequential flow akin to the natural progression of words in a text. This model, derived by implementing the semantic functor ℱ outlined in Fig. 2 , results in what we refer to as the path tensor network model, symbolized as ℱ ( ϕ ) [path] = PTN, and illustrated in Fig. 6 . This approach sequentially aligns with the reading order, mapping out a straightforward path through the sequence. Download figure Open in new tab Figure 6: Path tensor network (PTN) 2.2 Convolutional Tensor Networks Hierarchical and Uniform PTN Models Delving deeper, we introduce a nuanced layer to the model by attributing a hierarchical structure to the parameter sets { ϕ mi }, contingent upon their sequential position i , within the range of. {1, 2, …, | S | − 1 } This adjustment births the hierarchical PTN (hPTN) models. Proceeding further, we harmonize the parameters across all merging circuits ( m -circuits) to a singular set, ϕ m = ϕ mi for all i , thereby engendering a recurrent structure, which we term the uniform PTN (uPTN). This uPTN model distills down to a basic form of a recurrent quantum model, or equivalently, a matrix product state (MPS) model, in which all dimensions except for the last are either disregarded or selected based on outcomes, with the remaining dimension capturing the overall semantic value of the sentence. This initial compositional approach intentionally bypasses considerations of syntax and distant correlations, positioning it as an elementary framework for juxtaposition with models that incorporate syntactic awareness. While simplistic, this model is built upon a principle of local compositional application, serving as a critical benchmark. Our approach uses a refined compositional scheme, similar to convolutional neural networks, which serves as an advancement over the basic tree structure. This scheme is ingeniously crafted by integrating additional layers of filtering boxes (f-boxes) into the tree architecture. These f-boxes are strategically placed to operate prior to the merging boxes (m-boxes) along adjacent wires that do not converge into the same m-box. This setup results in a convolutional tensor network (CTN), distinguished by its ability to filter out superfluous entanglements at each layer through the f-circuits, followed by a consolidation of qubit wires by the m-circuits. This process effectively distills the sequence, preserving only the essential information pertinent to the designated task. Hierarchical and Uniform Variants The CTN model evolves into hierarchical (hCTN) and uniform (uCTN) variants based on the distribution and uniformity of the parameter sets across the layers. The hierarchical model shares parameter sets within the same layer, whereas the uniform model extends this sharing across the entire model. 2.3 Classical Model To establish a baseline for rigorously assessing the capabilities of our quantum models, we developed a classical model architecture ( Fig. 7 ) that exploits deep learning techniques, specifically tailored for processing and classifying protein sequences. Central to our classical model is the integration of embeddings derived from the ESM2 [ 3 ] pretrained model. Esteemed as a cutting-edge development in machine learning for bioinformatics, ESM2 is intricately designed to distill meaningful features from protein sequences, thereby representing a substantial leap forward in our ability to capture the complex patterns and functional attributes inherent in proteins. This model’s ability to learn from an extensive compendium of protein sequences endows it with the capacity to abstract a profound representation of amino acid interrelations, structural motifs, and other critical biochemical properties. Download figure Open in new tab Figure 7: Protein sequence is fed into the ESM2 embedding that passes through a fully connected classical layer to classify the input sequence. Within our classical framework, each protein sequence undergoes initial processing by the ESM2 model, yielding a fixed-size (1024) embedding vector. This vector serves as a condensed representation of the sequence’s biological and contextual nuances, primed for subsequent analysis via neural network techniques. These embeddings are then channeled into a fully connected network, comprising three hidden layers of size (512, 256, 128) of interconnected neurons. The training regimen for this model utilizes the same labeled dataset of protein sequences mentioned earlier, employing binary cross-entropy as the loss function to fine-tune the network’s weights. Adam, renowned for its optimization efficacy [ 12 ], was the algorithm of choice for this process, facilitating efficient and effective model training. By juxtaposing this classical model’s performance against that of our quantum approaches, we endeavor to elucidate the quantum computing paradigm’s potential benefits and efficacy in tackling the challenges associated with protein sequence classification. 3 METHODOLOGY 3.1 Dataset Compilation The dataset has been compiled using protein sequences obtained from UniProt [ 13 ]. The dataset of human protein sequences has been cleaned and preprocessed to have 80 to 200 amino acids in each sequence. The protein sequences have a categorization that has been based on subcellular localization into two groups—proteins in the cytosol or cytoplasm and those associated with the cell membrane. Structure The dataset is made of 1136 protein sequences after preprocessing which is divided into training, validation, and testing data subsets having 980, 123, and 123 protein sequences respectively. The size of the dataset has been scaled down to cater to the long runtime of quantum simulations while ensuring model validation and a thorough evaluation of its predictive performance. Significance The dataset comprises protein sequences that can be classified based on their location in the cell (cytoplasm or cell membrane), therefore facilitating a deeper understanding of the functional implications of proteins based on their cellular locales. On a broader scale, it contributes to the study of complex cellular functions, and biological processes. 3.2 Implementation Details The lambeq [ 14 ] library, a forefront tool in quantum natural language processing (QNLP), introduces a sophisticated method for translating diagrammatic representations of linguistic structures into quantum circuits, enabling the exploration of various quantum ansatzes for processing and analysis. The ansatzes explored include the IQPAnsatz , Sim14Ansatz , Sim15Ansatz , and MPSAnsatz , each offering unique advantages for different types of quantum computations [ 15 ] as explained next. IQPAnsatz (Instantaneous Quantum Polynomial-time Ansatz) :This ansatz constructs circuits that are believed to implement computations not efficiently simulatable by classical computers, focusing on problems that can be encoded in a certain polynomial structure. It is particularly suited for tasks where quantum advantage is explored. The mapping from diagrammatic representations to quantum circuits within lambeq is facilitated by parsers like spiders_reader , cups_reader , and stairs_reader [ 14 ]. spiders_reader translates complex syntactic interactions into quantum circuits using graphical elements known as spiders, facilitating the representation of non-linear word relationships. cups_reader captures entanglement between elements in a sentence through cups in diagrammatic notation, effectively modeling pairwise dependencies. stairs_reader leverages a staircase pattern to represent the sequential flow and dependencies of elements within a sentence, ideal for capturing long-range contextual information crucial in understanding protein sequences. The transition from these diagrammatic representations to quantum circuits involves encoding linguistic or biological data as initial quantum states, followed by the application of quantum gates as dictated by the chosen ansatz. This process effectively translates the structure and semantics of the input data into a form amenable to quantum computation, enabling the exploration of quantum mechanical advantages in processing complex sequences. Simulation and training of these models are facilitated through the tensornetwork library and JAX, respectively, with the latter enabling Just-In-Time compilation for efficient processing [ 16 ] [ 17 ]. Among various quantum ansatz ( IQPAnsatz , Sim14Ansatz , Sim15Ansatz [ 15 ], MPSAnsatz ), the expressive ansatz 14 was selected for its notable test performance [ 15 ]. Optimization is achieved using AdamW, with the aim of minimizing binary cross-entropy loss for accurate label prediction [ 18 ]. Considering the prospect of quantum computer training, we suggest the parameter-shift rule for gradient estimation or the use of SPSA for its practicality in near-term quantum computing environments [ 19 – 21 ]. The model selection process uses k-fold validation and early stopping, focusing on hyperparameters like embedding qubit count, ansatz depth (q, D), and learning rate. In k-fold validation, the data is split into k parts, training on k-1 and validating on the remaining part, iteratively. This ensures a comprehensive evaluation. The model’s generalization is finally tested on an unseen dataset, assessing prediction accuracy. We ensure a consistent comparison framework by reporting test accuracies at the peak of validation accuracy for specified hyperparameters and learning rate settings, under a fixed seed for reproducibility. 4 RESULTS Our model follows an encoder architecture, which essentially means they accept sentences as inputs and generate corresponding outputs, thereby serving the role of classifiers. These models are predominantly aimed at binary classification tasks, involving the measurement of a singular qubit ( q ′ = 1) from the output quantum state produced by the U c circuit. This process determines probabilities for two possible outcomes, p 0 and p 1 by measuring the average state of the qubits, with each outcome directly mapping to a class label. The tree-like design of our introduced model species promotes not only efficient computation but also a natural resistance to the occurrence of barren plateaus during the training process, a notable hurdle in the optimization of quantum models[ 22 – 24 ]. We have explored different Quantum Tensor Networks, viz., the Path Tensor Network (PTN) and the Convolutional Tensor Network (CTN)—with hierarchical and uniform parameter-sharing strategies. We analyze the performance of four models: the uniform path tensor network (uPTN), the hierarchical path tensor network (hPTN), the uniform convolutional tensor network (uCTN), and the hierarchical convolutional tensor network (hCTN). Our results, summarized in two tables ( Table 1 and 2 ), provide the test accuracy and F1-scores for each model, with and without post-selection. The hPTN model outperforms the others in both metrics significantly, indicating its superior capability in capturing the necessary features for classification. The uPTN follows, showing decent performance, but not matching the hierarchical counterpart. The uCTN and hCTN models exhibit lower performance, with uCTN scoring the lowest on both accuracy and F1-score. View this table: View inline View popup Download powerpoint Table 1: Test accuracy for evaluated models under different measurement procedures (discard) and (postselect) View this table: View inline View popup Download powerpoint Table 2: F1-score for evaluated models under different measurement procedures (discard) and (postselect) Model Performance Analysis The PTN-based models have a sequential architecture that aligns well with the natural sequence of amino acids in protein sequences allowing better classification accuracy. In hPTN models ( Fig. 8 ), the sequential information of protein sequences along with the varying dependencies are captured very effectively due to the presence of unique parameter sets in the model. uPTN models, based on recurrent quantum nets, perform consistently well but are limited in flexibility by a single shared parameter set across all PQCs. Although CTNs have enhanced tree-like architecture, they are not very adept at capturing the long-range correlations effectively. The slight improvement of the performance of hCTN models over uCTN models suggests that having a hierarchical unique parameter set helps in classification more than having a single shared parameter set. In the comparative analysis of quantum models for protein sequence classification, the ESM2-based classical model ( Fig. 7 ), with its high accuracy of 0.98, sets a significant benchmark for performance. Notably, the hierarchical Path Tensor Network (hPTN) quantum model closely rivals this benchmark with an impressive accuracy of 0.94. This near-parity highlights the substantial potential of quantum models to reach and possibly exceed the performance standards set by advanced classical models in complex biological computations. Download figure Open in new tab Figure 8: Accuracy and Loss plots for training, test, and validation data evaluated for PTN models under postselect and discard measurements. Plots A-D represent postselect measurement for hierarchical (A, B) and uniform (C, D) PTNs. Plots E-H represent discard measurement for hierarchical (E, F) and uniform (G, H) PTNs. Model runtime analysis CTN models, characterized by their substantial parameter count, require significantly longer training durations, approximately 8 hours per epoch. Conversely, PTN models exhibit a markedly swifter training pace, completing the entire simulation in about 1 hour. Intriguingly, CTN models achieve convergence in fewer epochs (around 4 to 5) compared to PTN which took much higher number of epochs as shown in Fig. 8 . Despite the classical model being trained within an hour, it’s important to acknowledge that the foundational pre-trained protein language model, ESM2, underwent a training process powered by extensive computational resources for nearly a week. Limitations The QTNs were evaluated under idealized conditions devoid of quantum noise, likely leading to an optimistic representation of their capabilities. Such an omission of quantum noise considerations prevents a fully equitable comparison to the classical ESM2 model and overlooks the challenges posed by the hardware noise in the devices of the current noisy quantum computers. Future research should prioritize addressing this gap by incorporating noise models in the simulation backends. 5 CONCLUSION We proposed and evaluated four flavors of Quantum Tensor Nets (QTNs) namely, hierarchical Path Tensor Network (hPTN), uniform Path Tensor Network (uPTN), hierarchical Convolutional Tensor Network (hCTN), and uniform Convolutional Tensor Network (uCTN) to classify protein sequences based on their cellular locales (cytosol/cytoplasm or cell membrane). The hPTN model demonstrated superior performance in classifying protein sequences compared to its uniform counterpart and the convolutional tensor network variants. The uniform models, particularly the uCTN, require further investigation and possible architectural or parameter adjustments to improve their performance. ACKNOWLEDGMENTS We extend our gratitude to Avimita Chatterjee (Pennsylvania State University) for her valuable insights and assistance in structuring the work. The work is supported in parts by the National Science Foundation (NSF) (CNS-1722557, CCF-1718474, OIA-2040667, DGE-1723687, and DGE-1821766). Footnotes dqk5620{at}psu.edu apg6127{at}psu.edu je5459{at}psu.edu juw1179{at}psu.edu nxd338{at}psu.edu szg212{at}psu.edu REFERENCES [1]. ↵ G Sudhakararao et al. Physiological role of proteins and their functions in human body . International Journal of Pharma Research and Health Sciences , 7 : 2874 – 2878 , 2019 . OpenUrl [2]. ↵ G Harvey Anderson et al. Dietary proteins in the regulation of food intake and body weight in humans . The Journal of nutrition , 134 ( 4 ): 974S – 979S , 2004 . OpenUrl Abstract / FREE Full Text [3]. ↵ Zeming Lin et al. Evolutionary-scale prediction of atomic-level protein structure with a language model . Science , 379 ( 6637 ): 1123 – 1130 , 2023 . OpenUrl CrossRef PubMed [4]. ↵ Patrick Bryant et al. Improved prediction of protein-protein interactions using alphafold2 . Nature communications , 13 ( 1 ): 1265 , 2022 . OpenUrl [5]. ↵ Bob Coecke et al. Mathematical foundations for a compositional distributional model of meaning . arXiv preprint arXiv: 1003.4394 , 2010 . [6]. ↵ Ofer et al. The language of proteins: Nlp, machine learning & protein sequences . Computational and Structural Biotechnology Journal , 19 : 1750 – 1758 , 2021 . OpenUrl [7]. ↵ Yandell et al. Genomics and natural language processing . Nature Reviews Genetics , 3 ( 8 ): 601 – 610 , 2002 . OpenUrl CrossRef PubMed Web of Science [8]. ↵ Mohammad Hassan Khatami et al. Gate-based quantum computing for protein design . PLOS Computational Biology , 19 ( 4 ): e1011033 , 2023 . OpenUrl [9]. ↵ Lloyd CL Hollenberg . Fast quantum search algorithms in protein sequence comparisons: Quantum bioinformatics . Physical Review E , 62 ( 5 ): 7532 , 2000 . OpenUrl PubMed [10]. ↵ Carys Harvey et al. Sequence processing with quantum tensor networks . arXiv preprint arXiv: 2308.07865 , 2023 . [11]. ↵ Philippe de Groote , Glyn Morrill , and Christian Retoré Wojciech Buszkowski . Lambek grammars based on pregroups . In Philippe de Groote , Glyn Morrill , and Christian Retoré , editors, Logical Aspects of Computational Linguistics , pages 95 – 109 , Berlin, Heidelberg , 2001 . Springer Berlin Heidelberg . [12]. ↵ Diederik P Kingma et al. Adam: A method for stochastic optimization . arXiv preprint arXiv: 1412.6980 , 2014 . [13]. ↵ Uniprot: the universal protein knowledgebase in 2023 . Nucleic Acids Research , 51 ( D1 ): D523 – D531 , 2023 . OpenUrl CrossRef [14]. ↵ Dimitri Kartsaklis et al. lambeq: An Efficient High-Level Python Library for Quantum NLP . arXiv preprint arXiv: 2110.04236 , 2021 . [15]. ↵ Sukin Sim et al. Expressibility and entangling capability of parameterized quantum circuits for hybrid quantum-classical algorithms . Advanced Quantum Technologies , 2 ( 12 ): 1900070 , 2019 . OpenUrl [16]. ↵ Chase Roberts et al. Tensornetwork: A library for physics and machine learning . arXiv preprint arXiv: 1905.01330 , 2019 . [17]. ↵ James Bradbury et al. JAX: composable transformations of Python+NumPy programs , 2018 . [18]. ↵ Ilya Loshchilov et al. Decoupled weight decay regularization , 2017 . [19]. ↵ Maria Schuld et al. Evaluating analytic gradients on quantum hardware . Physical Review A , 99 ( 3 ), Mar 2019 . [20]. J.C. Spall . Multivariate stochastic approximation using a simultaneous perturbation gradient approximation . IEEE Transactions on Automatic Control , 37 ( 3 ): 332 – 341 , 1992 . OpenUrl CrossRef Web of Science [21]. ↵ Xavier Bonet-Monroig et al. Performance comparison of optimization methods on variational quantum algorithms . Physical Review A , 107 ( 3 ), Mar 2023 . [22]. ↵ Arthur Pesah et al. Absence of barren plateaus in quantum convolutional neural networks . Physical Review X , 11 ( 4 ): 041011 , 2021 . OpenUrl [23]. Cervero Martín et al. Barren plateaus in quantum tensor network optimization . Quantum , 7 : 974 , Apr 2023 . [24]. ↵ Chen Zhao et al. Analyzing the barren plateau phenomenon in training quantum neural networks with the zx-calculus . Quantum , 5 : 466 , 2021 . View the discussion thread. Back to top Previous Next Posted March 14, 2024. Download PDF Email Thank you for your interest in spreading the word about bioRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following Application of Quantum Tensor Networks for Protein Classification Message Subject (Your Name) has forwarded a page to you from bioRxiv Message Body (Your Name) thought you would like to see this page from the bioRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share Application of Quantum Tensor Networks for Protein Classification Debarshi Kundu , Archisman Ghosh , Srinivasan Ekambaram , Jian Wang , Nikolay Dokholyan , Swaroop Ghosh bioRxiv 2024.03.11.584501; doi: https://doi.org/10.1101/2024.03.11.584501 Share This Article: Copy Citation Tools Application of Quantum Tensor Networks for Protein Classification Debarshi Kundu , Archisman Ghosh , Srinivasan Ekambaram , Jian Wang , Nikolay Dokholyan , Swaroop Ghosh bioRxiv 2024.03.11.584501; doi: https://doi.org/10.1101/2024.03.11.584501 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Bioinformatics Subject Areas All Articles Animal Behavior and Cognition (7651) Biochemistry (17746) Bioengineering (13928) Bioinformatics (42066) Biophysics (21499) Cancer Biology (18650) Cell Biology (25579) Clinical Trials (138) Developmental Biology (13409) Ecology (19947) Epidemiology (2067) Evolutionary Biology (24374) Genetics (15633) Genomics (22557) Immunology (17775) Microbiology (40505) Molecular Biology (17217) Neuroscience (88796) Paleontology (667) Pathology (2845) Pharmacology and Toxicology (4836) Physiology (7664) Plant Biology (15179) Scientific Communication and Education (2047) Synthetic Biology (4304) Systems Biology (9839) Zoology (2272)
Text is read by the "Ask this paper" AI Q&A widget below.
Extraction quality varies by source — PMC NXML preserves structure
cleanly, OA-HTML may include some navigation residue, and OA-PDF can
have broken hyphenation. The publisher copy
(via DOI)
is the canonical version.