Complex-based Ligand-Binding Proteins Redesign by Equivariant Diffusion-based Generative Models

preprint OA: closed
📄 Open PDF Full text JSON View at publisher

Abstract

Proteins, serving as the fundamental architects of biological processes, interact with ligands to perform a myriad of functions essential for life. The design and optimization of ligand-binding proteins are pivotal for advancing drug development and enhancing therapeutic efficacy. In this study, we introduce ProteinReDiff, a novel computational framework designed to revolutionize the redesign of ligand-binding proteins. Distinguished by its utilization of Equivariant Diffusion-based Generative Models and advanced computational modules, ProteinReDiff enables the creation of high-affinity ligand-binding proteins without the need for detailed structural information, leveraging instead the potential of initial protein sequences and ligand SMILES strings. Our thorough evaluation across sequence diversity, structural preservation, and ligand binding affinity underscores ProteinReDiff's potential to significantly advance computational drug discovery and protein engineering. Our source code is publicly available at https://github.com/HySonLab/Protein_Redesign
Full text 85,834 characters · extracted from preprint-html · click to expand
Complex-based Ligand-Binding Proteins Redesign by Equivariant Diffusion-based Generative Models | bioRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-M677548'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search New Results Complex-based Ligand-Binding Proteins Redesign by Equivariant Diffusion-based Generative Models Viet Thanh Duy Nguyen , Nhan D. Nguyen , View ORCID Profile Truong Son Hy doi: https://doi.org/10.1101/2024.04.17.589997 Viet Thanh Duy Nguyen † FPT Software AI Center , Ho Chi Minh, Vietnam Find this author on Google Scholar Find this author on PubMed Search for this author on this site Nhan D. Nguyen ‡ Pritzker School of Molecular Engineering, University of Chicago , Chicago, IL 60637, United States Find this author on Google Scholar Find this author on PubMed Search for this author on this site Truong Son Hy ¶ Department of Mathematics and Computer Science, Indiana State University , Terre Haute, IN 47807, United States Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Truong Son Hy For correspondence: TruongSon.Hy{at}indstate.edu Abstract Full Text Info/History Metrics Preview PDF Abstract Proteins, serving as the fundamental architects of biological processes, interact with ligands to perform a myriad of functions essential for life. Designing functional ligand-binding proteins is pivotal for advancing drug development and enhancing therapeutic efficacy. In this study, we introduce ProteinReDiff, an efficient computational framework targeting the redesign of ligand-binding proteins. Using equivariant diffusion-based generative models, ProteinReDiff enables the creation of high-affinity ligand-binding proteins without the need for detailed structural information, leveraging instead the potential of initial protein sequences and ligand SMILES strings. Our evaluations across sequence diversity, structural preservation, and ligand binding affinity underscore ProteinReDiff’s potential to advance computational drug discovery and protein engineering. Our source code is publicly available at https://github.com/HySonLab/Protein_Redesign . Introduction Proteins, often referred to as the molecular architects of life, play a critical role in virtually all biological processes. A significant portion of these functions involves interactions between proteins and ligands, underpinning the complex network of cellular activities. These interactions are not only pivotal for basic physiological processes, such as signal transduction and enzymatic catalysis, but also have broad implications in the development of therapeutic agents, diagnostic tools, and various biotechnological applications 1 – 3 . Despite the paramount importance of protein-ligand interactions, the majority of existing studies have primarily focused on protein-centric designs to optimize specific protein properties, such as stability, expression levels, and specificity 4 – 8 . This prevalent approach, despite leading to numerous advancements, does not fully exploit the synergistic potential of optimizing both proteins and ligands for redesigning ligand-binding proteins. By embracing an integrated design approach, it becomes feasible to refine control over binding affinity and specificity, leading to applications such as tailored therapeutics with reduced side effects, highly sensitive diagnostic tools, efficient biocatalysis, targeted drug delivery systems, and sustainable bioremediation solutions 9 – 11 , thus illustrating the transformative impact of redesigning ligand-binding proteins across various fields. Traditional methods for designing ligand-binding proteins have relied heavily on experimental techniques, characterized by systematic but often inefficient trial-and-error processes 12 – 14 . These methods, while foundational, are time-consuming, resource-intensive, and sometimes fall short in precision and efficiency. The emergence of computational design has marked a transformative shift, offering new pathways to accelerate the design process and gain deeper insights into the molecular basis of protein-ligand interactions. However, even with the advancements in computational approaches, significant challenges remain. Many existing models demand extensive structural information, such as protein crystal structures and specific binding pocket data, limiting their applicability, especially in urgent scenarios like the emergence of novel diseases 15 – 17 . For instance, during the outbreak of a new disease like COVID-19, the spike proteins of the virus may not have well-characterized binding sites, delaying the development of effective drugs 18 , 19 . Furthermore, the complexity of binding mechanisms, including allosteric effects and cryptic pockets, adds another layer of difficulty 20 , 21 . Specifically, many proteins do not exhibit clear binding pockets until ligands are in close vicinity, necessitating extensive simulations to reveal potential binding interfaces 21 , 22 . While molecular dynamics simulations offer detailed atomistic insights into binding mechanisms, they often prove inadequate for designing high-throughput sequences due to high computational cost 9 , 23 . This complexity underscores the need for a drug design methodology that is agnostic to predefined binding pockets. Our study addresses those identified challenges by introducing ProteinReDiff, a computational framework developed to enhance the process of redesigning ligand-binding proteins. Originating from the foundational concepts of the Equivariant Diffusion-Based Generative Model for Protein-Ligand Complexes (DPL) 24 , ProteinReDiff incorporates key improvements inspired by the representation learning modules from the AlphaFold2 (AF2) architecture 25 . Specifically, we integrate the Outer Product Update (adapted from outer product mean of AF2), Single Representation Attention (adapted from MSA row attention module), and Triangle Multiplicative Update modules into our Residual Feature Update procedure. These modules collectively enhance the framework’s ability to capture intricate protein-ligand interactions, improve the fidelity of binding affinity predictions, and enable more precise redesigns of ligand-binding proteins. The framework integrates the generation of diverse protein sequences with blind docking capabilities. Starting with a selected protein-ligand pair, our approach stochastically masks amino acids and equivariantly denoises the diffusion model to capture the joint distribution of ligand and protein complex conformations. Another key feature of our method is blind docking, which predicts how the redesigned protein interacts with its ligand without the need for predefined binding site information, while relying solely on initial protein sequences and ligand SMILES strings 26 . This streamlined approach significantly reduces reliance on detailed structural data, thus expanding the scope for sequence-based exploration of proteinligand interactions. In summary, the contributions of our paper are outlined as follows: We introduce ProteinReDiff, an efficient computational framework for ligand-binding protein redesign, rooted in equivariant diffusion-based generative models. Our innovation lies in integrating AF2’s representational learning modules to enhance the framework’s ability to capture intricate protein-ligand interactions. Our framework enables the design of high-affinity ligand-binding proteins without reliance on detailed structural information, relying solely on initial protein sequences and ligand SMILES strings. We comprehensively evaluate our model’s outcomes across multiple design aspects, including sequence diversity, structure preservation, and ligand binding affinity, ensuring a holistic assessment of its effectiveness and applicability in various contexts. Related Work Traditional Approaches in Protein Design Protein design has historically hinged on computational and experimental strategies that paved the way for modern advancements in the field. These foundational methodologies emphasized the intricate balance between understanding protein structure and engineering novel functionalities, albeit with inherent limitations in scalability and precision. Key traditional approaches include: Rational Design 27 – 29 focused on introducing specific mutations into proteins based on known structural and functional insights. This method required an in-depth understanding of the target protein structures and how changes might impact its function. Directed Evolution 30 – 33 mimicked natural selection in the laboratory, evolving proteins towards desired traits through iterative rounds of mutation and selection. Despite its effectiveness in discovering functional proteins, the process was often labor-intensive and time-consuming. These traditional methods have been instrumental in advancing our understanding and capability in protein design. However, their limitations in terms of efficiency, specificity, and the broad applicability of findings highlighted the need for more versatile and scalable approaches. As the field progressed, the integration of computational power and biological understanding opened new avenues for innovation in protein design, leading to the exploration and adoption of more advanced methodologies. Deep Generative Models in Protein Design Since their inception, deep generative models have significantly advanced fields like computer vision (CV) 34 and natural language processing (NLP) 35 , sparking interest in their application to protein design. This enthusiasm has led to numerous studies that harness these models for innovating within the protein design area. Among these, certain types of deep generative models have distinguished themselves through their effectiveness and the promising results they have achieved, including: Variational Autoencoders (VAEs) are harnessed for their ability to learn rich representations of protein sequences, enabling the generation of novel sequences through manipulation in the latent space 36 – 38 . Autoregressive models predict the probability of each amino acid in a sequential manner, facilitating the generation of coherent and functionally plausible protein sequences 39 , 40 . Generative Adversarial Networks (GANs) employ two networks that work in tandem to produce protein sequences indistinguishable from real ones, enhancing the realism and diversity of generated designs 41 , 42 . Diffusion models represent a step forward by gradually transforming noise into structured data, simulating the complex process of folding sequences into functional proteins 43 – 46 . However, the majority of these studies have focused on protein-centric designs, with a noticeable gap in research that integrates both proteins and ligands for the purpose of redesigning ligand-binding proteins. Such integration is crucial for a holistic understanding of the intricate dynamics between protein structures and their ligands, a domain that remains underexplored. Current Approaches in Ligand-Binding Protein Redesign Heavy Reliance on Detailed Structural Information Contemporary computational methodologies for designing proteins that target specific surfaces predominantly rely on structural insights from native complexes, underscoring the critical role of fine-tuning side-chain interactions and optimizing backbone configurations for optimal binding affinity 15–17,44,47,48 . These strategies often initiate with the generation of protein backbones, employing inverse folding techniques to identify sequences capable of folding into these pre-designed structures 6 , 7 , 48 , 49 . This approach signifies a paradigm shift by prioritizing structural prediction ahead of sequence identification, aiming to produce proteins that not only fit the desired conformations for potential ligand interactions but also navigate around the challenge of undefined binding sites. Despite the advantages, including the potential of computational docking to create binders via manipulation of antibody scaffolds and varied loop geometries 36 , 50 , 51 , a notable challenge persists in validating these binding modes with high-resolution structural evidence. Additionally, the traditional focus on a limited array of hotspot residues for guiding protein scaffold placement often restricts the exploration of possible interaction modes, particularly in cases where target proteins lack clear pockets or clefts for ligand accommodation 22 , 52 . Limited Training Data and Lack of Diversity Existing approaches often rely on a limited set of training data, which can restrict the diversity and generalizability of the resulting models. For instance, datasets like PDBBind provide detailed ligand information, but their scope is limited 53 . This limitation is further compounded when protein datasets lack corresponding ligand data, reducing the effectiveness of the training process. Traditional methodologies also tend to focus on a narrow range of protein-ligand interactions, potentially overlooking the broader spectrum of possible interactions. Single-Domain Denoising Focus Previous methodologies typically concentrate on denoising either in sequence space or structural space, but not both. Approaches like ProteinMPNN 6 , LigandMPNN 17 , and MIF 48 primarily operate in sequence space, while others like DPL function in structural space 24 . This single-domain focus can limit the ability to capture the full complexity of protein-ligand interactions, which inherently involve both sequence and structural dimensions. Consequently, these methodologies may fall short of accurately predicting the functional capabilities of redesigned proteins. Challenges in Generating Diverse Sequences with Structural Integrity While some approaches prioritize sequence similarity to generate functional proteins, they often do so at the expense of structural integrity. For example, ProteinMPNN and CARP focus heavily on sequence similarity, which can result in a lack of diversity and flexibility in the generated sequences 6 , 7 . This limitation can hinder the ability to explore a wider range of functional conformations, reducing the effectiveness of the protein design process. Distinct Improvements of Our Approach We address the weaknesses of available methodologies by integrating diverse datasets, employing a dual-domain denoising strategy, and ensuring the generation of diverse sequences while maintaining structural integrity. Our approach utilizes only protein sequences and ligand SMILES strings, eliminating the need for detailed structural information. By combining PDBBind 53 and CATH 54 datasets, we effectively double our training data, enhancing protein representations. Our equivariant and KL-divergence loss functions enable denoising across both sequence and structural dimensions, capturing the full complexity of protein-ligand interactions. This approach maintains structural fidelity and promotes sequence diversity, overcoming the limitations of methodologies prioritizing sequence similarity at the expense of diversity. Background Protein Language Models (PLMs) Protein Language Models (PLMs) harness the power of natural language processing (NLP) to unravel the intricate latency embedded within protein sequences. By analogizing amino acid sequences to human language sentences, PLMs unlock profound insights into protein functions, interactions, and evolutionary trajectories 55 . These models leverage advanced text processing techniques to predict structural, functional, and interactional properties of proteins based solely on their amino acid sequences 56 – 59 . Their adoption in protein design has catalyzed significant progress, with studies leveraging PLMs to translate protein sequence data 47 , 60 – 62 into actionable insights, thus guiding the precise engineering of proteins with targeted functional attributes. Mathematically, a PLM can be represented as a function F that maps a sequence of amino acids S = [ s 1 , s 2 , …, s n ], where s i denotes the i -th amino acid in the sequence, to a high-dimensional feature space that encapsulates the protein’s structural and functional properties: where X represents the continuous representation or embedding derived from the sequence S and d represents the dimensionality of the embedding space, determined by the PLM’s architecture. This embedding captures the complex dependencies and patterns underlying the protein’s structural information and biological functionality. Through training on known sequences and structures, PLMs discern the “grammar” governing protein folding and function, facilitating accurate predictions. We employ the ESM-2 model 59 , a state-of-the-art protein language model with 650 million parameters, pre-trained on nearly 65 million unique protein sequences from the UniRef 63 database, to feature initial masked protein sequences. ESM-2 enriches the latent representation of protein sequences, bypassing the need for conventional multiple sequence alignment (MSA) methods. By incorporating structural and evolutionary information from input sequences, ESM-2 enables us to unravel interaction patterns across protein families for effective ligand targeting. This understanding is crucial for designing and optimizing ligand-binding proteins. Equivariant Diffusion-based Generative Models We utilize a generative model driven by equivariant diffusion principles, drawing from the foundations laid by Variational Diffusion Models 64 and E(3) Equivariant Diffusion Models 65 . The Diffusion Procedure First, we employ a diffusion procedure that is equivariant with respect to the coordinates of atoms x , alongside a series of progressively more perturbed versions of x , known as latent variables z t , with t varying from 0 to 1. To maintain translational invariance within the distributions, we opt for distributions on a linear subspace that anchors the centroid of the molecular structure at the origin, and designate N x as a Gaussian distribution within this specific subspace. The conditional distribution of the latent variable z t given x , for any given t in the interval [0, 1], is defined as where α t and represent strictly positive scalar functions of t , dictating the extent of signal preservation versus noise introduction, respectively. We implement a variance-conserving mechanism where and posit that α t smoothly and monotonically decreases with t , ensuring α 0 ≈ 1 and α 1 ≈ 0. Given the Markov property of this diffusion process, it can be described via transition distributions as for any t > s , where α t | s = α t / α s and . The Gaussian posterior of these transitions, conditional on x , can be derived using Bayes’ theorem: With The Generative Denoising Process The construction of the generative model inversely mirrors the diffusion process, generating a reverse temporal sequence of latent variables z t from t = 1 back to t = 0. By dividing time into T equal intervals, the generative framework can be described as: with s ( i ) = ( i − 1)/ T and t ( i ) = i / T . Leveraging the variance-conserving nature and the premise that α 1 ≈ 0, we posit q ( z 1 ) = N x (0, I ), hence treating the initial distribution of z 1 as a standard Gaussian: Furthermore, under the variance-preserving framework and assuming α 0 ≈ 1, the distribution q ( z 0 | x ) is modeled as highly peaked 64 , 66 . This allows us to approximate p data ( x ) as nearly constant within this narrow peak region. This yields: Accordingly, we approximate q ( x | z 0 ) through: The generative model’s conditional distributions are then formulated as: which mirrors q ( z s | Fsz t , x ) but substitutes the actual coordinates x with the estimates from a temporal denoising model , which employs a neural network parameterized by θ to predict x from its noisier version z t . This denoising model’s framework, predicated on noise prediction , is articulated as: Consequently, the transition mean is determined by: Method In this section, we detail the methodology employed in our noise prediction model, which is depicted in Figure 1 and consists of three main procedures: (1) input featurization, (2) residual feature update, and (3) equivariant denoising. Through these steps, we transform raw protein and ligand data into structured representations, iteratively refine their features, and leverage denoising techniques inherent in the diffusion model to improve sampling quality. Download figure Open in new tab Figure 1: Overview of the proposed framework. The process begins with utilizing a protein amino acid sequence and a ligand SMILES string as inputs. The conformational sampling process includes iteratively applying input featurization, updating residual features, and denoising equivariantly, ultimately yielding novel protein sequences alongside their corresponding Cα protein backbone and ligand complexes. Input Featurization We develop both single and pair representations from protein sequences and ligand SMILES string ( Figure 2 ). For proteins, we initially applied stochastic masking to segments of the amino acid sequences. The protein representation is attained through the normalization and linear mapping of the output from the final layer of the ESM-2 model, which is subsequently combined with the amino acid and masked token embeddings. Additionally, for pair representations of proteins, we leveraged pairwise relative positional encoding techniques, drawing from established methodologies 25 . For ligand representations, we employed a comprehensive feature embedding approach, capturing atomic and bond properties such as atomic number, chirality, connectivity, formal charge, hydrogen attachment count, radical electron count, hybridization status, aromaticity, and ring presence for atoms; and bond type, stereochemistry, and conjugation status for bonds. These representations are subsequently merged, incorporating radial basis function embeddings of atomic distances and sinusoidal embeddings of diffusion times. Together, these steps culminate in the formation of preliminary complex representations, laying the foundation for our computational analyses. Download figure Open in new tab Figure 2: Overview of the input featurization procedure of the model. Download figure Open in new tab Figure 3: Overview of the residual feature update procedure of the model. Residual Feature Update Procedure Our approach deviates significantly from the residual feature update procedure employed in the original DPL model 24 . While the DPL model relied on Alphafold2’s Triangular Multiplicative Update for updating single and pair representations, where these representations mutually influence each other, our objective is to optimize this procedure for greater efficiency. Specifically, we incorporate enhancements such as the Outer Product Update and Single Representation Attention to formulate sequence representational hypotheses of pro-tein structures and to model suitable motifs for binding target ligands specifically. These modules, integral to Evoformer, the sequence-based module of AF2, play a crucial role in extracting essential connections among internal motifs that serve structural functions (i.e., ligand binding) when structural information is not explicitly provided during training. Importantly, we adapt and tailor these modules to fit within our model architecture, ensuring their effectiveness in capturing the intricate interplay between proteins and ligands. Single Representation Attention Module The Single Representation Attention (SRA) module, derived from the Alphafold2 model’s MSA row attention with pair bias, accounts for long-range interactions among residues and ligand atoms within a single protein-ligand embedding vector. In essence, the attention mechanism assigns importance to those involved in complex-based folding to denoise the equivariant loss (Section) in a self-supervised manner. While the original Alphafold2 MSA row attention mechanism processes input for a single sequence, the SRA module is designed to incorporate representations from multiple protein-ligand complexes concurrently. Specifically, the pair bias component of the SRA attention module captures dependencies between proteins and ligands, which was shown to fit the attention score better than the regular selfattention model without bias terms 67 . By considering both the single representation vector (which encodes the protein/ligand sequential representation) and the pairwise representation vector (which encodes protein-protein and protein-ligand interactions), this cross-attention mechanism exchanges information between pairwise and single representation to effectively preserves internal motifs, as evidenced by contact overlap metrics 55 , 68 . As transformer architecture is widely used for predicting protein functions 69 , we observed similar efficacy to our binding affinity prediction in section Results and Appendix ?? - ?? . For a detailed description of the computational steps implemented in this module, refer to Algorithm 1. Algorithm 1 Single Representation Attention pseudocode Download figure Open in new tab Outer Product Update Since the SRA encodings have a shape ( s, r, c m ) and the pair representation has a shape ( s, r, r, c z ), the outer product (OPU) layer merges insights by reshaping SRA encodings into pair representations. This module leverages evolutionary cues from ESM to generate plausible structural hypotheses for pair representations 70 . It first calculates the outer product of the SRA embeddings of protein-ligand pairs, then aggregates the outer products to yield a measure of co-evolution between every residue pair 55 . Analogous to Tensor Product Representations (TPR) in NLP, the outer product is akin to the filler-and-role binding relationship, where each entity (i.e. amino acid residue) on a sequence is attached to a rich functional embedding based on its relationship to one another 71 – 73 . This process integrates correlated information of residues i and j of a sequence s , resulting in the intermediate Kronecker product tensors (.i.e. role embeddings in NLP) 67 , 74 , 75 . Subsequently, an affine transformation projects those representations to hypotheses concerning the relative positions of residues i and j under biophysical constraints. Our implementation adapts the outer product without computing the mean to maintain the pair representations of multiple protein-ligand complexes. For a detailed description of the computational steps implemented in this module, refer to Algorithm 2. Algorithm 2 Outer product update pseudocode Download figure Open in new tab Triangle Multiplicative Updates After refining the pair representation, our model interprets the primary protein-ligand structure using principles from graph theory, treating each residue as a distinct entity interconnected through the pairwise matrix. These connections are then refined through triangular multiplicative updates to account for physical and geometric constraints, such as triangular inequality. While the SRA weights the importance of residues, the triangular multiplicative update acts as another stack of transformer-based layers where any two edges affect the third one to enforce triangle equivariance 55 , 76 . The starting and ending nodes propagate information in and out of neighbors in similar fashion as the message-passing framework 67 . These mechanisms enable the model to generate more accurate representations of protein-ligand complexes, leading to improved predictive performance in predicting binding affinities and structural characteristics. Equivariant Denoising During the equivariant denoising process, the final pair representation undergoes symmetrization and is then transformed using a multi-layer perceptron (MLP) into a weight matrix W . This matrix is utilized to compute the weighted sum of all relative differences in 3D space for each atom, as shown in the equation 24 : Afterward, the centroid is subtracted from this computation, resulting in the output of our noise prediction model . Additionally, it’s important to note that the described model maintains SE(3)-equivariance, meaning that: for any rotation R and translation t . This property is derived from the fact that the final representation, and hence the weight matrix W , depends solely on atom distances that are invariant to rotation and translation. Experiments Training Process Materials Our training strategy leverages a meticulously curated dataset encompassing a broad range of protein structures, including both ligand-bound (holo) and ligand-free (apo) forms, sourced from two key repositories: PDBBind v2020 53 and CATH 4.2 54 . PDBBind v2020 offers a diverse collection of protein-ligand complexes, while CATH 4.2 provides a substantial repository of protein structures. Each dataset was selected for its unique contributions to our understanding of protein-ligand interactions and structural diversity. This strategic selection of datasets ensures our model is exposed to a wide and varied spectrum of protein-ligand interactions and structural configurations, enabling comprehensive evaluation against diverse inverse folding benchmarks. By training on both holo and apo structures, our approach imbues the model with a robust understanding of protein-ligand dynamics, equipping it to navigate the complexities of unseen protein-ligand interaction scenarios effectively. To ensure robust model training and evaluation, we employ careful data partitioning techniques. Using MMseqs2 77 , we clustered and partitioned the protein sets for training, validation, and testing, maintaining sequence similarities between 40% and 50% to ensure unbiased training and predictions, following protocols from other protein models 25 , 48 . For ligands, we cluster based on the Tanimoto similarity of Morgan fingerprints 78 on ligand structures. Incorporating CATH 4.2 data into PDBBind not only preserves the objectivity of the train/test/validation partitions but also substantially decreases the similarities within ligand sets, as shown in Table 1 . View this table: View inline View popup Download powerpoint Table 1: Similarity between Train/Validation/Test Sets of Proteins and Ligands. The values represent similarity percentages for the original PDBBind dataset versus combined PDBBind with CATH datasets in parentheses. Table 2 provides an overview of the partitioning details, facilitating a clear understanding of the distribution of samples across different subsets of the dataset. View this table: View inline View popup Download powerpoint Table 2: Data Partitioning Overview (Unit: number of samples) PDBBind v2020 : For consistency and comparability with previous studies, we first adhered to the test/training/validation split settings outlined in established literature 79 , specifically following the configurations defined in the respective sources for the PDBBind v2020 datasets 80 . Then, we filtered out those highly similar sequences (above 95%) to keep the average similarities between 40%-50%. CATH 4.2 : In our approach, we deliberately focused on proteins with fewer than 400 amino acids and less similar (below 90%) sequences from the CATH 4.2 database. This selective criterion was chosen to prioritize smaller proteins, which often represent more druggable targets of interest in drug discovery and development endeavors. During both the training and validation phases, SMILES strings of CATH 4.2 proteins were represented as asterisks (masked tokens) to denote unspecified ligands. Notably, CATH 4.2 was excluded from the test set due to the absence of corresponding ligands required for evaluating protein-ligand interactions. Loss Functions Previous models typically denoise in only one domain, such as ProteinMPNN 6 , LigandMPNN 17 , and MIF 48 in sequence space, and DPL 24 in structural space. This limitation restricts their ability to fully capture the intricate interactions between proteins and ligands. To address this, we have introduced significant modifications to the loss function to better suit the task of ligand-binding protein redesign. By tailoring the loss function to integrate both sequence and structural spaces, our approach effectively addresses the unique challenges of proteinligand interactions. Specifically, the optimization of our model for ligand-binding protein redesign is governed by a composite loss function L , formulated as follows: Weighted Sum of Relative Differences ( L WS ) This component ensures the model’s sensitivity to the directional influence between atoms, supporting the accurate prediction of the denoised structure while maintaining physical symmetries. It is crucial for the equivariant denoising step, enabling accurate noise prediction for atoms in the protein-ligand complex. The loss is defined as: where T is the total number of time steps in the diffusion process, ϵ is the Gaussian noise vector 𝒩( 0, I ), and is the loss prediction at time step t parameterized by a weight MLP in Section. Kullback-Leibler Divergence ( L KL ) 81 This component quantifies the divergence between the model’s predictions and actual sequence data at timestep t − 1, playing a pivotal role in the denoising process. Defined as KL ( x pred t-1 , seq t-1 ), it contrasts the predicted distribution, x pred t-1 , against the true sequence distribution, seq t-1 , leveraging the diffusion process’s γ parameter for temporal adjustment. This loss is also applied in the Protein Generator 5 model to ensure the model’s predictions progressively align with actual data distributions, enhancing the accuracy of sequence and structure generation by minimizing the expected divergence. Cross-entropy Loss ( L CE ) This loss function is crucial for the accurate prediction of protein sequences, aligning them with the ground truth through effective classification. It denoises each amino acid from masked latent embedding to a specific class, leveraging categorical cross-entropy to rigorously penalize discrepancies between the model’s predicted probability distributions and the actual distributions for each amino acid type. Training Performance Throughout the training phase, we meticulously observed the model’s performance, paying close attention to the dynamics between training and validation losses, as demonstrated in Figure 4 . While the training loss consistently diminished, indicating effective learning, the validation loss exhibited more variability. Despite these fluctuations, the validation loss showed an overall downward trend, suggesting that the model is improving its generalization capabilities over time. The general alignment between the downward trends of training and validation losses indicates that the model is learning effectively without significant overfitting. Download figure Open in new tab Figure 4: Training history chart of ProteinReDiff, showcasing the evolution of training and validation losses over epochs. Evaluation Process Ligand Binding Affinity (LBA) Ligand binding affinity is a fundamental measure that quantifies the strength of the interaction between a protein and a ligand. This metric is crucial as it directly influences the effectiveness and specificity of potential therapeutic agents; higher affinity often translates to increased drug efficacy and lower chances of side effects 82 . Within this context, ProteinReDiff is evaluated on its ability to generate protein sequences for significantly improved binding affinity with specific ligands. We utilize a docking score-based approach for this assessment, where the docking score serves as a quantitative indicator of affinity. Expressed in kcal/mol, these scores inversely relate to binding strength — lower scores denote stronger, more desirable binding interactions. Sequence Diversity Sequence diversity is crucial for exploring protein’s functional space 83 . It reflects the capacity of our model, ProteinReDiff, to traverse the vast landscape of protein sequences and generate a wide array of variations. To quantitatively assess this diversity, we utilize the average edit distance (Levenshtein distance) 84 between all pairs of sequences generated by the model. This metric offers a nuanced measure of variability, surpassing traditional metrics that may overlook subtle yet significant differences. The diversity score is calculated using the formula: where d ( S i , S j ) represents the edit distance between any two sequences S i and S j . This calculation provides an empirical gauge of ProteinReDiff’s ability to enrich the protein sequence space with novel and diverse sequences, underlining the practical variance introduced by our model. Structure Preservation Structural preservation is paramount in the redesign of proteins, ensuring that essential functional and structural characteristics are maintained post-modification. To effectively measure structural preservation between the original and redesigned proteins, three key metrics: the Template Modeling Score (TM Score) 85 , the Root Mean Square Deviation (RMSD) 86 , and the Contact Overlap (CO) 87 . These two metrics collectively provide a comprehensive assessment of structural integrity and similarity, essential for evaluating the success of our protein redesign efforts. The Root Mean Square Deviation (RMSD) is a measure used to quantify the distance between two sets of points. In the context of protein structures, these points are the positions of the atoms in the protein. The RMSD is given by the formula: Where and denote two sequences of N 3D coordinates representing the atomic positions in the original and redesigned proteins, respectively. This formula calculates the minimum root mean square of distances between corresponding atoms, after optimal superposition, which involves finding the best-fit rotation R and translation t that aligns the two sets of points. A lower RMSD value indicates a higher degree of structural similarity, making it a direct measure of the extent to which structural deviation has been minimized. Achieving a low RMSD is desirable, as it signifies that the redesign process has successfully preserved the core structural configuration of the original protein. TM Score provides a normalized measure of structural similarity between protein configurations, which is less sensitive to local variations and more reflective of the overall topology. The TM Score is defined as follows: where d 0 is a scale parameter typically chosen based on the size of the proteins. The closer the TM Score is to 1, the more similar the structures are, indicating global structural alignment. Contact Overlap (CO) provides a complementary perspective to RMSD and TM Score by focusing on the preservation of local structural motifs rather than overall geometric similarity. Several studies show that having high CO indicates protein’s residue pairs having co-evolutionary signals 87 , 88 and performing related functions 89 . CO quantitatively measures the conservation of inter-atomic contacts between the original and redesigned protein structures, which are crucial for the protein’s structural integrity and functional capabilities. The metric is defined as: where C = { ( i, j ) : ∥ p i − p j ∥ < r c , i ≠ j} and represent the sets of contacts in the original and redesigned proteins, respectively. Here, p i and are the positions of atoms in the original and redesigned proteins, and r c is a predefined cutoff distance that determines when two atoms are considered to be in contact. A high CO score indicates that many of the original contacts are preserved in the redesigned structure, suggesting that the redesign maintains much of the original protein’s structural network, which is crucial for its stability and function. Experimental Setup To evaluate ProteinReDiff, we employed Omegafold 90 to predict the three-dimensional structures of all designed protein sequences. The choice of Omegafold over AF2 was favorable because Omegafold can more accurately fold proteins with low similarity to existing proteomes, making it suitable for proteins lacking available ligand-binding conformations. Next, we utilized AutoDock Vina 91 to conduct docking simulations and evaluate the binding affinity between the redesigned proteins and their respective ligands based on the predicted 3D structures. To ensure fair comparisons and mitigate potential biases introduced by predocked structures, we aligned our redesigned protein structures with reference structures before docking. This approach is crucial, particularly because the use of pre-docked structures may favor certain conformations, leading to inaccurate evaluations. Additionally, to provide context for our results, we compared the binding scores of our redesigned proteins not only with those of the original proteins but also with proteins generated by other protein design models. Although these models may exhibit different sequence characteristics compared to those explicitly designed for ligand binding affinity, comparing their scores offers valuable insights. Such comparisons help elucidate the interplay between protein sequence and structure in determining ligand interactions, enriching the interpretation of our findings and advancing our understanding of protein-ligand interactions. Benchmark Model Selection In selecting benchmark models for performance comparison, we focused on state-of-the-art approaches, particularly those relevant to protein design tasks. Traditionally, protein design has been primarily based on inverse folding, utilizing protein structure information. Our choices encompass a range of methodologies: MIF 48 , MIF-ST 48 , and ProteinMPNN 6 are notable for generating sequences with high identity and experimental significance, utilizing protein structure information. The Protein Generator 5 , a representative of RosettaFold models 44 , employs diffusionbased methods, making it an intriguing comparative candidate. The model also shares a similar loss function, L KL , in sequence space with our model but diverges in modules and training procedures (i.e., stochastic masking). ESMIF 49 , belonging to the ESM model family 92 , stands as another competitive benchmark, emphasizing the generation of high-quality sequences. CARP, while lacking ligand information, shares similar protein input and output characteristics with our models, warranting inclusion for comparison. DPL 24 , originally geared towards protein-ligand complex generation, was adapted for our purposes by modifying loss functions and incorporating a sequence prediction module, given its alignment with our model architecture. LigandMPNN 17 , resembling the most to our task in designing ligand-binding proteins, necessitates binding pocket information, unlike our model, which emphasizes a simplified yet effective approach for ligand-binding protein tasks. Our model’s design prioritizes simplicity in input while achieving effectiveness in output for ligand-binding protein tasks. For a comprehensive comparison of input-output dynamics across each model, please consult Table 3 . View this table: View inline View popup Download powerpoint Table 3: Comparison of protein design models based on input and output characteristics Results and Discussion We conducted comprehensive evaluation of ProteinReDiff, as detailed in Table 4 and visually represented in Figure 6 , across the metrics of ligand binding affinity, sequence diversity, and structure preservation. These evaluations provide a clear depiction of the model’s performance relative to established baselines and within its variations. View this table: View inline View popup Download powerpoint Table 4: Comparison of method performance across multiple metrics: Ligand binding affinity (LBA), sequence diversity, and structure preservation. Ligand binding affinity (LBA), TM Score, and RMSD are reported as mean values with their respective margins of error. For ProteinReDiff, we aimed to capture the diverse conformations of ligand-binding proteins, recognizing that they can adopt multiple structural states. To assess these conformations, we employed alignment metrics such as TM score, RMSD, and contact overlap (CO). In Figure 5 , we presented several instances where the contact overlap appeared to be maintained, yet the RMSD is large and TM score is low. This discrepancy suggests that while global alignment metrics like TM score and RMSD may not adequately capture the domain shift within these complex ensembles, the preservation of local motifs, as indicated by contact overlap, remains crucial in our framework. This underscores the importance of capturing both global and local structural features for a comprehensive understanding of protein-ligand interactions. Download figure Open in new tab Figure 5: Comparative visualizations of protein structures, each annotated with its corresponding PDB ID. The figure includes a succinct table detailing Contact Overlap (CO) and Root Mean Square Deviation (RMSD) metrics. Original protein structures are highlighted in green, and the redesigned versions by ProteinReDiff are depicted in pink, illustrating the precise structural changes and enhancements achieved through the redesign. Download figure Open in new tab Figure 6: Visualization of method performance across metrics. The metrics are plotted with mean values and margins of error. For LBA, the red bar (top right) shows the docking score of reference complexes. The horizontal dash lines indicate the regions of 15% masking model which is our standard for comparison. A pivotal observation from our study is ProteinReDiff’s unparalleled ability to enhance ligand binding affinity, particularly at a 15% masking ratio in Figure 6 . This configuration not only surpasses the performance of Inverse Folding (IF) models and the original DPL framework but also exceeds the binding efficiencies of the original protein designs. By incorporating attention modules from AlphaFold2, ProteinReDiff effectively captures the complex interplay between proteins and ligands, demonstrating its superiority over the original DPL model. While other masking ratios within ProteinReDiff show varying degrees of effectiveness, lower ratios, though at the same par as reference, do not achieve the peak LBA performance observed at 15%. For instance, the 5% masked model emphasizes structural consistency with a high TM-Score and low RMSD, but does not exhibit the same level of binding capability as the 15% masking. These findings are also consistent with ablation studies shown in Appendix ?? . Conversely, higher masking ratios fail to strike the necessary balance between introducing beneficial modifications and maintaining functional precision, underscoring the importance of optimizing the masking ratio. Our analysis of sequence diversity and structure preservation metrics reveals a delicate balance essential in protein redesign. The 15% masking ratio, identified as optimal for enhancing ligand binding affinity in our model, also aligns closely with benchmark methods in both sequence diversity and structure preservation. For instance, LigandMPNN excels in sequence diversity but faces challenges in obtaining binding pocket inputs for various design tasks, unlike our approach. Moreover, our models (at 30% and 40% maskings) significantly outperform others in contact overlap, crucial for diversifying structures while preserving functional motifs in protein redesign tasks. This equilibrium underscores ProteinReDiff’s ability to optimize ligand interactions without compromising the exploration of sequence diversity or the integrity of original protein structures. In contrast, extreme values in either sequence diversity or structure preservation, which could be seen in other masking ratios, do not lead to optimal ligand binding affinities. This finding highlights an inverse relationship between pushing the limits of diversity and preservation and achieving the primary goal of binding enhancement. Thus, the 15% masking ratio not only stands out for its ability to significantly improve ligand binding affinity but also for maintaining a balanced approach, ensuring that enhancements in functionality do not detract from the protein’s structural and functional viability. In Figure 7 , we compare the ligand-binding affinity (LBA) of original and redesigned proteins by ProteinReDiff. The redesigned proteins maintain their original folds while significantly enhancing LBA. In ablation studies (Section), we can apply various masking strategies to adjust both sequence diversity and structural integrity. This approach has potential applications in different settings to control the affinity of ligand binders. Download figure Open in new tab Figure 7: Comparative visualizations of protein-ligand complexes, each labeled with corresponding PDB IDs and accompanied by a small table showing Ligand Binding Affinity (LBA) before and after the redesign. Original structures are highlighted in green, while redesigned versions by ProteinReDiff appear in pink. Ligands are depicted in various colors to emphasize specific binding sites and molecular interaction enhancements post-redesign. Ablation Studies Here we conducted thorough ablation studies on ProteinReDiff’s model architecture, featurization, and masking ratios. For complete ablation setup, please refer to Table ?? (Appendix ?? ) Interpreting Model Architecture We trained ablated versions of ProteinReDiff without the SRA or OPU modules and compared them to the original DPL model. Initially designed for generating ensembles of complex structures, DPL was adapted for targeted protein redesign by adding sequence-based loss functions to generate new target sequences. In Figure 8 , we computed the performance score by averaging the sum of five evaluation metrics introduced in Sections,, and. Since the sequence diversity is not within the [0,1] range, we applied Min-Max normalization. For LBA and RMSD, we used inverse normalization to ensure that a score closer to 1.0 indicates better model performance. The average score is then compared with the baseline score of ProteinReDiff which was trained without any ablations. Download figure Open in new tab Figure 8: Ablation studies on ProteinReDiff’s model architecture and featurization. The dash line indicates the baseline’s average score obtained from ProteinReDiff without ablations. We observed that our model outperformed DPL by a large margin. Incorporating just the OPU module (without the SRA module) yields better performance than DPL, indicating OPU’s ability to exchange insights between single and pair representations. Firstly, the equivariant loss function is parameterized on the structural space, making the pairwise representations from the OPU critical to that loss. Secondly, without OPU, the model performs poorly on TMScore (the bottom brown line in Figure ?? , Appendix ?? ), which measures global structural preservation. Additionally, introducing SRA only without OPU hurts our model performance, suggesting the model would have been over-parameterized as the SRA updates primarily on the sequence representation. Therefore, combining both the OPU and SRA modules provides an effective approach for enhancing the representational learning of ProteinReDiff. A complete comparative assessment is presented in Table 4 and Appendix ?? . Ablations on Input Featurization Methods We conducted ablation studies to evaluate different input featurization methods, including manual feature engineering for ligands and the use of ESM-2 as a pre-trained LLM for protein featurization. We gradually reduced ligand features, starting with ligand distance and bond information (e.g., types, ring), and even omitted the entire bond and ligand. In Figure 8 , omitting bond features and distance caused less reduction in model performance than omitting the entire ligand. Ligand bond information is crucial for the model to learn the relative positions of ligand atoms and adhere to geometric constraints within the triangular update module (Section). We observed a significant decrease in model performance when ESM embeddings were excluded (the red bar in Figure 8 ). The ESM features alone (the brown bar) significantly boosted performance when training without ligand data, as these embeddings are enriched with protein evolutionary and biophysical information needed for both single and pair representations. Other protein features, such as position encodings and amino acid types, provided slight improvements, though they were minimal. However, excluding ligand information led to a reduction in model performance compared to the baseline, as the model relies on learning the overall structure of the complexes. Therefore, using pre-trained featurization methods, such as ESM and other protein BERT-like models, in combination with ligand input, significantly enhances model training and performance. Impact of Masking Ratios We examined ProteinReDiff’s performance with various percentages of masked amino acids, adjusting the masking ratio as a hyperparameter and retraining our model. In Figure 9 , we observed consistent top performance across the metrics with masking ratios between 5% and 15%. This range is crucial for the protein redesign strategy, enhancing binding affinity while preserving the structural and functional motifs of the target protein. The 15% masking ratio achieved the best ligand binding affinity, the most important metric for capturing protein function. Download figure Open in new tab Figure 9: Mask ablation studies on both validation and test sets. Each of the mask ratios (5%, 10%, 15%, 30%, 40%, 50%, 60%, 70%) is a hyperparameter and represented by a model. The performances of the masked models are evaluated for all metrics. The arrows on y-axes show directions of better performance. Interestingly, we noticed performance spikes for 50% masking in contact overlap and TMscore. This is because applying stochastic masks allows the model to learn representations with varied masking from 0 up to the set ratio. Although the 50% masking does not surpass the 15% masking’s performance, the improvement in the high masking regime demonstrates the robustness of our training scheme. Overall, this investigation highlights the optimal level of sequence masking needed to enhance ligand binding affinity, sequence diversity, and structural preservation. It also rein-forces training strategies for protein redesign as shown on the Discussion section (). Conclusions This study introduces ProteinReDiff, a computational framework developed to redesign ligand-binding proteins. By utilizing advanced techniques inspired by Equivariant DiffusionBased Generative Models and the attention mechanism from AlphaFold2, ProteinReDiff demonstrates its ability to enhance complex protein-ligand interactions. Our model excels in optimizing ligand binding affinity based solely on initial protein sequences and ligand SMILES strings, bypassing the need for detailed structural data. Experimental validations highlight ProteinReDiff’s capability to improve ligand binding affinity while preserving essential sequence diversity and structural integrity. These findings open new possibilities for protein-ligand complex modeling, indicating significant potential for ProteinReDiff in various biotechnological and pharmaceutical applications. Appendix Appendix A: Evaluating Protein-Ligand Complex Representation Evaluation Methodology In the continuation of our study’s exploration of protein-ligand complex representations, we extended the use of the PDBBind v2020 dataset, previously detailed in our training process, to evaluate the effectiveness of embeddings generated by ProteinReDiff. Employing these embeddings as input features, we trained a Gaussian Process (GP) model aimed at predicting ligand binding affinity. The choice of a GP model recognized for its probabilistic nature and adaptability to the nuanced, uncertain dynamics of biological interactions, was pivotal in assessing how well our embeddings encapsulate predictive information about protein-ligand interactions. Results and discussion The evaluation of our ProteinReDiff model on the PDBBind v2020 dataset demonstrates competitive results in predicting ligand binding affinity using protein-ligand complex representations, as evidenced in Table 5 . It’s important to note that this experiment aimed to verify the effectiveness of our protein-ligand complex representation rather than to fine-tune the model for this specific task. Consequently, while our results are promising and competitive with specialized studies focused solely on ligand binding affinity prediction, the primary goal was to validate the representation’s capability within the protein redesign framework of ProteinReDiff. This underscores the model’s utility in guiding the redesign process effectively, affirming the robustness and applicability of our protein-ligand complex representation strategy. View this table: View inline View popup Download powerpoint Table 5: Experimental results of ligand binding affinity prediction task on PDBBind v2020 dataset. Footnotes We include additional results into the manuscript. References (1). ↵ Du , X. ; Li , Y. ; Xia , Y.-L. ; Ai , S.-M. ; Liang , J. ; Sang , P. ; Ji , X.-L. ; Liu , S.-Q. Insights into protein-ligand interactions: Mechanisms, models, and methods . Int. J. Mol. Sci . 2016 , 17 , 144 . OpenUrl CrossRef PubMed (2). Wanat , K. Biological barriers, and the influence of protein binding on the passage of drugs across them . Molecular Biology Reports 2020 , 47 , 3221 – 3231 . OpenUrl (3). ↵ Skolnick , J. ; Zhou , H. Implications of the essential role of small molecule ligand binding pockets in protein–protein interactions . The Journal of Physical Chemistry B 2022 , 126 , 6853 – 6867 . OpenUrl (4). ↵ Listov , D. ; Goverde , C. A. ; Correia , B. E. ; Fleishman , S. J. Opportunities and challenges in design and optimization of protein function . Nature Reviews Molecular Cell Biology 2024 , (5). ↵ Lisanza , S. L. ; Gershon , J. M. ; Tipps , S. ; Arnoldt , L. ; Hendel , S. ; Sims , J. N. ; Li , X. ; Baker , D. Joint Generation of Protein Sequence and Structure with RoseTTAFold Sequence Space Diffusion . bioRxiv 2023 , (6). ↵ Dauparas , J. et al. Robust deep learning–based protein sequence design using Protein-MPNN . Science 2022 , 378 , 49 – 56 . OpenUrl CrossRef PubMed (7). ↵ Yang , K. K. ; Fusi , N. ; Lu , A. X. Convolutions are competitive with transformers for protein sequence pretraining . bioRxiv 2023 , (8). ↵ Iqbal , S. ; Ge , F. ; Li , F. ; Akutsu , T. ; Zheng , Y. ; Gasser , R. B. ; Yu , D.-J. ; Webb , G. I. ; Song , J. Prost: Alphafold2-aware sequence-based predictor to estimate protein stability changes upon missense mutations . Journal of Chemical Information and Modeling 2022 , 62 , 4270 – 4282 . OpenUrl CrossRef (9). ↵ Yang , W. ; Lai , L. Computational design of ligand-binding proteins . Current Opinion in Structural Biology 2017 , 45 , 67 – 73 , Engineering and design: New trends in designer proteins . OpenUrl (10). Ebrahimi , S. B. ; Samanta , D. Engineering protein-based therapeutics through structural and Chemical Design . Nature Communications 2023 , 14 . (11). ↵ Ruscito , A. ; DeRosa , M. C. Small-molecule binding aptamers: Selection strategies, characterization, and applications . Frontiers in Chemistry 2016 , 4 . (12). ↵ Creutznacher , R. ; Maass , T. ; Veselkova , B. ; Ssebyatika , G. ; Krey , T. ; Empting , M. ; Tautz , N. ; Frank , M. ; Kölbel , K. ; Uetrecht , C. ; Peters , T. NMR Experiments Provide Insights into Ligand-Binding to the SARS-CoV-2 Spike Protein Receptor-Binding Domain . Journal of the American Chemical Society 2022 , 144 , 13060 – 13065 . OpenUrl (13). Munk , C. ; Harpsøe , K. ; Hauser , A. S. ; Isberg , V. ; Gloriam , D. E. Integrating structural and mutagenesis data to elucidate GPCR ligand binding . Current Opinion in Pharmacology 2016 , 30 , 51 – 58 . OpenUrl CrossRef (14). ↵ Tavares , D. ; van der Meer , J. R. Ribose-binding protein mutants with improved interaction towards the non-natural ligand 1,3-cyclohexanediol . Frontiers in Bioengineering and Biotechnology 2021 , 9 . (15). ↵ Polizzi , N. F. ; DeGrado , W. F. A defined structural unit enables de novo design of small-molecule–binding proteins . Science 2020 , 369 , 1227 – 1233 . OpenUrl Abstract / FREE Full Text (16). Stärk , H. ; Jing , B. ; Barzilay , R. ; Jaakkola , T. Harmonic Self-Conditioned Flow Matching for Multi-Ligand Docking and Binding Site Design . 2023 . (17). ↵ Dauparas , J. ; Lee , G. R. ; Pecoraro , R. ; An , L. ; Anishchenko , I. ; Glasscock , C. ; Baker , D. Atomic context-conditioned protein sequence design using LigandMPNN . bioRxiv 2023 , (18). ↵ Lv , M. et al. Coronavirus disease (COVID-19): a scoping review . Euro Surveill . 2020 , 25 . (19). ↵ Schaub , J. M. ; Chou , C.-W. ; Kuo , H.-C. ; Javanmardi , K. ; Hsieh , C.-L. ; Goldsmith , J. ; DiVenere , A. M. ; Le , K. C. ; Wrapp , D. ; Byrne , P. O. ; et al. Expression and characterization of SARS-COV-2 spike proteins . Nature Protocols 2021 , 16 , 5339 – 5356 . OpenUrl (20). ↵ Agajanian , S. ; Alshahrani , M. ; Bai , F. ; Tao , P. ; Verkhivker , G. M. Exploring and learning the universe of protein allostery using artificial intelligence augmented biophysical and computational approaches . J. Chem. Inf. Model . 2023 , 63 , 1413 – 1428 . OpenUrl (21). ↵ Oleinikovas , V. ; Saladino , G. ; Cossins , B. P. ; Gervasio , F. L. Understanding cryptic pocket formation in protein targets by enhanced sampling simulations . J. Am. Chem. Soc . 2016 , 138 , 14257 – 14263 . OpenUrl CrossRef PubMed (22). ↵ Meller , A. ; Ward , M. ; Borowsky , J. ; Kshirsagar , M. ; Lotthammer , J. M. ; Oviedo , F. ; Ferres , J. L. ; Bowman , G. R. Predicting locations of cryptic pockets from single protein structures using the PocketMiner graph neural network . Nature Communications 2023 , 14 , 1177 . OpenUrl (23). ↵ Barros , E. P. ; Schiffer , J. M. ; Vorobieva , A. ; Dou , J. ; Baker , D. ; Amaro , R. E. Improving the efficiency of ligand-binding protein design with molecular dynamics simulations . Journal of Chemical Theory and Computation 2019 , 15 , 5703 – 5715 . OpenUrl (24). ↵ Nakata , S. ; Mori , Y. ; Tanaka , S. End-to-end protein–ligand complex structure generation with diffusion-based generative models . BMC Bioinformatics 2023 , 24 , 233 . OpenUrl (25). ↵ Jumper , J. et al. Highly accurate protein structure prediction with AlphaFold . Nature 2021 , 596 , 583 – 589 . OpenUrl CrossRef PubMed (26). ↵ Weininger , D. SMILES, a chemical language and information system. 1. Introduction to methodology and encoding rules . Journal of Chemical Information and Computer Sciences 1988 , 28 , 31 – 36 . OpenUrl CrossRef Web of Science (27). ↵ Korendovych , I. V. Rational and semirational protein design . Protein engineering: methods and protocols 2018 , 15 – 23 . (28). Song , Z. ; Zhang , Q. ; Wu , W. ; Pu , Z. ; Yu , H. Rational design of enzyme activity and enantioselectivity . Frontiers in Bioengineering and Biotechnology 2023 , 11 . (29). ↵ Alley , E. C. ; Khimulya , G. ; Biswas , S. ; AlQuraishi , M. ; Church , G. M. Unified Rational Protein Engineering with sequence-based deep representation learning . Nature Methods 2019 , 16 , 1315 – 1322 . OpenUrl (30). ↵ Arnold , F. H. ; Volkov , A. A. Directed evolution of biocatalysts . Current Opinion in Chemical Biology 1999 , 3 , 54 – 59 . OpenUrl CrossRef PubMed Web of Science (31). Wang , M. ; Zhao , H. Combined and iterative use of computational design and directed evolution for protein–ligand binding design . Methods in Molecular Biology 2016 , 139 – 153 . (32). Guntas , G. ; Mansell , T. J. ; Kim , J. R. ; Ostermeier , M. Directed evolution of protein switches and their application to the creation of ligand-binding proteins . Proceedings of the National Academy of Sciences 2005 , 102 , 11224 – 11229 . OpenUrl Abstract / FREE Full Text (33). ↵ Waltenspühl , Y. ; Jeliazkov , J. R. ; Kummer , L. ; Plückthun , A. Directed evolution for high functional production and stability of a challenging G protein-coupled receptor . Scientific Reports 2021 , 11 . (34). ↵ Raut , G. ; Singh , A. Generative AI in Vision: A Survey on Models, Metrics and Applications . 2024 . (35). ↵ Iqbal , T. ; Qureshi , S. The survey: Text generation models in deep learning . Journal of King Saud University - Computer and Information Sciences 2022 , 34 , 2515 – 2528 . OpenUrl (36). ↵ Lyu , S. ; Sowlati-Hashjin , S. ; Garton , M. ProteinVAE: Variational AutoEncoder for Translational Protein Design . bioRxiv 2023 , (37). Greener , J. G. ; Moffat , L. ; Jones , D. T. Design of metalloproteins and novel protein folds using variational autoencoders . Scientific Reports 2018 , 8 , 16189 . OpenUrl (38). ↵ Brookes , D. ; Park , H. ; Listgarten , J. Conditioning by adaptive sampling for robust design . Proceedings of the 36th International Conference on Machine Learning . 2019 ; pp 773 – 782 . (39). ↵ Trinquier , J. ; Uguzzoni , G. ; Pagnani , A. ; Zamponi , F. ; Weigt , M. Efficient generative modeling of protein sequences using simple autoregressive models . Nature Communications 2021 , 12 , 5800 . OpenUrl (40). ↵ Fannjiang , C. ; Bates , S. ; Angelopoulos , A. N. ; Listgarten , J. ; Jordan , M. I. Conformal prediction under feedback covariate shift for biomolecular design . Proceedings of the National Academy of Sciences 2022 , 119 . (41). ↵ Kucera , T. ; Togninalli , M. ; Meng-Papaxanthos , L. Conditional generative modeling for de novo protein design with hierarchical functions . Bioinformatics 2022 , 38 , 3454 – 3461 . OpenUrl CrossRef (42). ↵ Anand , N. ; Huang , P. Generative modeling for protein structures . Advances in Neural Information Processing Systems . 2018 . (43). ↵ Gruver , N. ; Stanton , S. ; Frey , N. C. ; Rudner , T. G. J. ; Hotzel , I. ; Lafrance-Vanasse , J. ; Rajpal , A. ; Cho , K. ; Wilson , A. G. Protein Design with Guided Discrete Diffusion . 2023 . (44). ↵ Watson , J. L. et al. De novo design of protein structure and function with RFdiffusion . Nature 2023 , 620 , 1089 – 1100 . OpenUrl (45). Wu , K. E. ; Yang , K. K. ; van den Berg , R. ; Zou , J. Y. ; Lu , A. X. ; Amini , A. P. Protein structure generation via folding diffusion . 2022 . (46). ↵ Fu , C. ; Yan , K. ; Wang , L. ; Au , W. Y. ; McThrow , M. ; Komikado , T. ; Maruhashi , K. ; Uchino , K. ; Qian , X. ; Ji , S. A Latent Diffusion Model for Protein Structure Generation . 2023 . (47). ↵ Zheng , Z. ; Deng , Y. ; Xue , D. ; Zhou , Y. ; Ye , F. ; Gu , Q. Structure-informed language models are protein designers . Proceedings of the 40th International Conference on Machine Learning . 2023 . (48). ↵ Yang , K. K. ; Zanichelli , N. ; Yeh , H. Masked inverse folding with sequence transfer for protein representation learning . Protein Engineering, Design and Selection 2022 , 36 , gzad015 . OpenUrl (49). ↵ Hsu , C. ; Verkuil , R. ; Liu , J. ; Lin , Z. ; Hie , B. ; Sercu , T. ; Lerer , A. ; Rives , A. Learning inverse folding from millions of predicted structures . ICML 2022 , (50). ↵ Bennett , N. R. et al. Atomically accurate de novo design of single-domain antibodies . bioRxiv 2024 , (51). ↵ Chungyoun , M. F. ; Gray , J. J. AI models for protein design are driving antibody engineering . Current Opinion in Biomedical Engineering 2023 , 28 , 100473 . OpenUrl (52). ↵ Gagliardi , L. ; Rocchia , W. SiteFerret: Beyond simple pocket identification in proteins . J. Chem. Theory Comput . 2023 , 19 , 5242 – 5259 . OpenUrl (53). ↵ Wang , R. ; Fang , X. ; Lu , Y. ; Wang , S. The PDBbind Database: Collection of Binding Affinities for Protein-Ligand Complexes with Known Three-Dimensional Structures . Journal of Medicinal Chemistry 2004 , 47 , 2977 – 2980 , PMID: 15163179 . OpenUrl CrossRef PubMed Web of Science (54). ↵ Sillitoe , I. ; Dawson , N. ; Lewis , T. E. ; Das , S. ; Lees , J. G. ; Ashford , P. ; Tolulope , A. ; Scholes , H. M. ; Senatorov , I. ; Bujan , A. ; Ceballos Rodriguez-Conde , F. ; Dowling , B. ; Thornton , J. ; Orengo , C. A. CATH: expanding the horizons of structure-based functional annotations for genome sequences . Nucleic Acids Research 2018 , 47 , D280 – D284 . OpenUrl (55). ↵ Yang , Z. ; Zeng , X. ; Zhao , Y. ; Chen , R. AlphaFold2 and its applications in the fields of biology and medicine . Signal Transduct. Target. Ther . 2023 , 8 , 115 . OpenUrl (56). ↵ Brandes , N. ; Ofer , D. ; Peleg , Y. ; Rappoport , N. ; Linial , M. ProteinBERT: a universal deep-learning model of protein sequence and function . Bioinformatics 2022 , 38 , 2102 – 2110 . OpenUrl (57). Elnaggar , A. ; Heinzinger , M. ; Dallago , C. ; Rehawi , G. ; Wang , Y. ; Jones , L. ; Gibbs , T. ; Feher , T. ; Angerer , C. ; Steinegger , M. ; Bhowmik , D. ; Rost , B. ProtTrans: Toward Understanding the Language of Life Through Self-Supervised Learning . IEEE Transactions on Pattern Analysis and Machine Intelligence 2022 , 44 , 7112 – 7127 . OpenUrl PubMed (58). Rives , A. ; Meier , J. ; Sercu , T. ; Goyal , S. ; Lin , Z. ; Liu , J. ; Guo , D. ; Ott , M. ; Zitnick , C. L. ; Ma , J. ; Fergus , R. Biological structure and function emerge from scaling unsupervised learning to 250 million protein sequences . Proceedings of the National Academy of Sciences 2021 , 118 , e2016239118 . OpenUrl Abstract / FREE Full Text (59). ↵ Lin , Z. ; Akin , H. ; Rao , R. ; Hie , B. ; Zhu , Z. ; Lu , W. ; Smetanin , N. ; Verkuil , R. ; Kabeli , O. ; Shmueli , Y. ; dos Santos Costa , A.; Fazel-Zarandi , M. ; Sercu , T. ; Candido , S. ; Rives , A. Evolutionary-scale prediction of atomic-level protein structure with a language model . Science 2023 , 379 , 1123 – 1130 . OpenUrl CrossRef PubMed (60). ↵ Madani , A. ; Krause , B. ; Greene , E. R. ; Subramanian , S. ; Mohr , B. P. ; Holton , J. M. ; Olmos , J. L. ; Xiong , C. ; Sun , Z. Z. ; Socher , R. ; Fraser , J. S. ; Naik , N. Large language models generate functional protein sequences across diverse families . Nature Biotechnology 2023 , 41 , 1099 – 1106 . OpenUrl (61). Ruffolo , J. A. ; Madani , A. Designing proteins with language models . Nature Biotechnology 2024 , 42 , 200 – 202 . OpenUrl (62). ↵ Min , X. ; Yang , C. ; Xie , J. ; Huang , Y. ; Liu , N. ; Jin , X. ; Wang , T. ; Kong , Z. ; Lu , X. ; Ge , S. ; Zhang , J. ; Xia , N. Tpgen: a language model for stable protein design with a specific topology structure . BMC Bioinformatics 2024 , 25 , 35 . OpenUrl (63). ↵ Suzek , B. E. ; Wang , Y. ; Huang , H. ; McGarvey , P. B. ; Wu , C. H. ; the UniProt Consortium UniRef clusters: a comprehensive and scalable alternative for improving sequence similarity searches . Bioinformatics 2014 , 31 , 926 – 932 . OpenUrl CrossRef (64). ↵ Kingma , D. P. ; Salimans , T. ; Poole , B. ; Ho , J. Variational Diffusion Models . 2023 . (65). ↵ Hoogeboom , E. ; Satorras , V. G. ; Vignac , C. ; Welling , M. Equivariant Diffusion for Molecule Generation in 3D . 2022 . (66). ↵ Song , Y. ; Sohl-Dickstein , J. ; Kingma , D. P. ; Kumar , A. ; Ermon , S. ; Poole , B. Score-Based Generative Modeling through Stochastic Differential Equations . arXiv 2020 , (67). ↵ Xu , T. ; Xu , Q. ; Li , J. Toward the appropriate interpretation of Alphafold2 . Front. Artif. Intell . 2023 , 6 , 1149748 . OpenUrl (68). ↵ Rao , R. M. ; Liu , J. ; Verkuil , R. ; Meier , J. ; Canny , J. ; Abbeel , P. ; Sercu , T. ; Rives , A. MSA Transformer . Proceedings of the 38th International Conference on Machine Learning . 2021 ; pp 8844 – 8856 . (69). ↵ Buton , N. ; Coste , F. ; Le Cunff , Y. Predicting enzymatic function of protein sequences with attention . Bioinformatics 2023 , 39 . (70). ↵ Ju , F. ; Zhu , J. ; Shao , B. ; Kong , L. ; Liu , T.-Y. ; Zheng , W.-M. ; Bu , D. CopulaNet: Learning residue co-evolution directly from multiple sequence alignment for protein structure prediction . Nat. Commun . 2021 , 12 , 2535 . OpenUrl CrossRef PubMed (71). ↵ Huang , Q. ; Smolensky , P. ; He , X. ; Deng , L. ; Wu , D. Tensor Product Generation Networks for Deep NLP Modeling . Proceedings of the 2018 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies , Volume 1 (Long Papers). New Orleans, Louisiana , 2018 ; pp 1263 – 1273 . OpenUrl (72). Smolensky , P. Tensor product variable binding and the representation of symbolic structures in connectionist systems . Artificial Intelligence 1990 , 46 , 159 – 216 . OpenUrl CrossRef Web of Science (73). ↵ Huang , Q. ; Deng , L. ; Wu , D. ; Liu , C. ; He , X. Attentive Tensor Product Learning . Proceedings of the AAAI Conference on Artificial Intelligence 2019 , 33 , 1344 – 1351 . OpenUrl (74). ↵ Schlag , I. ; Schmidhuber , J. Learning to reason with third-order tensor products . 2018 , (75). ↵ Chen , C. ; Lu , Q. ; Beukers , A. ; Baldassano , C. ; Norman , K. A. Learning to perform role-filler binding with schematic knowledge . PeerJ 2021 , 9 , e11046 . OpenUrl (76). ↵ Lin , Y. ; AlQuraishi , M. Generating novel, designable, and diverse protein structures by equivariantly diffusing oriented residue clouds . Proceedings of the 40th International Conference on Machine Learning . 2023 . (77). ↵ Steinegger , M. ; Söding , J. MMseqs2 enables sensitive protein sequence searching for the analysis of massive data sets . Nature Biotechnology 2017 , 35 , 1026 – 1028 . OpenUrl CrossRef PubMed (78). ↵ Morgan , H. L. The Generation of a Unique Machine Description for Chemical Structures-A Technique Developed at Chemical Abstracts Service . Journal of Chemical Documentation 1965 , 5 , 107 – 113 . OpenUrl (79). ↵ Ingraham , J. ; Garg , V. ; Barzilay , R. ; Jaakkola , T. Generative Models for Graph-Based Protein Design . Advances in Neural Information Processing Systems . 2019 . (80). ↵ Koh , H. Y. ; Nguyen , A. T. ; Pan , S. ; May , L. T. ; Webb , G. I. PSICHIC: physicochemical graph neural network for learning protein-ligand interaction fingerprints from sequence data . bioRxiv 2023 , 2023 – 09 . (81). ↵ Lovric , M. Joyce , J. M. In International Encyclopedia of Statistical Science ; Lovric , M. , Ed.; Springer Berlin Heidelberg : Berlin, Heidelberg , 2011 ; pp 720 – 722 . (82). ↵ Sawada , R. ; Sakajiri , Y. ; Shibata , T. ; Yamanishi , Y. Predicting therapeutic and side effects from drug binding affinities to human proteome structures . iScience 2024 , 27 , 110032 . OpenUrl (83). ↵ Ziegler , C. ; Martin , J. ; Sinner , C. ; Morcos , F. Latent generative landscapes as maps of functional diversity in protein sequence space . Nat. Commun . 2023 , 14 , 2222 . OpenUrl CrossRef (84). ↵ Miller , F. P. ; Vandome , A. F. ; McBrewster , J. Levenshtein Distance: Information theory, Computer science, String (computer science), String metric, Damerau?Levenshtein distance, Spell checker, Hamming distance ; Alpha Press , 2009 . (85). ↵ Zhang , Y. ; Skolnick , J. Scoring function for automated assessment of protein structure template quality . Proteins 2004 , 57 , 702 – 710 . OpenUrl CrossRef PubMed Web of Science (86). ↵ Laskowski , R. ; de Beer , T. Dictionary of Bioinformatics and Computational Biology ; John Wiley and Sons, Ltd , 2014 . (87). ↵ Bastolla , U. ; Abia , D. ; Piette , O. PC ali: a tool for improved multiple alignments and evolutionary inference based on a hybrid protein sequence and structure similarity score . Bioinformatics 2023 , 39 , btad630 . OpenUrl (88). ↵ Cheng , L. ; Liu , P. ; Wang , D. ; Leung , K.-S. Exploiting locational and topological overlap model to identify modules in protein interaction networks . BMC Bioinformatics 2019 , 20 , 23 . OpenUrl CrossRef (89). ↵ Iyer , M. ; Li , Z. ; Jaroszewski , L. ; Sedova , M. ; Godzik , A. Difference contact maps: From what to why in the analysis of the conformational flexibility of proteins . PLoS One 2020 , 15 , e0226702 . OpenUrl (90). ↵ Wu , R. ; Ding , F. ; Wang , R. ; Shen , R. ; Zhang , X. ; Luo , S. ; Su , C. ; Wu , Z. ; Xie , Q. ; Berger , B. ; Ma , J. ; Peng , J. High-resolution de novo structure prediction from primary sequence . bioRxiv 2022 , (91). ↵ Trott , O. ; Olson , A. J. AutoDock Vina: Improving the speed and accuracy of docking with a new scoring function, efficient optimization, and multithreading . Journal of Computational Chemistry 2010 , 31 , 455 – 461 . OpenUrl CrossRef PubMed Web of Science (92). ↵ Lin , Z. ; Akin , H. ; Rao , R. ; Hie , B. ; Zhu , Z. ; Lu , W. ; Smetanin , N. ; Verkuil , R. ; Kabeli , O. ; Shmueli , Y. ; dos Santos Costa , A. ; Fazel-Zarandi , M. ; Sercu , T. ; Candido , S. ; Rives , A. Evolutionary-scale prediction of atomic-level protein structure with a language model . Science 2023 , 379 , 1123 – 1130 . OpenUrl CrossRef PubMed (93). Stepniewska-Dziubinska , M. M. ; Zielenkiewicz , P. ; Siedlecki , P. Development and evaluation of a deep learning model for protein–ligand binding affinity prediction . Bioinformatics 2018 , 34 , 3666 – 3674 . OpenUrl CrossRef (94). Zheng , L. ; Fan , J. ; Mu , Y. OnionNet: a Multiple-Layer Intermolecular-Contact-Based Convolutional Neural Network for Protein–Ligand Binding Affinity Prediction . ACS Omega 2019 , 4 , 15956 – 15965 , PMID: 31592466 . OpenUrl CrossRef PubMed (95). Jiang , D. ; Hsieh , C.-Y. ; Wu , Z. ; Kang , Y. ; Wang , J. ; Wang , E. ; Liao , B. ; Shen , C. ; Xu , L. ; Wu , J. ; Cao , D. ; Hou , T. InteractionGraphNet: A Novel and Efficient Deep Graph Representation Learning Framework for Accurate Protein–Ligand Interaction Predictions . Journal of Medicinal Chemistry 2021 , 64 , 18209 – 18232 , PMID: 34878785 . OpenUrl CrossRef PubMed (96). Li , S. ; Zhou , J. ; Xu , T. ; Huang , L. ; Wang , F. ; Xiong , H. ; Huang , W. ; Dou , D. ; Xiong , H. Structure-Aware Interactive Graph Neural Networks for the Prediction of Protein-Ligand Binding Affinity . Proceedings of the 27th ACM SIGKDD Conference on Knowledge Discovery & Data Mining . New York, NY, USA , 2021 ; p 975 – 985 . (97). Koes , D. R. ; Baumgartner , M. P. ; Camacho , C. J. Lessons Learned in Empirical Scoring with smina from the CSAR 2011 Benchmarking Exercise . Journal of Chemical Information and Modeling 2013 , 53 , 1893 – 1904 . OpenUrl CrossRef PubMed (98). McNutt , A. T. ; Francoeur , P. ; Aggarwal , R. ; Masuda , T. ; Meli , R. ; Ragoza , M. ; Sunseri , J. ; Koes , D. R. GNINA 1.0: molecular docking with deep learning . Journal of Cheminformatics 2021 , 13 , 43 . OpenUrl (99). Sverrisson , F. ; Feydy , J. ; Correia , B. E. ; Bronstein , M. M. Fast end-to-end learning on protein surfaces . 2021 IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR) . 2021 ; pp 15267 – 15276 . (100). Lu , W. ; Wu , Q. ; Zhang , J. ; Rao , J. ; Li , C. ; Zheng , S. TANKBind: Trigonometry-Aware Neural NetworKs for Drug-Protein Binding Structure Prediction . bioRxiv 2022 , (101). Nguyen , T. ; Le , H. ; Quinn , T. P. ; Nguyen , T. ; Le , T. D. ; Venkatesh , S. GraphDTA: predicting drug–target binding affinity with graph neural networks . Bioinformatics 2020 , 37 , 1140 – 1147 . OpenUrl (102). Chen , L. ; Tan , X. ; Wang , D. ; Zhong , F. ; Liu , X. ; Yang , T. ; Luo , X. ; Chen , K. ; Jiang , H. ; Zheng , M. TransformerCPI: improving compound–protein interaction prediction by sequence-based deep learning with self-attention mechanism and label reversal experiments . Bioinformatics 2020 , 36 , 4406 – 4414 . OpenUrl CrossRef (103). Huang , K. ; Xiao , C. ; Glass , L. M. ; Sun , J. MolTrans: Molecular Interaction Transformer for drug–target interaction prediction . Bioinformatics 2020 , 37 , 830 – 836 . OpenUrl CrossRef (104). Bai , P. ; Miljković , F. ; John , B. ; Lu , H. Interpretable bilinear attention network with domain adaptation improves drug–target prediction . Nature Machine Intelligence 2023 , 5 , 126 – 136 . OpenUrl (105). Jiang , M. ; Li , Z. ; Zhang , S. ; Wang , S. ; Wang , X. ; Yuan , Q. ; Wei , Z. Drug–target affinity prediction using graph neural network and contact maps . RSC Adv . 2020 , 10 , 20701 – 20712 . OpenUrl CrossRef (106). Bai , P. ; Miljković , F. ; John , B. ; Lu , H. Interpretable bilinear attention network with domain adaptation improves drug–target prediction . Nature Machine Intelligence 2023 , 5 , 126 – 136 . OpenUrl (107). Wang , P. ; Zheng , S. ; Jiang , Y. ; Li , C. ; Liu , J. ; Wen , C. ; Patronov , A. ; Qian , D. ; Chen , H. ; Yang , Y. Structure-Aware Multimodal Deep Learning for Drug–Protein Interaction Prediction . Journal of Chemical Information and Modeling 2022 , 62 , 1308 – 1317 . OpenUrl CrossRef View the discussion thread. Back to top Previous Next Posted July 12, 2024. Download PDF Email Thank you for your interest in spreading the word about bioRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following Complex-based Ligand-Binding Proteins Redesign by Equivariant Diffusion-based Generative Models Message Subject (Your Name) has forwarded a page to you from bioRxiv Message Body (Your Name) thought you would like to see this page from the bioRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share Complex-based Ligand-Binding Proteins Redesign by Equivariant Diffusion-based Generative Models Viet Thanh Duy Nguyen , Nhan D. Nguyen , Truong Son Hy bioRxiv 2024.04.17.589997; doi: https://doi.org/10.1101/2024.04.17.589997 Share This Article: Copy Citation Tools Complex-based Ligand-Binding Proteins Redesign by Equivariant Diffusion-based Generative Models Viet Thanh Duy Nguyen , Nhan D. Nguyen , Truong Son Hy bioRxiv 2024.04.17.589997; doi: https://doi.org/10.1101/2024.04.17.589997 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Bioinformatics Subject Areas All Articles Animal Behavior and Cognition (7649) Biochemistry (17738) Bioengineering (13925) Bioinformatics (42059) Biophysics (21496) Cancer Biology (18643) Cell Biology (25577) Clinical Trials (138) Developmental Biology (13406) Ecology (19946) Epidemiology (2067) Evolutionary Biology (24370) Genetics (15627) Genomics (22551) Immunology (17772) Microbiology (40497) Molecular Biology (17212) Neuroscience (88786) Paleontology (667) Pathology (2845) Pharmacology and Toxicology (4835) Physiology (7663) Plant Biology (15177) Scientific Communication and Education (2047) Synthetic Biology (4304) Systems Biology (9838) Zoology (2272)

Text is read by the "Ask this paper" AI Q&A widget below. Extraction quality varies by source — PMC NXML preserves structure cleanly, OA-HTML may include some navigation residue, and OA-PDF can have broken hyphenation. The publisher copy (via DOI) is the canonical version.

My notes (saved in your browser only)

Ask this paper AI returns verbatim quotes from the full text · source: preprint-html

Answers must be backed by verbatim quotes from this paper's full text. Hallucinated quotes are dropped automatically; if no verbatim passage answers the question, we say so. How this works

Citation neighborhood (no data yet)

We don't have any in-corpus citations linked to this paper yet. This is a recent paper (2024) — citers typically take a year or two to land, and the OpenAlex reference graph may still be filling in.

Source provenance

europepmc
last seen: 2026-05-20T01:45:00.602351+00:00