A Method for Predicting Enzyme Substrate Specificity Residues Using Homologous Sequence Information

preprint OA: closed
📄 Open PDF Full text JSON View at publisher

Abstract

ABSTRACT Identifying amino acid residues that are critical for the catalytic function of enzymes is essential for elucidating reaction mechanisms, facilitating drug discovery, and advancing protein engineering. However, experimentally and computationally distinguishing residues that maintain structural integrity from those directly involved in enzymatic function remains a major challenge. In this study, we developed a methodology to identify amino acid residues that influence substrate specificity in enzymes with homologous structures. We framed the sequence comparison as a classification problem, treating each residue as a feature, thereby enabling the rapid and objective identification of key residues responsible for functional differences. To validate the proposed method, we applied it to three enzyme pairs— trypsin/chymotrypsin, adenylyl cyclase/guanylyl cyclase, and lactate dehydrogenase (LDH)/malate dehydrogenase (MDH). The results confirmed the accurate prediction of previously identified specificity-determining residues. Furthermore, we conducted experiments on the LDH/MDH pair and successfully introduced mutations into key residues to alter substrate specificity, enabling LDH to utilize oxaloacetate while maintaining its expression levels. These findings demonstrate the potential of this method for efficiently identifying residues that govern substrate specificity. We have further developed this approach into a practical tool, the EZSCAN: Enzyme Substrate-specificity and Conservation Analysis Navigator ( https://ezscan.pe-tools.com/ ), which enables rapid identification of amino acid residues critical for enzyme function. Abstract Figure
Full text 60,381 characters · extracted from preprint-html · click to expand
A Method for Predicting Enzyme Substrate Specificity Residues Using Homologous Sequence Information | bioRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-M677548'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search New Results A Method for Predicting Enzyme Substrate Specificity Residues Using Homologous Sequence Information Seiya Mori , View ORCID Profile Teppei Niide , View ORCID Profile Yoshihiro Toya , View ORCID Profile Hiroshi Shimizu doi: https://doi.org/10.1101/2025.05.25.656053 Seiya Mori 1 Department of Bioinformatic Engineering, Graduate School of Information Science and Technology, The University of Osaka , 1-5 Yamadaoka, Suita, Osaka 565-0871, Japan Find this author on Google Scholar Find this author on PubMed Search for this author on this site Teppei Niide 1 Department of Bioinformatic Engineering, Graduate School of Information Science and Technology, The University of Osaka , 1-5 Yamadaoka, Suita, Osaka 565-0871, Japan Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Teppei Niide For correspondence: tniide{at}ist.osaka-u.ac.jp shimizu{at}ist.osaka-u.ac.jp Yoshihiro Toya 1 Department of Bioinformatic Engineering, Graduate School of Information Science and Technology, The University of Osaka , 1-5 Yamadaoka, Suita, Osaka 565-0871, Japan Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Yoshihiro Toya Hiroshi Shimizu 1 Department of Bioinformatic Engineering, Graduate School of Information Science and Technology, The University of Osaka , 1-5 Yamadaoka, Suita, Osaka 565-0871, Japan Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Hiroshi Shimizu For correspondence: tniide{at}ist.osaka-u.ac.jp shimizu{at}ist.osaka-u.ac.jp Abstract Full Text Info/History Metrics Supplementary material Preview PDF ABSTRACT Identifying amino acid residues that are critical for the catalytic function of enzymes is essential for elucidating reaction mechanisms, facilitating drug discovery, and advancing protein engineering. However, experimentally and computationally distinguishing residues that maintain structural integrity from those directly involved in enzymatic function remains a major challenge. In this study, we developed a methodology to identify amino acid residues that influence substrate specificity in enzymes with homologous structures. We framed the sequence comparison as a classification problem, treating each residue as a feature, thereby enabling the rapid and objective identification of key residues responsible for functional differences. To validate the proposed method, we applied it to three enzyme pairs— trypsin/chymotrypsin, adenylyl cyclase/guanylyl cyclase, and lactate dehydrogenase (LDH)/malate dehydrogenase (MDH). The results confirmed the accurate prediction of previously identified specificity-determining residues. Furthermore, we conducted experiments on the LDH/MDH pair and successfully introduced mutations into key residues to alter substrate specificity, enabling LDH to utilize oxaloacetate while maintaining its expression levels. These findings demonstrate the potential of this method for efficiently identifying residues that govern substrate specificity. We have further developed this approach into a practical tool, the EZSCAN: Enzyme Substrate-specificity and Conservation Analysis Navigator ( https://ezscan.pe-tools.com/ ), which enables rapid identification of amino acid residues critical for enzyme function. Download figure Open in new tab INTRODUCTION Enzymes are biological catalysts that drive chemical reactions with high precision and efficiency. Their catalytic function originates from coordinated conformational changes and electron transfer mediated by amino acid side chains 1 - 4 . Identifying the amino acid residues essential for catalysis and substrate specificity is critical for understanding structure–function relationships. This knowledge provides practical insights into how mutations contribute to disease pathogenesis and how targeted mutagenesis can be used to redesign enzyme specificity for biotechnological applications, such as metabolic engineering and cell manipulation 5 , 6 . Advances in DNA synthesis and sequencing technologies have paved the way for in-depth studies of mutation–function relationships, such as deep mutational scanning and multiplexed variant assays 7 - 9 . These methods systematically mutate each amino acid residue into all 20 possible variants, generating extensive libraries. Using NGS, researchers can identify residues critical for enzyme function by subjecting these populations to functional selection and tracking changes in abundance. Although large-scale mutation analysis offers functional mutation profiles, distinguishing functionally critical residues from those conserved due to structural constraints remains a challenge, as does identifying synergistic mutations 10 . Moreover, as nearly half of the loss-of-function mutations result from decreased protein abundance 11 , new approaches are necessary to pinpoint residues directly influencing enzyme function. In contrast to experimental methods, computational techniques offer rapid, cost-effective, and scalable alternatives for identifying functionally important residues. Several computational approaches have been developed to predict these residues based on their amino acid sequences. Among these, conservation analysis—which identifies residues that are highly conserved across proteins—has emerged as a powerful tool for identifying functionally critical residues 12 - 15 . Although this method has provided valuable insights into protein–protein interactions, structural stability, and ligand recognition, it requires refinement to distinguish between residues essential for function and those conserved due to structural constraints. As both functional and structural constraints shape protein evolution, the challenge is to identify residues that are crucial for protein function without being confounded by structural conservation. Recent advances in molecular biology and machine learning have shed light on how specific amino acid residues determine cofactor specificity. Using supervised learning on amino acid sequence datasets, we previously identified key residues that distinguish between NAD(H)- and NADP(H)-dependent malic enzymes 16 . Despite clear differences in cofactor preferences, these enzymes retain a highly conserved overall structure across species. Guided by machine learning-based residue rankings, we introduced mutations that not only preserved soluble expression but also completely switched the enzyme’s cofactor specificity from NADP to NAD. Notably, these substitutions were well tolerated, underscoring the functional relevance of the identified sites and enabling the separation of structural and functional constraints, which is difficult to achieve through conventional conservation analysis. These findings pointed to a broader principle: functionally critical residues underlying substrate specificity can be identified by contrasting enzymes that are structurally conserved yet functionally distinct. In this study, we present a computational framework to uncover the molecular basis of enzyme substrate specificity. By analyzing the sequence datasets of homologous enzymes using supervised machine learning, we identified key amino acid residues that govern substrate recognition. Focusing on three well-studied enzyme pairs—trypsin/chymotrypsin, adenylyl/guanylyl cyclase (AC/GC), and lactate/malate dehydrogenase (LDH/MDH)—we recovered known specificity-conferring residues and revealed previously unreported sites critical for function. Experimental validation of the LDH/MDH pair confirmed that the newly identified residues contributed directly to differences in substrate preference. Although protein sequences and functions have diversified through evolution, their three-dimensional structures are often highly conserved 17 , 18 . Comparative analyses of these protein families can provide key insights into the evolution of functions while maintaining structural constraints. Our approach not only distinguishes functionally relevant positions in structurally similar enzymes but also provides a generalizable strategy for dissecting enzyme specificity. To support its broad adoption, we developed EZSCAN, a web tool that enables researchers to explore substrate recognition features across diverse enzyme families. RESULTS Prediction of substrate-specific residues in enzymes using EZSCAN The EZSCAN protocol represents an advancement in understanding enzyme functionality through a machine-learned binary classification algorithm tailored to extract critical amino acid residues linked to enzyme cofactor specificity. In this study, we extended this approach to identify amino acid residues vital for substrate specificity by leveraging two distinct sets of amino acid sequence data ( Figure 1 ). Initially, we obtained amino acid sequences of two sets of enzymes with homologous structures from a comprehensive database. These sequences were aligned using multiple sequence alignment, converted into one-hot vectors, and subsequently analyzed using a logistic regression model. Given the structural homology between the enzyme sets, our classification of residues associated with enzymatic function is expected to yield meaningful results. The key explanatory variable in the trained model was the amino acid type at each position, and the range between the maximum and minimum partial regression coefficients served as an important evaluative metric. Download figure Open in new tab Figure 1. Schematic illustration of estimation process for specificity-conferring residues using the EZSCAN protocol. The amino acid sequences of two structurally homologous enzyme groups are used as input data, and a logistic regression model is trained to extract specificity-conferring residues that distinguish the two enzyme groups. The extracted specificity-conferring residues can be ranked by their contribution and used for experimental evaluation. As a practical demonstration, we applied this method to three enzyme pairs— trypsin/chymotrypsin, AC/GC, and LDH/MDH—to identify residues critical for enzymatic function. Despite their differences in substrate preference, these enzyme pairs share homologous structures ( Figure 2 ). The average root mean square deviation (RMSD) and template modeling score (TM-score) values for each enzyme pair further confirmed their high degree of structural similarity. Details of RMSD and TM scores across all crystal structures are provided in the Supplementary Information (Supplementary Tables S1–S6). TM-scores range from 0 to 1, with values above 0.5 indicative of structural homology 19 . In the following sections, we present results identifying amino acid residues that are pivotal for substrate specificity across these different enzymes. Download figure Open in new tab Figure 2. Superimposed images. (A) Trypsin (PDB ID 6T5W, 1ANE, and 1OS8) and chymotrypsin (PDB ID 1KDQ, 1ACB, and 1EQ9). (B) AC (PDB ID 1AB8, 6R3Q, and 7YZI) and GC (PDB ID 2WZ1, 3ET6, and 6PAS). (C) LDH (PDB ID 1LDG, 1LDN, 3VPH, 4AJ2, and 6J9T) and MDH (PDB ID 1B8P, 1HLP, 2PWZ, 4CL3, and 5UJK). Red, yellow, and green regions represent α-helix, β-sheet, and random coil structures, respectively. Average values of metrics from the superimposed structures are shown. All scores compared between each structure are shown in Supplementary Tables S1–S6. Trypsin/chymotrypsin Trypsin and chymotrypsin, both serine proteases, exhibited significant structural homology ( Figure 2A ). Trypsin is known for cleaveing the C-terminal side of Arg and Lys at the P1 position of substrate peptides, whereas chymotrypsin targets Phe, Tyr and Trp at the same position 20 . This distinction in substrate specificity is primarily attributed to the pivotal role of residue S195 (in chymotrypsin numbering), which defines the S1 pocket adjacent to the active site 21 . Notably, the negative charge generated by the combination of D189, G216, and G226 (chymotrypsin numbering) significantly influences trypsin substrate specificity. In contrast, S189, G216, and G226 (chymotrypsin numbering) are critical for chymotrypsin specificity 22 . Interestingly, swapping D189 and S189 alone is insufficient to shift specificity from trypsin-like to chymotrypsin-like or vice versa 23 , 24 . Additionally, Y172, although not directly involved in substrate interaction, is essential when transitioning substrate specificity between two enzymes 25 . To explore the predictive potential of EZSCAN for identifying specificity-conferring residues, we applied the protocol to trypsin and chymotrypsin. Amino acid sequence data for these enzymes were obtained from the Kyoto Encyclopedia of Genes and Genomes (KEGG), focusing on sequences between 240–270 residues in length. A total of 793 trypsin sequences and 652 chymotrypsin sequences were used as input for EZSCAN analysis (Supplementary Figure S1A). We used trypsin and chymotrypsin structures from Rattus norvegicus as templates to display specificity-conferring residues. The model predicted Tyr (trypsin) and Trp (chymotrypsin) at residue 172 as the top specificity-conferring residues ( Figure 3A , Table 1 ). Notably, Asp (trypsin) and Ser (chymotrypsin) at residue 189 were ranked fourth, consistent with prior studies on substrate specificity. Download figure Open in new tab Figure 3. Mapping of specificity-conferring residues estimated by EZSCAN on the crystal structure. (A) Trypsin derived from R. norvegicus (PDB: 1ANE), (B) AC derived from R. norvegicus (PDB: 1AB8), (C) LDH derived from G. stearothermophilus (PDB: 1LDN). Cyan sticks represent the top 10 ranked amino acid residues estimated by the EZSCAN protocol. Residues previously reported to be involved in substrate specificity are shown in blue; all others are shown in black. The numbers in parentheses indicate their ranking. View this table: View inline View popup Download powerpoint Table 1. Top 10 ranked amino acid residues associated with substrate specificity for trypsin/chymotrypsin, AC/GC, and LDH/MDH as predicted by EZSCAN. The predicted residue positions correspond to the amino acid positions in each template enzyme. For trypsin/chymotrypsin, the residue numbering follows the chymotrypsin numbering scheme. The second ranked residue, Tyr (trypsin) and Trp (chymotrypsin) at residue 39, is located distally from the active site but is conserved in mesotrypsin, which is known to interact with protease inhibitors 26 . This suggests that the second ranked residue may also influence substrate recognition. Additionally, the third ranked residue at position 219 and the fifth ranked residue at position 221 form a loop near the substrate pocket, indicating their likely contribution to the substrate recognition process. AC/GC AC and GC are crucial enzymes in signal transduction, responsible for converting ATP and GTP into cAMP and cGMP, respectively. ACs are classified into five categories based on their structural characteristics 27 , 28 . Notably, Class III ACs exhibit a close phylogenetic relationship with GCs, and their catalytic domains demonstrate significant structural homology ( Figure 2B ). The catalytic domain of mammalian Class III AC exists as a heterodimer composed of C1 and C2 domains, whereas the bacterial and protozoan counterparts are homodimeric 29 , 30 . These enzymes exhibit strict selectivity for ATP or GTP, with specific amino acid residues at the dimer interface playing a pivotal role in determining substrate specificity 31 - 33 . For instance, it has been shown that GC from Bos taurus can be engineered to function similarly to AC through just two amino acid substitutions at the active site—E930K and C1002D 34 . In contrast, similar modifications at corresponding sites in AC from R. norvegicus did not produce a substantial change in substrate specificity. Residue I1019 has been implicated in the formation of hydrogen bonds with the N-6 amino group of ATP and is considered essential for this process 35 . We aimed to evaluate whether such specificity-conferring residues in AC and GC could be predicted using the EZSCAN software. Amino acid sequence data for AC and GC were obtained from the KEGG database. We analyzed 319 ACs and 572 GCs, with sequence lengths ranging from 1,090 to 1,130 amino acids (Supplementary Figure S1B). In this analysis, AC from R. norvegicus and GC from B. taurus served as template structures for identifying specificity-conferring residues. EZSCAN identified E930 and C1002 as the third and fourth highest-ranked residues for AC from R. norvegicus , consistent with previous findings demonstrating their role in shifting substrate specificity ( Figure 3B , Table 1 ). Another key residue, I1019, ranked second and is known to directly interact with the substrate. The top-ranked mutation in GC has been implicated in central areolar choroidal dystrophy, a genetic eye disease 36 , indicating its potential significance in cyclase function. Interestingly, although our analysis encompassed the full-length sequences of both AC and GC, the top-ranked residues predominantly resided in the C2 domain. This finding aligns with prior studies 34 , 35 , 37 , 38 and suggests that the presence of non-homologous transmembrane domains does not interfere with the identification of residues essential for substrate specificity. LDH/MDH LDH and MDH are essential redox enzymes characterized by structural homology and a reliance on NAD(H) as a cofactor ( Figure 2C ). LDH catalyzes the interconversion between lactate and pyruvate, while MDH is involved in the conversion of malate and oxaloacetate 39 , 40 . The differences in substrate specificity between these enzymes are primarily due to variations in amino acid residues within their substrate-binding sites. Notably, LDH from Geobacillus stearothermophilus can acquire MDH activity through a single Q86R mutation 41 . Conversely, introducing the reverse mutation into MDH from Escherichia coli does not result in a comparable enhancement in LDH activity 42 , 43 . Furthermore, one study reported that introducing five specific mutations—I12V, R81Q, M85E, G210A, and V214I—dramatically increased the k cat /K M value of E. coli MDH from 0.14 to 3,500 MlJ¹slJ¹ 44 . These findings raise an important question: can EZSCAN accurately identify the residues responsible for such differences in substrate specificity between LDH and MDH? We sourced amino acid sequences for LDH and MDH from the UniProtKB database and analyzed 228 LDH and 397 MDH sequences. The sequence lengths ranged from 300 to 340 amino acids (Supplementary Figure S1C). Using LDH from G. stearothermophilus and MDH from E. coli as template structures, we identified key specificity-conferring residues. Residues Q81, M85, and G210, previously reported to play crucial roles in substrate specificity, were ranked first, second, and fourth, respectively ( Figure 3C , Table 1 ). The fifth-ranked mutation, from Thr to Gly, was shown to reduce the enzymatic activity of G. stearothermophilus LDH by more than 1,000-fold, highlighting its importance in substrate recognition 41 . Interestingly, the third-ranked residue, I237, has not been previously reported but is located at the base of the substrate pocket, suggesting a potential role in influencing substrate specificity. We applied the EZSCAN protocol to three enzyme pairs based on prior experimental findings. This approach successfully identified amino acid residues known to determine substrate specificity and ranked them prominently. Additionally, EZSCAN revealed new candidate residues that have not been examined previously, offering valuable targets for future experimental validation. This method also provides a comprehensive view by quantifying the contribution of all 20 amino acid variants at each residue position, enabling a clearer understanding of the mechanisms underlying enzyme specificity (Supplementary Figures S2–S4). These findings support the conclusion that machine learning analysis of homologous enzyme sequences can effectively uncover substrate specificity-conferring residues. Selection of LDH for Experimental Validation We experimentally validated the amino acid residues associated with substrate specificity, as predicted by the EZSCAN protocol. LDH catalyzes the interconversion between pyruvate and lactate, whereas MDH catalyzes the conversion of oxaloacetate to malate ( Figure 4A ). To evaluate the accuracy of EZSCAN predictions, we assessed whether the specificity of template LDHs for pyruvate could be decreased and their specificity for oxaloacetate increased. Download figure Open in new tab Figure 4. (A) Reactions catalyzed by LDH and MDH. (B) Phylogenetic tree constructed from the LDH/MDH dataset used for model training. Blue and orange branches represent LDH and MDH, respectively. (C) Aligned amino acid sequences of pfLDH, gsLDH, and lcLDH. Dashes indicate alignment gaps. An insertion of SDKEW is observed at positions 89–93 in pfLDH. Residue numbering corresponds to pfLDH. (D) Design of pfLDH_trunc, in which residues 89–93 (SDKEW) of pfLDH are deleted. Black indicates the truncated region; magenta and cyan highlight residues ranked first and second in importance, respectively. Four LDHs from three species were used as template enzymes, and the EZSCAN-predicted specificity-conferring residues in these LDHs were replaced with MDH-like residues. The LDHs selected for mutagenesis were LDH from G. stearothermophilus (gsLDH; UniProt ID: P00344), LDH from Lactobacillus casei (lcLDH; UniProt ID: P00343), and LDH from Plasmodium falciparum (pfLDH; UniProt ID: Q27743). gsLDH has previously been reported to acquire MDH activity via a single Q86R mutation 41 . lcLDH was selected because it has a lower optimal pH (4.8) than other LDHs 45 , which could lead to distinct mutational effects compared with gsLDH. pfLDH is a unique LDH found within the MDH clade rather than the LDH clade ( Figure 4B ), suggesting it may have arisen by convergent evolution from an MDH ancestor 46 . Its sequence is therefore expected to resemble that of MDH, making it a suitable candidate for examining differences between enzyme templates. Alignment of pfLDH with gsLDH and lcLDH revealed a five-amino acid insertion from S89 to W93 in pfLDH ( Figure 4C ). These inserted residues are located in the same loop where EZSCAN ranked residues first and second in importance, suggesting that the insertion likely alters the shape of the substrate pocket (Supplementary Figure S5). In addition to wild-type pfLDH, we designed a truncated variant, pfLDH_trunc, in which the five-residue insertion was removed ( Figure 4D ). Structural prediction indicated that pfLDH_trunc is homologous to gsLDH and lcLDH (Supplementary Figure S6). Using these four types of LDHs as templates, we evaluated whether the EZSCAN-predicted specificity-conferring residues, when replaced, would alter substrate specificity. Experimental Validation of Substrate Specificity Conversion The investigation of four types of LDH involved substituting residues identified by the EZSCAN protocol with MDH-like amino acid residues to assess their impact on substrate specificity. These substitutions were introduced sequentially, one residue at a time, starting with the highest-ranked candidate. Because pfLDH retained its third-ranked Pro residue, pfLDH3 incorporated the fourth-ranked amino acid, pfLDH4 added the fifth-ranked residue, and pfLDH5 introduced the sixth-ranked residue. Each gene encoding the wild-type LDH and its corresponding mutants was cloned into a pET28a expression vector. Expression was carried out in E. coli BL21(DE3), and purification was performed using a Ni-NTA column. All LDH variants, including the mutants, were obtained in the soluble fraction. Expression levels ranged from 40.0 to 167.2% relative to the wild-type pfLDH, with no significant decrease in expression observed (Supplementary Figure S6). To analyze substrate specificity, the purified enzymes were tested using pyruvate and oxaloacetate, and initial reaction velocities were determined by monitoring NADH oxidation at 340 nm. Kinetic parameters were derived from these initial velocities via nonlinear fitting, as shown in Figure 5 . Notably, the Q86R mutation in gsLDH (gsLDH1) drastically reduced its LDH activity, decreasing its k cat /K M value to 1/438 of the wild-type level. At the same time, it enabled MDH activity that was undetectable in the wild-type enzyme. gsLDH1 exhibited a k cat /K M for MDH of 366.1 mMlJ¹ slJ¹, surpassing the wild-type LDH activity. This switch in substrate specificity due to the Q86R mutation is consistent with previous findings 41 . Further mutations in gsLDH3 caused LDH activity to fall below detection limits, yielding a purely MDH-like enzyme. gsLDH4 showed a 4.4-fold increase in MDH activity compared to gsLDH3, while gsLDH5 lost MDH activity entirely. Download figure Open in new tab Figure 5. Enzyme activity ( k cat /K M ) of each LDH variant before and after mutation, based on rankings from the EZSCAN protocol. Blue bars represent enzyme activity using pyruvate as the substrate, and red bars represent activity using oxaloacetate. Data were collected in triplicate. Asterisks indicate cases in which k cat /K M values could not be determined. Michaelis–Menten plots and fitted curves are provided in Supplementary Figure S7. The k cat and K M values for each LDH are listed in Supplementary Table S7. In contrast, lcLDH lost LDH activity with just a single mutation (Q88R), while showing a 15-fold increase in MDH activity, which had previously been nearly undetectable. The lcLDH2 variant, incorporating the additional second-ranked mutation (E92M), displayed an 80-fold increase in MDH activity—a 1,191-fold increase over the wild-type. However, as more mutations were added, MDH activity gradually declined, and lcLDH5 ultimately lost all MDH activity. For pfLDH, a consistent decline in LDH activity was observed with each successive mutation, ranked according to EZSCAN. pfLDH5 retained only 1/3987 of the wild-type LDH activity. Although pfLDH4 exhibited slight MDH activity, the conversion to MDH-like specificity was not as pronounced as in gsLDH or lcLDH. These results suggest that the SDKEW insertion sequence, spanning residues 159–163 in pfLDH, hinders the acquisition of MDH activity while preserving LDH function. We further evaluated pfLDH_trunc, in which the SDKEW sequence was removed. This truncation led to a complete loss of LDH activity, consistent with previous findings by Douglas et al. 46 , 47 . This outcome underscores the importance of the SDKEW loop structure in maintaining LDH activity. Interestingly, pfLDH_trunc gained progressively stronger MDH activity with the introduction of additional mutations. Among these, pfLDH_trunc5 exhibited the highest MDH activity. This result suggests that converting pfLDH to MDH functionality—likely governed by a distinct reaction mechanism— requires both structural adjustments to the backbone near functional residues and targeted substitutions. These findings validate that EZSCAN-predicted residues directly influence substrate specificity. All introduced mutations led to reductions in LDH activity while promoting MDH activity, without significantly affecting expression levels. This confirms that the amino acid residues identified by EZSCAN are indeed specificity-conferring residues. DISCUSSION Directed evolution is a widely employed and powerful strategy in protein engineering. However, the vast sequence space of proteins makes it challenging to efficiently identify variants with the desired functions efficiently 48 In typical directed evolution, random mutations or DNA shuffling are introduced into natural proteins, followed by high-throughput screening under selective pressure to enhance protein properties such as activity or stability. Although the number of substitutions required to achieve a functional shift varies depending on the direction of engineering, a few to a dozen amino acid changes are often sufficient for functional enhancement or adaptation. Importantly, protein function and stability often exhibit a trade-off relationship 49 , 50 , highlighting the need for strategies that improve function without compromising structural integrity. A key to the successful design for substrate specificity conversion is to distinguish amino acids that contribute to function from those responsible for structure, and to focus mutations only on the former. Computational tools are indispensable for identifying important amino acid residues in proteins, particularly for scalable and broadly applicable analyses 51 , 52 . Although recent methods using NMR chemical shifts have shown promise in narrowing down potential mutation sites 53 , 54 , in silico approaches remain central to efficiently exploring large sequence spaces. Among these, evolutionary conservation information is frequently used to estimate the functional or structural contribution of each amino acid residue 55 , 56 . Highly conserved residues are likely essential for protein folding and biological activity in native contexts. However, one of the remaining challenges is separating residues critical for specific functions from those involved in structural integrity—something that conventional conservation-based methods alone cannot achieve. In this study, we proposed a methodology in which hundreds of amino acid sequences from two structurally homologous enzyme groups are analyzed using a simple linear regression equation to extract amino acid residues where differences in substrate specificity between the enzyme groups appear. Three structurally homologous enzyme pairs (trypsin/chymotrypsin, AC/GC, and LDH/MDH) that had already been investigated in previous studies were selected, and an attempt was made to extract the amino acid residues responsible for substrate specificity from evolutionary information. By applying this method, we accurately estimated amino acid residues that are experimentally known to confer substrate specificity and identified new amino acid residues that have not yet been investigated. Furthermore, the deduced amino acid residues for the LDH/MDH pair were experimentally validated. Four LDHs from three phylogenetically distinct species were selected, and the amino acid residues identified by the ranking-based analysis were replaced with MDH-like residues to evaluate substrate specificity. The results showed that substrate specificity shifted stepwise from pyruvate to oxaloacetate as mutations accumulated. Furthermore, the expression levels of all mutants varied in the range of 40.0%–167.2%, and the introduced mutations had no significant effect on expression levels. These results suggest that the amino acid residues inferred using this method are indeed responsible for functional specificity. We also examined the involvement of conserved amino acid residues in substrate specificity. Using the 228 LDH sequences previously employed in our EZSCAN protocol to identify substrate specificity differences between LDH and MDH, we calculated the conservation level for each residue. Highly conserved residues were distributed throughout the protein sequence ( Figure 6A ). Interestingly, the top five residues predicted by EZSCAN as functionally important for LDH activity had conservation ratios ranging from 0.852 to 0.988. When the conservation ratios of all residues were mapped onto the three-dimensional structure of gsLDH, we observed that the conserved residues were particularly concentrated in the structural core, the multimer interface, and the substrate-binding pocket of the protein ( Figure 6B ). This pattern supports earlier findings that residues in the structural core evolve more slowly than those on the surface 18 . In total, 72 residues in the gsLDH sequence had conservation ratios above 0.8, making it difficult to determine which residues were specifically responsible for substrate recognition ( Figure 6C ). These findings suggest that substrate-specific residues can be hidden in highly conserved regions, making them difficult to identify using conservation-based methods alone. Notably, several key residues identified using our method were not completely conserved, implying that some functionally important residues may tolerate minor evolutionary variation. This highlights the need for comparison methods that consider not only sequence conservation but also functional divergence. Our approach, which contrasts sequences based on functional differences among structurally related enzymes, offers a promising strategy for identifying such critical residues. Download figure Open in new tab Figure 6. Visualization of conserved residues. (A) Conservation scores of each amino acid residue using gsLDH as the reference sequence. Scores closer to 1 indicate higher conservation. Red plots indicate the top 1–5 ranked positions predicted by the EZSCAN protocol as critical for LDH function. (B) Structural representation of gsLDH (PDB ID: 1LDB) with conservation scores mapped onto each residue. The color bar indicates the conservation ratio. (C) List of amino acid residues with conservation scores of 0.8 or higher. Residues highlighted in red represent the top 1–5 ranked positions predicted by the EZSCAN protocol as critical for LDH function. To make this method broadly accessible, we developed a user-friendly software tool called EZSCAN. Many proteins share structural similarities despite performing different biological functions. This structural similarity can be explored using databases such as SCOP 57 , CATH 58 , and InterPro 59 , which classify proteins based on their structural and evolutionary relationships. For example, the SCOP database defines over 5,900 families of proteins that share structural motifs but often fulfill distinct biological roles. In addition to these databases, online tools such as Foldseek 60 , DALI 61 , and PDBeFold 62 can rapidly identify proteins with similar structures. EZSCAN can be used not only to identify amino acid residues responsible for enzyme substrate specificity but also to highlight differences between other structurally similar protein pairs that differ in function. With the rapid expansion of amino acid sequence data from genome and metagenome projects, and the increasing availability of predicted protein structures driven by deep learning, the potential applications of EZSCAN are expected to grow significantly across both biology and biotechnology. In summary, we have proposed a methodology for extracting substrate specificity-conferring residues using a linear regression-based classification program that compares groups of enzymes with homologous sequences. The estimated amino acid residues altered substrate specificity without disrupting overall protein structure. This method enables the identification and extraction of shared sequence patterns important for protein function and structure—something difficult to achieve using conservation information alone. Furthermore, because the method is highly interpretable, relying only on a linear model, it may prove useful for experimental validation of protein function and for selecting mutational targets in protein engineering. ASSOCIATED CONTENT The Supporting Information is available free of charge at https://XXXXX. Histogram of amino acid sequence length that used in the machine learning dataset (Figure S1); Aligned heatmap of specificity-conferring residues in trypsin /chymotrypsin, AC/GC, and LDH/MDH (Figure S2–S4); Structural models of pfLDH_trunc (Figure S5); SDS-PAGE image of purified LDHs and their yield (Figure S6); Substrate-dependent kinetics of LDH and MDH reactions (Figure S7); RMSD and TM-score values of pairwise structure of trypsin/chymotrypsin, AC/GC, and LDH/MDH (Table S1–S6); Kinetic parameters of LDHs (Table S7) AUTHOR INFORMATION Corresponding Authors Teppei Niide − Department of Bioinformatic Engineering, Graduate School of Information Science and Technology, The University of Osaka, 1-5 Yamadaoka, Suita, Osaka 565-0871, Japan ; E-mail: tniide{at}ist.osaka-u.ac.jp Hiroshi Shimizu − Department of Bioinformatic Engineering, Graduate School of Information Science and Technology, The University of Osaka, 1-5 Yamadaoka, Suita, Osaka 565-0871, Japan ; E-mail: shimizu{at}ist.osaka-u.ac.jp Authors Seiya Mori − Department of Bioinformatic Engineering, Graduate School of Information Science and Technology, The University of Osaka, 1-5 Yamadaoka, Suita, Osaka 565-0871, Japan Yoshihiro Toya − Department of Bioinformatic Engineering, Graduate School of Information Science and Technology, The University of Osaka, 1-5 Yamadaoka, Suita, Osaka 565-0871, Japan Complete contact information is available at: https://XXXXX. Author Contributions T.N. and H.S. conceived and designed research, with contributions from Y.T., and S.M. conducted experiments. S.M. and T.N. analyzed and interpreted results. S.M. and T.N. wrote the manuscript with input from all authors. T.N. and H.S. supervised the study. All authors approved the content of the submitted manuscript. Notes There are no conflicts of interest directly relevant to the content of this work. ACKNOWLEDGMENTS We are grateful to Keiko Hiratomi and Satoko Wake for their expert technical assistance. This work was supported by Japan Society for the Promotion of Science KAKENHI (22K04841; T.N.), ACT-X (JPMJAX20BC; T.N.), PRESTO (JPMJPR24G5; T.N.), GteX (JPMJGX23B4; T.N. and Y.T.), BOOST (JPMJBS2402; S.M.) Japan Science and Technology Agency. Funder Information Declared Japan Society for the Promotion of Science, https://ror.org/00hhkn466 , 22K04841 Japan Science and Technology Agency, https://ror.org/00097mb19 , JPMJAX20BC , JPMJPR24G5 , JPMJGX23B4 , JPMJBS2402 REFERENCES (1). ↵ Hammes , G. G . Multiple conformational changes in enzyme catalysis . Biochemistry-Us 2002 , 41 ( 26 ), 8221 – 8228 . DOI: 10.1021/bi0260839 . OpenUrl CrossRef PubMed (2). Arora , K. ; Brooks , C. L . Large-scale allosteric conformational transitions of adenylate kinase appear to involve a population-shift mechanism . P Natl Acad Sci USA 2007 , 104 ( 47 ), 18496 – 18501 . DOI: 10.1073/pnas.0706443104 . OpenUrl Abstract / FREE Full Text (3). Siegbahn , P. E. ; Blomberg , M. R . Quantum chemical studies of proton-coupled electron transfer in metalloenzymes . Chem Rev 2010 , 110 ( 12 ), 7040 – 7061 . DOI: 10.1021/cr100070p . OpenUrl CrossRef PubMed Web of Science (4). ↵ Palmer , A. G . Enzyme Dynamics from NMR Spectroscopy . Accounts Chem Res 2015 , 48 ( 2 ), 457 – 465 . DOI: 10.1021/ar500340a . OpenUrl CrossRef PubMed (5). ↵ Yue , P. ; Li , Z. L. ; Moult , J . Loss of protein structure stability as a major causative factor in monogenic disease . J Mol Biol 2005 , 353 ( 2 ), 459 – 473 . DOI: 10.1016/j.jmb.2005.08.020 . OpenUrl CrossRef PubMed Web of Science (6). ↵ Adzhubei , I. A. ; Schmidt , S. ; Peshkin , L. ; Ramensky , V. E. ; Gerasimova , A. ; Bork , P. ; Kondrashov , A. S. ; Sunyaev , S. R . A method and server for predicting damaging missense mutations . Nat Methods 2010 , 7 ( 4 ), 248 – 249 . DOI: 10.1038/nmeth0410-248 . OpenUrl CrossRef PubMed Web of Science (7). ↵ Fowler , D. M. ; Fields , S . Deep mutational scanning: a new style of protein science . Nat Methods 2014 , 11 ( 8 ), 801 – 807 . DOI: 10.1038/nmeth.3027 . OpenUrl CrossRef PubMed Web of Science (8). Starita , L. M. ; Ahituv , N. ; Dunham , M. J. ; Kitzman , J. O. ; Roth , F. P. ; Seelig , G. ; Shendure , J. ; Fowler , D. M . Variant Interpretation: Functional Assays to the Rescue . Am J Hum Genet 2017 , 101 ( 3 ), 315 – 325 . DOI: 10.1016/j.ajhg.2017.07.014 . OpenUrl CrossRef PubMed (9). ↵ Kinney , J. B. ; McCandlish , D. M. Massively Parallel Assays and Quantitative Sequence-Function Relationships . Annu Rev Genomics Hum Genet 2019 , 20 , 99 – 127 . DOI: 10.1146/annurev-genom-083118-014845 . OpenUrl CrossRef PubMed (10). ↵ Li , X. ; Lehner , B . Biophysical ambiguities prevent accurate genetic prediction . Nat Commun 2020 , 11 ( 1 ), 4923 . DOI: 10.1038/s41467-020-18694-0 . OpenUrl CrossRef (11). ↵ Cagiada , M. ; Johansson , K. E. ; Valanciute , A. ; Nielsen , S. V. ; Hartmann-Petersen , R. ; Yang , J. J. ; Fowler , D. M. ; Stein , A. ; Lindorff-Larsen , K . Understanding the Origins of Loss of Protein Function by Analyzing the Effects of Thousands of Variants on Activity and Abundance . Mol Biol Evol 2021 , 38 ( 8 ), 3235 – 3246 . DOI: 10.1093/molbev/msab095 . OpenUrl CrossRef PubMed (12). ↵ Lichtarge , O. ; Bourne , H. R. ; Cohen , F. E . An evolutionary trace method defines binding surfaces common to protein families . J Mol Biol 1996 , 257 ( 2 ), 342 – 358 . DOI: 10.1006/jmbi.1996.0167 . OpenUrl CrossRef PubMed Web of Science (13). Ng , P. C. ; Henikoff , S . SIFT: Predicting amino acid changes that affect protein function . Nucleic Acids Res 2003 , 31 ( 13 ), 3812 – 3814 . DOI: 10.1093/nar/gkg509 . OpenUrl CrossRef PubMed Web of Science (14). Lo , L. W. ; Shakhnovich , E. I. ; Mirny , L. A . Amino acids determining enzyme-substrate specificity in prokaryotic and eukaryotic protein kinases . P Natl Acad Sci USA 2003 , 100 ( 8 ), 4463 – 4468 . DOI: 10.1073/pnas.0737647100 . OpenUrl Abstract / FREE Full Text (15). ↵ Kumar , P. ; Henikoff , S. ; Ng , P. C . Predicting the effects of coding non-synonymous variants on protein function using the SIFT algorithm . Nat Protoc 2009 , 4 ( 7 ), 1073 – 1081 . DOI: 10.1038/nprot.2009.86 . OpenUrl CrossRef PubMed Web of Science 16. ↵ Sugiki , S. ; Niide , T. ; Toya , Y. ; Shimizu , H. Logistic Regression-Guided Identification of Cofactor Specificity-Contributing Residues in Enzyme with Sequence Datasets Partitioned by Catalytic Properties . ACS Synth Biol 2022 , 11 ( 12 ), 3973 - 3985 . DOI: 10.1021/acssynbio.2c00315 . OpenUrl CrossRef PubMed (17). ↵ Orengo , C. A. ; Jones , D. T. ; Thornton , J. M. Protein Superfamilies and Domain Superfolds . Nature 1994 , 372 ( 6507 ), 631 – 634 . DOI : DOI 10.1038/372631a0 . OpenUrl CrossRef PubMed Web of Science (18). ↵ Illergård , K. ; Ardell , D. H. ; Elofison , A . Structure is three to ten times more conserved than sequence-A study of structural response in protein cores . Proteins 2009 , 77 ( 3 ), 499 – 508 . DOI: 10.1002/prot.22458 . OpenUrl CrossRef PubMed Web of Science (19). ↵ Zhang , Y. ; Skolnick , J . Scoring function for automated assessment of protein structure template quality . Proteins 2004 , 57 ( 4 ), 702 – 710 . DOI: 10.1002/prot.20264 . OpenUrl CrossRef PubMed Web of Science (20). ↵ Vajda , T. ; Szabo , T . Specificity of Trypsin and Alpha-Chymotrypsin Towards Neutral Substrates . Acta Biochim Biophys 1976 , 11 ( 4 ), 287 – 294 . OpenUrl (21). ↵ Steitz , T. A. ; Henderson , R. ; Blow , D. M. Structure of Crystalline Alpha-Chymotrypsin .3. Crystallographic Studies of Substrates and Inhibitors Bound to Active Site of Alpha-Chymotrypsin . J Mol Biol 1969 , 46 ( 2 ), 337 -+. DOI: 10.1016/0022-2836(69)90426-4 . OpenUrl CrossRef PubMed Web of Science (22). ↵ Hedstrom , L. Serine protease mechanism and specificity . Chemical Reviews 2002 , 102 ( 12 ), 4501 – 4523 . DOI: 10.1021/cr000033x . OpenUrl CrossRef PubMed Web of Science (23). ↵ Venekei , I. ; Szilagyi , L. ; Graf , L. ; Rutter , W. J . Attempts to convert chymotrypsin to trypsin . Febs Lett 1996 , 379 ( 2 ), 143 – 147 . DOI: 10.1016/0014-5793(95)01484-5 . OpenUrl CrossRef PubMed Web of Science (24). ↵ Graf , L. ; Jancso , A. ; Szilagyi , L. ; Hegyi , G. ; Pinter , K. ; Narayszabo , G. ; Hepp , J. ; Medzihradszky , K. ; Rutter , W. J . Electrostatic Complementarity within the Substrate-Binding Pocket of Trypsin . P Natl Acad Sci USA 1988 , 85 ( 14 ), 4961 – 4965 . DOI: 10.1073/pnas.85.14.4961 . OpenUrl Abstract / FREE Full Text (25). ↵ Hedstrom , L. ; Perona , J. J. ; Rutter , W. J . Converting Trypsin to Chymotrypsin - Residue-172 Is a Substrate-Specificity Determinant . Biochemistry-Us 1994 , 33 ( 29 ), 8757 – 8763 . DOI: 10.1021/bi00195a017 . OpenUrl CrossRef PubMed (26). ↵ Salameh , M. A. ; Soares , A. S. ; Alloy , A. ; Radisky , E. S . Presence versus absence of hydrogen bond donor Tyr-39 influences interactions of cationic trypsin and mesotrypsin with protein protease inhibitors . Protein Sci 2012 , 21 ( 8 ), 1103 – 1112 . DOI: 10.1002/pro.2097 . OpenUrl CrossRef PubMed (27). ↵ Barzu , O. ; Danchin , A . Adenylyl cyclases: a heterogeneous class of ATP-utilizing enzymes . Prog Nucleic Acid Res Mol Biol 1994 , 49 , 241 – 283 . DOI: 10.1016/s0079-6603(08)60052-5 . OpenUrl CrossRef PubMed Web of Science (28). ↵ Sismeiro , O. ; Trotot , P. ; Biville , F. ; Vivares , C. ; Danchin , A . Aeromonas hydrophila adenylyl cyclase 2: a new class of adenylyl cyclases with thermophilic properties and sequence similarities to proteins from hyperthermophilic archaebacteria . J Bacteriol 1998 , 180 ( 13 ), 3339 – 3344 . DOI: 10.1128/JB.180.13.3339-3344.1998 . OpenUrl Abstract / FREE Full Text (29). ↵ Zhang , G. ; Liu , Y. ; Ruoho , A. E. ; Hurley , J. H . Structure of the adenylyl cyclase catalytic core . Nature 1997 , 386 ( 6622 ), 247 – 253 . DOI: 10.1038/386247a0 . OpenUrl CrossRef PubMed Web of Science (30). ↵ Liu , Y. ; Ruoho , A. E. ; Rao , V. D. ; Hurley , J. H . Catalytic mechanism of the adenylyl and guanylyl cyclases: Modeling and mutational analysis . P Natl Acad Sci USA 1997 , 94 ( 25 ), 13414 – 13419 . DOI: 10.1073/pnas.94.25.13414 . OpenUrl Abstract / FREE Full Text (31). ↵ Whisnant , R. E. ; Gilman , A. G. ; Dessauer , C. W . Interaction of the two cytosolic domains of mammalian adenylyl cyclase . P Natl Acad Sci USA 1996 , 93 ( 13 ), 6621 – 6625 . DOI: 10.1073/pnas.93.13.6621 . OpenUrl Abstract / FREE Full Text (32). Tesmer , J. J. ; Sunahara , R. K. ; Gilman , A. G. ; Sprang , S. R . Crystal structure of the catalytic domains of adenylyl cyclase in a complex with Gsalpha.GTPgammaS . Science 1997 , 278 ( 5345 ), 1907 – 1916 . DOI: 10.1126/science.278.5345.1907 . OpenUrl Abstract / FREE Full Text (33). ↵ Tesmer , J. J. ; Sunahara , R. K. ; Johnson , R. A. ; Gosselin , G. ; Gilman , A. G. ; Sprang , S. R . Two-metal-Ion catalysis in adenylyl cyclase . Science 1999 , 285 ( 5428 ), 756 – 760 . DOI: 10.1126/science.285.5428.756 . OpenUrl Abstract / FREE Full Text (34). ↵ Tucker , C. L. ; Hurley , J. H. ; Miller , T. R. ; Hurley , J. B . Two amino acid substitutions convert a guanylyl cyclase, RetGC-1, into an adenylyl cyclase . P Natl Acad Sci USA 1998 , 95 ( 11 ), 5993 - 5997 . DOI: 10.1073/pnas.95.11.5993 . OpenUrl Abstract / FREE Full Text (35). ↵ Sunahara , R. K. ; Beuve , A. ; Tesmer , J. J. G. ; Sprang , S. R. ; Garbers , D. L. ; Gilman , A. G . Exchange of substrate and inhibitor specificities between adenylyl and guanylyl cyclases . J Biol Chem 1998 , 273 ( 26 ), 16332 – 16338 . DOI: 10.1074/jbc.273.26.16332 . OpenUrl Abstract / FREE Full Text (36). ↵ Hughes , A. E. ; Meng , W. H. ; Lotery , A. J. ; Bradley , D. T. A Novel Mutation, V933A, Causes Central Areolar Choroidal Dystrophy . Invest Ophth Vis Sci 2012 , 53 ( 8 ), 4748 – 4753 . DOI: 10.1167/iovs.12-10061 . OpenUrl Abstract / FREE Full Text (37). ↵ Childers , K. C. ; Garcin , E. D . Structure/function of the soluble guanylyl cyclase catalytic domain . Nitric Oxide-Biol Ch 2018 , 77 , 53 – 64 . DOI: 10.1016/j.niox.2018.04.008 . OpenUrl CrossRef PubMed (38). ↵ Linder , J. U . Substrate selection by class III adenylyl cyclases and guanylyl cyclases . Iubmb Life 2005 , 57 ( 12 ), 797 – 803 . DOI: 10.1080/15216540500415636 . OpenUrl CrossRef PubMed Web of Science (39). ↵ Goward , C. R. ; Nicholls , D. J . Malate-Dehydrogenase - a Model for Structure, Evolution, and Catalysis . Protein Sci 1994 , 3 ( 10 ), 1883 – 1888 . DOI: 10.1002/pro.5560031027 . OpenUrl CrossRef PubMed Web of Science (40). ↵ Adeva-Andany , M. ; López-Ojén , M. ; Funcasta-Calderón , R. ; Ameneiros-Rodríguez , E. ; Donapetry-García , C. ; Vila-Altesor , M. ; Rodríguez-Seijas , J . Comprehensive review on lactate metabolism in human health . Mitochondrion 2014 , 17 , 76 – 100 . DOI: 10.1016/j.mito.2014.05.007 . OpenUrl CrossRef PubMed (41). ↵ Wilks , H. M. ; Hart , K. W. ; Feeney , R. ; Dunn , C. R. ; Muirhead , H. ; Chia , W. N. ; Barstow , D. A. ; Atkinson , T. ; Clarke , A. R. ; Holbrook , J. J. A Specific, Highly-Active Malate-Dehydrogenase by Redesign of a Lactate-Dehydrogenase Framework . Science 1988 , 242 ( 4885 ), 1541 – 1544 . DOI: 10.1126/science.3201242 . OpenUrl Abstract / FREE Full Text (42). ↵ Nicholls , D. J. ; Miller , J. ; Scawen , M. D. ; Clarke , A. R. ; Holbrook , J. J. ; Atkinson , T. ; Goward , C. R . The Importance of Arginine-102 for the Substrate-Specificity of Escherichia-Coli Malate-Dehydrogenase . Biochem Bioph Res Co 1992 , 189 ( 2 ), 1057 – 1062 . DOI: 10.1016/0006-291X(92)92311-K . OpenUrl CrossRef PubMed Web of Science (43). ↵ Cendrin , F. ; Chroboczek , J. ; Zaccai , G. ; Eisenberg , H. ; Mevarech , M. Cloning, Sequencing, and Expression in Escherichia-Coli of the Gene Coding for Malate-Dehydrogenase of the Extremely Halophilic Archaebacterium Haloarcula-Marismortui . Biochemistry-Us 1993 , 32 ( 16 ), 4308 – 4313 . DOI: 10.1021/bi00067a020 . OpenUrl CrossRef PubMed (44). ↵ Yin , Y. ; Kirsch , J. F . Identification of functional paralog shift mutations: Conversion of Escherichia coli malate dehydrogenase to a lactate dehydrogenase . P Natl Acad Sci USA 2007 , 104 ( 44 ), 17353 – 17357 . DOI: 10.1073/pnas.0708265104 . OpenUrl Abstract / FREE Full Text (45). ↵ Hensel , R. ; Mayr , U. ; Fujiki , H. ; Kandler , O . Comparative studies of lactate dehydrogenases in lactic acid bacteria. Amino-acid composition of an active-site region and chemical properties of the L-lactate dehydrogenase of Lactobacillus casei, Lactobacillus curvatus, Lactobacillus plantarum, and Lactobacillus acidophilus . Eur J Biochem 1977 , 80 ( 1 ), 83 – 92 . DOI: 10.1111/j.1432-1033.1977.tb11859.x . OpenUrl CrossRef PubMed (46). ↵ Boucher , J. I. ; Jacobowitz , J. R. ; Beckett , B. C. ; Classen , S. ; Theobald , D. L . An atomic-resolution view of neofunctionalization in the evolution of apicomplexan lactate dehydrogenases . Elife 2014 , 3 . DOI: 10.7554/eLife.02304 . OpenUrl CrossRef PubMed (47). ↵ Wirth , J. D. ; Boucher , J. I. ; Jacobowitz , J. R. ; Classen , S. ; Theobald , D. L . Functional and Structural Resilience of the Active Site Loop in the Evolution of Plasmodium Lactate Dehydrogenase . Biochemistry-Us 2018 , 57 ( 45 ), 6434 – 6442 . DOI: 10.1021/acs.biochem.8b00913 . OpenUrl CrossRef (48). ↵ Zeymer , C. ; Hilvert , D . Directed Evolution of Protein Catalysts . Annu Rev Biochem 2018 , 87 , 131 – 157 . DOI: 10.1146/annurev-biochem-062917-012034 . OpenUrl CrossRef PubMed (49). ↵ Tokuriki , N. ; Stricher , F. ; Serrano , L. ; Tawfik , D. S . How Protein Stability and New Functions Trade Off . Plos Comput Biol 2008 , 4 ( 2 ). DOI: ARTN e1000002 10.1371/journal.pcbi.1000002 . OpenUrl CrossRef PubMed (50). ↵ Bigman , L. S. ; Levy , Y . Proteins: molecules defined by their trade-offs . Curr Opin Struc Biol 2020 , 60 , 50 – 56 . DOI: 10.1016/j.sbi.2019.11.005 . OpenUrl CrossRef PubMed (51). ↵ Ebert , M. C. C. J. C. ; Pelletier , J. N . Computational tools for enzyme improvement: why everyone can - and should - use them . Curr Opin Chem Biol 2017 , 37 , 89 – 96 . DOI: 10.1016/j.cbpa.2017.01.021 . OpenUrl CrossRef PubMed (52). ↵ Planas-Iglesias , J. ; Marques , S. M. ; Pinto , G. P. ; Musil , M. ; Stourac , J. ; Damborsky , J. ; Bednar , D . Computational design of enzymes for biotechnological applications . Biotechnol Adv 2021 , 47 . DOI: ARTN 107696 10.1016/j.biotechadv.2021.107696 . OpenUrl CrossRef PubMed (53). ↵ Ashkenazy , H. ; Abadi , S. ; Martz , E. ; Chay , O. ; Mayrose , I. ; Pupko , T. ; Ben-Tal , N. ConSurf 2016: an improved methodology to estimate and visualize evolutionary conservation in macromolecules . Nucleic Acids Research 2016 , 44 ( W1 ), W344 - W350 . DOI: 10.1093/nar/gkw408 . OpenUrl CrossRef PubMed (54). ↵ Khersonsky , O. ; Lipsh , R. ; Avizemer , Z. ; Ashani , Y. ; Goldsmith , M. ; Leader , H. ; Dym , O. ; Rogotner , S. ; Trudeau , D. L. ; Prilusky , J. ;, et al. Automated Design of Efficient and Functionally Diverse Enzyme Repertoires . Mol Cell 2018 , 72 ( 1 ), 178 -+. DOI: 10.1016/j.molcel.2018.08.033 . OpenUrl CrossRef PubMed (55). ↵ Bhattacharya , S. ; Margheritis , E. G. ; Takahashi , K. ; Kulesha , A. ; D’Souza , A. ; Kim , I. ; Yoon , J. H. ; Tame , J. R. H. ; Volkov , A. N. ; Makhlynets , O. V. ;, et al. NMR-guided directed evolution . Nature 2022 , 610 ( 7931 ), 389 -+. DOI: 10.1038/s41586-022-05278-9 . OpenUrl CrossRef PubMed (56). ↵ Gutierrez-Rus , L. I. ; Vos , E. ; Pantoja-Uceda , D. ; Hoffka , G. ; Gutierrez-Cardenas , J. ; Ortega-Muñoz , M. ; Risso , V. A. ; Jimenez , M. A. ; Kamerlin , S. C. L. ; Sanchez-Ruiz , J. M . Enzyme Enhancement Through Computational Stability Design Targeting NMR-Determined Catalytic Hotspots . J Am Chem Soc 2025 . DOI: 10.1021/jacs.4c09428 . OpenUrl CrossRef (57). ↵ Andreeva , A. ; Kulesha , E. ; Gough , J. ; Murzin , A. G . The SCOP database in 2020: expanded classification of representative family and superfamily domains of known protein structures . Nucleic Acids Res 2020 , 48 ( D1 ), D376 – D382 . DOI: 10.1093/nar/gkz1064 . OpenUrl CrossRef PubMed (58). ↵ Sillitoe , I. ; Bordin , N. ; Dawson , N. ; Waman , V. P. ; Ashford , P. ; Scholes , H. M. ; Pang , C. S. M. ; Woodridge , L. ; Rauer , C. ; Sen , N. ;, et al. CATH: increased structural coverage of functional space . Nucleic Acids Res 2021 , 49 ( D1 ), D266 – D273 . DOI: 10.1093/nar/gkaa1079 . OpenUrl CrossRef PubMed (59). ↵ Blum , M. ; Andreeva , A. ; Florentino , L. C. ; Chuguransky , S. R. ; Grego , T. ; Hobbs , E. ; Pinto , B. L. ; Orr , A. ; Paysan-Lafosse , T. ; Ponamareva , I. ;, et al. InterPro: the protein sequence classification resource in 2025 . Nucleic Acids Res 2025 , 53 ( D1 ), D444 – D456 . DOI: 10.1093/nar/gkae1082 From NLM Medline. OpenUrl CrossRef (60). ↵ van Kempen , M. ; Kim , S. S. ; Tumescheit , C. ; Mirdita , M. ; Lee , J. ; Gilchrist , C. L. M. ; Soding , J. ; Steinegger , M . Fast and accurate protein structure search with Foldseek . Nat. Biotechnol . 2024 , 42 ( 2 ), 243 – 246 . DOI: 10.1038/s41587-023-01773-0 . OpenUrl CrossRef PubMed (61). ↵ Holm , L. ; Laiho , A. ; Toronen , P. ; Salgado , M . DALI shines a light on remote homologs: One hundred discoveries . Protein Sci 2023 , 32 ( 1 ), e4519 . DOI: 10.1002/pro.4519 . OpenUrl CrossRef PubMed (62). ↵ Krissinel , E. ; Henrick , K . Secondary-structure matching (SSM), a new tool for fast protein structure alignment in three dimensions . Acta Crystallogr D Biol Crystallogr 2004 , 60 ( Pt 12 Pt 1 ), 2256 - 2268 . DOI: 10.1107/S0907444904026460 . OpenUrl CrossRef PubMed Web of Science View the discussion thread. Back to top Previous Next Posted May 29, 2025. Download PDF Supplementary Material Email Thank you for your interest in spreading the word about bioRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following A Method for Predicting Enzyme Substrate Specificity Residues Using Homologous Sequence Information Message Subject (Your Name) has forwarded a page to you from bioRxiv Message Body (Your Name) thought you would like to see this page from the bioRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share A Method for Predicting Enzyme Substrate Specificity Residues Using Homologous Sequence Information Seiya Mori , Teppei Niide , Yoshihiro Toya , Hiroshi Shimizu bioRxiv 2025.05.25.656053; doi: https://doi.org/10.1101/2025.05.25.656053 Share This Article: Copy Citation Tools A Method for Predicting Enzyme Substrate Specificity Residues Using Homologous Sequence Information Seiya Mori , Teppei Niide , Yoshihiro Toya , Hiroshi Shimizu bioRxiv 2025.05.25.656053; doi: https://doi.org/10.1101/2025.05.25.656053 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Bioengineering Subject Areas All Articles Animal Behavior and Cognition (7640) Biochemistry (17707) Bioengineering (13903) Bioinformatics (41980) Biophysics (21465) Cancer Biology (18613) Cell Biology (25528) Clinical Trials (138) Developmental Biology (13387) Ecology (19920) Epidemiology (2067) Evolutionary Biology (24332) Genetics (15615) Genomics (22519) Immunology (17747) Microbiology (40424) Molecular Biology (17194) Neuroscience (88664) Paleontology (667) Pathology (2839) Pharmacology and Toxicology (4827) Physiology (7650) Plant Biology (15160) Scientific Communication and Education (2046) Synthetic Biology (4302) Systems Biology (9826) Zoology (2271)

Text is read by the "Ask this paper" AI Q&A widget below. Extraction quality varies by source — PMC NXML preserves structure cleanly, OA-HTML may include some navigation residue, and OA-PDF can have broken hyphenation. The publisher copy (via DOI) is the canonical version.

My notes (saved in your browser only)

Ask this paper AI returns verbatim quotes from the full text · source: preprint-html

Answers must be backed by verbatim quotes from this paper's full text. Hallucinated quotes are dropped automatically; if no verbatim passage answers the question, we say so. How this works

Citation neighborhood (no data yet)

We don't have any in-corpus citations linked to this paper yet. This is a recent paper (2025) — citers typically take a year or two to land, and the OpenAlex reference graph may still be filling in.

Source provenance

europepmc
last seen: 2026-05-20T01:45:00.602351+00:00