Deep learning guided design of protease substrates

preprint OA: closed
📄 Open PDF Full text JSON View at publisher

Abstract

Proteases, a class of enzymes that play critical roles in health and disease, exert their function through the cleavage of peptide bonds. Identifying substrates that are efficiently and selectively cleaved by target proteases is essential for studying protease activity and for harnessing their activity in protease-activated diagnostics and therapeutics. However, the vast design space of possible substrates (c.a. 20 10 unique amino acid combinations for a 10-mer peptide) and the limited accessibility of high-throughput activity profiling tools hinder the speed and success of substrate design. We present CleaveNet, an end-to-end AI pipeline for the design of protease substrates. Applied to matrix metalloproteinases, CleaveNet enhances the scale, tunability, and efficiency of substrate design. CleaveNet generates peptide substrates that exhibit sound biophysical properties and capture not only well-established but also novel cleavage motifs. To enable precise control over substrate design, CleaveNet incorporates a conditioning tag that enables generation of peptides guided by a target cleavage profile, enabling targeted design of efficient and selective substrates. CleaveNet-generated substrates were validated experimentally through a large-scale in vitro screen, even in the challenging case of designing highly selective substrates for MMP13. We envision that CleaveNet will accelerate our ability to study and capitalize on protease activity, paving the way for new in silico design tools across enzyme classes.
Full text 97,973 characters · extracted from preprint-html · click to expand
Deep learning guided design of protease substrates | bioRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-M677548'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search New Results Deep learning guided design of protease substrates View ORCID Profile Carmen Martin-Alonso , View ORCID Profile Sarah Alamdari , View ORCID Profile Tahoura S. Samad , View ORCID Profile Kevin K. Yang , View ORCID Profile Sangeeta N. Bhatia , View ORCID Profile Ava P. Amini doi: https://doi.org/10.1101/2025.02.27.640681 Carmen Martin-Alonso 1 Koch Institute for Integrative Cancer Research, Massachusetts Institute of Technology , Cambridge, MA, USA 2 Harvard-MIT Division of Health Sciences and Technology, Massachusetts Institute of Technology , Cambridge, MA, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Carmen Martin-Alonso Sarah Alamdari 3 Microsoft Research , Cambridge, MA, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Sarah Alamdari Tahoura S. Samad 1 Koch Institute for Integrative Cancer Research, Massachusetts Institute of Technology , Cambridge, MA, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Tahoura S. Samad Kevin K. Yang 3 Microsoft Research , Cambridge, MA, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Kevin K. Yang Sangeeta N. Bhatia 1 Koch Institute for Integrative Cancer Research, Massachusetts Institute of Technology , Cambridge, MA, USA 2 Harvard-MIT Division of Health Sciences and Technology, Massachusetts Institute of Technology , Cambridge, MA, USA 4 Institute for Medical Engineering and Science, Massachusetts Institute of Technology , Cambridge, MA, USA 10 Howard Hughes Medical Institute , Cambridge, MA, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Sangeeta N. Bhatia For correspondence: sbhatia{at}mit.edu ava.amini{at}microsoft.com Ava P. Amini 3 Microsoft Research , Cambridge, MA, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Ava P. Amini For correspondence: sbhatia{at}mit.edu ava.amini{at}microsoft.com Abstract Full Text Info/History Metrics Supplementary material Data/Code Preview PDF Abstract Proteases, a class of enzymes that play critical roles in health and disease, exert their function through the cleavage of peptide bonds. Identifying substrates that are efficiently and selectively cleaved by target proteases is essential for studying protease activity and for harnessing their activity in protease-activated diagnostics and therapeutics. However, the vast design space of possible substrates (c.a. 20 10 unique amino acid combinations for a 10-mer peptide) and the limited accessibility of high-throughput activity profiling tools hinder the speed and success of substrate design. We present CleaveNet, an end-to-end AI pipeline for the design of protease substrates. Applied to matrix metalloproteinases, CleaveNet enhances the scale, tunability, and efficiency of substrate design. CleaveNet generates peptide substrates that exhibit sound biophysical properties and capture not only well-established but also novel cleavage motifs. To enable precise control over substrate design, CleaveNet incorporates a conditioning tag that enables generation of peptides guided by a target cleavage profile, enabling targeted design of efficient and selective substrates. CleaveNet-generated substrates were validated experimentally through a large-scale in vitro screen, even in the challenging case of designing highly selective substrates for MMP13. We envision that CleaveNet will accelerate our ability to study and capitalize on protease activity, paving the way for new in silico design tools across enzyme classes. Introduction Proteases are a diverse class of enzymes that play critical roles in both health and disease, including in coagulation, tissue remodeling, and cancer ( 1 – 3 ). Because proteases exert their function through the cleavage of peptide bonds, studying and capitalizing on protease activity relies on the identification of peptide substrates that can be used as molecular probes ( 4 , 5 ), peptide-based inhibitors ( 6 – 8 ), or conditionally-activated triggers in engineered diagnostics ( 9 – 11 ) and therapeutics ( 12 – 15 ). Yet, designing substrates that are both sensitive and specific to target proteases remains a significant challenge rooted in the complex biochemistry of proteases ( 16 , 17 ) and their substrates ( 18 – 20 ). To achieve such diverse functions, proteases have evolved to display a broad range of cleavage specificities, mediated by interactions between a protease active site and a natural or synthetic substrate, often up to 10 amino acids long ( 21 , 22 ). Substrate design is thus a combinatorial problem: for a 10-amino acid peptide, approximately 20 10 (c.a. 10 13 ) sequences are possible, even without accounting for non-natural amino acids. Additionally, given that functionally-related proteases often evolved from a common ancestor, they tend to share overlapping substrate sets, making it challenging to identify substrates that are truly selective for a target protease. Given that all possible protease-substrate pairs have not yet been comprehensively profiled, substrate design typically involves two steps. The first is the nomination of candidate substrate sequences from a large, combinatorial space of possible sequences; the second is the selection of substrates as a function of their cleavage profiles after screening against proteases of interest. The most common approach for substrate nomination entails surveying the literature for existing substrates, informed by cleavage sites in naturally-occurring proteins ( 23 ). However, this search is inefficient and seldom results in synthetic substrates with desired cleavage profiles ( 19 ). Alternatively, rational design leveraging expert knowledge of chemical biology can optimize synthetic substrates for specific targets, but this process is inherently resource intensive, low throughput, and bespoke to each protease of interest ( 24 ). To circumvent bottlenecks in substrate nomination, recent technological advances have significantly increased the throughput of the selection step ( 25 ). Large-scale chemical or displaybased libraries, such as mRNA, ribosome, and phage display, have enabled high-throughput in vitro screening of 10 5 -10 13 unique substrates against proteases of interest ( 26 – 28 ). However, these methods are sophisticated and costly, and thus not broadly usable. In contrast to highthroughput experimental approaches, computational methods offer the possibility of accelerating substrate selection through virtual screening. Methods leveraging inferred substitution matrices ( 29 – 31 ) or supervised learning ( 32 – 35 ) have focused on identifying substrate cleavage sites to predict whether a substrate will be recognized by a target protease. However, these methods return a binary cleavage prediction (i.e. cleaved vs. not cleaved), making it challenging to rank-order substrates by their cleavage efficiencies against a target protease, which is critical for engineering applications and the study of nuanced protease specificities. Recent deep learning models that extract patterns from amino acid sequence data have enabled the accurate prediction of protein sequence-function relationships ( 36 ) and the generative design of new proteins ( 37 – 39 ), but remain challenging to apply to enzymes and have been underexplored in their application to protease substrate design ( 40 ). New in silico methods that learn from large-scale screening data to design new protease substrates with pre-defined cleavage preferences would increase the scale, diversity, and tunability of substrate design. We present CleaveNet, an AI-based pipeline for the end-to-end design of protease substrates ( Fig. 1 ). CleaveNet consists of a predictive model, the CleaveNet Predictor, which assigns cleavage scores across proteases of interest ( Fig. 1A ), and a generative model, the CleaveNet Generator, which produces new peptide sequences optionally-conditioned on a desired protease cleavage profile ( Fig. 1B ). We validate our approach on the widely-studied family of matrix metalloproteinases (MMPs), which play important roles in health and a variety of diseases ( 3 , 41 ), by training CleaveNet on a dataset of mRNA-display peptides screened against 18 MMPs ( 42 ) and evaluating it both in silico and in vitro . We demonstrate that the CleaveNet Predictor achieves accurate and robust prediction of cleavage scores on an independent, experimentally-distinct test set and show that the CleaveNet Generator produces sequences that capture relevant cleavage profiles across MMPs and that exhibit both known and novel cleav-age motifs. We use the end-to-end CleaveNet pipeline to design novel synthetic substrates for MMP13, a collagenase that plays a critical role in cancer metastasis, wound healing, and osteoarthritis ( 43 , 44 ), optimizing for both efficiency and selectivity ( Fig. 1C ), and experimentally assess their cleavage profiles through a large-scale in vitro screen of 95 substrates against 12 MMPs ( Fig. 1D ). All CleaveNet-designed substrates are cleavable by MMP13, and, strikingly, some conditionally-designed selective substrates are uniquely cleaved by MMP13. Our experimentally-validated, AI-based design of novel protease substrates uncovers underappreciated determinants of MMP efficiency and selectivity, opening the door to overcoming the well-known promiscuity of MMPs. We envision that CleaveNet will advance the tunability, speed, and accuracy of protease substrate design across diverse applications. Download figure Open in new tab Figure 1: A deep learning approach for protease substrate design. ( A ) For a given peptide, the CleaveNet Predictor predicts cleavage scores across 18 matrix metalloproteinases (MMPs). ( B ) The CleaveNet Generator can generate new substrates unconditionally (left) or conditionally, guided by a desired protease cleavage profile (right).( C ) To design candidate substrates for a target protease, new peptide sequences are generated by the CleaveNet Generator and prioritized by the cleavage scores from the CleaveNet Predictor. ( D ) An in vitro screen of 95 fluorogenic substrates against 12 recombinant MMPs was performed to validate the cleavage profiles of CleaveNet-designed substrates. Results CleaveNet predicts cleavage efficiencies of peptides by MMPs We trained our CleaveNet models on a publicly available dataset of c.a. 18,500 mRNA-displayed synthetic substrates and their continuous cleavage efficiencies across 18 MMPs, quantified as a normalized Z-score ( Z s m ) representing the strength of cleavage of substrate s by protease m ( 42 ). Although sequences with higher Z-scores can generally be expected to be cleaved with higher likelihoods, we note that this readout does not directly distinguish cleaved from non-cleaved substrates. Since a critical component of substrate design is the down-selection of candidate substrates from a large combinatorial space, we first developed a predictive deep learning model, the CleaveNet Predictor, to score and virtually screen sequences in silico . We formulated the task of the CleaveNet Predictor as a multi-output sequence-to-function regression problem ( Fig. 2A ). Given an input amino acid sequence s , the model predicts continuous cleavage values Z ^ sm for 18 MMPs and their associated uncertainty scores σ s m , quantified by training an ensemble of 5 predictor models over the mRNA-display dataset. Continuous Z ^ sm values can optionally be converted to a binary classification output (i.e., cleaved vs. not cleaved) based on a desired threshold Z t . Download figure Open in new tab Figure 2: CleaveNet accurately predicts cleavage efficiencies of synthetic peptides against MMPs. ( A ) Multi-output cleavage score regression and uncertainty estimation by the CleaveNet Predictor. Two biochemically-distinct sets of synthetic peptides, from mRNA display ( n =3,717 peptides) and fluorometric ( n =71 peptides) activity assays, are used as test sets for evaluation across different MMPs. IceLogos are normalized by natural amino acid frequencies (see Methods). Individual amino acids are colored by the chemical properties of their side chain: hydrophobic aromatic (lavender), hydrophobic (blue), hydrophilic (yellow), acidic (orange), and basic (red). ( B ) Mean absolute error (MAE) between true and predicted MMP-cleavage scores across the test sets, comparing the transformer (light blue) and LSTM (dark blue) models (mean ± s.d., n =5 independent training trials; ns p > 0.05, one-way ANOVA). ( C ) Correlation of true (x-axis) and transformer-predicted (y-axis) MMP13-cleavage scores on the mRNA-display test set ( n =3,717, Pearson’s r =0.80). ( D ) Receiver operating characteristic (ROC) evaluation of the CleaveNet transformer predictor over the mRNA-display test set for MMP13. Individual lines represent performance for different true cleavage score thresholds Z t , with area under the curve (AUC) provided. ( E ) Correlation of true (x-axis) and transformer-predicted (y-axis) MMP13-cleavage scores on the fluorescence test set ( n =71, Pearson’s r =0.80). ( F-G ) Fluorescence test substrates rank-ordered by predicted MMP13-cleavage score and colored by a binary indicator of true MMP13 cleavage (red, cleaved; blue, not cleaved; F ) and by model uncertainty ( G ). We investigated two model architectures common for sequence modeling – a recurrent bidirectional LSTM architecture ( 45 ) and a transformer architecture ( 46 ). LSTMs learn patterns sequentially, while transformers learn patterns by looking at all elements of a sequence simultaneously and are the dominant architecture for protein language modeling ( 47 , 48 ). To assess model performance and generalizability, we evaluated prediction performance across two datasets: sequences obtained from a 20% random split of the training dataset never seen during training (mRNA-display test) and sequences obtained from an independent set of 71 FRET-paired sequences that were previously screened against 7 recombinant MMPs in vitro (fluorescence test) ( 49 ). Substrates across the two test sets differed in length (10-mers vs. 7- to 14-mers for mRNA-display vs. fluorescence, respectively) and amino acid composition ( Fig. 2A ), and were assayed by very distinct experimental methods. Both the LSTM and transformer models displayed similar predictive performance on the test sets, as supported by their comparable mean absolute errors (MAE) between the true and predicted cleavage scores ( Fig. 2B , Table S1 ). With the goal of optimizing performance for substrates in the larger and more diverse mRNA-display dataset, we used the transformer-based model as the CleaveNet Predictor for subsequent analyses. The models performed better on some MMPs than others, with better performance achieved for MMPs that had either a higher fraction of training examples with high Z-scores or that displayed a wider dynamic range of Z-scores between cleaved and uncleaved sequences ( Fig. S1 ). These results suggest that our models may be more effective at learning cleavage patterns for proteases with more cleaved training examples or more specific cleavage patterns, respectively. We observed strong correspondence between predicted Z ^ sm and true Z s m for MMP13 (Pear-son’s r = 0.80; Fig. 2C ) and for other MMPs in the mRNA-display test set ( Fig. S2 ). To assess model performance for classification, we assigned “true cleaved” labels to sequences at varying Z-score thresholds: Z s m ≥ Z t , where { Z t = 0, 1.0, 1.5, 2.0, 2.5}. We then calculated receiveroperator curves given a predicted Z ^ sm , over this range of possible cleavage thresholds. CleaveNet predictions were robust over the range of Z t evaluated, and classification performance improved as the Z-threshold increased, reaching an AUC value of 0.98 at Z t = 2.5 ( Fig. 2D , Fig. S3 - S4 ); a threshold of 2.5 was previously reported to be broadly consistent with confident cleavage across all MMPs for this dataset ( 42 ). To assess the robustness and generalizability of the models, we next evaluated their performance on the biochemically distinct fluorescence set ( Fig. S5 - S8 ). The Z-scores predicted by the transformer model displayed a strong positive linear correlation with true Z-scores (m = MMP13, Pearson’s r = 0.80, Fig. 2E ), especially for Z s 13 > 0 that are more likely to correspond to cleaved substrates. Given that Z s m values are dependent on the nature of the experimental assay and cleavage range of the substrate library being profiled, which differ be-tween the mRNA-display and fluorescence sets, the relative rank of the predicted Z-scores is a more appropriate metric to assess performance on the fluorescence test set. Independent of their exact value, we expect low Z-scores to be non-cleaved and higher scores to be cleaved. The CleaveNet Predictor consistently predicted lower Z ^ sm for sequences that were not cleaved (blue) relative to substrates that were cleaved (red) ( Fig. 2F ), and the degree of reliability of these predictions was reflected in the model’s uncertainty for each individual prediction ( Fig. 2G , Fig. S9 , S10 ). Taken together, these results support the accuracy and robustness of the CleaveNet Predictor in producing MMP cleavage scores for synthetic substrates, even across biochemically distinct assay setups. The CleaveNet Generator yields plausible and novel MMP substrates Having validated the CleaveNet Predictor and with the aim of automating the substrate nomination step, we next sought to develop a generative model – the CleaveNet Generator – that would learn pan-MMP cleavage specificities and enable unconditional sampling of diverse MMP-cleavable substrates without further input. Ideally, generated sequences should capture the amino acid distribution and the biophysical and functional properties of training sequences, while still being diverse from each other and distinct from the train set. To generate sequences relevant to MMP activity, we trained an autoregressive transformer model (see Methods for details) on the training split of the mRNA-display dataset. As a baseline, we generated sequences by sampling randomly and independently from the position-wise distribution of amino acids across all training sequences (“site-independent baseline”, Fig. 3A ). Download figure Open in new tab Figure 3: CleaveNet generates novel, biophysically plausible MMP substrates. ( A ) Unconditional generation from the CleaveNet Generator. Baseline sequences are produced via position-wise sampling from the amino acid distribution of peptides in the mRNA display train set. ( B ) IceLogo visualization of position-specific amino acid composition of CleaveNet unconditional generations ( n = 4, 000), with the Kullback-Leibler (KL) divergence between the generated and mRNA-display test distributions denoted (normalized by natural amino acid frequencies). Position-wise distributions are included in Fig. S11 . ( C ) Probability density functions of in silico -computed biophysical properties of generated (green, n = 4, 000) and mRNA-display test (blue, n = 3, 717) sequences. ( D ) CleaveNet was used to generate 20,000 candidate substrates and score their cleavage profiles across 18 MMPs. ( E ) Distributions of predicted cleavage scores, across 18 MMPs, for CleaveNet unconditional generations (green, n = 4, 000), site-independent baseline sequences (yellow, n = 4, 000), and sequences from the mRNA-display test set (blue, n = 3, 717). Box plots show median, interquartile range, and minimum to maximum. ( F ) Cumulative distribution of the frequency of unique k -mers present in CleaveNet generations (green, n = 19, 905), site-independent baseline sequences (yellow, n = 20, 000), and mRNA-display test sequences (blue, n = 18, 583) as a function of the total number of k -mers considered. The top-occurring k -mers in each set are provided (right). We unconditionally generated 20,000 sequences using the CleaveNet Generator and the site-independent baseline and compared them to the mRNA-display test sequences ( Fig. 3B , Fig. S11 ). As expected for a method that considers each site independently, the baseline closely matched the test set position-wise amino acid distributions (average KL divergence = 0.237; Fig. S11B ). The CleaveNet Generator was effective at capturing the canonical MMP motif Proline-X-X-Hy, where “Hy” is a hydrophobic residue, and exceeded the site-independent baseline at recapitulating the amino acid distributions in positions P3 to P2’, which are most relevant for cleavage (position-wise KL divergence averaged over P3 to P2’ = 0.25 for CleaveNet generated vs. 0.404 for site-independent baseline; Fig. 3B , Fig. S11 ). The CleaveNet-generated and the site-independent baseline sequences exhibited biophysical properties consistent with those of the mRNA-display test dataset ( Fig. 3C , Fig. S12 ), supporting their quality and plausibility. Scoring each set of sequences with the CleaveNet Predictor ( Fig. 3D ) revealed that, across MMPs, the cleavage scores for the CleaveNet-generated sequences closely matched the distribution of scores for mRNA-display sequences, while the sequences from the site-independent baseline displayed significantly lower values across all MMPs ( Fig. 3E ). To further evaluate the plausibility of generated sequences, we inspected the cumulative density function (CDF) of unique k -mers in each set of generated or mRNA-display sequences ( Fig. 3F ). Each of the mRNA-display, CleaveNet-generated, and site-independent baseline sequence sets effectively sampled the true 3-mer space, as demonstrated by saturation of the CDF by 8,000 unique k -mers. However, as the space of possible unique k -mers became increasingly large for 4-, 5-, and 6-mers, the site-independent baseline covered the space near linearly, suggesting generation closer to random across the space of all possible k -mers. In contrast, the distribution of unique k -mers was more consistent between the mRNA-display and CleaveNet-generated sequences across all k -mer lengths, indicating that CleaveNet gener-ations exhibit similar k-mer diversity to sequences from the mRNA display set. Inspection of the top occurring 5- and 6-mers in each set of sequences revealed that those from CleaveNet were closer to canonical MMP motifs than 5-mers from the site-independent baseline ( Fig. 3F ). Further, only 95 of the 20,000 sequences generated by CleaveNet were exact matches to train set sequences, indicating that CleaveNet was not simply memorizing training data. Together, these results demonstrate that the CleaveNet Generator can produce biophysically-plausible sequences that exhibit expected cleavage profiles across MMPs and possess diverse, novel motifs while retaining sequence identities compatible with MMP cleavage. Generated sequences recapitulate biologically relevant cleavage patterns across MMP functional classes To delve deeper into the biological validity of the CleaveNet pipeline, we virtually screened the 20,000 unconditional generations and analyzed the sequence determinants and activity profiles of top-scoring substrates across individual MMPs. First, we inspected position-wise determinants of specificity by plotting IceLogos, which reflect position-specific amino acid compositions, for each set ( 50 ). Despite containing no overlapping sequences, the amino acid profiles of top-scoring generated sequences closely matched those in the mRNA display test set for individual MMPs ( Fig. 4A ). Shared trends also emerged across MMPs within the same subclass (i.e., collagenases, gelatinases, MT1-MMP, MT2-MMP, and stromelysins). Although the PXXL motif was prominent across all MMPs, less expected amino acid preferences also emerged in the CleaveNet-generated sequences. Notably, a preference for methionine at P4 was identified (fold-increase above natural frequency of: 11.4-fold for gelatinases, 10.9-fold for collagenases, 14.0-fold for MT1-MMP, 9.2-fold MT2-MMP and 12-2 fold stromelysins), suggesting a previously unexplored relevance of this amino acid to the cleavage efficiency of short peptides for the MMP catalytic class. Download figure Open in new tab Figure 4: Generated sequences recapitulate biologically-relevant cleavage patterns across MMP functional classes. ( A ) IceLogos for individual MMPs, grouped by relevant subclasses, are shown for the top 100 Z-scoring sequences in each set: CleaveNet-generated (green boxes) and mRNA-display training (blue boxes). The IceLogos are normalized by amino acid frequencies from the mRNA-display set.. ( B ) Histograms of 3-mer frequencies shared across the top 100 Z-scoring MMP13 sequences in each of the CleaveNet generated (top), mRNA-display training (middle), and site-independent baseline (bottom) sets. The frequencies of the top-5 occurring 3-mers in each set are summarized as tables on the right, with shared 3-mers between the CleaveNet generated and mRNA-display sets bolded. ( C ) Heat map, colored by CleaveNetpredicted cleavage score and annotated by hierarchical clustering, of the 25-top Z-scoring substrates per MMP, including the similarity cutoff and subsequent clustering of proteases into 5 groups with shared phylogeny. Closer inspection of specificity determinants for the gelatinases MMP2 and MMP9, which have well-characterized cleavage preferences, revealed trends consistent to those previously reported ( 51 , 52 ) ( Fig. 4A ). For instance, we noted a higher prevalence of proline at P3 (99% of sequences) in the top Z-scoring generated sequences for gelatinases over other MMP classes, a preference for small amino acids such as glycine and alanine over aromatic and larger aliphatic residues at P1 and P3’, and a preference towards the positively charged amino acid arginine (27.5% of sequences) at P2’. We next evaluated our pipeline’s ability to learn complex inter-amino acid relationships critical to substrate recognition, such as subsite cooperativity. Such interactions between multiple subsites help enable cleavage but cannot be captured by position-wise IceLogos. To understand whether our pipeline was capturing such effects, we inspected the frequency and the identity of 3-mers shared across the 50 top-ranking MMP13 substrates in the CleaveNet-generated, mRNA display, and site-independent baseline sets ( Fig. 4B ). We observed significantly higher frequencies of some 3-mers over others for the generated and the mRNA display sets, suggesting a strong advantage for cleavage when specific amino acids are positioned contiguously. The canonical motif PLG emerged as the top occurring 3-mer for both datasets, and the top-4 occurring 3-mers were shared between the generated and mRNA-display sets, suggesting a high degree of meaningful conservation in generated sequences. In contrast, the site-independent baseline set displayed little preference towards any 3-mer and did not enrich for the canonical PLG motif. These results lend further support to the CleaveNet generation strategy and suggest that CleaveNet can learn complex inter-amino acid relationships that are otherwise overlooked by simply sampling from position-wise amino acid distributions independently. As a final measure of biological validity, we used the CleaveNet Predictor to score the cleavage profiles of generated sequences, and then clustered the activity profiles of the top 25 Z-scoring sequences for each MMP ( Fig. 4C ). The cleavage profiles for all MMPs, except MMP11 and MMP12, clustered based on the phylogenetic distance of their catalytic domains ( Fig. S13 ) ( 53 ). Group 1 consisted of the 4 transmembrane-type MT-MMP family members (MMP14, MMP15, MMP16, MMP24); group 2 was collagenases (MMP1 and MMP8); group 3 included gelatinases (MMP2 and MMP9), the collagenase MMP13, and the stromelysin MMP3; group 4 was the glycosylphosphatidylinositol (GPI)-anchored MT-MMPs (MMP17 and MMP25); and group 5 included the non-furin regulated MMPs MMP7, MMP10 and MMP20. Altogether, these results reinforce the biological validity of CleaveNet sequences by demonstrating their alignment with well-established MMP cleavage preferences, their ability to capture subsite cooperativity patterns, and the clustering of their activity profiles according to the phylogenetic relationships of MMPs. Moreover, CleaveNet generated comparable – but nonoverlapping – data to those obtained through the time and resource-intensive mRNA display screen used to collect the training data, providing a rapid in silico method to increase the scale and diversity of sequences that can be explored for tasks of interest. Generated substrates with high predicted Z-scores for MMP13 are cleaved by MMP13 in vitro To further validate the CleaveNet pipeline, we used CleaveNet to nominate efficient substrates for MMP13 and tested them experimentally through an in vitro cleavage assay. To this end, we virtually screened the 20,000 unconditionally generated substrates with the CleaveNet Predictor and rank-ordered them by their uncertainty-aware predicted cleavage scores for MMP13 ( Fig. 5A ). To maximize the likelihood of identifying novel substrates, we selected 24 substrates with the highest predicted MMP13 cleavage scores and non-overlapping 5-mers. As controls for this CleaveNet-guided substrate design pipeline, we included site-independent substrates that were obtained using the site-independent baseline alone or after rank-ordering with the CleaveNet Predictor. The latter group was included given the poor performance of the siteindependent baseline alone in previous analyses and to assess the value of in silico screening with the CleaveNet Predictor on substrates not designed by the CleaveNet Generator. The 5 top MMP13 substrates and 5 uncleaved substrates from the mRNA display training set were included as positive and negative controls, respectively. We synthesized these substrates as fluorogenic Förster resonance energy transfer (FRET)-probes and screened them in vitro against recombinant MMP13 by measuring increases in fluorescence indicative of cleavage over time ( Fig. 5A ). Fold changes in fluorescence at end point were calculated as a proxy for cleavage activity for each substrate. Download figure Open in new tab Figure 5: CleaveNet-designed substrates are efficiently cleaved by MMP13 in vitro . ( A ) Candidate sequences were generated by the CleaveNet Generator, ranked by uncertainty-aware MMP13-cleavage scores from the CleaveNet Predictor, and filtered based on redundant 5-mers. Nominated sequences were synthesized as FRET-substrates and screened in vitro against recombinant MMP13. ( B ) In vitro substrate cleavage efficiencies of CleaveNet-generated sequences as described in (A)( n =24; green), sequences from the mRNA-display dataset ( n =5 each; Negative controls, grey; Positive controls, blue), and sequences from the site-independent baselines ( n =8 each; site-independent alone, yellow; site-independent + CleaveNet Predictor, red). ** p < 0.01; Kruskal-Wallis test. ( C ) From top to bottom: IceLogo profiles, across all groups and displaying raw frequencies, of the top-7 substrates most cleaved by MMP13, all MMP13-cleavable substrates, and substrates not cleaved by MMP13. ( D ) Cleavage scores, from the CleaveNet Predictor, across all 18 MMPs for the two most efficient MMP13 substrates identified, DL73 (top; site-independent + CleaveNet Predictor group) and DL6 (bottom; CleaveNet-generated group). For ease of interpretation, fluorescence fold changes were transformed to cleavage efficiencies, with a value of 0 for substrates that were not cleaved, a value of 1 for the substrate with the highest cleavage rate FC max , and a fractional cleavage efficiency between 0 and 1 for all other cleaved substrates. We first assessed the hit rate, defined as the fraction of substrates with a measurable cleavage in vitro , across the different sequence groups. In agreement with our CleaveNet-predicted cleavage scores, all CleaveNet-generated substrates were cleaved (100%, 24/24), whereas only one of the eight substrates nominated with the site-independent baseline was cleaved (12.5%, 1/8) ( Fig. 5B ). Complementing the site-independent baseline with the CleaveNet Predictor enriched for sequences with high predicted cleavage scores and increased the hit rate to 100% (8/8) ( Fig. 5B ). We next assessed the absolute values of MMP13 cleavage efficiencies across the groups ( Table S3 ). In addition to significantly outperforming the site-independent baseline ( p < 0.01), both CleaveNet-guided approaches yielded substrates with cleavage efficiencies superior to those of positive controls from the training set (median cleavage efficiencies of 0.22, 0.37, and 0.64 for the mRNA-displayed, CleaveNet-generated, and Site-independent + CleaveNet Predictor groups, respectively; Fig. 5B , Table S3 ). Across its two design groups, CleaveNet produced 18 substrates with MMP13 cleavage efficiencies higher than that of the most highly cleaved training substrate, DL57 (RMPLGLRAPA, efficiency 0.46). To identify sequences that were cleaved preferentially by MMP13, we compared the IceL-ogos for all sequences cleaved by MMP13 ( n =38; 24 CleaveNet-generated, 1 site-independent baseline, 8 site-independent baseline + CleaveNet Predictor, and 5 training sequences) with the subset that displayed efficiencies greater than 0.6 ( n =7; 3 CleaveNet-generated, 4 siteindependent baseline + CleaveNet Predictor) ( Fig. 5C ). Whereas MMP13 was able to cleave 38/50 substrates and thus tolerated a large diversity of sequences, those that were most efficiently cleaved displayed more constrained sequence preferences. The top 7 best cleaved substrates shared PL at positions P3-P2, suggesting a strong bias of MMP13 for leucine at P2, which may have been previously overlooked ( 51 ). The P1 and P1’ sites were restricted to hydrophobic amino acids, as is canonical for MMPs, but displayed greater diversity than P3 and P2, with P1 allowing glycine, alanine, or proline and P1’ allowing methionine, leucine, or isoleucine. Albeit less dominant, a preference for alanine at P4 and for alanine or aspartic acid at P3’ also emerged in the 7 best-cleaved substrates. The 4-mers shared across the top 7 sequences – namely APLG, LGLT, PLAM, PLGI, and PLGL – mostly overlapped with the P4-P2’ region, but the 4-mer TASG overlapped with the P2’-P5’ region. The top MMP13 substrate DL73 (LFPLAMMDMT) exhibited an unconventional MMP cleavage sequence, with phenylalanine at P4 and methionines at P1’ and P2’. The two next best cleaved sequences, DL6 and DL50, shared the 6-mer APLGLT. These observations of the CleaveNet-guided substrates indicate that positions beyond the canonical P3-P1’ region may still be important contributors to MMP13 efficiency. Performing a similar analysis with the training dataset revealed distinct and less conserved amino acid preferences at all positions, with the exception of proline at P3 ( Fig. S14 ). These findings suggest that CleaveNet-designed substrates reveal novel sequence determinants of MMP13 efficiency beyond those that could be inferred from the training set. Curious about the novelty of these 7 highest MMP13 scoring substrates, we identified the sequences in training with which they shared the longest k -mers and their corresponding cleavage Z-scores ( Table S4 ). Sequences DL5, DL6, and DL50 shared 5-, 6-, and 8-mers with at least one training sequence with high Z-scores, suggesting a degree of overlap with training sequences. In contrast, DL73 (the top MMP13 substrate) and DL42 shared at most a single 4-mer with a small subset of training sequences exhibiting variable Z-scores. Further, the sequences with the longest k -mer that matched DL3 and DL49 were not cleaved in the training set, suggesting high novelty for some of the CleaveNet-guided sequences. Performing in silico screening of the top 2 MMP13 sequences across other MMPs using the CleaveNet Predictor suggested that, albeit MMP13-cleavable, these substrates could also be susceptible to cleavage by other MMPs ( Fig. 5D ), making them inappropriate for applications where MMP13 selectivity is required. Altogether, these results suggest that substrates designed for efficient MMP13 cleavage using CleaveNet-guided strategies are cleavable by recombinant MMP13 in vitro , overperform the site-independent baseline, and exhibit higher cleavage efficiencies and distinct motifs than top-scoring substrates in the mRNA-display training set. However, these substrates are predicted to be cleaved promiscuously across multiple MMPs. To enrich for sequences with target cleavage profiles, such as high selectivity for MMP13, we next developed a guided generation approach. Conditional generation with CleaveNet enables in silico design of selective MMP substrates Identifying substrates that meet a predefined cleavage profile can be achieved by generating sequences unconditionally and filtering them based on their predicted cleavage profiles across MMPs. For instance, to design for substrates with high selectivity, sequences may be filtered by a selectivity metric calculated from predicted cleavage scores. However, this approach is untargeted (i.e., it may require a large number of unconditional generations to arrive at a selective sequence) and would be hard to generalize to nuanced cleavage patterns (e.g., if designing for substrates that are cleaved by multiple proteases of interest but not by others). To overcome these challenges, the CleaveNet Generator was trained with conditioning tags specifying target cleavage profiles across MMPs to guide generations to match desired profiles ( Fig. 6A ). Download figure Open in new tab Figure 6: Conditional generation enables targeted design of selective substrate sequences. ( A ) The CleaveNet Generator can generate candidate substrates conditioned on a desired cleavage profile. Conditional generation from CleaveNet is compared to position-wise sampling from the amino acid distribution of the top 50 sequences that have the desired cleavage pro-file. ( B ) Predicted MMP13-selectivity scores, S ^ s 13 , for sequences conditionally generated for MMP13 selectivity ( n =20,000, dark green), relative to sequences from the mRNA-display dataset ( n =18,583, blue), unconditional generations ( n =19,905, light green), and unconditional and conditional site-independent baselines ( n =20,000 each, light and dark yellow, respectively). **** p < 0.0001; Kruskal-Wallis test. ( C ) IceLogo profile of the amino acid distributions for the CleaveNet-generated MMP13-selective sequences ( n =20,000 IceLogo normalized by natural amino acid frequencies). ( D ) Breakdown of shared and unique k -mers in a combined pool of mRNA-display sequences ( n =18,583, blue) and sequences conditionally generated by CleaveNet for MMP13 selectivity ( n =20,000, dark green). k -mers shared between the two sources are denoted in grey, k -mers only observed in the mRNA-display dataset are shown in blue, and k -mers only observed in the CleaveNet conditional generations are shown in dark green. The top three most frequently occurring k -mers in each subset are denoted to the right of each bar; the percentage covered by a given subset is denoted to the left. To test this capability, we sought to use CleaveNet to design substrates selective for MMP13. We used the CleaveNet Generator to produce 20k sequences unconditionally (as previously described, “unconditional generated”), or conditionally given a conditioning tag specifying high MMP13-selectivity (“conditional generated”). As a baseline, 20k sequences were also sampled from the amino acid distribution of all training sequences (“unconditional site-independent baseline”) versus the distribution of the top 50 MMP13-selective sequences (“conditional siteindependent baseline”). Here, the selectivity score refers to the cleavage of a sequence by one MMP relative to the average cleavage across all other MMPs, as previously described ( 42 ) (see Methods). Both sets of CleaveNet-guided sequences displayed significantly higher selectivity scores than the mRNA-display sequences ( p < 0.0001; Fig. 6B ). However, conditionally generated sequences achieved even greater selectivity than unconditionally generated ones ( p < 0.0001; 5.5-fold higher median selectivity, Fig. 6B ), highlighting the effectiveness of the conditional generation strategy. Accordingly, the MMP13-selective conditionally generated substrates ( Fig. 6C ) were quite distinct from those in the mRNA display train dataset ( Fig. 3A , IceLogo). To investigate the novelty of the conditionally generated sequences, we calculated the fraction of shared and unique k -mers in the combined pool of mRNA-display sequences and the sequences conditionally generated for MMP13 selectivity ( Fig. 6D ). While most 3-mers were shared, the fraction of shared k -mers decreased with k -mer length, leading to almost mutually exclusive sets of 6-mers between the two datasets. This observation indicates a divergence of CleaveNet’s conditional generations from training sequences and raises the prospect of identifying novel motifs dictating MMP13-specificity. Together these in silico results suggest that the CleaveNet conditional generation strategy may offer a powerful tool to guide generations towards novel sequences with desired cleavage profiles. CleaveNet designs and nominates substrates selectively cleaved by MMP13 We next sought to validate experimentally that conditionally generated CleaveNet sequences indeed exhibited superior selectivity over other design strategies. To this end, we performed an in vitro screen of 95 FRET-paired, fluorogenic substrates ( n =40 selected for MMP13 efficiency, n =40 designed for MMP13 selectivity, n =15 total sequences from the mRNA-display set as controls for top efficient, top selective, and negative cleavage) against 12 recombinant proteases spanning all activity clusters in Figure 4C ( Fig. 7A , Fig. S15 ). Download figure Open in new tab Figure 7: CleaveNet designs novel substrates that are selectively cleaved by MMP13 in vitro . ( A ) Schematic overview of nomination strategies for substrates included in the in vitro screen ( n =95 substrates total). (i) Substrates were selected from unconditional generations for efficient cleavage by MMP13 (top) or conditionally designed to be cleaved selectively by MMP13 (bottom). In addition to CleaveNet-generated substrates (green, n =24 per group), appropriate baselines consisting of site-independent baseline alone (yellow, n =8 per group) and site-independent + CleaveNet Predictor (burgundy, n =8 per group) were added. Controls from the mRNA-display training set corresponding to substrates that were efficiently, selectively, or not cleaved by MMP13 (light blue, dark blue, and grey respectively, n=5 per group) were also included ( Fig. S15 ). (ii) 95 FRET-paired substrates were screened against 12 recombinant MMPs in vitro . ( B ) Heatmap showing in vitro cleavage efficiencies for all substrate-protease pairs. Substrates were ordered by their expected MMP13 cleavage profile: efficient (top), selective (middle), uncleaved (bottom). ( C ) In vitro selectivity for substrates in groups designed for high MMP13 selectivity. Box plots show median, interquartile range, and minimum to maximum. * p < 0.05, ** p 0.4 and S > 2.4 (dotted gray lines) divides substrates into four quadrants of activity. ( E ) Comparison of in vitro efficiencies observed for the single-most selective substrate from each of the mRNA-display group (blue) versus the CleaveNet conditionally-generated group, delineating substrates with high selectivity-low efficiency (green striped) versus high selectivity-high efficiency (green solid). ( F ) IceLogos characterizing substrates in distinct activity quadrants (IceLogos normalized by natural amino acid frequencies). There was strong agreement between technical duplicates for each MMP (0.74 < R 2 < 0.98; Fig. S16 ) and no dependence of cleavage on the properties of the crude peptides used ( Fig. S17 ). Moreover, given that multiple substrates were cleaved by each MMP, the screen served to identify individual thresholds of predicted Z-scores associated with true cleavage for each MMP (ranging between CleaveNet-predicted threshold scores of 0.3 and 2.4; Table S5 ). Importantly, such cleavage thresholds could not previously be inferred from the mRNA-display dataset alone, as this dataset did not distinguish cleaved from non-cleaved substrates. Such thresholds can be employed for more accurate binary classification of substrates when utilizing our CleaveNet Predictor and yielded high classification accuracy for substrate cleavage across all MMPs screened (0.77 < ROC AUC < 0.98; Table S5 ; Fig. S18 ). Visualizing the cleavage efficiencies of all protease-substrate pairs demonstrated that substrates designed for MMP13 efficiency were highly cleaved by MMP13 but also fairly promiscuously cleaved by other proteases, while substrates designed for MMP13 selectivity were more selectively cleaved by MMP13, consistent with design expectations ( Fig. 7B ). We next compared selectivity scores across the different MMP13-selective design groups ( Fig. 7C ; Table S6 ). Similar to the results observed for MMP13 efficiency, both CleaveNet-guided approaches (conditional CleaveNet generated and conditional site-independent baseline + CleaveNet Predictor) outperformed the conditional site-independent alone baseline. However, given that some of the selective substrates from training were uniquely cleaved by MMP13 and thus had the highest selectivity score attainable, the CleaveNet-guided substrates at best only reached equivalent selectivity to these training examples. Albeit highly specific, the MMP13-selective training substrates had relatively low efficiency scores ( E < 0.16; Table S6 ). We were thus curious to investigate how substrates from different groups compared as a function of both their cleavage efficiency and selectivity for MMP13. Plotting selectivity scores vs. efficiency scores for each substrate enabled division of substrates into 4 quadrants ( 49 ): those with low efficiency-low selectivity ( n =56), low efficiency-high selectivity ( n =19), high efficiency-low selectivity ( n =15), and high efficiency-high selectivity ( n =5) ( Fig. 7D ). There were three examples of generated substrates that achieved perfect selectivity for MMP13 (i.e. DL41, DL32, DL28), matching the selectivity score of the best selective substrate in training (DL93) but at the expense of low efficiency ( Fig. 7D, E ). A handful of substrates, like DL48, emerged in the upper right quadrant ( Fig. 7D, E ), characterized by much higher efficiency while maintaining high selectivity. By virtue of these properties, substrates in this quadrant may prove useful for engineering applications, such as the numerous conditionally-activated therapeutics in development ( 12 – 15 ). Notably, this quadrant contained only fully CleaveNet-designed substrates. Additionally, all other substrates in this quadrant were novel, with the closest train-set sequences seldom being cleaved by MMP13, but never selectively ( Table S7 ), with the exception of DL16, which shared a 4-mer with the top 3rd most selective substrate in training. Closer inspection of the sequences in each quadrant of interest reinforced previously identified determinants of high MMP13 efficiency ( Fig. 5C ), while shedding light into determinants of MMP13 selectivity ( Fig. 7F ). Substrates with high selectivity but low efficiency were highly enriched for arginine at P2, aromatic residues, especially phenylalanine, at P1’, and aspartic acid at P3’, while those with both high efficiency and high selectivity shared traits from both. Taken together, this large 95 substrate versus 12 MMP in vitro screen validated and calibrated our CleaveNet Predictor model across MMPs spanning all catalytic activity clusters, and successfully established the ability of the CleaveNet Generator to design substrates conditioned on a target cleavage profile. A subset of sequences conditionally designed for high MMP13 selectivity matched the maximum selectivity levels observed in training (low efficiency-high selectivity quadrant), while a different set exhibited higher cleavage efficiency while maintaining high selectivity (high efficiency-high selectivity quadrant), a cleavage profile that is desirable for engineering applications and that was largely absent from training. Discussion Given that the vast repertoire of possible protease substrates is larger than what has been profiled to date, the identification of substrates with target cleavage profiles remains a significant challenge ( 8 , 18 , 19 ). We developed CleaveNet, an AI-based pipeline for next-generation protease substrate design, that is capable of designing substrates with pre-defined cleavage profiles without prior expertise in chemical biology. The CleaveNet pipeline consists of a generator model and a predictor model that replace the traditional substrate nomination and selection steps and greatly increase the scale, tunability, and efficacy of the design process. We demonstrate the validity of our pipeline for the design of substrates for MMPs, a protease class critical in health and disease. Given that CleaveNet integrates continuous cleavage scores across 18 MMPs, it allows users to optimize for not only cleavage efficiency but also selectivity. Using in silico analyses, we demonstrated that CleaveNet generates substrates that are distinct in sequence yet maintain comparable biophysical properties and cleavage profiles to training data. The pipeline was also able to recapitulate biologically-relevant motifs across 18 MMPs and to capture intricate relationships dictating MMP-cleavability, such as subsite cooperativity. Additionally, introducing a conditioning tag enabled guidance of substrate generation towards specific cleavage profiles, such as high MMP13 selectivity. We validated CleaveNet by experimentally testing substrates designed for high MMP13 efficiency or selectivity in vitro against 12 recombinant MMPs. All CleaveNet-designed substrates were cleaved by MMP13. Among substrates designed for high MMP13 selectivity, we identified three substrates that were uniquely cleaved by MMP13 but exhibited low efficiency, comparable to the most selective substrates in training. Additionally, we discovered a distinct set of CleaveNet-designed substrates with considerable selectivity for MMP13 but significantly higher efficiency – a cleavage profile entirely absent in the training data. These findings highlight the potential of CleaveNet to realize new substrates with novel motifs and desirable cleavage profiles. We open source the CleaveNet models, datasets, and codebase to empower the broader research community. The most significant contribution of our work is the design and deployment of generative models for protease substrate design, a process that to date has remained highly manual, dictated by trial-and-error, and plagued by low hit rates. Our AI-based approach enables new capabilities beyond those of state-of-the-art display-based strategies for substrate design. By learning from peptide sequence data and incorporating a flexible conditioning tag, CleaveNet inherently embeds biological plausibility into library design and provides the first method to steer generations towards sequences that exhibit a cleavage profile of interest across multiple proteases. Far from replacing display-based technologies, CleaveNet complements them and will serve as a hypothesis generator to guide new and highly informative library design and to maximize the information that can be extracted from sparse datasets, greatly decreasing the complexity and cost of screening experiments. Additionally, when used in isolation, the CleaveNet Predictor outputs continuous cleavage scores, as opposed to binary cleavage labels, greatly improving upon other available tools for virtual substrate screening. Taken together, the CleaveNet pipeline is the first end-to-end AI design pipeline for protease substrates, enabling in silico generation of synthetic datasets comparable in scale and quality to the highest throughput display strategies available. Despite its successes, CleaveNet has several limitations that highlight avenues for continued work. First, CleaveNet is currently limited to synthetic substrates and could be extended to encompass natural substrates with appropriate training data or to design full-length proteins containing substrates for proteases of interest to expand beyond applications enabled by short peptide substrates. Such work could benefit from integrating CleaveNet models with other AI-based protein structure prediction ( 54 , 55 ) and generative design ( 37 , 39 ) models. Second, CleaveNet is so far only validated for MMP substrate design. Efforts to create similar datasets and models across other protease subclasses could greatly enhance CleaveNet’s reach and the utility. Finally, even with high predictive accuracy and generation quality, experimental validation remains critical to ensure the real-world applicability of designed substrates, and thus CleaveNet does not completely replace the need for in vitro testing. CleaveNet opens the door to numerous lines of future research. With respect to MMP13 substrate design, docking analyses could reveal a more mechanistic understanding of MMP13 specificity determinants, and the selectivity and efficiency of designed substrates could continue to be refined by additional rounds of conditioning around top hits or by incorporating non-natural amino acids. CleaveNet could be deployed to design MMP substrates for different use cases, such as the design or identification of substrates selective to other MMPs beyond MMP13. Moreover, CleaveNet is also compatible with use cases where multiple protease cleavage events are required, for instance in the design of substrates that require cleavage by multiple MMPs co-expressed in a pathologic condition to maximize pro-drug activation ( 56 , 57 ) or for substrate logic design ( 58 , 59 ). Given that CleaveNet is trained across 18 MMPs, it also offers a method for systematic dissection of the determinants, redundancies, and limits of MMP specificity, and is poised to uncover important insights in protease biology. Looking forward, combining high throughput activity data and the CleaveNet modeling framework could enable the creation of a protease substrate atlas that characterizes the complete repertoire of substrates across protease classes beyond MMPs. Such an atlas would set the stage for similar datasets and models for substrate design across additional enzyme classes, such as nucleases, kinases, and phosphatases. In sum, CleaveNet provides a streamlined approach for the design of peptide substrates that target specific proteases. We envision that the public availability of the open-source CleaveNet models and datasets will democratize substrate design, making it accessible to the multidisciplinary groups doing protease research that may otherwise not have access to sophisticated display-based strategies or deep-learning and chemical biology expertise. By exploring vast se-quence spaces in silico , CleaveNet will catalyze protease research and the development of new protease-targeted profiling, diagnostic, and therapeutic tools. Funding C.M.A. acknowledges support from a fellowship from La Caixa Foundation (ID: 100010434, code: LCF/BQ/AA19/11720039) and from The Ludwig Center Fellowship at MIT’s Koch Insti-tute. T.S.S. was supported by a postdoctoral fellowship from the Ludwig Center at MIT’s Koch Institute for Integrative Cancer Research. Additional support was received from the Koch Institute’s Marble Center for Cancer Nanomedicine. S.N.B. is a Howard Hughes Medical Institute Investigator. Competing interests C.M.A. is an employee of Amplifyer Bio. S.N.B. reports compensation for consulting or board membership by Amplifyer Bio, Catalio Capital, Danaher, Earli Inc., Impilo Therapeutics, Matrisome Bio, Ochre Bio, Port Therapeutics, Ropirio Therapeutics, Satellite Bio, Sunbird Bio, Vertex Pharmaceuticals, and Xilio Therapeutics. All other authors declare no competing interests. Author contributions Conceptualization: C.M.A., S.N.B., A.P.A.; Methodology: C.M.A., S.A., T.S., K.K.Y., S.N.B., A.P.A.; Software Programming: C.M.A., S.A., A.P.A.; Experimental Design: C.M.A., S.A., T.S., K.K.Y., A.P.A.; Investigation: C.M.A., S.A., T.S., A.P.A.; Validation: C.M.A., S.A., T.S., A.P.A.; Formal analysis: C.M.A., S.A., T.S., A.P.A.; Resources Provision: S.N.B, A.P.A.; Data Curation: C.M.A., S.A., S.N.B., A.P.A.; Visualization: C.M.A., S.A., A.P.A.; Writing – Original Draft: C.M.A., S.A., A.P.A.; Writing - Review & Editing: C.M.A., S.A., T.S., K.K.Y., S.N.B., A.P.A.; Supervision: S.N.B., A.P.A. Resource availability Code, model weights, generated sequences, and computed metrics are available at https://github.com/microsoft/cleavenet . Methods Dataset A library of 18,583 10-mer peptide sequences profiled via mRNA display for their cleavage by 18 MMPs ( 42 ) was used for training and validation. Each protease-substrate pair is associated with a normalized score ( Z s m ) representing the strength of cleavage of substrate s by a protease m . The data was randomly split 80/20 into training and testing splits. This test set containing 3,717 sequences was held out from all models during training and is referred to as the mRNA-display test set. An additional, independent out-of-distribution dataset containing 71 peptide substrates screened in vitro against recombinant MMPs ( 49 ) was used as an additional test set. This is referred to as the fluorescence test set. Training the CleaveNet predictor Given an input sequence s , a neural network is tasked to learn a multi-task regression with the objective to predict a continuous value Z ^ sm for all 18 MMPs. The network takes a single sequence s as an input, tokenized via indexing each of the canonical amino acids (plus an additional [ PAD ] token), indexed at 0. For the transformer models, a [ CLS ] token is added to the start of each sequence. For the predictor task, two model architectures were evaluated: a bidirectional LSTM and a Transformer. A grid search was done over the following hyperparameters to nominate the best set of parameters for each model: batch size ( 32 , 64, 128, 256); model hidden dimension ( 16 , 32 , 64, 128); model hidden layers ( 2 , 4 , 6 ); and dropout rate (0, 0.1, 0.25, 0.3). The set of hyperparamters that produced the lowest overall test loss was chosen for each model. The final transformer model was a 56k parameter model comprised of a 2 layer encoderonly transformer (model dimension=32) with 6 attention heads. The model was trained with positional encodings applied to inputs, using a batch size of 64. The output of the transformer was pooled by taking the representation of the [ CLS ] token before the final layer. The final LSTM model was a 44k parameter model comprised of a 2 layer fully-connected biderectional LSTM (model dimension=32), trained using a batch size of 32. Dropout was included after each LSTM layer ( d =0.25) to prevent rapid overfitting. Both models were trained with a 32 dimensional embedding layer. The transformer predictor model was trained using the learning rate formula from the original transformer paper ( 46 ): with a 4k step linear warmup, followed by a decrease proportional to the inverse square root of the step number; the LSTM predictor model was trained using a learning rate of 5e-3. Both models were trained using the Adam optimizer over 70 epochs on 4 NVIDIA A6000 GPUs. To quantify model uncertainty and performance, an ensemble ( 60 ) of 5 predictor models were trained over 5 independent 80/20 train/validation splits of the original training data set, and evaluated using the independent mRNA display test set. The checkpoint with the lowest validation loss, per ensemble, was used for model evaluations. Both LSTM and transformer models performed well on the validation and test sets, and as such both are included in the final set of CleaveNet predictor models. Training the CleaveNet generator The generator model is an autoresgressive model, which learns to predict the next amino acid residue x i in a sequence s , from previous residues ( x 1 …x i −1 ). The probability of a sequence can be factorized as a set of conditional probabilities: The network takes in a single sequence s tokenized via indexing of the canonical amino acids, plus a [ START ] and [ STOP ] token added to the beginning and end of the sequence, respectively. Both a Transformer decoder and LSTM network were evaluated for this task. A grid search was done over the following hyperparameters to nominate the best set of parameters and model architecture for this task: batch size ( 32 , 64, 128); model hidden dimension ( 16 , 32 , 64, 128); layers ( 2 , 3 ); and dropout rate (0, 0.1, 0.2). The set of hyperparameters that produced the lowest overall test loss for unconditional generation was chosen for model comparisons. We observed that the LSTM loss did not improve with an increased number of model parameters (test loss of 42k parameter LSTM: 2.283 vs. 25k parameter LSTM: 2.220, Table S2 ). Alternatively, the transformer-based generator model’s performance improved with scale and performed superior to a similar size LSTM network (test loss of 56k parameter transformer: 2.203 vs. 42k parameter LSTM: 2.283, Table S2 ). As such, the decoder-only transformer was selected as the model architecture for the final CleaveNet generator. The generator’s learning capabilities were assessed on two tasks: unconditional and conditional generation. To train the model conditionally, samples were trained with a vector of 18 MMP scores, rounded to the nearest tenth, in place of the [ START ] token. The model performed much better on the conditional task, compared to the unconditional task (test loss conditional-only: 1.998 vs unconditional-only: 2.139, Table S2 ). To maximize for a flexible and performant model, the final CleaveNet Generator was trained 50% of the time unconditionally and the other 50% conditionally on a set of 18 Z -scores rounded to the nearest tenth place. This 50-50 training scheme yielded a performance comparable to a conditional-only model by increasing model parameters of the joint unconditional-conditional model (test loss unconditional-conditional: 1.980 vs. conditional-only: 1.998, Table S2 ). Accordingly, the final CleaveNet Generator model architecture has 3 decoder-only layers (328k parameters) with 64 hidden model dimensions, 6 attention heads, trained using a batch size of 128 and the transformer learning rate, Equation 1 ( 46 ) with a 4k step linear warmup. The model was trained on 4 NVIDIA A6000 GPUs for 50 epochs, at which the test loss began to plateau. The checkpoint with the lowest test loss was used for model evaluations. At inference, the trained CleaveNet Generator can generate sequences unconditionally, by prompting with a [ START ] token, or conditionally based on a Z-score profile, by prompting with a set of 18 Z-scores. Selectivity score calculation To evaluate substrate selectivity towards an individual protease, selectivity scores were computed as previously described ( 42 ). Briefly, selectivity scores were obtained by normalizing Z-scores across MMPs via the following equation: Unconditional generations To evaluate the CleaveNet Generator, 20k sequences were generated unconditionally by prompting the model with a [ START ] token. These were generated with a standard sampling temperature of 1 and a repeat penalty of 1.2. The repeat penalty was applied to penalize consecutive sampling of a single amino acid. Generated sequences shorter or longer than 10 residues in length were filtered out for comparison purposes; this filtered out 87 sequences. Given the smaller design space, sometimes the model would generate exact matches to the Kukreja dataset; 95 exact matches were generated and filtered out. This resulted in a total of 19,905 sequences used to measure generation metrics. For the purpose of comparing generation quality to the mRNA-display test set, this set of 19,905 generations was downsampled to 4,000 sequences for a fair head-to-head comparison. Conditional generations Conditionally-generated substrates were designed by prompting the CleaveNet Generator with a vector containing 18 Z-scores. To obtain seeds for conditional design, Z-score profiles were predicted for the entire mRNA display dataset using the CleaveNet Predictor. The top 50 Z ^ s 13 scoring substrates were selected as MMP13-efficient seed sequences, and the top 50 S ^ s 13 scoring substrates were selected as MMP13-selective seed sequences (100 total). Predicted Z-score profiles for these 100 seeds rounded to the nearest tenth were were used to seed all conditional generations. Each Z-score profile was used to seed 400 generations each, with a sampling temperature of 1.2 and a repeat penalty of 1.2, resulting in a total of 20k efficient and 20k selective generations. To generate sequences for each of the efficient and selective baselines, sequences were produced via random, position-wise, independent sampling from the position-wise amino acid distribution of the top 50 MMP13-efficient and MMP13-selective designs, respectively. Selection of CleaveNet-generated substrates for in vitro experiments A total of 48 CleaveNetgenerated sequences were used for in vitro validation. 24 of these were selected for predicted MMP13 efficiency from a pool of unconditionally generated sequences, and the other 24 were chosen for predicted MMP13 selectivity from a pool of conditionally generated sequences. Starting from a set of 20k unconditionally generated sequences, designs were ranked by an “uncertainty-aware cleavage score” defined as Z ^ sm − σ s m and filtered for diversity. To encourage a diverse final pool, the set of generations were reduce such that all 5-mers were unique, by keeping only the top scoring sequence corresponding to each 5-mer. The top 24 MMP13-efficient designs were selected after this procedure. For the conditionally-generated sequences, seeds for MMP13 selectivity were chosen by taking the top 50 MMP13-selective Z-score profiles from the mRNA-display dataset and used to prompt the generation of 20k new designs, as previously described. Sequences were then ranked by the predicted selectivity score S s m , with-out any uncertainty filters, and then filtered for diversity as was done for the efficient designs. The top 24 MMP13 selective sequences were chosen. The sequences of the 48 CleaveNetgenerated substrates, annotated by generation procedure, are provided in Supplementary File 1. Selection of baseline and control substrates for in vitro experiments Sequences sampled position-wise from the mRNA-display amino acid distribution were used as a baseline for the CleaveNet models (“site-independent baselines”). For the efficiency baseline, 20k sequences were sampled directly from the mRNA-display test distribution then filtered for diversity as described previously. From this pool 5 sequences were randomly selected for in vitro testing; these are denoted as the “efficient site-independent” set. As a stronger baseline, siteindependent baseline sequences were then evaluated with the CleaveNet Predictor and ranked by their uncertainty-aware cleavage scores. The top 5 highest-scoring sequences were nominated as a second baseline, denoted the “efficient site-independent + CleaveNet Predictor” set. For the selective baseline, 20k sequences were sampled via site-independent, position-wise random sampling from the amino acid distribution of the 50 top MMP13 selective sequences in the mRNA-display dataset and filtered for diversity as previously described. From this pool 5 sequences were randomly selected for in vitro testing; these are denoted as the “selective siteindependent” set. As a stronger baseline, site-independent baseline sequences were evaluated with the CleaveNet Predictor and ranked by their selectivity score. The top 5 highest-scoring sequences were nominated as a second selective baseline, denoted the “selective site-independent + CleaveNet Predictor” baseline. For in vitro experiments, 15 control sequences – 10 positive and 5 negative – were also tested. The positive controls were selected as the top 5 each of selective and efficient MMP13 cleaved sequences from the original mRNA display dataset. The 5 negative controls were also chosen from the mRNA display dataset, these controls were reported to have negative Z-scores across all 18 MMPs. Biophysical property prediction Biophysical properties (aliphatic index, hydrophobicity, Boman solubility, charge, and isoelectric point) were measured using the peptides.py package. K -mer analysis Was performed using the paa.substrate.generate kmers and paa.substrate.search kmer functions in the Protease Activity Analysis package ( 49 ). IceLogo representations IceLogos were generated using the logomaker package. As appropriate for each use case and as specified in each panel, amino acid frequencies were either displayed as raw frequencies or after normalization by the frequencies of amino acids in nature or by the background frequencies of amino acids in the mRNA-display set. Amino acids were colored as a function of their properties into: 1) lavender for hydrophobic aromatic (F, W, Y), 2) blue for other hydrophobic (A, I, L, M, P, V, G), 3) yellow for hydrophilic (C, N, Q, S, T), 4) orange for acidic (D, E, H), and 5) red for basic (K, R). In vitro screening of peptides against recombinant MMP FRET-paired substrates (Mca-DNP) were synthesized by CPC Scientific as crude peptides. The full list of sequences, including N- and C-terminal modifications is provided in data S1 . Recombinant proteases were purchased from Enzo Life Sciences (human MMP1, human MMP2, human MMP3, human MMP8, human MMP9, human MMP10, human MMP11, human MMP12, human MMP14, human MMP20) or R&D Systems (human MMP17). For recombinant protease assays, fluorogenic substrates (10 µ M final concentration) were incubated with recombinant proteases at 37°C for 3 to 24 hours, allowing for signal saturation for multiple peptides. Proteases were incubated at 10nM, with the exception of MMP2 (75nM), MMP10 (75nM), and MMP7(30nM), due to their lower activity levels. A standard MMP buffer (50mM TRIS, 10 mM CaCl2, 300 mM NaCl, 20 uM ZnCl2, 0.02% Brij 35, 0.1% BS at pH 7.5) was used for all MMPs, except for MMP3 that is active at a lower pH (50mM MES, 10 mM CaCl2, 300 mM NaCl, 10mM ZnCl2, 0.02% Brij 35, 0.1% BSA, pH 6). Activation was only required for MMP17, and was performed per manufacturer’s recommendations. Proteolytic cleavage of substrates was quantified by increases in fluorescence over time by a fluorimeter (Tecan Infinite M200 Pro) and analyzed with the Protease Activity Analysis Python package ( 49 ). Each plate contained duplicate reactions for each protease-substrate pair and two replicates corresponding to two independent runs with different peptide plates and protease preparations were performed. Calculation of cleavage efficiencies from in vitro data To facilitate the interpretation of cleavage rates across different MMPs, raw cleavage rates were transformed to cleavage efficiencies, with a value of 0 for substrates that were not cleaved, a value of 1 for the substrate with the highest cleavage rate FC max (fluorescent units/min), and a fractional cleavage efficiency between 0 and 1 defined as for all other cleaved substrates. Supplementary Information for View this table: View inline View popup Download powerpoint Table S1: Performance of the CleaveNet Predictor. For each MMP and test set, the average mean absolute error (MAE) and standard deviation over an ensemble of n =5 is provided for each model class. View this table: View inline View popup Download powerpoint Table S2: CleaveNet generator ablation experiments . The training task, model parameters, and lowest cross entropy test loss is given for each experiment. The final CleaveNet generator is in bold. Download figure Open in new tab Figure S1: Model performance varies as a function of training cleavage profiles across MMPs. The fraction of training substrates with high Z-scores and the training distribution skewness both negatively correlate with the RMSE obtained on the test set (Pearson’s r ). Download figure Open in new tab Figure S2: Performance of the LSTM model on Z-score prediction over the mRNA-display test set for each MMP. The correlation coefficient (Pearson’s r ) is denoted for each plot. Download figure Open in new tab Figure S3: ROC-AUC plot evaluating the transformer predictor over the mRNA-display test set for each MMP. Individual lines represent performance for different Z-score thresholds, with AUCs provided. Download figure Open in new tab Figure S4: ROC-AUC plot evaluating the LSTM predictor over the mRNA-display test set for each MMP. Individual lines represent performance for different Z-score thresholds, with AUCs provided. Download figure Open in new tab Figure S5: Performance of the transformer model on Z-score prediction over the fluorescence test set for each MMP. The correlation coefficient (Pearson’s r ) is denoted for each plot. Download figure Open in new tab Figure S6: Performance of the LSTM model on Z-score prediction over the fluorescence test set for each MMP. The correlation coefficient (Pearson’s r ) is denoted for each plot. Download figure Open in new tab Figure S7: ROC-AUC plot evaluating the transformer predictor over the fluorescence test set for each MMP. Individual lines represent performance for different Z-score thresholds, with AUCs provided. Download figure Open in new tab Figure S8: ROC-AUC plot evaluating the LSTM predictor over the fluorescence test set for each MMP. Individual lines represent performance for different Z-score thresholds, with AUCs provided. Download figure Open in new tab Figure S9: Rank-ordered prediction scores for the transformer predictor, evaluated over the fluorescence test set substrates for each MMP. ( A ) Each substrate is colored by true cleavage with Z t =0, where red are cleaved substrates and blue are not-cleaved substrates. Dotted line is at a predicted Z-score of 0. ( B ) Each substrate is colored by the predicted model uncertainty. Download figure Open in new tab Figure S10: Rank-ordered prediction scores for the LSTM predictor, evaluated over the fluorescence test set substrates for each MMP. ( A ) Each substrate is colored by true cleavage with Z t =0, where red are cleaved substrates and blue are not-cleaved substrates. Dotted line is at a predicted Z-score of 0. ( B ) Each substrate is colored by the predicted model uncertainty. Download figure Open in new tab Figure S11: Position-wise amino acid distributions of ( A ) generated (green, n =4,000) and ( B ) site-independent baseline (yellow, n =4,000) sequences compared to sequences from the mRNA-display test dataset (blue, n =3,717), with the position-wise KL between the CleaveNetgenerated or site-independent baseline distributions and the test distribution denoted. Red box outlines indicate positions in the canonical MMP cleavage site. Download figure Open in new tab Figure S12: Sequence-level biophysical properties measured for sequences from the siteindependent baseline (yellow, n =4,000) against the mRNA-display test sequences (blue, n =3,717). Figure S13: Phylogenetic tree of (A) catalytic domain sequences and (B) entire MMP sequences. Solid line: GPI-anchored MMPs; broken line: trans-membrane MMPs; line-dotted: gelatinases; dotted: non furin regulated MMPs. Modified from Andreini et al. ( 53 ). Download figure Open in new tab Figure S14: IceLogos of all MMP13 cleavable (bottom) and top-7 most efficiently cleaved (top) substrates in the mRNA-display training set (left) versus the CleaveNet-guided designed substrates tested in vitro (right). Raw amino acid frequencies are displayed. View this table: View inline View popup Download powerpoint Table S3: Summary statistics of in vitro MMP13 cleavage efficiency metrics by group. View this table: View inline View popup Download powerpoint Table S4: Summary of novelty features for top MMP13-efficient substrates identified. Sequences for each substrate are provided, with the longest shared k -mer highlighted in blue. Substrates are ranked by their in vitro efficiencies (Eff.). The maximum Levenshtein similarity (Max. Similarity), length of the longest shared k -mer (Shared k -mer length), and maximum Z-score attained for MMP13 by a substrate sharing the longest k -mer (Max Z ^ 13 Training) are reported. T he range of cleavable Z-scores in the mRNA display set was 1-3.32. Download figure Open in new tab Figure S15: Schematic overview of substrates screened in vitro . Substrates were selected from unconditional generations for efficient cleavage by MMP13 (left), or from conditional generations designed to be cleaved selectively by MMP13 (right). In addition to CleaveNet-generated substrates (green, n =24 per group), appropriate baselines consisting of site-independent baseline only (yellow, n =8 per group) and site-independent baseline + CleaveNet Predictor (burgundy, n =8 per group) were added. Controls from the mRNA-display training set corresponding to substrates that were efficiently, selectively, and not cleaved by MMP13 (not shown, n =5 per group) were also included. Download figure Open in new tab Figure S16: Scatter plots of cleavage rates between technical MMP replicates in the in vitro screen. R 2 denotes the correlation coefficient between replicates, which range between 0.73 for MMP3 and 0.98 for MMP13. Higher correlation is achieved for cleaved substrates above the noise floor of the assay. Download figure Open in new tab Figure S17: Scatter plots of in vitro MMP13 cleavage rates vs. CleaveNet-predicted cleavage scores, color-coded by peptide properties. Crude peptides were ( A ) manually annotated by their perceived dullness (negligible, low, medium and high), and manufacturer-reported ( B ) solubility, ( C ) molecular weight, and ( D ) purity. View this table: View inline View popup Download powerpoint Table S5: Z ^ -score cleavage thresholds predicted with the CleaveNet transformer predictor and LSTM and respective cleavage ROC-AUCs for the 95 substrates screened in vitro . Threshold values for either model should be utilized when trying to assess whether a substrate is cleaved or not, given a predicted Z ^ -score. Download figure Open in new tab Figure S18: Scatter plots of CleaveNet predicted scores (with Transformer) vs. in vitro efficiency scores for all substrates in the screen across MMPs. Substrates are color-coded by group and cleavage thresholds in Table S5 denoted by vertical dotted lines. View this table: View inline View popup Download powerpoint Table S6: Summary statistics of in vitro MMP13 selectivity metrics by group. “Uncond.” and “Cond.” represent unconditionally and conditionally, respectively. View this table: View inline View popup Download powerpoint Table S7: Summary of novelty features for high efficiency-high selectivity MMP13 substrates. Sequences for each are provided, with the longest shared k -mer highlighted in blue. Substrates are ranked by their in vitro MMP13 selectivities (Sel.). We report the in vitro MMP13 efficiency, maximum Levenshtein similarity (Max. Similarity), the length of the longest shared k -mer, the maximum MMP13 Z ^ -score (range 1-3.32) and maximum MMP13 selectivity score (ra nge -2.15-2.74) attained by a substrate sharing the longest k -mer in training are reported. Acknowledgements The authors thank The Center for the Development of Therapeutics (CDoT) at the Broad Institute for their assistance with automated peptide plating; Nicolo Fusi for feedback on the project; Philip Rosenfield for assistance in project management; Hannah Richardson for assistance with software release; Heather Fleming for valuable feedback on the manuscript; and Melodi Anahtar for insight on specificity versus efficiency analyses and support with PAA. Footnotes ↵ * These authors jointly supervised the work https://github.com/microsoft/cleavenet References 1. ↵ C. López-Otín , J. S. Bond , J . Biol. Chem . 283 , 30433 – 30437 ( 2008 ). Proteases: multifunctional enzymes in life and disease . OpenUrl CrossRef 2. ↵ G. A. Cabral-Pacheco , et al. , International journal of molecular sciences 21 , 9739 ( 2020 ). The roles of matrix metalloproteinases and their inhibitors in human diseases . OpenUrl CrossRef PubMed 3. ↵ J. S. Dudani , A. D. Warren , S. N. Bhatia , Annu. Rev. Cancer Biol . 2 , 353 – 376 ( 2018 ). Harnessing protease activity to improve cancer care . OpenUrl CrossRef 4. ↵ D. Kato , et al. , Nature Chemical Biology 1 , 33 – 38 ( 2005 ). Activity-based probes that target diverse cysteine protease families . OpenUrl CrossRef PubMed 5. ↵ A. J. O’Donoghue , et al. , Nature Methods 9 , 1095 – 1100 ( 2012 ). Global identification of peptidase specificity by multiplex substrate profiling . OpenUrl CrossRef PubMed 6. ↵ M. Bogyo , et al. , Proceedings of the National Academy of Sciences 94 , 6629 – 6634 ( 1997 ). Covalent modification of the active site threonine of proteasomal β subunits and the escherichia coli homolog hslv by a new class of inhibitors . OpenUrl Abstract / FREE Full Text 7. B. Leiting , et al. , Biochemical Journal 371 , 525 – 532 ( 2003 ). Catalytic properties and inhibition of proline-specific dipeptidyl peptidases ii, iv and vii . OpenUrl Abstract / FREE Full Text 8. ↵ M. Poreba , The FEBS Journal 287 , 1936 – 1969 ( 2020 ). Protease-activated prodrugs: strategies, challenges, and future directions . OpenUrl CrossRef PubMed 9. ↵ A. P. Soleimany , S. N. Bhatia , Trends in Molecular Medicine 26 , 450 – 468 ( 2020 ). Activitybased diagnostics: an emerging paradigm for disease detection and monitoring . OpenUrl CrossRef PubMed 10. E. S. Hwang , et al. , JAMA Surgery 157 , 573 – 580 ( 2022 ). Clinical impact of intraoperative margin assessment in breast-conserving surgery with a novel pegulicianine fluorescence– guided system: a nonrandomized controlled trial . OpenUrl CrossRef PubMed 11. ↵ J. D. Kirkpatrick , et al. , Science Translational Medicine 12 , eaaw0262 ( 2020 ). Urinary detection of lung cancer in mice via noninvasive pulmonary protease profiling . OpenUrl Abstract / FREE Full Text 12. ↵ W. M. Kavanaugh , Expert Opinion on Biological Therapy 20 , 163 – 171 ( 2020 ). Antibody prodrugs for cancer . OpenUrl CrossRef PubMed 13. S. J. Lin , et al. , Cancer Research 81 , 933 – 933 ( 2021 ). Protritac is a modular and robust t cell engager prodrug platform with therapeutic index expansion observed across multiple tumor targets . OpenUrl CrossRef 14. C. Ngambenjawong , L. W. Chan , H. E. Fleming , S. N. Bhatia , ACS Nano 16 , 15779 – 15791 ( 2022 ). Conditional antimicrobial peptide therapeutics . OpenUrl CrossRef PubMed 15. ↵ 15. Q. Zhong , et al. , bioRxiv ( 2024 ). Conditional fusogenic lipid nanocarriers for cytosolic delivery of macromolecular therapeutics . 16. ↵ C. M. Overall , Molecular Biotechnology 22 , 51 – 86 ( 2002 ). Molecular determinants of metalloproteinase substrate specificity: matrix metalloproteinase substrate binding domains, modules, and exosites . OpenUrl CrossRef PubMed Web of Science 17. ↵ Y. Choe , et al. , Journal of Biological Chemistry 281 , 12824 – 12832 ( 2006 ). Substrate profiling of cysteine proteases using a combinatorial peptide library identifies functionally unique specificities . OpenUrl Abstract / FREE Full Text 18. ↵ L. Ducry , B. Stump , Bioconjugate Chemistry 21 , 5 – 13 ( 2010 ). Antibodydrug conjugates: linking cytotoxic payloads to monoclonal antibodies . OpenUrl CrossRef PubMed 19. ↵ P. Kasperkiewicz , M. Poreba , K. Groborz , M. Drag , The FEBS journal 284 , 1518 – 1539 ( 2017 ). Emerging challenges in the design of selective substrates, inhibitors and activitybased probes for indistinguishable proteases . OpenUrl CrossRef PubMed 20. ↵ B. Turk , Nature Reviews Drug Discovery 5 , 785 – 799 ( 2006 ). Targeting proteases: successes, failures and future prospects . OpenUrl CrossRef PubMed Web of Science 21. ↵ J. E. Fuchs , et al. , PLoS Computational Biology 9 , e1003007 ( 2013 ). Cleavage entropy as quantitative measure of protease specificity . OpenUrl CrossRef 22. ↵ I. Schechter , A. Berger , Biochemical and Biophysical Research Communications 425 , 497 – 502 ( 2012 ). On the size of the active site in proteases. I. Papain (reprinted from biochemical and biophysical research communications , vol 27, pg 157, 1967). OpenUrl CrossRef PubMed 23. ↵ M. Geiger , et al. , Nature Communications 11 , 3196 ( 2020 ). Protease-activation using antiidiotypic masks enables tumor specificity of a folate receptor 1-T cell bispecific antibody . OpenUrl CrossRef PubMed 24. ↵ M. Poreba , et al. , Cell Death & Differentiation 21 , 1482 – 1492 ( 2014 ). Unnatural amino acids increase sensitivity and provide for the design of highly selective caspase substrates . OpenUrl CrossRef PubMed 25. ↵ S. Chen , J. J. Yim , M. Bogyo , Biological Chemistry 401 , 165 – 182 ( 2019 ). Synthetic and biological approaches to map substrate specificities of proteases . OpenUrl CrossRef PubMed 26. ↵ J. Zhou , et al. , Proceedings of the National Academy of Sciences 117 , 25464 – 25475 ( 2020 ). Deep profiling of protease substrate specificity enabled by dual random and scanned human proteome substrate phage libraries . OpenUrl Abstract / FREE Full Text 27. I. A. Kozlov , et al. , PLoS One 7 , e37441 ( 2012 ). A highly scalable peptide-based assay system for proteomics . OpenUrl CrossRef PubMed 28. ↵ E. L. Schneider , C. S. Craik , Proteases and Cancer: Methods and Protocols pp. 59 – 78 ( 2009 ). Positional scanning synthetic combinatorial libraries for substrate profiling . 29. ↵ S. E. Boyd , R. N. Pike , G. B. Rudy , J. C. Whisstock , M. G. De La Banda , Journal of Bioinformatics and Computational Biology 3 , 551 – 585 ( 2005 ). PoPS: a computational tool for modeling and predicting protease specificity . OpenUrl CrossRef PubMed 30. M. Ayyash , H. Tamimi , Y. Ashhab , BMC Bioinformatics 13 , 1 – 14 ( 2012 ). Developing a powerful in silico tool for the discovery of novel caspase-3 substrates: a preliminary screening of the human proteome . OpenUrl CrossRef PubMed 31. ↵ 31. T. Lohmüller, et al. ( 2003 ). Toward computer-based cleavage site prediction of cysteine endopeptidases. 32. ↵ F. Li , et al. , Bioinformatics 36 , 1057 – 1065 ( 2020 ). DeepCleave: a deep learning predictor for caspase and matrix metalloprotease substrates and cleavage sites . OpenUrl CrossRef PubMed 33. J. Song , et al. , Bioinformatics 34 , 684 – 687 ( 2018 ). PROSPERous: high-throughput prediction of substrate cleavage sites for 90 proteases with improved accuracy . OpenUrl CrossRef PubMed 34. J. Song , et al. , Bioinformatics 26 , 752 – 760 ( 2010 ). Cascleave: towards more accurate prediction of caspase substrate cleavage sites . OpenUrl CrossRef PubMed Web of Science 35. ↵ F. Li , et al. , Genomics, Proteomics & Bioinformatics 18 , 52 – 64 ( 2020 ). Procleave: predicting protease-specific substrate cleavage sites by combining sequence and structural information . OpenUrl CrossRef PubMed 36. ↵ R. Michael , et al. , PLOS Computational Biology 20 , e1012061 ( 2024 ). A systematic analysis of regression models for protein engineering . OpenUrl CrossRef PubMed 37. ↵ J. L. Watson , et al. , Nature 620 , 1089 – 1100 ( 2023 ). De novo design of protein structure and function with RFdiffusion . OpenUrl CrossRef PubMed 38. A. H.-W. Yeh , et al. , Nature 614 , 774 – 780 ( 2023 ). De novo design of luciferases using deep learning . OpenUrl CrossRef PubMed 39. ↵ S. Alamdari , et al. , bioRxiv ( 2024 ). Protein generation with evolutionary diffusion: sequence is all you need . 40. ↵ W. J. Xie , A. Warshel , National Science Review 10 , nwad331 ( 2023 ). Harnessing generative AI to decode enzyme catalysis and evolution for enhanced engineering . OpenUrl CrossRef PubMed 41. ↵ S. Quintero-Fabián, et al. , Frontiers in Oncology 9 , 1370 ( 2019 ). Role of matrix metalloproteinases in angiogenesis and cancer . OpenUrl CrossRef PubMed 42. ↵ M. Kukreja , et al. , Chemistry & Biology 22 , 1122 – 1133 ( 2015 ). High-throughput multiplexed peptide-centric profiling illustrates both substrate cleavage redundancy and specificity in the MMP family . OpenUrl CrossRef PubMed 43. ↵ Q. Hu , M. Ecker , International Journal of Molecular Sciences 22 , 1742 ( 2021 ). Overview of MMP-13 as a promising target for the treatment of osteoarthritis . OpenUrl CrossRef PubMed 44. ↵ N. Hattori , et al. , The American Journal of Pathology 175 , 533 – 546 ( 2009 ). MMP-13 plays a role in keratinocyte migration, angiogenesis, and contraction in mouse skin wound healing . OpenUrl CrossRef PubMed Web of Science 45. ↵ S. Hochreiter , J. Schmidhuber, Neural Computation 9 , 1735 – 1780 ( 1997 ). Long short-term memory . OpenUrl 46. ↵ A. Vaswani , et al. , Advances in Neural Information Processing Systems 30 ( 2017 ). Attention is all you need . 47. ↵ Z. Lin , et al. , Science 379 , 1123 – 1130 ( 2023 ). Evolutionary-scale prediction of atomiclevel protein structure with a language model . OpenUrl CrossRef PubMed 48. ↵ A. Madani , et al. , Nature Biotechnology 41 , 1099 – 1106 ( 2023 ). Large language models generate functional protein sequences across diverse families . OpenUrl CrossRef PubMed 49. ↵ A. P. Soleimany , C. Martin-Alonso , M. Anahtar , C. S. Wang , S. N. Bhatia , ACS Omega 7 , 24292 – 24301 ( 2022 ). Protease activity analysis: A toolkit for analyzing enzyme activity data . OpenUrl CrossRef PubMed 50. ↵ N. Colaert , K. Helsens , L. Martens , J. Vandekerckhove , K. Gevaert , Nature Methods 6 , 786 – 787 ( 2009 ). Improved visualization of protein consensus sequences by IceLogo . OpenUrl CrossRef PubMed 51. ↵ U. Eckhard , et al. , Data in Brief 7 , 299 – 310 ( 2016 ). Active site specificity profiling datasets of matrix metalloproteinases (MMPs) 1, 2, 3, 7, 8, 9, 12, 13 and 14 . OpenUrl CrossRef PubMed 52. ↵ B. I. Ratnikov , et al. , Proceedings of the National Academy of Sciences 111 , E4148 – E4155 ( 2014 ). Basis for substrate recognition and distinction by matrix metalloproteinases . OpenUrl Abstract / FREE Full Text 53. ↵ C. Andreini , L. Banci , I. Bertini , C. Luchinat , A. Rosato , Journal of Proteome Research 3 , 21 – 31 ( 2004 ). Bioinformatic comparison of structures and homology-models of matrix metalloproteinases . OpenUrl CrossRef PubMed 54. ↵ J. Jumper , et al. , Nature 596 , 583 – 589 ( 2021 ). Highly accurate protein structure prediction with AlphaFold . OpenUrl CrossRef PubMed 55. ↵ K. Tunyasuvunakool , J. Adler , Z. Wu , et al. , Nature 596 , 590 – 596 ( 2021 ). Highly accurate protein structure prediction for the human proteome . OpenUrl CrossRef PubMed 56. ↵ C. Ryppa , et al. , Bioconjugate Chemistry 19 , 1414 – 1422 ( 2008 ). In vitro and in vivo evaluation of doxorubicin conjugates with the divalent peptide E-[c (RGDfK) 2] that targets integrin αvβ3 . OpenUrl CrossRef PubMed 57. ↵ N. Ueki , et al. , Theranostics 6 , 808 ( 2016 ). Synthesis and preclinical evaluation of a highly improved anticancer prodrug activated by histone deacetylases and cathepsin L . OpenUrl CrossRef PubMed 58. ↵ B. A. Holt , G. A . Kwong , Nature Communications 11 , 5021 ( 2020 ). Protease circuits for processing biological information . OpenUrl 59. ↵ A. Sivakumar , et al. , Nature Nanotechnology pp. 1 – 10 ( 2025 ). AND-gated proteaseactivated nanosensors for programmable detection of anti-tumour immunity . 60. ↵ B. Lakshminarayanan , A. Pritzel , C. Blundell , Advances in Neural Information Processing Systems 30 ( 2017 ). Simple and scalable predictive uncertainty estimation using deep ensembles . View the discussion thread. Back to top Previous Next Posted March 02, 2025. Download PDF Supplementary Material Data/Code Email Thank you for your interest in spreading the word about bioRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following Deep learning guided design of protease substrates Message Subject (Your Name) has forwarded a page to you from bioRxiv Message Body (Your Name) thought you would like to see this page from the bioRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share Deep learning guided design of protease substrates Carmen Martin-Alonso , Sarah Alamdari , Tahoura S. Samad , Kevin K. Yang , Sangeeta N. Bhatia , Ava P. Amini bioRxiv 2025.02.27.640681; doi: https://doi.org/10.1101/2025.02.27.640681 Share This Article: Copy Citation Tools Deep learning guided design of protease substrates Carmen Martin-Alonso , Sarah Alamdari , Tahoura S. Samad , Kevin K. Yang , Sangeeta N. Bhatia , Ava P. Amini bioRxiv 2025.02.27.640681; doi: https://doi.org/10.1101/2025.02.27.640681 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Biochemistry Subject Areas All Articles Animal Behavior and Cognition (7636) Biochemistry (17704) Bioengineering (13898) Bioinformatics (41967) Biophysics (21460) Cancer Biology (18599) Cell Biology (25525) Clinical Trials (138) Developmental Biology (13384) Ecology (19909) Epidemiology (2067) Evolutionary Biology (24326) Genetics (15613) Genomics (22512) Immunology (17740) Microbiology (40423) Molecular Biology (17191) Neuroscience (88645) Paleontology (667) Pathology (2835) Pharmacology and Toxicology (4825) Physiology (7646) Plant Biology (15158) Scientific Communication and Education (2046) Synthetic Biology (4302) Systems Biology (9825) Zoology (2271)

Text is read by the "Ask this paper" AI Q&A widget below. Extraction quality varies by source — PMC NXML preserves structure cleanly, OA-HTML may include some navigation residue, and OA-PDF can have broken hyphenation. The publisher copy (via DOI) is the canonical version.

My notes (saved in your browser only)

Ask this paper AI returns verbatim quotes from the full text · source: preprint-html

Answers must be backed by verbatim quotes from this paper's full text. Hallucinated quotes are dropped automatically; if no verbatim passage answers the question, we say so. How this works

Citation neighborhood (no data yet)

We don't have any in-corpus citations linked to this paper yet. This is a recent paper (2025) — citers typically take a year or two to land, and the OpenAlex reference graph may still be filling in.

Source provenance

europepmc
last seen: 2026-05-20T01:45:00.602351+00:00